Compare commits
401 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 274afafde6 | |||
| 60fa958da2 | |||
| 9950361bc9 | |||
| 9f74b6619a | |||
| f687046450 | |||
| c6058652be | |||
| 2a95da2ff4 | |||
| 7f9137fcb6 | |||
| 3bf3968bc7 | |||
| 261aa056f9 | |||
| 042b8c99dd | |||
| 4bfab6b718 | |||
| 008a457557 | |||
| 9494a6b99a | |||
| eb0557621e | |||
| a0eed6f01b | |||
| e2a91e883e | |||
| 62646957ea | |||
| 802c0ab701 | |||
| 4bab23e241 | |||
| a94262271b | |||
| 5c12865c25 | |||
| 3f8c956325 | |||
| 2051ceeb26 | |||
| 325d0771a4 | |||
| 7b97aae85b | |||
| 49a404ddf3 | |||
| 72d6a6878b | |||
| 4466ee0ef2 | |||
| 25ba7f16bb | |||
| 5ba69c9cf2 | |||
| 17052bb515 | |||
| e95ed99bf7 | |||
| 1477e4358a | |||
| 9e4e423ad6 | |||
| 6c2d6e93cb | |||
| 789b6a8716 | |||
| 01462c9695 | |||
| 8b4ff78546 | |||
| d4a2cd720c | |||
| 5a467e1f8b | |||
| 1fb6176783 | |||
| 49df79203c | |||
| 235644c0f0 | |||
| 7772b41993 | |||
| eccd0548ce | |||
| c4e23eebad | |||
| 7df503dfc2 | |||
| f5e02fedd6 | |||
| 29e7a06c49 | |||
| 92a96fcbd8 | |||
| 13482872bb | |||
| c1ca6273fc | |||
| 5f1b260c81 | |||
| 30d6872779 | |||
| 9d1306d442 | |||
| e54e3d87ea | |||
| bf027f10b9 | |||
| 3c5873dfe2 | |||
| e29227d5f4 | |||
| 45aca9eb3e | |||
| 2af13ab1ff | |||
| 0788d84be8 | |||
| 20c1094cbf | |||
| 9011c59b9f | |||
| c11ad71ed0 | |||
| bdcf285265 | |||
| d4f93a7b13 | |||
| cfebc575ea | |||
| 822327eed5 | |||
| 5289eb509f | |||
| e4c703a51a | |||
| 703a05db41 | |||
| 3f036b2a62 | |||
| 82fae94c55 | |||
| b1f34c2e6b | |||
| 3f807d9f1b | |||
| e70263062c | |||
| 1a1e586b62 | |||
| c16d118f09 | |||
| 4b10d02207 | |||
| c5fbfdbf4a | |||
| 12cff28abb | |||
| 77a6a7142e | |||
| 9f3671b801 | |||
| 84034b34d1 | |||
| 1e9b2c9b7e | |||
| 5d422f85fa | |||
| ed2027b202 | |||
| 6b0a99b2b7 | |||
| 6b7caba248 | |||
| f8b0d42a5c | |||
| d1e7d71eee | |||
| 7fd914df1a | |||
| b066eb1903 | |||
| a196d34455 | |||
| 051d320ea0 | |||
| 766772763f | |||
| eab8185d7b | |||
| 7f672f0fb8 | |||
| e2801b9bbc | |||
| ea02c7b248 | |||
| ce74e164c6 | |||
| e60f892efd | |||
| ce05886831 | |||
| be123d0ac7 | |||
| 2d09c8b027 | |||
| ed54f0224e | |||
| 8d5bc3ee89 | |||
| ab0cc71aa4 | |||
| 4cd9046353 | |||
| 4e98a74047 | |||
| a502ba53e0 | |||
| 446cc11d1d | |||
| ddd81fe174 | |||
| ae74cc081f | |||
| cf8da1d5fa | |||
| b8aedeafcb | |||
| 3d61af6f6f | |||
| ed99c209ac | |||
| af4c88d54b | |||
| 44c735f6f5 | |||
| b540a1744b | |||
| 0f51d53098 | |||
| c670792ffe | |||
| 3982ace544 | |||
| 7180b1aad0 | |||
| c26f695402 | |||
| 48877315ca | |||
| 5c08054533 | |||
| b6db9c31f5 | |||
| e7b33fe3a0 | |||
| e3e403e5c8 | |||
| 799014e99d | |||
| 69e09b10fa | |||
| b9d09e044e | |||
| fd8650cda4 | |||
| 11050e24ed | |||
| 769f282408 | |||
| 6f71f40047 | |||
| fb36c5238f | |||
| cc9cdc938b | |||
| 5fede82468 | |||
| 1515025804 | |||
| 7754f53662 | |||
| a507f7b31b | |||
| 71c322f104 | |||
| fde2c15627 | |||
| 2830735644 | |||
| 4e3ac91a22 | |||
| 29d3f0b41f | |||
| cbc732444f | |||
| 6f828b8c38 | |||
| 145a8c8862 | |||
| 7057291739 | |||
| 2302b3bc11 | |||
| 3a004dc1b3 | |||
| a9a3c12232 | |||
| 94ec77a1bc | |||
| 49f285cfda | |||
| 5a12ae7930 | |||
| 127e6832a9 | |||
| 6442a583ae | |||
| 24b96d29ae | |||
| 4721771052 | |||
| 2f71a30bd7 | |||
| 22cdebbdbe | |||
| 154971c2b8 | |||
| fef287c346 | |||
| a37acd5ee3 | |||
| d6ef0c8013 | |||
| dfb70871b4 | |||
| 4ee7b16929 | |||
| 8557289dc0 | |||
| dd2efd8541 | |||
| 5af786d135 | |||
| 380eb63277 | |||
| 6c61355f8f | |||
| 96c406b968 | |||
| a5d81c3f70 | |||
| 92c0f164f1 | |||
| 9e813ec179 | |||
| 395b3b5c46 | |||
| 6e9e464d62 | |||
| 105c065615 | |||
| 01492059d4 | |||
| f84824ee29 | |||
| c4d40fbc2b | |||
| 7c684e40d3 | |||
| 6938f52155 | |||
| 0df34f3220 | |||
| 457458437f | |||
| 3759c41f99 | |||
| 09159f2857 | |||
| 29cd1194c2 | |||
| 815e8f8b23 | |||
| 1e60ac0745 | |||
| 650a4c146b | |||
| 23f299e105 | |||
| dbf6fef0e9 | |||
| d292522d00 | |||
| 0241e0d3a8 | |||
| e4973eb8a4 | |||
| b6b88c5f1c | |||
| 86dddfe240 | |||
| 0d5944af63 | |||
| 4a5030a5c6 | |||
| c1c8794c48 | |||
| 65a78932c1 | |||
| f429ca1a50 | |||
| 73aab3f83e | |||
| 887aca0183 | |||
| 3fd23ecafa | |||
| f379847942 | |||
| ea12107497 | |||
| 591df91de1 | |||
| 6a814176f0 | |||
| d11d1d157c | |||
| d057d56156 | |||
| d703ce1313 | |||
| b32a30fd47 | |||
| 464dbc0930 | |||
| a8cadd9150 | |||
| 57b8c0b56d | |||
| 147f50c19e | |||
| eee4d576a2 | |||
| b4f9d7f53a | |||
| 3aca53b967 | |||
| 4aa1fae296 | |||
| 0c865032f9 | |||
| ea41bbf6b9 | |||
| 7b918c51ff | |||
| 554395b104 | |||
| 823976c1b5 | |||
| c8388a7f92 | |||
| 02e6aef98c | |||
| 6d493bc7bb | |||
| e5cb51a90e | |||
| e545c08082 | |||
| efa0deb9b2 | |||
| ca47e90c01 | |||
| fa1f49675b | |||
| 8426c3528f | |||
| 65f98ba910 | |||
| d05205d1eb | |||
| 667254df47 | |||
| 2926cd1784 | |||
| c801851c66 | |||
| de70aa38f1 | |||
| 53a533afb4 | |||
| 77ad88631b | |||
| d88017807b | |||
| 2159a5a94a | |||
| b9c2cf69f4 | |||
| 8beae50fe7 | |||
| b2a58cb966 | |||
| b8b25cf74c | |||
| f159ca7d27 | |||
| 83f2aea60f | |||
| 002329adb5 | |||
| a49671ceb9 | |||
| 76672ff016 | |||
| 9379f92c23 | |||
| 21ff63b11d | |||
| 21844b54d7 | |||
| 4769481515 | |||
| bbbb4c1eb3 | |||
| 38c248e617 | |||
| 9debc0de27 | |||
| f34361b263 | |||
| be5ba22c75 | |||
| 85c90d440a | |||
| e97502d550 | |||
| aef14ff46e | |||
| cba516bda4 | |||
| ba51e0c6cc | |||
| 086c59848e | |||
| 0c10079755 | |||
| ece2091b53 | |||
| b3f917e6f5 | |||
| fa39a5f55e | |||
| d5128a1d35 | |||
| 61097e5cf0 | |||
| ef507bcd12 | |||
| 94f50e507a | |||
| 75b15086b0 | |||
| dab9645906 | |||
| e93b5f6512 | |||
| bfabe13e8f | |||
| 7e49c6eca2 | |||
| 11cbfa79b4 | |||
| c2c2746922 | |||
| 6ed70700a0 | |||
| 0087645da4 | |||
| 19cacf5b62 | |||
| 4dd12083ab | |||
| 30e21adec7 | |||
| 719b79f892 | |||
| 66e5247b6d | |||
| 1e41bd63b4 | |||
| 4887d03d88 | |||
| 7d5434455d | |||
| f71ee4926e | |||
| 282a2fc2b8 | |||
| f04e934b94 | |||
| 3fae35c357 | |||
| 18aecbfe67 | |||
| 5d75f72473 | |||
| e028a0ae54 | |||
| 2fa673d4c0 | |||
| 9020d01b40 | |||
| 1006805027 | |||
| 27aefbf9a0 | |||
| d42c2bc204 | |||
| 3916adc372 | |||
| fa97f598dd | |||
| ea9aa4fd77 | |||
| a2b8caf6b5 | |||
| b5ddbe5757 | |||
| d223a93039 | |||
| 96d8191149 | |||
| f0e7ac73d6 | |||
| e2fe861b4d | |||
| 279d6f5fbd | |||
| 34480cebef | |||
| 9d0bf14c46 | |||
| 5a3ab5764c | |||
| 3bad9f5785 | |||
| 8308c0b68f | |||
| 3b3063eb2b | |||
| df9086263d | |||
| eb568ff451 | |||
| ac790e4cce | |||
| 6b5f3f472f | |||
| 38dec72152 | |||
| 0e8bfb74fc | |||
| 51f7b0a3ca | |||
| 21c539f22e | |||
| bd2774b5f1 | |||
| c50f5b2d61 | |||
| 01a840cc14 | |||
| c4deef08be | |||
| c796eac09c | |||
| e897e5257b | |||
| 2afa3652bb | |||
| 80092ff359 | |||
| 9d37f3aa29 | |||
| eaf89abaf6 | |||
| d895f02bc1 | |||
| 43206cac2f | |||
| 8bba3a8184 | |||
| 5cf3ca9a89 | |||
| ac474981e4 | |||
| ba04b2359b | |||
| 2e5b63f6f6 | |||
| 3437d6313d | |||
| 743377d6cd | |||
| 5952d559c7 | |||
| 205ad823b0 | |||
| d654ccb818 | |||
| a89dcc9b7e | |||
| 0d7b4fb026 | |||
| 321d8dcbb5 | |||
| ef8c97871e | |||
| 26bafe824b | |||
| 838a701109 | |||
| e5eb3534c7 | |||
| 959c83534f | |||
| 31b028e860 | |||
| 7840e9adf6 | |||
| c3672f5472 | |||
| cf54aed451 | |||
| bbf68f3e3c | |||
| 776743cbe2 | |||
| 826e0aeb2a | |||
| c935b181dd | |||
| ee932fd85b | |||
| fe2e5ede34 | |||
| c325054242 | |||
| 7662e2d0c8 | |||
| 4877992a70 | |||
| 1c051c4e47 | |||
| adb7a67880 | |||
| 723fe494e9 | |||
| 0f08b93659 | |||
| 432c1d92d1 | |||
| 60b7e67b42 | |||
| 3fbd43fe3f | |||
| 32ebf065ac | |||
| 1178b3f684 | |||
| 2a434ced2f | |||
| cc919aa2b6 | |||
| 6417b0edd9 | |||
| 39c7ce76f3 | |||
| b40f477210 | |||
| 5c56cb347f | |||
| e694deace3 | |||
| 748367b7d6 | |||
| dcf5fb3be3 | |||
| f9d2ee2a2b | |||
| 866c7f2e9a |
@@ -1,15 +1,15 @@
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"name": "fleetd",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the fleetd MCP gateway.",
|
||||
"owner": {
|
||||
"name": "LTMS"
|
||||
},
|
||||
"plugins": [
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"name": "fleet",
|
||||
"source": "./plugin",
|
||||
"description": "Make a project bridge-ready: mount the fleetd MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.1.0",
|
||||
"description": "Mount the fleetd MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.2.0",
|
||||
"author": {
|
||||
"name": "LTMS"
|
||||
}
|
||||
|
||||
@@ -0,0 +1,226 @@
|
||||
---
|
||||
name: handover
|
||||
description: Procedure for an outgoing lead to write the handover file that a fresh lead session inherits. Load this when your context is filling up and you are about to be replaced, whether you hand off by hand or fleetd does it for you. The file is the new lead's only inheritance — follow it exactly.
|
||||
---
|
||||
|
||||
# Handover — write the file the next lead depends on
|
||||
|
||||
A lead session fills up its context and has to be replaced by a fresh one. The outgoing lead
|
||||
writes a handover file, and the new session reads that file and carries on.
|
||||
|
||||
**There are two ways to hand off, and the file is the same either way.**
|
||||
|
||||
- **By hand.** You write the file, then tell the operator where it is. The operator starts the new
|
||||
session and points it at the file. This always works.
|
||||
- **With `fleet_handover`** (fleetd #480, merged 2026-09-11). You ask fleetd to do the swap: it
|
||||
checks the file, clears your pane, and tells the fresh session to read it. This needs
|
||||
`leadRollover:` in `fleetd.yaml`; without it every action answers a clean refusal naming
|
||||
`NOT_CONFIGURED`, and you fall back to the manual path. Section 11 below is the procedure.
|
||||
|
||||
Nothing else in this skill changes between the two. Only who performs the swap changes.
|
||||
|
||||
**The new lead's only inheritance is that file.** It does not see your conversation, your plan,
|
||||
or your screen. If the file is thin or wrong, the new lead re-derives what you already knew, and
|
||||
that wastes hours. Writing a good handover file is real work. It is not paperwork you rush
|
||||
through at the end of a session.
|
||||
|
||||
This skill is the procedure for writing it. Every rule below earned its place because a past
|
||||
handover got it wrong.
|
||||
|
||||
## 1. Confirm you are the right session to write this
|
||||
|
||||
Run `fleet_whoami` first. It must answer `primary`. Only a primary (lead) session writes a
|
||||
handover file. A worker's job ends with its own pull request, not a fleet-wide handoff.
|
||||
|
||||
## 2. Every number needs a command, run in this turn
|
||||
|
||||
A number is a claim: a count, a commit hash, a process id, a percentage, a queue depth. Before
|
||||
you write one, run the command that produces it — now, in this turn, against the live state.
|
||||
|
||||
Never take a number from:
|
||||
|
||||
- earlier in your own conversation — the state has moved since then,
|
||||
- a peer lead's report — that is their measurement, not yours,
|
||||
- your own memory of an earlier session.
|
||||
|
||||
Put the command, or its real output, next to the number. That lets the next lead re-run it and
|
||||
check it still matches. If you cannot measure something yourself, say so instead of guessing:
|
||||
"the fleet01 lead reports 91 commits behind; I have not checked this myself."
|
||||
|
||||
## 3. Say what you measured and what you did not
|
||||
|
||||
Mark every claim as one of two things:
|
||||
|
||||
- **"I checked this myself, in the code or on this host, at `<time>`."**
|
||||
- **"I did not check this myself; `<who>` reported it."**
|
||||
|
||||
Never present someone else's measurement as your own. This matters most for cross-host claims —
|
||||
a peer lead's daemon, a worker's report, or something the operator said earlier that you cannot
|
||||
re-verify from here.
|
||||
|
||||
## 4. Record open decisions, and who owns them
|
||||
|
||||
List three things:
|
||||
|
||||
- what the operator actually asked for, in their own words where you have them,
|
||||
- what is still unanswered,
|
||||
- any question you decided yourself instead of asking, with your reason.
|
||||
|
||||
Write the decision so it cannot be mistaken for the operator's instruction. Say plainly: "the
|
||||
operator never answered X; I decided Y, because Z." Without this, the next lead either silently
|
||||
reopens a closed question or assumes the operator chose something they never did.
|
||||
|
||||
## 5. Record live hazards
|
||||
|
||||
List anything that will break if the next lead does the obvious thing next. This includes:
|
||||
|
||||
- unpushed commits or unmerged branches,
|
||||
- a build, a spawn, or a redeploy still running,
|
||||
- code merged to `main` but not yet redeployed to the live daemon,
|
||||
- any trap that looks safe and is not — say what goes wrong and why, not only that something is
|
||||
"tricky."
|
||||
|
||||
## 6. Record what is explicitly not owed
|
||||
|
||||
List work that is finished, and work that another party has said they do not want touched. Name
|
||||
who said so and when. Without this line, the next lead re-does closed work or reopens a question
|
||||
a peer already declined to revisit.
|
||||
|
||||
## 7. Open the file with three re-measurement commands
|
||||
|
||||
The file's own first section must give the next lead three concrete commands to run before
|
||||
acting on anything else in the file:
|
||||
|
||||
1. confirm role — for example `fleet_whoami`,
|
||||
2. confirm the state of the working tree — for example `git status` and
|
||||
`git rev-list origin/main..HEAD`,
|
||||
3. read the live fleet — for example `fleet_list`.
|
||||
|
||||
Record what each command answered when you wrote the file, and tell the reader to run it again
|
||||
rather than trust your answer. The point of this section is that the reader checks live state
|
||||
before acting on any claim in the rest of the file, including yours.
|
||||
|
||||
## 8. Stamp the file with time and commit
|
||||
|
||||
Near the top of the file, write:
|
||||
|
||||
- the date and time you wrote it,
|
||||
- the commit the tree was on (`git rev-parse HEAD`),
|
||||
- whether the tree was clean (`git status`).
|
||||
|
||||
Without this, nobody can tell how old the file is, or which code it describes.
|
||||
|
||||
## 9. State plainly that the file goes stale fast
|
||||
|
||||
Say near the top: **re-measure anything you act on.** The file goes stale the moment anyone
|
||||
merges a branch, spawns a member, or restarts the daemon. Everything in the file is a snapshot
|
||||
of one moment, not a live fact.
|
||||
|
||||
## 10. What to leave out
|
||||
|
||||
Do not include:
|
||||
|
||||
- narration of how the session felt, or how hard something was,
|
||||
- anything the repo already records — code structure, git history, or a rule already written in
|
||||
`CLAUDE.md`. Point at it instead of repeating it,
|
||||
- advice that is only true for the session that is ending — a half-open terminal, a local
|
||||
variable, a train of thought with no state behind it.
|
||||
|
||||
A handover file is a record of state and decisions. It is not a diary.
|
||||
|
||||
## 11. Using `fleet_handover` (only if `leadRollover:` is configured)
|
||||
|
||||
**Run the three steps in this order. The order is not a style choice — the wrong order is
|
||||
refused.**
|
||||
|
||||
1. **`fleet_handover{action: "open", reason: "<why now>"}`.** It returns a `token` and the
|
||||
`handoverPath` you must write to. Nothing has happened to your pane yet.
|
||||
|
||||
**Write to exactly that path, and do not resolve it yourself.** It is always absolute, even when
|
||||
the operator configured a relative `handoverPath`: fleetd resolves a relative one against your
|
||||
own workspace before it hands it to you. The daemon and your pane can run in different
|
||||
directories, so a path you resolve yourself can point at a different file from the one the daemon
|
||||
will check.
|
||||
|
||||
**Check that the path is ignored by git before you write to it (#491).** A relative
|
||||
`handoverPath` resolves inside YOUR workspace, which is usually a repository — and usually not
|
||||
the `fleetd` one, so an ignore rule added to `fleetd` does not protect it. Run
|
||||
`grep -n handover <your workspace>/.gitignore`. No output means the file you are about to write
|
||||
will show up as untracked content in that repo. The file is a snapshot of live state and must
|
||||
never be committed, so tell the operator rather than committing it or silently editing their
|
||||
`.gitignore`.
|
||||
2. **Write the handover file at that path**, following sections 1–10 above.
|
||||
3. **Ask the operator, then `fleet_handover{action: "confirm", token, operatorConfirmed: true}`.**
|
||||
|
||||
Why that order: `confirm` refuses with `HANDOVER_STALE` unless the file was modified **after** the
|
||||
`open` request. That check stops a leftover file from an earlier session being accepted as this
|
||||
one's handover. So writing the file first and then calling `open` — the obvious order — always
|
||||
fails.
|
||||
|
||||
`{action: "cancel", token}` drops a pending request without rolling.
|
||||
|
||||
**Things that will surprise you:**
|
||||
|
||||
- **`accepted` does not mean your pane has been cleared.** It means every gate passed and the roll
|
||||
is scheduled to run once your current turn ends. Say your goodbye in the same turn — you will not
|
||||
get another one.
|
||||
- **There is no terminal or session parameter, on purpose.** The pane is always your own, resolved
|
||||
from your connection, so you can only ever roll yourself.
|
||||
- **`operatorConfirmed` is your report of what a human told you.** Do not pass `true` because you
|
||||
are confident. Ask, wait for the answer, then pass what they said. `requireOperatorConfirm`
|
||||
defaults to `true` and this is the only thing standing between a judgement call and a wiped
|
||||
session.
|
||||
- **The roll can still refuse after `confirm` returns**, and by then there is no caller to tell.
|
||||
Those outcomes are logged only, as `lead-rollover:` lines in the daemon log.
|
||||
- **The bootstrap prompt has never yet landed, and the fix is unproven (fleetd #489).** The first
|
||||
real rollover, on 2026-09-12, joined `/clear` and the bootstrap text into one line and Claude Code
|
||||
refused it as `Unknown command: /clearFresh`. The pane was never cleared and no context was lost,
|
||||
so the failure was safe — the roll simply did nothing. PR #490 fixed the cause and is deployed,
|
||||
but no roll has bootstrapped a fresh session end to end yet. **Assume it may still fail, and tell
|
||||
the operator so before you confirm.** The recovery is the same either way: the file is already
|
||||
written, so the operator starts a session and points it at the file. That is why you write the
|
||||
file before you confirm, and never the other way round.
|
||||
|
||||
## Writing style
|
||||
|
||||
Write in plain English. Use everyday words, one idea per sentence, and active voice. Keep every
|
||||
class, method, file, flag, and config key exactly as it appears in the code — replacing a
|
||||
precise term with a vague one makes the sentence wrong, not simpler. Explain an abbreviation the
|
||||
first time you use it.
|
||||
|
||||
If a diagram genuinely helps, put it in the `.md` file as a fenced ` ```mermaid ` block with no
|
||||
hardcoded colors, so it stays readable on light and dark backgrounds. Quote any label that has
|
||||
brackets, colons, or slashes.
|
||||
|
||||
## Template
|
||||
|
||||
```markdown
|
||||
# Handover — <fleet name> lead session, <date and time>
|
||||
|
||||
Written at commit `<output of git rev-parse HEAD>`. Tree was <clean, or dirty: `<git status
|
||||
summary>`>. Re-measure anything you act on — this file goes stale the moment anyone merges,
|
||||
spawns, or restarts.
|
||||
|
||||
## 0. Do these three things first
|
||||
|
||||
1. Confirm your role: `fleet_whoami` — must answer `primary`. (Answered `<result>` at `<time>`.)
|
||||
2. Confirm tree state: `git status`, `git rev-list origin/main..HEAD`. (`<result>` at `<time>`.)
|
||||
3. Read the live fleet: `fleet_list`. (`<result>` at `<time>`.)
|
||||
|
||||
## 1. What the operator asked for
|
||||
|
||||
<the live instructions, in their words where you have them; what is still open; any decision
|
||||
you made yourself, and why>
|
||||
|
||||
## 2. Open decisions, and who owns them
|
||||
|
||||
<one line per decision: who owns it, what is unanswered>
|
||||
|
||||
## 3. Live hazards
|
||||
|
||||
<one entry per hazard: what breaks, and why, if the next lead does the obvious thing>
|
||||
|
||||
## 4. What is not owed
|
||||
|
||||
<finished work, and work another party has declined; name who said so and when>
|
||||
```
|
||||
@@ -0,0 +1,102 @@
|
||||
---
|
||||
name: hunter
|
||||
description: Defect-hunt procedure for a fleetd worker — sweep an assigned package for real bugs and report several ranked findings without fixing anything. Load this when the lead asks you to hunt or audit a scope rather than review one diff. Do NOT load `reviewer` for this; the two want different output.
|
||||
---
|
||||
|
||||
# Hunter worker — procedure
|
||||
|
||||
The turn contract (one `fleet_reply`, `fleet_ask` for the lead's decisions, honest reporting,
|
||||
never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and already applies.
|
||||
|
||||
**This skill is not `reviewer`.** `reviewer` judges one diff and reports the *single* most
|
||||
important issue in about 90 words. A hunt sweeps a whole package and reports *several* findings
|
||||
in a long structured form. Loading both gives you two contradictory output contracts, and the
|
||||
usual result is a worker that writes a good report into its terminal and ends the turn without
|
||||
sending it. Load exactly one.
|
||||
|
||||
## 0. Read this before you read code: how the report gets home
|
||||
|
||||
Your terminal reaches nobody. The lead sees **only** the text inside your `fleet_reply` call.
|
||||
|
||||
A long report is exactly the case where this goes wrong, so plan for it:
|
||||
|
||||
- **Write the report into the `fleet_reply` argument itself.** Do not compose it in your terminal
|
||||
and then summarise it into the call.
|
||||
- If the report is long, **send it anyway** — one `fleet_reply` with everything.
|
||||
- If you end the turn without replying, the bridge scrapes your pane instead. That scrape carries
|
||||
at most the last 4000 characters, and on a hunt it usually captures the tail of the lead's own
|
||||
brief rather than your findings. The lead then has nothing and has to ask you again.
|
||||
|
||||
## 1. Change nothing
|
||||
|
||||
A hunt is read-only. Do not edit a production file, do not "quickly fix" what you find, and do
|
||||
not run a formatter. You may run the build and tests to *check* a claim, and you should say so
|
||||
when you did.
|
||||
|
||||
## 2. Read the whole scope first
|
||||
|
||||
Read every file in the assigned package before you judge any of it. A defect that a caller
|
||||
elsewhere in the same package makes unreachable is not a defect, and you cannot know that from
|
||||
one file.
|
||||
|
||||
Stay inside the scope. If a defect there depends on a class outside it, read that class to
|
||||
confirm — but the defect itself must live in the scope you were given.
|
||||
|
||||
## 3. The bar — this matters more than the count
|
||||
|
||||
**Name the path into the bad state.** Say which caller, in which state, reaches it. A defect on
|
||||
paper is not a reachable defect. If you cannot name that path, keep the finding but mark it
|
||||
`unproven` and say exactly what you could not check. Do not drop it, and do not dress it up.
|
||||
|
||||
**Say which direction the harm goes.** Data loss, privilege escalation and silent wrong answers
|
||||
are worth reporting even when the window is narrow. A finding whose worst outcome is a worse log
|
||||
line is not worth a block.
|
||||
|
||||
Two workers once ran the same scope: the one that applied the direction-of-harm filter found ten
|
||||
real defects, the one that did not found none. Fewer findings the lead can act on beat many the
|
||||
lead has to triage.
|
||||
|
||||
## 4. Shapes that have produced real merged fixes here
|
||||
|
||||
Read for these first:
|
||||
|
||||
1. **A one-way gate.** A guard added after an incident closes only the direction that incident
|
||||
came from. Do not only ask what closes the gate — ask **which states still open it**.
|
||||
2. **A value read once, then used later to authorise something destructive**, after something
|
||||
else has had a chance to change it.
|
||||
3. **A failure downgraded to a value that looks like a legitimate result** — `-1`, `null`, an
|
||||
empty list, `false` — which a caller then trusts.
|
||||
4. **A lock held for one half of a read-modify-write and not the other**, or two collections
|
||||
updated under different locks.
|
||||
5. **A comment or javadoc stating an invariant the code no longer keeps.** Comments are
|
||||
load-bearing in this repo; a stale one has already caused a bug.
|
||||
|
||||
## 5. What you cannot check, and must not claim you did
|
||||
|
||||
- `fleetd/fleetd.yaml` is gitignored and **absent from your worktree**. You cannot read it. If a
|
||||
finding depends on live configuration, name the key and say you could not check it.
|
||||
- `.mcp.json`, `opencode.json` and `.autoenv` in your worktree are neutralised stubs, not the
|
||||
repo's real files.
|
||||
- The `wiki/` submodule pointer is months old. Do not cite it.
|
||||
|
||||
Reporting a fact you took from the lead's brief as something you measured yourself is a false
|
||||
report, even when the fact is correct. Say where each fact came from.
|
||||
|
||||
## 6. The report — what goes in `fleet_reply`
|
||||
|
||||
One block per finding, most severe first:
|
||||
|
||||
```
|
||||
FINDING N — <one line>
|
||||
file:line
|
||||
Path in: <which caller, in which state, reaches this>
|
||||
Direction: <data loss | escalation | silent wrong answer | outage | ...>
|
||||
Window/trigger: <when it actually happens>
|
||||
Confidence: <confirmed by reading | unproven — say what you could not check>
|
||||
Why nothing else catches it: <the guard or test you checked, and why it misses>
|
||||
```
|
||||
|
||||
End with one line naming every file you read, so the lead knows the denominator.
|
||||
|
||||
**Nothing clears the bar?** Reply `NO FINDINGS`, name the files you read, and say what you ruled
|
||||
out. A clean sweep is a valid result; an invented defect is worse than none.
|
||||
@@ -39,6 +39,19 @@ a worker made all 59 of its edits in the primary's tree and never noticed.
|
||||
test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-toplevel)"
|
||||
```
|
||||
|
||||
**Never run `git stash` (or `git stash pop`/`apply`/`drop`).** Your worktree is isolated, but the
|
||||
stash is **not**: `refs/stash` is one stack shared by the primary's checkout and every other
|
||||
worker's worktree of this repo. Measured on 2026-09-04 — `git stash list` from a worker's worktree
|
||||
and from the primary's tree returned byte-identical output. So a `git stash` you run can be popped
|
||||
into someone else's tree, and a `git stash pop` you run can drop **another worker's** uncommitted
|
||||
edits on top of yours. This has already happened here: two workers were running in parallel and one
|
||||
of them had its in-progress edit silently overwritten by the other's stash.
|
||||
|
||||
The branch is your isolation, so use it instead. To set work aside, commit it on your own branch
|
||||
(`git commit -m "wip: ..."`) and carry on; to try something and back out, use
|
||||
`git diff > /tmp/<your-branch>.patch` then `git checkout -- <file>`. Both stay inside your worktree.
|
||||
If you find a stash entry you did not create, leave it alone and say so in your report.
|
||||
|
||||
## 2. Implement
|
||||
|
||||
- Implement exactly the scope the lead named. Keep the diff focused; note anything out of scope
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
---
|
||||
name: redeploy-fleetd
|
||||
description: Rebuild and restart the live fleetd daemon after a merge (lead / primary only). Load this before redeploying — it holds the script, the drain step, the permission grant, and the five checks that have each gone wrong here before. Workers must never do this.
|
||||
---
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
**A merge is not a deployment.** The running `fleetd` holds the jar it was started with, so a
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. Saying "shipped"
|
||||
about code the live daemon has never loaded is a false report. The lead **may and should** redeploy
|
||||
rather than hand the job back to the operator.
|
||||
|
||||
Workers must never do this. A worker has no business restarting the daemon it is talking through,
|
||||
and stopping it kills the worker's own channel mid-turn.
|
||||
|
||||
**Use the script — do not hand-roll the steps.**
|
||||
|
||||
```bash
|
||||
scripts/redeploy-fleetd.sh --check # report state, change nothing
|
||||
scripts/redeploy-fleetd.sh # build, confirm drain, restart, verify
|
||||
scripts/redeploy-fleetd.sh --yes # skip the drain prompt (fleet already checked)
|
||||
scripts/redeploy-fleetd.sh --no-build # restart the jar already on disk
|
||||
```
|
||||
|
||||
`--no-build` skips the build and restarts whatever jar is at `fleetd/target/fleetd.jar`. Use it only
|
||||
when you just built and nothing changed since. It gives up the protection in the next paragraph: no
|
||||
build runs, so a stale or missing jar is not caught early. The script still checks the file is there
|
||||
and dies with `no jar at … — run without --no-build` if it is not, but it cannot tell you the jar is
|
||||
old. A `mvn clean` in the tree deletes that jar while the daemon keeps running on it, and nothing
|
||||
degrades until the next restart. Run `--check` first: it prints the jar's hash and its modification
|
||||
time, so you can see for yourself whether the jar is missing or older than the code you mean to ship.
|
||||
|
||||
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
||||
old process to exit rather than assuming; it polls `/healthz`; and it anchors its log checks to a
|
||||
line marker taken before the restart, so old errors cannot be misread as new ones. Run `--check`
|
||||
first — it is read-only and reports whether the forge token resolves, which nothing else tells you.
|
||||
|
||||
The script encodes the five things below, each of which has gone wrong here before. Read them anyway:
|
||||
if the script is unavailable or a step fails, this is what it was protecting you from.
|
||||
|
||||
1. **Login shell, or workers silently lose their forge token.** The daemon inherits
|
||||
`WORKER_GITEA_TOKEN` from the shell that starts it, and that comes from
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Start it from a non-login shell and the variable is empty, the
|
||||
daemon starts fine, and the failure appears much later as workers that cannot open a PR. Nothing
|
||||
logs this at startup — the script's `--check` is the only thing that reports it, and it checks
|
||||
whether the name resolves without ever printing the value.
|
||||
2. **Drain live members first.** `fleet_list`, then `fleet_stop` each member, and collect anything
|
||||
you still want with `fleet_poll` before you kill anything. A restart drops in-flight tickets and
|
||||
rendezvous, and a member's report is not recoverable once its ticket is gone.
|
||||
3. **A restart is the only way deferred config keys take effect.** That is usually the reason to do
|
||||
it. The startup log names which keys it accepted and which it deferred — read those lines rather
|
||||
than assuming.
|
||||
4. **Re-check identity afterwards.** Call `fleet_whoami` and confirm it still answers `primary`. The
|
||||
lead is found by its tab label (`fleet.leaders.*.tab`), and a lead whose tab no longer matches is
|
||||
demoted to worker, which refuses every orchestration call.
|
||||
5. **Prove the new jar is the one running.** Confirm a *fresh* `fleetd listening` line at the end of
|
||||
`fleetd/fleetd.out`, dated after the restart. An old daemon that never died looks identical from
|
||||
the outside.
|
||||
|
||||
**Permission.** A `CLAUDE.md` rule grants intent, not tool permission — the command classifier
|
||||
refuses a bare `kill` on the daemon whatever this file says. The script is the seam that fixes that:
|
||||
it is one auditable command, so the operator allow-lists it once instead of approving a stop and a
|
||||
start every time. The rule lives in the operator's Claude Code settings:
|
||||
|
||||
```json
|
||||
{ "permissions": { "allow": ["Bash(scripts/redeploy-fleetd.sh:*)"] } }
|
||||
```
|
||||
|
||||
Granted by the operator on 2026-08-15. If a call is still refused, do **not** route around it by
|
||||
running the stop and start as separate commands — that is exactly the approval the script replaced.
|
||||
Say what you were going to run and why, and let the operator decide.
|
||||
|
||||
@@ -10,6 +10,11 @@ never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and alrea
|
||||
skill is only the *review procedure*: how to work the scope, and the exact shape of what you
|
||||
send back.
|
||||
|
||||
**Wrong skill for a sweep.** This one reviews *one* diff or scope and reports the *single* most
|
||||
important issue. If the lead asked you to hunt or audit a whole package for several defects, load
|
||||
`hunter` instead and ignore this file — the two want different output, and following both is how a
|
||||
worker ends its turn with a good report that never gets sent.
|
||||
|
||||
## 1. Read the whole scope before you judge
|
||||
|
||||
The delegation names your scope — a file, a diff, a PR, a function. **Read all of it first.**
|
||||
|
||||
+10
-4
@@ -87,12 +87,18 @@ jobs:
|
||||
apt-get update && apt-get install -y --no-install-recommends maven
|
||||
mvn -version
|
||||
|
||||
# The `contract` profile clears the default-excludes group, so the @Tag("contract") AMQP test
|
||||
# runs against the RabbitMQ service container (AMQP_URI). Pinned to the one contract test to
|
||||
# avoid re-running the unit suite already covered by the `build` job.
|
||||
# The `contract` profile clears the default-excludes group, so `-Dgroups=contract` runs every
|
||||
# @Tag("contract") test and nothing from the unit suite the `build` job already covered — a
|
||||
# tag selects the whole group, so a test added to it later runs here automatically. A prior
|
||||
# version of this step pinned `-Dtest=AmqpReplyInboxContractTest` by class name instead: that
|
||||
# silently excluded every other contract test (including the herdr ones) from CI, and nobody
|
||||
# noticed until the herdr protocol drifted out from under a test that never ran here
|
||||
# (fleetd #449). If this runner has no herdr socket, the herdr-backed tests in the group
|
||||
# skip on their own `assumeTrue` and only the broker-backed ones actually run — check the
|
||||
# step output rather than assuming which.
|
||||
- name: Contract tests
|
||||
working-directory: fleetd
|
||||
run: mvn -B -Pcontract test -Dtest=AmqpReplyInboxContractTest
|
||||
run: mvn -B -Pcontract test -Dgroups=contract
|
||||
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
|
||||
@@ -19,3 +19,9 @@
|
||||
fleetd.out
|
||||
fleetd/fleetd.out
|
||||
logs/
|
||||
|
||||
# fleetd #480: the lead rollover handover file. `leadRollover.handoverPath` points here, and the
|
||||
# outgoing lead rewrites it on every rollover. It is a snapshot of one moment's live state —
|
||||
# unpushed branches, running builds, open questions — so it is stale the moment it is written and
|
||||
# has no business in git history.
|
||||
.handover/
|
||||
|
||||
@@ -7,6 +7,14 @@
|
||||
> wiki ([Use Cases](https://git.ltms.dev/fleet/fleetd/wiki/7-Use-Cases) → *The portable
|
||||
> CLAUDE.md block*); improvements go to the template first, then out to each project. Anything
|
||||
> specific to *this* repo lives under §Project addendum below, never inline above it.
|
||||
>
|
||||
> **Anything you measure in an addendum is perishable.** Date it, give the command that
|
||||
> re-measures it and what each outcome means, and tell the reader to delete the section once
|
||||
> it stops reproducing. The four parts work together: deciding what would falsify a claim is
|
||||
> the expensive step, and a reader in the middle of another task will not pay it, so a bare
|
||||
> "verify before relying on this" costs the same space and does nothing. The case this is for
|
||||
> is a note that goes stale as a live restriction — it will tell a future session it cannot do
|
||||
> the thing at the moment doing it becomes the job.
|
||||
|
||||
If no `fleet_*` MCP tools are mounted in this session, this section does not apply — skip it.
|
||||
|
||||
@@ -50,8 +58,9 @@ and the sender silently receives nothing. Fail toward the recoverable error.
|
||||
re-send because a call looks slow — the bridge delivers when the peer is `idle`, `blocked` or
|
||||
`done`. A spawned member must **also** have mounted the bridge MCP: until it has, it is not
|
||||
deliverable, and a send waits on that gate for ~60s and then fails without ever reaching its pane.
|
||||
5. **Never drive the terminal multiplexer directly** (no `herdr` CLI, no socket). The bridge owns
|
||||
policy; the multiplexer owns PTYs. Going around the bridge bypasses every rule above.
|
||||
5. **Never move a fleet session, pane or peer except through the bridge.** The bridge owns policy;
|
||||
the multiplexer owns PTYs. Any route that changes fleet state without the bridge's checks
|
||||
bypasses every rule above — the `herdr` CLI and its socket are the usual example.
|
||||
|
||||
### Primary (lead) — run this on every task, in order
|
||||
|
||||
@@ -71,8 +80,11 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
the final judgment call, verification, merges, and anything that depends on context only you
|
||||
hold. Nothing else is yours by default.
|
||||
3. **Spawn every delegated unit first** — `fleet_spawn{profile, worktree:true, ticket}`, one per
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model and cost, not in
|
||||
tier, so the default is rarely what you want.
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model, cost and
|
||||
LIVENESS, not in tier, so the default is rarely what you want. The default is whatever the
|
||||
daemon reports, and on a host where it sits on an exhausted or withdrawn credential every
|
||||
unqualified spawn fails — sometimes loudly, sometimes as a member that spawns fine and then
|
||||
produces nothing. `fleet_profiles` reports the default; check it once per session.
|
||||
4. **Then send them all** — `fleet_send{sessionId, content, wait:false}`. Line 1 of every brief is
|
||||
`Load the <name> skill.` naming the worker's playbook; those skills are opt-in and that line is
|
||||
what makes them reliable. Where the project ships no such skill, spell the procedure out in the
|
||||
@@ -94,6 +106,18 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
8. **Adjudicate, merge, tear down — yours alone.** Read the diff yourself: fully if it is small,
|
||||
targeted at the reported findings and the risky paths if it is large. Reviewer findings direct
|
||||
your attention; they never substitute for it. Then merge, then `fleet_stop{paneId}`.
|
||||
**If the forge refuses you the merge** — a protected branch, a token without the grant — the
|
||||
adjudication is still yours. Read the diff, decide, and hand the operator a merge-ready queue
|
||||
with the refusal quoted. Never report a PR as merged, and never call one "ready to merge"
|
||||
without having read the diff yourself. A refusal is exactly when that shortcut is tempting,
|
||||
because no action is left that forces you to look, and taking it turns this step into
|
||||
forwarding a reviewer's verdict — which is delegating the merge by proxy, two lines above.
|
||||
**Test a refusal; do not read it off a permissions field.** A protected branch holds its merge
|
||||
rights separately from the repository permissions, so that field can say yes while the merge is
|
||||
refused, and still say no after a grant makes it work. Probe instead, with a request that cannot
|
||||
succeed on its merits, so a rejection can only mean the refusal. Treat a transport failure as a
|
||||
third answer that proves nothing: a timeout, a DNS error or a bad URL is not a refusal, and
|
||||
counting it as one makes you sure of something you never measured.
|
||||
|
||||
**Steps 3 and 4 are separate on purpose** — spawning and sending in one loop is how parallel work
|
||||
silently becomes serial, and it is the most common way this layer is wasted. For the same reason,
|
||||
@@ -114,9 +138,11 @@ the merge — and merging on a reviewer's word is delegating it by proxy.
|
||||
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId` |
|
||||
| Message a **peer lead** on this host | `fleet_send{sessionId: <their terminal>, content}` — `fleet_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Message a **peer lead** on another daemon or host | `fleet_send{coordId: <their coord-id>, content}` — needs a `coordinator:` block; your own coord-id is in `fleet_list`. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `fleet_reply{content}` — the one case a lead replies |
|
||||
| Answer a peer lead that messaged you | `fleet_send{coordId}` — or `{sessionId}` if they are on this host. **Not** `fleet_reply`: it has no peer route and the publish is refused |
|
||||
| Read your own held lead-to-lead mail (no ack) | `fleet_poll{coordId: <your own coord-id, from fleet_list's coordinator.selfId>}` — primary-only; never acks, so `fleet_list`'s `held[]` still shows it after. `fleet_list`'s `held[]` gives only a truncated preview — this is the only way to read the full body |
|
||||
| Collect a held reply | `fleet_poll{target}` · then `fleet_ack{target, msgId}` |
|
||||
| Tear down a member | `fleet_stop{paneId}` |
|
||||
| Replace your OWN lead session when its context is full | `fleet_handover{action:"open", reason?}` → write the handover file it names → `fleet_handover{action:"confirm", token, operatorConfirmed}`. Primary-only. **In that order**: the file must be modified *after* `open`, or `confirm` refuses it as stale. There is no terminal parameter — the pane is always your own, so you can never roll another lead. `{action:"cancel", token}` drops a pending request |
|
||||
|
||||
### Lead ↔ lead — coordinate, never delegate
|
||||
|
||||
@@ -140,10 +166,21 @@ The traffic between leads is coordination and nothing else:
|
||||
3. **Verify a peer exactly as you verify yourself.** Peer status buys nothing: check the claim
|
||||
against the code, and re-run the build. A peer's correction gets the same treatment — right or
|
||||
wrong on the evidence, not on who said it. Neither of you merges the other's work unreviewed.
|
||||
**N observations are N data points only if they differ in the axis you are trusting.** This cuts
|
||||
both ways. N *failures* blamed on one cause are one data point when the cases share what you are
|
||||
not varying. N *agreeing measurements* are also one data point when they share an instrument —
|
||||
two hosts, two operators and the same formula is one formula, not two confirmations.
|
||||
4. **Ask a peer to read your project addendum.** Your addendum is instruction surface: every future
|
||||
session on your host obeys it, and a wrong one is obeyed just as faithfully as a right one. The
|
||||
author is the worst reader of their own qualifier placement — measured here, one addendum carried
|
||||
two defects and a non-author found both. If you have no peer, at least re-read it asking "which
|
||||
sentence goes false first, and would a reader reach the caveat before acting?"
|
||||
|
||||
Being messaged by a peer does not make you its worker: answer with `fleet_reply`, and push back on
|
||||
the substance if it is wrong. A peer that simply complies has thrown away the reason there are two of
|
||||
you.
|
||||
Being messaged by a peer does not make you its worker: answer the way you would open —
|
||||
`fleet_send{coordId}` for another daemon, `fleet_send{sessionId}` on this host — and push back on
|
||||
the substance if it is wrong. `fleet_reply` resolves a member's blocked `fleet_send`; a peer's
|
||||
coord-id message is durable and non-blocking, so there is nothing for it to resolve. A peer that
|
||||
simply complies has thrown away the reason there are two of you.
|
||||
|
||||
### Member (worker or architect) — the turn contract
|
||||
|
||||
@@ -188,76 +225,85 @@ must obey belongs in the charter, not here.
|
||||
- **This repo is the bridge.** The daemon is `fleetd`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
and the code behind the rules above is `mcp/FleetMcp` (tools), `auth/Authz` (the role table),
|
||||
`mcp/ConnectionIdentity` (connection→role), and `worker/*Launcher` (`REPLY_CHARTER`).
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR) and
|
||||
`reviewer` (scoped review → one structured finding). Name one in every delegation.
|
||||
- **Herdr socket tests (measured 2026-09-10).** In this repo, herdr is a subject under test. A
|
||||
worker assigned to herdr code, and the lead, may let a test open the herdr socket directly in a
|
||||
throwaway workspace that the test tears down. This only covers
|
||||
`fleetd/src/test/java/dev/ltms/fleet/herdr/AgentControlContractTest.java`,
|
||||
`fleetd/src/test/java/dev/ltms/fleet/herdr/HerdrContractTest.java`,
|
||||
`fleetd/src/test/java/dev/ltms/fleet/herdr/PaneLocatorContractTest.java`, and
|
||||
`fleetd/src/test/java/dev/ltms/fleet/herdr/WorkspacePlacementContractTest.java`. It is not a
|
||||
general licence. Using the herdr CLI or socket to move a real fleet session, pane, or peer stays
|
||||
banned. That is the control plane that invariant 5 protects. Re-measure with
|
||||
`grep -rl 'UnixSocketHerdrClient.connect()' fleetd/src/test/java --include='*.java'`. A non-empty
|
||||
result means tests still open the socket and this note still applies. An empty result means nobody
|
||||
does this any more; delete this section. Canonical invariant 5 restatement is tracked in #458 and
|
||||
is not part of this change.
|
||||
- **`fleet_profiles`/`fleet_list` report two separate outage states, and they are not the same
|
||||
thing.** *Quarantined* (CB-578) means the backend told us it is out of capacity — a long,
|
||||
1800s-default cooldown. *Cooling off* (fleetd #201/#227) means a profile's credential threw two
|
||||
distinct backend errors (a non-exhaustion failure such as an HTTP 5xx) within 60 seconds — a
|
||||
short, fixed 60s cooldown, not configurable per profile. Each check runs independently, so a
|
||||
profile can show both at once. In the JSON: a cooling profile carries `credentialId` and
|
||||
`coolingOffForSeconds`; a quarantined profile carries `quarantinedForSeconds`; a profile hit by
|
||||
both carries all three fields, and either state alone already sets that profile's `free` to `0`.
|
||||
A `fleet_spawn` naming a cooling-off profile is refused before it ever reaches the backend
|
||||
adapter, with a message naming the credential and the remaining seconds ("cooling off after
|
||||
repeated backend errors") — distinct wording from a quarantine refusal, so don't conflate the
|
||||
two when reading a spawn failure.
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR),
|
||||
`reviewer` (one diff → one structured finding) and `hunter` (sweep a package → several ranked
|
||||
findings, change nothing). Name exactly one in every delegation. **`reviewer` and `hunter` are
|
||||
not interchangeable** — `reviewer` caps the answer at one finding in about 90 words, so naming
|
||||
it for a multi-finding sweep hands the worker two contradictory output contracts. That has
|
||||
already cost three workers' turns: each wrote a good report to its terminal and ended the turn
|
||||
with no `fleet_reply`, and the scrape returned the tail of the brief instead.
|
||||
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace) and
|
||||
`fleets-status` (report every fleet that shares one LavinMQ instance).
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace),
|
||||
`fleets-status` (report every fleet that shares one LavinMQ instance),
|
||||
`redeploy-fleetd` (rebuild and restart the live daemon after a merge) and
|
||||
`handover` (write the file a fresh lead session inherits when the outgoing one hands off,
|
||||
fleetd #480).
|
||||
- **This repo is also a Claude Code marketplace, and ships a plugin.** `.claude-plugin/marketplace.json`
|
||||
points at `plugin/`, which carries the MCP mount and the `setup` skill
|
||||
(`/claude-bridge:setup` — make any project bridge-ready). It was added in CB-527 and then went
|
||||
unmentioned by every instruction file, so it drifted and a later session planned it from scratch
|
||||
(#362). **Read `plugin/` before designing anything about onboarding a project.** Two limits are
|
||||
structural, not bugs: a plugin cannot carry the role agent files, because
|
||||
`ClaudeCodeLauncher.java:371` requires `<cwd>/.claude/agents/<role>.md` in the member's own
|
||||
worktree; and a plugin cannot deliver anything to members at all, because
|
||||
`ClaudeCodeLauncher.java:285` exports `CLAUDE_CONFIG_DIR` and every Claude profile here sets it,
|
||||
so a member never reads the operator's plugin store. **The plugin is the lead-side surface;
|
||||
member-facing assets travel in the worktree.**
|
||||
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
||||
(a submodule with its own remote).
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, turn-done fallback —
|
||||
are diagrammed in `docs/MCP-Contract.md` **§6 only**. The rest of that page is a pre-build design
|
||||
doc whose tool names, parameter names and REST paths never caught up with the code, so do not use
|
||||
it as the tool reference (CB-609). Section 6 is kept out of this file because this file loads into
|
||||
every session's context.
|
||||
- **A provisioned worktree neutralizes `.mcp.json`, `opencode.json` and `.autoenv`** — the repo's
|
||||
committed copies would otherwise mount the primary's IDE and forge servers (fleetd #134). The
|
||||
worktree's copy of each is a stub, **not** the repo's real file, so a worker that reads one and
|
||||
reports what it found is reporting on the stub. The daemon logs a per-spawn summary, but the
|
||||
worker cannot see that log. From inside its own worktree a worker — or a lead debugging one —
|
||||
reads the list with `git config --worktree --get-all fleet.neutralizedConfig`, and the
|
||||
consequence with `git config --worktree --get fleet.neutralizedConfigNote`. Never brief a worker
|
||||
to edit one of these files: the edit cannot be committed, and it will not tell you so.
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, the turn-done
|
||||
fallback and status gating — are diagrammed in `docs/MCP-Contract.md`. That page is now flows
|
||||
only: its pre-build tool catalogue, parameter tables and REST paths were deleted rather than
|
||||
corrected, because a hand-maintained second copy of the tool surface is what drifted for a month
|
||||
while this line pointed every session at it (CB-609 / #114). **The live MCP schema is the tool
|
||||
reference**, with the intent→tool table above as the short form. `McpContractDocTest` fails if
|
||||
that page names a `fleet_*` tool the server does not register. The flows are kept out of this
|
||||
file because this file loads into every session's context.
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
**A merge is not a deployment.** The running `fleetd` holds the jar it was started with, so a
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. Saying "shipped"
|
||||
about code the live daemon has never loaded is a false report. The lead **may and should** redeploy
|
||||
rather than hand the job back to the operator.
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. The lead **may and
|
||||
should** redeploy rather than hand the job back to the operator. **Workers must never do this** — a
|
||||
worker has no business restarting the daemon it is talking through, and stopping it kills the
|
||||
worker's own channel mid-turn.
|
||||
|
||||
Workers must never do this. A worker has no business restarting the daemon it is talking through,
|
||||
and stopping it kills the worker's own channel mid-turn.
|
||||
|
||||
**Use the script — do not hand-roll the steps.**
|
||||
|
||||
```bash
|
||||
scripts/redeploy-fleetd.sh --check # report state, change nothing
|
||||
scripts/redeploy-fleetd.sh # build, confirm drain, restart, verify
|
||||
scripts/redeploy-fleetd.sh --yes # skip the drain prompt (fleet already checked)
|
||||
```
|
||||
|
||||
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
||||
old process to exit rather than assuming; it polls `/healthz`; and it anchors its log checks to a
|
||||
line marker taken before the restart, so old errors cannot be misread as new ones. Run `--check`
|
||||
first — it is read-only and reports whether the forge token resolves, which nothing else tells you.
|
||||
|
||||
The script encodes the five things below, each of which has gone wrong here before. Read them anyway:
|
||||
if the script is unavailable or a step fails, this is what it was protecting you from.
|
||||
|
||||
1. **Login shell, or workers silently lose their forge token.** The daemon inherits
|
||||
`WORKER_GITEA_TOKEN` from the shell that starts it, and that comes from
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Start it from a non-login shell and the variable is empty, the
|
||||
daemon starts fine, and the failure appears much later as workers that cannot open a PR. Nothing
|
||||
logs this at startup — the script's `--check` is the only thing that reports it, and it checks
|
||||
whether the name resolves without ever printing the value.
|
||||
2. **Drain live members first.** `fleet_list`, then `fleet_stop` each member, and collect anything
|
||||
you still want with `fleet_poll` before you kill anything. A restart drops in-flight tickets and
|
||||
rendezvous, and a member's report is not recoverable once its ticket is gone.
|
||||
3. **A restart is the only way deferred config keys take effect.** That is usually the reason to do
|
||||
it. The startup log names which keys it accepted and which it deferred — read those lines rather
|
||||
than assuming.
|
||||
4. **Re-check identity afterwards.** Call `fleet_whoami` and confirm it still answers `primary`. The
|
||||
lead is found by its tab label (`fleet.leaders.*.tab`), and a lead whose tab no longer matches is
|
||||
demoted to worker, which refuses every orchestration call.
|
||||
5. **Prove the new jar is the one running.** Confirm a *fresh* `fleetd listening` line at the end of
|
||||
`fleetd/fleetd.out`, dated after the restart. An old daemon that never died looks identical from
|
||||
the outside.
|
||||
|
||||
**Permission.** A `CLAUDE.md` rule grants intent, not tool permission — the command classifier
|
||||
refuses a bare `kill` on the daemon whatever this file says. The script is the seam that fixes that:
|
||||
it is one auditable command, so the operator allow-lists it once instead of approving a stop and a
|
||||
start every time. The rule lives in the operator's Claude Code settings:
|
||||
|
||||
```json
|
||||
{ "permissions": { "allow": ["Bash(scripts/redeploy-fleetd.sh:*)"] } }
|
||||
```
|
||||
|
||||
Granted by the operator on 2026-08-15. If a call is still refused, do **not** route around it by
|
||||
running the stop and start as separate commands — that is exactly the approval the script replaced.
|
||||
Say what you were going to run and why, and let the operator decide.
|
||||
**Load the `redeploy-fleetd` skill before you redeploy.** It holds `scripts/redeploy-fleetd.sh`
|
||||
and its flags, the drain step, the operator's permission grant, and the five checks that have each
|
||||
gone wrong here before. Do not hand-roll the steps from memory.
|
||||
|
||||
### The prompt is part of the product — update it with the code (mandatory)
|
||||
|
||||
@@ -305,6 +351,16 @@ print("in sync:", w[i:w.index("\n```\n", i) + 1] == block)
|
||||
PY
|
||||
```
|
||||
|
||||
**Only the lead can run that check (measured 2026-09-10).** A member's provisioned worktree has
|
||||
`wiki/` uninitialized, so the script dies with `FileNotFoundError: wiki/7-Use-Cases.md`. Measured
|
||||
in three worker worktrees: `git submodule status` printed a leading `-` and `wiki/` held 0
|
||||
entries; the primary's own clone printed a leading `+` and the file was there. So never make this
|
||||
check a member's acceptance criterion — it is unsatisfiable for them, and a brief that asks for it
|
||||
is asking a worker to invent a pass. A member told to check it must say it could not run it, and
|
||||
must never report it as passed. The lead runs it in the main clone before merging. Re-measure with
|
||||
`git submodule status` in a member's worktree: a leading `-` means this still applies; once it
|
||||
prints a commit with no `-`, delete this paragraph.
|
||||
|
||||
## IDE MCP tools & validation workflow (enforced)
|
||||
|
||||
> **Primary only.** Workers have no IDE MCP mount — if you are a worker, skip this section and
|
||||
|
||||
+54
-30
@@ -1,56 +1,80 @@
|
||||
# CB-504 — systemd unit for fleetd (Linux).
|
||||
#
|
||||
# The macOS launchd agent (deploy/dev.ltms.fleetd.plist) is the supervision target for the
|
||||
# current single-host deployment. This unit exists for the per-host gateways CB-308 introduces,
|
||||
# which will run on Linux.
|
||||
# fleetd #360: the previous version of this file started clean and broke the daemon in three ways
|
||||
# that nothing logs (see the DO NOT block and the ExecStart/PrivateTmp comments below for what and
|
||||
# why). The unit below, plus its companion deploy/herdr.service, is the version that has actually
|
||||
# run on fleet01 without those failures. Do not "improve" it back toward the old shape without
|
||||
# re-reading why each line is the way it is.
|
||||
#
|
||||
# Install (user service — fleetd drives the user's herdr, not a system daemon):
|
||||
# mkdir -p ~/.config/systemd/user
|
||||
# cp deploy/fleetd.service ~/.config/systemd/user/
|
||||
# # edit ExecStart / WorkingDirectory / Environment below, then:
|
||||
# cp deploy/fleetd.service deploy/herdr.service ~/.config/systemd/user/
|
||||
# # edit WorkingDirectory / ExecStart below for your host's paths and java location
|
||||
# systemctl --user daemon-reload
|
||||
# systemctl --user enable --now fleetd
|
||||
# systemctl --user enable --now herdr fleetd
|
||||
# loginctl enable-linger $USER # REQUIRED -- see below
|
||||
# journalctl --user -u fleetd -f
|
||||
#
|
||||
# `loginctl enable-linger` is not optional and is easy to miss, because leaving it out looks like
|
||||
# success: `systemctl --user enable` reports "enabled" and both units run for as long as you stay
|
||||
# logged in. A user manager without lingering starts at your first login and stops at your last
|
||||
# logout, so the fleet simply does not come back after a reboot -- which is the whole reason to
|
||||
# use systemd here rather than the setsid scripts these units replaced. Check it with
|
||||
# `loginctl show-user $USER -p Linger`; the answer must be `Linger=yes`.
|
||||
#
|
||||
# Secrets (AI_GATEWAY_TOKEN, WORKER_GITEA_TOKEN, LAVINMQ_URI, COORD_AMQP_URI, ...) are not set
|
||||
# here and need no systemd drop-in: ExecStart runs a login shell, so they come from wherever your
|
||||
# login shell already sources them (this host: ~/.fleet/secrets.sh via ~/.zprofile). If a token is
|
||||
# missing there, fleetd still starts — the daemon reports every secret a configured profile
|
||||
# references, by name, never by value:
|
||||
# journalctl --user -u fleetd | grep 'startup secret'
|
||||
# A resolved one logs "startup secret NAME: set (profile 'x' tokenEnv)"; a missing one logs
|
||||
# "startup secret NAME: MISSING" at WARN and the daemon starts anyway — the first visible symptom
|
||||
# is a member that cannot open a pull request, hours later and in a different component.
|
||||
|
||||
[Unit]
|
||||
Description=fleetd — claude-bridge message server
|
||||
Description=fleetd — fleet message server
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
# Ordering only: herdr is a user process and its socket may appear after us. This is advisory —
|
||||
# fleetd retries the herdr socket rather than exiting, which is what actually makes a late
|
||||
# socket survivable. Do NOT add Requires=: a herdr restart must not take fleetd down with it.
|
||||
# Ordering only. fleetd retries the herdr socket rather than exiting, which is what actually makes
|
||||
# a late socket survivable. Do NOT add Requires=: a herdr restart must not take fleetd down too.
|
||||
After=herdr.service
|
||||
Wants=herdr.service
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
WorkingDirectory=%h/src/claude-bridge/fleetd
|
||||
ExecStart=/usr/lib/jvm/temurin-25-jdk/bin/java -jar target/fleetd.jar fleetd.yaml
|
||||
WorkingDirectory=%h/LTMS/fleetd/fleetd
|
||||
|
||||
Environment=HERDR_SOCKET_PATH=%h/.config/herdr/herdr.sock
|
||||
# PATH matters more than it looks (CB-511): fleetd propagates its own PATH to every worker it
|
||||
# spawns, so this line decides whether the fleet can run a build at all. systemd does not source a
|
||||
# login shell, so without it the daemon — and every worker — gets a bare default with no JDK/Maven.
|
||||
Environment=PATH=/usr/lib/jvm/temurin-25-jdk/bin:/usr/share/maven/bin:/usr/local/bin:/usr/bin:/bin
|
||||
# Secrets are NOT set here — this file is committed. Put the API/worker tokens in a private
|
||||
# drop-in that systemd reads with restrictive permissions:
|
||||
# systemctl --user edit fleetd → [Service] / Environment=FLEETD_API_TOKEN=...
|
||||
# or point EnvironmentFile at a 0600 file:
|
||||
# EnvironmentFile=%h/.config/fleetd/env
|
||||
# A LOGIN shell, not java directly. Every secret this daemon needs (AI_GATEWAY_TOKEN,
|
||||
# WORKER_GITEA_TOKEN, LAVINMQ_URI, COORD_AMQP_URI) lives in ~/.fleet/secrets.sh, which only
|
||||
# ~/.zprofile sources. systemd runs no login shell. Started any other way the daemon boots fine
|
||||
# and looks healthy, and the failure appears hours later as a member that cannot open a pull
|
||||
# request. exec keeps it one process, so systemd tracks the right PID.
|
||||
# This also avoids a SECOND copy of the secrets in a systemd drop-in: one source of truth.
|
||||
ExecStart=/bin/zsh -lc "exec java -jar target/fleetd.jar fleetd.yaml"
|
||||
|
||||
# PrivateTmp MUST stay false -- see herdr.service. fleetd creates the member ZDOTDIR scrub dir and
|
||||
# the opencode config dir under java.io.tmpdir, and the member pane (a herdr child, a different
|
||||
# unit) has to read them. A private /tmp turns the credential scrub into a silent no-op.
|
||||
PrivateTmp=false
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=10s
|
||||
# A bad config (e.g. a non-loopback bind without token auth) makes fleetd fail fast by design.
|
||||
# Give up rather than restart-loop on a permanent error.
|
||||
# A bad config makes fleetd fail fast by design. Give up rather than restart-loop forever.
|
||||
StartLimitBurst=5
|
||||
StartLimitIntervalSec=120
|
||||
|
||||
# The daemon reads the repo, writes worktrees, and talks to a Unix socket — it needs no more.
|
||||
# DO NOT add ProtectSystem=, ProtectHome=, ProtectKernelTunables= or ProtectControlGroups=.
|
||||
# Measured on fleet01 2026-09-05: each of those gives the unit its own mount namespace, and
|
||||
# fleetd resolves a caller role by running lsof to find the loopback peer PID
|
||||
# (mcp/LsofPeerPidLookup). Inside such a namespace lsof returns nothing, every caller falls back
|
||||
# to ANONYMOUS, and the primary is refused every orchestration call with
|
||||
# "unauthenticated: anonymous may not SPAWN".
|
||||
# The daemon still starts, healthz still returns ok and the secrets still resolve - the only
|
||||
# symptom is that the fleet cannot be driven at all. Verified by bisecting the directives:
|
||||
# no sandbox 3 lsof lines | ProtectSystem=strict 0 | ProtectHome=read-only 0
|
||||
# ProtectKernelTunables 0 | ProtectControlGroups 0 | RestrictSUIDSGID 3 | NoNewPrivileges 3
|
||||
# The two below add no mount namespace and are safe.
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectSystem=strict
|
||||
ProtectHome=read-write
|
||||
ProtectKernelTunables=true
|
||||
ProtectControlGroups=true
|
||||
RestrictSUIDSGID=true
|
||||
|
||||
StandardOutput=journal
|
||||
|
||||
Executable
+19
@@ -0,0 +1,19 @@
|
||||
#!/bin/zsh
|
||||
# fleetd #360 — template for the script deploy/herdr.service's ExecStart wraps in a pty.
|
||||
#
|
||||
# `script -qfec <this> /dev/null` needs a real command to run, and that command has to be a LOGIN
|
||||
# shell script: herdr itself needs the same secrets fleetd.service's login shell picks up (this
|
||||
# host: ~/.fleet/secrets.sh via ~/.zprofile), because members it spawns inherit its environment.
|
||||
# systemd's own Environment= lines in herdr.service are not enough for that -- they set TERM and a
|
||||
# bare PATH so the pty starts at all, nothing more.
|
||||
#
|
||||
# Copy this file to the path deploy/herdr.service's ExecStart names
|
||||
# (%h/LTMS/fleetd/fleetd-run/herdr-inner.sh by default) and `chmod +x` it. Not committed under
|
||||
# that path itself because the session name below is host-specific.
|
||||
|
||||
# A 0x0 pty makes every pane spawn fail with "ghostty error -2" (see herdr-multi-instance-facts /
|
||||
# fleet01-headless-herdr-standup) -- give it a real size before herdr ever touches it.
|
||||
stty rows 50 cols 200
|
||||
|
||||
# -l: login shell, so herdr and everything it spawns gets the real secrets and PATH.
|
||||
exec zsh -lc 'exec herdr --session <name>'
|
||||
@@ -0,0 +1,40 @@
|
||||
# fleetd #360 — systemd unit for herdr (Linux), the terminal multiplexer fleetd drives.
|
||||
#
|
||||
# This is fleetd.service's companion: fleetd.service's After=/Wants=herdr.service assumes this
|
||||
# unit exists. Before this ticket it did not, so on a fresh host fleetd started against a herdr
|
||||
# that systemd never supervised at all.
|
||||
#
|
||||
# Install: see deploy/fleetd.service's header comment (both units install the same way).
|
||||
#
|
||||
# ExecStart below runs deploy/herdr-inner.sh (copy the template of that name from this directory
|
||||
# to the path in ExecStart, or point ExecStart at wherever you keep it, and make it executable).
|
||||
# It is a separate file rather than an inline command because it must itself be a login shell (see
|
||||
# its own header for why) and systemd's ExecStart does not run one.
|
||||
|
||||
[Unit]
|
||||
Description=herdr terminal multiplexer (fleet session)
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
# script(1) gives herdr a real pty. Without it the client reports a 0x0 window and every pane
|
||||
# spawn fails with "ghostty error -2" -- which surfaces as a fleetd spawn failure, not a herdr one.
|
||||
ExecStart=/usr/bin/script -qfec %h/LTMS/fleetd/fleetd-run/herdr-inner.sh /dev/null
|
||||
StandardInput=null
|
||||
Environment=TERM=xterm-256color
|
||||
Environment=PATH=%h/.local/bin:/usr/local/bin:/usr/bin:/bin
|
||||
|
||||
# PrivateTmp MUST stay false. fleetd writes the member ZDOTDIR scrub dir and the opencode config
|
||||
# dir under its own java.io.tmpdir, and the member pane -- a child of THIS process -- has to read
|
||||
# them. A private /tmp here silently breaks the credential scrub instead of failing loudly.
|
||||
PrivateTmp=false
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=5s
|
||||
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=herdr
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -0,0 +1,531 @@
|
||||
# CB-201 and CB-227 refinement
|
||||
|
||||
Date: 2026-09-03
|
||||
|
||||
## Decision
|
||||
|
||||
#201 and #227 are one delivery program, but they are not one implementation unit.
|
||||
|
||||
#201 has a real seam: `CompletionResolver` can publish a typed backend-error event only after its
|
||||
waiter resolution wins. #227 can consume that event without knowing any pane text. The classifier
|
||||
must land before the final #227 wiring. However, the policy engine, roster state, and lead nudge can
|
||||
be built in parallel with the classifier.
|
||||
|
||||
I propose five units. Units 1 to 4 own separate files and can run in parallel. Unit 5 owns all
|
||||
composition files and lands after them. It also depends on the #234 defect 2 fix named in the task.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
U1["Unit 1: typed backend-error classification"] --> U5["Unit 5: wire policy, spawn gate, and fleet views"]
|
||||
U2["Unit 2: credential outage policy"] --> U5
|
||||
U3["Unit 3: lead outage nudge"] --> U5
|
||||
U4["Unit 4: durable member outcome"] --> U5
|
||||
D234["#234 defect 2: fail-loud target resolution"] --> U5
|
||||
```
|
||||
|
||||
*Figure 1. Four file-disjoint foundations feed one composition unit.*
|
||||
|
||||
This split keeps `Fleetd.java` under one owner. It also keeps every other production file under one
|
||||
unit in this plan.
|
||||
|
||||
## Evidence checked in the current branch
|
||||
|
||||
I read both issue pages in full. Each page reports zero comments.
|
||||
|
||||
| Evidence | What the code says now |
|
||||
|---|---|
|
||||
| `inject/CompletionResolver.java:229-237` | A turn below two seconds fails before normal scrape classification. A matching fast backend error is therefore only a generic failure today. |
|
||||
| `inject/CompletionResolver.java:260-276` and `:332-367` | #211 already added raw-screen classification when `lastAssistantBlock` is empty. The “dead-code question” in #201 is stale on this branch. |
|
||||
| `inject/CompletionResolver.java:288-317` | Exhaustion wins before the hard-coded `API Error:` match. A backend error then goes through generic `fail(...)`. |
|
||||
| `inject/CompletionResolver.java:311-316` | The code admits that the pattern is a heuristic. A member report which quotes an API error may match it. |
|
||||
| `inject/CompletionResolver.java:449-467` | Startup coverage exists only for `exhaustedPattern`. |
|
||||
| `inject/ExhaustedPatternLookup.java:13-25` | The current lookup and explicit `none()` value are a good shape for the new classifier seam. |
|
||||
| `Fleetd.java:196-207` | One `BackendQuarantine` is shared by placement and the exhaustion sink. Its cooldown comes from `quarantineCooldownSeconds`. |
|
||||
| `Fleetd.java:322-363` | Pattern compilation, target-to-profile lookup, and the live `ExhaustionSink` are composed in `Fleetd.main`. The sink on this branch still ends in `.ifPresent(...)`. This plan assumes #234 replaces that silent path. |
|
||||
| `placement/BackendQuarantine.java:60-87` | A repeated exhaustion restarts one long quarantine. The store is credential-keyed and uses an injected monotonic clock. |
|
||||
| `member/CompositePeerLauncher.java:260-317` | Explicit and policy-selected spawns have separate gates. Both paths must learn about outage cool-off. |
|
||||
| `member/CompositePeerLauncher.java:347-379` | Exhaustion refusal already checks a credential for explicit spawns and filters policy candidates. Its error text says “exhausted”. |
|
||||
| `placement/PlacementContext.java:10-22` and `PlacementPolicyUtil.java:14-83` | Automatic placement has only one transient exclusion set named `quarantined`. Reusing it would make outage errors say “backend exhausted”. |
|
||||
| `mcp/FleetMcp.java:913-1025` | `fleet_list` sets `free: 0` and adds `credentialId` plus `quarantinedForSeconds` when quarantine is active. |
|
||||
| `session/MemberSession.java:51-59` | The roster has `DONE` and generic `FAILED`, but no backend-error state or stored reason. |
|
||||
| `session/SessionManager.java:695-773` | A normal boundary moves `BUSY` to `DONE`. A failure moves any non-released session to `FAILED`. The async completion resolver can race the `DONE` update. |
|
||||
| `session/SessionManager.java:648-687` | `rosterView` reports the session state, but it reports no terminal reason. |
|
||||
| `msg/MessageService.java:922-940` | CB-588 already nudges for every terminal async ticket, including failures. Current code would report failed tickets, but it would not report one correlated outage. |
|
||||
| `msg/ReplyPushLoop.java:20-48` | Replies, terminal tickets, and questions share one per-lead schedule. This prevents two push sources from injecting competing turns. |
|
||||
| `msg/ReplyPushLoop.java:305-395` | Each push entry point resolves the owning lead through `PrimaryRegistry`. Missing ownership is logged and the durable or pending item remains the backstop. |
|
||||
| `msg/ReplyPushLoop.java:496-547` | One tick builds one combined nudge. Pending items have separate reminder counts. |
|
||||
| `health/FleetHealthMonitor.java:91-143` | Health is a slow periodic observer of members and message-layer facts. It does not receive completion classifications. |
|
||||
| `health/FleetHealthMonitor.java:206-208` | `healthCoverage` means health enabled plus webhook configured. It does not describe lead-pane alerts. |
|
||||
| `Fleetd.java:465-486` | Health stays `detection-only` without the webhook notification setting. |
|
||||
|
||||
I also read the related unit tests for `CompletionResolver`, `ReplyPushLoop`, `BackendQuarantine`,
|
||||
`CompositePeerLauncher`, `PlacementPolicyUtil`, `SessionManager`, `MessageService`, and `FleetMcp`.
|
||||
|
||||
I did not inspect the in-progress #234 branch. I only used the two measured facts in the task. No
|
||||
peer architect was named, so I did not exchange a design with one.
|
||||
|
||||
## Required behaviour
|
||||
|
||||
The policy should use these first values:
|
||||
|
||||
- Threshold: **2** classified backend errors.
|
||||
- Window: **60 seconds**, measured from the first error to the second.
|
||||
- Cool-off: **60 seconds**, starting when the threshold is reached.
|
||||
- Correlation key: `credentialId`, never profile name and never error text.
|
||||
- Incident rule: one active incident per credential. Errors during its cool-off do not extend it and
|
||||
do not create more lead notices.
|
||||
- Rearm rule: after cool-off ends, two fresh errors are needed for another incident.
|
||||
|
||||
Two errors are the smallest threshold which protects the honest one-turn failure. A 60-second window
|
||||
fits the measured two-member outage. A 60-second cool-off blocks immediate repeat spawns without
|
||||
turning a short backend fault into the default 1,800-second exhaustion quarantine.
|
||||
|
||||
A single classified error still fails its send and marks its member `backend_error`. It does not
|
||||
cool a credential and does not send an outage notice. This is what “a single error changes nothing”
|
||||
must mean at the credential level. It cannot mean that the failed member still looks successful.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant R1 as Resolver for member A
|
||||
participant R2 as Resolver for member B
|
||||
participant P as Outage policy
|
||||
participant S as Spawn gate
|
||||
participant N as Lead push loop
|
||||
participant L as Lead pane
|
||||
|
||||
R1->>P: backend error for credential C
|
||||
Note over P: Count 1, no cool-off
|
||||
R2->>P: backend error for credential C within 60s
|
||||
P->>P: Start one 60s incident
|
||||
P->>S: Credential C is cooling off
|
||||
P->>N: Queue one incident notice
|
||||
N->>L: Inject when lead is idle, blocked, or done
|
||||
L->>S: Request another spawn on credential C
|
||||
S-->>L: Refuse and report remaining cool-off
|
||||
```
|
||||
|
||||
*Figure 2. The second independent classification creates the fleet-level event.*
|
||||
|
||||
Against the 2026-09-01 case, the second failed member would start cool-off. `fleet_list` would show
|
||||
zero free capacity and both members as `backend_error`. The push loop would inject one outage notice
|
||||
even if the lead had not polled either ticket yet. The design reports the outage. It does not recover
|
||||
uncommitted work from the members.
|
||||
|
||||
## Unit 1 — Typed backend-error classification
|
||||
|
||||
### Scope
|
||||
|
||||
Replace the direct hard-coded check inside `CompletionResolver` with a lookup and a sink. Keep the
|
||||
public send result as a failed send. The typed internal event is the seam #227 consumes.
|
||||
|
||||
The lookup returns the pattern for a target. The sink receives the target, matched line, and full
|
||||
failure reason. It fires only after `Rendezvous.resolveFailure(...)` wins for that exact captured
|
||||
waiter. This copies the race rule already used by `ExhaustionSink`.
|
||||
|
||||
The classifier must run in all three current paths:
|
||||
|
||||
1. a normal non-empty assistant block;
|
||||
2. the #211 raw scrape fallback;
|
||||
3. a turn inside `MIN_TURN_NANOS`, before it becomes a generic too-fast failure.
|
||||
|
||||
In every path, the order stays: stale-baseline guard, exhaustion, backend error, then generic
|
||||
failure or completion. A fast turn still fails when no configured pattern matches.
|
||||
|
||||
Keep `(?i)\bAPI Error\s*:` as a compatibility pattern for profiles without `errorPattern` until the
|
||||
operator config is updated. Do not call this full coverage. Startup reporting in Unit 5 must name
|
||||
profiles using this weaker legacy default.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/inject/BackendErrorPatternLookup.java`.
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/inject/BackendErrorSink.java`.
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/inject/CompletionResolver.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/inject/CompletionResolverTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. A target-specific error pattern matches a normal assistant block and resolves the send as failed.
|
||||
2. The same match calls `BackendErrorSink` exactly once after the waiter resolution wins.
|
||||
3. A late classification which loses to `fleet_reply` does not call the sink.
|
||||
4. An exhausted line that also matches the generic error pattern stays `BACKEND_EXHAUSTED`. It calls
|
||||
only `ExhaustionSink`.
|
||||
5. A raw pane with leading Terminal User Interface (TUI) chrome and no assistant marker still uses
|
||||
the #211 fallback and calls the backend-error sink.
|
||||
6. A matching error inside the two-second floor is typed and sent to the sink. A non-matching fast
|
||||
turn stays a generic failure.
|
||||
7. An unchanged delivery baseline which contains old backend-error text is suppressed. It never
|
||||
increments outage evidence.
|
||||
8. A non-match keeps the existing completion result and text.
|
||||
9. Constructors used by current callers keep compiling. They use the legacy default lookup and an
|
||||
explicit inert sink until Unit 5 supplies the production objects.
|
||||
10. Unit tests pass. The developer runs the focused test first, then `mvn clean install` from
|
||||
`fleetd/`.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 1 can run with Units 2, 3, and 4.
|
||||
|
||||
Unit 5 depends on its new lookup, sink, and constructor.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact classifier order in all three paths.
|
||||
- The focused test command and result.
|
||||
- The test which proves a losing waiter race does not publish an event.
|
||||
- The test which proves a fast matching failure is typed.
|
||||
- The final `mvn clean install` result.
|
||||
- Any constructor kept only for transition and where Unit 5 replaces it.
|
||||
|
||||
## Unit 2 — Credential outage policy
|
||||
|
||||
### Scope
|
||||
|
||||
Build a small credential-keyed state machine. It accepts already-classified backend-error events.
|
||||
It does not read pane text, profiles, sessions, or lead state.
|
||||
|
||||
Use an injected monotonic clock. A call records `credentialId`, target, and reason. It returns a new
|
||||
incident only on the threshold crossing. The incident contains a stable event id, credential id,
|
||||
the distinct affected targets, evidence count, window, and remaining cool-off.
|
||||
|
||||
This class owns both correlation and short cool-off. Keeping them together makes threshold crossing
|
||||
and the cool-off deadline one atomic state change.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/placement/BackendOutagePolicy.java`.
|
||||
- Add `fleetd/src/test/java/dev/ltms/fleet/placement/BackendOutagePolicyTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. One error creates no incident and no cool-off.
|
||||
2. Two errors for one credential within 60 seconds create exactly one incident and a 60-second
|
||||
cool-off.
|
||||
3. Two errors more than 60 seconds apart do not create an incident.
|
||||
4. The exact 60-second boundary has a pinned result. Use inclusive `<= 60s` so scheduler delay does
|
||||
not discard evidence at the boundary.
|
||||
5. Different credentials never share evidence.
|
||||
6. Different profiles which supply the same credential id do share evidence. The policy itself only
|
||||
sees the credential id.
|
||||
7. More errors during active cool-off do not extend its deadline and do not return another incident.
|
||||
8. After expiry, old evidence is cleared. Two fresh errors are needed to create the next incident.
|
||||
9. Remaining seconds round up, matching `BackendQuarantine` reporting.
|
||||
10. Concurrent second and third errors cannot return two incidents.
|
||||
11. The focused tests and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 2 can run with Units 1, 3, and 4.
|
||||
|
||||
Unit 5 depends on the policy API.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The state transition table and locking method.
|
||||
- The exact threshold, window, cool-off, and boundary rule.
|
||||
- The test which proves one incident under concurrent calls.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 3 — Lead outage nudge
|
||||
|
||||
### Scope
|
||||
|
||||
Add backend incidents as a fourth pending source in `ReplyPushLoop`. Do not create another scheduler
|
||||
or call `AgentControl.send` from `Fleetd`. The existing combined per-lead schedule is the control
|
||||
which prevents competing injected turns.
|
||||
|
||||
The entry point takes an incident id, affected worker targets, credential id, affected profile
|
||||
names, and remaining cool-off. It resolves distinct owning leads through `PrimaryRegistry`.
|
||||
|
||||
Each `(incidentId, lead)` item is one-shot. It waits while the lead is not injectable. After one
|
||||
successful `agents.send`, remove it. A send exception keeps it pending for a bounded retry. It never
|
||||
uses the repeated reminder behaviour of an uncollected ticket.
|
||||
|
||||
Also add a fail-loud entry point for a classified target that Unit 5 cannot map to a credential. It
|
||||
uses `PrimaryRegistry.nudgeTargetFor(target)` and says that correlation could not run. If no lead is
|
||||
known, log at `WARN`, not `DEBUG`.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/msg/ReplyPushLoop.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/msg/ReplyPushLoopTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. One incident affecting two workers owned by one lead causes one successful pane injection.
|
||||
2. Two affected workers owned by two leads cause one successful injection per affected lead. This
|
||||
is one notice per event per lead, not one notice per member.
|
||||
3. Repeating the same incident id is idempotent.
|
||||
4. A busy or unknown lead is not injected. The item stays pending until the lead becomes injectable
|
||||
or its attempt cap is reached.
|
||||
5. After one successful injection, later ticks do not mention that incident again.
|
||||
6. A failed `agents.send` is retried within the existing bound. A successful retry still gives only
|
||||
one successful send.
|
||||
7. A pending ticket and an outage incident for one lead appear in one combined nudge, not two
|
||||
competing turns.
|
||||
8. The text names the credential, profiles, affected workers, and remaining cool-off. It tells the
|
||||
lead to run `fleet_list`.
|
||||
9. An unmapped target produces a direct warning notice when a lead is known. If no lead is known,
|
||||
the code logs a `WARN` naming the target and reason.
|
||||
10. `stop()` clears incident state as it clears other push state.
|
||||
11. Existing reply, ticket, and question tests stay green. The focused tests and
|
||||
`mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. The API uses plain values, not the Unit 2 incident class. This lets Unit 3 run in parallel.
|
||||
|
||||
Unit 5 adapts the Unit 2 incident into this entry point.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact one-shot and retry rules.
|
||||
- The test showing one combined nudge with a failed ticket.
|
||||
- The test showing one successful send for two affected workers.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 4 — Durable member backend-error outcome
|
||||
|
||||
### Scope
|
||||
|
||||
Make a classified backend failure remain visible after its ticket is collected or expires.
|
||||
|
||||
Add `BACKEND_ERROR` to `MemberSession.State`. Add a nullable failure detail to `MemberSession` and
|
||||
render it as `failureReason` in `SessionManager.rosterView`. Add
|
||||
`SessionManager.onBackendError(target, reason)`.
|
||||
|
||||
The transition must handle both completion orderings:
|
||||
|
||||
- `BUSY -> BACKEND_ERROR` when classification wins before the normal completion state update;
|
||||
- `DONE -> BACKEND_ERROR` when the async resolver runs after `SessionManager.onTurnComplete`.
|
||||
|
||||
It must use a compare-and-set retry or another atomic update. `RELEASED` must never return to the
|
||||
roster. A backend-error member is terminal and cannot accept another delivery.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/session/MemberSession.java`.
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/session/SessionManager.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/session/SessionManagerTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. `onBackendError` moves a `BUSY` member to `BACKEND_ERROR` and stores the reason.
|
||||
2. It also moves `DONE` to `BACKEND_ERROR`, covering the resolver race.
|
||||
3. A later normal `onTurnComplete` cannot change `BACKEND_ERROR` back to `DONE`.
|
||||
4. A released or unknown member is not recreated. The unknown case logs at `WARN` and returns an
|
||||
explicit false result to its caller.
|
||||
5. `onDelivered` refuses a `BACKEND_ERROR` member, just as it refuses generic `FAILED`.
|
||||
6. `rosterView` reports `state: backend_error` and `failureReason` after the send ticket is gone.
|
||||
7. Ordinary members do not gain a blank or invented `failureReason` field.
|
||||
8. Existing constructors keep source compatibility for tests and adapters.
|
||||
9. The focused tests and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 4 can run with Units 1, 2, and 3.
|
||||
|
||||
Unit 5 calls the new session method from the production sink.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The two race orderings and the tests for both.
|
||||
- The exact roster JSON shape.
|
||||
- The unknown-target result and log level.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 5 — Production wiring, spawn gate, and fleet views
|
||||
|
||||
### Scope
|
||||
|
||||
Compose Units 1 to 4 in production. This is the only unit which edits `Fleetd.java`.
|
||||
|
||||
Add per-profile `errorPattern` config beside `exhaustedPattern`. Compile both once at startup. A
|
||||
configured pattern wins over the legacy default. Report configured profiles and legacy-default
|
||||
profiles separately at startup. A bad regex must stop startup with the profile and key in the
|
||||
message.
|
||||
|
||||
Wire one production `BackendErrorSink` with this order:
|
||||
|
||||
1. mark the member `backend_error` with its reason;
|
||||
2. resolve the profile and its current `effectiveCredentialId()` through the fail-loud #234 seam;
|
||||
3. record the error in `BackendOutagePolicy`;
|
||||
4. on a new incident, submit one event to `ReplyPushLoop`.
|
||||
|
||||
If target metadata cannot be resolved, do not end in `Optional.ifPresent`. Log an error and call the
|
||||
Unit 3 unmapped-target notice. The failed send still reaches its ticket through CB-588.
|
||||
|
||||
Teach both spawn paths about a separate cool-off source. Exhaustion quarantine has priority when
|
||||
both states are active. Automatic placement needs a distinct `coolingOff` set so its refusal does
|
||||
not say “exhausted”.
|
||||
|
||||
Extend the MCP (Model Context Protocol) views:
|
||||
|
||||
- A cooling profile has `free: 0`, `credentialId`, and `coolingOffForSeconds` in `fleet_list`.
|
||||
- It does not have `quarantinedForSeconds` unless exhaustion quarantine is also active.
|
||||
- `fleet_profiles` has a separate `coolingOff` map, not an entry in `quarantined`.
|
||||
- A direct spawn refusal says the credential is cooling off after repeated backend errors and gives
|
||||
the remaining seconds.
|
||||
|
||||
Do not change `FleetHealthMonitor.coverage`. It still describes the periodic health webhook path.
|
||||
Lead-pane outage delivery is a separate capability.
|
||||
|
||||
### Files owned
|
||||
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/Fleetd.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/config/ConfigRef.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/member/CompositePeerLauncher.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/placement/PlacementContext.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/placement/PlacementPolicyUtil.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/config/FleetConfigTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/config/ConfigRefTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/member/CompositePeerLauncherTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/placement/PlacementPolicyTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/mcp/FleetMcpTest.java`
|
||||
- Add `fleetd/src/test/java/dev/ltms/fleet/BackendOutageFlowTest.java`.
|
||||
- `fleetd/fleetd.example.yaml`
|
||||
- `CLAUDE.md`
|
||||
|
||||
No earlier unit edits these files.
|
||||
|
||||
The lead, not a worker, must update `wiki/7-Use-Cases.md`, `wiki/9-Implementation.md`, and
|
||||
`wiki/11-Features.md`. Project rules forbid workers from committing `wiki/`. The portable block in
|
||||
`CLAUDE.md` and `wiki/7-Use-Cases.md` must remain byte-identical.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. `errorPattern` binds per profile. Blank uses the legacy default and is reported as degraded
|
||||
coverage. A config reload which changes it is reported as deferred because patterns are compiled
|
||||
at startup.
|
||||
2. A malformed `errorPattern` stops startup and names `profiles.<name>.errorPattern`.
|
||||
3. The real production sink never silently drops an unknown target. A test captures its error log
|
||||
and the fallback notice call.
|
||||
4. One real backend-error classification through `CompletionResolver` marks only its member. It does
|
||||
not cool the credential and does not send an outage notice.
|
||||
5. Two real classifications for one credential within 60 seconds start one incident.
|
||||
6. The integration test then calls the real explicit-profile spawn gate. It is refused before any
|
||||
adapter spawn call, with “cooling off” and remaining seconds in the message.
|
||||
7. The same test calls an automatic placement path. A cooling candidate is skipped. If every
|
||||
candidate is cooling, the error names cool-off rather than exhaustion.
|
||||
8. Profiles sharing the credential are all blocked. A profile on another credential stays usable.
|
||||
9. `fleet_list` from the same fixture shows both members as `backend_error`, preserves each failure
|
||||
reason, and reports `free: 0`, the credential, and `coolingOffForSeconds`.
|
||||
10. `fleet_profiles` reports cool-off separately from quarantine.
|
||||
11. The real `ReplyPushLoop` receives one incident and makes one successful lead-pane send. Existing
|
||||
failed-ticket notice content may share that same combined send.
|
||||
12. Exhaustion still wins when a line matches both patterns. A simultaneous exhaustion quarantine
|
||||
also wins in spawn errors and fleet views.
|
||||
13. After the 60-second cool-off, spawn is allowed again. A new incident needs two fresh errors.
|
||||
14. `healthCoverage` has the same value before and after this change for the same health config.
|
||||
15. `fleetd.example.yaml` explains `errorPattern`, the legacy fallback, 2/60/60 policy, and the
|
||||
difference between cool-off and exhaustion quarantine.
|
||||
16. `CLAUDE.md` tells leads how `fleet_profiles` and `fleet_list` report cool-off. The lead later
|
||||
applies the matching wiki updates and runs the documented byte-sync check.
|
||||
17. The developer records the new end-to-end test failing before implementation, then passing. The
|
||||
focused suites and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
Unit 5 starts only after Units 1 to 4 are merged or rebased into its branch. It also starts after the
|
||||
#234 defect 2 fix lands, because both areas touch the same target-resolution control path.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact commits used for Units 1 to 4 and #234.
|
||||
- The startup coverage line with one configured and one legacy-default profile.
|
||||
- The red test output before the implementation and its green result after.
|
||||
- The explicit and automatic spawn refusal text.
|
||||
- Sample `fleet_list` and `fleet_profiles` JSON for cool-off and exhaustion.
|
||||
- The number and text of lead-pane sends in the real-path test.
|
||||
- The focused test commands and final `mvn clean install` result.
|
||||
- The exact `CLAUDE.md` change and the wiki edits the lead must apply.
|
||||
|
||||
## File ownership summary
|
||||
|
||||
| Area | Unit | Shared edit risk |
|
||||
|---|---:|---|
|
||||
| Completion classification | 1 | Only Unit 1 edits `CompletionResolver` and its test. |
|
||||
| Correlation and cool-off state | 2 | New files only. |
|
||||
| Lead push scheduling | 3 | Only Unit 3 edits `ReplyPushLoop` and its test. |
|
||||
| Member terminal state | 4 | Only Unit 4 edits `MemberSession`, `SessionManager`, and their test. |
|
||||
| Main composition, config, placement, MCP views, shipped prompt | 5 | Only Unit 5 edits `Fleetd`, `FleetConfig`, `CompositePeerLauncher`, placement context, `FleetMcp`, and `CLAUDE.md`. |
|
||||
| Wiki propagation | Lead after Unit 5 | Workers do not commit the wiki submodule. |
|
||||
|
||||
## What I would not build
|
||||
|
||||
1. **Do not reuse `BackendQuarantine` for outages.** Its repeat call restarts a long credential
|
||||
quarantine. Its fields and errors say “exhausted”. That is wrong for a short outage.
|
||||
2. **Do not merge the exhaustion and generic error patterns.** Exhaustion must win because it has a
|
||||
different policy and duration.
|
||||
3. **Do not group by error string.** One outage can produce different text. The shared operational
|
||||
limit is the credential.
|
||||
4. **Do not mark a profile unusable until config changes.** The current classifier cannot safely
|
||||
tell a permanent malformed request from a transient service fault. A permanent state would need
|
||||
a stronger error taxonomy first.
|
||||
5. **Do not quarantine on the first generic backend error.** That would turn one bad request or one
|
||||
false pattern match into a fleet-wide capacity loss.
|
||||
6. **Do not add this to `FleetHealthMonitor`.** The monitor samples slow member health. The exact
|
||||
backend event already exists at completion resolution, and moving it to polling would lose type
|
||||
and time.
|
||||
7. **Do not add another direct lead injector.** `ReplyPushLoop` already owns status gating,
|
||||
per-lead coalescing, retry bounds, and heartbeat stand-down.
|
||||
8. **Do not change `healthCoverage` to `full`.** That field still means a webhook notification sink
|
||||
exists for periodic health. A backend outage nudge does not make every health event visible.
|
||||
9. **Do not persist incident history across daemon restart in this work.** Existing exhaustion
|
||||
quarantine is also in memory. A 60-second state does not justify a new durable store.
|
||||
10. **Do not build work recovery.** The PR-body survival story proves why checkpoint-first work is
|
||||
useful, but these tickets are about detection, capacity, and signalling.
|
||||
11. **Do not remove the legacy `API Error:` fallback in the first release.** Doing so would turn an
|
||||
unedited config back into a false successful completion. Report it as degraded coverage instead.
|
||||
12. **Do not reorder or add the old `visibleTurn` fallback from #201.** #211 already implemented the
|
||||
narrow raw-scrape fallback at `CompletionResolver.classifyRawScrapeFallback`.
|
||||
|
||||
## Riskiest assumption and cheapest experiment
|
||||
|
||||
The riskiest assumption is that a configured error regex means “the backend failed this turn”. The
|
||||
current code and test already show the counterexample: a worker may quote `API Error:` while writing
|
||||
a valid report. Two such false matches on one credential would now remove capacity for 60 seconds.
|
||||
|
||||
The cheapest experiment is a replay corpus before Unit 5 ships:
|
||||
|
||||
1. Save the full pane text from the measured 2026-09-01 outage.
|
||||
2. Produce one safe failure per backend with a disposable invalid endpoint or request.
|
||||
3. Save one valid member report which quotes each error line.
|
||||
4. Replay all samples through the real `CompletionResolver` test fixture.
|
||||
5. Require outage samples to match and quoted-report samples not to match after assistant-block
|
||||
extraction and baseline checks.
|
||||
|
||||
This costs no outage deployment and no real sleep. If quoted reports still match, narrow the profile
|
||||
patterns before enabling correlation. Do not raise the threshold to hide a bad classifier.
|
||||
|
||||
## Sequencing with three developers
|
||||
|
||||
First wave:
|
||||
|
||||
1. Developer A: Unit 1, typed classification.
|
||||
2. Developer B: Unit 2, credential outage policy.
|
||||
3. Developer C: Unit 3, lead outage nudge.
|
||||
|
||||
As soon as one slot is free, start Unit 4. It is file-disjoint from every first-wave unit. Merge and
|
||||
review Units 1 to 4 independently.
|
||||
|
||||
Start Unit 5 only after all four foundations and #234 are available. Unit 5 is the only high-conflict
|
||||
integration branch, so no other active unit should touch its file list.
|
||||
|
||||
## Checks performed for this refinement
|
||||
|
||||
- Read issue #201 and issue #227 through their Gitea pages. Both showed zero comments.
|
||||
- Read the source and tests named in the evidence section.
|
||||
- Ran `git status --short --branch`; the branch was clean before this document was added.
|
||||
- Ran `git log --oneline -12` to identify the branch base.
|
||||
- I did not run Maven because this change adds only a design document.
|
||||
- Rendered both Mermaid blocks with `npx @mermaid-js/mermaid-cli`; both commands succeeded.
|
||||
@@ -192,7 +192,7 @@ Deliberately small; every one maps to a failure mode we have actually hit.
|
||||
| `fleet_send_duration_seconds` | histogram | delegated turn latency |
|
||||
| `fleet_replies_total{path}` | counter | path ∈ rendezvous\|inbox — how often a reply strands (CB-307's whole reason to exist) |
|
||||
| `fleet_inbox_depth{target}` | gauge | undrained replies; steady-state should be 0 |
|
||||
| `fleet_push_nudges_total{outcome}` | counter | outcome ∈ delivered\|exhausted — a rising `exhausted` means the primary is not draining |
|
||||
| `fleet_push_nudges_total{outcome}` | counter | outcome ∈ sent\|exhausted — a rising `exhausted` means the primary is not draining. `sent` was called `delivered` until fleetd #365; it counts the herdr paste-and-submit call returning, never a confirmation the pane read it |
|
||||
| `fleet_spawns_total{kind,outcome}` | counter | outcome ∈ ready\|timeout\|guard_rejected; per peer kind (CB-402) |
|
||||
| `fleet_sessions{state}` | gauge | SPAWNING/READY/BUSY/DONE census |
|
||||
| `fleet_herdr_calls_total{method,outcome}` | counter | socket health — the dependency everything rests on |
|
||||
|
||||
+123
-324
@@ -1,311 +1,162 @@
|
||||
# MCP Contract — `fleetd`'s unified gateway
|
||||
# MCP flows and error model — `fleetd`
|
||||
|
||||
> **Status: 🔴 HISTORICAL DESIGN — do NOT use as the tool reference.** Written 2026-07-14, before
|
||||
> any MCP code existed. The system shipped and this page never caught up, so **its tool names,
|
||||
> parameter names and REST paths are wrong today**. Audited 2026-08-17; the specific drift:
|
||||
> **What this page is.** The **flows**: how a delegation, a clarification, a detached task and a
|
||||
> silent member each travel through `fleetd`. These shapes are what shipped, and they are hard to
|
||||
> read off the code because they span the MCP face, the rendezvous registry, the `Injector` and
|
||||
> herdr.
|
||||
>
|
||||
> - **Tools it names that do not exist:** `fleet_read`, `fleet_cancel`.
|
||||
> - **Shipped tools it omits:** `fleet_poll`, `fleet_ack`, `fleet_profiles`, `fleet_whoami`.
|
||||
> - **Parameter names are wrong nearly everywhere** — it says `message`/`target`/`timeout_seconds`/
|
||||
> `block` where the code takes `content`/`sessionId`/`timeoutMs`/`wait`; `text` where
|
||||
> `fleet_reply` takes `content`; `target` where `fleet_stop` takes `paneId`.
|
||||
> - **REST paths are wrong:** it says `POST /workers` and `DELETE /workers/{paneId}`; the daemon
|
||||
> serves `POST /members` and `DELETE /members/{paneId}`.
|
||||
> **What this page is NOT: a tool reference.** It deliberately holds no tool catalogue, no
|
||||
> parameter tables and no REST paths. **The live MCP schema is the authority** — each tool's own
|
||||
> description and parameters, as mounted — with the intent→tool table in `CLAUDE.md` as the short
|
||||
> form.
|
||||
>
|
||||
> **The authoritative tool surface is the live MCP schema** (each tool's own description and
|
||||
> parameters, as mounted), with the intent→tool table in `CLAUDE.md` as the short form. Both were
|
||||
> checked against `mcp/FleetMcp.java` on 2026-08-17 and are accurate.
|
||||
> That absence is the fix for fleetd #114 (CB-609), and it is worth stating why. This page used to
|
||||
> carry a full tool catalogue written in July 2026, before any MCP code existed. The code shipped;
|
||||
> the page did not follow. By August it named two tools that do not exist, omitted five that do,
|
||||
> had the wrong name for nearly every parameter, pointed at REST paths the daemon does not serve,
|
||||
> and — worst — still described an identity model (*"any connection that does not map to a known
|
||||
> worker is treated as a primary"*) that was a real privilege bug, fixed since by the ancestry
|
||||
> walk in fleetd #161. Every one of those errors is the same error: **a second, hand-maintained
|
||||
> copy of something the code already states**. So the second copy is gone rather than corrected.
|
||||
> Only the flows remain, because a flow is a shape rather than a name, and shapes are what this
|
||||
> page was ever good for.
|
||||
>
|
||||
> What is still worth reading here is **§6 — the flows and the error model** (rendezvous,
|
||||
> `fleet_ask`, detached delivery, the turn-done fallback). The shapes it describes are the ones
|
||||
> that shipped; only the names around them drifted. Rewriting this page is tracked as **CB-609**.
|
||||
|
||||
`fleetd` is the **sole communication gateway** for every Claude session in the bridge. Both
|
||||
the **primary** (Opus, on subscription) and every **worker** (off-subscription Claude Code)
|
||||
mount the *same* MCP server with a single `claude mcp add` line, and talk only through its
|
||||
tools. No Claude session ever addresses a broker, a peer, or the network directly.
|
||||
|
||||
This document defines every MCP tool that face must expose, who may call it, its blocking
|
||||
semantics, and how it maps onto the code already in the tree.
|
||||
> The names that do appear below are checked by `McpContractDocTest`, which fails if this page
|
||||
> names a `fleet_*` tool the server does not register. That test is the whole reason it is safe to
|
||||
> write a tool name here at all.
|
||||
|
||||
---
|
||||
|
||||
## 1. Design constraints (non-negotiable)
|
||||
## 1. Rendezvous flows
|
||||
|
||||
These come from the project's core invariants and bound every decision below.
|
||||
### 1.1 Delegation — happy path
|
||||
|
||||
1. **One server, both roles.** The primary and all workers mount an identical server. The
|
||||
catalog must serve both, and `fleetd` must decide *who is calling* from the connection —
|
||||
never from a caller-supplied argument that could be spoofed.
|
||||
2. **Subscription-safe by construction.** No MCP tool ever reads, sets, or forwards
|
||||
`ANTHROPIC_BASE_URL`. Mounting the bridge cannot move a session off subscription.
|
||||
Enforced today by [`SubscriptionGuard`](1-Architecture).
|
||||
3. **Blocking rendezvous, no busy-poll.** The primary consumes a worker's reply through a
|
||||
*single* MCP call that `fleetd` holds open — never a cross-turn poll loop that would burn
|
||||
subscription quota.
|
||||
4. **Status-gated delivery.** Anything that puts text into a worker flows through the existing
|
||||
[`Injector`](1-Architecture): delivered only when the worker is `idle`/`blocked`, at most
|
||||
one message per turn.
|
||||
5. **`fleetd` owns policy; herdr owns PTYs.** MCP tools express *intent*; `fleetd`
|
||||
translates it into guard checks, rendezvous bookkeeping, and herdr `agent.*` calls.
|
||||
|
||||
---
|
||||
|
||||
## 2. Topology
|
||||
|
||||
Both faces live in the one daemon. The **north face** is MCP (this document); the **south
|
||||
face** is the herdr Unix socket. REST/SSE remains only for non-Claude clients and dashboards.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
OPUS["Opus — primary<br/>(Claude Code, env CLEAN)<br/>MCP client"]
|
||||
subgraph BD["fleetd — standalone daemon"]
|
||||
MCP["MCP server (north face)<br/>fleet_send · fleet_reply<br/>fleet_ask · fleet_status · lifecycle"]
|
||||
RDV["rendezvous registry<br/>(blocking-call waiters)"]
|
||||
INJ["Injector + StatusPoller<br/>(status-gated writer)"]
|
||||
SOCK["herdr socket client (south face)"]
|
||||
MCP --> RDV
|
||||
RDV --> INJ
|
||||
INJ --> SOCK
|
||||
MCP --> SOCK
|
||||
end
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
|
||||
OPUS -->|"fleet_send (blocks)"| MCP
|
||||
W -.->|"fleet_reply / fleet_ask"| MCP
|
||||
SOCK -->|"agent.start · agent.send<br/>agent.get · pane.close"| HERDR
|
||||
HERDR -->|"drives PTY"| W
|
||||
|
||||
classDef ext fill:#2b6cb0,stroke:#1a365d,color:#ffffff;
|
||||
classDef core fill:#2f855a,stroke:#22543d,color:#ffffff;
|
||||
class OPUS,W ext
|
||||
class MCP,RDV,INJ,SOCK core
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. Identity & addressing
|
||||
|
||||
Because the same server is mounted by everyone, `fleetd` resolves the caller's role on every
|
||||
request — this is the linchpin of the whole contract and has no code yet.
|
||||
|
||||
- **Workers are known.** `fleetd` spawns every worker
|
||||
([`WorkerService`](1-Architecture)) and records its herdr session UUID / `terminal_id` on
|
||||
the returned [`Agent`]. When a call arrives on a connection that maps to a known worker,
|
||||
the caller is *that* worker — so **workers never pass a target**; routing is implicit.
|
||||
- **The primary is "not a worker".** Any connection that does not map to a known worker is
|
||||
treated as a primary. It addresses workers **explicitly** by `target` — a session UUID,
|
||||
a `terminal_id`, or a friendly `profile` name.
|
||||
- **Turn correlation.** A blocking `fleet_send` registers a *waiter* keyed by worker
|
||||
identity. A worker's later `fleet_reply` / `fleet_ask` on the same identity resolves that
|
||||
waiter. A `turn_id` is minted per exchange so a clarification round-trip
|
||||
(§6.2) rejoins the right turn.
|
||||
|
||||
---
|
||||
|
||||
## 4. Transport
|
||||
|
||||
`fleetd` is a long-lived daemon serving **multiple** concurrent clients (one primary + N
|
||||
workers), so a per-client stdio child is the wrong shape. The recommended transport is
|
||||
**streamable-HTTP / SSE** on the same bind as the REST face:
|
||||
|
||||
```bash
|
||||
# identical on primary and every worker
|
||||
claude mcp add --transport http fleetd http://127.0.0.1:8080/mcp
|
||||
```
|
||||
|
||||
This adds an MCP-server dependency the pom does not yet carry. See [Open decisions](#10-open-decisions).
|
||||
|
||||
---
|
||||
|
||||
## 5. Tool catalog
|
||||
|
||||
| Tool | Caller | Blocks? | Backing (exists today?) |
|
||||
|---|---|---|---|
|
||||
| [`fleet_send`](#fleet_send) | primary | yes (default) | `Injector.enqueue` ✅ · rendezvous registry ❌ (CB-104) |
|
||||
| [`fleet_reply`](#fleet_reply) | worker | no | rendezvous ❌ · pane injection via `Injector` ✅ |
|
||||
| [`fleet_ask`](#fleet_ask) | worker | yes | reverse rendezvous ❌ |
|
||||
| [`fleet_status`](#fleet_status) | either | no | `AgentControl.status` ✅ · `Injector.activeTargets` ✅ |
|
||||
| [`fleet_spawn`](#lifecycle) | primary | no | `WorkerService.spawn` ✅ (`POST /workers`) |
|
||||
| [`fleet_list`](#lifecycle) | either | no | `WorkerService.list` ✅ (`/agents`) |
|
||||
| [`fleet_stop`](#lifecycle) | primary | no | `WorkerService.stop` ✅ (`DELETE /workers/{paneId}`) |
|
||||
| [`fleet_read`](#fleet_read) | primary | no | `AgentControl.read` ✅ |
|
||||
| [`fleet_cancel`](#fleet_cancel) | primary | no | — ❌ (future) |
|
||||
|
||||
### Core: delegation & rendezvous
|
||||
|
||||
#### `fleet_send`
|
||||
*(primary → worker — the headline tool, CB-104)*
|
||||
|
||||
- **Params:** `message` (required); `target` (optional — defaults to the sole worker / default
|
||||
profile); `timeout_seconds` (default 600); `block` (default `true`); `auto_spawn`
|
||||
(default `true`); `turn_id` (optional — supplied when answering a worker's `fleet_ask`).
|
||||
- **Blocking (`block:true`):** enqueue `message` via the `Injector`, then hold the call open
|
||||
until exactly one of:
|
||||
- worker calls `fleet_reply` → `{ outcome:"reply", text }`
|
||||
- worker calls `fleet_ask` → `{ outcome:"question", text, turn_id }`
|
||||
- worker's `agent_status` reaches done/idle with no reply → `{ outcome:"turn_done", text:<terminal tail> }`
|
||||
- deadline elapses → `{ outcome:"timeout" }`
|
||||
- worker gone → error `worker_gone`
|
||||
- **Detached (`block:false`):** enqueue and return `{ outcome:"dispatched", dispatch_id }`
|
||||
immediately. The eventual reply is injected into the primary's idle pane (§6.3), or drained
|
||||
via `fleet_status` on a split-host primary.
|
||||
|
||||
#### `fleet_reply`
|
||||
*(worker → primary)*
|
||||
|
||||
- **Params:** `text` (required); `final` (default `true`).
|
||||
- **Behavior:** resolve the primary waiter registered against this worker with `text`. If no
|
||||
waiter exists (detached delegation), `fleetd` **injects the primary's idle pane** instead.
|
||||
Returns `{ delivered:true, mode:"resolved"|"injected" }`. No `target` — identity is implicit.
|
||||
|
||||
#### `fleet_ask`
|
||||
*(worker → primary — the reverse rendezvous)*
|
||||
|
||||
- **Params:** `question` (required); `timeout_seconds`.
|
||||
- **Behavior:** blocks the *worker's* call. Surfaces the question to the primary (resolving its
|
||||
open `fleet_send` with `outcome:"question"`, or injecting its pane). When the primary
|
||||
answers — a `fleet_send` carrying the matching `turn_id` — that unblocks this call and
|
||||
returns `{ answer }` to the worker, which continues **in the same turn**.
|
||||
|
||||
### Worker lifecycle
|
||||
<a id="lifecycle"></a>
|
||||
Thin adapters over [`WorkerService`](1-Architecture) — parity with the existing REST routes.
|
||||
|
||||
- **`fleet_spawn`** — `{ profile? }` → worker view (`sessionId`, `terminalId`, `paneId`,
|
||||
`status`). Guard-checked; a boundary breach returns error `subscription_boundary` (the
|
||||
REST `403`).
|
||||
- **`fleet_list`** — no params → all workers + `agent_status`. Read-only, either role.
|
||||
- **`fleet_stop`** — `{ target }` → tears down the pane and its dedicated tab. Idempotent.
|
||||
|
||||
### Observability
|
||||
|
||||
#### `fleet_status`
|
||||
*(either role — the README's 4th named tool)*
|
||||
|
||||
- **Params:** `target?`.
|
||||
- **Behavior:** per-worker `agent_status`, queue depth (`Injector.activeTargets`), whether a
|
||||
rendezvous is open, and ids. For the *calling* session it also reports/drains **pending
|
||||
messages addressed to me** — the path a split-host primary's `Stop`-hook uses to wake and
|
||||
collect replies without being injectable. Read-only, non-blocking.
|
||||
|
||||
#### `fleet_read`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`; `source` ∈ `visible | recent | recent_unwrapped | detection`.
|
||||
- **Behavior:** returns the worker's terminal text so the primary can peek at a *detached*
|
||||
worker's progress. Adapter over `AgentControl.read`.
|
||||
|
||||
### Control (future)
|
||||
|
||||
#### `fleet_cancel`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`. Interrupt the worker's current turn / abandon the rendezvous. No
|
||||
backing code yet.
|
||||
|
||||
---
|
||||
|
||||
## 6. Rendezvous flows
|
||||
|
||||
### 6.1 Delegation — happy path
|
||||
|
||||
One blocking call, zero polls.
|
||||
One blocking call, zero polls. The lead's call is held open by `fleetd` until the member answers.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary (Opus)
|
||||
participant B as fleetd (MCP + Injector)
|
||||
participant P as "Lead (primary)"
|
||||
participant B as "fleetd (MCP + Injector)"
|
||||
participant H as herdr
|
||||
participant W as Worker (Claude)
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B->>B: register waiter(w)
|
||||
B->>H: agent.send(w, "do X") (idle window)
|
||||
H-->>W: prompt injected
|
||||
W->>W: works the turn
|
||||
W->>B: fleet_reply("result")
|
||||
B->>B: resolve waiter(w)
|
||||
B-->>P: { outcome:"reply", text:"result" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B->>B: "register waiter(sessionId)"
|
||||
B->>H: "agent.send — only in an injectable window"
|
||||
H-->>W: "prompt injected"
|
||||
W->>W: "works the turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B->>B: "resolve waiter"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.2 Clarification — reverse rendezvous (`fleet_ask`)
|
||||
**The cap that matters:** a blocking `fleet_send` is bounded by the *caller's own* MCP client
|
||||
timeout, about 60 seconds — not by the task. Anything slower than that must use the detached flow
|
||||
in §1.3, or the lead's call returns while the member is still working.
|
||||
|
||||
The worker pauses mid-turn to ask; the primary answers; the worker resumes in the same turn.
|
||||
### 1.2 Clarification — reverse rendezvous
|
||||
|
||||
The member pauses mid-turn to ask, the lead answers, and the member resumes **the same turn** with
|
||||
its context intact.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>B: fleet_ask("which config?") — worker blocks
|
||||
B-->>P: { outcome:"question", text:"which config?", turn_id }
|
||||
P->>B: fleet_send("config.yaml", target=w, turn_id) — blocks again
|
||||
B-->>W: resolve fleet_ask → { answer:"config.yaml" }
|
||||
W->>W: resumes same turn
|
||||
W->>B: fleet_reply("done")
|
||||
B-->>P: { outcome:"reply", text:"done" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B-->>W: "content injected"
|
||||
W->>B: "fleet_ask{question} — member blocks"
|
||||
B-->>P: "{ outcome: question, turnId }"
|
||||
P->>B: "fleet_send{turnId, content} — answers THIS turn"
|
||||
B-->>W: "fleet_ask returns the answer"
|
||||
W->>W: "resumes the same turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.3 Detached delegation — pane injection
|
||||
**Answer with `turnId`, never `sessionId`.** A `sessionId` send starts a new turn; it does not
|
||||
resolve the waiting `fleet_ask`.
|
||||
|
||||
The primary does not block; the reply arrives later in its idle pane.
|
||||
**The window is about 55 seconds and no nudge extends it.** So never brief a member to "ask me":
|
||||
decide before delegating, or give the member an explicit default to fall back on.
|
||||
|
||||
### 1.3 Detached delegation — the lead does not block
|
||||
|
||||
The lead gets a ticket immediately and collects the answer later. This is the flow for any real
|
||||
task, because of the ~60s cap in §1.1.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w, block=false)
|
||||
B-->>P: { outcome:"dispatched", dispatch_id }
|
||||
P->>P: continues its own work
|
||||
W->>B: fleet_reply("result")
|
||||
Note over B: no waiter → detached path
|
||||
B->>B: Injector.enqueue(primary_pane, "result")
|
||||
B-->>P: injected into idle pane (status-gated)
|
||||
P->>B: "fleet_send{sessionId, content, wait:false}"
|
||||
B-->>P: "accepted — ticket"
|
||||
P->>P: "continues its own work"
|
||||
W->>B: "fleet_reply{content}"
|
||||
Note over B: "no waiter is blocked — the reply is held"
|
||||
B->>B: "nudge the lead's own pane (status-gated)"
|
||||
P->>B: "fleet_poll{ticket}"
|
||||
B-->>P: "the member's report"
|
||||
P->>B: "fleet_ack{target, msgId}"
|
||||
```
|
||||
|
||||
### 6.4 Uncooperative worker — turn-done fallback
|
||||
A terminal ticket nudges the lead's pane by itself, so a detached task does not need watching. The
|
||||
nudge needs an injectable lead pane and is capped, so it is a convenience rather than a guarantee.
|
||||
|
||||
A worker that never calls `fleet_reply` still returns a result: `fleetd` reads its terminal
|
||||
tail when the turn completes.
|
||||
### 1.4 The member never replies — turn-done fallback
|
||||
|
||||
A member that ends its turn without `fleet_reply` still produces something: `fleetd` reads its
|
||||
pane tail. This is a **fallback, not a channel** — it is lossy in three separate ways, and every
|
||||
one of them has produced a wrong answer in practice.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>W: works, never calls fleet_reply
|
||||
B->>B: StatusPoller sees agent_status → idle/done
|
||||
B->>B: AgentControl.read(w, "recent")
|
||||
B-->>P: { outcome:"turn_done", text:<terminal tail> }
|
||||
P->>B: "fleet_send — blocks or detaches"
|
||||
B-->>W: "content injected"
|
||||
W->>W: "works, never calls fleet_reply"
|
||||
B->>B: "StatusPoller sees the turn end"
|
||||
B->>B: "read the pane tail"
|
||||
B->>B: "classify: exhausted? echoed brief? real report?"
|
||||
B-->>P: "{ outcome: turn_done } or a named failure"
|
||||
```
|
||||
|
||||
The three ways it goes wrong, and what each looks like now:
|
||||
|
||||
| What happened | What the lead used to get | What it gets today |
|
||||
|---|---|---|
|
||||
| The report is longer than the scrape window | The **end** silently cut off | Still clipped, but marked partial |
|
||||
| The member never started — spent credential | The lead's **own brief** echoed back as a report | A named failure: backend exhausted |
|
||||
| The member is simply slow | A tail of work in progress | Unchanged — read it as a hint, not a result |
|
||||
|
||||
The echoed-brief case is the one to remember: it reads as a long, on-topic report with nothing in
|
||||
it from the member. It is suppressed now, but the general rule stands — **check the member's
|
||||
worktree with `git log` before believing a report you did not watch arrive.**
|
||||
|
||||
---
|
||||
|
||||
## 7. Status gating
|
||||
## 2. Status gating
|
||||
|
||||
Delivery only happens in a safe window. This is the state machine the `Injector` already
|
||||
enforces via `AgentStatus.injectable()`; MCP `fleet_send` is simply its producer.
|
||||
Delivery only happens in a safe window. `fleet_send` is a producer for the `Injector`, which
|
||||
already enforces this through `AgentStatus.injectable()`.
|
||||
|
||||
```mermaid
|
||||
stateDiagram-v2
|
||||
[*] --> IDLE
|
||||
IDLE --> WORKING: message delivered / picks up
|
||||
WORKING --> IDLE: turn done
|
||||
WORKING --> BLOCKED: awaits input
|
||||
BLOCKED --> WORKING: input delivered
|
||||
IDLE --> UNKNOWN: detection glitch
|
||||
BLOCKED --> UNKNOWN: detection glitch
|
||||
UNKNOWN --> IDLE: re-detected
|
||||
IDLE --> WORKING: "message delivered, picked up"
|
||||
WORKING --> IDLE: "turn done"
|
||||
WORKING --> BLOCKED: "awaits input"
|
||||
BLOCKED --> WORKING: "input delivered"
|
||||
IDLE --> UNKNOWN: "detection glitch"
|
||||
BLOCKED --> UNKNOWN: "detection glitch"
|
||||
UNKNOWN --> IDLE: "re-detected"
|
||||
|
||||
note right of IDLE
|
||||
injectable — deliver head of FIFO
|
||||
@@ -321,69 +172,17 @@ stateDiagram-v2
|
||||
end note
|
||||
```
|
||||
|
||||
At most one message is delivered per turn: after a send the `Injector` waits for a `WORKING`
|
||||
pickup before delivering the next, with a `PICKUP_GRACE_POLLS` fallback for turns faster than
|
||||
the poll interval. A herdr `events.subscribe` stream can later replace the sampling without
|
||||
touching this state machine.
|
||||
**At most one message per turn.** After a send, the `Injector` waits for a `WORKING` pickup before
|
||||
delivering the next, with a grace-poll fallback for turns that finish faster than the poll
|
||||
interval.
|
||||
|
||||
---
|
||||
Two consequences a lead feels directly:
|
||||
|
||||
## 8. Error model
|
||||
- **A second send to a busy member never lands.** It reports as queued and times out. The member
|
||||
is fine; the message simply waits, and then restarts the member when it next goes idle.
|
||||
- **A spawned member is not deliverable until it has mounted the MCP.** Until then a send waits on
|
||||
that gate for about 60 seconds and then fails without ever reaching the pane.
|
||||
|
||||
| Condition | `fleet_send` result | Notes |
|
||||
|---|---|---|
|
||||
| Worker replies | `{ outcome:"reply" }` | normal |
|
||||
| Worker asks | `{ outcome:"question", turn_id }` | answer with `fleet_send(turn_id)` |
|
||||
| Turn ends, no reply | `{ outcome:"turn_done" }` | terminal tail as text |
|
||||
| Deadline elapsed | `{ outcome:"timeout" }` | message may still be queued/delivered |
|
||||
| Worker vanished | error `worker_gone` | `Injector.drop` fails the queued future |
|
||||
| Guard breach on spawn | error `subscription_boundary` | REST `403` parity |
|
||||
| Delivery failed at herdr | error, message dropped | poisoned message not left blocking the FIFO |
|
||||
|
||||
`fleet_reply` from a worker with no open waiter is **not** an error — it falls through to
|
||||
detached pane injection (§6.3).
|
||||
|
||||
---
|
||||
|
||||
## 9. Mapping to existing code
|
||||
|
||||
The MCP face is a thin adapter layer; nearly every capability already exists behind the REST
|
||||
seam. Only the **rendezvous registry** and the **caller-identity resolver** are new.
|
||||
|
||||
| MCP tool | Existing collaborator | New work |
|
||||
|---|---|---|
|
||||
| `fleet_send` | `Injector.enqueue`, `AgentControl.send` | waiter registry, timeout, outcome mux (CB-104) |
|
||||
| `fleet_reply` / `fleet_ask` | `Injector` (pane injection) | reverse rendezvous, identity resolver |
|
||||
| `fleet_status` | `AgentControl.status`, `Injector.activeTargets` | pending-drain projection |
|
||||
| `fleet_spawn` / `list` / `stop` | `WorkerService.{spawn,list,stop}` | MCP adapter only |
|
||||
| `fleet_read` | `AgentControl.read` | MCP adapter only |
|
||||
|
||||
Because the REST routes in `FleetApp` already exercise the collaborators, MCP tools are
|
||||
validated by **parity** against those routes, not by re-testing behavior.
|
||||
|
||||
---
|
||||
|
||||
## 10. Open decisions
|
||||
|
||||
1. **`fleet_ask` direction.** This page defines it as *worker-asks-primary* (a genuine reverse
|
||||
channel, matching the "inject the primary's pane" language). The alternative — a synonym for
|
||||
a blocking primary→worker send — is weaker and produces different plumbing. **Recommend
|
||||
worker-asks-primary.**
|
||||
2. **Detached delivery shape.** A `block:false` param on `fleet_send` (keeps the catalog
|
||||
small) vs. a separate `fleet_dispatch` tool. **Recommend the param.**
|
||||
3. **Auto-spawn on send.** `fleet_send` provisions a worker per profile when none exists
|
||||
(simplest primary UX) vs. requiring an explicit `fleet_spawn` first. **Recommend
|
||||
auto-spawn, defaulting on.**
|
||||
4. **Transport & SDK.** Streamable-HTTP/SSE co-located with the REST bind (recommended) vs.
|
||||
stdio. Requires choosing a Java MCP server SDK and adding it to the pom.
|
||||
|
||||
---
|
||||
|
||||
## 11. Implementation staging
|
||||
|
||||
- **CB-104** — blocking `fleet_send` + rendezvous registry + caller-identity resolver
|
||||
(the producer that finally drives the inert `StatusPoller`).
|
||||
- **CB-1xx** — `fleet_reply` / `fleet_ask` reverse rendezvous + detached pane injection.
|
||||
- **CB-1xx** — lifecycle + observability adapters (`fleet_spawn/list/stop/status/read`).
|
||||
- **CB-1xx** — transport wiring + `claude mcp add` docs; parity tests vs. REST.
|
||||
- **Later** — `fleet_cancel`; swap `StatusPoller` for herdr `events.subscribe`.
|
||||
`UNKNOWN` is deliberately neither injectable nor a pickup. A pane whose status cannot be read is
|
||||
not a pane that is safe to write to — see fleetd #176 for what happens when a gate treats an
|
||||
unreadable pane as a ready one.
|
||||
|
||||
+252
-17
@@ -88,6 +88,47 @@ bind:
|
||||
# backoffMs: 60000
|
||||
# quietNudgeCap: 3
|
||||
|
||||
# Lead rollover (fleetd #480): replace a lead session that has decided it is ready to be replaced,
|
||||
# without an operator doing it by hand. A lead writes a handover file, then asks fleetd to clear its
|
||||
# own pane and bootstrap a fresh session against that file.
|
||||
#
|
||||
# Opt-in on purpose — it clears the lead's own pane on request, so upgrading the daemon must never
|
||||
# acquire that ability for you. Absent block = feature off, and nothing is constructed at all. Even
|
||||
# once present, nothing but an explicit confirm() call — one that passes every check — can ever
|
||||
# cause a /clear: there is no recurring timer, heartbeat or scheduler anywhere in this feature that
|
||||
# fires one on its own initiative. confirm() itself is called FROM the calling lead's own turn, so
|
||||
# it cannot clear the pane inline (that pane is still WORKING); instead it schedules a one-shot
|
||||
# continuation that waits for the SAME confirm() call's turn to end, then does the actual work. See
|
||||
# dev.ltms.fleet.lead.LeadRollover's class javadoc for the exact order (fleetd #480 correction).
|
||||
#
|
||||
# handoverPath: REQUIRED when this block is present — where the handover file a fresh lead session
|
||||
# reads must live. No default (an operator-specific path); a present block with no
|
||||
# handoverPath refuses to start. May be relative: it then resolves against the
|
||||
# CALLING lead's own fleet.leaders.<name>.cwd (falling back to the daemon's own
|
||||
# working directory when that lead has none configured) — never against whatever
|
||||
# directory the daemon process happens to have been started in. An absolute path is
|
||||
# used unchanged. Prefer an absolute path if the daemon and the lead's pane might not
|
||||
# share a working directory (fleetd #480 follow-up).
|
||||
# requireOperatorConfirm: true # default true — confirm() refuses unless the caller also passes
|
||||
# # operatorConfirmed: true
|
||||
# maxDocAgeSeconds: 3600 # default 3600 — refuse a handover file older than this
|
||||
# turnSettleSeconds: 20 # default 20 — how long the deferred roll waits for the CALLING
|
||||
# # lead's own turn to end (its pane to report injectable again)
|
||||
# # before sending /clear at all. If this elapses, /clear is NEVER
|
||||
# # sent — a lead that never goes idle is still doing real work.
|
||||
# clearSettleSeconds: 20 # default 20 — how long to wait for the pane to become injectable
|
||||
# # again AFTER /clear before giving up (never sends bootstrapText
|
||||
# # if this elapses). A separate, second wait from turnSettleSeconds.
|
||||
# bootstrapText: "..." # default names the RESOLVED (absolute) handoverPath — sent to
|
||||
# # the lead once its pane settles after /clear
|
||||
# leadRollover:
|
||||
# handoverPath: /path/to/handover.md
|
||||
# requireOperatorConfirm: true
|
||||
# maxDocAgeSeconds: 3600
|
||||
# turnSettleSeconds: 20
|
||||
# clearSettleSeconds: 20
|
||||
# bootstrapText: "Fresh lead session: read the handover file and carry on."
|
||||
|
||||
# Fleet health detection is dormant unless enabled (CB-573). It reads one whole-fleet agent list
|
||||
# per tick.
|
||||
# intervalSeconds → how often a tick runs (default 30). ENFORCED floor of 15: the code computes
|
||||
@@ -110,6 +151,14 @@ bind:
|
||||
# notifications:
|
||||
# mode: disabled
|
||||
|
||||
# Idle-sleep guard: while at least one member is live, hold an OS-level assertion against idle
|
||||
# sleep (macOS only — a `caffeinate -i` child; a no-op elsewhere or if caffeinate is missing), so
|
||||
# an unattended host does not idle-sleep out from under a member's long turn. Unlike health/
|
||||
# configReload above, this is ON BY DEFAULT — omitting the block entirely leaves it enabled, the
|
||||
# same as `enabled: true`. Uncomment only to turn it off:
|
||||
# idleSleepGuard:
|
||||
# enabled: false
|
||||
|
||||
# herdr Unix socket. Omit to use the client default
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
@@ -172,7 +221,11 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# skills/MCP/hooks. Omit to leave the worker on the host default.
|
||||
# parityOverlay → repo-relative paths copied primary→worktree so a worker in a provisioned
|
||||
# worktree sees the same local config (CB-301-ext). Omit for the default set:
|
||||
# [.env, .envrc]. (.claude/settings.local.json is NOT in the default — it
|
||||
# [.env] only (CB-148). .envrc is left out of the default on purpose: it is
|
||||
# executable shell that direnv runs on every cd, so copying it carries
|
||||
# behaviour into the worker, not just values, unlike .env. An operator who
|
||||
# wants it copied can still write parityOverlay: [.env, .envrc] explicitly.
|
||||
# (.claude/settings.local.json is NOT in the default — it
|
||||
# pre-approves IDE/tool grants a member must not hold ambiently; CB-525/CB-634.)
|
||||
#
|
||||
# Do NOT add .mcp.json (CB-525). A worker's tools are whatever its launcher
|
||||
@@ -195,8 +248,12 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# omit and this profile's completion fallback behaves exactly as before.
|
||||
# Every backend words its refusal differently, so this is config, never a
|
||||
# vendor string baked into fleetd itself.
|
||||
# DEFERRED: compiled once into a startup pattern map — editing it needs a
|
||||
# daemon restart, same as this profile's model/baseUrl/argv.
|
||||
# HOT (fleetd #446): read live, cached by profile name, at every
|
||||
# completion-fallback check AND by fleet_profiles' exhaustionDetectionArmed —
|
||||
# editing it and reloading arms or disarms usage-limit detection for this
|
||||
# profile with no daemon restart. (Before fleetd #446 this was DEFERRED,
|
||||
# compiled once into a startup pattern map like model/baseUrl/argv still are —
|
||||
# see errorPattern below, which is still deferred that way on purpose.)
|
||||
# credentialId → CB-578 stage B: the credential this profile quarantines WITH when a
|
||||
# BACKEND_EXHAUSTED classification fires. Two profiles that set the SAME
|
||||
# credentialId share one quarantine — the case this exists for is two models
|
||||
@@ -206,6 +263,40 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# profile quarantines alone, under its own name, exactly as if the field did
|
||||
# not exist. Cooldown length is the top-level quarantineCooldownSeconds below.
|
||||
# HOT: read live at every spawn/exhaustion check — no restart needed.
|
||||
# errorPattern → fleetd #201 / #227: regex matched against a completion-fallback scrape to
|
||||
# classify a turn that ended with no fleet_reply as a BACKEND ERROR — a
|
||||
# credential outage or a provider 5xx — rather than a real answer or a
|
||||
# usage-limit exhaustion (exhaustedPattern above always wins when a line
|
||||
# matches both). Opt-in. Omit it and this profile falls back to fleetd's
|
||||
# built-in legacy pattern `(?i)\bAPI Error\s*:` — classification still
|
||||
# happens, just without a profile-specific match; every backend words its
|
||||
# failure differently, so a hardcoded sentence would only ever match one
|
||||
# of them.
|
||||
# DEFERRED: compiled once into a startup pattern map — editing it needs a
|
||||
# daemon restart. Unlike exhaustedPattern above (made hot by fleetd #446),
|
||||
# errorPattern was scoped out of that ticket on purpose and stays deferred.
|
||||
# # errorPattern: "503 Service Unavailable" # opt-in: classify a backend outage
|
||||
#
|
||||
# What happens once a match fires (BackendOutagePolicy, credentialId-keyed,
|
||||
# SEPARATE from the CB-578 stage B quarantine above and never merged with it):
|
||||
# - threshold 2 — TWO DISTINCT TARGETS (never raw events) on the same
|
||||
# effective credential inside a 60-second window start an "incident" and a
|
||||
# 60-second cool-off for that credential. One member repeating the same
|
||||
# classified line twice never cools anything off — a real outage hits
|
||||
# every target on that credential, so requiring a second, independent
|
||||
# target loses nothing against the case this guards against, while
|
||||
# protecting against a heuristic misfire on one flaky member.
|
||||
# - a fresh error while a credential is already cooling off is ignored
|
||||
# outright: it neither extends the 60s deadline nor starts a new incident.
|
||||
# - `fleet_list`/`fleet_profiles` report a cooling credential with
|
||||
# `coolingOffForSeconds` (never `quarantinedForSeconds`, unless CB-578
|
||||
# exhaustion quarantine is ALSO independently active for the same
|
||||
# credential — the two checks can both fire at once). A spawn onto a
|
||||
# cooling profile is refused with a message naming the credential and
|
||||
# remaining seconds — "cooling off", never "exhausted", so an operator can
|
||||
# tell a short transient fault from a spent subscription at a glance.
|
||||
# - the lead gets ONE nudge per incident (not one per affected target), via
|
||||
# the same push loop that already delivers ticket/question reminders.
|
||||
# env → extra environment for this profile's workers, as a literal key/value map
|
||||
# (CB-511). Use it to give workers a toolchain.
|
||||
#
|
||||
@@ -269,13 +360,49 @@ profiles:
|
||||
# GOTCHA 2 — `maxLoad` is the ONLY throttle you have here. There is no metering, no budget
|
||||
# and no refusal on cost; the cap on live members is the single thing standing between a
|
||||
# fan-out and your monthly limit. Set it deliberately and keep it small.
|
||||
#
|
||||
# GOTCHA 3 (fleetd #176, corrected by fleetd #257) — `maxLoad` counts members, never the lead
|
||||
# itself. The lead is a live `claude` session on this SAME account (a lead is never moved
|
||||
# off-subscription, whatever its own profile says), so it already holds one seat before any
|
||||
# member spawns. If a lead's `fleet.leaders.<name>.profile` names THIS profile — or ANY OTHER
|
||||
# `subscription: true` profile that shares this one's account (see THE SENTINEL, just below,
|
||||
# next to `credentialId:`) — `fleet_list` reports that seat count under `leadSeats`; see
|
||||
# `profile:` under THE FLEET below. `free` itself is NEVER reduced by `leadSeats`: `free` means
|
||||
# "what the real placement gate (`CompositePeerLauncher#enforceMaxLoad`) will actually grant a
|
||||
# fresh `fleet_spawn` right now", and that gate only ever compares live members against
|
||||
# `maxLoad` — it has no notion of the lead's own seat. An earlier cut of this feature
|
||||
# subtracted `leadSeats` from `free` on the theory it made `free` describe the true ceiling on
|
||||
# the account, but no backend seat ceiling shared with the lead has ever actually been
|
||||
# measured, and the subtraction just made `free` disagree with the one thing it is supposed to
|
||||
# describe — the fleetd #257 fix. `maxLoad: 3` means 3 member slots, full stop; a lead sharing
|
||||
# the account is a fact you can see in `leadSeats`, not a reason `free` undercounts spawns that
|
||||
# will, in practice, succeed.
|
||||
#
|
||||
# THE SENTINEL (fleetd #176 stage 2, correcting an inert stage 1 fix): every `subscription:
|
||||
# true` profile that leaves `credentialId` unset shares ONE implicit account-wide credential
|
||||
# id with every other such profile on this host — because a subscription profile doesn't
|
||||
# authenticate with a credential of its own, it authenticates as the operator's own Claude
|
||||
# login, and there is exactly one of those. So on a typical host, `opus` (the lead's profile)
|
||||
# and `sonnet` (the members' profile) are linked automatically, with NOTHING to set here — that
|
||||
# is what makes GOTCHA 3 above work without also writing matching `credentialId:` values on
|
||||
# both. This linkage is not just cosmetic: it is the same key `BackendQuarantine`/cool-off use,
|
||||
# so a usage-limit hit on `opus` now quarantines `sonnet` too (and vice versa) — correct, since
|
||||
# they are one Claude account, but worth knowing before you wonder why an unrelated-looking
|
||||
# profile went quarantined.
|
||||
#
|
||||
# WHEN TO OVERRIDE — set explicit, DIFFERENT `credentialId:` values on two `subscription: true`
|
||||
# profiles only when they are genuinely two separate Claude logins on the same host (a real,
|
||||
# if unusual, setup). An explicit `credentialId` always wins over the sentinel, so this is the
|
||||
# one way to keep two subscription profiles from being treated as one account for lead-seat
|
||||
# counting AND for quarantine/cool-off grouping alike.
|
||||
# gitTokenEnv: GITEA_TOKEN # opt-in: let this profile's workers open their own PR (CB-302)
|
||||
# gitHostEnv: GITEA_HOST # defaults to GITEA_HOST; injected only with gitTokenEnv
|
||||
# exhaustedPattern: "usage limit has been reached" # opt-in: classify a usage-limit refusal (CB-578)
|
||||
# credentialId: shared-openai # opt-in: quarantine together with every other profile sharing this id (CB-578)
|
||||
# errorPattern: "503 Service Unavailable" # opt-in: classify a backend outage (fleetd #201/#227) — see the key doc above
|
||||
# configDir: /Users/me/.ccs/instances/gx10 # CLAUDE_CONFIG_DIR — inherit that profile's skills/MCP
|
||||
# cwd: /Users/me/src/myrepo # pin the working dir; omit to inherit the primary's
|
||||
# parityOverlay: [".env", ".envrc"] # the default; never add .mcp.json or .claude/settings.local.json — see above
|
||||
# parityOverlay: [".env"] # the default; add ".envrc" explicitly if you want it copied too (CB-148) — never add .mcp.json or .claude/settings.local.json — see above
|
||||
# ideMcpUrl: http://127.0.0.1:29170/index-mcp/streamable-http # opt-in (CB-634): IDE code intelligence, pinned to the worktree
|
||||
# ideProjectDir: fleetd # CB-634: module dir the IDE opens + the overlay pins (this repo's pom is in fleetd/)
|
||||
# ideOpenCommand: env DISPLAY=:10.0 idea {dir} # CB-634 auto-open: opens {dir} in the IDE at spawn; omit to open by hand
|
||||
@@ -366,9 +493,24 @@ placement: weighted
|
||||
# seconds, before a spawn may land on it again. Applies to every profile's effective credential
|
||||
# (its own name, or its credentialId if set above) — there is no per-profile override. Default
|
||||
# 1800 (30 minutes) when omitted or non-positive.
|
||||
#
|
||||
# fleetd #466: this is now only the BASE of an escalating backoff, not a flat retry rate. A
|
||||
# credential quarantined again within one base cooldown of the previous quarantine ending (still
|
||||
# reporting exhausted — e.g. a weekly subscription limit that hasn't reset) backs off further:
|
||||
# cooldown doubles each such time, capped at 12x this value (~6 hours at the 1800s default). A
|
||||
# quarantine that starts after a base-cooldown's worth of quiet resets back to this value. Not
|
||||
# configurable per se — the multiplier and ceiling are constants in BackendQuarantine, not new
|
||||
# YAML keys; see its class doc for the exact formula and why there is no automatic probe to clear
|
||||
# it early (the operator's own design constraint — a probe spends the quota it's measuring).
|
||||
# DEFERRED: baked once into the BackendQuarantine built at startup — a running quarantine keeps
|
||||
# its original cooldown regardless; a new value only applies to a quarantine that starts after a
|
||||
# restart. Editing this needs a daemon restart to take effect.
|
||||
#
|
||||
# This does NOT govern the fleetd #201 / #227 backend-error cool-off documented under errorPattern
|
||||
# above — that mechanism is a separate, shorter-lived, NOT-configurable policy (threshold 2 distinct
|
||||
# targets, 60-second window, 60-second cool-off), on purpose: it exists to survive a brief transient
|
||||
# fault, not to replace this 30-minute exhaustion quarantine. Do not conflate the two when reading
|
||||
# fleet_list/fleet_profiles — coolingOffForSeconds and quarantinedForSeconds are independent facts.
|
||||
# quarantineCooldownSeconds: 1800
|
||||
|
||||
# Re-read this file without restarting the daemon (CB-559). Off unless you add this block, so an
|
||||
@@ -381,9 +523,10 @@ placement: weighted
|
||||
# when the reload happens — not about how important the key is:
|
||||
# HOT → takes effect on the next spawn: the whole `fleet:` block (every role pool,
|
||||
# `charters`, and `tabLabel`), `placement:`, and an existing profile's weight / maxLoad
|
||||
# / credentialId. Those are hot because the placement policy (and, for credentialId,
|
||||
# the CB-578 stage B quarantine check) reads them through a supplier — being config is
|
||||
# not by itself enough to make a key hot.
|
||||
# / credentialId / exhaustedPattern. Those are hot because the placement policy (and,
|
||||
# for credentialId, the CB-578 stage B quarantine check; for exhaustedPattern, fleetd
|
||||
# #446's LiveExhaustedPatterns) reads them through a supplier — being config is not by
|
||||
# itself enough to make a key hot.
|
||||
# EXCEPT `fleet.leaders`: Fleetd.main reads it once at startup to build the lead tab
|
||||
# scanner and launcher, and neither is rebuilt on reload. A changed/added/removed
|
||||
# `fleet.leaders` entry is silently accepted — the reload reports "config reloaded"
|
||||
@@ -395,8 +538,10 @@ placement: weighted
|
||||
# stage B — baked once into the quarantine tracker built at startup), ADDING or
|
||||
# REMOVING a profile (a new backend needs its own launcher, and launchers are built
|
||||
# once), AND an existing profile's launch settings — model, baseUrl, argv, env,
|
||||
# configDir, mcpUrl, tabLabel, exhaustedPattern. The launcher takes a copy of
|
||||
# `profiles:` at startup and resolves every spawn out of that copy, so those never
|
||||
# configDir, mcpUrl, tabLabel, errorPattern (fleetd #201 / #227 — compiled once into
|
||||
# a startup pattern map; exhaustedPattern used to be compiled the same way until
|
||||
# fleetd #446 made it hot — see above). The launcher takes a copy of `profiles:` at
|
||||
# startup and resolves every spawn out of that copy, so those never
|
||||
# reach a launch until you restart. The reload logs them by name rather than
|
||||
# pretending they applied.
|
||||
# COLD → cannot change at all: `bind:`, `herdrSocket:`, `broker:` and `auth:`. The socket is
|
||||
@@ -457,6 +602,23 @@ fleet:
|
||||
# recognised: give it a `profile:` and the daemon launches the shortfall when fewer than
|
||||
# `instances` are live. Omit `profile:` and it is recognise-only, as before.
|
||||
#
|
||||
# `profile:` has a SECOND job as of fleetd #176, even for a recognise-only lead you never want
|
||||
# auto-launched: it is also how fleetd learns which account this lead's own session shares. A
|
||||
# `subscription: true` profile bills the operator's Claude account, and the lead itself is always
|
||||
# a live `claude` session on that same account — `maxLoad` never counted that seat. If a lead
|
||||
# entry here names a profile that shares a worker profile's account, `fleet_list` reports the
|
||||
# lead's live seat(s) on that worker profile under `leadSeats` — informational only, as of fleetd
|
||||
# #257 it is NEVER subtracted from `free` (see GOTCHA 3, next to `maxLoad:`, in THE WORKERS above,
|
||||
# for why). "Shares the account" is decided by matching `effectiveCredentialId()`, which (fleetd
|
||||
# #176 stage 2 — see THE SENTINEL, next to `credentialId:`, in THE WORKERS above) means: an
|
||||
# explicit, matching `credentialId:` on both, OR — the common case, needing NO extra config — both
|
||||
# being `subscription: true` with `credentialId` left unset, since those all share one implicit
|
||||
# account-wide id. A lead on `opus` and workers on `sonnet` link automatically this way; they do
|
||||
# NOT need the same profile name. Setting `profile:` on an already-running, recognise-only lead is
|
||||
# safe — the daemon only launches the SHORTFALL below `instances`, so naming a profile here does
|
||||
# not, by itself, start anything. Omit it and fleetd has no way to derive the sharing — there is
|
||||
# no other reliable signal on the daemon's side — so that lead's seat never appears in `leadSeats`.
|
||||
#
|
||||
# `tab:` (CB-579) is REQUIRED and is the only field identity depends on — the exact label of the
|
||||
# tab hosting the lead, matched case-insensitively. Label the tab yourself and put that same
|
||||
# string here, and the pane is recognised on the next rescan. Reopen the tab later, or the session
|
||||
@@ -578,20 +740,35 @@ guard:
|
||||
# every name here NOT also in `allow` is overlaid with a non-secret sentinel value before
|
||||
# the pane's login shell runs — real protection only for names that shell does not itself
|
||||
# re-export (see the ROUND-2 CORRECTION note above). Under allow-list: reporting only.
|
||||
# sshAuthSock → whether SSH_AUTH_SOCK may pass through under allow-list ("allow") or must be
|
||||
# blanked like any other non-derived name ("block", the default). This is a decision you
|
||||
# have to make explicitly: SSH_AUTH_SOCK is a handle to YOUR ssh-agent, and a member
|
||||
# holding it can sign with your keys — it sits in no secret file and looks like no
|
||||
# credential, which is why it slipped past three earlier tickets (gitea #110). Blocking
|
||||
# it breaks git over SSH inside members (push/fetch authenticate as you); use HTTPS
|
||||
# remotes or scoped deploy keys instead of allowing it lightly.
|
||||
# sshAgentEnv → whether SSH_AUTH_SOCK may pass through under allow-list ("inherit") or is omitted
|
||||
# from the member environment ("omit", the default). Omitting it only omits the
|
||||
# inherited ssh-agent path. It discourages automatic use of the operator's agent.
|
||||
# It does not deny same-user access to that socket. It also does not block SSH keys that
|
||||
# are readable on disk. Git over SSH may still work from inside a member. Keep the block:
|
||||
# it is correct and costs nothing, but it is not a control. A member runs as the same OS
|
||||
# user as the lead. Inside one uid, ordinary Unix permissions provide no meaningful
|
||||
# confidentiality boundary. A real boundary needs a different OS user or OS-level
|
||||
# confinement, such as a container or VM. That is the open question in fleetd #184.
|
||||
#
|
||||
# Still do not set this to "inherit" casually. SSH_AUTH_SOCK is a live handle to YOUR
|
||||
# ssh-agent, so a member holding it can sign with EVERY key the agent holds. It sits in
|
||||
# no secret file and looks like no credential, which is why it slipped past three
|
||||
# earlier tickets (gitea #110). Blocking it does not contain a member, but allowing it
|
||||
# hands one a signing capability for no gain — the block costs nothing, so keep it.
|
||||
#
|
||||
# Both halves of this are measured, not argued. 2026-08-28: a member with
|
||||
# SSH_AUTH_SOCK blanked pushed to the forge over SSH successfully, because `ssh -G`
|
||||
# resolves an IdentityFile outside ~/.ssh that is readable and has no passphrase. An
|
||||
# earlier version of this comment claimed blocking the socket BREAKS git over SSH. It
|
||||
# does not. That claim came from looking only in ~/.ssh, which holds nothing but four
|
||||
# `Include` lines — looking in one place and concluding about the whole host.
|
||||
#
|
||||
# HOT-RELOADABLE the same way `fleet:` is (CB-559): read fresh on every spawn, so editing this list
|
||||
# and reloading config (or restarting) changes what the NEXT spawn inherits; already-running members
|
||||
# are unaffected either way.
|
||||
# memberCredentials:
|
||||
# policy: deny-by-default # or "deny-list", or "allow-list" (CB-633) — see above
|
||||
# sshAuthSock: block # allow-list only; see the sshAuthSock note above
|
||||
# sshAgentEnv: omit # allow-list only; see the sshAgentEnv note above
|
||||
# allow:
|
||||
# - AI_GATEWAY_TOKEN # named in a profile's tokenEnv (local/gx) — a member reaching the
|
||||
# # gateway is by design, not a leak
|
||||
@@ -661,6 +838,28 @@ guard:
|
||||
# so fleetd falls back to the weaker CB-596 sentinel overlay instead (a WARN names the gap).
|
||||
# worktreeGroup: fleet-workers
|
||||
|
||||
# fleetd #362: a directory of skill folders (each a subdirectory holding a SKILL.md, the same
|
||||
# shape as this repo's own .claude/skills/) copied into every PROVISIONED worktree's
|
||||
# .claude/skills/, so a member spawned against ANY repo — not only one that already ships its own
|
||||
# copy — can load a bridge skill (e.g. implementer). Unset (the default): no worktree is touched
|
||||
# beyond today's behaviour. A skill folder the target repo already carries under
|
||||
# .claude/skills/<name> is never overwritten — the repo's own copy always wins. Best-effort like
|
||||
# worktreeGroup above: a missing/unreadable directory here is logged and skipped, never a failed
|
||||
# spawn. Every non-hidden subdirectory of this directory is copied wholesale, with no per-file
|
||||
# allowlist — don't park scratch files or drafts alongside the real skill folders, they will be
|
||||
# copied into every provisioned worktree too.
|
||||
#
|
||||
# fleetd #393: which member KINDS actually consume this once it is copied. kind: claude-code —
|
||||
# the Claude Code CLI discovers .claude/skills/ on its own; nothing else is needed. kind: opencode
|
||||
# — opencode has no such discovery, so OpenCodeLauncher reads whatever landed under
|
||||
# .claude/skills/ and appends each seeded skill's SKILL.md to the generated instructions[] file
|
||||
# (opencode's only channel for static guidance text; unlike Claude Code's Skill tool, the content
|
||||
# is always part of the system prompt, not loaded on demand). Both kinds are covered as of #393 —
|
||||
# earlier builds copied the files for every kind but only claude-code could read them, and the
|
||||
# seeding log said "N of M" regardless. Check the per-spawn launcher log (not just the seeding
|
||||
# log) to see what a given member actually got.
|
||||
# memberSkills: /path/to/fleetd/checkout/.claude/skills
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
# the feature disabled. By default the daemon never reaps, caps, or drains sessions.
|
||||
# idleTtlSeconds → reap READY/DONE sessions idle longer than this (never BUSY/SPAWNING)
|
||||
@@ -708,10 +907,17 @@ guard:
|
||||
# across every daemon sharing this vhost.
|
||||
# prefetch → consumer basicQos, capping how many unacked messages the mailbox holds in-heap.
|
||||
# Default 32 when omitted.
|
||||
# peers → fleetd #361: the coord-ids of the OTHER daemons on this vhost, declared by the
|
||||
# operator (the daemon never guesses). fleet_list reports each one's live reachability
|
||||
# (a passive queue check, never a presence protocol) alongside this daemon's own
|
||||
# mailbox state. Omit, or leave empty, for a daemon with no known peers yet — an
|
||||
# undeclared peer can still reach you and be reached by fleet_send, it just will not
|
||||
# show up as a row in fleet_list.
|
||||
# coordinator:
|
||||
# uriEnv: LEAD_COORD_URI
|
||||
# selfId: mac-opus
|
||||
# prefetch: 32
|
||||
# peers: [fleet01-lead]
|
||||
|
||||
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open fleet_send,
|
||||
# the ReplyPushLoop injects a *drain nudge* (never the payload) into the primary's own herdr
|
||||
@@ -736,3 +942,32 @@ guard:
|
||||
# terminal: term_65619bd6174568
|
||||
# pushReminders: 5
|
||||
# pushBackoffMs: 15000
|
||||
|
||||
# Central allow-list of models any profiles: entry may name. Nothing checked a profile's model:
|
||||
# value before this block existed — it was a free-form string handed straight to the backend
|
||||
# adapter, and a withdrawn or misspelled name failed silently instead of at config load (opencode
|
||||
# falls back to a default model rather than erroring on an unknown -m).
|
||||
#
|
||||
# Absent, or present with an empty allow:, is OFF: no profile's model: is checked, exactly like
|
||||
# before this block existed. fleetd.yaml is gitignored on every host, so an upgrade must not force
|
||||
# every operator to enumerate their models before the daemon will start.
|
||||
#
|
||||
# The list is the authority; profiles: is checked against it, never the reverse — adding or
|
||||
# editing a profiles: entry cannot, by itself, widen what is permitted here.
|
||||
#
|
||||
# Enforcement is at CONFIG LOAD only (a bad model: fails the daemon at startup, naming both the
|
||||
# model and the profile). There is no spawn-time enforcement, no runtime on/off switch, and no
|
||||
# interaction with BackendQuarantine — those are separate, later units.
|
||||
#
|
||||
# allow → the permitted models. Each entry is its own block (not a bare string) so a later unit
|
||||
# can add an on/off state or a load limit per model without changing this shape.
|
||||
# model → the model id exactly as a profiles: entry's model: field would write it. One flat,
|
||||
# opaque-string namespace: a bare Claude id (claude-sonnet-5) and an opencode
|
||||
# provider-prefixed id (openai/gpt-5.6-terra) both fit here unchanged — the check is a
|
||||
# plain string match, never a parse of the provider prefix or a branch on kind:.
|
||||
# models:
|
||||
# allow:
|
||||
# - model: claude-sonnet-5
|
||||
# - model: claude-opus-5
|
||||
# - model: openai/gpt-5.6-terra
|
||||
# - model: amazon.nova-pro-v1:0
|
||||
|
||||
@@ -29,6 +29,7 @@
|
||||
<commons-compress.version>1.27.1</commons-compress.version>
|
||||
<commons-lang3.version>3.18.0</commons-lang3.version>
|
||||
<sqlite-jdbc.version>3.53.4.0</sqlite-jdbc.version>
|
||||
<archunit.version>1.5.0</archunit.version>
|
||||
</properties>
|
||||
|
||||
<!--
|
||||
@@ -176,6 +177,14 @@
|
||||
<version>${testcontainers.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
|
||||
<!-- fleetd #131: package-boundary and cycle enforcement (PackageCyclesTest). -->
|
||||
<dependency>
|
||||
<groupId>com.tngtech.archunit</groupId>
|
||||
<artifactId>archunit-junit5</artifactId>
|
||||
<version>${archunit.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
|
||||
<build>
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -30,8 +30,27 @@ public final class Authz {
|
||||
DRAIN,
|
||||
/** Read-only observation: status, roster, profiles, task polling. */
|
||||
READ,
|
||||
/**
|
||||
* Read (never ack) this daemon's own held lead-to-lead coordination mail (fleetd #421).
|
||||
*
|
||||
* <p>Deliberately <strong>not</strong> folded into {@link #READ}. {@code READ}'s grant
|
||||
* rests on "the roster carries no secrets" (see its case below) — a lead-to-lead body is
|
||||
* not the roster; it is where leads discuss host shapes, credentials and unmerged work.
|
||||
* Mapping this to {@code READ} would let any worker read every peer lead's mail in full
|
||||
* and would silently falsify that comment for every other {@code READ} caller.
|
||||
*/
|
||||
COORD_READ,
|
||||
/** Scrape the metrics endpoint. */
|
||||
METRICS
|
||||
METRICS,
|
||||
/**
|
||||
* Drive the lead-rollover executor ({@code fleet_handover}: open/confirm/cancel a
|
||||
* self-replace, fleetd #480 Unit C). Primary-only, same as {@link #SPAWN}/{@link #STOP}/
|
||||
* {@link #DRAIN} — and, unlike those, the terminal it acts on is never even an argument:
|
||||
* {@code LeadRollover#open}/{@code #confirm} are always called with the CALLER's own
|
||||
* connection-resolved terminal (see {@code dev.ltms.fleet.lead.LeadRollover}'s class
|
||||
* javadoc, fleetd #480 correction 2), so a primary can only ever roll itself.
|
||||
*/
|
||||
HANDOVER
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -50,7 +69,7 @@ public final class Authz {
|
||||
// deliberately does NOT get these (CB-548), so it cannot tear down or stand up workers
|
||||
// even though it coordinates them; and a worker driving any of these would be a worker
|
||||
// escalating into the orchestrator role.
|
||||
case SPAWN, STOP, DRAIN -> caller.isPrimary();
|
||||
case SPAWN, STOP, DRAIN, HANDOVER -> caller.isPrimary();
|
||||
|
||||
// Delivering a turn is open to the primary and the architect: an architect delegates
|
||||
// to workers (that is the role's point) but still has no lifecycle rights. A worker is
|
||||
@@ -68,6 +87,11 @@ public final class Authz {
|
||||
// Observation is open to every authenticated role: a worker legitimately polls its own
|
||||
// status, and the roster carries no secrets.
|
||||
case READ, METRICS -> caller.isPrimary() || caller.isWorker() || caller.isArchitect();
|
||||
|
||||
// fleetd #421: reading held lead-to-lead mail is the primary's alone. An architect
|
||||
// holds READ today (CB-548), so "not primary" must mean not-architect here too — this
|
||||
// is coordination between leads, not observation of the roster.
|
||||
case COORD_READ -> caller.isPrimary();
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -236,7 +236,15 @@ public final class CallerResolver {
|
||||
// loopback-trust: same-host callers that are not workers are the primary. A non-loopback
|
||||
// caller is anonymous even here — and startup refuses that combination anyway
|
||||
// (FleetConfig.validateAuthExposure), so this is defence in depth, not the control.
|
||||
return isLoopback(remoteAddr) ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
//
|
||||
// fleetd #317: "not a worker" must not be conflated with "identity unresolved". The real
|
||||
// primary is a real process — its pid resolves (c.resolved()), it just owns no herdr pane.
|
||||
// A caller whose peer-PID lookup failed (LsofPeerPidLookup's -1 sentinel — on any failure,
|
||||
// silently including "lsof found no match") has no such pid, and PaneLocator's own javadoc
|
||||
// already names what happens if that case is handed the primary role: a worker→primary
|
||||
// escalation. So an unresolved caller is refused (ANONYMOUS — the same clean, already-tested
|
||||
// "authenticated as nothing" outcome used everywhere else in this method), never promoted.
|
||||
return isLoopback(remoteAddr) && c.resolved() ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
}
|
||||
|
||||
private boolean presentedTokenMatches(String authorizationHeader) {
|
||||
@@ -262,11 +270,15 @@ public final class CallerResolver {
|
||||
return token.isEmpty() ? null : token;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #305: delegates to {@link ConnectionIdentity#isLoopback}. This used to be a second,
|
||||
* independent copy of the same rule, and the two drifted: this one accepted all of
|
||||
* {@code 127.0.0.0/8}, {@code ConnectionIdentity}'s accepted only {@code 127.0.0.1}. A caller
|
||||
* from {@code 127.0.0.2} therefore had its identity skipped (so it had no terminal) and was
|
||||
* then read as loopback here — which under loopback-trust is the primary. Sharing the inputs
|
||||
* would not have prevented that; only sharing the computation does.
|
||||
*/
|
||||
private static boolean isLoopback(String remoteAddr) {
|
||||
if (remoteAddr == null) {
|
||||
return false;
|
||||
}
|
||||
return remoteAddr.equals("127.0.0.1") || remoteAddr.equals("::1")
|
||||
|| remoteAddr.equals("0:0:0:0:0:0:0:1") || remoteAddr.startsWith("127.");
|
||||
return ConnectionIdentity.isLoopback(remoteAddr);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,17 +5,70 @@ import dev.ltms.fleet.peer.MemberRole;
|
||||
/** Optional session lifecycle hook for live member-slot bindings. */
|
||||
public interface MemberLifecycle {
|
||||
|
||||
/** A slot held before a member process starts. */
|
||||
record SlotReservation(String slot, String profile) { }
|
||||
|
||||
MemberLifecycle NONE = new MemberLifecycle() {
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
return role; // no registry configured — nothing to bind against, so the request stands
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
// no registry configured — nothing to validate against, so nothing is refused
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
return null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
}
|
||||
};
|
||||
|
||||
void acquired(MemberRole role, String profile, String terminal);
|
||||
/**
|
||||
* Try to bind a newly spawned {@code terminal} into the role it was granted.
|
||||
*
|
||||
* @return the role this session actually holds: {@code role} unchanged for a role with no
|
||||
* live slot-binding semantics (dev, reviewer), or when the bind succeeded; a fallback
|
||||
* role — never {@code role} — when a slot-bound role (architect) could not be bound.
|
||||
* Callers must record THIS value on the session, never the requested {@code role}, so
|
||||
* a later roster read never reports a role the session does not hold (CB-619). In
|
||||
* normal operation this fallback should not happen once a reservation has been bound.
|
||||
* It remains the honest answer if a caller has no reservation, or if binding a
|
||||
* reservation unexpectedly fails.
|
||||
*/
|
||||
MemberRole acquired(MemberRole role, String profile, String terminal);
|
||||
|
||||
void released(String terminal);
|
||||
|
||||
/**
|
||||
* Refuse an acquire before anything spawns when {@code role} requires a live slot binding and
|
||||
* no configured slot carries {@code profile} (CB-619 / fleetd #123). A no-op for a role with
|
||||
* no slot-binding semantics.
|
||||
*
|
||||
* @throws IllegalArgumentException naming the role, the profile, and the pools that do carry it
|
||||
*/
|
||||
void requireSlotFor(MemberRole role, String profile);
|
||||
|
||||
/** Reserve a matching slot before launch, or refuse before a charter can be delivered. */
|
||||
SlotReservation reserve(MemberRole role, String profile);
|
||||
|
||||
/** Convert a reservation into a live terminal binding. */
|
||||
boolean bind(SlotReservation reservation, String terminal);
|
||||
|
||||
/** Return an unbound reservation after a failed launch. */
|
||||
void release(SlotReservation reservation);
|
||||
}
|
||||
|
||||
@@ -8,8 +8,10 @@ import org.slf4j.LoggerFactory;
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The architect-slot registry (CB-548): every gateway-local architect name and the strong-model
|
||||
@@ -18,16 +20,43 @@ import java.util.Objects;
|
||||
*
|
||||
* <p>Two halves, split by who owns each:
|
||||
* <ul>
|
||||
* <li><b>slots</b> — configured once, keyed by the gateway-local unique name; each carries the
|
||||
* {@code profile} reference the spawn lifecycle reads when it stands the slot up. A read-only
|
||||
* snapshot taken at construction.</li>
|
||||
* <li><b>terminal bindings</b> — owned by this registry and initially <em>empty</em>. Config
|
||||
* declares no architect terminal, so at startup every slot is idle and nothing resolves to an
|
||||
* architect; a session only becomes one when the spawn lifecycle {@linkplain #bind(String,
|
||||
* String) binds} its terminal to a slot. {@link CallerResolver} reads this through
|
||||
* {@link #snapshot()} to turn a pane into an {@link Role#ARCHITECT}.</li>
|
||||
* <li><b>slots</b> — read from {@code fleet.architects}/{@code developers}/{@code reviewers}
|
||||
* (see {@link #slots()}), each carrying the {@code profile} reference the spawn lifecycle
|
||||
* reads when it stands the slot up. <strong>Live, since fleetd #424</strong>: {@link #live}
|
||||
* re-reads {@code fleet:} on every call, through a supplier the same shape as
|
||||
* {@code CompositePeerLauncher}'s (see {@code ConfigRef}'s class doc) — so a config reload
|
||||
* that removes or adds an architect slot governs the <em>next</em> spawn with no restart.
|
||||
* Only {@link #MemberRegistry(FleetConfig.Fleet)} freezes the pool at construction, and that
|
||||
* constructor exists for tests and for the (rare) case of wiring a fixed, code-built config.</li>
|
||||
* <li><b>terminal bindings</b> — owned by this registry, initially <em>empty</em>, and
|
||||
* <strong>never</strong> touched by a reload. Config declares no architect terminal, so at
|
||||
* startup every slot is idle and nothing resolves to an architect; a session only becomes one
|
||||
* when the spawn lifecycle {@linkplain #bind(String, String) binds} its terminal to a slot.
|
||||
* {@link CallerResolver} reads this through {@link #snapshot()} to turn a pane into an
|
||||
* {@link Role#ARCHITECT}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>The binding rule (fleetd #424): config governs what a bound slot still grants, as
|
||||
* well as what may be bound next.</strong> Removing a slot from config revokes it — that is the
|
||||
* ticket's entire point ("Revoking an architect slot does not revoke it"). Revoking it means an
|
||||
* architect already bound to that slot loses the ARCHITECT privilege on its very next request:
|
||||
* {@link #roleForSlot} and {@link #nameForSlot} read {@link #slots()} directly, with no cache, so
|
||||
* the moment a slot drops out of config, {@link CallerResolver#resolve} (which calls both on every
|
||||
* request from a bound pane, {@code CallerResolver.java:220}) can no longer confirm the pane's slot
|
||||
* is an architect slot, and the pane falls through to {@code Principal.worker(...)}. What does
|
||||
* <em>not</em> change is the {@code terminalToSlot} <em>occupancy</em> — the binding created by
|
||||
* {@link #bind} is untouched by a reload, on purpose: unbinding it here would double-book the slot
|
||||
* key (a second terminal could then bind to the "freed" key while the first is still the terminal
|
||||
* the operator actually meant to demote) and would silently break {@link #unbind}'s compare-safe
|
||||
* contract, which needs the original {@code terminal → slot} pair intact to remove it cleanly. So
|
||||
* the demoted session keeps occupying its slot — {@link #slotForTerminal} and {@link #snapshot()}
|
||||
* still name it — it just no longer resolves as an architect through that occupancy, and a fresh
|
||||
* spawn still cannot bind to the same key while it is occupied ({@link #reserve}/
|
||||
* {@link #requireSlotFor} refuse it anyway, since it is gone from {@link #slots()}). The demoted
|
||||
* session's own turn is unaffected: {@code fleet_reply}'s authorization
|
||||
* ({@code Authz.Action.REPLY}) is {@code caller.ownsSession(targetSession)} — identity by terminal,
|
||||
* not by role — so a demoted architect can still end its own turn normally.
|
||||
*
|
||||
* <p>Spawning/lifecycle is deliberately a separate unit: this class only owns the bindings and
|
||||
* exposes the map the resolver resolves against plus the profile lookup lifecycle will call.
|
||||
* Nothing here creates or manages an architect session.
|
||||
@@ -54,12 +83,42 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
}
|
||||
}
|
||||
|
||||
private final Map<String, Entry> slots;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code this}. */
|
||||
private final Supplier<FleetConfig.Fleet> fleet;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code terminalToSlot}. */
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
/** Slot keys held between reservation and the terminal binding. Guarded by terminalToSlot. */
|
||||
private final java.util.Set<String> reservedSlots = new java.util.HashSet<>();
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one registry. Leaders are not members. */
|
||||
/**
|
||||
* Freeze the pool at construction — for tests, and for the rare case of wiring a fixed,
|
||||
* code-built config. Production wiring should prefer {@link #live}, which re-reads
|
||||
* {@code fleet:} on every call.
|
||||
*/
|
||||
public MemberRegistry(FleetConfig.Fleet fleet) {
|
||||
this(() -> fleet);
|
||||
}
|
||||
|
||||
private MemberRegistry(Supplier<FleetConfig.Fleet> fleet) {
|
||||
this.fleet = fleet;
|
||||
}
|
||||
|
||||
/**
|
||||
* Live variant (fleetd #424): {@code fleet} is read fresh on every {@link #slots()} call — pass
|
||||
* {@code () -> config.get().fleet()}, the same supplier shape {@code CompositePeerLauncher}
|
||||
* already uses for placement — so a reload that adds or removes an architect slot governs the
|
||||
* next spawn's {@link #reserve}/{@link #requireSlotFor} check with no restart. A separate,
|
||||
* private constructor rather than a same-arity public overload of
|
||||
* {@link #MemberRegistry(FleetConfig.Fleet)}: a {@code FleetConfig.Fleet} and a
|
||||
* {@code Supplier<FleetConfig.Fleet>} overload are ambiguous for a literal {@code null} — the
|
||||
* same reason {@code CallerResolver.withLeads} is a static factory rather than a fourth
|
||||
* constructor overload.
|
||||
*/
|
||||
public static MemberRegistry live(Supplier<FleetConfig.Fleet> fleet) {
|
||||
return new MemberRegistry(Objects.requireNonNull(fleet, "fleet"));
|
||||
}
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one map. Leaders are not members. */
|
||||
private static Map<String, Entry> flatten(FleetConfig.Fleet fleet) {
|
||||
Map<String, Entry> flat = new LinkedHashMap<>();
|
||||
if (fleet != null) {
|
||||
for (MemberRole role : MemberRole.values()) {
|
||||
@@ -71,18 +130,22 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
});
|
||||
}
|
||||
}
|
||||
this.slots = Collections.unmodifiableMap(flat);
|
||||
return Collections.unmodifiableMap(flat);
|
||||
}
|
||||
|
||||
/** The configured slots, keyed by qualified {@link Entry#key()}. Unmodifiable snapshot. */
|
||||
/**
|
||||
* The configured slots, keyed by qualified {@link Entry#key()}. Unmodifiable snapshot of
|
||||
* {@code fleet:} <em>as of this call</em> — see the class doc for which constructor makes that
|
||||
* live versus frozen.
|
||||
*/
|
||||
public Map<String, Entry> slots() {
|
||||
return slots;
|
||||
return flatten(fleet.get());
|
||||
}
|
||||
|
||||
/** The slots belonging to {@code role}, in definition order. */
|
||||
/** The slots belonging to {@code role}, in definition order, as of this call. */
|
||||
public Map<String, Entry> slotsFor(MemberRole role) {
|
||||
Map<String, Entry> out = new LinkedHashMap<>();
|
||||
slots.forEach((key, e) -> {
|
||||
slots().forEach((key, e) -> {
|
||||
if (e.role() == role) {
|
||||
out.put(key, e);
|
||||
}
|
||||
@@ -114,31 +177,49 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
}
|
||||
|
||||
/**
|
||||
* The strong-model profile a slot runs under — what the spawn lifecycle reads.
|
||||
* The strong-model profile a slot runs under, as of this call.
|
||||
*
|
||||
* <p>Nothing in {@code src/main} calls this (fleetd #431 — grepped both the {@code
|
||||
* .profileForSlot(} and the {@code ::profileForSlot} form). This javadoc used to say "what the
|
||||
* spawn lifecycle reads", and that seam does not exist: the spawn lifecycle takes its profile
|
||||
* from the {@link MemberLifecycle.SlotReservation} that {@code reserve} returns, never from
|
||||
* here. Kept and pinned rather than deleted because it is the natural accessor for that seam
|
||||
* if one is added; live for the same reason as {@link #roleForSlot}, so a reload cannot leave
|
||||
* it answering for the old config.
|
||||
*
|
||||
* @return the slot's configured {@code profile}, or {@code null} if the slot is unknown or
|
||||
* declares none
|
||||
*/
|
||||
public String profileForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
Entry e = slots().get(slotName);
|
||||
return (e == null || e.profile() == null) ? null : e.profile();
|
||||
}
|
||||
|
||||
/** The role a qualified slot key belongs to, or {@code null} when the key is unknown. */
|
||||
/**
|
||||
* The role a qualified slot key belongs to, or {@code null} when the key is not currently
|
||||
* configured. Deliberately live, with no cache (fleetd #424, see the class doc's binding rule):
|
||||
* removing a slot from config must make {@link CallerResolver#resolve} stop granting the
|
||||
* ARCHITECT role for it on the very next request from a terminal that was bound to it, which is
|
||||
* the ticket's whole point — revoking a slot must actually revoke it, not just refuse the next
|
||||
* spawn.
|
||||
*/
|
||||
public MemberRole roleForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
Entry e = slots().get(slotName);
|
||||
return e == null ? null : e.role();
|
||||
}
|
||||
|
||||
/** The unqualified configured name for a slot, or {@code null} if it is unknown. */
|
||||
/**
|
||||
* The unqualified configured name for a slot, or {@code null} if it is not currently configured.
|
||||
* Live for the same reason as {@link #roleForSlot} — see the class doc's binding rule.
|
||||
*/
|
||||
public String nameForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
Entry e = slots().get(slotName);
|
||||
return e == null ? null : e.name();
|
||||
}
|
||||
|
||||
/** True when {@code slotName} is a configured architect slot. */
|
||||
public boolean isSlot(String slotName) {
|
||||
return slots.containsKey(slotName);
|
||||
return slots().containsKey(slotName);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -166,7 +247,7 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
if (existingSlot != null) {
|
||||
return slot.equals(existingSlot); // already this slot (idempotent) or a different one
|
||||
}
|
||||
if (terminalToSlot.containsValue(slot)) {
|
||||
if (terminalToSlot.containsValue(slot) || reservedSlots.contains(slot)) {
|
||||
return false; // slot already hosts a terminal — no second one
|
||||
}
|
||||
terminalToSlot.put(terminal, slot);
|
||||
@@ -206,19 +287,116 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
*
|
||||
* <p>The role check is lifecycle policy. {@link CallerResolver} repeats it when resolving a
|
||||
* binding, so a later lifecycle regression cannot turn a worker into an architect.
|
||||
*
|
||||
* <p>CB-619 / fleetd #123: the return value is the role this session actually holds, and the
|
||||
* caller is required to record THAT — never the requested {@code role} — on the session. Before
|
||||
* this fix the caller kept the requested role regardless of whether the bind below succeeded, so
|
||||
* a demoted session's {@code GET /members} row still said {@code "architect"} while
|
||||
* {@code fleet_whoami} (which reads the live binding, not the request) correctly said
|
||||
* {@code "worker"} — three sources of truth that disagreed about one live member, silently.
|
||||
*/
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
if (role != MemberRole.ARCHITECT || terminal == null || terminal.isBlank()) {
|
||||
return;
|
||||
return role;
|
||||
}
|
||||
// slotsFor preserves definition order, so duplicate-profile slots use the first free one.
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if (Objects.equals(profile, entry.profile()) && bind(entry.key(), terminal)) {
|
||||
return;
|
||||
return MemberRole.ARCHITECT;
|
||||
}
|
||||
}
|
||||
// fleetd #123: at least WARN — a role downgrade that the roster must now also reflect is
|
||||
// not routine bookkeeping. requireSlotFor already refuses the config-gap case (no slot at
|
||||
// all carries this profile) before a process ever spawns; reaching here means the config DID
|
||||
// carry a matching slot but every one of them was already bound to a different terminal — a
|
||||
// race this pre-spawn check cannot close on its own (see requireSlotFor's javadoc).
|
||||
log.warn("member slot: no free architect slot for profile={} terminal={}; holding the session "
|
||||
+ "as {} instead of the architect it asked for — every configured slot for this "
|
||||
+ "profile is already bound to a different terminal", profile, terminal,
|
||||
MemberRole.DEV.wireName());
|
||||
return MemberRole.DEV;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-619 / fleetd #123: refuse an architect acquire before anything spawns when no configured
|
||||
* slot carries {@code profile} — the config-gap case from the original defect report (a spawn
|
||||
* asked for {@code role=architect, profile=sonnet}, and {@code fleet.architects} carried only
|
||||
* {@code opus} and {@code sol}). A dev/reviewer acquire is always a no-op: those pools are
|
||||
* placement candidates only (see {@code CompositePeerLauncher}), never a live identity binding,
|
||||
* so there is nothing here to refuse — an explicit profile outside the pool for those roles is a
|
||||
* documented operator override, not a defect.
|
||||
*
|
||||
* <p>This closes the config-gap case, not the live-capacity case: a profile that DOES carry a
|
||||
* slot can still lose the race to a concurrent spawn between this check and the actual
|
||||
* {@link #bind}, which is why {@link #acquired} must still answer honestly even after this
|
||||
* check has passed.
|
||||
*/
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return;
|
||||
}
|
||||
boolean hasSlot = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.anyMatch(e -> Objects.equals(profile, e.profile()));
|
||||
if (hasSlot) {
|
||||
return;
|
||||
}
|
||||
List<String> pools = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.map(Entry::profile)
|
||||
.distinct()
|
||||
.toList();
|
||||
throw new IllegalArgumentException(
|
||||
"no " + role.wireName() + " slot for profile '" + profile + "' — an architect's "
|
||||
+ "identity IS the slot it is bound to, so there is nothing to bind this "
|
||||
+ "session's identity to. fleet." + role.configKey() + " carries profiles: "
|
||||
+ (pools.isEmpty() ? "(none configured)" : String.join(", ", pools))
|
||||
+ "; add profile '" + profile + "' there, or spawn " + role.wireName()
|
||||
+ " on one of those profiles instead");
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return null;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if ((profile == null || profile.isBlank() || Objects.equals(profile, entry.profile()))
|
||||
&& !terminalToSlot.containsValue(entry.key()) && reservedSlots.add(entry.key())) {
|
||||
return new SlotReservation(entry.key(), entry.profile());
|
||||
}
|
||||
}
|
||||
}
|
||||
throw new IllegalArgumentException("no free architect slot for profile '" + profile
|
||||
+ "' — every matching slot is already bound or reserved");
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
if (reservation == null || terminal == null || terminal.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
if (!reservedSlots.remove(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
if (!isSlot(reservation.slot()) || terminalToSlot.containsKey(terminal)
|
||||
|| terminalToSlot.containsValue(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
terminalToSlot.put(terminal, reservation.slot());
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
if (reservation != null) {
|
||||
synchronized (terminalToSlot) {
|
||||
reservedSlots.remove(reservation.slot());
|
||||
}
|
||||
}
|
||||
log.info("member slot: no free architect slot for profile={}; session remains a worker", profile);
|
||||
}
|
||||
|
||||
/** Unbind a released terminal using the compare-safe registry operation. */
|
||||
|
||||
@@ -11,6 +11,7 @@ import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
@@ -22,63 +23,296 @@ import java.util.function.Supplier;
|
||||
* choice rather than an accident of where the field was initialised.
|
||||
*
|
||||
* <h2>Not every key can change under a running daemon</h2>
|
||||
* Keys fall into three classes, and the difference is about what already exists when the reload
|
||||
* Keys fall into four classes, and the difference is about what already exists when the reload
|
||||
* happens — not about how important the key is.
|
||||
*
|
||||
* <ul>
|
||||
* <li><strong>Hot</strong> — re-read per use, so a reload takes effect on the next spawn:
|
||||
* {@code fleet:} (every role pool, {@code charters}, and {@code tabLabel}),
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Those
|
||||
* three are read through a supplier on {@code CompositePeerLauncher}, which is what makes
|
||||
* them hot — not the fact that they are config. <strong>This does NOT include
|
||||
* {@code fleet.leaders}</strong>: {@code Fleetd.main} reads {@code cfg.fleet().leaders()}
|
||||
* once at startup to build the {@code LeadTabScanner} and the {@code LeadLauncher}, and
|
||||
* neither is reconstructed on reload — so a lead added, removed, or re-{@code tab}'d under
|
||||
* {@code fleet.leaders} needs a restart, the same as any deferred key below.</li>
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Both are
|
||||
* read through a supplier on {@code CompositePeerLauncher}, which is what makes them hot —
|
||||
* not the fact that they are config. Most of {@code fleet:} — every role pool
|
||||
* ({@code architects}/{@code developers}/{@code reviewers}), {@code charters}, and
|
||||
* {@code tabLabel} — is read the same live way, through the same supplier
|
||||
* ({@code () -> config.get().fleet()}). {@code architects} in particular is hot for
|
||||
* <strong>two independent consumers</strong> (fleetd #424): {@code CompositePeerLauncher}
|
||||
* reads it live for placement (which profile an unqualified architect spawn may land on), and
|
||||
* {@code MemberRegistry} separately reads it live, through its own instance of the same
|
||||
* supplier shape, for identity — both which slot a spawn may bind to <em>and</em> what a slot
|
||||
* already bound still grants. Removing an architect slot from config therefore revokes the
|
||||
* {@link dev.ltms.fleet.auth.Role#ARCHITECT} role on the bound pane's very next request; only
|
||||
* the slot <em>occupancy</em> survives, so the demoted session still holds its slot key until
|
||||
* it unbinds. See {@code MemberRegistry}'s class doc for that binding rule.
|
||||
* <strong>But {@code fleet:} as a whole is NOT in this class</strong>: {@code fleet.leaders}
|
||||
* inside the same key is frozen, which is exactly what makes {@code fleet:} split rather than
|
||||
* hot — see below. {@code models:} (fleetd #422) joined this class whole: {@link
|
||||
* FleetConfig#validateModels()} re-runs fully against the fresh config on every {@link
|
||||
* #reload()} (via {@link FleetConfig#validateAll()}), refusing a bad edit outright rather than
|
||||
* caching a stale copy anywhere, and the on/off half added by fleetd #422 is read live both by
|
||||
* {@code CompositePeerLauncher}'s spawn gate ({@code enforceModelEnabled} and its candidate
|
||||
* filter) and by {@code fleet_profiles}/{@code GET /profiles} (via
|
||||
* {@code PeerLauncher.disabledModels()}). Nothing about {@code models:} is baked into an
|
||||
* object built at startup, so — unlike the deferred keys below — there is no frozen half left
|
||||
* to report; it moved here from deferred rather than joining split. An existing profile's
|
||||
* {@code exhaustedPattern} (fleetd #446) joined this class the same way: it used to be
|
||||
* compiled once into {@code Fleetd.main}'s startup pattern map (see the Deferred bullet's old
|
||||
* wording, and {@code LiveExhaustedPatterns}'s class doc for the history), and is now read
|
||||
* live, cached by profile name, by both {@code CompletionResolver}'s classification (via
|
||||
* {@code LiveExhaustedPatterns.patternFor}) and {@code fleet_profiles}'s {@code
|
||||
* exhaustionDetectionArmed} (via {@code LiveExhaustedPatterns.armed}) — the one live object
|
||||
* both read, so a reload that arms or disarms a profile's usage-limit detection takes effect
|
||||
* on the next check with no restart. {@code errorPattern}, {@code exhaustedPattern}'s sibling
|
||||
* key for backend-error (not usage-limit) classification, was deliberately left OUT of this
|
||||
* fleetd #446 change and stays deferred below — the ticket scoped it out explicitly.
|
||||
* {@code leadRollover:} (fleetd #480) joined this class whole, the same shape as
|
||||
* {@code models:} above: {@code dev.ltms.fleet.lead.LeadRollover} holds a
|
||||
* {@code Supplier<FleetConfig.LeadRollover>} (the same {@code () -> config.get().x()} shape)
|
||||
* and reads {@code handoverPath}/{@code requireOperatorConfirm}/{@code maxDocAgeSeconds}/
|
||||
* {@code turnSettleSeconds}/{@code clearSettleSeconds}/{@code bootstrapText} fresh on every
|
||||
* {@code open()}/{@code confirm()} call (and on the deferred post-{@code confirm()}
|
||||
* continuation fleetd #480's correction added — see {@code LeadRollover}'s class doc) rather
|
||||
* than capturing them into fields at construction — unlike its closest
|
||||
* structural cousin {@code leadHeartbeat:}, whose {@code LeadHeartbeatLoop} bakes
|
||||
* {@code idleAfterNanos}/{@code backoffMs}/{@code quietNudgeCap} into final fields. The one
|
||||
* restart-only edge is structural, not a stale value: {@code Fleetd.java} decides whether to
|
||||
* construct the {@code LeadRollover} object at all off the startup snapshot (the same
|
||||
* presence gate {@code leadHeartbeat:} uses), so a block ADDED where it was absent at boot
|
||||
* needs a restart before anything exists to call — the same fact already true of adding a
|
||||
* brand-new {@code profiles:} entry.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code idleSleepGuard:} ({@code Fleetd.java} reads it once, at startup, to decide whether
|
||||
* to construct an {@code IdleSleepGuard} and wire {@code SessionManager}'s
|
||||
* {@code onAcquire}/{@code onRelease} hooks to it — neither is rebuilt on reload, so a
|
||||
* running daemon keeps whatever this was at startup regardless of a later edit),
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code quarantineCooldownSeconds}
|
||||
* (CB-578 stage B — baked once into the {@code BackendQuarantine} built at startup),
|
||||
* {@code guard:}, {@code worktreeRoot:}, adding or removing a profile (a new backend needs its own launcher,
|
||||
* {@code guard:}, {@code worktreeRoot:}, {@code worktreeGroup:} and {@code memberSkills:}
|
||||
* (all three of the latter baked once into the {@code GitWorktrees} built at
|
||||
* {@code Fleetd.java:251} and never rebuilt — fleetd #323 instance 2 found
|
||||
* {@code worktreeGroup} missing from this list and from {@link #changedDeferredKeys};
|
||||
* {@code memberSkills} (fleetd #362) followed the same shape), {@code primary:} (fleetd #326 — {@code Fleetd.java:506, 519,
|
||||
* 520} read {@code cfg.primary()} only off the startup snapshot to build {@code
|
||||
* PrimaryRegistry} and size {@code ReplyPushLoop}'s reminder cap/backoff, and neither is
|
||||
* rebuilt on reload. Say the consequence exactly: {@code primary.terminal} is DEPRECATED
|
||||
* (CB-532, and {@code Fleetd.java:511} warns about it at startup) — a lead's identity comes
|
||||
* from {@code leaders:}/{@code leadScan:}, so changing this pin does not demote or promote a
|
||||
* lead that uses those. What a changed pin still does not take effect on until a restart is
|
||||
* the fallback nudge destination the pin remains, the deprecated identity path for an operator
|
||||
* who still relies on it, and {@code pushReminders}/{@code pushBackoffMs}), {@code configReload:} (fleetd #326 — {@code
|
||||
* Fleetd.java:679-680} read it only at startup to decide whether to build a {@code
|
||||
* ConfigWatcher} at all and with what interval; the watcher that would apply a later change is
|
||||
* itself built once, so a running watcher keeps polling on its original enabled flag and
|
||||
* interval regardless of what a reload changes it to, the same shape as {@code lifecycle} —
|
||||
* not cold, because no already-open resource goes inconsistent with the new value, the watcher
|
||||
* (if any) simply keeps its old settings), adding or removing a profile (a new backend needs its own launcher,
|
||||
* which is constructed once), <em>and an existing profile's launch settings</em> —
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl},
|
||||
* {@code exhaustedPattern} (CB-578 stage A — compiled once into {@code Fleetd.main}'s
|
||||
* pattern map at startup), and the rest. {@code credentialId} (CB-578 stage B) is NOT on
|
||||
* this list — it is read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so it is hot instead.
|
||||
* {@code errorPattern} (fleetd #201 Unit 5 — compiled once into
|
||||
* {@code Fleetd.main}'s backend-error pattern map at startup; deliberately NOT made hot
|
||||
* alongside {@code exhaustedPattern} by fleetd #446 — that ticket scoped {@code errorPattern}
|
||||
* and cooling-off out explicitly),
|
||||
* {@code ideProjectDir} / {@code ideOpenCommand} / {@code autoCompactWindow} (fleetd #323
|
||||
* instance 1 — all three are read at spawn off the same frozen profile map and were missing
|
||||
* from {@link #sameLaunchSettings}), and the rest of {@link #sameLaunchSettings}.
|
||||
* {@code credentialId} (CB-578 stage B) and {@code exhaustedPattern} (fleetd #446) are NOT on
|
||||
* this list — both are read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so both are hot instead
|
||||
* (see the Hot bullet above for {@code exhaustedPattern}'s history).
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} at construction and resolves
|
||||
* each spawn out of that copy, so those never reach a launch until the daemon restarts. A
|
||||
* reload logs these rather than pretending they applied.</li>
|
||||
* <li><strong>Cold</strong> — cannot change at all under a running daemon: {@code bind:},
|
||||
* {@code herdrSocket:}, {@code broker:} and {@code auth:}. The socket is bound, the broker
|
||||
* <li><strong>Split</strong> (fleetd #330; extended to a third key by fleetd #333) — read
|
||||
* <em>both</em> ways at different sites, so the key does not fit any class above as a whole:
|
||||
* {@code health:}, {@code coordinator:} and {@code fleet:}. Each is read off the startup
|
||||
* snapshot to build a long-lived object, and read live off {@link #get()} at a different,
|
||||
* unrelated site — so half of a reload's effect already applies while the other half waits
|
||||
* for a restart, and a bare "config reloaded" would under-claim by exactly that half.
|
||||
* <ul>
|
||||
* <li>{@code health:} — the monitor itself ({@code enabled}, {@code intervalSeconds},
|
||||
* {@code workingSuspectAfterSeconds}) is built once at {@code Fleetd.java:556-563}
|
||||
* and never rebuilt, so a changed value needs a restart to actually start, stop, or
|
||||
* retime it. The coverage string {@code fleet_profiles} reports
|
||||
* ({@code Fleetd.java:648-650}) is read live off {@link #get()} on every call, so it
|
||||
* already reflects the new value.</li>
|
||||
* <li>{@code coordinator:} — the {@code LeadMailbox} connection ({@code uri},
|
||||
* {@code uriEnv}, {@code selfId}, {@code prefetch}) is opened once at
|
||||
* {@code Fleetd.java:502} and never reopened, so a changed value needs a restart —
|
||||
* {@code selfId} in particular names this daemon's own AMQP inbox queue, and a peer
|
||||
* lead that learned the old name would not discover a new one on its own. The broker
|
||||
* URI env-var <em>name</em> that {@code MemberEnvAllowList} keeps out of a member's
|
||||
* environment is read live off {@link #get()} on every spawn
|
||||
* ({@code HerdrPeerLauncher.java:1530}), so it already applies.</li>
|
||||
* <li>{@code fleet:} (fleetd #333) — {@code fleet.leaders} is the frozen half:
|
||||
* {@code Fleetd.java:281} reads {@code cfg.fleet().leaders()} off the startup snapshot
|
||||
* for two long-lived objects built right after it and never rebuilt — the
|
||||
* {@code LeadTabScanner}'s {@code tab label → lead name} map ({@code Fleetd.java:301},
|
||||
* wired into {@code CallerResolver.withLeadsAndMembers} at {@code Fleetd.java:620/624},
|
||||
* which is how a caller's pane is recognised as a lead at all) and, when herdr answered
|
||||
* at startup, {@code LeadLauncher(...).ensureLeads()} ({@code Fleetd.java:315}), which
|
||||
* auto-launches each declared lead up to its {@code instances} count. So a lead added,
|
||||
* removed, or given a new {@code tab:} label under {@code fleet.leaders} needs a
|
||||
* restart — until then it is invisible to identity resolution, and this is exactly the
|
||||
* scenario fleetd #333 named: an operator edits a lead's {@code tab:} to match a
|
||||
* renamed pane, sees "config reloaded", and the pane keeps resolving as a worker,
|
||||
* because {@code CallerResolver} is still matching against the old label. The live
|
||||
* half is the rest of {@code fleet:} — {@code architects}/{@code developers}/
|
||||
* {@code reviewers}, {@code charters}, {@code tabLabel} — read live through the same
|
||||
* {@code CompositePeerLauncher} supplier the Hot bullet above names, so a reload that
|
||||
* only touches those already applies with nothing to report. Because the hot and frozen
|
||||
* halves of {@code fleet:} are disjoint sub-fields rather than the same fields read two
|
||||
* ways (contrast {@code coordinator.uriEnv} above), {@link #changedSplitKeys} compares
|
||||
* {@code fleet.leaders} alone, not the whole {@code Fleet} record — comparing the whole
|
||||
* record would report "split" for a {@code tabLabel}-only change that is actually fully
|
||||
* hot, over-claiming in exactly the direction this class exists to avoid under-claiming
|
||||
* in.</li>
|
||||
* </ul>
|
||||
* A split change is still accepted — {@link Outcome#applied()} stays {@code true}, the same
|
||||
* as a deferred change — because the live half genuinely took effect; refusing the whole
|
||||
* reload would leave the operator worse off than today. {@link Outcome#split()} names the
|
||||
* key and says which half is which each time, rather than trying to score "how changed" a
|
||||
* mixed key is or handle "both halves changed in one reload" as a special case.</li>
|
||||
* <li><strong>Cold</strong> — cannot change at all under a running daemon. All five of
|
||||
* {@link #COLD_KEYS}: {@code bind:}, {@code herdrSocket:}, {@code memberHerdrSocket:},
|
||||
* {@code broker:} and {@code auth:}. The sockets are already connected, the broker
|
||||
* connection is open, and the auth mode decides who may reach the port that is already
|
||||
* listening.</li>
|
||||
* listening. This bullet omitted {@code memberHerdrSocket:} until fleetd #333 — say "all
|
||||
* five of COLD_KEYS" rather than re-listing them, so prose and set cannot drift again.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>The denominator, measured on 2026-09-04 (fleetd #330; recounted for fleetd #333);
|
||||
* recounted again for fleetd #362, again after {@code idleSleepGuard:} was added, again after
|
||||
* {@code models:} was added as deferred, again for fleetd #422, which moved {@code models:}
|
||||
* from deferred to hot-excluded once its on/off half was read live everywhere, and again after
|
||||
* {@code leadRollover:} was added (fleetd #480).</strong>
|
||||
* {@code FleetConfig} has 26 top-level record components: 5 cold, 13 deferred, 3 split, 5
|
||||
* hot-excluded. Five of them are named nowhere in this file, and the reason is the same for all
|
||||
* five: {@code placement}, {@code memberCredentials}, {@code memberLoginShell}, {@code models} and
|
||||
* {@code leadRollover} are <strong>hot</strong> and correctly absent — all five are read live off
|
||||
* {@code config.get()} (placement through the {@code CompositePeerLauncher} supplier the Hot bullet
|
||||
* names; {@code memberCredentials}/{@code memberLoginShell} at spawn time, {@code Fleetd.java:198,
|
||||
* 205, 729} and {@code HerdrPeerLauncher#configuredMemberLoginShell}; {@code models} the same way,
|
||||
* through the Hot bullet's {@code models:} paragraph; {@code leadRollover} through the Hot bullet's
|
||||
* {@code leadRollover:} paragraph), so a reload takes effect on the next spawn (or, for
|
||||
* {@code models}, the next reported status; for {@code leadRollover}, the next {@code open()}/
|
||||
* {@code confirm()} call) with no entry needed here.
|
||||
* {@code health} and {@code coordinator} used to be a third kind — <strong>undecided</strong>, not
|
||||
* hot — until fleetd #330 added the <strong>split</strong> class above and gave them a home. A
|
||||
* reload touching either used to report a bare "config reloaded", which under-claimed; now it names
|
||||
* the key and says which half is which. {@code fleet} was the same story in reverse: fleetd #330's
|
||||
* own fact-find named {@code health}/{@code coordinator} as "the complete set of split-shaped keys"
|
||||
* and filed {@code fleet.leaders}'s restart requirement as a documented caveat sitting in the
|
||||
* <strong>hot-excluded</strong> escape hatch instead — correctly documented, but in the one bucket
|
||||
* this file's own coverage test cannot check the truth of (see that test's javadoc). fleetd #333
|
||||
* moved it into <strong>split</strong>, where {@link #changedSplitKeys} actually reports it.
|
||||
* <p>The point of writing the count down: "not mentioned in this file" looks identical for a key
|
||||
* that is correctly hot and for a key nobody triaged. Three times now — {@code worktreeGroup} (#323),
|
||||
* {@code primary}/{@code configReload} (#326), and {@code fleet.leaders} sitting in the escape hatch
|
||||
* (#333) — the second kind hid among the first. A top-level coverage checker in the
|
||||
* {@link ConfigRefProfileCoverageTest} shape (one level up, over {@code FleetConfig} itself rather
|
||||
* than {@code FleetConfig.Profile}) proves this file's four classes exhaust the record's components
|
||||
* — see {@code ConfigRefTopLevelCoverageTest}. That test proves the record's <em>shape</em> is fully
|
||||
* triaged; it does NOT prove a {@code SPLIT_KEYS}/{@code COLD_KEYS}/{@code DEFERRED_KEYS} member has
|
||||
* any reporting code behind it at all — {@code ConfigRefTopLevelReportingCoverageTest} is what
|
||||
* fleetd #333 added for that, after measuring that a {@code SPLIT_KEYS} entry with its reporting
|
||||
* branch deleted passes both this file's own "kept in step" assert and
|
||||
* {@code ConfigRefTopLevelCoverageTest} unchanged. fleetd #337 extended it to {@code DEFERRED_KEYS}
|
||||
* after measuring the same one-way gap there directly: dropping {@code guard}'s branch out of
|
||||
* {@link #changedDeferredKeys} while {@code "guard"} stayed in the set left the whole suite green.
|
||||
*
|
||||
* <p><strong>A cold change refuses the whole reload.</strong> Not the hot half applied and the cold
|
||||
* half warned about: that would leave the running daemon in a state matching no file on disk, which
|
||||
* is the worst thing a reload can do to an operator debugging one. Refusing keeps the invariant that
|
||||
* the live config is always some version of the file, and the message names the keys that must
|
||||
* change through a restart.
|
||||
* change through a restart. A split change does <em>not</em> refuse, for a different reason than a
|
||||
* deferred change does not: its live half genuinely took effect, so refusing would throw that away
|
||||
* and leave the operator worse off than the partial-but-honest report {@link Outcome#split()} gives.
|
||||
*
|
||||
* <p>A reload that fails to parse or fails validation is also refused, and the previous config keeps
|
||||
* running. A config file being edited is normally read once mid-save; degrading a working daemon
|
||||
* because it caught a half-written file would be a bad trade.
|
||||
*
|
||||
* <p><strong>fleetd #474</strong> — {@link FleetConfig#validateAll()} is not the only gate startup
|
||||
* runs before a config takes effect: {@code Fleetd.main} also calls {@code
|
||||
* dev.ltms.fleet.mcp.CharterToolSurface#assertChartersNameOnlyRegisteredTools}, right after {@code
|
||||
* cfg.validateAll()}, to refuse a charter that names an MCP tool the server does not register. That
|
||||
* check cannot live inside {@link FleetConfig} — {@code CharterToolSurface} lives in the {@code mcp}
|
||||
* package because the canonical tool set ({@code FleetTool}) does, and config is loaded before the
|
||||
* MCP server exists, so {@code FleetConfig} must not gain a dependency on {@code mcp}. {@link
|
||||
* #reload} cannot import {@code mcp} either, for the same reason applied one layer up: {@code
|
||||
* dev.ltms.fleet.config} is loaded before {@code dev.ltms.fleet.mcp} exists, same as {@code
|
||||
* FleetConfig}. So this class accepts the check as a {@code Consumer<FleetConfig>} —
|
||||
* {@link #extraValidation} — supplied by whichever caller already sits at the seam that holds both
|
||||
* a loaded {@code FleetConfig} and the {@code mcp} package: {@code Fleetd.main}. It is invoked
|
||||
* inside the same try/catch as {@code fresh.validateAll()}, so a charter that would have refused to
|
||||
* boot refuses a reload too, and keeps the running config exactly like any other {@code
|
||||
* validateAll()} failure. A ref built through the two-argument constructor (every test fixture that
|
||||
* does not care about this check, and {@link #fixed}) gets a no-op consumer, so nothing outside
|
||||
* {@code Fleetd.main} needs to know this hook exists.
|
||||
*/
|
||||
public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigRef.class);
|
||||
|
||||
/** Keys that cannot change under a running daemon — see the class doc. */
|
||||
private static final Set<String> COLD_KEYS =
|
||||
/**
|
||||
* Keys that cannot change under a running daemon — see the class doc.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelCoverageTest} can fold it
|
||||
* into the top-level triage it checks, the same way it reads {@link #SPLIT_KEYS}.
|
||||
*/
|
||||
static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "memberHerdrSocket", "broker", "auth");
|
||||
|
||||
/**
|
||||
* Keys read BOTH off the startup snapshot and live off {@link #get()} at different sites, so
|
||||
* neither the hot, deferred nor cold class fits them as a whole — see the class doc's Split
|
||||
* bullet (fleetd #330). A changed split key is accepted ({@link Outcome#applied()} stays
|
||||
* {@code true}) and reported by name, with a message naming which half is live and which needs
|
||||
* a restart.
|
||||
*/
|
||||
static final Set<String> SPLIT_KEYS = Set.of("health", "coordinator", "fleet");
|
||||
|
||||
/**
|
||||
* Top-level keys {@link #changedDeferredKeys} compares — see the class doc's Deferred bullet.
|
||||
* Promoted here from a test-side copy in {@code ConfigRefTopLevelCoverageTest} by fleetd #337,
|
||||
* the same reason {@link #COLD_KEYS} and {@link #SPLIT_KEYS} live here rather than in a test: a
|
||||
* second, hand-maintained copy of this set is exactly the kind of thing that silently drifts
|
||||
* from the method it is supposed to describe. {@code spawnReadyTimeoutMs} and
|
||||
* {@code spawnReadyPollMs} are compared together in one branch and reported under the combined
|
||||
* label {@code "spawnReady*"}; {@code profiles} is compared twice over (added/removed names,
|
||||
* then an existing profile's launch settings) — see {@link #changedDeferredKeys}.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelCoverageTest} and
|
||||
* {@code ConfigRefTopLevelReportingCoverageTest} can both read it, the same way they already
|
||||
* read {@link #COLD_KEYS} and {@link #SPLIT_KEYS}.
|
||||
*/
|
||||
static final Set<String> DEFERRED_KEYS = Set.of(
|
||||
"guard", "worktreeRoot", "worktreeGroup", "memberSkills", "primary", "configReload",
|
||||
"leadHeartbeat", "lifecycle", "spawnReadyTimeoutMs", "spawnReadyPollMs",
|
||||
"quarantineCooldownSeconds", "profiles", "idleSleepGuard");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<FleetConfig> current;
|
||||
private final Consumer<FleetConfig> extraValidation;
|
||||
|
||||
/** Equivalent to the three-argument constructor with a no-op {@code extraValidation}. */
|
||||
public ConfigRef(Path path, FleetConfig initial) {
|
||||
this(path, initial, cfg -> { });
|
||||
}
|
||||
|
||||
/**
|
||||
* @param extraValidation run on every {@link #reload} candidate, inside the same try/catch as
|
||||
* {@code fresh.validateAll()} — see the class doc's fleetd #474 note.
|
||||
* {@code Fleetd.main} passes {@code
|
||||
* Fleetd::assertChartersNameOnlyRegisteredTools} (a package-private
|
||||
* {@code FleetConfig -> void} adapter over {@code
|
||||
* CharterToolSurface#assertChartersNameOnlyRegisteredTools}), so a reload
|
||||
* runs the same gate startup does without this class depending on the
|
||||
* {@code mcp} package.
|
||||
*/
|
||||
public ConfigRef(Path path, FleetConfig initial, Consumer<FleetConfig> extraValidation) {
|
||||
this.path = path;
|
||||
this.current = new AtomicReference<>(Objects.requireNonNull(initial, "initial config"));
|
||||
this.extraValidation = Objects.requireNonNull(extraValidation, "extraValidation");
|
||||
}
|
||||
|
||||
/** A fixed reference that never reloads — for tests and for wiring built from a config in code. */
|
||||
@@ -100,25 +334,37 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
/**
|
||||
* What a reload attempt did.
|
||||
*
|
||||
* <p>{@code split} is a separate field from {@code deferred} rather than a differently-worded
|
||||
* entry inside it, because the two carry different guarantees for any caller that branches on
|
||||
* them rather than just printing {@link #summary()}: every {@code deferred} entry means "this
|
||||
* key's whole change waits for a restart", while every {@code split} entry means "part of this
|
||||
* key's change already applied, and the message says which part" — collapsing them would force
|
||||
* a caller to re-parse the message to tell those apart. See the class doc's Split bullet
|
||||
* (fleetd #330) for why the key needs this at all.
|
||||
*
|
||||
* @param applied true when the new config is now live
|
||||
* @param coldKeys cold keys whose value changed, which is why an unapplied reload was refused
|
||||
* @param deferred keys that changed and were accepted, but whose effect waits for a restart
|
||||
* @param deferred keys that changed and were accepted, but whose effect waits entirely on a
|
||||
* restart
|
||||
* @param split split keys that changed and were accepted, each named with which half of it
|
||||
* is already live and which half waits for a restart
|
||||
* @param error the parse or validation failure that refused the reload, else {@code null}
|
||||
*/
|
||||
public record Outcome(boolean applied, List<String> coldKeys, List<String> deferred,
|
||||
String error) {
|
||||
List<String> split, String error) {
|
||||
|
||||
public Outcome {
|
||||
coldKeys = List.copyOf(coldKeys);
|
||||
deferred = List.copyOf(deferred);
|
||||
split = List.copyOf(split);
|
||||
}
|
||||
|
||||
static Outcome refusedCold(List<String> keys) {
|
||||
return new Outcome(false, keys, List.of(), null);
|
||||
return new Outcome(false, keys, List.of(), List.of(), null);
|
||||
}
|
||||
|
||||
static Outcome failed(String error) {
|
||||
return new Outcome(false, List.of(), List.of(), error);
|
||||
return new Outcome(false, List.of(), List.of(), List.of(), error);
|
||||
}
|
||||
|
||||
/** A one-line summary for the operator — the reason, not just the verdict. */
|
||||
@@ -130,11 +376,18 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
return "config reload refused — these keys cannot change under a running daemon: "
|
||||
+ String.join(", ", coldKeys) + ". Restart fleetd to apply them.";
|
||||
}
|
||||
if (!deferred.isEmpty()) {
|
||||
return "config reloaded; these changes need a restart to take effect: "
|
||||
+ String.join(", ", deferred);
|
||||
if (deferred.isEmpty() && split.isEmpty()) {
|
||||
return "config reloaded";
|
||||
}
|
||||
return "config reloaded";
|
||||
StringBuilder out = new StringBuilder("config reloaded");
|
||||
if (!deferred.isEmpty()) {
|
||||
out.append("; these changes need a restart to take effect: ")
|
||||
.append(String.join(", ", deferred));
|
||||
}
|
||||
if (!split.isEmpty()) {
|
||||
out.append("; partially live — ").append(String.join(" | ", split));
|
||||
}
|
||||
return out.toString();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -155,12 +408,18 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
fresh = FleetConfig.load(path);
|
||||
// The same gate startup runs. A config that would have refused to boot must not be able
|
||||
// to slip in through a reload — that is how a daemon ends up in a state it could never
|
||||
// have started in, which is the hardest kind to debug.
|
||||
fresh.validateAuthExposure();
|
||||
fresh.validateLeadTabPrefixes();
|
||||
fresh.validateSubscriptionProfiles();
|
||||
fresh.validateCharters();
|
||||
fresh.validateMembers();
|
||||
// have started in, which is the hardest kind to debug. fleetd ticket "central allow-list
|
||||
// of usable models" follow-up: this used to be six individual validateXxx() calls, and
|
||||
// mutation testing found two of the six unpinned here even though startup pinned nothing
|
||||
// at all — see FleetConfig#validateAll's javadoc for why the fix is one reflective call,
|
||||
// not a longer hand-maintained list.
|
||||
fresh.validateAll();
|
||||
// fleetd #474: validateAll() does not cover everything startup refuses on — the charter
|
||||
// tool-surface check (Fleetd.main, right after cfg.validateAll()) lives outside
|
||||
// FleetConfig on purpose (see this class's doc) and is supplied here as extraValidation.
|
||||
// Same try/catch as validateAll() above, on purpose: either failure must refuse the whole
|
||||
// reload and keep the running config the same way.
|
||||
extraValidation.accept(fresh);
|
||||
} catch (RuntimeException e) {
|
||||
String msg = e.getMessage() == null ? e.toString() : e.getMessage();
|
||||
log.warn("config reload from {} refused, keeping the running config: {}", path, msg);
|
||||
@@ -175,14 +434,21 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
}
|
||||
|
||||
List<String> deferred = changedDeferredKeys(old, fresh);
|
||||
List<String> split = changedSplitKeys(old, fresh);
|
||||
current.set(fresh);
|
||||
Outcome out = new Outcome(true, List.of(), deferred, null);
|
||||
Outcome out = new Outcome(true, List.of(), deferred, split, null);
|
||||
log.info(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Cold keys whose value differs between the running config and the candidate. */
|
||||
private static List<String> changedColdKeys(FleetConfig old, FleetConfig fresh) {
|
||||
/**
|
||||
* Cold keys whose value differs between the running config and the candidate.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #sameLaunchSettings} is package-private — see that test's class doc (fleetd #333).
|
||||
*/
|
||||
static List<String> changedColdKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.bind(), fresh.bind())) {
|
||||
changed.add("bind");
|
||||
@@ -204,8 +470,16 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
return changed;
|
||||
}
|
||||
|
||||
/** Changed keys that were accepted but whose effect waits for a restart. */
|
||||
private static List<String> changedDeferredKeys(FleetConfig old, FleetConfig fresh) {
|
||||
/**
|
||||
* Changed keys that were accepted but whose effect waits for a restart.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #changedColdKeys} and {@link #changedSplitKeys} already are (fleetd #333, extended to
|
||||
* this method by fleetd #337 — membership in {@link #DEFERRED_KEYS} proved nothing about this
|
||||
* method on its own until then; see that test's class doc).
|
||||
*/
|
||||
static List<String> changedDeferredKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.lifecycle(), fresh.lifecycle())) {
|
||||
changed.add("lifecycle");
|
||||
@@ -219,6 +493,44 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
if (!Objects.equals(old.worktreeRoot(), fresh.worktreeRoot())) {
|
||||
changed.add("worktreeRoot");
|
||||
}
|
||||
// Baked into the same GitWorktrees as worktreeRoot (Fleetd.java:251) and never rebuilt
|
||||
// either — see the class doc. Missing this check was fleetd #323 instance 2: a reload
|
||||
// that changed only worktreeGroup reported "config reloaded" with nothing deferred, and
|
||||
// newly provisioned worktrees kept the old sharing behaviour.
|
||||
if (!Objects.equals(old.worktreeGroup(), fresh.worktreeGroup())) {
|
||||
changed.add("worktreeGroup");
|
||||
}
|
||||
// fleetd #362: baked into the same GitWorktrees as worktreeRoot/worktreeGroup
|
||||
// (Fleetd.java:251) and never rebuilt either — a reload that changes only memberSkills
|
||||
// must be reported the same way, or a newly provisioned worktree keeps seeding from (or
|
||||
// skipping) the old source directory with nothing telling the operator why.
|
||||
if (!Objects.equals(old.memberSkills(), fresh.memberSkills())) {
|
||||
changed.add("memberSkills");
|
||||
}
|
||||
// fleetd #326: Fleetd.java:506, 519, 520 read cfg.primary() only off the startup snapshot
|
||||
// (PrimaryRegistry's pinned terminal, ReplyPushLoop's reminder cap and backoff) — neither is
|
||||
// rebuilt on reload, so a changed value needs a restart. Note what it does NOT mean:
|
||||
// primary.terminal is deprecated (CB-532), identity comes from leaders:/leadScan:, so a lead
|
||||
// using those is unaffected by this pin either way. See the class doc for the exact scope.
|
||||
if (!Objects.equals(old.primary(), fresh.primary())) {
|
||||
changed.add("primary");
|
||||
}
|
||||
// fleetd #326: Fleetd.java:679-680 read cfg.configReload() only at startup to decide whether
|
||||
// to build a ConfigWatcher at all and with what interval — the watcher that would apply a
|
||||
// later change is itself built once, so a running watcher keeps its original enabled flag and
|
||||
// interval regardless of what a reload changes it to. Not cold: no already-open resource goes
|
||||
// inconsistent with the new value, a watcher (if any) simply keeps polling on the old settings.
|
||||
if (!Objects.equals(old.configReload(), fresh.configReload())) {
|
||||
changed.add("configReload");
|
||||
}
|
||||
// Fleetd.java reads cfg.idleSleepGuard() once, at startup, to decide whether to construct
|
||||
// an IdleSleepGuard at all and wire SessionManager's onAcquire/onRelease hooks to it —
|
||||
// neither is rebuilt on reload, so a running daemon keeps whatever this was at startup
|
||||
// (armed or not) regardless of a later edit here. Not cold: nothing already-open goes
|
||||
// inconsistent with the new value, an armed-or-not guard just keeps its original answer.
|
||||
if (!Objects.equals(old.idleSleepGuard(), fresh.idleSleepGuard())) {
|
||||
changed.add("idleSleepGuard");
|
||||
}
|
||||
if (!Objects.equals(old.spawnReadyTimeoutMs(), fresh.spawnReadyTimeoutMs())
|
||||
|| !Objects.equals(old.spawnReadyPollMs(), fresh.spawnReadyPollMs())) {
|
||||
changed.add("spawnReady*");
|
||||
@@ -263,14 +575,114 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically. Compares every component
|
||||
* the launcher reads at spawn; {@code weight}, {@code maxLoad} and {@code credentialId} are
|
||||
* excluded because those are read live (by the placement policy and, for credentialId, by
|
||||
* {@code CompositePeerLauncher}/the CB-578 stage B exhaustion sink) and really do take effect on
|
||||
* the next spawn.
|
||||
* Split keys whose value differs between the running config and the candidate — see the class
|
||||
* doc's Split bullet (fleetd #330, extended for {@code fleet:} by fleetd #333). Unlike
|
||||
* {@link #changedDeferredKeys}, this does not try to tell which sub-field moved for {@code
|
||||
* health:} or {@code coordinator:}: any change to either gets the same fixed message, because
|
||||
* the message already names both halves every time, so there is no "which half changed"
|
||||
* question left for the caller to answer. {@code fleet:} is different on purpose — see below.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #sameLaunchSettings} is package-private — see that test's class doc (fleetd #333). That
|
||||
* test exists because membership in {@link #SPLIT_KEYS} proves nothing about this method on its
|
||||
* own: fleetd #333 measured that dropping the {@code coordinator} branch out of this method
|
||||
* while leaving {@code "coordinator"} in {@code SPLIT_KEYS} left the whole suite green except a
|
||||
* hand-written {@code ConfigRefTest} case — neither {@code ConfigRefTopLevelCoverageTest} (it
|
||||
* only reads the set) nor the "kept in step" assert below (it only checks the reported keys are
|
||||
* a SUBSET of {@code SPLIT_KEYS}, never that every {@code SPLIT_KEYS} member has a branch here)
|
||||
* would have caught it.
|
||||
*/
|
||||
private static boolean sameLaunchSettings(FleetConfig.Profile a, FleetConfig.Profile b) {
|
||||
return Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
static List<String> changedSplitKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.health(), fresh.health())) {
|
||||
changed.add("health: the monitor itself (enabled, interval, workingSuspectAfter) is "
|
||||
+ "frozen at startup and needs a restart; the coverage status fleet_profiles "
|
||||
+ "reports is read live and already applied");
|
||||
}
|
||||
if (!Objects.equals(old.coordinator(), fresh.coordinator())) {
|
||||
changed.add("coordinator: the LeadMailbox connection (uri, uriEnv, selfId, prefetch) is "
|
||||
+ "opened once and needs a restart; the broker URI env-var name kept out of a "
|
||||
+ "member's environment is read live on every spawn and already applied");
|
||||
}
|
||||
// fleetd #333: unlike health/coordinator above, most of `fleet:` (developers, reviewers,
|
||||
// charters, tabLabel) is genuinely hot — ConfigRefTest.aHotChangeIsAppliedAndRead-
|
||||
// ThroughGet and aCharterChangeIsHotAndReachesTheLiveConfig prove it reaches the live config
|
||||
// with no restart note. `architects` is hot too, and — since fleetd #424 — hot for BOTH of
|
||||
// its consumers, not just the one this comment used to name: CompositePeerLauncher reads it
|
||||
// live for PLACEMENT through the () -> config.get().fleet() supplier named in the class doc's
|
||||
// Hot bullet, and MemberRegistry separately reads it live for IDENTITY (which slot a spawn
|
||||
// may bind to, AND what a slot already bound still grants) through its own instance of that
|
||||
// same supplier shape — see MemberRegistry.live and its class doc for the binding rule:
|
||||
// removing a slot revokes ARCHITECT on the bound pane's very next request, and only the slot
|
||||
// OCCUPANCY survives, so the demoted session keeps its slot key until it unbinds. Only
|
||||
// fleet.leaders is frozen (Fleetd.java:281 reads cfg.fleet().leaders() off the startup
|
||||
// snapshot to build both the LeadTabScanner's tab-label-to-name map, wired into
|
||||
// CallerResolver.withLeadsAndMembers at Fleetd.java:620/624, and — when herdr answered —
|
||||
// LeadLauncher(...).ensureLeads() at Fleetd.java:315, which auto-launches each lead up to its
|
||||
// `instances` count; neither is rebuilt on reload). So this compares fleet.leaders alone, not
|
||||
// the whole Fleet record: comparing the whole record would report "split" for a tabLabel-only
|
||||
// or architects-only change that is actually fully hot, which is the over-claim mirror of the
|
||||
// under-claim bug this class exists to prevent.
|
||||
if (!Objects.equals(leadersOf(old), leadersOf(fresh))) {
|
||||
changed.add("fleet: fleet.leaders (each lead's tab, workspace, cwd, profile and "
|
||||
+ "instances count) is read once at startup to build the LeadTabScanner's "
|
||||
+ "identity map and to auto-launch leads, and neither is rebuilt on reload, so a "
|
||||
+ "lead added, removed, or given a new tab: label needs a restart — until then it "
|
||||
+ "stays unrecognised, and a caller from its new tab resolves as a worker, not a "
|
||||
+ "lead; the rest of fleet: (developers, reviewers, charters, tabLabel) is read "
|
||||
+ "live through the supplier on CompositePeerLauncher, and architects is read "
|
||||
+ "live through that same supplier for placement AND through a separate supplier "
|
||||
+ "on MemberRegistry for spawn-time identity — both already applied");
|
||||
}
|
||||
// Kept in step with SPLIT_KEYS the same way changedColdKeys is kept in step with COLD_KEYS —
|
||||
// every message here must be traceable to one of the split keys the class doc documents.
|
||||
// NOTE what this does NOT prove, per the javadoc above: it does not catch a SPLIT_KEYS
|
||||
// member with no branch above at all, only a branch whose message is mis-worded relative to
|
||||
// the set. ConfigRefTopLevelReportingCoverageTest is what proves the former.
|
||||
assert changed.stream().allMatch(m -> SPLIT_KEYS.stream().anyMatch(k -> m.startsWith(k + ":")))
|
||||
: "a split entry was reported that does not start with a SPLIT_KEYS name: " + changed;
|
||||
return changed;
|
||||
}
|
||||
|
||||
/** {@code cfg.fleet().leaders()}, defensively, in case a caller hands in a non-defaulted config. */
|
||||
private static Map<String, FleetConfig.Leader> leadersOf(FleetConfig cfg) {
|
||||
return cfg.fleet() == null ? Map.of() : cfg.fleet().leaders();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@link FleetConfig.Profile} record components deliberately left out of
|
||||
* {@link #sameLaunchSettings} because they are read <em>live</em>, not baked in at spawn — see
|
||||
* the class doc's <em>Hot</em> bullet. {@code weight} and {@code maxLoad} are read live by the
|
||||
* placement policy on every spawn; {@code credentialId} is read live by
|
||||
* {@code CompositePeerLauncher} and the CB-578 stage B exhaustion sink; {@code exhaustedPattern}
|
||||
* (fleetd #446) is read live, cached by profile name, by {@code LiveExhaustedPatterns} — see
|
||||
* that class's doc and the class doc's <em>Hot</em> bullet for the history (it used to be
|
||||
* compared here, deferred, like its sibling {@code errorPattern} still is). Nothing else is
|
||||
* excluded — see {@code sameLaunchSettingsComparesEveryProfileComponentOrExcludesIt} in
|
||||
* {@code ConfigRefProfileCoverageTest}, which enumerates every {@code Profile} record component
|
||||
* by reflection and fails the build if one is neither compared below nor named here.
|
||||
*/
|
||||
static final Set<String> LAUNCH_SETTINGS_EXCLUDED =
|
||||
Set.of("weight", "maxLoad", "credentialId", "exhaustedPattern");
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically.
|
||||
*
|
||||
* <p>This must compare every {@link FleetConfig.Profile} record component except the three in
|
||||
* {@link #LAUNCH_SETTINGS_EXCLUDED}. That is not a claim this javadoc can make good on by
|
||||
* itself — a javadoc saying "compares every component" is exactly what fleetd #323 found to be
|
||||
* false for three fields (and a sibling method's field list, for a fourth). The actual
|
||||
* guarantee comes from {@code ConfigRefProfileCoverageTest}: it enumerates every record
|
||||
* component of {@code FleetConfig.Profile} by reflection, mutates each one not in
|
||||
* {@code LAUNCH_SETTINGS_EXCLUDED} on a base profile, and asserts this method reports a
|
||||
* difference — so a new component that is neither compared here nor added to
|
||||
* {@code LAUNCH_SETTINGS_EXCLUDED} (with a reason) fails that test by name, rather than
|
||||
* silently reporting "config reloaded" for a value the daemon never picked up.
|
||||
*/
|
||||
static boolean sameLaunchSettings(FleetConfig.Profile a, FleetConfig.Profile b) {
|
||||
return Objects.equals(a.profile(), b.profile())
|
||||
&& Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
&& Objects.equals(a.model(), b.model())
|
||||
&& Objects.equals(a.configDir(), b.configDir())
|
||||
&& Objects.equals(a.tokenEnv(), b.tokenEnv())
|
||||
@@ -289,9 +701,25 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
&& Objects.equals(a.kind(), b.kind())
|
||||
&& Objects.equals(a.env(), b.env())
|
||||
&& Objects.equals(a.subscription(), b.subscription())
|
||||
// CB-578 stage B: exhaustedPattern is compiled once into Fleetd.main's pattern map
|
||||
// at startup (see ExhaustedPatternLookup wiring) — a reload never re-reads it, so a
|
||||
// changed pattern must be reported as deferred, exactly like model/baseUrl/argv.
|
||||
&& Objects.equals(a.exhaustedPattern(), b.exhaustedPattern());
|
||||
// fleetd #446: exhaustedPattern moved to LAUNCH_SETTINGS_EXCLUDED — it is now read
|
||||
// live, cached by profile name, through LiveExhaustedPatterns (see that class's doc
|
||||
// and ConfigRef's class doc Hot bullet), so it must NOT be compared here any more: a
|
||||
// reload that changes only exhaustedPattern must report "config reloaded", not
|
||||
// "these changes need a restart".
|
||||
// fleetd #201 Unit 5: errorPattern is compiled once into Fleetd.main's backend-error
|
||||
// pattern map at startup (see BackendErrorPatternLookup wiring) — unlike its sibling
|
||||
// exhaustedPattern above (fleetd #446), a reload still never re-reads it; fleetd #446
|
||||
// scoped errorPattern out on purpose (see the class doc's Hot bullet).
|
||||
&& Objects.equals(a.errorPattern(), b.errorPattern())
|
||||
// fleetd #323 instance 1: ideProjectDir and ideOpenCommand are read at spawn off the
|
||||
// same frozen profile map as ideMcpUrl above (ClaudeCodeLauncher.java:267/269,
|
||||
// OpenCodeLauncher.java:474/480/486) and were missing from this comparison.
|
||||
&& Objects.equals(a.ideProjectDir(), b.ideProjectDir())
|
||||
&& Objects.equals(a.ideOpenCommand(), b.ideOpenCommand())
|
||||
// fleetd #323 instance 1: autoCompactWindow is read at spawn the same way
|
||||
// (ClaudeCodeLauncher.java:926, OpenCodeLauncher.java:650). Comparing it here only
|
||||
// makes the reload REPORT that a restart is needed — it deliberately does not make
|
||||
// autoCompactWindow take effect live, which is a separate, larger change.
|
||||
&& Objects.equals(a.autoCompactWindow(), b.autoCompactWindow());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import com.fasterxml.jackson.annotation.JsonCreator;
|
||||
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
|
||||
import com.fasterxml.jackson.annotation.JsonProperty;
|
||||
import com.fasterxml.jackson.core.JsonParser;
|
||||
import com.fasterxml.jackson.core.JsonToken;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
@@ -14,16 +16,22 @@ import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.lang.reflect.InvocationTargetException;
|
||||
import java.lang.reflect.Method;
|
||||
import java.lang.reflect.Modifier;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Arrays;
|
||||
import java.util.Collections;
|
||||
import java.util.Comparator;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.regex.PatternSyntaxException;
|
||||
|
||||
/**
|
||||
* {@code fleetd} configuration, loaded from a YAML file (see
|
||||
@@ -62,12 +70,14 @@ import java.util.Set;
|
||||
* {@code fixed} (default), {@code round-robin}, or {@code weighted}
|
||||
* @param auth API authentication mode ({@code null} → {@code loopback-trust}, the
|
||||
* historical behaviour), CB-501
|
||||
* @param quarantineCooldownSeconds how long a credential stays quarantined after a
|
||||
* @param quarantineCooldownSeconds the BASE cooldown a credential is quarantined for after a
|
||||
* {@code BACKEND_EXHAUSTED} classification (CB-578 stage B); {@code null}/{@code
|
||||
* <=0} → {@link #DEFAULT_QUARANTINE_COOLDOWN_SECONDS}. Baked once into the
|
||||
* {@code BackendQuarantine} built at startup, so it is DEFERRED: changing it
|
||||
* needs a restart, and a quarantine already running keeps whatever cooldown was
|
||||
* live when it started.
|
||||
* <=0} → {@link #DEFAULT_QUARANTINE_COOLDOWN_SECONDS}. Since fleetd #466 this is
|
||||
* only the first occurrence's length — a credential quarantined again shortly
|
||||
* after this cooldown ends backs off further, up to a ceiling; see {@code
|
||||
* BackendQuarantine}'s class doc. Baked once into the {@code BackendQuarantine}
|
||||
* built at startup, so it is DEFERRED: changing it needs a restart, and a
|
||||
* quarantine already running keeps whatever cooldown was live when it started.
|
||||
* @param memberCredentials deny-by-default policy (CB-596) for which of the operator's own host
|
||||
* credentials a spawned member's pane inherits. {@code null} (the block
|
||||
* omitted) blocks nothing — see {@link MemberCredentials}.
|
||||
@@ -90,12 +100,48 @@ import java.util.Set;
|
||||
* fleetd's own process, so fleetd's own {@code $SHELL} says nothing about what
|
||||
* that pane runs. There is no channel to ask herdr for another user's shell, so
|
||||
* this must be told, never guessed. {@code null}/blank (or a value not ending
|
||||
* in {@code zsh}) is treated the same as "not zsh": the {@code
|
||||
* memberCredentials.policy: allow-list} ZDOTDIR scrub is skipped in favour of
|
||||
* the CB-596 sentinel overlay — a degraded control, never a refusal to spawn.
|
||||
* in {@code zsh}) is treated the same as "not zsh". fleetd #155: what that means
|
||||
* now depends on {@code memberCredentials.policy}. Under {@code deny-by-default}
|
||||
* it stays a degraded control, never a refusal to spawn — the pane-creation
|
||||
* overlay is unaffected by shell type, so the spawn proceeds with a WARN naming
|
||||
* the shell. Under {@code allow-list} the spawn is REFUSED instead: that policy's
|
||||
* whole point is a control a sourced file cannot undo, so silently falling back
|
||||
* to the weaker overlay would be the same "control silently does nothing" defect
|
||||
* #155 exists to remove — configure this field (or move the account to zsh, or
|
||||
* switch policy back to {@code deny-by-default}) to unblock the spawn.
|
||||
* When {@code memberHerdrSocket} is NOT configured this field is never
|
||||
* consulted at all; fleetd keeps reading its own {@code $SHELL}, exactly as
|
||||
* before this field existed.
|
||||
* @param memberSkills fleetd #362: nullable directory of skill folders (each a subdirectory
|
||||
* holding a {@code SKILL.md}, the same shape as this repo's own {@code
|
||||
* .claude/skills/}) copied into every provisioned worktree's {@code
|
||||
* .claude/skills/}, so a member spawned against ANY repo — not only one that
|
||||
* already ships its own copy — can load a bridge skill such as {@code
|
||||
* implementer}. {@code null}/blank ⇒ off: no worktree is touched beyond
|
||||
* today's behaviour. A skill folder the target repo already carries is never
|
||||
* overwritten — see {@link dev.ltms.fleet.session.GitWorktrees}. Claude Code
|
||||
* members only; an opencode member's equivalent lives under a different path
|
||||
* ({@code .opencode/agent}) and is not covered by this key. Every non-hidden
|
||||
* subdirectory of this directory is copied wholesale, with no per-file
|
||||
* allowlist — do not park scratch files or drafts alongside the real skill
|
||||
* folders, they will be copied into every provisioned worktree too.
|
||||
* @param idleSleepGuard opt-in-by-default: hold an OS-level assertion against idle sleep while at
|
||||
* least one member is live, so an unattended host does not sleep out from
|
||||
* under a member's long turn. {@code null} (the block omitted) behaves the
|
||||
* same as an explicit {@code enabled: true}; set {@code enabled: false} to
|
||||
* turn it off. See {@link dev.ltms.fleet.power.IdleSleepGuard}.
|
||||
* @param models central allow-list of models any {@code profiles:} entry may name. {@code
|
||||
* null} or an empty {@code allow:} ⇒ off: {@link #validateModels()} checks
|
||||
* nothing and every existing config keeps working exactly as it does today.
|
||||
* When non-empty, a profile whose {@code model:} is not one of {@link
|
||||
* Models#ids()} fails config load, naming both the model and the profile —
|
||||
* see {@link #validateModels()}. This block decides what may be CONFIGURED;
|
||||
* fleetd #422 added the separate on/off question — whether a configured model
|
||||
* may be spawned onto RIGHT NOW ({@link Models.ModelEntry#enabled}) — enforced
|
||||
* live at spawn by {@code CompositePeerLauncher}, not here. See {@link Models}.
|
||||
* @param leadRollover opt-in lead rollover (fleetd #480): {@code null} ⇒ off, and no
|
||||
* {@code dev.ltms.fleet.lead.LeadRollover} is constructed at all — an upgraded
|
||||
* daemon never clears a lead's pane on its own initiative. See {@link LeadRollover}.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record FleetConfig(
|
||||
@@ -120,7 +166,68 @@ public record FleetConfig(
|
||||
MemberCredentials memberCredentials,
|
||||
Coordinator coordinator,
|
||||
String worktreeGroup,
|
||||
String memberLoginShell) {
|
||||
String memberLoginShell,
|
||||
String memberSkills,
|
||||
IdleSleepGuard idleSleepGuard,
|
||||
Models models,
|
||||
LeadRollover leadRollover) {
|
||||
|
||||
/** Back-compat form before the {@code leadRollover:} block was added. */
|
||||
public FleetConfig(Bind bind, String herdrSocket, String memberHerdrSocket, Map<String, Profile> profiles,
|
||||
Guard guard, String worktreeRoot, Lifecycle lifecycle, Integer spawnReadyTimeoutMs,
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, Health health, String placement, Auth auth,
|
||||
ConfigReload configReload, Integer quarantineCooldownSeconds,
|
||||
MemberCredentials memberCredentials, Coordinator coordinator, String worktreeGroup,
|
||||
String memberLoginShell, String memberSkills, IdleSleepGuard idleSleepGuard,
|
||||
Models models) {
|
||||
this(bind, herdrSocket, memberHerdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, coordinator, worktreeGroup,
|
||||
memberLoginShell, memberSkills, idleSleepGuard, models, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the {@code models:} block was added. */
|
||||
public FleetConfig(Bind bind, String herdrSocket, String memberHerdrSocket, Map<String, Profile> profiles,
|
||||
Guard guard, String worktreeRoot, Lifecycle lifecycle, Integer spawnReadyTimeoutMs,
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, Health health, String placement, Auth auth,
|
||||
ConfigReload configReload, Integer quarantineCooldownSeconds,
|
||||
MemberCredentials memberCredentials, Coordinator coordinator, String worktreeGroup,
|
||||
String memberLoginShell, String memberSkills, IdleSleepGuard idleSleepGuard) {
|
||||
this(bind, herdrSocket, memberHerdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, coordinator, worktreeGroup,
|
||||
memberLoginShell, memberSkills, idleSleepGuard, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the {@code idleSleepGuard:} block was added. */
|
||||
public FleetConfig(Bind bind, String herdrSocket, String memberHerdrSocket, Map<String, Profile> profiles,
|
||||
Guard guard, String worktreeRoot, Lifecycle lifecycle, Integer spawnReadyTimeoutMs,
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, Health health, String placement, Auth auth,
|
||||
ConfigReload configReload, Integer quarantineCooldownSeconds,
|
||||
MemberCredentials memberCredentials, Coordinator coordinator, String worktreeGroup,
|
||||
String memberLoginShell, String memberSkills) {
|
||||
this(bind, herdrSocket, memberHerdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, coordinator, worktreeGroup,
|
||||
memberLoginShell, memberSkills, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the {@code memberSkills} key was added. */
|
||||
public FleetConfig(Bind bind, String herdrSocket, String memberHerdrSocket, Map<String, Profile> profiles,
|
||||
Guard guard, String worktreeRoot, Lifecycle lifecycle, Integer spawnReadyTimeoutMs,
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, Health health, String placement, Auth auth,
|
||||
ConfigReload configReload, Integer quarantineCooldownSeconds,
|
||||
MemberCredentials memberCredentials, Coordinator coordinator, String worktreeGroup,
|
||||
String memberLoginShell) {
|
||||
this(bind, herdrSocket, memberHerdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, coordinator, worktreeGroup,
|
||||
memberLoginShell, null, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the {@code memberLoginShell} key was added. */
|
||||
public FleetConfig(Bind bind, String herdrSocket, String memberHerdrSocket, Map<String, Profile> profiles,
|
||||
@@ -131,7 +238,7 @@ public record FleetConfig(
|
||||
MemberCredentials memberCredentials, Coordinator coordinator, String worktreeGroup) {
|
||||
this(bind, herdrSocket, memberHerdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, coordinator, worktreeGroup, null);
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, coordinator, worktreeGroup, null, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the {@code worktreeGroup} key was added. */
|
||||
@@ -240,7 +347,8 @@ public record FleetConfig(
|
||||
* @param configDir {@code CLAUDE_CONFIG_DIR} so the worker inherits the profile's
|
||||
* skills/MCP/hooks (may be {@code null})
|
||||
* @param tokenEnv name of the host env var holding the worker's auth token; its value
|
||||
* is injected as {@code ANTHROPIC_AUTH_TOKEN} (never stored in config)
|
||||
* is injected as {@code ANTHROPIC_AUTH_TOKEN} (never stored in config).
|
||||
* {@code null}/blank means this profile needs no token.
|
||||
* @param argv launch command; defaults to {@code ["claude"]}
|
||||
* @param placement where a worker lands: {@code "tab"} (default — its own tab in the
|
||||
* worker space) or {@code "pane"} (legacy — split the focused tab)
|
||||
@@ -256,7 +364,13 @@ public record FleetConfig(
|
||||
* @param cwd fixed working directory for this profile's workers (CB-112 "told otherwise");
|
||||
* {@code null}/blank → inherit the primary's cwd, else the daemon's
|
||||
* @param parityOverlay repo-relative paths copied primary→worktree for config parity; null/empty
|
||||
* defaults to a sensible set of local config files.
|
||||
* defaults to {@code [.env]} only (CB-148 point 2). {@code .env} is data — a
|
||||
* copy of it can only carry values. {@code .envrc} is executable shell that
|
||||
* {@code direnv} evaluates on every {@code cd}, so copying it moves behaviour
|
||||
* into the worker, not just values, and that is a different risk from copying
|
||||
* data. It is deliberately left out of the default for that reason. The knob
|
||||
* still supports it: an operator who wants it copied writes
|
||||
* {@code parityOverlay: [.env, .envrc]} explicitly and owns that choice.
|
||||
* <p><strong>Every overlay path must be gitignored or tracked-and-skipped.</strong>
|
||||
* CB-576 made {@code release()} preserve a worktree that {@code git status
|
||||
* --porcelain} reports as dirty, and it deliberately counts untracked files —
|
||||
@@ -266,11 +380,10 @@ public record FleetConfig(
|
||||
* worktrees then accumulate with no error anywhere.
|
||||
* <p>Checked on 2026-08-15 (CB-581): inert as configured. Tracked overlay
|
||||
* files carry {@code --skip-worktree} so {@code --porcelain} cannot see them,
|
||||
* {@code fleetd.yaml} is gitignored, and the default pair {@code .env} /
|
||||
* {@code .envrc} does not exist in this repo. Note the default applies to
|
||||
* <em>every</em> profile, so creating either file at the repo root is enough
|
||||
* to make it live. Add a new overlay path to {@code .gitignore} in the same
|
||||
* change that adds it here.
|
||||
* {@code fleetd.yaml} is gitignored, and {@code .env} does not exist in this
|
||||
* repo. Note the default applies to <em>every</em> profile, so creating
|
||||
* {@code .env} at the repo root is enough to make it live. Add a new overlay
|
||||
* path to {@code .gitignore} in the same change that adds it here.
|
||||
* <p>CB-578 stage C raised the stakes: a preserved dirty worktree is now also
|
||||
* committed to {@code refs/wip/<branch>} via {@code git add -A}. The overlay
|
||||
* carries the primary's own environment files into the worktree, so a
|
||||
@@ -306,8 +419,9 @@ public record FleetConfig(
|
||||
* statement that does not stop being true just because the profile was
|
||||
* named directly. A negative value has no sane meaning (there is no
|
||||
* "excluded" to degrade to below zero) and is refused at config load
|
||||
* instead, naming the profile and the key. Live means any session the
|
||||
* registry still owns (acquired and not yet released), in any state.
|
||||
* instead, naming the profile and the key. Live means a session that can
|
||||
* receive another delivery. The roster keeps terminal {@code BACKEND_ERROR}
|
||||
* and {@code FAILED} sessions for diagnostics, but they do not use capacity.
|
||||
* @param kind which peer launcher spawns this profile: {@code "claude-code"} (default —
|
||||
* the {@link dev.ltms.fleet.member.ClaudeCodeLauncher}) or {@code "opencode"}.
|
||||
* The {@code CompositePeerLauncher} routes {@code spawn}/reap by this value, so
|
||||
@@ -331,6 +445,12 @@ public record FleetConfig(
|
||||
* {@code null}/blank ⇒ the classification never fires for this profile and
|
||||
* today's completion-fallback behaviour is unchanged. Every backend words
|
||||
* its refusal differently, so this is config, never a vendor string in code.
|
||||
* Read live off the current config through {@code LiveExhaustedPatterns},
|
||||
* cached by profile name (fleetd #446), so it is HOT: an operator can arm
|
||||
* or disarm this profile's usage-limit detection by editing this key and
|
||||
* reloading, with no restart. Before fleetd #446 this was compiled once at
|
||||
* daemon startup, the same deferred shape its sibling {@code errorPattern}
|
||||
* (below) still has.
|
||||
* @param credentialId the shared account this profile authenticates as (CB-578 stage B). Two
|
||||
* or more profiles setting the <em>same</em> non-blank value are quarantined
|
||||
* together by one {@code exhaustedPattern} classification on any one of
|
||||
@@ -355,6 +475,16 @@ public record FleetConfig(
|
||||
* {@code limit.context} instead, which bounds the window opencode compacts
|
||||
* <em>within</em>, and only when {@code model} resolves to a
|
||||
* {@code provider/model} pair.
|
||||
* @param errorPattern regex matched against a completion-fallback scrape (fleetd #201 Unit 5)
|
||||
* to classify a turn that ended with no {@code fleet_reply} as a backend
|
||||
* error (credential outage, provider 5xx) rather than a real answer or an
|
||||
* exhausted usage limit. {@code null}/blank ⇒ this profile relies on
|
||||
* {@link dev.ltms.fleet.inject.CompletionResolver}'s built-in
|
||||
* {@code (?i)\bAPI Error\s*:} compatibility pattern instead — classification
|
||||
* still happens, just without a profile-specific match (reported separately
|
||||
* at startup as legacy-default coverage, never full coverage). Compiled once
|
||||
* at startup ({@code Fleetd.main}), like {@code exhaustedPattern}, so a
|
||||
* reload of this key is DEFERRED (see {@code ConfigRef}).
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record Profile(String profile, String baseUrl, String model,
|
||||
@@ -373,7 +503,8 @@ public record FleetConfig(
|
||||
String ideMcpUrl,
|
||||
String ideProjectDir,
|
||||
String ideOpenCommand,
|
||||
Integer autoCompactWindow) {
|
||||
Integer autoCompactWindow,
|
||||
String errorPattern) {
|
||||
|
||||
/** Peer kind spawned by {@link dev.ltms.fleet.member.ClaudeCodeLauncher} (the default). */
|
||||
public static final String KIND_CLAUDE_CODE = "claude-code";
|
||||
@@ -389,7 +520,7 @@ public record FleetConfig(
|
||||
? (KIND_CLAUDE_CODE.equals(k) ? List.of("claude") : List.of(k))
|
||||
: List.copyOf(argv);
|
||||
kind = k;
|
||||
tokenEnv = (tokenEnv == null || tokenEnv.isBlank()) ? "FLEETD_WORKER_TOKEN" : tokenEnv;
|
||||
tokenEnv = (tokenEnv == null || tokenEnv.isBlank()) ? null : tokenEnv;
|
||||
placement = (placement == null || placement.isBlank()) ? "tab" : placement.toLowerCase();
|
||||
workspace = (workspace == null || workspace.isBlank()) ? "fleet" : workspace;
|
||||
// CB-557: no per-profile default any more. A label is generated from the member's ROLE
|
||||
@@ -407,7 +538,7 @@ public record FleetConfig(
|
||||
// servers do not exist there to be enabled. This is defence in depth, not a live fix: the
|
||||
// worker stays isolated only because this separate mechanism already removes the servers.
|
||||
parityOverlay = (parityOverlay == null || parityOverlay.isEmpty())
|
||||
? List.of(".env", ".envrc")
|
||||
? List.of(".env")
|
||||
: List.copyOf(parityOverlay);
|
||||
// gitTokenEnv stays null when unset (opt-in). gitHostEnv defaults so operators enabling
|
||||
// checkpoints need only set gitTokenEnv; it is injected only alongside a resolved token.
|
||||
@@ -447,6 +578,32 @@ public record FleetConfig(
|
||||
// there is no blank-string form to normalize (it's an Integer), and the [100000, 1000000]
|
||||
// range is enforced eagerly at config load (rejectAutoCompactWindowOutOfRange), naming the
|
||||
// profile, rather than silently clamped here. A profile that never sets it keeps null.
|
||||
// errorPattern stays null when unset/blank (opt-in) — same rule as exhaustedPattern: no
|
||||
// defaulting, no vendor wording. A profile that never sets it relies on
|
||||
// CompletionResolver's built-in compatibility pattern instead (never "off").
|
||||
errorPattern = (errorPattern == null || errorPattern.isBlank()) ? null : errorPattern;
|
||||
}
|
||||
|
||||
/**
|
||||
* Backward-compatible constructor without the fleetd #201 {@code errorPattern} field — the
|
||||
* profile relies on {@code CompletionResolver}'s built-in compatibility pattern instead
|
||||
* (legacy-default coverage). This is the shape the canonical constructor had before the
|
||||
* field was added; every pre-existing Java call site (and any YAML that omits the key) keeps
|
||||
* compiling and behaving identically. Jackson binds the canonical (longest) constructor, so
|
||||
* YAML omitting {@code errorPattern:} still lands here as {@code null} via that path too.
|
||||
*/
|
||||
public Profile(String profile, String baseUrl, String model,
|
||||
String configDir, String tokenEnv, List<String> argv,
|
||||
String placement, String workspace, String tabLabel, String mcpUrl,
|
||||
String cwd, List<String> parityOverlay, String gitTokenEnv, String gitHostEnv,
|
||||
String kind, Map<String, String> env, Float weight, Integer maxLoad,
|
||||
Boolean subscription, String exhaustedPattern, String credentialId,
|
||||
String ideMcpUrl, String ideProjectDir, String ideOpenCommand,
|
||||
Integer autoCompactWindow) {
|
||||
this(profile, baseUrl, model, configDir, tokenEnv, argv, placement, workspace, tabLabel,
|
||||
mcpUrl, cwd, parityOverlay, gitTokenEnv, gitHostEnv, kind, env, weight, maxLoad,
|
||||
subscription, exhaustedPattern, credentialId, ideMcpUrl, ideProjectDir, ideOpenCommand,
|
||||
autoCompactWindow, null);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -491,7 +648,8 @@ public record FleetConfig(
|
||||
public Profile withProfile(String p) {
|
||||
return new Profile(p, baseUrl, model, configDir, tokenEnv, argv, placement, workspace, tabLabel,
|
||||
mcpUrl, cwd, parityOverlay, gitTokenEnv, gitHostEnv, kind, env, weight, maxLoad, subscription,
|
||||
exhaustedPattern, credentialId, ideMcpUrl, ideProjectDir, ideOpenCommand, autoCompactWindow);
|
||||
exhaustedPattern, credentialId, ideMcpUrl, ideProjectDir, ideOpenCommand, autoCompactWindow,
|
||||
errorPattern);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -589,14 +747,55 @@ public record FleetConfig(
|
||||
}
|
||||
|
||||
/**
|
||||
* The credential group this profile quarantines with (CB-578 stage B): the configured
|
||||
* {@link #credentialId} when set, else this profile's own name — so an unconfigured profile
|
||||
* quarantines alone, exactly as it did before this field existed. Two profiles that set the
|
||||
* same non-blank {@code credentialId} share one quarantine: a {@code BACKEND_EXHAUSTED}
|
||||
* classification on either one quarantines both.
|
||||
* True when this profile configures its own fleetd #201 Unit 5 backend-error classification
|
||||
* pattern. {@code false} means this profile relies on {@code CompletionResolver}'s built-in
|
||||
* {@code (?i)\bAPI Error\s*:} compatibility pattern instead (legacy-default coverage).
|
||||
*/
|
||||
public boolean hasErrorPattern() {
|
||||
return errorPattern != null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The shared credential id every {@code subscription: true} profile falls back to when it
|
||||
* sets no explicit {@link #credentialId} (fleetd #176 stage 2, correcting an inert first cut
|
||||
* of that ticket). A subscription profile has no credential of its own to fall back to its
|
||||
* name for: it authenticates as the operator's own Claude login, and a host has exactly one
|
||||
* of those, whatever names the operator gives the profiles running on it. Falling back to the
|
||||
* profile's own name (the way an ordinary off-subscription profile does) would keep two
|
||||
* subscription profiles on one login apart from each other, which is the opposite of what
|
||||
* "one account" means.
|
||||
*
|
||||
* <p>Measured live and what it broke: a lead on profile {@code opus}, members on profile
|
||||
* {@code sonnet}, same Claude login, neither setting {@code credentialId}. Before this
|
||||
* sentinel, {@code opus.effectiveCredentialId()} was {@code "opus"} and {@code sonnet
|
||||
* .effectiveCredentialId()} was {@code "sonnet"} — so fleetd #176's lead-seat matcher (and,
|
||||
* this sentinel now also fixes, {@code CompositePeerLauncher}'s quarantine/cool-off spawn
|
||||
* refusal and {@code BackendOutagePolicy}'s incident grouping) silently never linked them: the
|
||||
* fix shipped, and stayed inert on the one host it was written for.
|
||||
*/
|
||||
public static final String SUBSCRIPTION_CREDENTIAL_ID = "<subscription>";
|
||||
|
||||
/**
|
||||
* The credential group this profile quarantines with (CB-578 stage B; extended fleetd #176
|
||||
* stage 2 — see {@link #SUBSCRIPTION_CREDENTIAL_ID}): the configured {@link #credentialId}
|
||||
* when set — that always wins, so an operator with two separate Claude logins on one host can
|
||||
* still keep them apart. Otherwise, a {@code subscription: true} profile falls back to
|
||||
* {@link #SUBSCRIPTION_CREDENTIAL_ID} rather than its own name; an ordinary off-subscription
|
||||
* profile falls back to its own name, exactly as it did before this field existed, so an
|
||||
* unconfigured off-subscription profile still quarantines alone.
|
||||
*
|
||||
* <p>A {@code BACKEND_EXHAUSTED} (or repeated backend-error) classification on any profile
|
||||
* sharing the result quarantines/cools off every profile that shares it — including, now,
|
||||
* every {@code subscription: true} profile with no explicit {@code credentialId}. That is
|
||||
* intended, not incidental: one Claude subscription hitting a usage limit really does take out
|
||||
* every profile running on it, the same way {@code credentialId: openai-shared} already lets
|
||||
* two OpenAI-backed profiles share one quarantine.
|
||||
*/
|
||||
public String effectiveCredentialId() {
|
||||
return (credentialId == null || credentialId.isBlank()) ? profile : credentialId;
|
||||
if (credentialId != null && !credentialId.isBlank()) {
|
||||
return credentialId;
|
||||
}
|
||||
return isSubscription() ? SUBSCRIPTION_CREDENTIAL_ID : profile;
|
||||
}
|
||||
|
||||
/** True when this profile's workers are granted a forge token to open their own PR (CB-302). */
|
||||
@@ -604,6 +803,11 @@ public record FleetConfig(
|
||||
return gitTokenEnv != null && !gitTokenEnv.isBlank();
|
||||
}
|
||||
|
||||
/** True when this profile explicitly names a host token environment variable. */
|
||||
public boolean hasTokenEnv() {
|
||||
return tokenEnv != null && !tokenEnv.isBlank();
|
||||
}
|
||||
|
||||
/**
|
||||
* True when this profile's {@code env:} block names a worker-side Anthropic binding variable.
|
||||
* Those two keys are the adapter's, never the operator's: on a claude-code worker
|
||||
@@ -787,12 +991,18 @@ public record FleetConfig(
|
||||
* {@code null} ⇒ kept as {@code null} (no self id configured).
|
||||
* @param prefetch the consumer's {@code basicQos} prefetch count. {@code null}/non-positive ⇒
|
||||
* {@link LeadMailbox#DEFAULT_PREFETCH}.
|
||||
* @param peers fleetd #361: the coord-ids the operator declares as this daemon's peers — the
|
||||
* daemon never guesses who else exists. {@code fleet_list} reports each one's
|
||||
* live reachability. Blank entries are dropped; {@code null} ⇒ an empty list, so
|
||||
* a config written before this field existed still parses unchanged.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record Coordinator(String uri, String uriEnv, String selfId, Integer prefetch) {
|
||||
public record Coordinator(String uri, String uriEnv, String selfId, Integer prefetch, List<String> peers) {
|
||||
|
||||
public Coordinator {
|
||||
selfId = (selfId == null || selfId.isBlank()) ? null : selfId;
|
||||
peers = peers == null ? List.of()
|
||||
: peers.stream().filter(p -> p != null && !p.isBlank()).toList();
|
||||
}
|
||||
|
||||
/** True when a {@code uriEnv} is configured by name, whether or not its variable resolves. */
|
||||
@@ -886,7 +1096,15 @@ public record FleetConfig(
|
||||
* survives restarts of the agent inside it — so identity is now the tab label alone.
|
||||
*
|
||||
* @param profile the {@code profiles:} entry to launch this lead on when one must
|
||||
* be created; {@code null} ⇒ recognise-only, never create
|
||||
* be created; {@code null} ⇒ recognise-only, never create.
|
||||
* <p>fleetd #176: also the field {@code Fleetd.leadSeatLookup} reads
|
||||
* to learn which account this lead's own live session shares — set it
|
||||
* (safely, even on an already-running recognise-only lead: naming a
|
||||
* profile here never starts anything beyond {@code instances}) so a
|
||||
* {@code subscription: true} worker profile sharing its
|
||||
* {@code effectiveCredentialId()} has this lead's seat subtracted from
|
||||
* {@code fleet_list}'s {@code free}. {@code null} here also means this
|
||||
* lead's seat cannot be derived and is not counted.
|
||||
* @param tab the exact tab label hosting this lead, matched case-insensitively;
|
||||
* the only field identity depends on. Required — a lead with no
|
||||
* {@code tab} can never be discovered, launched or not
|
||||
@@ -1111,6 +1329,104 @@ public record FleetConfig(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Opt-in lead rollover (fleetd #480): a lead that decides it is ready to be replaced writes a
|
||||
* handover file, then asks fleetd to clear its own pane and bootstrap a fresh session against
|
||||
* that file. Config + a pure decision/verification layer only — see
|
||||
* {@code dev.ltms.fleet.lead.LeadRollover} for the executor this block feeds, and
|
||||
* {@code fleet_*} tool wiring is a later ticket.
|
||||
*
|
||||
* <p>Deliberately opt-in ({@code null} ⇒ off, exactly like {@code leadHeartbeat:}): absent this
|
||||
* block, {@code Fleetd.java} never constructs a {@code LeadRollover} object at all, so an
|
||||
* upgraded daemon cannot silently acquire the ability to clear the lead's own pane. Even once
|
||||
* present, nothing but an explicit {@code confirm()} call can ever roll a pane — there is no
|
||||
* timer, heartbeat or timeout anywhere in this feature that fires one on its own; see that
|
||||
* class's javadoc.
|
||||
*
|
||||
* <p><strong>Hot, not deferred</strong> (see {@code ConfigRef}'s class doc): every field below
|
||||
* is read live, through a {@code Supplier<LeadRollover>} the same {@code () -> config.get().x()}
|
||||
* shape {@code fleet}/{@code placement}/{@code models} already use, so an edit to any of the
|
||||
* six fields takes effect on the very next {@code open()}/{@code confirm()} call once the block
|
||||
* has been present since startup — nothing here is captured into a frozen field the way {@code
|
||||
* leadHeartbeat}'s {@code idleAfterNanos}/{@code backoffMs}/{@code quietNudgeCap} are. The one
|
||||
* restart-only edge left is structural, not a stale value: {@code Fleetd.java} decides whether
|
||||
* to construct the {@code LeadRollover} object at all off the startup snapshot, the same gate
|
||||
* {@code leadHeartbeat:} uses, so a block ADDED where it was absent at boot needs a restart
|
||||
* before anything exists to call — the same fact already true of adding a brand-new
|
||||
* {@code profiles:} entry.
|
||||
*
|
||||
* <p><strong>{@code turnSettleSeconds} (fleetd #480 correction):</strong> {@code confirm()} is
|
||||
* called FROM the calling lead's own turn, so its pane is still {@code WORKING} the instant
|
||||
* {@code confirm()} validates every gate and schedules the roll. {@code
|
||||
* dev.ltms.fleet.lead.LeadRollover}'s deferred continuation waits up to this many seconds for
|
||||
* that SAME pane to report an injectable state again — i.e. for the calling turn to actually
|
||||
* end — before it sends {@code /clear} at all. If that wait times out, no {@code /clear} is
|
||||
* ever sent: a lead that never goes idle is still doing real work, and clearing it would
|
||||
* destroy live context. This is a separate wait from {@code clearSettleSeconds} below, which
|
||||
* bounds the SECOND wait, for the pane to re-settle AFTER {@code /clear} has already gone out.
|
||||
*
|
||||
* @param handoverPath required when this block is present — where the handover file a fresh
|
||||
* lead session reads must live. There is no sane non-null default for an
|
||||
* operator-specific path, so a present block with a {@code null}/blank
|
||||
* {@code handoverPath} is refused at config load; see
|
||||
* {@link #validateLeadRollover()}. May be relative: {@code
|
||||
* dev.ltms.fleet.lead.LeadRollover#open} resolves a relative path against
|
||||
* the CALLING lead's configured {@code fleet.leaders.<name>.cwd}, falling
|
||||
* back to the daemon's own working directory ({@code
|
||||
* System.getProperty("user.dir")}) when that lead has none configured —
|
||||
* never against whatever directory the daemon process happens to have been
|
||||
* started in for its own sake. An absolute path is used unchanged.
|
||||
* @param requireOperatorConfirm default {@code true} — {@code confirm()} refuses unless the
|
||||
* caller also passes {@code operatorConfirmed: true}. Set {@code false} to
|
||||
* let the three handover-file checks alone gate the roll.
|
||||
* @param maxDocAgeSeconds default 3600 — refuse a handover file whose modified time is older
|
||||
* than this many seconds, so a stale leftover from an earlier rollover
|
||||
* attempt can never be mistaken for a fresh one.
|
||||
* @param turnSettleSeconds default 20 — bound on how long the deferred roll waits for the
|
||||
* CALLING lead's own turn to end (its pane to report injectable again)
|
||||
* before sending {@code /clear} at all. See the paragraph above.
|
||||
* @param clearSettleSeconds default 20 — bound on how long to wait for the lead's pane to
|
||||
* report an injectable state again after {@code /clear} before giving up. A
|
||||
* roll that times out here never sends {@code bootstrapText}.
|
||||
* @param bootstrapText default a sentence naming the RESOLVED handover path — sent to the
|
||||
* lead's pane once it settles after {@code /clear}, telling the fresh
|
||||
* session where to read the handover and carry on. Left {@code null} here
|
||||
* when the operator configures none: the default sentence cannot be built
|
||||
* at construction time because it must name the path AFTER {@code
|
||||
* dev.ltms.fleet.lead.LeadRollover#open} has resolved a relative {@code
|
||||
* handoverPath} against the calling lead's workspace, which this record has
|
||||
* no way to know — see {@link #bootstrapTextFor(String)}.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record LeadRollover(String handoverPath, Boolean requireOperatorConfirm,
|
||||
Integer maxDocAgeSeconds, Integer turnSettleSeconds,
|
||||
Integer clearSettleSeconds, String bootstrapText) {
|
||||
public LeadRollover {
|
||||
requireOperatorConfirm = requireOperatorConfirm == null || requireOperatorConfirm;
|
||||
maxDocAgeSeconds = (maxDocAgeSeconds == null || maxDocAgeSeconds <= 0) ? 3600 : maxDocAgeSeconds;
|
||||
turnSettleSeconds = (turnSettleSeconds == null || turnSettleSeconds <= 0) ? 20 : turnSettleSeconds;
|
||||
clearSettleSeconds = (clearSettleSeconds == null || clearSettleSeconds <= 0) ? 20 : clearSettleSeconds;
|
||||
bootstrapText = (bootstrapText == null || bootstrapText.isBlank()) ? null : bootstrapText;
|
||||
}
|
||||
|
||||
/**
|
||||
* The text actually sent to the lead's pane once it settles after {@code /clear}: the
|
||||
* operator's configured {@link #bootstrapText} when one is set, otherwise the default
|
||||
* sentence built from {@code resolvedHandoverPath}.
|
||||
*
|
||||
* @param resolvedHandoverPath the ABSOLUTE path {@code dev.ltms.fleet.lead.LeadRollover
|
||||
* #open} already resolved — never the raw configured {@link
|
||||
* #handoverPath}, which may still be relative and would name a
|
||||
* directory the fresh lead session's own pane cannot resolve
|
||||
*/
|
||||
public String bootstrapTextFor(String resolvedHandoverPath) {
|
||||
return bootstrapText != null
|
||||
? bootstrapText
|
||||
: "Fresh lead session: read the handover file at " + resolvedHandoverPath
|
||||
+ " and carry on from there.";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Watch {@code fleetd.yaml} and re-read it when it changes (CB-559).
|
||||
*
|
||||
@@ -1136,6 +1452,145 @@ public record FleetConfig(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Hold an OS-level assertion against idle sleep while at least one member is live (see
|
||||
* {@link dev.ltms.fleet.power.IdleSleepGuard}).
|
||||
*
|
||||
* <p>Unlike most opt-in blocks in this file, this one defaults to <em>on</em>: an unattended
|
||||
* host idle-sleeping mid-turn is a correctness problem (a dropped AMQP link, a frozen member),
|
||||
* not a convenience, so the safer default is armed. An operator who wants the previous
|
||||
* behaviour (no assertion held, ever) sets {@code enabled: false} explicitly.
|
||||
*
|
||||
* @param enabled {@code false} turns the guard off; {@code null} (the block omitted
|
||||
* entirely) or {@code true} leaves it on
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record IdleSleepGuard(Boolean enabled) {
|
||||
public boolean isEnabled() {
|
||||
return !Boolean.FALSE.equals(enabled);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Central allow-list of models any {@code profiles:} entry may name (fleetd ticket: "a central
|
||||
* allow-list of usable models"). Nothing before this block checked a profile's {@code model:}
|
||||
* against anything — it was a free-form string handed straight to the backend adapter, and a
|
||||
* withdrawn or misspelled name failed silently (opencode falls back to a default model rather
|
||||
* than erroring on an unknown {@code -m}) rather than at config load, where a mistake is cheap.
|
||||
*
|
||||
* <p><b>Absent or empty {@code allow:} is "off"</b>, on purpose: this is a large deployment
|
||||
* with gitignored {@code fleetd.yaml} on more than one host, and a change that forced every
|
||||
* operator to enumerate their models before the daemon would start would break every one of
|
||||
* them on upgrade. See {@link FleetConfig#validateModels()}, which is where the allow-list is
|
||||
* actually enforced, at config load.
|
||||
*
|
||||
* <p><b>The list is the authority; a {@code profiles:} entry is checked against it, never the
|
||||
* other way around.</b> Adding or editing a {@code profiles:} entry cannot, by itself, widen
|
||||
* the set of permitted models — only editing {@code models.allow:} itself can. This is the
|
||||
* invariant the ticket asked for: the two blocks are validated in one direction only.
|
||||
*
|
||||
* <p><b>Spawn-time enforcement (fleetd #422, units 2+3) lives outside this record</b> —
|
||||
* {@code CompositePeerLauncher.enforceModelEnabled} and its candidate-set filter read this
|
||||
* block LIVE (through the same kind of supplier {@code weight}/{@code maxLoad} already use),
|
||||
* so the on/off state below is hot: no restart needed. This block itself still only decides
|
||||
* what may be CONFIGURED (membership in {@link #allow}); {@link ModelEntry#enabled} decides
|
||||
* whether a member of that list is currently spawnable. The two questions are deliberately
|
||||
* separate — see {@link ModelEntry}'s javadoc for why turning a model off must never mean
|
||||
* removing it from {@link #allow}.
|
||||
*
|
||||
* @param allow the permitted models, each its own {@link ModelEntry} rather than a bare
|
||||
* string — see that record's javadoc for why. {@code null}/empty ⇒ the block is
|
||||
* treated as absent: {@link FleetConfig#validateModels()} checks nothing.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record Models(List<ModelEntry> allow) {
|
||||
|
||||
public Models {
|
||||
allow = (allow == null) ? List.of() : List.copyOf(allow);
|
||||
}
|
||||
|
||||
/**
|
||||
* One permitted model, named as a record rather than a bare string on purpose: this ticket
|
||||
* (fleetd #422) is the "later unit" the original comment here predicted — it hangs an on/off
|
||||
* state ({@link #enabled}) off each entry, and a bare {@code List<String>} could not have
|
||||
* grown that field without changing the YAML shape underneath every operator who already
|
||||
* wrote one. {@link #model()} is intentionally a single flat, opaque-string namespace — a
|
||||
* bare Claude id ({@code claude-sonnet-5}) and an opencode provider-prefixed id
|
||||
* ({@code openai/gpt-5.6-terra}) both fit it unchanged, because {@link
|
||||
* FleetConfig#validateModels()} only ever compares a profile's {@code model:} value against
|
||||
* this string for exact equality; it never parses a provider prefix or branches on a
|
||||
* profile's {@code kind:}.
|
||||
*
|
||||
* @param model the model id exactly as a {@code profiles:} entry's {@code model:} field
|
||||
* would name it
|
||||
* @param enabled {@code false} turns spawning onto this model off; {@code null} (the field
|
||||
* omitted — every config written before fleetd #422 is this shape) or
|
||||
* {@code true} leaves it on. Turning a model off must NEVER remove it from
|
||||
* {@link Models#allow} — {@link FleetConfig#validateModels()} checks
|
||||
* <em>membership</em> only, never the on/off state, so an off entry stays a
|
||||
* valid thing for a {@code profiles:} entry to name; only
|
||||
* {@code CompositePeerLauncher}'s spawn-time gate reads {@link #enabled}.
|
||||
* Collapsing the two — turning a model off by deleting its {@code allow:}
|
||||
* entry — would make {@link FleetConfig#validateModels()} refuse the whole
|
||||
* config reload the moment a still-configured profile names it, which is
|
||||
* exactly the restart-to-flip-a-switch problem this field exists to avoid.
|
||||
* <p>A model id can be named by more than one {@code profiles:} entry (e.g.
|
||||
* {@code deepseek-v4-flash} backs both {@code local} and {@code
|
||||
* local-direct} in the live config) — turning it off disables every profile
|
||||
* that names it, on purpose: the model is what a subscription's rate limit
|
||||
* actually constrains, not any one profile alias for it.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record ModelEntry(String model, Boolean enabled) {
|
||||
public ModelEntry {
|
||||
model = (model == null || model.isBlank()) ? null : model.trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* Back-compat form before {@link #enabled} was added (fleetd #422) — the model is
|
||||
* unconditionally on, exactly as every {@code ModelEntry} behaved before this field
|
||||
* existed. Keeps pre-#422 call sites (and any YAML that omits {@code enabled:})
|
||||
* compiling and behaving identically.
|
||||
*/
|
||||
public ModelEntry(String model) {
|
||||
this(model, null);
|
||||
}
|
||||
|
||||
/** {@code true} unless {@link #enabled} is explicitly {@code false} — absent means on. */
|
||||
public boolean isEnabled() {
|
||||
return !Boolean.FALSE.equals(enabled);
|
||||
}
|
||||
}
|
||||
|
||||
/** {@link #allow}'s model ids, as a set for membership checks. Blank/null entries are dropped. */
|
||||
public Set<String> ids() {
|
||||
Set<String> ids = new java.util.LinkedHashSet<>();
|
||||
for (ModelEntry e : allow) {
|
||||
if (e != null && e.model() != null) {
|
||||
ids.add(e.model());
|
||||
}
|
||||
}
|
||||
return Collections.unmodifiableSet(ids);
|
||||
}
|
||||
|
||||
/**
|
||||
* Model ids currently turned off (fleetd #422: {@link ModelEntry#isEnabled()} {@code
|
||||
* false}). Read live by {@code CompositePeerLauncher}'s spawn gate and by {@code
|
||||
* fleet_profiles}/{@code GET /profiles} — both must read this same accessor off the same
|
||||
* live config so the two surfaces cannot disagree about which model is off (the fleetd
|
||||
* #404 lesson: a status field must read the source the behaviour reads).
|
||||
*/
|
||||
public Set<String> offIds() {
|
||||
Set<String> off = new java.util.LinkedHashSet<>();
|
||||
for (ModelEntry e : allow) {
|
||||
if (e != null && e.model() != null && !e.isEnabled()) {
|
||||
off.add(e.model());
|
||||
}
|
||||
}
|
||||
return Collections.unmodifiableSet(off);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The terminal → lead-name map seeded from the legacy singular {@code primary:} pin (CB-530).
|
||||
*
|
||||
@@ -1251,19 +1706,19 @@ public record FleetConfig(
|
||||
* deny-list/deny-by-default every name here that is NOT also in {@link #allow} is
|
||||
* overlaid with a non-secret sentinel value. Under allow-list this list is
|
||||
* reporting only.
|
||||
* @param sshAuthSock whether the member may inherit {@code SSH_AUTH_SOCK} under the allow-list
|
||||
* policy ({@code "allow"}) or must have it blanked ({@code "block"}, the default).
|
||||
* @param sshAgentEnv whether the member may inherit {@code SSH_AUTH_SOCK} under the allow-list
|
||||
* policy ({@code "inherit"}) or must have it omitted ({@code "omit"}, the default).
|
||||
* This is a DECISION, never a default: {@code SSH_AUTH_SOCK} is a handle to the
|
||||
* operator's ssh-agent, and a member holding it can sign with the operator's own
|
||||
* keys — but it appears in no secret file and is credential-shaped like nothing on
|
||||
* any list, which is why three earlier tickets missed it (gitea #110 / CB-607).
|
||||
* Blocking it breaks git over SSH inside the member; allow it only when members do
|
||||
* Omitting it does not prevent git over SSH inside the member; inherit it only when members do
|
||||
* not need to authenticate as the operator over SSH. Ignored under deny-list /
|
||||
* deny-by-default, which never touch the name.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record MemberCredentials(String policy, List<String> allow, List<String> known,
|
||||
String sshAuthSock) {
|
||||
String sshAgentEnv) {
|
||||
|
||||
/** Default policy: block every {@code known} name not in {@code allow}, via the env overlay. */
|
||||
public static final String POLICY_DENY_BY_DEFAULT = "deny-by-default";
|
||||
@@ -1280,11 +1735,25 @@ public record FleetConfig(
|
||||
*/
|
||||
public static final String POLICY_ALLOW_LIST = "allow-list";
|
||||
|
||||
/** The pre-CB-633 three-field form — {@code sshAuthSock} defaults to blocked. */
|
||||
/** The pre-CB-633 three-field form — {@code sshAgentEnv} defaults to omitted. */
|
||||
public MemberCredentials(String policy, List<String> allow, List<String> known) {
|
||||
this(policy, allow, known, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Reads both the current {@code sshAgentEnv} key and the compatible {@code sshAuthSock} key.
|
||||
* When both keys are present, {@code sshAgentEnv} wins, even if its value is unrecognised.
|
||||
*/
|
||||
@JsonCreator
|
||||
public static MemberCredentials fromYaml(@JsonProperty("policy") String policy,
|
||||
@JsonProperty("allow") List<String> allow,
|
||||
@JsonProperty("known") List<String> known,
|
||||
@JsonProperty("sshAgentEnv") String sshAgentEnv,
|
||||
@JsonProperty("sshAuthSock") String sshAuthSock) {
|
||||
return new MemberCredentials(policy, allow, known,
|
||||
sshAgentEnv != null ? sshAgentEnv : sshAuthSock);
|
||||
}
|
||||
|
||||
public MemberCredentials {
|
||||
String normalizedPolicy = (policy == null || policy.isBlank())
|
||||
? POLICY_DENY_BY_DEFAULT : policy.toLowerCase(java.util.Locale.ROOT);
|
||||
@@ -1293,8 +1762,9 @@ public record FleetConfig(
|
||||
policy = POLICY_DENY_LIST.equals(normalizedPolicy) ? POLICY_DENY_BY_DEFAULT : normalizedPolicy;
|
||||
allow = allow == null ? List.of() : List.copyOf(allow);
|
||||
known = known == null ? List.of() : List.copyOf(known);
|
||||
sshAuthSock = (sshAuthSock != null && "allow".equalsIgnoreCase(sshAuthSock.trim()))
|
||||
? "allow" : "block";
|
||||
sshAgentEnv = (sshAgentEnv != null
|
||||
&& ("inherit".equalsIgnoreCase(sshAgentEnv.trim())
|
||||
|| "allow".equalsIgnoreCase(sshAgentEnv.trim()))) ? "inherit" : "omit";
|
||||
}
|
||||
|
||||
/** True when this block selects the CB-633 derived-allow-list policy. */
|
||||
@@ -1303,8 +1773,8 @@ public record FleetConfig(
|
||||
}
|
||||
|
||||
/** True when {@code SSH_AUTH_SOCK} may pass through under the allow-list policy. Default: no. */
|
||||
public boolean sshAuthSockAllowed() {
|
||||
return "allow".equals(sshAuthSock);
|
||||
public boolean sshAgentEnvInherited() {
|
||||
return "inherit".equals(sshAgentEnv);
|
||||
}
|
||||
|
||||
/** {@link #allow} as a set, for membership checks. */
|
||||
@@ -1371,7 +1841,8 @@ public record FleetConfig(
|
||||
"bind", "herdrSocket", "memberHerdrSocket", "profiles", "guard", "worktreeRoot",
|
||||
"lifecycle", "spawnReadyTimeoutMs", "spawnReadyPollMs", "broker", "primary", "fleet",
|
||||
"leadHeartbeat", "health", "placement", "auth", "configReload", "quarantineCooldownSeconds",
|
||||
"memberCredentials", "coordinator", "worktreeGroup", "memberLoginShell");
|
||||
"memberCredentials", "coordinator", "worktreeGroup", "memberLoginShell", "memberSkills",
|
||||
"idleSleepGuard", "models", "leadRollover");
|
||||
|
||||
/** Load and validate config from {@code path}. */
|
||||
public static FleetConfig load(Path path) {
|
||||
@@ -1383,6 +1854,7 @@ public record FleetConfig(
|
||||
rejectDuplicateMemberSlots(yaml);
|
||||
rejectNegativeMaxLoad(yaml);
|
||||
rejectAutoCompactWindowOutOfRange(yaml);
|
||||
rejectMalformedProfilePatterns(yaml);
|
||||
rejectUnknownKind(yaml);
|
||||
rejectUnknownAuthMode(yaml);
|
||||
rejectUnknownPlacement(yaml);
|
||||
@@ -1731,6 +2203,59 @@ public record FleetConfig(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Reject a profile whose {@code errorPattern} (fleetd #201 Unit 5) or {@code exhaustedPattern}
|
||||
* (CB-578 stage A) is not a valid Java regex, naming the profile, the key, and the parser's own
|
||||
* message.
|
||||
*
|
||||
* <p>Unset/{@code null} means, for {@code errorPattern}, "use {@code CompletionResolver}'s
|
||||
* built-in {@code (?i)\bAPI Error\s*:} compatibility pattern", and for {@code exhaustedPattern},
|
||||
* "opt out of that classification" — either way it passes silently. A profile that DOES set
|
||||
* either key gets it compiled once at daemon startup ({@code Fleetd.main}) — an uncaught
|
||||
* {@link java.util.regex.PatternSyntaxException} there crashes startup without naming which
|
||||
* profile or key is at fault (fleetd #273: this happened for {@code exhaustedPattern}, which had
|
||||
* no validator here even though its sibling {@code errorPattern} did). Validate eagerly here
|
||||
* instead, at config load, the same "fail loud at load, not lazily later" reasoning as
|
||||
* {@link #rejectAutoCompactWindowOutOfRange}. Both keys are checked from a single load, and any
|
||||
* failures from either are collected together into one message.
|
||||
*
|
||||
* @param yaml the raw config text
|
||||
* @throws IllegalStateException when any profile's {@code errorPattern} or
|
||||
* {@code exhaustedPattern} fails to compile
|
||||
*/
|
||||
static void rejectMalformedProfilePatterns(String yaml) {
|
||||
Map<?, ?> raw;
|
||||
try {
|
||||
raw = YAML.readValue(yaml, Map.class);
|
||||
} catch (IOException | IllegalArgumentException e) {
|
||||
return; // a malformed file is reported by the real parse, not here
|
||||
}
|
||||
if (raw == null || !(raw.get("profiles") instanceof Map<?, ?> profiles)) {
|
||||
return;
|
||||
}
|
||||
List<String> bad = new ArrayList<>();
|
||||
for (Map.Entry<?, ?> e : profiles.entrySet()) {
|
||||
if (!(e.getValue() instanceof Map<?, ?> p)) {
|
||||
continue;
|
||||
}
|
||||
for (String key : List.of("errorPattern", "exhaustedPattern")) {
|
||||
if (!(p.get(key) instanceof String pattern) || pattern.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
try {
|
||||
Pattern.compile(pattern);
|
||||
} catch (PatternSyntaxException ex) {
|
||||
bad.add("profiles." + e.getKey() + "." + key + " (\"" + pattern + "\"): " + ex.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
bad.sort(String::compareTo);
|
||||
if (!bad.isEmpty()) {
|
||||
throw new IllegalStateException("refusing to start: malformed pattern — "
|
||||
+ String.join("; ", bad));
|
||||
}
|
||||
}
|
||||
|
||||
/** The peer kinds this build has an adapter for — {@link Profile#kind()}'s only valid values. */
|
||||
private static final Set<String> KNOWN_KINDS = Set.of(Profile.KIND_CLAUDE_CODE, Profile.KIND_OPENCODE);
|
||||
|
||||
@@ -1994,9 +2519,27 @@ public record FleetConfig(
|
||||
// memberLoginShell is left as-is (fleetd #213), like worktreeGroup: null/blank is "not
|
||||
// configured", and there is no sane non-null default — a member's login shell is
|
||||
// operator-specific and only meaningful when memberHerdrSocket is also set.
|
||||
// memberSkills is left as-is (fleetd #362), like worktreeGroup/memberLoginShell: null/blank
|
||||
// is "off", and there is no sane non-null default — the daemon may not even run from a
|
||||
// checkout that ships its own .claude/skills/.
|
||||
// idleSleepGuard is left as-is, like leadHeartbeat/configReload above, but for the opposite
|
||||
// reason: it is on by default already (its own isEnabled() treats null the same as
|
||||
// enabled: true — see its javadoc), so defaulting the block here would change nothing a
|
||||
// reader observes and would only obscure that "block omitted" and "block present and
|
||||
// enabled" are deliberately the same outcome.
|
||||
// models is left as-is, like broker/primary/coordinator above: null/empty is "off", and an
|
||||
// absent block must validate nothing (see Models's javadoc) — defaulting it here to an
|
||||
// empty Models would be a no-op for validateModels() either way, since an empty allow-list
|
||||
// already means "check nothing", so there is nothing to gain and one more null check to
|
||||
// avoid by leaving it exactly as configured.
|
||||
// leadRollover is left as-is, like leadHeartbeat above: null is "off", and LeadRollover's
|
||||
// own compact constructor defaults the fields of a block that IS present. Defaulting it
|
||||
// here would construct a LeadRollover object (via Fleetd.java's presence gate) for every
|
||||
// config that never mentioned it.
|
||||
return new FleetConfig(b, herdrSocket, memberHerdrSocket, profiles, g, worktreeRoot, l, timeout, pollMs,
|
||||
broker, primary, f, leadHeartbeat, health, placementOrDefault, a, configReload,
|
||||
quarantineCooldown, mc, coordinator, worktreeGroup, memberLoginShell);
|
||||
quarantineCooldown, mc, coordinator, worktreeGroup, memberLoginShell, memberSkills,
|
||||
idleSleepGuard, models, leadRollover);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -2079,6 +2622,26 @@ public record FleetConfig(
|
||||
+ "lead tabs cannot be confused.");
|
||||
}
|
||||
|
||||
/**
|
||||
* Reject a present {@code leadRollover:} block with no (or a blank) {@code handoverPath}
|
||||
* (fleetd #480). There is no sane non-null default for an operator-specific file path, unlike
|
||||
* every other field on {@link LeadRollover}, which {@link LeadRollover}'s own compact
|
||||
* constructor already defaults — so this is the one field that must be refused at load rather
|
||||
* than silently defaulted to something that would never match a real handover file.
|
||||
*
|
||||
* @throws IllegalStateException when {@code leadRollover} is present but {@code handoverPath}
|
||||
* is {@code null} or blank
|
||||
*/
|
||||
public void validateLeadRollover() {
|
||||
if (leadRollover == null) {
|
||||
return;
|
||||
}
|
||||
if (leadRollover.handoverPath() == null || leadRollover.handoverPath().isBlank()) {
|
||||
throw new IllegalStateException("refusing to start: leadRollover.handoverPath is "
|
||||
+ "required when leadRollover: is present.");
|
||||
}
|
||||
}
|
||||
|
||||
/** Case-insensitive prefix test that tolerates a null/blank label. */
|
||||
private static boolean startsWithIgnoreCase(String label, String prefix) {
|
||||
if (label == null || prefix == null || prefix.isBlank()) {
|
||||
@@ -2216,6 +2779,118 @@ public record FleetConfig(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Reject a {@code profiles:} entry whose {@code model:} is not on the configured {@link
|
||||
* #models} allow-list.
|
||||
*
|
||||
* <p>Absent or empty {@code models.allow:} validates nothing — see {@link Models}'s javadoc:
|
||||
* an existing config with no such block must keep working exactly as it does today. Once the
|
||||
* operator declares at least one entry, every profile's {@code model:} (when set — a profile
|
||||
* may legitimately leave it {@code null}, e.g. a {@code subscription: true} profile relying on
|
||||
* the account's own default) must equal one of {@link Models#ids()} exactly. The comparison is
|
||||
* a flat string match: a bare Claude id and an opencode {@code provider/model} id are both
|
||||
* just opaque strings here, so nothing here needs to know which {@code kind:} a profile runs.
|
||||
*
|
||||
* <p>The check runs one direction only, by construction: it reads {@link #profiles} and
|
||||
* {@link #models}, and only ever adds to {@code bad} when a profile's model is missing from
|
||||
* the allow-list. Nothing here can be satisfied by widening a {@code profiles:} entry — only
|
||||
* editing {@code models.allow:} itself changes what passes. That is the invariant the ticket
|
||||
* asked for: the list is the authority, profiles are checked against it.
|
||||
*
|
||||
* @throws IllegalStateException when any profile names a model outside the configured
|
||||
* allow-list, naming both the model and the profile that wanted it
|
||||
*/
|
||||
public void validateModels() {
|
||||
if (models == null || models.allow().isEmpty()) {
|
||||
return;
|
||||
}
|
||||
Set<String> allowed = models.ids();
|
||||
List<String> bad = new ArrayList<>();
|
||||
profiles.forEach((name, p) -> {
|
||||
String model = p.model();
|
||||
if (model != null && !model.isBlank() && !allowed.contains(model)) {
|
||||
bad.add("profile '" + name + "' names model '" + model + "', which is not in "
|
||||
+ "models.allow: (have: " + allowed + ").");
|
||||
}
|
||||
});
|
||||
if (!bad.isEmpty()) {
|
||||
throw new IllegalStateException("refusing to start: " + String.join(" ", bad));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Runs every validator this class declares — found by reflection, not by name.
|
||||
*
|
||||
* <p>fleetd ticket "central allow-list of usable models", follow-up: mutation testing found
|
||||
* that although each of the six validators above was well pinned on its own, nothing proved
|
||||
* either real caller ({@code Fleetd.main} and {@link ConfigRef#reload()}) still
|
||||
* invoked it — deleting a call site left the full suite green. The fix is not a seventh test
|
||||
* per caller; a hand-maintained list of six names here would have the exact same defect its
|
||||
* own javadoc would warn against: the seventh validator someone adds next month has no reason
|
||||
* to be added to it. So this method does not name any validator. It sweeps {@link
|
||||
* #getClass()}'s own public, no-argument, {@code void} methods whose name starts with {@code
|
||||
* "validate"} (excluding itself) and invokes every one it finds, via {@link
|
||||
* #invokeAllValidators}. A new {@code validateXxx()} method is therefore wired into both
|
||||
* callers the moment it is written — there is no second step to forget, and so no state in
|
||||
* which it silently never runs.
|
||||
*
|
||||
* <p>{@code Fleetd.main} and {@link ConfigRef#reload()} each call this one method instead of
|
||||
* the six individually — see the comments at those two call sites for why
|
||||
* each must run it.
|
||||
*
|
||||
* <p>Methods run in a fixed (alphabetical) order, so a config with more than one violation
|
||||
* always names the same one first, on every run.
|
||||
*
|
||||
* @throws IllegalStateException (or whatever unchecked exception a validator itself throws),
|
||||
* propagated unchanged from the first validator, in that order,
|
||||
* that finds a problem
|
||||
*/
|
||||
public void validateAll() {
|
||||
invokeAllValidators(this);
|
||||
}
|
||||
|
||||
/**
|
||||
* The reflective sweep behind {@link #validateAll()}, kept as its own method — taking any
|
||||
* {@code target}, not just {@code this} — so a test can prove the MECHANISM is generic (it
|
||||
* would sweep a seventh {@code validateXxx()} method added to any class, not just something
|
||||
* special-cased to today's six on {@link FleetConfig}) without needing to add a real, unwanted
|
||||
* seventh validator to this class just to exercise that claim. See {@code
|
||||
* FleetConfigValidateAllTest} for that proof.
|
||||
*
|
||||
* @param target an object whose public, no-argument, {@code void} methods named {@code
|
||||
* validateXxx} (any name starting with {@code "validate"}, excluding {@code
|
||||
* validateAll} itself) should all run, in alphabetical-by-name order
|
||||
*/
|
||||
static void invokeAllValidators(Object target) {
|
||||
List<Method> methods = new ArrayList<>();
|
||||
for (Method m : target.getClass().getMethods()) {
|
||||
if (Modifier.isPublic(m.getModifiers())
|
||||
&& m.getParameterCount() == 0
|
||||
&& m.getReturnType() == void.class
|
||||
&& m.getName().startsWith("validate")
|
||||
&& !m.getName().equals("validateAll")) {
|
||||
methods.add(m);
|
||||
}
|
||||
}
|
||||
methods.sort(Comparator.comparing(Method::getName));
|
||||
for (Method m : methods) {
|
||||
try {
|
||||
m.invoke(target);
|
||||
} catch (InvocationTargetException e) {
|
||||
Throwable cause = e.getCause();
|
||||
if (cause instanceof RuntimeException re) {
|
||||
throw re;
|
||||
}
|
||||
if (cause instanceof Error err) {
|
||||
throw err;
|
||||
}
|
||||
throw new IllegalStateException("validator " + m.getName() + " failed", cause);
|
||||
} catch (IllegalAccessException e) {
|
||||
throw new IllegalStateException("cannot invoke validator " + m.getName(), e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** True for the loopback addresses and the unspecified-but-local forms we treat as same-host. */
|
||||
private static boolean isLoopbackBind(String host) {
|
||||
if (host == null || host.isBlank()) {
|
||||
|
||||
@@ -14,6 +14,7 @@ import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiConsumer;
|
||||
@@ -28,17 +29,65 @@ public final class FleetHealthMonitor {
|
||||
static final int MAX_FAIL_TARGET_ATTEMPTS = 3;
|
||||
// CB-641: Match the injector's 60s readiness gate so health allows a full first boot.
|
||||
static final long READINESS_GRACE_NANOS = TimeUnit.SECONDS.toNanos(60);
|
||||
/**
|
||||
* fleetd #280: how long after a terminal transition to wait before the one bounded re-check
|
||||
* fires. Must exceed the worst-case reverse-rendezvous {@code fleet_ask} window (55-115s, see
|
||||
* {@code FleetMcp.ASK_DEFAULT_TIMEOUT_MS} / {@code FleetApp.MAX_ASK_TIMEOUT_MS}) so that, if the
|
||||
* target was genuinely {@code ASKING} when {@code state} was first observed, its own ask has had
|
||||
* time to lapse (clearing {@code Task#question} back to {@code null}) before this fires.
|
||||
*/
|
||||
static final long ASK_LAPSE_RECHECK_DELAY_SECONDS = 120;
|
||||
|
||||
/**
|
||||
* fleetd #386: {@code System.nanoTime()} (or whatever {@link #clock} is) does not advance while
|
||||
* macOS sleeps, so a raw {@code nowNanos - lastActivityAtNanos} comparison freezes with the
|
||||
* host and can never cross {@link #workingSuspectAfterNanos}. This is a second, wall-clock
|
||||
* source used ONLY inside the stall check ({@link #stallElapsedNanos}) to detect and correct
|
||||
* for that freeze. Nothing else in this class reads it — every other decision (readiness grace,
|
||||
* the fault classification itself) stays exactly on {@link #clock}, as the ticket requires.
|
||||
*/
|
||||
private static final LongSupplier DEFAULT_REALTIME_CLOCK =
|
||||
() -> TimeUnit.MILLISECONDS.toNanos(System.currentTimeMillis());
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final MessageService messages;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final LongSupplier realtimeClock;
|
||||
private final long intervalSeconds;
|
||||
private final long tickIntervalNanos;
|
||||
private final long workingSuspectAfterNanos;
|
||||
private final BiConsumer<String, String> failTarget;
|
||||
private final Map<String, HealthPrior> priors = new HashMap<>();
|
||||
private final Map<String, HealthState> states = new HashMap<>();
|
||||
/**
|
||||
* fleetd #386 clock-drift bookkeeping. {@code haveClockBaseline}/{@code lastTickMonoNanos}/
|
||||
* {@code lastTickRealNanos} track the previous tick's pair of readings so each new tick can
|
||||
* measure how far the two clocks moved apart since then. {@code accumulatedDriftNanos} is the
|
||||
* running total of every such divergence observed since this monitor started (never decreases —
|
||||
* the monotonic clock can only lag real time, never lead it). {@code busyDriftBaselineNanos}/
|
||||
* {@code busyBaselineActivityNanos} record, per target, the value of {@code accumulatedDriftNanos}
|
||||
* at the moment this monitor first saw that target's CURRENT {@code lastActivityAtNanos} while
|
||||
* BUSY — so {@link #stallElapsedNanos} adds back only the drift observed DURING this BUSY span,
|
||||
* never drift from a sleep that happened before the member went busy. All five fields are touched
|
||||
* only from {@code tick()}, like {@link #priors}.
|
||||
*/
|
||||
private boolean haveClockBaseline = false;
|
||||
private long lastTickMonoNanos;
|
||||
private long lastTickRealNanos;
|
||||
private long accumulatedDriftNanos = 0;
|
||||
private final Map<String, Long> busyDriftBaselineNanos = new HashMap<>();
|
||||
private final Map<String, Long> busyBaselineActivityNanos = new HashMap<>();
|
||||
/**
|
||||
* The live classification per member, and the only one of this class's three maps that more
|
||||
* than one scheduler task touches. {@code tick} writes it (and prunes it to the roster);
|
||||
* fleetd #280's delayed {@link #recheckTerminalTarget} reads it from its own separate scheduled
|
||||
* task. Both run on the single-threaded scheduler {@code Fleetd} passes in today, so they are
|
||||
* serialised — but nothing in this class enforces that, and an unsynchronised {@link HashMap}
|
||||
* read racing a resize can spin a CPU forever rather than fail visibly. {@code priors} and
|
||||
* {@code orphanStreaks} stay plain maps because {@code tick} is still their only toucher.
|
||||
*/
|
||||
private final Map<String, HealthState> states = new ConcurrentHashMap<>();
|
||||
/**
|
||||
* CB-643: consecutive ticks on which a target looked like an orphaned delegation. The fact
|
||||
* {@link MessageService#hasOrphanedDelegation} reports is a true snapshot, but it can read true
|
||||
@@ -70,12 +119,29 @@ public final class FleetHealthMonitor {
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, long intervalSeconds,
|
||||
long workingSuspectAfterSeconds, BiConsumer<String, String> failTarget) {
|
||||
this(agents, roster, messages, scheduler, clock, DEFAULT_REALTIME_CLOCK, intervalSeconds,
|
||||
workingSuspectAfterSeconds, failTarget);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param realtimeClock fleetd #386: a wall-clock nanosecond source (e.g.
|
||||
* {@code System.currentTimeMillis()} converted to nanos) that keeps
|
||||
* advancing while {@code clock} is frozen by a host sleep. Used only to
|
||||
* correct the stall check — see the class-level javadoc on the
|
||||
* clock-drift fields.
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, LongSupplier realtimeClock,
|
||||
long intervalSeconds, long workingSuspectAfterSeconds,
|
||||
BiConsumer<String, String> failTarget) {
|
||||
this.agents = agents;
|
||||
this.roster = roster;
|
||||
this.messages = messages;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.realtimeClock = Objects.requireNonNull(realtimeClock, "realtimeClock");
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.tickIntervalNanos = TimeUnit.SECONDS.toNanos(intervalSeconds);
|
||||
this.workingSuspectAfterNanos = TimeUnit.SECONDS.toNanos(workingSuspectAfterSeconds);
|
||||
this.failTarget = Objects.requireNonNull(failTarget, "failTarget");
|
||||
}
|
||||
@@ -105,6 +171,7 @@ public final class FleetHealthMonitor {
|
||||
for (Agent agent : agentsNow) live.put(agent.terminalId(), agent);
|
||||
HashSet<String> current = new HashSet<>();
|
||||
long nowNanos = clock.getAsLong();
|
||||
long driftBeforeThisTick = observeClockDrift(nowNanos);
|
||||
for (MemberSession session : rosterNow) {
|
||||
current.add(session.terminalId());
|
||||
Agent agent = live.get(session.terminalId());
|
||||
@@ -115,7 +182,7 @@ public final class FleetHealthMonitor {
|
||||
&& session.state() != MemberSession.State.SPAWNING;
|
||||
boolean readinessGraceElapsed = nowNanos - session.spawnedAtNanos() >= READINESS_GRACE_NANOS;
|
||||
boolean stalled = session.state() == MemberSession.State.BUSY
|
||||
&& nowNanos - session.lastActivityAtNanos() >= workingSuspectAfterNanos;
|
||||
&& stallElapsedNanos(session, nowNanos, driftBeforeThisTick) >= workingSuspectAfterNanos;
|
||||
// CB-643: the three message-layer facts CB-640 published. Read them here rather than
|
||||
// leaving them false — that constant is what made 8 of the 9 fault states dead.
|
||||
boolean queuedDelivery = messages.hasQueuedDelivery(session.terminalId());
|
||||
@@ -133,6 +200,8 @@ public final class FleetHealthMonitor {
|
||||
priors.keySet().retainAll(current);
|
||||
states.keySet().retainAll(current);
|
||||
orphanStreaks.keySet().retainAll(current);
|
||||
busyDriftBaselineNanos.keySet().retainAll(current);
|
||||
busyBaselineActivityNanos.keySet().retainAll(current);
|
||||
} catch (Throwable error) {
|
||||
// Any unclassified collection failure must never kill the monitor's only scheduler task.
|
||||
log.warn("fleet health collection failed; will retry next tick", error);
|
||||
@@ -157,6 +226,65 @@ public final class FleetHealthMonitor {
|
||||
return streak >= ORPHAN_CONFIRM_TICKS;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #386: compare this tick's monotonic and real-time readings against the previous
|
||||
* tick's, and fold any positive divergence into {@link #accumulatedDriftNanos} (a ratchet — it
|
||||
* never decreases, since the monotonic clock can only fall behind real time, never ahead of
|
||||
* it). Logs once, at WARN, when that single tick's divergence exceeds one full tick interval —
|
||||
* the signature of a host that slept between the two ticks (a tick literally cannot run while
|
||||
* the process itself is suspended, so the whole sleep duration lands inside one tick's gap).
|
||||
*
|
||||
* @return {@link #accumulatedDriftNanos} as it stood BEFORE this tick's divergence was folded
|
||||
* in — the baseline {@link #stallElapsedNanos} needs when a target is observed BUSY
|
||||
* for the first time this tick, so a sleep that happened before this member went busy
|
||||
* is not attributed to it.
|
||||
*/
|
||||
private long observeClockDrift(long nowNanos) {
|
||||
long nowRealNanos = realtimeClock.getAsLong();
|
||||
long driftBeforeThisTick = accumulatedDriftNanos;
|
||||
if (haveClockBaseline) {
|
||||
long monoDelta = nowNanos - lastTickMonoNanos;
|
||||
long realDelta = nowRealNanos - lastTickRealNanos;
|
||||
long tickDrift = realDelta - monoDelta;
|
||||
if (tickDrift > tickIntervalNanos) {
|
||||
log.warn("fleet health: the monotonic clock did not advance for about {}s that the "
|
||||
+ "real clock did since the last tick (host likely slept); the stall "
|
||||
+ "detector could not see that time", TimeUnit.NANOSECONDS.toSeconds(tickDrift));
|
||||
}
|
||||
if (tickDrift > 0) {
|
||||
accumulatedDriftNanos = driftBeforeThisTick + tickDrift;
|
||||
}
|
||||
}
|
||||
lastTickMonoNanos = nowNanos;
|
||||
lastTickRealNanos = nowRealNanos;
|
||||
haveClockBaseline = true;
|
||||
return driftBeforeThisTick;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #386: {@code nowNanos - lastActivityAtNanos} alone freezes across a host sleep, since
|
||||
* both come from the monotonic {@link #clock}. This adds back the real-time drift observed
|
||||
* since this BUSY span started — not the monitor's whole lifetime, so a sleep that happened
|
||||
* before this member went busy never leaks into its stall reading (see the class-level javadoc
|
||||
* on the drift fields). The baseline resets whenever {@code lastActivityAtNanos} changes (a new
|
||||
* turn) or the member is not currently BUSY.
|
||||
*/
|
||||
private long stallElapsedNanos(MemberSession session, long nowNanos, long driftBeforeThisTick) {
|
||||
String target = session.terminalId();
|
||||
if (session.state() != MemberSession.State.BUSY) {
|
||||
busyDriftBaselineNanos.remove(target);
|
||||
busyBaselineActivityNanos.remove(target);
|
||||
return nowNanos - session.lastActivityAtNanos();
|
||||
}
|
||||
Long baselineActivity = busyBaselineActivityNanos.get(target);
|
||||
if (baselineActivity == null || baselineActivity != session.lastActivityAtNanos()) {
|
||||
busyBaselineActivityNanos.put(target, session.lastActivityAtNanos());
|
||||
busyDriftBaselineNanos.put(target, driftBeforeThisTick);
|
||||
}
|
||||
long driftSinceBusyStart = accumulatedDriftNanos - busyDriftBaselineNanos.get(target);
|
||||
return (nowNanos - session.lastActivityAtNanos()) + driftSinceBusyStart;
|
||||
}
|
||||
|
||||
void reportTransition(String target, HealthState next) {
|
||||
HealthState previous = states.put(target, next);
|
||||
if (previous == next) return;
|
||||
@@ -171,6 +299,7 @@ public final class FleetHealthMonitor {
|
||||
// member stayed terminal.
|
||||
if (terminal(next)) {
|
||||
failTerminalTarget(target, next);
|
||||
scheduleTerminalRecheck(target, next);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -191,6 +320,48 @@ public final class FleetHealthMonitor {
|
||||
target, state, MAX_FAIL_TARGET_ATTEMPTS, last);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #280: schedule the one bounded, delayed follow-up for a terminal transition — never a
|
||||
* per-tick retry (CB-580 rejected that shape; {@link #reportTransition} still fires
|
||||
* {@link #failTerminalTarget} exactly once per transition, unconditionally on the tick loop).
|
||||
* This is a single one-shot task, scheduled once per transition into GONE/NEVER_READY, so a
|
||||
* member stuck terminal for the rest of its life gets exactly one extra attempt, not one per
|
||||
* tick. See {@link #recheckTerminalTarget} for why the extra attempt is safe.
|
||||
*/
|
||||
private void scheduleTerminalRecheck(String target, HealthState state) {
|
||||
if (scheduler.isShutdown()) return;
|
||||
try {
|
||||
scheduler.schedule(() -> recheckTerminalTarget(target, state),
|
||||
ASK_LAPSE_RECHECK_DELAY_SECONDS, TimeUnit.SECONDS);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("fleet health: could not schedule terminal re-check for member={} state={}",
|
||||
target, state, e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #280: the delayed re-check {@link #scheduleTerminalRecheck} scheduled for one terminal
|
||||
* transition. By now, a {@code fleet_ask} that was still open when {@code state} was first
|
||||
* observed has had time to lapse on its own (see {@link #ASK_LAPSE_RECHECK_DELAY_SECONDS}),
|
||||
* clearing {@code Task#question} back to {@code null} — which is exactly what
|
||||
* {@link MessageService#abandon(String, String, boolean)}'s {@code sweepAsking=false} filter
|
||||
* needs to finally match it. Calling {@link #failTerminalTarget} again is safe only because
|
||||
* {@code sweepAsking} stays {@code false}: a task genuinely still {@code ASKING} is skipped
|
||||
* exactly as it was on the very first attempt — this never fails a ticket whose ask has not yet
|
||||
* lapsed.
|
||||
*
|
||||
* <p><strong>Guarded on "target is still classified {@code state}."</strong> Without this guard,
|
||||
* a member that recovered (or was released and dropped from the roster) between the transition
|
||||
* and this re-check would still take a blind {@code failTarget} call — reaching into whatever
|
||||
* brand-new, unrelated turn it has since picked up and failing it too. {@link #states} already
|
||||
* carries the live classification (updated every tick, pruned to the current roster on release),
|
||||
* so a stale or recovered target simply reads as a mismatch here and this is a no-op.
|
||||
*/
|
||||
void recheckTerminalTarget(String target, HealthState state) {
|
||||
if (states.get(target) != state) return;
|
||||
failTerminalTarget(target, state);
|
||||
}
|
||||
|
||||
private static boolean terminal(HealthState state) {
|
||||
return state == HealthState.GONE || state == HealthState.NEVER_READY;
|
||||
}
|
||||
|
||||
@@ -3,7 +3,7 @@ package dev.ltms.fleet.herdr;
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
/**
|
||||
* Client face onto the herdr daemon (protocol 14, herdr 0.7.0).
|
||||
* Client face onto the herdr daemon (protocol 19, herdr 0.8.0).
|
||||
*
|
||||
* <p>This is the ONLY thing in {@code fleetd} that speaks to herdr. Every method
|
||||
* maps to a herdr JSON-RPC call over its Unix domain socket. Requests are
|
||||
|
||||
@@ -8,7 +8,7 @@ import com.fasterxml.jackson.databind.node.ObjectNode;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
|
||||
/**
|
||||
* Wire codec for herdr's newline-delimited JSON-RPC (protocol 14).
|
||||
* Wire codec for herdr's newline-delimited JSON-RPC (protocol 19).
|
||||
*
|
||||
* <p>Split out from the socket so the framing rules — the ones that actually bit us
|
||||
* during the spike (id MUST be a string; response carries {@code result} or
|
||||
|
||||
@@ -5,6 +5,7 @@ import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
@@ -53,17 +54,40 @@ import java.util.function.Supplier;
|
||||
* daemon and found again by this scan. The trust direction above is unaffected — fleetd writing a
|
||||
* name for a lead it just started is not a pane promoting itself — but <em>staleness</em> becomes
|
||||
* real: a label left behind by a session that has since died would read as a live lead forever.
|
||||
* This scanner does not solve that (its job is naming, and a stale name costs nothing here); the
|
||||
* launcher does, by requiring a running agent in the tab before it counts the lead as live. If you
|
||||
* ever make a decision that <em>removes</em> something based on this map, add the same check.
|
||||
* The remaining hazard is an <em>operator</em> one — a worker {@code tabLabel} template that
|
||||
* happens to start with the same prefix would promote the whole fleet — and that is refused at
|
||||
* startup by {@code FleetConfig.validateLeadTabPrefixes} rather than documented here.
|
||||
*
|
||||
* <p><strong>fleetd #359 — the staleness check this class used to skip.</strong> This used to say
|
||||
* "a stale name costs nothing here" and leave liveness to {@code LeadLauncher}, on the theory that
|
||||
* naming and removing are different decisions. That was wrong: {@code dev.ltms.fleet.msg.LeadCoordLoop}
|
||||
* makes exactly the kind of removal decision the old javadoc warned about, by reading this map to
|
||||
* pick which pane a peer message goes into — and a stale entry there is not free. On a host where
|
||||
* {@code fleetd} had restarted more than once, a labelled-but-dead tab from a previous life was
|
||||
* reported right alongside the live one; {@code LeadCoordLoop.resolveLocalLead()} saw more than one
|
||||
* candidate and refused to guess (safe), but the fix the daemon's own WARN suggests — name a lead
|
||||
* after {@code coordinator.selfId} — stops being safe once two tabs can share a label: step 1 of
|
||||
* that resolution picks whichever matching entry it finds first, which can be the dead one, and
|
||||
* typing a peer's message into a dead shell does not fail — it is silently gone instead of merely
|
||||
* held. {@link #scan()} now cross-checks every labelled tab against {@code agent.list} (the same
|
||||
* signal {@code LeadLauncher.countLeads} already trusts for the same purpose) and drops any tab
|
||||
* with no agent running in it, so a dead tab is never in the map for a caller to pick at all.
|
||||
*
|
||||
* <p><strong>Caching.</strong> {@link #get()} is on the request path (every resolve), so the scan
|
||||
* is TTL-cached and a stale-but-valid map is preferred to a herdr round-trip. A failed scan keeps
|
||||
* the previous answer instead of emptying it — a herdr hiccup must not silently demote a live lead
|
||||
* mid-session.
|
||||
* is TTL-cached and a stale-but-valid map is preferred to a herdr round-trip. A failed scan (herdr
|
||||
* throws) keeps the previous answer instead of emptying it — a herdr hiccup must not silently
|
||||
* demote a live lead mid-session.
|
||||
*
|
||||
* <p><strong>fleetd #359 review, finding 2 — a successful-but-wrong scan is the same hazard.</strong>
|
||||
* The catch above only fires when a call throws. It does nothing for a call that returns 200 with an
|
||||
* incomplete answer — exactly what the ticket's own evidence showed {@code agent.list} can do. Once
|
||||
* this class started trusting that signal, an empty read would otherwise get cached as fact and
|
||||
* silently drop a lead {@code CallerResolver} had, until then, correctly resolved — turning it into a
|
||||
* {@code Role.WORKER}, which refuses every orchestration call. So a terminal this class already
|
||||
* reported as live is not dropped the first time {@code agent.list} loses it: {@link #scan()} grants
|
||||
* it one grace scan (see {@code gracedTerminals}) and only drops it if a <em>later</em> scan still
|
||||
* finds no agent. A terminal never reported live before gets no grace — that would weaken the
|
||||
* original #359 fix itself, which this class's own test suite already pins.
|
||||
*/
|
||||
public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
|
||||
@@ -79,6 +103,15 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
private long scannedAtNanos;
|
||||
private boolean everScanned;
|
||||
|
||||
/**
|
||||
* Terminals currently on their one grace scan: {@code cached} reported them live, the most
|
||||
* recent {@link #scan()} found no agent for them, and they were re-included anyway. Cleared for
|
||||
* a terminal the instant it is seen live again; a terminal still here on the <em>next</em> scan
|
||||
* is finally dropped. Scoped separately from {@link #cached} so a graced terminal cannot renew
|
||||
* its own grace forever just by staying in the exposed map (fleetd #359 review, finding 2).
|
||||
*/
|
||||
private Set<String> gracedTerminals = Set.of();
|
||||
|
||||
/**
|
||||
* @param herdr the herdr client to query ({@code workspace.list},
|
||||
* {@code tab.list}, {@code pane.list} — all read-only)
|
||||
@@ -141,7 +174,7 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
return cached;
|
||||
}
|
||||
|
||||
/** One full pass: labelled tabs → their panes → those panes' terminals. */
|
||||
/** One full pass: labelled tabs → live agents in them → those panes' terminals. */
|
||||
private Map<String, String> scan() {
|
||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
||||
for (JsonNode w : herdr.call("workspace.list").path("workspaces")) {
|
||||
@@ -158,17 +191,47 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
}
|
||||
}
|
||||
|
||||
Map<String, String> byTerminal = new LinkedHashMap<>();
|
||||
if (!nameByTab.isEmpty()) {
|
||||
// One pane.list for every tab: panes carry tab_id, so the join is local.
|
||||
for (JsonNode p : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String name = nameByTab.get(p.path("tab_id").asText(null));
|
||||
String terminal = p.path("terminal_id").asText(null);
|
||||
if (name != null && terminal != null && !terminal.isBlank()) {
|
||||
byTerminal.put(terminal, name);
|
||||
}
|
||||
if (nameByTab.isEmpty()) {
|
||||
gracedTerminals = Set.of();
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
// fleetd #359: a labelled tab is only a lead when herdr also reports a running agent in
|
||||
// it — the same liveness signal LeadLauncher.countLeads trusts for the identical purpose.
|
||||
// Without this, a tab left behind by a session that has since died reads as live forever.
|
||||
Set<String> tabsWithAgent = new HashSet<>();
|
||||
for (JsonNode a : herdr.call("agent.list").path("agents")) {
|
||||
String tabId = a.path("tab_id").asText(null);
|
||||
if (tabId != null) {
|
||||
tabsWithAgent.add(tabId);
|
||||
}
|
||||
}
|
||||
|
||||
Map<String, String> byTerminal = new LinkedHashMap<>();
|
||||
Set<String> stillGraced = new HashSet<>();
|
||||
// One pane.list for every tab: panes carry tab_id, so the join is local.
|
||||
for (JsonNode p : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String tabId = p.path("tab_id").asText(null);
|
||||
String name = nameByTab.get(tabId);
|
||||
String terminal = p.path("terminal_id").asText(null);
|
||||
if (name == null || terminal == null || terminal.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
if (tabsWithAgent.contains(tabId)) {
|
||||
byTerminal.put(terminal, name);
|
||||
continue;
|
||||
}
|
||||
// No agent reported for this tab, but its tab/pane are still here — this is the
|
||||
// ambiguous case review finding 2 named: a successful agent.list that came back short
|
||||
// does not prove the lead is dead. Grant one grace scan to a terminal we had already
|
||||
// reported as live; a terminal we never reported live gets none, so the original #359
|
||||
// fix (a genuinely dead tab is never reported) is unaffected for the common case.
|
||||
if (cached.containsKey(terminal) && !gracedTerminals.contains(terminal)) {
|
||||
byTerminal.put(terminal, name);
|
||||
stillGraced.add(terminal);
|
||||
}
|
||||
}
|
||||
gracedTerminals = stillGraced;
|
||||
return Collections.unmodifiableMap(byTerminal);
|
||||
}
|
||||
|
||||
@@ -177,12 +240,14 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
*
|
||||
* <p>Exact match (case-insensitive, ends stripped) against {@link #tabToName} — no prefix
|
||||
* stripping, so an operator's {@code "lead: something-else"} tab is never mistaken for a
|
||||
* configured lead just because it shares a prefix.
|
||||
* configured lead just because it shares a prefix. The match strips a trailing
|
||||
* {@link PendingCloseMarker} first, so a tab {@code LeadLauncher} has flagged as maybe-dead but
|
||||
* not yet closed keeps resolving normally while that reconcile is pending.
|
||||
*/
|
||||
private String leadNameOf(String label) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
return tabToName.get(label.strip().toLowerCase(Locale.ROOT));
|
||||
return tabToName.get(PendingCloseMarker.strip(label).toLowerCase(Locale.ROOT));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
/**
|
||||
* The suffix {@code dev.ltms.fleet.lead.LeadLauncher} appends to a lead tab's label the first time a
|
||||
* reconcile finds no running agent in it, before it is sure enough to close the tab outright.
|
||||
*
|
||||
* <p><strong>fleetd #359 review, finding 1.</strong> The daemon's own evidence showed
|
||||
* {@code agent.list} can read "no agent" for a tab that genuinely has one running — so a single such
|
||||
* reading must never be treated as proof a tab is dead. {@code LeadLauncher} now writes this marker
|
||||
* on the first miss, and only closes the tab if a <em>later</em>, independent reconcile still finds
|
||||
* it dead while the marker is still there. Two consecutive misses, one restart apart, is a much
|
||||
* stronger claim than one.
|
||||
*
|
||||
* <p>{@link LeadTabScanner} strips the same suffix before matching a label against a configured
|
||||
* lead's {@code tab}, so a flagged-but-actually-still-live tab keeps resolving normally — the marker
|
||||
* changes nothing about which pane {@code LeadCoordLoop} can reach while the flag is pending. Both
|
||||
* classes must use exactly this suffix, which is why it lives here rather than as a private constant
|
||||
* on either.
|
||||
*/
|
||||
public final class PendingCloseMarker {
|
||||
|
||||
public static final String SUFFIX = " [fleetd:pending-close]";
|
||||
|
||||
private PendingCloseMarker() {
|
||||
}
|
||||
|
||||
/** The label with any trailing pending-close marker removed, for name matching. */
|
||||
public static String strip(String label) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
String stripped = label.strip();
|
||||
return stripped.endsWith(SUFFIX)
|
||||
? stripped.substring(0, stripped.length() - SUFFIX.length()).strip()
|
||||
: stripped;
|
||||
}
|
||||
|
||||
/** Whether a label currently carries the marker. */
|
||||
public static boolean isFlagged(String label) {
|
||||
return label != null && label.strip().endsWith(SUFFIX);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Per-target lookup for a profile's configured backend-error pattern (fleetd#201 / #227): how
|
||||
* {@link CompletionResolver} recognizes a backend that failed a turn outright (a credential
|
||||
* outage, a provider 5xx) from whatever ends up on the member's pane, apart from a genuine
|
||||
* completion.
|
||||
*
|
||||
* <p>The pattern is always profile config, never a vendor string in Java source: every backend
|
||||
* words its failure differently, so a hardcoded sentence would only ever match one of them. A
|
||||
* target with no configured pattern is not "off" the way {@link ExhaustedPatternLookup#none()}
|
||||
* is — {@link CompletionResolver} falls back to its narrow {@code (?i)\bAPI Error\s*:}
|
||||
* compatibility pattern instead, so classification still happens, just without a profile-specific
|
||||
* match. {@link #legacy()} is the explicit stand-in existing (pre-fleetd#201) callers pass to get
|
||||
* exactly that fallback-only behavior until a caller wires a real, profile-driven lookup
|
||||
* (fleetd#201 Unit 5).
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface BackendErrorPatternLookup {
|
||||
|
||||
/** The compiled pattern configured for {@code target}'s profile, or {@code null} if none. */
|
||||
Pattern patternFor(String target);
|
||||
|
||||
/**
|
||||
* Legacy lookup — no profile has a configured pattern, so {@link CompletionResolver} classifies
|
||||
* every target using only its built-in {@code (?i)\bAPI Error\s*:} compatibility pattern. The
|
||||
* explicit stand-in existing constructors pass so their behavior is unchanged until a caller
|
||||
* wires a real, profile-driven lookup.
|
||||
*/
|
||||
static BackendErrorPatternLookup legacy() {
|
||||
return target -> null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
/**
|
||||
* Notified when {@link CompletionResolver} has a backend-error match at the start of a pane line,
|
||||
* or both a match and its too-fast crash signature, for a waiting send (fleetd#201 / #227). A text
|
||||
* match inside ordinary pane prose can be a member's report about an error, so it fails the send
|
||||
* without notifying this sink.
|
||||
* {@link CompletionResolver} calls this only after {@code Rendezvous.resolveFailure} returns
|
||||
* {@code true} for that exact waiter, mirroring the win-only race rule {@link ExhaustionSink} uses.
|
||||
*
|
||||
* <p>The public send result is unchanged by this classification — it is still a failed send
|
||||
* ({@code Rendezvous.Kind#FAILED}); this sink is the internal seam a later stage (fleetd#201 Unit
|
||||
* 5, the policy that cools a credential off after two such failures in 60 seconds and tells the
|
||||
* lead) consumes. {@link CompletionResolver} knows only {@code target} (a herdr terminal id); it
|
||||
* has no notion of profiles or credentials, so mapping {@code target} to whatever should be
|
||||
* quarantined is entirely the sink's job — exactly like {@link ExhaustionSink}.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface BackendErrorSink {
|
||||
|
||||
/**
|
||||
* @param target the herdr terminal id whose turn was classified as a backend error
|
||||
* @param matchedLine the single pane line that matched the backend-error pattern
|
||||
* @param reason the full failure reason carried by the classification (the matched line
|
||||
* plus whatever pane context the classifying path attaches)
|
||||
*/
|
||||
void onBackendError(String target, String matchedLine, String reason);
|
||||
|
||||
/**
|
||||
* Inert sink — nothing happens on a backend-error classification. The explicit stand-in a
|
||||
* caller (or a test not exercising this feature) passes instead of a defaulting overload,
|
||||
* exactly like {@link ExhaustionSink#none()}.
|
||||
*/
|
||||
static BackendErrorSink none() {
|
||||
return (target, matchedLine, reason) -> { };
|
||||
}
|
||||
}
|
||||
@@ -14,6 +14,7 @@ import java.util.TreeSet;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Function;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
@@ -85,11 +86,29 @@ public final class CompletionResolver implements TurnListener {
|
||||
private static final String CLIPPED_PANE_TAIL_MARKER =
|
||||
"[Pane tail clipped: member did not call fleet_reply.]";
|
||||
|
||||
/** A pane echo must be this large before it can replace a completion report. */
|
||||
static final int ECHO_MIN_CHARS = 400;
|
||||
|
||||
/** Normalised TUI chrome may add this many characters to an otherwise echoed brief. */
|
||||
static final int MAX_ECHO_EXCESS_CHARS = 160;
|
||||
|
||||
/** The explicit result returned instead of a lead's echoed injected brief. */
|
||||
public static final String NO_REPORT_PREFIX = "[no report — the member ended its turn without fleet_reply, "
|
||||
+ "and the pane still shows the injected brief. Nothing was produced on the pane. Check the "
|
||||
+ "member's worktree and branch for committed work before re-delegating.";
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ExhaustedPatternLookup exhaustedPatterns;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
private final BackendErrorPatternLookup backendErrorPatterns;
|
||||
private final BackendErrorSink backendErrorSink;
|
||||
private final LongSupplier nowNanos;
|
||||
private final Function<String, WorktreeBranch> worktreeBranches;
|
||||
|
||||
/** Known member location, used only to guide a lead after an echoed brief. */
|
||||
public record WorktreeBranch(String worktree, String branch) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-target record of the turn currently in flight: the exact {@link Rendezvous} waiter its
|
||||
@@ -106,7 +125,8 @@ public final class CompletionResolver implements TurnListener {
|
||||
* the other half of the {@link #MIN_TURN_NANOS} floor check, compared against a fresh reading at
|
||||
* resolution time.
|
||||
*/
|
||||
record InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline, long deliveredAtNanos) {
|
||||
record InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline, long deliveredAtNanos,
|
||||
String injectedText) {
|
||||
|
||||
/**
|
||||
* Convenience for tests exercising scrape/suppression logic that don't care about turn
|
||||
@@ -114,7 +134,11 @@ public final class CompletionResolver implements TurnListener {
|
||||
* Not used by production code — {@link #captureBaseline} always records a real reading.
|
||||
*/
|
||||
InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline) {
|
||||
this(waiter, baseline, Long.MIN_VALUE / 2);
|
||||
this(waiter, baseline, Long.MIN_VALUE / 2, null);
|
||||
}
|
||||
|
||||
InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline, long deliveredAtNanos) {
|
||||
this(waiter, baseline, deliveredAtNanos, null);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -129,10 +153,55 @@ public final class CompletionResolver implements TurnListener {
|
||||
* classification actually resolves a waiter. Required for the same
|
||||
* reason as {@code exhaustedPatterns} — pass {@link ExhaustionSink#none()}
|
||||
* to opt out.
|
||||
*
|
||||
* <p>Transition constructor (fleetd#201 Unit 1): delegates to the full constructor below with
|
||||
* {@link BackendErrorPatternLookup#legacy()} and {@link BackendErrorSink#none()}, so every
|
||||
* existing caller keeps today's behavior — the narrow built-in {@code API Error:} match, no
|
||||
* sink notified — until a caller wires a real, profile-driven backend-error lookup and sink
|
||||
* (fleetd#201 Unit 5).
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, System::nanoTime);
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
|
||||
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), System::nanoTime);
|
||||
}
|
||||
|
||||
/** Production constructor with a lookup for the member worktree and branch. */
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, Function<String, WorktreeBranch> worktreeBranches) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
|
||||
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), System::nanoTime, worktreeBranches);
|
||||
}
|
||||
|
||||
/**
|
||||
* Transition constructor (fleetd#201 Unit 1): same legacy backend-error defaults as the 4-arg
|
||||
* constructor above, but with the injectable clock. Kept so existing fleetd#164 timing tests
|
||||
* (and any other caller that wants a controllable clock but not the backend-error feature)
|
||||
* keep compiling unchanged.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, LongSupplier nowNanos) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
|
||||
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), nowNanos);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor (fleetd#201 Unit 1): adds the target-keyed backend-error pattern
|
||||
* lookup and sink alongside the existing exhaustion pair. Required, like {@code exhaustedPatterns}
|
||||
* and {@code exhaustionSink} — pass {@link BackendErrorPatternLookup#legacy()} and
|
||||
* {@link BackendErrorSink#none()} to opt out.
|
||||
*
|
||||
* @param backendErrorPatterns fleetd#201: per-target lookup for a profile's configured
|
||||
* backend-error pattern; a target with none configured is matched
|
||||
* against the built-in compatibility pattern instead (never "off").
|
||||
* @param backendErrorSink fleetd#201: notified when a backend-error classification actually
|
||||
* resolves a waiter — never on a race that lost.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
|
||||
BackendErrorSink backendErrorSink) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, backendErrorPatterns, backendErrorSink,
|
||||
System::nanoTime, _ -> null);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -145,12 +214,25 @@ public final class CompletionResolver implements TurnListener {
|
||||
* package.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, LongSupplier nowNanos) {
|
||||
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
|
||||
BackendErrorSink backendErrorSink, LongSupplier nowNanos) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, backendErrorPatterns, backendErrorSink,
|
||||
nowNanos, _ -> null);
|
||||
}
|
||||
|
||||
/** Full constructor with injectable clock and member location lookup. */
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
|
||||
BackendErrorSink backendErrorSink, LongSupplier nowNanos,
|
||||
Function<String, WorktreeBranch> worktreeBranches) {
|
||||
this.agents = agents;
|
||||
this.rendezvous = rendezvous;
|
||||
this.exhaustedPatterns = Objects.requireNonNull(exhaustedPatterns, "exhaustedPatterns");
|
||||
this.exhaustionSink = Objects.requireNonNull(exhaustionSink, "exhaustionSink");
|
||||
this.backendErrorPatterns = Objects.requireNonNull(backendErrorPatterns, "backendErrorPatterns");
|
||||
this.backendErrorSink = Objects.requireNonNull(backendErrorSink, "backendErrorSink");
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
this.worktreeBranches = Objects.requireNonNull(worktreeBranches, "worktreeBranches");
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -180,7 +262,7 @@ public final class CompletionResolver implements TurnListener {
|
||||
baseline = null; // fail open: no baseline ⇒ no suppression
|
||||
log.debug("delivery baseline for {} failed: {}", target, e.getMessage());
|
||||
}
|
||||
inFlight.put(target, new InFlight(waiter, baseline, nowNanos.getAsLong()));
|
||||
inFlight.put(target, new InFlight(waiter, baseline, nowNanos.getAsLong(), token.injectedText()));
|
||||
}
|
||||
|
||||
/** The turn currently baselined for {@code target}, or {@code null} — a test hook for the captureBaseline path. */
|
||||
@@ -232,7 +314,7 @@ public final class CompletionResolver implements TurnListener {
|
||||
// screen, since that is usually the backend's own error.
|
||||
long elapsedNanos = nowNanos.getAsLong() - turn.deliveredAtNanos();
|
||||
if (elapsedNanos < MIN_TURN_NANOS) {
|
||||
fail(target, turn, tooFastReason(target, elapsedNanos));
|
||||
failTooFast(target, turn, waiter, elapsedNanos);
|
||||
return;
|
||||
}
|
||||
String tail;
|
||||
@@ -298,25 +380,40 @@ public final class CompletionResolver implements TurnListener {
|
||||
+ "matched the profile's exhausted pattern): {}", target, reason);
|
||||
// CB-578 stage B: only on the resolution that actually won the race — a late
|
||||
// duplicate must never quarantine a credential twice for one refusal.
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
if (startsWithExhaustion(matchedLine, exhausted)) {
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
// fleetd#164 (part 2): a scrape that read cleanly and produced content still isn't a real
|
||||
// reply when that content is the backend's own rejection (e.g. an HTTP 400 before the worker
|
||||
// did any work). Classify it as a failure naming the member, rather than handing the caller a
|
||||
// scrape that reads like a completed answer.
|
||||
String backendError = firstMatchingLine(assistantBlock, BACKEND_ERROR);
|
||||
// fleetd#164 (part 2) / fleetd#201: a scrape that read cleanly and produced content still
|
||||
// isn't a real reply when that content is the backend's own rejection (e.g. an HTTP 400
|
||||
// before the worker did any work). Classify it as a failure naming the member, rather than
|
||||
// handing the caller a scrape that reads like a completed answer. A text match alone is not
|
||||
// enough to notify the typed backend-error sink: this assistant block can be a member's
|
||||
// normal prose about an error. A line that starts with the error match is stronger evidence;
|
||||
// the too-fast path below also has its crash signature before it records a credential failure.
|
||||
String backendError = firstMatchingLine(assistantBlock, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
// Carry the whole scrape, not just the matched line. The pattern is a heuristic: a member
|
||||
// that forgot fleet_reply while reporting *about* a backend error matches it too. Failing
|
||||
// is still right — the caller must not read a scrape as an answer — but dropping the rest
|
||||
// of the pane would destroy the report, which is the same defect fleetd#164 is about.
|
||||
fail(target, turn, "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + tail);
|
||||
String reason = "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + tail;
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
if (startsWithBackendError(backendError, backendErrorPatternOrFallback(target))) {
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
String completion = clipped ? tail + "\n" + CLIPPED_PANE_TAIL_MARKER : tail;
|
||||
if (echoesInjectedBrief(tail, turn.injectedText())) {
|
||||
completion = noReportMessage(target) + (clipped ? "\n" + CLIPPED_PANE_TAIL_MARKER : "");
|
||||
}
|
||||
if (rendezvous.resolveCompletion(waiter, completion)) {
|
||||
inFlight.remove(target, turn);
|
||||
if (clipped) {
|
||||
@@ -329,6 +426,40 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A full echoed brief is at least 400 normalised characters. A scrape that contains the brief may
|
||||
* add no more than 160 normalised characters of TUI chrome. This accepts harmless status text, but
|
||||
* preserves a real report that restates the full brief before adding substantive content.
|
||||
*/
|
||||
static boolean echoesInjectedBrief(String scrape, String injectedText) {
|
||||
String normalScrape = normalize(scrape);
|
||||
String normalInjected = normalize(injectedText);
|
||||
if (normalScrape.length() < ECHO_MIN_CHARS || normalInjected.length() < ECHO_MIN_CHARS) {
|
||||
return false;
|
||||
}
|
||||
if (normalInjected.contains(normalScrape)) {
|
||||
return true;
|
||||
}
|
||||
return normalScrape.contains(normalInjected)
|
||||
&& normalScrape.length() - normalInjected.length() <= MAX_ECHO_EXCESS_CHARS;
|
||||
}
|
||||
|
||||
private static String normalize(String text) {
|
||||
return text == null ? "" : text.toLowerCase().replaceAll("[^a-z0-9]+", "");
|
||||
}
|
||||
|
||||
private String noReportMessage(String target) {
|
||||
WorktreeBranch location = worktreeBranches.apply(target);
|
||||
if (location == null || (location.worktree() == null && location.branch() == null)) {
|
||||
return NO_REPORT_PREFIX + "]";
|
||||
}
|
||||
String locationText = location.worktree() == null ? "" : " worktree=" + location.worktree();
|
||||
if (location.branch() != null) {
|
||||
locationText += " branch=" + location.branch();
|
||||
}
|
||||
return NO_REPORT_PREFIX + locationText + "]";
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#211: the raw-scrape fallback classification, run only when {@link #lastAssistantBlock}
|
||||
* found nothing usable (see the call site in {@link #resolve}). Mirrors the two classifications
|
||||
@@ -349,18 +480,27 @@ public final class CompletionResolver implements TurnListener {
|
||||
+ "usable assistant block; no fleet_reply): {}", target, reason);
|
||||
// CB-578 stage B: only on the resolution that actually won the race — a late
|
||||
// duplicate must never quarantine a credential twice for one refusal.
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
if (startsWithExhaustion(matchedLine, exhausted)) {
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
String backendError = firstMatchingLine(raw, BACKEND_ERROR);
|
||||
String backendError = firstMatchingLine(raw, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
// Carry the pane, not just the matched line — the same fleetd#164 rule the normal path
|
||||
// above applies. Here it matters more, not less: the trimmed block was empty, so the raw
|
||||
// scrape is the ONLY copy of whatever the member managed to say. Clipped to the same cap
|
||||
// the normal path uses, since a raw screen has no boundary trimming to bound it.
|
||||
fail(target, turn, "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + clip(raw));
|
||||
String reason = "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + clip(raw);
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback from the raw scrape: {}", target, reason);
|
||||
if (startsWithBackendError(backendError, backendErrorPatternOrFallback(target))) {
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
@@ -402,22 +542,71 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#164: the failure reason for a turn that settled inside {@link #MIN_TURN_NANOS} — names
|
||||
* the member and both timings, and appends whatever the pane shows (usually the backend's own
|
||||
* error) so the caller sees the cause, not just "it failed".
|
||||
* fleetd#164 (floor) / fleetd#201 (classification): fail a turn that settled inside
|
||||
* {@link #MIN_TURN_NANOS} — a crash signature (e.g. a backend HTTP 400 before the worker did
|
||||
* anything) that a bare {@code BUSY -> DONE} transition cannot be told apart from a genuinely
|
||||
* fast completion. Runs the same backend-error classification the normal and raw-scrape paths
|
||||
* apply, against whatever is on screen right now: a match together with the too-fast crash
|
||||
* signature notifies {@link #backendErrorSink} (only on the resolution that wins the race). A
|
||||
* non-match stays the generic too-fast failure, naming the member and both timings, with
|
||||
* whatever the pane shows appended so the caller sees the evidence, not just "it failed".
|
||||
*
|
||||
* <p>fleetd#376: <strong>this path must never resolve a completion.</strong> A fast backend can
|
||||
* genuinely answer inside the floor, so the failure is sometimes wrong — but it is wrong in the
|
||||
* loud direction, and the fix for that is honest wording, not a guess at the pane's meaning.
|
||||
* Reclassifying from the scrape was tried and rejected: there is no reliable positive marker for
|
||||
* "this is a real reply" across backends. {@link #lastAssistantBlock} falls back to the entire
|
||||
* pane when it finds no {@code ⏺} marker, so on a crash the candidate "reply" is the whole
|
||||
* screen; and {@code ⏺} itself is a Claude Code marker that an opencode pane never carries — the
|
||||
* very backend whose speed raised this ticket. Any weaker test (non-blank, or "contains sentence
|
||||
* punctuation") passes on almost every crash, because a pane holding a file path, a version
|
||||
* number or a hostname contains a full stop. That trades a loud wrong answer for a silent one.
|
||||
*/
|
||||
private String tooFastReason(String target, long elapsedNanos) {
|
||||
private void failTooFast(String target, InFlight turn, CompletableFuture<Rendezvous.Resolution> waiter,
|
||||
long elapsedNanos) {
|
||||
String scrape;
|
||||
try {
|
||||
scrape = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
scrape = agents.read(target, SCRAPE_SOURCE);
|
||||
} catch (RuntimeException e) {
|
||||
scrape = "";
|
||||
}
|
||||
String reason = String.format(
|
||||
"member %s went BUSY -> DONE in %dms (floor %dms) — too fast to be real work, most "
|
||||
+ "likely a backend error before any work started",
|
||||
String clippedScrape = clip(scrape);
|
||||
String timing = String.format(
|
||||
"member %s went BUSY -> DONE in %dms (floor %dms)",
|
||||
target, elapsedNanos / 1_000_000, MIN_TURN_NANOS / 1_000_000);
|
||||
return scrape.isBlank() ? reason : reason + ": " + scrape;
|
||||
String backendError = firstMatchingLine(scrape, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
String reason = timing + " — too fast to be real work, and the pane carries a backend "
|
||||
+ "error: " + clippedScrape;
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
// fleetd#201 Unit 1: only on the resolution that actually won the race.
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
// fleetd#376: no pattern matched, so the cause is genuinely unknown. Say that, rather than
|
||||
// asserting a backend error the way this message used to — a fast backend really can finish
|
||||
// inside the floor, and a reader who trusts a wrong cause stops looking at the pane.
|
||||
String reason = timing + " — inside the floor. That is usually a backend error before any "
|
||||
+ "work started, but a fast backend can answer inside it too, and nothing here can "
|
||||
+ "tell those apart, so the turn is reported failed rather than guessed. Read the "
|
||||
+ "pane below before deciding which it was";
|
||||
fail(target, turn, clippedScrape.isBlank() ? reason : reason + ": " + clippedScrape);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#201: the pattern to classify a backend error against for {@code target} — its
|
||||
* profile's configured {@link BackendErrorPatternLookup} entry when there is one, else the
|
||||
* built-in {@link #BACKEND_ERROR} compatibility pattern. A target with no configured pattern is
|
||||
* never "off": it always falls back to this narrow default, exactly like the classification
|
||||
* behaved before fleetd#201 (Unit 5 reports a target relying on this fallback separately, as
|
||||
* legacy-default coverage rather than full per-profile coverage).
|
||||
*/
|
||||
private Pattern backendErrorPatternOrFallback(String target) {
|
||||
Pattern configured = backendErrorPatterns.patternFor(target);
|
||||
return configured != null ? configured : BACKEND_ERROR;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -447,17 +636,119 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for the CB-578 stage A exhausted-pattern classification, logged at startup
|
||||
* True when nothing before the match on this pane line ends a sentence — that is, the match is
|
||||
* still inside the line's first sentence rather than inside prose a member wrote about it.
|
||||
* Used to decide whether an exhaustion match may quarantine a credential (fleetd #348).
|
||||
*
|
||||
* <p><strong>Why this is looser than {@link #startsWithBackendError}.</strong> An
|
||||
* {@code exhaustedPattern} is written per profile and may name only the decisive words of a
|
||||
* provider message — {@code "usage limit has been reached"} without its leading {@code "The"}.
|
||||
* A start-of-line check would then reject the genuine refusal. That is the false negative
|
||||
* fleetd #348's invariant 1 calls the worse direction: an unrecorded exhaustion leaves the
|
||||
* fleet spawning into a credential with no capacity, and a quarantine runs 1800s against the
|
||||
* backend-error cooldown's fixed 60s.
|
||||
*
|
||||
* <p>This rule accepts a superset of what a start-of-line check accepts: if the match begins
|
||||
* right after the chrome, there is nothing in front of it, so there is no sentence ending
|
||||
* either. So moving to it cannot add a false negative.
|
||||
*
|
||||
* <p><strong>No chrome skipping here, deliberately.</strong> The first version of this method
|
||||
* copied {@code startsWithBackendError}'s leading-chrome loop. Measured on merge: deleting that
|
||||
* loop left all 1369 tests green, and it must — the scan only looks for {@code . ! ?}, and no
|
||||
* terminal chrome character is one of those. A step that cannot change the result is worse than
|
||||
* no step, because the next reader takes it as evidence that chrome was handled.
|
||||
*
|
||||
* <p>It stays a heuristic. Prose whose <em>first</em> sentence carries the pattern still
|
||||
* notifies the sink, and a genuine refusal behind an earlier full stop (a hostname, a version
|
||||
* number) still does not. Both are known and neither is fixed here.
|
||||
*/
|
||||
private static boolean startsWithExhaustion(String line, Pattern pattern) {
|
||||
var matcher = pattern.matcher(line);
|
||||
if (!matcher.find()) {
|
||||
return false;
|
||||
}
|
||||
for (int prefix = 0; prefix < matcher.start(); prefix++) {
|
||||
if (".!?".indexOf(line.charAt(prefix)) >= 0) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the error pattern begins the matched pane line, rather than appearing in prose.
|
||||
*
|
||||
* <p>Leading terminal chrome is skipped first — box-drawing characters, bullets, gutter bars and
|
||||
* spaces. #339 introduced this check with a bare {@code lookingAt}, and that rejected a genuine
|
||||
* error line rendered as {@code "| 503 Service Unavailable: ..."}: the send still failed, but the
|
||||
* credential outage was never recorded. That is the false negative #339's own invariant 3 called
|
||||
* worse than the false positive it set out to fix — measured with a throwaway probe on the
|
||||
* raw-scrape path, which is exactly the path whose comment says to expect leading chrome.
|
||||
*
|
||||
* <p>Skipping only a leading run of non-letter, non-digit characters keeps the fix's intent. A
|
||||
* member's prose ({@code "I checked the retry path. An API Error: makes it back off."}) still
|
||||
* does not match, because there the pattern sits after words, not after chrome.
|
||||
*/
|
||||
private static boolean startsWithBackendError(String line, Pattern pattern) {
|
||||
int i = 0;
|
||||
while (i < line.length() && !Character.isLetterOrDigit(line.charAt(i))) {
|
||||
i++;
|
||||
}
|
||||
return pattern.matcher(line.substring(i)).lookingAt();
|
||||
}
|
||||
|
||||
/**
|
||||
* What an unset pattern key means for the classification it configures (fleetd#415).
|
||||
* {@code coverage()} cannot infer this from the key's name — the two keys it currently
|
||||
* describes disagree on it, and a string comparison on the name would just move the same bug
|
||||
* to a new spot — so every caller must state it explicitly.
|
||||
*
|
||||
* <p><strong>This alone does not prove a caller passes the right one for its key.</strong> A
|
||||
* test that calls {@code coverage()} directly and supplies the meaning itself only proves this
|
||||
* enum is worded correctly, never that {@code Fleetd}'s two call sites pair each key with its
|
||||
* true meaning — that pairing is #415's actual defect. Measured on review: swapping the two
|
||||
* {@code UnsetMeaning} arguments at those call sites (giving {@code exhaustedPattern} the
|
||||
* built-in-default wording and {@code errorPattern} the off wording — #415's exact defect with
|
||||
* the keys exchanged) compiled with 0 errors and left all 1506 existing tests green. See
|
||||
* {@code dev.ltms.fleet.Fleetd#exhaustedPatternCoverageLine}/{@code #errorPatternCoverageLine}
|
||||
* and {@code FleetdPatternCoverageLineTest}, which exists specifically to catch that swap.
|
||||
*/
|
||||
public enum UnsetMeaning {
|
||||
/** No fallback exists: a profile with no configured pattern truly has this classification off. */
|
||||
OFF,
|
||||
/** A built-in pattern applies when unset: the classification still runs for that profile. */
|
||||
BUILT_IN_DEFAULT
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for a fleetd#201/CB-578-style pattern-key classification, logged at startup
|
||||
* the way {@link dev.ltms.fleet.health.FleetHealthMonitor#coverage} is — so an operator can
|
||||
* see whether the classification is on, and for which profiles, without reading every
|
||||
* profile's config by hand.
|
||||
*
|
||||
* <p>fleetd#415: this method measures <em>pattern coverage</em> — how many profiles set the
|
||||
* key — which is not the same thing as <em>feature state</em> for a key with a fallback. For
|
||||
* {@code errorPattern}, an empty {@code configuredProfiles} still runs the classification
|
||||
* against {@code CompletionResolver}'s built-in compatibility pattern ({@link #BACKEND_ERROR}
|
||||
* at line ~84); for {@code exhaustedPattern} there is no fallback, so empty really does mean
|
||||
* off. {@code unsetMeaning} is the single, required source of that fact — see
|
||||
* {@link dev.ltms.fleet.config.FleetConfig#rejectMalformedProfilePatterns} lines ~2029-2032 for
|
||||
* where it is documented for config authors. It is a required parameter, not a defaulted
|
||||
* overload: a third pattern key added later must supply one to compile at all, rather than
|
||||
* silently inheriting whichever wording this method happened to default to.
|
||||
*
|
||||
* @param allProfiles every configured profile name
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry an exhausted pattern
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry the pattern
|
||||
*/
|
||||
public static String coverage(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
public static String coverage(String patternKey, UnsetMeaning unsetMeaning, Set<String> allProfiles,
|
||||
Set<String> configuredProfiles) {
|
||||
if (configuredProfiles.isEmpty()) {
|
||||
return "off (no profile has an exhaustedPattern configured; profiles: " + sorted(allProfiles) + ")";
|
||||
return switch (unsetMeaning) {
|
||||
case OFF -> "off (no profile has an " + patternKey + " configured; profiles: "
|
||||
+ sorted(allProfiles) + ")";
|
||||
case BUILT_IN_DEFAULT -> "built-in default for all profiles (no profile customises "
|
||||
+ patternKey + "; profiles: " + sorted(allProfiles) + ")";
|
||||
};
|
||||
}
|
||||
Set<String> unconfigured = new TreeSet<>(allProfiles);
|
||||
unconfigured.removeAll(configuredProfiles);
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* Notified when {@link CompletionResolver} actually delivers a {@code BACKEND_EXHAUSTED}
|
||||
* classification to a waiting send (CB-578 stage B) — never on a race that lost (see
|
||||
@@ -10,22 +12,81 @@ package dev.ltms.fleet.inject;
|
||||
* profiles or credentials, so mapping {@code target} to whatever should be quarantined is entirely
|
||||
* the sink's job — see {@code Fleetd.main}'s wiring, which resolves target → session → profile →
|
||||
* {@code effectiveCredentialId()} and calls {@code BackendQuarantine.quarantine} on it.
|
||||
*
|
||||
* <p><strong>The 3-arg overload is the single abstract method — fleetd #234, round 4.</strong> Two
|
||||
* earlier rounds each shipped a caller that silently dropped the profile hint (see {@link
|
||||
* #onExhausted(String, String, String)}): a lambda written against this interface can only ever
|
||||
* implement whichever overload is abstract, and while the 2-arg form held that position, EVERY
|
||||
* lambda site — a call-site forwarder in {@code Fleetd.java}, a hand-built test double — bound to
|
||||
* it and silently inherited the profile-dropping default, whether or not its author remembered the
|
||||
* hazard. Making the 3-arg form abstract instead removes the shape entirely: a lambda declared
|
||||
* against this interface today is <em>forced</em> by the compiler to take {@code (target, reason,
|
||||
* profile)}, so there is no overload left for it to bind to that can drop the hint. This is a type
|
||||
* change, not a test — it holds even for a caller that has never heard of fleetd #234.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface ExhaustionSink {
|
||||
|
||||
/**
|
||||
* @param target the herdr terminal id whose turn was classified {@code BACKEND_EXHAUSTED}
|
||||
* @param reason the matched-line reason carried by the classification
|
||||
* @param target the herdr terminal id whose turn was classified {@code BACKEND_EXHAUSTED}
|
||||
* @param reason the matched-line reason carried by the classification
|
||||
* @param profile the profile the caller already knows should be quarantined, or {@code null}
|
||||
* when the caller has no better answer than {@code target} alone (fleetd #234):
|
||||
* a caller whose {@code target} is not yet resolvable through whatever roster
|
||||
* the sink's implementation consults — {@link
|
||||
* dev.ltms.fleet.member.OpenCodeLauncher}'s model-mismatch check fires from
|
||||
* {@code SessionAwareHandle.agentSessionId()}, which runs during {@code
|
||||
* SessionManager.acquire()} <em>before</em> that session is registered, so a
|
||||
* target -> session -> profile lookup finds nothing at that point. That launcher
|
||||
* already has its own {@code FleetConfig.Profile} in hand and does not need the
|
||||
* roster to know which profile to quarantine, so it supplies this directly
|
||||
* instead of leaving the sink to guess.
|
||||
*/
|
||||
void onExhausted(String target, String reason);
|
||||
void onExhausted(String target, String reason, String profile);
|
||||
|
||||
/**
|
||||
* Convenience for a caller with no profile to offer — every existing call site that predates
|
||||
* the hint (fleetd #234): {@link dev.ltms.fleet.inject.CompletionResolver}'s two call sites
|
||||
* always call with a {@code target} that IS live in the roster at the time of the call, so they
|
||||
* need no hint and keep working exactly as before, unchanged by this default.
|
||||
*
|
||||
* @param target as {@link #onExhausted(String, String, String)}
|
||||
* @param reason as {@link #onExhausted(String, String, String)}
|
||||
*/
|
||||
default void onExhausted(String target, String reason) {
|
||||
onExhausted(target, reason, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Inert sink — nothing happens on exhaustion. The explicit stand-in a caller (or a test not
|
||||
* exercising this feature) passes instead of a defaulting overload, exactly like
|
||||
* {@link ExhaustedPatternLookup#none()}.
|
||||
* {@link ExhaustedPatternLookup#none()}. Safe as a lambda: the 3-arg form is now the interface's
|
||||
* single abstract method, so a lambda here has no other overload to silently bind to instead —
|
||||
* it simply does nothing with all three arguments.
|
||||
*/
|
||||
static ExhaustionSink none() {
|
||||
return (target, reason) -> { };
|
||||
return (target, reason, profile) -> { };
|
||||
}
|
||||
|
||||
/**
|
||||
* A sink that forwards to whatever {@code target} currently supplies (fleetd #234). Exists to
|
||||
* break a genuine construction-order cycle: {@code Fleetd.main} builds its adapters (including
|
||||
* {@link dev.ltms.fleet.member.OpenCodeLauncher}) before {@code sessions} exists, so it cannot
|
||||
* hand them the real sink yet — it hands them a forwarder pointed at an {@code
|
||||
* AtomicReference<ExhaustionSink>} that starts at {@link #none()} and gets {@code .set()} to the
|
||||
* real sink once {@code sessions} is built. {@code target} is evaluated on every call, never
|
||||
* cached, so the forwarder keeps working after the reference is repointed.
|
||||
*
|
||||
* <p>Now safe as a one-line lambda (round 4): forwarding the single 3-arg abstract method
|
||||
* forwards everything a caller can supply — there is no separate 2-arg overload left for a
|
||||
* forwarder to bind to instead and silently lose the hint. Kept as a named factory rather than
|
||||
* written inline at each call site anyway, so a test can call the exact object {@code
|
||||
* Fleetd.java} builds instead of asserting a rebuilt copy of its shape (round 3's lesson).
|
||||
*
|
||||
* @param target supplies the sink to forward to, evaluated fresh on every call
|
||||
* @return a sink whose call delegates to {@code target.get()}
|
||||
*/
|
||||
static ExhaustionSink forwardingTo(Supplier<ExhaustionSink> target) {
|
||||
return (t, r, p) -> target.get().onExhausted(t, r, p);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -145,8 +145,58 @@ public final class Injector {
|
||||
return router != null ? router.agentsFor(target) : agents;
|
||||
}
|
||||
|
||||
/** The result of trying to remove an undelivered message from the injector. */
|
||||
public enum Cancellation {
|
||||
CANCELLED,
|
||||
DELIVERED,
|
||||
NOT_DELIVERED
|
||||
}
|
||||
|
||||
/**
|
||||
* An identity handle for one queued delivery. It is the only value accepted by
|
||||
* {@link #cancel(Delivery)}, so a caller cannot cancel a different message with the same target
|
||||
* or text.
|
||||
*/
|
||||
public static final class Delivery {
|
||||
private final Pending pending;
|
||||
|
||||
private Delivery(Pending pending) {
|
||||
this.pending = pending;
|
||||
}
|
||||
|
||||
public CompletableFuture<Void> completion() {
|
||||
return pending.delivered;
|
||||
}
|
||||
}
|
||||
|
||||
/** A pending message and the future that completes when it has been delivered. */
|
||||
private record Pending(String text, TurnToken token, CompletableFuture<Void> delivered) {
|
||||
private static final class Pending {
|
||||
enum State { QUEUED, DELIVERED, NOT_DELIVERED, CANCELLED }
|
||||
|
||||
final String target;
|
||||
final String text;
|
||||
final TurnToken token;
|
||||
final CompletableFuture<Void> delivered;
|
||||
volatile State state = State.QUEUED; // written under the owning Target monitor
|
||||
|
||||
Pending(String target, String text, TurnToken token, CompletableFuture<Void> delivered) {
|
||||
this.target = target;
|
||||
this.text = text;
|
||||
this.token = token;
|
||||
this.delivered = delivered;
|
||||
}
|
||||
|
||||
String text() {
|
||||
return text;
|
||||
}
|
||||
|
||||
TurnToken token() {
|
||||
return token;
|
||||
}
|
||||
|
||||
CompletableFuture<Void> delivered() {
|
||||
return delivered;
|
||||
}
|
||||
}
|
||||
|
||||
/** Per-worker delivery state, guarded by its own monitor (single writer per worker). */
|
||||
@@ -157,6 +207,7 @@ public final class Injector {
|
||||
boolean awaitingCompletion; // a delivered message's turn is not yet known-complete
|
||||
boolean turnObserved; // saw a real `working` sample since that delivery (turn ran)
|
||||
int unknownSinceTurn; // consecutive `unknown` samples while a delegation is outstanding (CB-109)
|
||||
int unknownSincePostTurn; // the same, for the post-turn housekeeping phase (fleetd #306)
|
||||
int notReadySincePoll; // consecutive injectable samples a queued message waited on the readiness gate (CB-114)
|
||||
boolean postTurnPending; // completion observed; adapter housekeeping has not started yet
|
||||
boolean awaitingPostTurnPickup;
|
||||
@@ -176,15 +227,47 @@ public final class Injector {
|
||||
* <p>Uses an atomic map update so a concurrent {@link #drop} cannot slip between "find the
|
||||
* target" and "queue the message" and orphan it in a target it just removed.
|
||||
*/
|
||||
public CompletableFuture<Void> enqueue(String target, String text, TurnToken token) {
|
||||
public Delivery enqueue(String target, String text, TurnToken token) {
|
||||
CompletableFuture<Void> delivered = new CompletableFuture<>();
|
||||
Pending p = new Pending(text, token, delivered);
|
||||
Pending p = new Pending(target, text, token, delivered);
|
||||
targets.compute(target, (_, existing) -> {
|
||||
Target t = (existing != null) ? existing : new Target();
|
||||
t.add(p); // synchronized on the Target monitor — atomic with a concurrent drop
|
||||
return t;
|
||||
});
|
||||
return delivered;
|
||||
return new Delivery(p);
|
||||
}
|
||||
|
||||
/**
|
||||
* Cancel this exact queued delivery. The target monitor serializes this operation with
|
||||
* {@link #onStatus}: if delivery wins that race, this returns {@link Cancellation#DELIVERED}
|
||||
* rather than claiming the message remained queued.
|
||||
*/
|
||||
public Cancellation cancel(Delivery delivery) {
|
||||
Pending p = delivery.pending;
|
||||
Target t = targets.get(p.target);
|
||||
if (t == null) {
|
||||
return cancellationOf(p);
|
||||
}
|
||||
synchronized (t) {
|
||||
if (p.state != Pending.State.QUEUED || !t.queue.remove(p)) {
|
||||
return cancellationOf(p);
|
||||
}
|
||||
p.state = Pending.State.CANCELLED;
|
||||
if (isQuiescent(t)) {
|
||||
targets.remove(p.target, t);
|
||||
}
|
||||
return Cancellation.CANCELLED;
|
||||
}
|
||||
}
|
||||
|
||||
private static Cancellation cancellationOf(Pending p) {
|
||||
return p.state == Pending.State.DELIVERED ? Cancellation.DELIVERED : Cancellation.NOT_DELIVERED;
|
||||
}
|
||||
|
||||
private static boolean isQuiescent(Target t) {
|
||||
return t.queue.isEmpty() && !t.awaitingPickup && !t.awaitingCompletion
|
||||
&& !t.postTurnPending && !t.awaitingPostTurnPickup && !t.postTurnObserved;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -217,10 +300,12 @@ public final class Injector {
|
||||
t.awaitingPickup = false;
|
||||
t.injectableSincePickup = 0;
|
||||
t.unknownSinceTurn = 0;
|
||||
t.unknownSincePostTurn = 0;
|
||||
t.notReadySincePoll = 0;
|
||||
if (t.awaitingCompletion) t.turnObserved = true;
|
||||
} else if (status.injectable()) { // IDLE or BLOCKED
|
||||
t.unknownSinceTurn = 0;
|
||||
t.unknownSincePostTurn = 0;
|
||||
if (t.awaitingPostTurnPickup) {
|
||||
if (++t.injectableSincePostTurnPickup >= PICKUP_GRACE_POLLS) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
@@ -271,6 +356,7 @@ public final class Injector {
|
||||
try {
|
||||
agentsFor(target).send(target, p.text());
|
||||
t.queue.poll();
|
||||
p.state = Pending.State.DELIVERED;
|
||||
t.awaitingPickup = true;
|
||||
t.awaitingCompletion = true;
|
||||
t.turnObserved = false;
|
||||
@@ -280,6 +366,7 @@ public final class Injector {
|
||||
// Delivery failed at herdr; drop the poisoned message and surface it
|
||||
// rather than blocking the queue behind it.
|
||||
t.queue.poll();
|
||||
p.state = Pending.State.NOT_DELIVERED;
|
||||
sent = p;
|
||||
sendError = e;
|
||||
}
|
||||
@@ -290,6 +377,9 @@ public final class Injector {
|
||||
// fail every queued message and release the target (CB-114) instead of
|
||||
// polling it indefinitely with the caller's future never completing.
|
||||
notReady = new ArrayList<>(t.queue);
|
||||
for (Pending pending : notReady) {
|
||||
pending.state = Pending.State.NOT_DELIVERED;
|
||||
}
|
||||
log.warn("readiness grace for {} expired after {} polls ({}s): target never "
|
||||
+ "became deliverable, so failing {} queued message(s) that never "
|
||||
+ "reached its pane",
|
||||
@@ -315,12 +405,29 @@ public final class Injector {
|
||||
t.unknownSinceTurn = 0;
|
||||
turnFailed = true;
|
||||
}
|
||||
// fleetd #306: the same escape for the post-turn housekeeping phase. Four latches
|
||||
// gate delivery (awaitingCompletion, postTurnPending, awaitingPostTurnPickup,
|
||||
// postTurnObserved) and only the first had a way out of a sustained unknown streak —
|
||||
// a gate that closed one direction only. The other two below are released here as
|
||||
// well; postTurnPending needs no escape because it is cleared unconditionally on the
|
||||
// line after the listener call that sets it.
|
||||
//
|
||||
// This does NOT set turnFailed. The delegated turn already completed and its waiter
|
||||
// already resolved — what is outstanding is adapter housekeeping (the `/clear`).
|
||||
// Reporting a turn failure here would drive SessionManager.onFailed on a session
|
||||
// that genuinely finished its work, which is a worse lie than the wedge.
|
||||
if ((t.awaitingPostTurnPickup || t.postTurnObserved)
|
||||
&& ++t.unknownSincePostTurn >= TURN_STALL_GRACE_POLLS) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
t.postTurnObserved = false;
|
||||
t.injectableSincePostTurnPickup = 0;
|
||||
t.unknownSincePostTurn = 0;
|
||||
}
|
||||
}
|
||||
|
||||
// Reclaim the entry once the worker is fully quiescent (nothing queued, no pickup or
|
||||
// completion awaited), so the map cannot grow without bound across short-lived workers.
|
||||
if (t.queue.isEmpty() && !t.awaitingPickup && !t.awaitingCompletion
|
||||
&& !t.postTurnPending && !t.awaitingPostTurnPickup && !t.postTurnObserved) {
|
||||
if (isQuiescent(t)) {
|
||||
targets.remove(target, t);
|
||||
}
|
||||
}
|
||||
@@ -411,6 +518,9 @@ public final class Injector {
|
||||
boolean hadDeliveredTurn;
|
||||
synchronized (t) {
|
||||
pending = new ArrayList<>(t.queue);
|
||||
for (Pending p : pending) {
|
||||
p.state = Pending.State.NOT_DELIVERED;
|
||||
}
|
||||
t.queue.clear();
|
||||
hadDeliveredTurn = t.awaitingCompletion;
|
||||
t.awaitingCompletion = false;
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* fleetd #446: the single live source for "is {@code exhaustedPattern} configured for this
|
||||
* profile, and what does it compile to". Read fresh off the config supplier on every call — the
|
||||
* same reason {@code CompositePeerLauncher#models0} is a live supplier read rather than a value
|
||||
* captured at construction (see that class's doc) — so an operator can arm or disarm usage-limit
|
||||
* detection for a profile by editing {@code exhaustedPattern} and reloading, with no restart.
|
||||
*
|
||||
* <p>Before this class, {@code exhaustedPattern} was compiled once into a {@code Map<String,
|
||||
* Pattern>} built inside {@code Fleetd.main} at startup ({@code ConfigRef}'s class doc used to
|
||||
* list it under <em>Deferred</em>, CB-578 stage A) — the asymmetry fleetd #446 exists to close:
|
||||
* an operator could turn a model off at runtime ({@code models.allow}'s {@code enabled: false},
|
||||
* hot since fleetd #422) but could not arm the detector that would tell them to, without a
|
||||
* restart. That was backwards for a feature whose whole point is to react while the fleet runs.
|
||||
*
|
||||
* <p>{@link #patternFor} is what {@link CompletionResolver} enforces on (via the {@link
|
||||
* ExhaustedPatternLookup} production wiring in {@code Fleetd.main}); {@link #armed} is what
|
||||
* {@code fleet_profiles}/{@code GET /profiles} report as {@code exhaustionDetectionArmed}. Both
|
||||
* read this ONE object, so the report can never disagree with the behaviour — the same rule
|
||||
* {@code CompositePeerLauncher#modelGateState()}'s javadoc states for the model gate's own
|
||||
* armed/off pair (fleetd #404): "armed" and "which models are off" must come from one read of the
|
||||
* same accessor the gate enforces on.
|
||||
*
|
||||
* <h2>Cache eviction</h2>
|
||||
* Cached by profile NAME, not by pattern text. A {@code Map<String, Pattern>} keyed by the
|
||||
* pattern STRING would grow by one entry per distinct regex ever typed for any profile across the
|
||||
* daemon's uptime — unbounded in practice, because tuning a regex to match a backend's exact
|
||||
* wording is exactly the kind of edit an operator makes several times while getting it right, and
|
||||
* every edit-and-reload cycle would leave the previous attempt's compiled {@link Pattern} behind
|
||||
* forever. Keyed by profile name instead, this cache holds at most one entry per profile name that
|
||||
* has ever been looked up — and that key space is bounded by the (small, human-authored) set of
|
||||
* configured profiles, which changes only on a restart: adding or removing a profile is itself a
|
||||
* deferred key (a new backend needs its own launcher, built once — see {@code ConfigRef}'s class
|
||||
* doc), so profile names do not churn the way pattern text does. Re-editing an EXISTING profile's
|
||||
* {@code exhaustedPattern} — the case this class exists to make hot — simply overwrites that
|
||||
* profile's one cache entry; it never adds a new one.
|
||||
*/
|
||||
public final class LiveExhaustedPatterns {
|
||||
|
||||
/** A compiled pattern paired with the source string it was compiled from, for change detection. */
|
||||
private record Cached(String source, Pattern compiled) {
|
||||
}
|
||||
|
||||
private final Supplier<Map<String, FleetConfig.Profile>> profiles;
|
||||
private final ConcurrentHashMap<String, Cached> cache = new ConcurrentHashMap<>();
|
||||
|
||||
public LiveExhaustedPatterns(Supplier<Map<String, FleetConfig.Profile>> profiles) {
|
||||
this.profiles = Objects.requireNonNull(profiles, "profiles");
|
||||
}
|
||||
|
||||
/**
|
||||
* The compiled {@code exhaustedPattern} currently configured for {@code profileName}, or
|
||||
* {@code null} when that profile is unknown or has none configured. Recompiles only when the
|
||||
* live pattern text differs from what is cached for this profile name; {@code
|
||||
* FleetConfig#rejectMalformedProfilePatterns} already refuses a config (at load and at reload)
|
||||
* whose {@code exhaustedPattern} does not compile, so this is not expected to throw in
|
||||
* production — it is not defended against here for that reason, the same trust
|
||||
* {@code CompositePeerLauncher#models0} places in config validation having already run.
|
||||
*/
|
||||
public Pattern patternFor(String profileName) {
|
||||
if (profileName == null) {
|
||||
return null;
|
||||
}
|
||||
FleetConfig.Profile profile = profiles.get().get(profileName);
|
||||
if (profile == null || !profile.hasExhaustedPattern()) {
|
||||
return null;
|
||||
}
|
||||
String source = profile.exhaustedPattern();
|
||||
Cached cached = cache.get(profileName);
|
||||
if (cached != null && cached.source().equals(source)) {
|
||||
return cached.compiled();
|
||||
}
|
||||
Cached fresh = new Cached(source, Pattern.compile(source));
|
||||
cache.put(profileName, fresh);
|
||||
return fresh.compiled();
|
||||
}
|
||||
|
||||
/** Whether {@code profileName} currently has a usage-limit pattern configured (live). */
|
||||
public boolean armed(String profileName) {
|
||||
return patternFor(profileName) != null;
|
||||
}
|
||||
}
|
||||
@@ -4,6 +4,7 @@ import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.PendingCloseMarker;
|
||||
import dev.ltms.fleet.herdr.Tab;
|
||||
import dev.ltms.fleet.herdr.Workspace;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
@@ -12,9 +13,11 @@ import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
|
||||
/**
|
||||
@@ -44,8 +47,33 @@ import dev.ltms.fleet.peer.PeerLauncher;
|
||||
* privilege escalation (the tab label never granted anything a pane could take for itself; see that
|
||||
* class's javadoc), but <em>staleness</em>: a label left behind by a crashed session would otherwise
|
||||
* read as a live lead forever, and the lead would never be relaunched. So a lead counts as live only
|
||||
* when herdr also reports a <em>running agent</em> in that tab — see {@link #liveLeads}. A labelled
|
||||
* when herdr also reports a <em>running agent</em> in that tab — see {@link #countLeads}. A labelled
|
||||
* tab with no agent in it is not a lead.
|
||||
*
|
||||
* <p><strong>fleetd #359 — a stale label used to pile up, not just mislead.</strong> Finding "not
|
||||
* live" here used to mean only one thing: launch another. The old label was left exactly where it
|
||||
* was, so a daemon that restarted enough times — or hit one herdr read that missed a genuinely
|
||||
* running agent — accumulated one more identically-labelled dead tab per occurrence, and
|
||||
* {@code LeadTabScanner} (before its own #359 fix) reported every one of them as a lead.
|
||||
*
|
||||
* <p><strong>Review finding 1 — closing on one reading is worse than the bug.</strong> The first
|
||||
* version of this fix closed a name's dead tabs the moment a single {@link #countLeads} reading
|
||||
* called them dead. The ticket's own live evidence rules that out: on a real host, {@code
|
||||
* agent.list} was seen reporting "0 live" for a tab that a plain {@code ps} confirmed was running a
|
||||
* real session. Closing on that reading would have destroyed the operator's actual lead — a worse
|
||||
* failure than the extra tab it replaces. So {@link #ensureLeads()} now needs the same dead reading
|
||||
* <em>twice</em>, one restart apart, before it closes anything: the first time a labelled tab reads
|
||||
* dead, it is only flagged ({@link PendingCloseMarker}), left running, untouched; it is closed only
|
||||
* if a <em>later</em>, independently-connected reconcile still finds it dead while the flag is
|
||||
* still there. A transient miss self-heals — the next reconcile sees the agent again and clears the
|
||||
* flag (see {@code toUnflag} below) — so the worst case for a single bad reading is one extra tab
|
||||
* surviving one more restart, never a live session destroyed. Two alternatives were considered and
|
||||
* rejected: corroborating {@code agent.list} against a second, truly independent signal was dropped
|
||||
* because nothing else herdr exposes proves "is a process attached to this pane" any better — a
|
||||
* second call to the same unreliable source is not independent evidence; capping the close to "all
|
||||
* but the most recent dead tab" was dropped because "most recent" has no reliable ordering across
|
||||
* tab ids and would leave the true failure mode (a name that is <em>never</em> reconfirmed) growing
|
||||
* by one tab per bad reading forever, which is the exact defect this ticket exists to fix.
|
||||
*/
|
||||
public final class LeadLauncher {
|
||||
|
||||
@@ -79,9 +107,9 @@ public final class LeadLauncher {
|
||||
return 0;
|
||||
}
|
||||
|
||||
Map<String, Integer> live;
|
||||
Map<String, LeadCount> live;
|
||||
try {
|
||||
live = liveLeads(leaders);
|
||||
live = countLeads(leaders);
|
||||
} catch (HerdrException e) {
|
||||
// Counting is the whole safety mechanism against double-spawning. If we cannot count, we
|
||||
// must not guess — spawning a second orchestrator is worse than starting none.
|
||||
@@ -93,9 +121,53 @@ public final class LeadLauncher {
|
||||
for (Map.Entry<String, FleetConfig.Leader> e : leaders.entrySet()) {
|
||||
String name = e.getKey();
|
||||
FleetConfig.Leader lead = e.getValue();
|
||||
int running = live.getOrDefault(name, 0);
|
||||
LeadCount state = live.getOrDefault(name, LeadCount.NONE);
|
||||
int running = state.running();
|
||||
int wanted = lead.instances();
|
||||
|
||||
// fleetd #359 review finding 1: a tab already flagged pending-close, still labelled for
|
||||
// this lead, and STILL hosting no agent on this separate reconcile — two independent
|
||||
// readings agree, so close it. A tab found dead for the first time is only flagged below,
|
||||
// never closed on the spot.
|
||||
for (String tabId : state.toClose()) {
|
||||
log.info("lead '{}': closing tab {} — flagged pending-close on a previous reconcile "
|
||||
+ "and still no agent running in it", name, tabId);
|
||||
try {
|
||||
spaces.closeTab(tabId);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("could not close stale tab {} for lead '{}': {}",
|
||||
tabId, name, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
// A tab labelled for this lead, with no agent running in it, seen dead for the first
|
||||
// time — flag it rather than closing it. One reading of `agent.list` is not enough
|
||||
// evidence to destroy a tab that might genuinely be live (see the class javadoc).
|
||||
for (String tabId : state.toFlag()) {
|
||||
String flagged = lead.tabLabel() + PendingCloseMarker.SUFFIX;
|
||||
log.info("lead '{}': tab {} has no agent running in it this reconcile — flagging it "
|
||||
+ "'{}' rather than closing; it is only closed if a later reconcile still "
|
||||
+ "finds it dead", name, tabId, flagged);
|
||||
try {
|
||||
spaces.renameTab(tabId, flagged);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("could not flag stale tab {} for lead '{}': {}",
|
||||
tabId, name, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
// A previously-flagged tab that is running an agent again — the miss that flagged it was
|
||||
// transient. Clear the flag so a future, unrelated miss starts its own two-reading count
|
||||
// rather than closing on the strength of this one's already-spent flag.
|
||||
for (String tabId : state.toUnflag()) {
|
||||
log.info("lead '{}': tab {} is running an agent again — clearing its pending-close flag",
|
||||
name, tabId);
|
||||
try {
|
||||
spaces.renameTab(tabId, lead.tabLabel());
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("could not clear the pending-close flag on tab {} for lead '{}': {}",
|
||||
tabId, name, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
if (running >= wanted) {
|
||||
log.info("lead '{}': {} live, {} wanted — nothing to start", name, running, wanted);
|
||||
continue;
|
||||
@@ -126,18 +198,29 @@ public final class LeadLauncher {
|
||||
}
|
||||
|
||||
/**
|
||||
* How many live leads exist per configured name: a running agent in a tab labelled with that
|
||||
* lead's exact {@code tab} (CB-579). Member workspaces are excluded, exactly as the scanner
|
||||
* excludes them: a member must not be counted as a lead because it happens to sit in a matching
|
||||
* tab.
|
||||
* How many live leads exist per configured name, and which of that name's labelled tabs are
|
||||
* <em>not</em> live: a running agent in a tab labelled with that lead's exact {@code tab}
|
||||
* (CB-579). Member workspaces are excluded, exactly as the scanner excludes them: a member must
|
||||
* not be counted as a lead because it happens to sit in a matching tab.
|
||||
*
|
||||
* <p>There used to be a second path here — a running agent on the terminal a
|
||||
* {@code fleet.leaders.<name>.terminal} pin named, for a lead opened and pinned by hand. That
|
||||
* pin is retired: {@code tab} is now the only field identity depends on, and {@link Agent}
|
||||
* already carries {@link Agent#tabId()} directly, so a hand-opened lead is found the same way an
|
||||
* auto-launched one is — by labelling its tab to match.
|
||||
*
|
||||
* <p>fleetd #359 review finding 1: a labelled tab with nothing running in it is split into
|
||||
* {@code toClose} (already flagged pending-close by a previous reconcile, and still dead — two
|
||||
* independent readings agree) and {@code toFlag} (dead for the first time — not enough evidence
|
||||
* to close yet). {@code toUnflag} is the reverse: a tab flagged pending-close that is running an
|
||||
* agent again, so the flag it carries no longer means anything and {@link #ensureLeads()} clears
|
||||
* it.
|
||||
*/
|
||||
private Map<String, Integer> liveLeads(Map<String, FleetConfig.Leader> leaders) {
|
||||
private record LeadCount(int running, List<String> toClose, List<String> toFlag, List<String> toUnflag) {
|
||||
static final LeadCount NONE = new LeadCount(0, List.of(), List.of(), List.of());
|
||||
}
|
||||
|
||||
private Map<String, LeadCount> countLeads(Map<String, FleetConfig.Leader> leaders) {
|
||||
// A lead and the members share ONE workspace now (the operator asked for a single "session"
|
||||
// with many tabs), so a workspace can no longer be excluded wholesale — the lead lives in the
|
||||
// member workspace by design. The sole discriminator is the exact tab label: a lead carries
|
||||
@@ -145,6 +228,7 @@ public final class LeadLauncher {
|
||||
// profile's `worker: {profile} #{n}` template. These never collide, so an exact-label match
|
||||
// separates them without needing to know which workspace anyone is in.
|
||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
||||
Set<String> flaggedTabIds = new LinkedHashSet<>();
|
||||
for (Workspace ws : spaces.listWorkspaces()) {
|
||||
if (ws.workspaceId() == null) {
|
||||
continue;
|
||||
@@ -153,31 +237,65 @@ public final class LeadLauncher {
|
||||
String declared = leadNameOf(tab.label(), leaders);
|
||||
if (declared != null && tab.tabId() != null) {
|
||||
nameByTab.put(tab.tabId(), declared);
|
||||
if (PendingCloseMarker.isFlagged(tab.label())) {
|
||||
flaggedTabIds.add(tab.tabId());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Set<String> liveTabIds = new LinkedHashSet<>();
|
||||
Map<String, Integer> counts = new LinkedHashMap<>();
|
||||
for (Agent a : agents.list()) {
|
||||
String name = nameByTab.get(a.tabId());
|
||||
if (name != null) {
|
||||
counts.merge(name, 1, Integer::sum);
|
||||
liveTabIds.add(a.tabId());
|
||||
}
|
||||
}
|
||||
return counts;
|
||||
|
||||
Map<String, List<String>> toCloseByName = new LinkedHashMap<>();
|
||||
Map<String, List<String>> toFlagByName = new LinkedHashMap<>();
|
||||
Map<String, List<String>> toUnflagByName = new LinkedHashMap<>();
|
||||
nameByTab.forEach((tabId, name) -> {
|
||||
boolean live = liveTabIds.contains(tabId);
|
||||
boolean flagged = flaggedTabIds.contains(tabId);
|
||||
if (live) {
|
||||
if (flagged) {
|
||||
toUnflagByName.computeIfAbsent(name, k -> new ArrayList<>()).add(tabId);
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (flagged) {
|
||||
toCloseByName.computeIfAbsent(name, k -> new ArrayList<>()).add(tabId);
|
||||
} else {
|
||||
toFlagByName.computeIfAbsent(name, k -> new ArrayList<>()).add(tabId);
|
||||
}
|
||||
});
|
||||
|
||||
Map<String, LeadCount> out = new LinkedHashMap<>();
|
||||
for (String name : leaders.keySet()) {
|
||||
out.put(name, new LeadCount(counts.getOrDefault(name, 0),
|
||||
toCloseByName.getOrDefault(name, List.of()),
|
||||
toFlagByName.getOrDefault(name, List.of()),
|
||||
toUnflagByName.getOrDefault(name, List.of())));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* The configured lead a tab label names, or {@code null} for a label that names none.
|
||||
*
|
||||
* <p>Matched exactly (case-insensitively) against each lead's configured {@code tab}, so an
|
||||
* operator's {@code "lead: something-else"} tab is not mistaken for a configured lead.
|
||||
* operator's {@code "lead: something-else"} tab is not mistaken for a configured lead. A
|
||||
* trailing {@link PendingCloseMarker} is stripped first, so a tab this class flagged on a
|
||||
* previous reconcile is still recognised as the same lead's tab on this one.
|
||||
*/
|
||||
private String leadNameOf(String label, Map<String, FleetConfig.Leader> leaders) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
String l = label.strip();
|
||||
String l = PendingCloseMarker.strip(label);
|
||||
for (Map.Entry<String, FleetConfig.Leader> e : leaders.entrySet()) {
|
||||
String tab = e.getValue().tabLabel();
|
||||
if (tab != null && l.equalsIgnoreCase(tab.strip())) {
|
||||
|
||||
@@ -0,0 +1,588 @@
|
||||
package dev.ltms.fleet.lead;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* fleetd #480: replace a lead session that has decided it is ready to be rolled over, without an
|
||||
* operator doing it by hand. A lead writes a handover file, calls {@link #open}, and then — once
|
||||
* every gate ({@link #confirm}'s own checks) has passed — a deferred, single-shot continuation
|
||||
* clears the lead's own pane and bootstraps a fresh session against that file.
|
||||
*
|
||||
* <p>This is the executor only. Nothing in this ticket wires an MCP tool onto {@link #open}/
|
||||
* {@link #confirm}/{@link #cancel} — that is a separate, later unit; until it lands, nothing calls
|
||||
* this class at all.
|
||||
*
|
||||
* <p><strong>{@code confirm()} cannot roll inline — a fleetd #480 correction.</strong> The first
|
||||
* version of this class called {@code agents.send(lead, "/clear")} directly from inside {@code
|
||||
* confirm()}, then polled for the pane to become injectable again. That is wrong, because {@code
|
||||
* confirm()} is called BY the lead, FROM the lead's own turn: the lead's pane is {@code WORKING}
|
||||
* for the whole duration of that call and cannot possibly report injectable until {@code confirm()}
|
||||
* itself returns. The poll always timed out — but only after the {@code /clear} had already been
|
||||
* sent and queued in the pane, where it fired the instant the turn ended anyway. The result was the
|
||||
* worst outcome this feature can produce: a silently destroyed lead context with no fresh session
|
||||
* ever started, and a refusal return value that claimed nothing had happened.
|
||||
*
|
||||
* <p>The fix: {@link #confirm} validates every gate, then does no I/O against the lead's own pane
|
||||
* at all — it only records that the request is approved and hands a one-shot continuation to
|
||||
* {@code continuationRunner} before returning. That continuation is what actually touches the pane,
|
||||
* once the calling turn has ended, in this order:
|
||||
* <ol>
|
||||
* <li>wait for the lead's own pane to report a real turn boundary — {@code IDLE} or {@code
|
||||
* DONE}, never merely {@code BLOCKED} — i.e. wait for the very {@code confirm()} call that
|
||||
* approved this roll to finish its turn — bounded by {@code turnSettleSeconds}. <strong>If
|
||||
* this never happens, nothing else in this list runs: no {@code /clear} is ever sent.</strong>
|
||||
* A lead that never goes idle is a lead still doing real work, and clearing it would throw
|
||||
* away live context — exactly the failure this correction exists to prevent.</li>
|
||||
* <li>{@code agents.send(lead, "/clear")}</li>
|
||||
* <li>wait for {@code /clear} to be picked up and settle, bounded by {@code clearSettleSeconds}
|
||||
* (fleetd #489: no longer a plain re-check of the same boundary — {@code /clear} starts no
|
||||
* turn of its own, so this instead nudges the submit keystroke while no pickup has been seen,
|
||||
* then waits for a real {@code WORKING} → {@code IDLE}/{@code DONE} boundary once one has;
|
||||
* see {@link #waitForClearPickupAndSettle})</li>
|
||||
* <li>{@code agents.send(lead, cfg.bootstrapTextFor(p.handoverPath()))}</li>
|
||||
* </ol>
|
||||
* A {@link #confirm} that returns {@link RollDecision#approved()} therefore means <em>"every gate
|
||||
* passed and the roll is scheduled"</em>, never <em>"the pane has been cleared"</em> — the pane may
|
||||
* still be mid-turn, possibly for a long time, when the caller gets that answer back.
|
||||
*
|
||||
* <p><strong>The safety invariant survives this change, restated precisely.</strong> The ticket
|
||||
* that first defined this class required "no timer, no scheduler, no background thread" so that
|
||||
* nothing but an explicit {@link #confirm} call could ever cause a {@code /clear}. That invariant
|
||||
* is about INITIATIVE, not about synchronicity, and this correction keeps it: {@code
|
||||
* continuationRunner} launches a single-shot task that exists only because one specific,
|
||||
* already-approved {@link #confirm} call created it — it is not recurring, it is not started at
|
||||
* construction time or on any schedule, and no two invocations of it ever share state. A recurring
|
||||
* heartbeat or timer that could decide on its own initiative to roll a pane is still, and will
|
||||
* always be, absent from this class. <strong>Nothing but an explicit {@link #confirm} call that
|
||||
* passes every gate can ever cause a {@code /clear} — that call may simply finish its own work
|
||||
* slightly later than the method return, as a continuation of the same approved request, rather
|
||||
* than entirely inside the method body.</strong>
|
||||
*
|
||||
* <p><strong>Identity is resolved by the caller, never looked up here — a second fleetd #480
|
||||
* correction.</strong> The first version resolved the pane to clear via {@code
|
||||
* PrimaryRegistry#primaryTerminal()}. That is correct for a background loop with no caller (see
|
||||
* {@code dev.ltms.fleet.msg.LeadHeartbeatLoop}), but wrong here and a violation of this project's
|
||||
* own charter invariant 3 — "identity comes from the connection, never an argument." This daemon
|
||||
* can hold more than one labelled lead tab (see {@code LeadLauncher}'s fleetd #359 two-reading
|
||||
* dead-tab cleanup), so a single-slot lookup lets lead X's {@link #confirm} clear lead Y's pane: an
|
||||
* unrecoverable loss of someone else's context, and a different lead's context at that. Both
|
||||
* {@link #open} and {@link #confirm} now take the lead's terminal id as a parameter instead —
|
||||
* {@link #open} stores it on the {@link PendingRollover}, and {@link #confirm} refuses with {@link
|
||||
* RefusalReason#NOT_YOUR_ROLLOVER} unless the caller's terminal matches the one {@link #open}
|
||||
* recorded. <strong>The terminal id passed to both methods must come from the MCP layer's own
|
||||
* connection-based caller resolution — the same source {@code auth/CallerResolver#resolve} uses to
|
||||
* build a {@code Principal.leader(...)} (see its use of {@code ConnectionIdentity.Caller#terminal})
|
||||
* — never a value the client supplies or chooses.</strong> The later MCP-tool unit that wires
|
||||
* {@link #open}/{@link #confirm} must pass the resolved caller terminal, not a request field.
|
||||
*
|
||||
* <p><strong>A relative {@code handoverPath} resolves against the CALLING lead's workspace, never
|
||||
* the daemon's own cwd — a fleetd #480 follow-up.</strong> The daemon and a lead's own pane can
|
||||
* have different working directories (this repo nests {@code fleetd/} inside its own root, so the
|
||||
* daemon's cwd and {@code fleet.leaders.<name>.cwd} already differ on this host). {@link #open}
|
||||
* resolves {@code cfg.handoverPath()} to an ABSOLUTE path exactly once — against {@code
|
||||
* leadWorkspace.apply(leadTerminal)} when that lookup returns a non-null, non-blank workspace, and
|
||||
* against {@code System.getProperty("user.dir")} otherwise (the same fallback {@code
|
||||
* LeadLauncher#launch} already uses for a lead with no configured {@code cwd}) — and stores only
|
||||
* that absolute path on {@link PendingRollover}. Every later read of {@code
|
||||
* PendingRollover#handoverPath()} (the freshness/exists/empty checks in {@link #checkHandover},
|
||||
* the value {@code FleetMcp} hands back to the lead in the {@code open} response so it knows where
|
||||
* to WRITE the file, and {@link FleetConfig.LeadRollover#bootstrapTextFor} which names it in the
|
||||
* text typed into the fresh session) therefore already sees the resolved absolute form and never
|
||||
* needs to resolve anything itself.
|
||||
*/
|
||||
public final class LeadRollover {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadRollover.class);
|
||||
|
||||
/** Poll interval while waiting for the lead's pane to settle after {@code /clear}. */
|
||||
static final long SETTLE_POLL_MS = 250;
|
||||
|
||||
/**
|
||||
* How many consecutive not-yet-picked-up polls {@link #waitForClearPickupAndSettle} allows
|
||||
* before releasing rather than wedging the roll — the same constant and the same
|
||||
* release-not-wedge choice {@link dev.ltms.fleet.inject.Injector} already makes for its own
|
||||
* post-turn {@code /clear} housekeeping (fleetd #306). <strong>This bounds the number of
|
||||
* consecutive polls, not the number of nudges:</strong> the first {@code PICKUP_GRACE_POLLS - 1}
|
||||
* of those polls each send a nudge, and the {@code PICKUP_GRACE_POLLS}th releases instead of
|
||||
* nudging again — so 8 polls produce 7 nudges, not 8.
|
||||
*/
|
||||
static final int PICKUP_GRACE_POLLS = 8;
|
||||
|
||||
/**
|
||||
* One request opened by {@link #open}, pending its {@link #confirm} (or {@link #cancel}).
|
||||
*
|
||||
* @param leadTerminal the lead pane that opened this request — the only terminal that may
|
||||
* later {@link #confirm} it (see {@link RefusalReason#NOT_YOUR_ROLLOVER})
|
||||
* @param handoverPath the ABSOLUTE, resolved handover path — never the raw configured value,
|
||||
* which may have been relative. {@link #open} resolves it once, against the
|
||||
* calling lead's workspace, before storing it here; see this class's
|
||||
* javadoc. This is the value the MCP layer hands back to the lead as
|
||||
* "write your file here", so callers may rely on it always being absolute.
|
||||
*/
|
||||
public record PendingRollover(String token, String leadTerminal, String handoverPath,
|
||||
long requestedAtMillis) {}
|
||||
|
||||
/** Which check refused a {@link #confirm} call, named so a caller can act on it. */
|
||||
public enum RefusalReason {
|
||||
/** {@code leadRollover:} is not configured — absent at construction, or removed since. */
|
||||
NOT_CONFIGURED,
|
||||
/** {@code token} names no pending request: never opened, already confirmed, or cancelled. */
|
||||
UNKNOWN_TOKEN,
|
||||
/**
|
||||
* The caller's terminal does not match the terminal that {@link #open} recorded for this
|
||||
* token. Only the lead that opened a request may confirm it (fleetd #480 correction 2).
|
||||
*/
|
||||
NOT_YOUR_ROLLOVER,
|
||||
/** {@code requireOperatorConfirm: true} and the caller passed {@code operatorConfirmed: false}. */
|
||||
OPERATOR_NOT_CONFIRMED,
|
||||
/** The handover file does not exist. */
|
||||
HANDOVER_MISSING,
|
||||
/** The handover file exists but is empty. */
|
||||
HANDOVER_EMPTY,
|
||||
/**
|
||||
* The handover file's modified time is not after {@link #open}'s request timestamp, or is
|
||||
* older than {@code maxDocAgeSeconds}.
|
||||
*/
|
||||
HANDOVER_STALE
|
||||
}
|
||||
|
||||
/**
|
||||
* The outcome of a {@link #confirm} call. {@link #approved()} means every gate passed and the
|
||||
* roll has been handed to a one-shot continuation — <strong>not</strong> that the pane has been
|
||||
* cleared; the continuation may still be waiting for the calling turn to end when this returns.
|
||||
* Whether the deferred roll itself later goes on to clear the pane, refuse for never going
|
||||
* idle, or refuse for never re-settling after {@code /clear} is logged only (see this class's
|
||||
* javadoc) — there is deliberately no synchronous caller left by that point to hand a result to.
|
||||
*/
|
||||
public record RollDecision(boolean accepted, RefusalReason reason, String detail) {
|
||||
static RollDecision approved() {
|
||||
return new RollDecision(true, null, "confirmed; the roll will run once the calling turn ends");
|
||||
}
|
||||
|
||||
static RollDecision refused(RefusalReason reason, String detail) {
|
||||
return new RollDecision(false, reason, detail);
|
||||
}
|
||||
}
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<FleetConfig.LeadRollover> configSupplier;
|
||||
/**
|
||||
* Terminal id → that lead's configured workspace directory (their {@code
|
||||
* fleet.leaders.<name>.cwd}), or {@code null} when the terminal names no currently-recognised
|
||||
* lead. {@link #open} calls this to resolve a relative {@code handoverPath} — see this class's
|
||||
* javadoc. Required: there is no sane default that would not silently reintroduce the
|
||||
* daemon-cwd bug this parameter exists to fix.
|
||||
*/
|
||||
private final Function<String, String> leadWorkspace;
|
||||
private final LongSupplier nowMillis;
|
||||
private final Runnable settleSleeper;
|
||||
/**
|
||||
* Launches the post-{@code confirm()} continuation. Production uses a single unstarted virtual
|
||||
* thread per confirmed request — see this class's javadoc for why that is a single-shot task,
|
||||
* not a background scheduler. Tests inject {@code Runnable::run} to make the continuation run
|
||||
* synchronously and deterministically on the calling thread.
|
||||
*/
|
||||
private final Consumer<Runnable> continuationRunner;
|
||||
private final Map<String, PendingRollover> pending = new ConcurrentHashMap<>();
|
||||
|
||||
/** Production constructor — wall clock, real sleep between settle polls, a real virtual thread. */
|
||||
public LeadRollover(AgentControl agents, Supplier<FleetConfig.LeadRollover> configSupplier,
|
||||
Function<String, String> leadWorkspace) {
|
||||
this(agents, configSupplier, leadWorkspace, System::currentTimeMillis,
|
||||
() -> sleepUninterruptibly(SETTLE_POLL_MS),
|
||||
r -> Thread.ofVirtual().name("lead-rollover-continuation-").start(r));
|
||||
}
|
||||
|
||||
/**
|
||||
* Full constructor — an injectable wall-clock supplier, settle-poll sleeper, and continuation
|
||||
* runner, for tests. {@code nowMillis} MUST be a wall-clock source (e.g. {@code
|
||||
* System.currentTimeMillis()}), never {@code System.nanoTime()}: the freshness check compares
|
||||
* against a file's modified time, which only a wall clock is comparable to, and {@code
|
||||
* nanoTime} freezes while the host sleeps (fleetd #386).
|
||||
*/
|
||||
LeadRollover(AgentControl agents, Supplier<FleetConfig.LeadRollover> configSupplier,
|
||||
Function<String, String> leadWorkspace, LongSupplier nowMillis,
|
||||
Runnable settleSleeper, Consumer<Runnable> continuationRunner) {
|
||||
this.agents = agents;
|
||||
this.configSupplier = configSupplier;
|
||||
this.leadWorkspace = leadWorkspace;
|
||||
this.nowMillis = nowMillis;
|
||||
this.settleSleeper = settleSleeper;
|
||||
this.continuationRunner = continuationRunner;
|
||||
}
|
||||
|
||||
private static void sleepUninterruptibly(long ms) {
|
||||
try {
|
||||
Thread.sleep(ms);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
// preserve the interrupt flag but continue — this poll loop should not be aborted by an
|
||||
// interrupt that was not meant for it.
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The lead says it is ready to be replaced. Generates a token and records the resolved
|
||||
* handover path, this moment's wall-clock timestamp (the baseline {@link #confirm} checks the
|
||||
* handover file's modified time against), and {@code leadTerminal} — only that exact terminal
|
||||
* may later {@link #confirm} this token.
|
||||
*
|
||||
* @param leadTerminal the calling lead's terminal id, resolved by the MCP layer from the
|
||||
* connection (see this class's javadoc) — never a client-supplied value
|
||||
* @param reason free-text audit note (logged only; not otherwise interpreted or stored)
|
||||
* @throws IllegalStateException if {@code leadRollover:} is not configured
|
||||
* @throws IllegalArgumentException if {@code leadTerminal} is null or blank
|
||||
*/
|
||||
public PendingRollover open(String leadTerminal, String reason) {
|
||||
FleetConfig.LeadRollover cfg = configSupplier.get();
|
||||
if (cfg == null) {
|
||||
throw new IllegalStateException("leadRollover: is not configured");
|
||||
}
|
||||
if (leadTerminal == null || leadTerminal.isBlank()) {
|
||||
throw new IllegalArgumentException("leadTerminal is required — it must be resolved from "
|
||||
+ "the caller's connection, never accepted as a client-chosen argument");
|
||||
}
|
||||
String token = UUID.randomUUID().toString();
|
||||
long requestedAt = nowMillis.getAsLong();
|
||||
String resolvedPath = resolveHandoverPath(cfg.handoverPath(), leadTerminal);
|
||||
PendingRollover p = new PendingRollover(token, leadTerminal, resolvedPath, requestedAt);
|
||||
pending.put(token, p);
|
||||
if (resolvedPath.equals(cfg.handoverPath())) {
|
||||
log.info("lead-rollover: open token={} lead={} handoverPath={} reason={}",
|
||||
token, leadTerminal, resolvedPath, reason);
|
||||
} else {
|
||||
// The configured value was relative (or merely un-normalized) and resolved to a
|
||||
// different string — log both, so an operator reading this line can see which
|
||||
// directory the daemon actually looked in, not just the value it was given.
|
||||
log.info("lead-rollover: open token={} lead={} configuredHandoverPath={} "
|
||||
+ "resolvedHandoverPath={} reason={}",
|
||||
token, leadTerminal, cfg.handoverPath(), resolvedPath, reason);
|
||||
}
|
||||
return p;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve {@code configured} to an absolute path exactly once, here, so nothing downstream
|
||||
* ({@link #checkHandover}, the MCP layer's {@code open} response, {@link
|
||||
* FleetConfig.LeadRollover#bootstrapTextFor}) ever has to resolve — or worse, silently
|
||||
* mis-resolve — a relative path again.
|
||||
*
|
||||
* <p><strong>The return value is GUARANTEED absolute, not merely usually absolute.</strong>
|
||||
* {@code leadWorkspace.apply(leadTerminal)} returns an operator-configured {@code
|
||||
* fleet.leaders.<name>.cwd} string, and nothing forces an operator to write an absolute one —
|
||||
* a relative {@code cwd} resolved with plain {@link Path#resolve} would still yield a relative
|
||||
* result, silently reopening the exact bug this class exists to fix (every later reader back to
|
||||
* interpreting an ambiguous string against ITS OWN working directory). {@link
|
||||
* Path#toAbsolutePath()} closes that: it resolves any remaining relative path against {@code
|
||||
* user.dir} (the JVM's own cwd), which is the correct base for an operator-written path the
|
||||
* daemon process itself is meant to interpret, exactly like the {@code user.dir} fallback used
|
||||
* below. Applying it unconditionally on both branches means the ALREADY-absolute branch stays a
|
||||
* no-op (an absolute path is unaffected by {@code toAbsolutePath()}) while the relative-{@code
|
||||
* cwd} branch above is closed the same way.
|
||||
*
|
||||
* <ul>
|
||||
* <li>already absolute → returned unchanged (normalized)</li>
|
||||
* <li>relative → resolved against {@code leadWorkspace.apply(leadTerminal)} when that is
|
||||
* non-null and non-blank; otherwise against {@code System.getProperty("user.dir")} — the
|
||||
* same fallback {@code LeadLauncher#launch} uses for a lead with no configured {@code
|
||||
* cwd}. If {@code leadWorkspace}'s own answer is itself relative (an operator wrote a
|
||||
* relative {@code cwd:}), the result is finished off against the daemon's own
|
||||
* {@code user.dir} — see the paragraph above.</li>
|
||||
* </ul>
|
||||
*/
|
||||
private String resolveHandoverPath(String configured, String leadTerminal) {
|
||||
Path path = Path.of(configured);
|
||||
if (path.isAbsolute()) {
|
||||
// toAbsolutePath() is a no-op for an already-absolute path — kept here anyway so both
|
||||
// branches call the exact same guarantee, rather than one branch relying on
|
||||
// isAbsolute() alone to already imply what toAbsolutePath() enforces.
|
||||
return path.toAbsolutePath().normalize().toString();
|
||||
}
|
||||
String workspace = leadWorkspace.apply(leadTerminal);
|
||||
Path base = (workspace == null || workspace.isBlank())
|
||||
? Path.of(System.getProperty("user.dir"))
|
||||
: Path.of(workspace);
|
||||
return base.resolve(path).toAbsolutePath().normalize().toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate every gate, then — if and only if all of them pass — hand a one-shot continuation
|
||||
* that performs the actual roll to {@code continuationRunner} and return. <strong>This method
|
||||
* never itself sends anything to the lead's pane</strong> — see this class's javadoc for why
|
||||
* (it is called FROM the lead's own turn, so the pane cannot possibly be injectable yet).
|
||||
*
|
||||
* <p>Order: token lookup, then ownership ({@code callerTerminal} must match the terminal
|
||||
* {@link #open} recorded — {@link RefusalReason#NOT_YOUR_ROLLOVER}), then the
|
||||
* operator-confirmation gate, then the three handover-file checks (exists, not empty, fresh —
|
||||
* see {@link #checkHandover}). The first failing check is returned and {@code token} stays
|
||||
* pending (so a caller can fix the problem — e.g. rewrite the handover file — and retry with
|
||||
* the same token); it is consumed only once every gate passes and the continuation is launched.
|
||||
*
|
||||
* @param callerTerminal the CALLING lead's terminal id, resolved by the MCP layer from the
|
||||
* connection — never a client-supplied value (see this class's javadoc)
|
||||
* @param token the token {@link #open} returned
|
||||
* @param operatorConfirmed the caller's answer to "has an operator confirmed this roll" —
|
||||
* consulted only when the live config's {@code requireOperatorConfirm}
|
||||
* is true
|
||||
*/
|
||||
public RollDecision confirm(String callerTerminal, String token, boolean operatorConfirmed) {
|
||||
FleetConfig.LeadRollover cfg = configSupplier.get();
|
||||
if (cfg == null) {
|
||||
return RollDecision.refused(RefusalReason.NOT_CONFIGURED, "leadRollover: is not configured");
|
||||
}
|
||||
PendingRollover p = pending.get(token);
|
||||
if (p == null) {
|
||||
return RollDecision.refused(RefusalReason.UNKNOWN_TOKEN,
|
||||
"token " + token + " names no pending rollover request");
|
||||
}
|
||||
if (!p.leadTerminal().equals(callerTerminal)) {
|
||||
return RollDecision.refused(RefusalReason.NOT_YOUR_ROLLOVER,
|
||||
"token " + token + " was opened by a different lead terminal");
|
||||
}
|
||||
if (cfg.requireOperatorConfirm() && !operatorConfirmed) {
|
||||
return RollDecision.refused(RefusalReason.OPERATOR_NOT_CONFIRMED,
|
||||
"requireOperatorConfirm is true and operatorConfirmed was false");
|
||||
}
|
||||
|
||||
RollDecision docCheck = checkHandover(p, cfg);
|
||||
if (docCheck != null) {
|
||||
return docCheck;
|
||||
}
|
||||
|
||||
pending.remove(token);
|
||||
log.info("lead-rollover: confirmed token={} lead={} — roll scheduled once the calling turn ends",
|
||||
token, callerTerminal);
|
||||
continuationRunner.accept(() -> runRollover(p, cfg));
|
||||
return RollDecision.approved();
|
||||
}
|
||||
|
||||
/**
|
||||
* The single-shot continuation {@link #confirm} hands to {@code continuationRunner}. Runs
|
||||
* entirely after {@link #confirm} has returned to its caller — see this class's javadoc for the
|
||||
* four-step order. There is no result to return to by this point, so every outcome is logged
|
||||
* only.
|
||||
*/
|
||||
private void runRollover(PendingRollover p, FleetConfig.LeadRollover cfg) {
|
||||
String lead = p.leadTerminal();
|
||||
boolean turnSettled = waitUntilAtTurnBoundary(lead, cfg.turnSettleSeconds());
|
||||
if (!turnSettled) {
|
||||
log.warn("lead-rollover: pane {} did not reach a turn boundary (IDLE or DONE) within {}s "
|
||||
+ "after confirm() — refusing to send /clear at all; the calling lead's "
|
||||
+ "own turn is still live and clearing it now would destroy live context "
|
||||
+ "(token={})",
|
||||
lead, cfg.turnSettleSeconds(), p.token());
|
||||
return;
|
||||
}
|
||||
|
||||
// This deliberately bypasses Injector, exactly like ClaudeCodeLauncher#clearContext:
|
||||
// /clear is housekeeping, not a delegated turn, and routing it through Injector wedges the
|
||||
// pane forever (see this class's javadoc).
|
||||
agents.send(lead, "/clear");
|
||||
boolean clearSettled = waitForClearPickupAndSettle(lead, cfg.clearSettleSeconds());
|
||||
if (!clearSettled) {
|
||||
log.warn("lead-rollover: pane {} did not reach a turn boundary (IDLE or DONE) within {}s "
|
||||
+ "after /clear — NOT sending bootstrapText (token={})",
|
||||
lead, cfg.clearSettleSeconds(), p.token());
|
||||
return;
|
||||
}
|
||||
agents.send(lead, cfg.bootstrapTextFor(p.handoverPath()));
|
||||
log.info("lead-rollover: rolled token={} lead={}", p.token(), lead);
|
||||
}
|
||||
|
||||
/** Drop a pending request without rolling. @return whether a pending request existed for {@code token} */
|
||||
public boolean cancel(String token) {
|
||||
return pending.remove(token) != null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The three handover-file checks, in order: exists, not empty, fresh (modified after
|
||||
* {@link #open}'s timestamp and not older than {@code maxDocAgeSeconds}). Stats {@code
|
||||
* p.handoverPath()} directly — {@link #open} already resolved it to an absolute path, so this
|
||||
* never has to guess which directory it means.
|
||||
*
|
||||
* @return the first failing check's refusal, or {@code null} when all three pass
|
||||
*/
|
||||
private RollDecision checkHandover(PendingRollover p, FleetConfig.LeadRollover cfg) {
|
||||
Path path = Path.of(p.handoverPath());
|
||||
if (!Files.exists(path)) {
|
||||
return RollDecision.refused(RefusalReason.HANDOVER_MISSING,
|
||||
"handover file " + p.handoverPath() + " does not exist");
|
||||
}
|
||||
long size;
|
||||
long mtimeMillis;
|
||||
try {
|
||||
size = Files.size(path);
|
||||
mtimeMillis = Files.getLastModifiedTime(path).toMillis();
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("failed to stat handover file " + p.handoverPath(), e);
|
||||
}
|
||||
if (size == 0) {
|
||||
return RollDecision.refused(RefusalReason.HANDOVER_EMPTY,
|
||||
"handover file " + p.handoverPath() + " is empty");
|
||||
}
|
||||
if (mtimeMillis <= p.requestedAtMillis()) {
|
||||
return RollDecision.refused(RefusalReason.HANDOVER_STALE,
|
||||
"handover file " + p.handoverPath() + " was not modified after the open() "
|
||||
+ "request (mtime=" + mtimeMillis + "ms, requestedAt=" + p.requestedAtMillis() + "ms)");
|
||||
}
|
||||
long ageMillis = nowMillis.getAsLong() - mtimeMillis;
|
||||
long maxAgeMillis = TimeUnit.SECONDS.toMillis(cfg.maxDocAgeSeconds());
|
||||
if (ageMillis > maxAgeMillis) {
|
||||
return RollDecision.refused(RefusalReason.HANDOVER_STALE,
|
||||
"handover file " + p.handoverPath() + " is " + TimeUnit.MILLISECONDS.toSeconds(ageMillis)
|
||||
+ "s old, older than maxDocAgeSeconds=" + cfg.maxDocAgeSeconds());
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll {@link AgentControl#status} until {@code target} reports a real turn boundary — {@link
|
||||
* AgentStatus#IDLE} or {@link AgentStatus#DONE} — bounded by {@code settleSeconds}. Used once by
|
||||
* {@link #runRollover}, to wait for the CALLING turn's own pane to settle before {@code /clear}
|
||||
* is ever sent at all — the {@code turnSettleSeconds} gate that makes this correction safe. The
|
||||
* SECOND wait, after {@code /clear}, is {@link #waitForClearPickupAndSettle} instead (fleetd
|
||||
* #489) — a plain boundary check is not enough there, because {@code /clear} starts no turn of
|
||||
* its own, so this method would (wrongly) report "settled" on its very first poll whether or not
|
||||
* {@code /clear} was actually picked up. A failed status read degrades to "not yet settled" and
|
||||
* is retried on the next poll, the same posture {@code LeadHeartbeatLoop} and {@code
|
||||
* HerdrPeerLauncher}'s readiness gate already take toward an unreadable status.
|
||||
*
|
||||
* <p><strong>Deliberately not {@link AgentStatus#injectable()}.</strong> {@code injectable()}
|
||||
* answers the {@code Injector}'s question — "may I deliver a message without stepping on a live
|
||||
* turn" — and it accepts {@link AgentStatus#BLOCKED} for that purpose, because a pane paused on
|
||||
* an approval prompt is safe to queue a message behind. This class asks a stricter question —
|
||||
* "has the turn actually ended" — and {@code BLOCKED} answers no: it is a live turn that is
|
||||
* merely paused, not one that has finished. Reusing {@code injectable()} here would let this
|
||||
* wait fire {@code /clear} while the lead's own {@code confirm()}-calling turn is still live and
|
||||
* paused on a prompt — exactly the live-context-destroying failure the {@code turnSettleSeconds}
|
||||
* gate exists to prevent. Do not "simplify" this back to {@code injectable()}. ({@link
|
||||
* #waitForClearPickupAndSettle} keeps the same exclusion of {@code BLOCKED}, for the same
|
||||
* reason, on the second wait.)
|
||||
*/
|
||||
private boolean waitUntilAtTurnBoundary(String target, int settleSeconds) {
|
||||
long deadline = nowMillis.getAsLong() + TimeUnit.SECONDS.toMillis(settleSeconds);
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
AgentStatus status;
|
||||
try {
|
||||
status = agents.status(target);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("lead-rollover: status check failed while waiting for {} to settle: {}",
|
||||
target, e.toString());
|
||||
status = null;
|
||||
}
|
||||
if (status == AgentStatus.IDLE || status == AgentStatus.DONE) {
|
||||
return true;
|
||||
}
|
||||
settleSleeper.run();
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* The SECOND wait in {@link #runRollover} — after {@code /clear} has been sent, waits for it to
|
||||
* settle, bounded by {@code settleSeconds}. <strong>fleetd #489 — the paste-race fix.</strong>
|
||||
* {@code /clear} does not start a real turn of its own, so a pane with no submit race simply
|
||||
* stays {@link AgentStatus#IDLE} the whole time: {@link #waitUntilAtTurnBoundary} would (wrongly)
|
||||
* call that "settled" on its very first poll, whether or not the {@code /clear} Enter actually
|
||||
* landed. That was Fault 1, measured live on 2026-09-12 — the second gate was a no-op, so a
|
||||
* {@code bootstrapText} send followed immediately, racing Fault 2: {@link AgentControl#submit}'s
|
||||
* own javadoc already records that the submit accompanying a delivery "can race the paste —
|
||||
* especially right as the worker's TUI becomes interactive — leaving the text unsubmitted"
|
||||
* (CB-113). Because {@code runRollover} deliberately bypasses {@code Injector} for {@code
|
||||
* /clear} (see this class's javadoc), it inherited none of {@code Injector}'s nudging — so the
|
||||
* lost {@code /clear} Enter sat in the input box and {@code bootstrapText} was typed right after
|
||||
* it, landing as one concatenated line.
|
||||
*
|
||||
* <p>This method copies the pickup-nudge pattern {@link dev.ltms.fleet.inject.Injector} already
|
||||
* ships for exactly this, on its own post-turn {@code /clear} housekeeping (fleetd #306; see
|
||||
* {@code Injector.java:288-340} and {@code Injector.java:437-442}):
|
||||
* <ul>
|
||||
* <li>an {@link AgentStatus#WORKING} sample means {@code /clear} was picked up as a real
|
||||
* turn;</li>
|
||||
* <li>until that happens, each poll that still reports {@link AgentStatus#IDLE} or {@link
|
||||
* AgentStatus#DONE} re-sends the submit keystroke ({@link AgentControl#submit}) to nudge
|
||||
* the raced Enter — for the first {@code PICKUP_GRACE_POLLS - 1} of {@link
|
||||
* #PICKUP_GRACE_POLLS} consecutive such polls (i.e. {@code PICKUP_GRACE_POLLS - 1}
|
||||
* nudges: 7, not 8, given {@code PICKUP_GRACE_POLLS = 8}). A second Enter on an empty
|
||||
* Claude Code prompt is a no-op, so repeating it is safe;</li>
|
||||
* <li>the {@code PICKUP_GRACE_POLLS}th consecutive such poll, with {@code WORKING} still never
|
||||
* observed, releases rather than wedges the roll instead of nudging again — the same
|
||||
* choice {@code Injector} makes — and returns {@code true} anyway, logged at {@code info}
|
||||
* so an operator can see which path ran;</li>
|
||||
* <li>once {@code WORKING} has been observed, nudging stops and this instead waits for a real
|
||||
* {@code working → IDLE/DONE} completion boundary before returning {@code true}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>{@link AgentStatus#BLOCKED} is deliberately excluded from both the nudge and the
|
||||
* boundary check</strong> — the same reasoning as {@link #waitUntilAtTurnBoundary}'s own
|
||||
* javadoc: a paused live turn is not a settled one, and re-sending Enter into an open approval
|
||||
* prompt could wrongly answer it. A {@code BLOCKED} sample (or an unreadable/{@link
|
||||
* AgentStatus#UNKNOWN} one) simply keeps this polling, with no nudge and no release, until either
|
||||
* a real boundary is reached or {@code settleSeconds} runs out.
|
||||
*
|
||||
* <p>{@link AgentControl#submit} can itself throw; a {@link RuntimeException} from it is
|
||||
* swallowed and logged at {@code debug}, exactly like {@code Injector.java:437-442} — a failed
|
||||
* nudge must not abort the roll.
|
||||
*
|
||||
* @return {@code true} once {@code /clear} has settled, or once the nudge budget was exhausted
|
||||
* with no pickup ever observed (released rather than wedged); {@code false} if {@code
|
||||
* settleSeconds} elapses first — the caller must NOT send {@code bootstrapText} in that
|
||||
* case, exactly as before this fix
|
||||
*/
|
||||
private boolean waitForClearPickupAndSettle(String target, int settleSeconds) {
|
||||
long deadline = nowMillis.getAsLong() + TimeUnit.SECONDS.toMillis(settleSeconds);
|
||||
boolean pickedUp = false; // a WORKING sample has been observed since /clear was sent
|
||||
int idlePollsAwaitingPickup = 0;
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
AgentStatus status;
|
||||
try {
|
||||
status = agents.status(target);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("lead-rollover: status check failed while waiting for {} to settle after "
|
||||
+ "/clear: {}", target, e.toString());
|
||||
status = null;
|
||||
}
|
||||
if (status == AgentStatus.WORKING) {
|
||||
pickedUp = true;
|
||||
} else if (status == AgentStatus.IDLE || status == AgentStatus.DONE) {
|
||||
if (pickedUp) {
|
||||
return true; // a real WORKING -> IDLE/DONE completion boundary
|
||||
}
|
||||
if (++idlePollsAwaitingPickup >= PICKUP_GRACE_POLLS) {
|
||||
log.info("lead-rollover: /clear on {} was never observed as WORKING after {} "
|
||||
+ "consecutive IDLE/DONE polls ({} of those were nudged) — "
|
||||
+ "releasing rather than wedging the roll",
|
||||
target, PICKUP_GRACE_POLLS, PICKUP_GRACE_POLLS - 1);
|
||||
return true;
|
||||
}
|
||||
try {
|
||||
agents.submit(target); // nudge a raced Enter (CB-113) so /clear actually submits
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("lead-rollover: resubmit to {} failed (will retry next poll): {}",
|
||||
target, e.getMessage());
|
||||
}
|
||||
}
|
||||
// AgentStatus.BLOCKED or UNKNOWN (or an unreadable status, above): neither a pickup
|
||||
// signal nor a boundary — keep polling without nudging or releasing.
|
||||
settleSleeper.run();
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,81 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* fleetd #469: a launch charter that names an MCP tool the server does not register must stop the
|
||||
* daemon at startup, not wait for a member to discover the gap by calling something that is not
|
||||
* there.
|
||||
*
|
||||
* <p>Deliberately its own class outside {@code dev.ltms.fleet.config}, not a case in {@link
|
||||
* dev.ltms.fleet.config.FleetConfig#validateCharters()}. The canonical tool surface ({@link
|
||||
* FleetTool}) lives in the {@code mcp} package; config is loaded before the MCP server exists and
|
||||
* must not gain a dependency on it. So this check belongs at the seam that already holds both a
|
||||
* loaded {@code FleetConfig} and the {@code mcp} package: {@code Fleetd.main}, called right after
|
||||
* {@code cfg.validateAll()} and before anything opens a socket or spawns a member.
|
||||
*
|
||||
* <p>{@code #464}'s {@code CharterToolSurfaceTest} proved the same comparison against a charter
|
||||
* fixture it wrote itself into a {@code @TempDir}, which meant nothing anyone wrote into the live
|
||||
* {@code fleetd.yaml} could ever fail it. This class is what a real charter is actually checked
|
||||
* against at boot; {@code FleetdStartupValidationTest} exercises it through {@code Fleetd.main}
|
||||
* itself, the same way it proves every other {@code validateXxx()} still runs there.
|
||||
*
|
||||
* <p><strong>fleetd #474</strong> — startup was not the only door: {@code
|
||||
* dev.ltms.fleet.config.ConfigRef#reload()} used to run {@code FleetConfig#validateAll()} alone,
|
||||
* which does not look at what a charter's text names, so a charter naming an unregistered tool
|
||||
* that could not have booted the daemon could still be installed into a running one through a
|
||||
* reload. This class still knows nothing about {@code ConfigRef} — {@code Fleetd.main} wires a
|
||||
* small {@code FleetConfig -> void} adapter over {@link #assertChartersNameOnlyRegisteredTools}
|
||||
* ({@code Fleetd::assertChartersNameOnlyRegisteredTools}) into {@code ConfigRef}'s constructor as
|
||||
* its {@code Consumer<FleetConfig>} {@code extraValidation}, run inside {@code reload()}'s same
|
||||
* try/catch as {@code validateAll()}, so both call sites — {@code Fleetd.main} at startup and
|
||||
* {@code ConfigRef#reload()} afterwards — go through this one method and can never check different
|
||||
* things. {@code dev.ltms.fleet.config.ConfigRefTest} and {@code
|
||||
* FleetdConfigRefCharterToolSurfaceWiringTest} are what prove the reload call site, the same way
|
||||
* {@code FleetdStartupValidationTest} proves the startup one.
|
||||
*/
|
||||
public final class CharterToolSurface {
|
||||
|
||||
/** A {@code fleet_…} (current) or {@code bridge_…} (pre-CB-634) tool-shaped token in prose. */
|
||||
private static final Pattern TOOL_REFERENCE = Pattern.compile("(fleet_[a-z_]+|bridge_[a-z_]+)");
|
||||
|
||||
private CharterToolSurface() {
|
||||
}
|
||||
|
||||
/**
|
||||
* @param charters the configured {@code fleet.charters:} map (role wire name → charter text);
|
||||
* {@code null} or empty is a no-op, same as an absent {@code fleet:} block
|
||||
* @throws IllegalStateException naming the charter key and every tool it names that {@link
|
||||
* FleetTool} does not list, when any charter does so
|
||||
*/
|
||||
public static void assertChartersNameOnlyRegisteredTools(Map<String, String> charters) {
|
||||
if (charters == null || charters.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
Set<String> registered = FleetTool.wireNames();
|
||||
List<String> bad = new ArrayList<>();
|
||||
charters.forEach((key, text) -> {
|
||||
if (text == null) {
|
||||
return;
|
||||
}
|
||||
Set<String> named = new LinkedHashSet<>();
|
||||
Matcher m = TOOL_REFERENCE.matcher(text);
|
||||
while (m.find()) {
|
||||
named.add(m.group(1));
|
||||
}
|
||||
named.stream()
|
||||
.filter(t -> !registered.contains(t))
|
||||
.forEach(unknown -> bad.add("fleet.charters." + key + " names '" + unknown
|
||||
+ "', which the server does not register (registered: " + registered + ")."));
|
||||
});
|
||||
if (!bad.isEmpty()) {
|
||||
throw new IllegalStateException("refusing to start: " + String.join(" ", bad));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -36,6 +36,25 @@ public final class ConnectionIdentity {
|
||||
* primary / an off-host client) and its {@code pid} (or {@code -1} if not resolvable).
|
||||
*/
|
||||
public record Caller(String terminal, long pid) {
|
||||
|
||||
/**
|
||||
* Whether the OS peer-PID lookup actually succeeded — {@code false} means {@code pid} is
|
||||
* the {@code -1} sentinel, not a real process id, so this caller's identity could not be
|
||||
* established at all. That is a different fact from a real pid that simply owns no worker
|
||||
* pane (the primary's own connection): the primary is {@code resolved()} and has a
|
||||
* {@code null terminal}; an unresolvable caller is {@code !resolved()} and also has a
|
||||
* {@code null terminal}. The two look identical through {@link #terminal} alone, which is
|
||||
* exactly how fleetd #317 happened — a failed {@code lsof} lookup and a genuine primary both
|
||||
* fell through to {@code Principal.primary(...)}.
|
||||
*
|
||||
* <p>Centralised here, next to the sentinel it tests, for the same reason
|
||||
* {@link ConnectionIdentity#isLoopback} is centralised rather than left for each caller to
|
||||
* reimplement: a raw {@code pid > 0} check duplicated at every call site is precisely the
|
||||
* "one rule, two copies" shape that let #305 drift.
|
||||
*/
|
||||
public boolean resolved() {
|
||||
return pid > 0;
|
||||
}
|
||||
}
|
||||
|
||||
/** Resolve the caller's terminal and PID from one peer-PID lookup. */
|
||||
@@ -60,7 +79,44 @@ public final class ConnectionIdentity {
|
||||
return pid > 0 ? cwds.cwdForPid(pid) : null;
|
||||
}
|
||||
|
||||
private static boolean isLoopback(String addr) {
|
||||
return "127.0.0.1".equals(addr) || "::1".equals(addr) || "0:0:0:0:0:0:0:1".equals(addr);
|
||||
/**
|
||||
* Whether {@code addr} is a same-host address, and therefore one whose peer PID is worth
|
||||
* looking up. <strong>This is the one definition of loopback in the daemon</strong> —
|
||||
* {@code CallerResolver} calls it rather than keeping its own, because the two used to differ
|
||||
* and that difference was a privilege escalation (fleetd #305).
|
||||
*
|
||||
* <p>The whole of {@code 127.0.0.0/8} counts, not just {@code 127.0.0.1}. On Linux every
|
||||
* address in that range is bound to {@code lo} by default, so a process can connect to
|
||||
* {@code 127.0.0.1:8765} with a source address of {@code 127.0.0.2} — measured on the Linux
|
||||
* fleet host, where binding that source succeeds.
|
||||
*
|
||||
* <p><strong>What excluding an address costs, stated as it is today.</strong> This paragraph
|
||||
* used to say that narrowing this range turned a worker into the lead, and that widening the
|
||||
* check was what closed the hole. That was true only while there were <em>two</em> definitions
|
||||
* that disagreed: {@code ConnectionIdentity} skipped the identity lookup for {@code 127.0.0.2}
|
||||
* while {@code CallerResolver} read the same address as loopback and granted the primary role.
|
||||
* #305 removed the second copy, and with one shared definition the old sentence no longer holds.
|
||||
*
|
||||
* <p>Measured on 2026-09-04 by narrowing this method back to exactly {@code 127.0.0.1} and
|
||||
* running {@code CallerResolverTest} and {@code ConnectionIdentityTest}: a caller from
|
||||
* {@code 127.0.0.2} then resolves to {@code ANONYMOUS}, not {@code PRIMARY} — for a worker
|
||||
* ({@code aWorkerOnAnyLoopbackSourceAddressIsStillAWorkerNotThePrimary}) and for a non-worker
|
||||
* ({@code aNonWorkerOnAnyLoopbackSourceAddressIsStillThePrimary}) alike. Excluding an address
|
||||
* now <em>refuses</em> its caller; it does not promote one.
|
||||
*
|
||||
* <p>So keep the whole range, but for the plain reason: a genuine worker or primary that
|
||||
* connects from {@code 127.0.0.2} must be identifiable at all, and narrowing this predicate
|
||||
* locks it out. That is an outage, and an outage is the direction to fail in — which is exactly
|
||||
* why the range must not be narrowed casually and also why doing so is no longer a security
|
||||
* hole. This predicate still does not decide whether a caller is trusted; it decides whether the
|
||||
* caller's identity is <em>resolved at all</em>. What makes an unresolved caller safe is
|
||||
* {@link Caller#resolved()} (#317), not this method.
|
||||
*/
|
||||
public static boolean isLoopback(String addr) {
|
||||
if (addr == null) {
|
||||
return false;
|
||||
}
|
||||
String a = addr.startsWith("::ffff:") ? addr.substring(7) : addr; // IPv4-mapped IPv6
|
||||
return a.startsWith("127.") || "::1".equals(a) || "0:0:0:0:0:0:0:1".equals(a);
|
||||
}
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,83 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* The one canonical set of MCP tool names this daemon registers (fleetd #469, follow-up to #464).
|
||||
*
|
||||
* <p>Before this enum, the tool surface was written twice with nothing tying the copies together:
|
||||
* once as the literal {@code "fleet_…"} string passed to each tool-schema builder in
|
||||
* {@link FleetMcp}, and again as the case labels of {@link FleetMcp}'s authorization switch. A
|
||||
* reader that needed "what does this server register" — a charter check, in particular — had no
|
||||
* source to ask except scraping {@code FleetMcp.java}'s source text for {@code tool("…")} calls: a
|
||||
* third copy of the same list, and the weakest of the three forms.
|
||||
*
|
||||
* <p>Every reader that needs the registered tool surface now asks this enum instead:
|
||||
*
|
||||
* <ul>
|
||||
* <li>the tool-schema builders in {@code FleetMcp} pass {@code wireName()} rather than a literal;
|
||||
* <li>{@code FleetMcp}'s constructor asserts, at startup, that the set of tool names it actually
|
||||
* registers with the MCP SDK equals {@link #wireNames()} exactly — a canonical entry that is
|
||||
* never registered, or a registration with no canonical entry backing it, fails the daemon's
|
||||
* own boot rather than only a test's;
|
||||
* <li>{@code FleetMcp.toolAction}'s dispatch onto {@code Authz.Action} switches on the enum
|
||||
* (not the raw string) with no {@code default}, so adding a tool here without pinning its
|
||||
* action is a compile error, not a run-time throw;
|
||||
* <li>{@link CharterToolSurface} asks {@link #wireNames()} to check a configured launch charter
|
||||
* against the live tool surface, instead of scraping source text a third time.
|
||||
* </ul>
|
||||
*/
|
||||
public enum FleetTool {
|
||||
|
||||
SEND("fleet_send"),
|
||||
REPLY("fleet_reply"),
|
||||
ASK("fleet_ask"),
|
||||
STATUS("fleet_status"),
|
||||
POLL("fleet_poll"),
|
||||
ACK("fleet_ack"),
|
||||
SPAWN("fleet_spawn"),
|
||||
LIST("fleet_list"),
|
||||
STOP("fleet_stop"),
|
||||
PROFILES("fleet_profiles"),
|
||||
WHOAMI("fleet_whoami"),
|
||||
HANDOVER("fleet_handover");
|
||||
|
||||
private final String wireName;
|
||||
|
||||
FleetTool(String wireName) {
|
||||
this.wireName = wireName;
|
||||
}
|
||||
|
||||
/** The name this tool is registered under, and called by, on the wire ({@code "fleet_send"}, …). */
|
||||
public String wireName() {
|
||||
return wireName;
|
||||
}
|
||||
|
||||
private static final Map<String, FleetTool> BY_WIRE_NAME;
|
||||
private static final Set<String> WIRE_NAMES;
|
||||
|
||||
static {
|
||||
Map<String, FleetTool> byName = new LinkedHashMap<>();
|
||||
Set<String> names = new LinkedHashSet<>();
|
||||
for (FleetTool tool : values()) {
|
||||
byName.put(tool.wireName, tool);
|
||||
names.add(tool.wireName);
|
||||
}
|
||||
BY_WIRE_NAME = Map.copyOf(byName);
|
||||
WIRE_NAMES = Set.copyOf(names);
|
||||
}
|
||||
|
||||
/** The tool named {@code wireName}, or empty when this daemon registers no such tool. */
|
||||
public static Optional<FleetTool> byWireName(String wireName) {
|
||||
return Optional.ofNullable(BY_WIRE_NAME.get(wireName));
|
||||
}
|
||||
|
||||
/** Every wire name this daemon registers — the canonical tool surface. */
|
||||
public static Set<String> wireNames() {
|
||||
return WIRE_NAMES;
|
||||
}
|
||||
}
|
||||
@@ -41,6 +41,14 @@ public final class LsofPeerPidLookup implements PeerPidLookup {
|
||||
if (!p.waitFor(2, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
}
|
||||
if (found < 0) {
|
||||
// fleetd #317: this is the silent path — lsof ran clean and simply reported no
|
||||
// matching process (e.g. queried before the OS socket table settles). Previously
|
||||
// this logged nothing at all, which is exactly why the escalation went unnoticed;
|
||||
// the exception path below already logs. A caller now refused because of this is
|
||||
// still refused (never promoted) — this line only makes the refusal diagnosable.
|
||||
log.debug("lsof peer-pid lookup for port {} found no matching process", port);
|
||||
}
|
||||
return found;
|
||||
} catch (Exception e) {
|
||||
log.debug("lsof peer-pid lookup for port {} failed: {}", port, e.getMessage());
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.fasterxml.jackson.databind.node.ObjectNode;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
@@ -14,6 +17,9 @@ import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.nio.file.attribute.PosixFileAttributeView;
|
||||
import java.util.Arrays;
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
@@ -47,6 +53,21 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ClaudeCodeLauncher.class);
|
||||
|
||||
/** JSON codec for the additive workspace-trust seed (fleetd #149) — Jackson's default settings. */
|
||||
private static final ObjectMapper TRUST_JSON = new ObjectMapper();
|
||||
|
||||
/**
|
||||
* Serialises every {@link #seedTrustDialog} read-modify-write for the whole daemon process.
|
||||
* Two claude-code spawns starting at once are normal (fleetd runs several members in parallel
|
||||
* routinely) and both would otherwise read the same {@code .claude.json}, add their own entry
|
||||
* to their own in-memory copy, and write — the second write wins and the first spawn's trust
|
||||
* entry silently disappears. A single process-wide lock is enough because every spawn on this
|
||||
* daemon runs in this one JVM; it does not protect against a second daemon process or the
|
||||
* operator's own Claude Code process writing at the same instant, which {@link #writeAtomically}
|
||||
* covers instead (each writer only ever sees a fully-old or fully-new file, never a torn one).
|
||||
*/
|
||||
private static final Object TRUST_JSON_LOCK = new Object();
|
||||
|
||||
private final SubscriptionGuard guard;
|
||||
|
||||
/**
|
||||
@@ -258,11 +279,15 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
workerEnv.remove("ANTHROPIC_AUTH_TOKEN");
|
||||
} else {
|
||||
workerEnv.put("ANTHROPIC_BASE_URL", baseUrl);
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", env.apply(cfg.tokenEnv()));
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", resolveEnv(cfg.tokenEnv()));
|
||||
}
|
||||
putIfPresent(workerEnv, "ANTHROPIC_MODEL", cfg.model());
|
||||
putIfPresent(workerEnv, "CLAUDE_CONFIG_DIR", cfg.configDir());
|
||||
applyGitToken(workerEnv, cfg);
|
||||
// fleetd #149: seed the workspace-trust entry BEFORE this spawn ever reaches herdr — see
|
||||
// seedTrustDialog for why, and isProvisionedWorktree for why this is gated to a worktree
|
||||
// fleetd itself provisioned (never a real checkout, never an un-configured fallback cwd).
|
||||
seedTrustDialog(cfg.configDir(), spec.cwd());
|
||||
|
||||
// CB-547a: Claude Code can MINT its own session id, so fleetd chooses it — a fresh spawn
|
||||
// gets a UUID we pass as --session-id and return from agentSessionId(), so the resume
|
||||
@@ -412,12 +437,12 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
*/
|
||||
private static void writeIdeOverlay(String cwd, String projectPath) {
|
||||
try {
|
||||
Path dotGit = Path.of(cwd, ".git");
|
||||
if (!Files.isRegularFile(dotGit)) {
|
||||
if (!isProvisionedWorktree(cwd)) {
|
||||
// Not a provisioned worktree (primary's real checkout has a .git directory, or the
|
||||
// cwd is not a repo at all). Never write into it.
|
||||
return;
|
||||
}
|
||||
Path dotGit = Path.of(cwd, ".git");
|
||||
// The overlay FILE lives at the worktree root (claude-code's cwd), but its CONTENT pins
|
||||
// project_path to the module dir the IDE opened (projectPath), not the worktree root.
|
||||
Files.writeString(Path.of(cwd, "CLAUDE.local.md"), PeerLauncher.ideOverlayText(projectPath));
|
||||
@@ -447,6 +472,321 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #149: pre-seed the workspace-trust entry for {@code cwd} in Claude Code's own config
|
||||
* file, BEFORE this launch ever reaches herdr (called from {@link #buildLaunch}, which always
|
||||
* runs before the base class starts the process). Claude Code asks an interactive, un-timed
|
||||
* "Is this a project you created or one you trust?" the first time it starts in a directory it
|
||||
* has not seen before, and every {@code worktree: true} spawn lands in a brand-new directory —
|
||||
* so without this seed the member sits on that dialog forever, never mounts the bridge MCP, and
|
||||
* never calls {@code fleet_reply}. herdr reports it as healthy the whole time ({@code
|
||||
* agent_status: blocked}, {@code interactive_ready: true}), so nothing else catches it. Measured
|
||||
* live on fleet01 2026-08-23: an unseeded fresh cwd sat on the dialog indefinitely; a seeded one
|
||||
* reached {@code idle} clean.
|
||||
*
|
||||
* <p>This is not a new grant — the operator already trusted this repo by configuring the
|
||||
* profile against it, and a worktree is a checkout of that same repo.
|
||||
*
|
||||
* <p>The key Claude Code reads is per-project, in {@code .claude.json}: {@code
|
||||
* projects.<cwd>.hasTrustDialogAccepted}. The file lives at {@code <configDir>/.claude.json}
|
||||
* when the profile sets {@code CLAUDE_CONFIG_DIR} (mirrors this launcher's own env var above),
|
||||
* else the default {@code ~/.claude.json} — the same file Claude Code itself would read either
|
||||
* way, so this seeds exactly what the spawned peer is about to open.
|
||||
*
|
||||
* <p><b>Additive, not a rewrite.</b> {@code .claude.json} is large (tens of KB, dozens of
|
||||
* projects) and Claude Code itself rewrites it while running, so this reads the file as a JSON
|
||||
* tree (missing or unreadable → treated as an empty object) and changes only
|
||||
* {@code projects.<cwd>.hasTrustDialogAccepted} (that key alone — see fleetd #247) —
|
||||
* every other top-level key and every other project entry is written back untouched. Only the
|
||||
* one project entry for {@code cwd} is replaced/created; an existing entry for a DIFFERENT cwd
|
||||
* (or the operator's own project history) is never touched.
|
||||
*
|
||||
* <p>Best-effort, like {@link #writeIdeOverlay}: a failure here (unwritable configDir, a
|
||||
* corrupt existing file, …) must never fail the spawn — it is logged at debug and swallowed. A
|
||||
* peer that starts without the seed still starts; it just may hit the dialog fleetd #149
|
||||
* describes.
|
||||
*
|
||||
* <p><b>Gated to a provisioned worktree</b> ({@link HerdrPeerLauncher#isProvisionedWorktree}) — see that
|
||||
* method's javadoc for the incident that made this gate mandatory, not optional: this must
|
||||
* never run against a real checkout or an un-configured fallback cwd, only the exact
|
||||
* always-fresh-directory population fleetd #149 describes.
|
||||
*
|
||||
* <p><b>Atomic and lock-protected.</b> {@code .claude.json} is a live file — Claude Code itself
|
||||
* rewrites it while running, and this daemon routinely spawns several members at once, each
|
||||
* calling this method for its own cwd. Every write goes through {@link #writeAtomically} (a
|
||||
* sibling-temp-file + {@code ATOMIC_MOVE}, never a truncate-in-place) so a crash mid-write or a
|
||||
* concurrent reader never observes a half-written file, and through {@link #TRUST_JSON_LOCK} so
|
||||
* two concurrent spawns' entries both survive instead of the second write silently discarding
|
||||
* the first. Both exist because of a real incident: see {@link HerdrPeerLauncher#isProvisionedWorktree}'s javadoc
|
||||
* and {@link #writeAtomically}'s javadoc.
|
||||
*
|
||||
* <p><b>Compare-and-swap against a writer the lock cannot reach (fleetd #247).</b>
|
||||
* {@code TRUST_JSON_LOCK} only serialises calls this launcher itself makes inside this one JVM.
|
||||
* It does nothing about a writer outside it — and on a host where the profile's
|
||||
* {@code configDir} is the operator's own {@code CLAUDE_CONFIG_DIR}, the file this method writes
|
||||
* IS the operator's own live Claude Code session's config file, being read and written by that
|
||||
* session while it runs. Measured 2026-09-03: its mtime moved minutes after a spawn while that
|
||||
* session was active. A plain read-modify-write there is a routine lost update, not a rare one:
|
||||
* fleetd reads v1, the operator's session reads v1 and writes v2 with their own change, fleetd's
|
||||
* {@code ATOMIC_MOVE} then lands v3 built from v1 — atomic, but v2's change is gone. So before
|
||||
* the move this method re-reads {@code target}'s exact bytes and compares them with the bytes it
|
||||
* built its update from; a mismatch means someone else wrote in between, and it discards its
|
||||
* work and rebuilds from the fresh bytes, up to {@link #MAX_TRUST_JSON_CAS_ATTEMPTS} times.
|
||||
* <b>Exhausting the retries writes nothing</b> — see the WARN at the end of the loop for why
|
||||
* that, not a last write-anyway, is the safe failure: the member shows the trust dialog and
|
||||
* fails to reach an injectable state, which is visible, logged and recoverable; overwriting the
|
||||
* operator's live config with a stale copy is neither. This narrows the lost-update window, it
|
||||
* does not close it — a write landing between the final re-read and the {@code ATOMIC_MOVE}
|
||||
* itself is still lost, because there is no OS-level compare-and-swap on a plain file, only this
|
||||
* cooperative narrowing of the gap.
|
||||
*
|
||||
* <p><b>fleetd #285: refuses under {@code memberHerdrSocket} rather than writing somewhere the
|
||||
* member cannot read.</b> Under {@code memberHerdrSocket:} the member pane runs as a
|
||||
* <em>different OS user with its own {@code $HOME}</em> — the same reason {@link
|
||||
* #writeCharterFile} routes the role/reply charter under {@code worktreeRoot} instead of
|
||||
* {@code java.io.tmpdir} and refuses the spawn when it cannot. This method has no equivalent
|
||||
* relocation available: unlike the charter (fleetd's own content, free to place anywhere and
|
||||
* hand to the peer via an argv flag), {@code .claude.json} is a file Claude Code looks up for
|
||||
* ITSELF at a fixed location — {@code CLAUDE_CONFIG_DIR/.claude.json}, or else the member OS
|
||||
* user's own {@code ~/.claude.json}, a path fleetd has no channel to learn. So when {@code
|
||||
* configDir} is unset, there is no member-readable target to seed at all — writing the
|
||||
* unqualified default would land in <em>fleetd's own</em> {@code ~/.claude.json} instead, the
|
||||
* exact defect this fix closes, not a workable fallback. And even with {@code configDir} set,
|
||||
* the file this method itself just wrote is {@code 0600} (owner-only — see {@link
|
||||
* #copyPosixPermissionsIfPresent}), unreadable by a different-uid member unless shared with
|
||||
* {@code worktreeGroup}, the same group {@link EnvAllowListScrub#shareWithGroup} already uses
|
||||
* for the ZDOTDIR scrub (fleetd #213) and the charter file (fleetd #219/#222). So under {@code
|
||||
* memberHerdrSocket} this method requires BOTH {@code configDir} and {@code worktreeGroup}
|
||||
* before it ever touches a file, and refuses the spawn — naming exactly which one is missing —
|
||||
* rather than silently corrupt fleetd's own home or hand the member an unreadable path. This
|
||||
* mirrors {@link #writeCharterFile}'s "refuse, don't degrade" decision: a member spawned without
|
||||
* a readable trust seed is not degraded, it sits on the interactive dialog forever and never
|
||||
* calls {@code fleet_reply} — exactly the failure fleetd #149 exists to prevent, so trading it
|
||||
* for "spawn something" is not worth it. When {@code configDir} and {@code worktreeGroup} are
|
||||
* both present, the write proceeds exactly as below and the resulting file is additionally
|
||||
* chgrp'd/chmod'd group-readable ({@code rw-r-----}) via {@link
|
||||
* EnvAllowListScrub#shareFileWithGroup(Path, String)} so the member's OS user can actually
|
||||
* open it — the
|
||||
* directory itself (unlike {@code worktreeRoot} or the charter's per-spawn directory) is not
|
||||
* fleetd-managed, so its own traversal permissions remain the operator's setup, same as they
|
||||
* already must be for the member to read anything else fleetd points {@code CLAUDE_CONFIG_DIR}
|
||||
* at. With {@code memberHerdrSocket} ABSENT (today's only live mode) every branch below is
|
||||
* byte-identical to before this fix.
|
||||
*
|
||||
* @param configDir the profile's {@code CLAUDE_CONFIG_DIR} ({@code cfg.configDir()}), or
|
||||
* {@code null}/blank to target the default {@code ~/.claude.json} — refused
|
||||
* outright when {@code memberHerdrSocket} is configured, see above
|
||||
* @param cwd the spawn's resolved working directory — the exact key Claude Code will look
|
||||
* up for itself once it starts there
|
||||
* @throws IllegalStateException when {@code memberHerdrSocket} is configured but {@code
|
||||
* configDir} and/or {@code worktreeGroup} is not — the same
|
||||
* refusal shape as {@link #writeCharterFile}
|
||||
*/
|
||||
private void seedTrustDialog(String configDir, String cwd) {
|
||||
if (!isProvisionedWorktree(cwd)) {
|
||||
return;
|
||||
}
|
||||
boolean unsetConfigDir = configDir == null || configDir.isBlank();
|
||||
boolean memberHerdrSocket = memberHerdrSocketConfigured();
|
||||
String group = memberHerdrSocket ? memberGroup() : null;
|
||||
if (memberHerdrSocket && (unsetConfigDir || group == null)) {
|
||||
throw new IllegalStateException("memberHerdrSocket is configured, so the workspace-trust "
|
||||
+ "seed (.claude.json, which gates Claude Code's interactive trust dialog) must be "
|
||||
+ "placed where the member's OS user can read it — configDir, shared via "
|
||||
+ "worktreeGroup — but " + (unsetConfigDir ? "configDir" : "worktreeGroup")
|
||||
+ " is not configured. Refusing to spawn rather than write fleetd's own default "
|
||||
+ "'~/.claude.json' or hand the member a config file it cannot read: that member "
|
||||
+ "would sit on the interactive trust dialog forever and never reach an "
|
||||
+ "injectable state. Configure configDir on this profile and worktreeGroup on the "
|
||||
+ "fleet to enable claude-code member spawns under memberHerdrSocket.");
|
||||
}
|
||||
Path target = unsetConfigDir
|
||||
? Path.of(System.getProperty("user.home"), ".claude.json")
|
||||
: Path.of(configDir, ".claude.json");
|
||||
if (unsetConfigDir) {
|
||||
// fleetd #247: configDir unset is the ONLY path that targets ~/.claude.json — the
|
||||
// operator's own home file, not a per-profile one — and it is the default, so a
|
||||
// profile that simply forgot to set configDir gets no signal at all short of the
|
||||
// operator noticing their own file changing. Say so loudly, every time it is about to
|
||||
// happen, rather than only once ever: each occurrence is a live write to a real
|
||||
// person's home config and deserves its own log line. (Reached only when
|
||||
// memberHerdrSocket is absent — the block above already refused otherwise.)
|
||||
log.warn("seedTrustDialog: profile has no configDir set, so the workspace-trust seed "
|
||||
+ "for cwd '{}' is about to write the operator's own default '{}' — set "
|
||||
+ "configDir on this profile to target a per-member config file instead",
|
||||
cwd, target);
|
||||
}
|
||||
synchronized (TRUST_JSON_LOCK) {
|
||||
boolean written = false;
|
||||
try {
|
||||
if (target.getParent() != null) {
|
||||
Files.createDirectories(target.getParent());
|
||||
}
|
||||
for (int attempt = 1; attempt <= MAX_TRUST_JSON_CAS_ATTEMPTS && !written; attempt++) {
|
||||
byte[] before = Files.isRegularFile(target) ? Files.readAllBytes(target) : null;
|
||||
ObjectNode root = parseTrustJsonOrEmpty(before);
|
||||
JsonNode projectsNode = root.get("projects");
|
||||
ObjectNode projects = projectsNode instanceof ObjectNode projectsObject
|
||||
? projectsObject : TRUST_JSON.createObjectNode();
|
||||
if (!(projectsNode instanceof ObjectNode)) {
|
||||
root.set("projects", projects);
|
||||
}
|
||||
JsonNode projectNode = projects.get(cwd);
|
||||
ObjectNode project = projectNode instanceof ObjectNode projectObject
|
||||
? projectObject : TRUST_JSON.createObjectNode();
|
||||
if (!(projectNode instanceof ObjectNode)) {
|
||||
projects.set(cwd, project);
|
||||
}
|
||||
// fleetd #247: ONLY hasTrustDialogAccepted. We used to write
|
||||
// hasCompletedProjectOnboarding beside it; do not put it back. Measured on
|
||||
// 2026-09-03, minutes after a live spawn seeded this file: 28 of 28 project
|
||||
// entries carried hasTrustDialogAccepted and 0 of 28 carried the onboarding key
|
||||
// — including the 27 entries Claude Code wrote for itself. Claude Code
|
||||
// normalises the whole file when it saves and drops that key every time, so
|
||||
// writing it achieved nothing except making the next reader think it mattered.
|
||||
// The member reached idle with the trust flag alone, which is the only outcome
|
||||
// this seed exists for. If a future Claude Code needs the second flag the
|
||||
// symptom returns as the trust dialog fleetd #149 describes — re-measure then,
|
||||
// do not restore it on a guess.
|
||||
project.put("hasTrustDialogAccepted", true);
|
||||
String newContent = TRUST_JSON.writerWithDefaultPrettyPrinter().writeValueAsString(root);
|
||||
|
||||
trustJsonCasTestHook.run();
|
||||
|
||||
// fleetd #247 CAS: re-read immediately before the move and compare with what
|
||||
// this attempt built its update from. A mismatch means another writer (most
|
||||
// plausibly the operator's own live Claude Code — see this method's javadoc)
|
||||
// landed a change in between; discard this attempt's work and rebuild from the
|
||||
// fresh bytes rather than blindly overwriting it.
|
||||
byte[] atMove = Files.isRegularFile(target) ? Files.readAllBytes(target) : null;
|
||||
if (!Arrays.equals(before, atMove)) {
|
||||
continue;
|
||||
}
|
||||
writeAtomically(target, newContent);
|
||||
written = true;
|
||||
}
|
||||
if (!written) {
|
||||
// fleetd #247: deliberately do NOT write here. A member that starts without the
|
||||
// seed still starts — it may hit the trust dialog fleetd #149 describes and fail
|
||||
// to reach an injectable state, but that failure is visible (herdr reports it,
|
||||
// the spawn-readiness gate times out) and recoverable (retry the spawn). Writing
|
||||
// our stale copy over whatever the other writer left would be silent and, if that
|
||||
// other writer is the operator's own live session, could destroy real
|
||||
// configuration — fail toward the recoverable outcome, not the silent one.
|
||||
log.warn("seedTrustDialog: gave up seeding workspace-trust for cwd '{}' into '{}' "
|
||||
+ "after {} attempts — another writer (most plausibly the operator's own "
|
||||
+ "live Claude Code sharing this file) kept changing it faster than we "
|
||||
+ "could re-read it, so nothing was written; the member may show the "
|
||||
+ "trust dialog instead", cwd, target, MAX_TRUST_JSON_CAS_ATTEMPTS);
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.debug("cannot seed workspace-trust entry for cwd '{}' into '{}'", cwd, target, e);
|
||||
return;
|
||||
}
|
||||
// fleetd #285: the write above lands as fleetd's own OS user; under memberHerdrSocket
|
||||
// that is NOT the member's OS user, so without this the member still cannot read the
|
||||
// file it exists to seed — a silent readiness timeout with the write looking "done".
|
||||
// Deliberately OUTSIDE the swallow-all catch above: a group that fails to resolve here
|
||||
// means the seed is unreadable despite a successful write, which must fail as loudly as
|
||||
// writeCharterFile's own EnvAllowListScrub.shareWithGroup call already does.
|
||||
if (written && memberHerdrSocket) {
|
||||
EnvAllowListScrub.shareFileWithGroup(target, group);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/** Bound on {@link #seedTrustDialog}'s fleetd #247 compare-and-swap retry loop. */
|
||||
private static final int MAX_TRUST_JSON_CAS_ATTEMPTS = 5;
|
||||
|
||||
/**
|
||||
* Test-only seam for fleetd #247: invoked once per CAS attempt inside {@link #seedTrustDialog}'s
|
||||
* retry loop, after that attempt has read the target's bytes and built its replacement content,
|
||||
* but immediately before the final re-read/compare that decides whether to write. A no-op in
|
||||
* production. Package-visible (not {@code private}) so {@code ClaudeCodeLauncherTest} can install
|
||||
* a hook here that writes to the target file, deterministically simulating a writer racing
|
||||
* fleetd's own read-modify-write at the exact instant the CAS is meant to catch — the same race
|
||||
* an external process (most plausibly the operator's own live Claude Code) creates, without
|
||||
* depending on real thread scheduling to land the interleaving. A test that sets this MUST
|
||||
* restore it to the no-op default in a {@code finally} block — it is shared, static state.
|
||||
*/
|
||||
static Runnable trustJsonCasTestHook = () -> {};
|
||||
|
||||
/**
|
||||
* Parse {@code bytes} as a {@code .claude.json} tree, or hand back a fresh empty object when
|
||||
* {@code bytes} is {@code null} (no file yet) or does not parse to a JSON object — the same
|
||||
* missing-or-unreadable-is-empty fallback {@link #seedTrustDialog} always used, factored out so
|
||||
* the fleetd #247 CAS loop can call it once per attempt.
|
||||
*/
|
||||
private static ObjectNode parseTrustJsonOrEmpty(byte[] bytes) throws IOException {
|
||||
if (bytes == null) {
|
||||
return TRUST_JSON.createObjectNode();
|
||||
}
|
||||
JsonNode existing = TRUST_JSON.readTree(bytes);
|
||||
return existing instanceof ObjectNode existingObject ? existingObject : TRUST_JSON.createObjectNode();
|
||||
}
|
||||
|
||||
/**
|
||||
* Write {@code content} to {@code target} atomically: serialise to a sibling temp file in the
|
||||
* <strong>same directory</strong> as {@code target} (an atomic move is only guaranteed within
|
||||
* one filesystem — a different directory could mean a different filesystem), then
|
||||
* {@link StandardCopyOption#ATOMIC_MOVE} it into place. A reader — Claude Code itself, or
|
||||
* another {@code seedTrustDialog} call — only ever observes the fully-old file or the
|
||||
* fully-new one, never a truncated or half-written one.
|
||||
*
|
||||
* <p><b>fleetd #149 incident.</b> The original implementation used
|
||||
* {@code Files.writeString(target, content)} directly, which truncates {@code target} in place
|
||||
* before writing the replacement bytes. Combined with an ungated {@code cwd} (see
|
||||
* {@link HerdrPeerLauncher#isProvisionedWorktree}'s javadoc), a mutation-testing run hit that truncation window
|
||||
* against the operator's real {@code ~/.claude.json} and left it at 178 bytes. The gate closes
|
||||
* <em>which file</em> this can ever target; this closes <em>how</em> the target is written, so
|
||||
* that even a legitimate write against a real, live, concurrently-read {@code .claude.json}
|
||||
* cannot leave it observably empty or partial.
|
||||
*
|
||||
* <p>Preserves {@code target}'s existing POSIX permissions (Claude Code ships {@code
|
||||
* .claude.json} as {@code 0600}) when the filesystem reports them; a freshly created temp file
|
||||
* already defaults to owner-only permissions on a POSIX filesystem, so a first-ever write (no
|
||||
* existing {@code target}) is no less private without this. On a non-POSIX filesystem (e.g.
|
||||
* Windows) the permission copy is a silent no-op rather than a failure.
|
||||
*
|
||||
* <p>Package-visible (not {@code private}) so a test can drive it directly with a concurrent
|
||||
* reader thread and prove the torn-file property this method exists for — the LOCK in
|
||||
* {@link #seedTrustDialog} already fully serialises every call this launcher itself makes, so a
|
||||
* test that only ever goes through {@code seedTrustDialog}/{@code spawn()} could never observe
|
||||
* a torn file regardless of whether this method is atomic; it would be proving the lock, not
|
||||
* this method. Atomicity's actual job is protecting against a writer the lock cannot reach at
|
||||
* all — a second daemon process, or the operator's own live Claude Code — so the test for it
|
||||
* has to reach this method on its own.
|
||||
*/
|
||||
static void writeAtomically(Path target, String content) throws IOException {
|
||||
Path parent = target.getParent();
|
||||
Path tmp = Files.createTempFile(parent, target.getFileName() + ".", ".tmp");
|
||||
try {
|
||||
Files.writeString(tmp, content);
|
||||
copyPosixPermissionsIfPresent(target, tmp);
|
||||
Files.move(tmp, target, StandardCopyOption.ATOMIC_MOVE, StandardCopyOption.REPLACE_EXISTING);
|
||||
} catch (IOException e) {
|
||||
Files.deleteIfExists(tmp);
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
/** Copy {@code target}'s POSIX permissions onto {@code tmp}, or no-op where either is unsupported. */
|
||||
private static void copyPosixPermissionsIfPresent(Path target, Path tmp) {
|
||||
try {
|
||||
if (!Files.isRegularFile(target)) {
|
||||
return; // nothing to inherit from — first-ever write, temp file's own default stands
|
||||
}
|
||||
PosixFileAttributeView view = Files.getFileAttributeView(target, PosixFileAttributeView.class);
|
||||
if (view == null) {
|
||||
return; // non-POSIX filesystem — nothing this JVM can read/set here
|
||||
}
|
||||
Files.setPosixFilePermissions(tmp, Files.getPosixFilePermissions(target));
|
||||
} catch (IOException e) {
|
||||
log.debug("cannot preserve permissions of '{}' onto its replacement", target, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** {@code s}, or {@code null} when {@code s} is null/blank — the charter-presence test used above. */
|
||||
private static String nonBlank(String s) {
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
@@ -459,12 +799,83 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
* {@link OpenCodeLauncher#writeConfig} already uses for its charter file, since the process that
|
||||
* reads this file (the spawned peer) outlives this JVM call and there is no spawn-scoped teardown
|
||||
* hook to delete it synchronously.
|
||||
*
|
||||
* <p><b>fleetd #222.</b> {@code Files.createTempFile(prefix, suffix)} with no directory argument
|
||||
* resolves against {@code java.io.tmpdir} — on macOS the per-user {@code $TMPDIR} under
|
||||
* {@code /var/folders/...}, mode {@code 0700}, both resolved against FLEETD's own OS user. Under
|
||||
* {@code memberHerdrSocket:} the member pane runs as a DIFFERENT OS user, so that user cannot even
|
||||
* traverse the directory, let alone read the file — and since fleetd #220 the charter file is the
|
||||
* ONLY delivery path for {@code --append-system-prompt-file}, always, not merely the fallback it
|
||||
* used to be. A member handed a path it cannot read is not degraded, it is broken: see this
|
||||
* method's refusal branch below.
|
||||
*
|
||||
* <p><b>Measured severity (fleetd #222 real-binary check, claude 2.1.258):</b> an unreadable
|
||||
* {@code --append-system-prompt-file} is the LOUD failure, not the silent one. {@code claude}
|
||||
* checks the file before touching auth or the network — invoked with a bogus API key against a
|
||||
* {@code chmod 000} file, it printed {@code Error reading append system prompt file: EACCES:
|
||||
* permission denied, open '<path>'} and exited 1 immediately (a nonexistent path gets {@code
|
||||
* Error: Append system prompt file not found: <path>}, same exit code). So the pre-fix bug did
|
||||
* NOT leave a charter-less member silently occupying a pane and never calling {@code
|
||||
* fleet_reply} — it made the herdr pane exit immediately, which the CB-306 spawn-readiness gate
|
||||
* (this launcher's {@code spawnReadyTimeoutMs} poll) would have surfaced as "did not reach
|
||||
* injectable state", the same unexplained-timeout shape fleetd #220 already describes. Still a
|
||||
* real defect (every claude-code member under {@code memberHerdrSocket} would have failed to
|
||||
* spawn), but not the worse, undetectable failure mode.
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@code memberHerdrSocket} ABSENT (today's only live mode): byte-identical to before this
|
||||
* fix — {@code Files.createTempFile("fleetd-role-charter-", ".md")} with no directory
|
||||
* argument, i.e. still resolved against {@code java.io.tmpdir}.</li>
|
||||
* <li>{@code memberHerdrSocket} PRESENT: a fresh per-spawn directory is created under {@code
|
||||
* worktreeRoot} (never {@code java.io.tmpdir}) holding just the charter file, then shared
|
||||
* read-only with {@code worktreeGroup} via {@link EnvAllowListScrub#shareWithGroup} — the
|
||||
* SAME mechanism fleetd #213 built for the ZDOTDIR scrub and fleetd #219 reused for {@link
|
||||
* OpenCodeLauncher#writeConfig}'s {@code opencode.json} directory, reused here rather than
|
||||
* duplicated a third time. A per-spawn subdirectory (not {@code worktreeRoot} itself) is the
|
||||
* unit {@code shareWithGroup} chmods, so this never touches permissions on anything else
|
||||
* under {@code worktreeRoot}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><b>Follows fleetd #219's REFUSAL decision, not #213's degrade decision.</b> The ZDOTDIR
|
||||
* scrub is a credential CONTROL — a degraded control (the CB-596 sentinel overlay) still has
|
||||
* value, so #213 falls back rather than refusing. A charter is NOT a control, it is the member's
|
||||
* TURN CONTRACT (the rule that ends every turn with {@code fleet_reply}). A member spawned with no
|
||||
* charter is not degraded, it is broken: either claude-code exits on the unreadable
|
||||
* {@code --append-system-prompt-file} path and the spawn dies at the readiness gate (loud), or it
|
||||
* starts anyway with no charter and never calls {@code fleet_reply} — the sender silently gets
|
||||
* nothing (silent). Neither outcome is worth trading for "spawn something." So a missing {@code
|
||||
* worktreeRoot}/{@code worktreeGroup} under {@code memberHerdrSocket} refuses the spawn here,
|
||||
* naming the missing key, exactly like {@link OpenCodeLauncher#configParentDir()}.
|
||||
*
|
||||
* @throws IllegalStateException when {@code memberHerdrSocket} is configured but {@code
|
||||
* worktreeRoot} and/or {@code worktreeGroup} is not
|
||||
*/
|
||||
private static Path writeCharterFile(String charterText) {
|
||||
private Path writeCharterFile(String charterText) {
|
||||
try {
|
||||
Path file = Files.createTempFile("fleetd-role-charter-", ".md");
|
||||
if (!memberHerdrSocketConfigured()) {
|
||||
Path file = Files.createTempFile("fleetd-role-charter-", ".md");
|
||||
Files.writeString(file, charterText);
|
||||
file.toFile().deleteOnExit();
|
||||
return file;
|
||||
}
|
||||
Path parentDir = memberScrubParentDir();
|
||||
String group = memberGroup();
|
||||
if (parentDir == null || group == null) {
|
||||
throw new IllegalStateException("memberHerdrSocket is configured, so the role/reply "
|
||||
+ "charter file (mounted via --append-system-prompt-file) must be placed where "
|
||||
+ "the member's OS user can read it — worktreeRoot, shared via worktreeGroup — "
|
||||
+ "but " + (parentDir == null ? "worktreeRoot" : "worktreeGroup") + " is not "
|
||||
+ "configured. Refusing to spawn rather than hand the member a charter path it "
|
||||
+ "cannot read: that member's turn contract (the fleet_reply rule) would never "
|
||||
+ "reach it. Configure both worktreeRoot and worktreeGroup to enable claude-code "
|
||||
+ "member spawns under memberHerdrSocket.");
|
||||
}
|
||||
Path dir = Files.createTempDirectory(parentDir, "fleetd-role-charter-");
|
||||
dir.toFile().deleteOnExit();
|
||||
Path file = dir.resolve("charter.md");
|
||||
Files.writeString(file, charterText);
|
||||
file.toFile().deleteOnExit();
|
||||
EnvAllowListScrub.shareWithGroup(dir, group);
|
||||
return file;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot write role charter temp file", e);
|
||||
|
||||
@@ -10,9 +10,11 @@ import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementCandidate;
|
||||
import dev.ltms.fleet.placement.PlacementContext;
|
||||
import dev.ltms.fleet.placement.PlacementDecision;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import dev.ltms.fleet.placement.PlacementPolicy;
|
||||
@@ -89,9 +91,34 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
private final Supplier<Map<String, FleetConfig.Profile>> profileConfigs;
|
||||
private final Supplier<PlacementPolicy> placementPolicy;
|
||||
|
||||
/**
|
||||
* fleetd #422: the central model allow-list's on/off state, read live per spawn — same reason
|
||||
* {@link #profileConfigs} is a supplier rather than a captured map (see the class doc above and
|
||||
* {@link #enforceModelEnabled}). A caller with no {@code models:} block to read from (the
|
||||
* simpler, map-based constructors used throughout this class's own tests) wires this to a
|
||||
* constant {@code null}, which {@link #models0} treats as "nothing configured, gate never
|
||||
* fires" — the pre-#422 behaviour.
|
||||
*/
|
||||
private final Supplier<FleetConfig.Models> models;
|
||||
|
||||
/** The value {@link #models0} normalizes a {@code null} supplier result to. */
|
||||
private static final FleetConfig.Models NO_MODELS_CONFIGURED = new FleetConfig.Models(List.of());
|
||||
|
||||
/** CB-578 stage B: credential cooldown, checked before an explicit spawn and filtered into placement. */
|
||||
private final BackendQuarantine quarantine;
|
||||
|
||||
/**
|
||||
* fleetd #201 Unit 5: credential cool-off after repeated backend errors, checked before an
|
||||
* explicit spawn and filtered into placement — a SEPARATE, shorter-lived source from
|
||||
* {@link #quarantine}. A never-{@code record}-called instance is naturally inert (its
|
||||
* {@code remainingCoolOffSeconds} always returns empty), so back-compat constructors that predate
|
||||
* this feature share one fixed instance rather than needing a {@code none()} sentinel.
|
||||
*/
|
||||
private final BackendOutagePolicy outagePolicy;
|
||||
|
||||
/** The shared inert instance back-compat constructors wire in — never {@code record}-called. */
|
||||
private static final BackendOutagePolicy NO_OUTAGE_POLICY = new BackendOutagePolicy(System::nanoTime);
|
||||
|
||||
/**
|
||||
* CB-557: the role pools an unqualified spawn draws its candidates from. A supplier that yields
|
||||
* {@code null}, and an empty pool for a role, both fall back to every configured profile — the
|
||||
@@ -130,7 +157,8 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
Map<String, FleetConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, null, BackendQuarantine.none());
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, null,
|
||||
BackendQuarantine.none(), NO_OUTAGE_POLICY);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -148,13 +176,14 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet, BackendQuarantine.none());
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet,
|
||||
BackendQuarantine.none(), NO_OUTAGE_POLICY);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with role pools and quarantine (CB-578 stage B). The full-featured
|
||||
* non-reloading form; {@link #CompositePeerLauncher(List, String, Supplier, Function, BackendQuarantine)}
|
||||
* is what {@code Fleetd.main} actually wires up.
|
||||
* Production constructor with role pools and quarantine (CB-578 stage B), no cool-off (fleetd
|
||||
* #201 Unit 5). Kept for callers that predate the cool-off feature; use the 8-arg overload below
|
||||
* to wire a real {@link BackendOutagePolicy}.
|
||||
*
|
||||
* @param quarantine required — pass {@link BackendQuarantine#none()} for a caller that does not
|
||||
* want the feature, never a defaulting overload (CB-578 stage B's own rule).
|
||||
@@ -166,17 +195,74 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet,
|
||||
BackendQuarantine quarantine) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet, quarantine,
|
||||
NO_OUTAGE_POLICY);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with role pools, quarantine (CB-578 stage B), and cool-off (fleetd
|
||||
* #201 Unit 5). The full-featured non-reloading form;
|
||||
* {@link #CompositePeerLauncher(List, String, Supplier, Function, BackendQuarantine, BackendOutagePolicy)}
|
||||
* is what {@code Fleetd.main} actually wires up.
|
||||
*
|
||||
* @param quarantine required — pass {@link BackendQuarantine#none()} for a caller that does not
|
||||
* want the feature, never a defaulting overload (CB-578 stage B's own rule).
|
||||
* @param outagePolicy required — pass a fresh, never-{@code record}-called {@link
|
||||
* BackendOutagePolicy} for a caller that does not want the feature; same
|
||||
* "explicit opt-out, never a silent default" rule as {@code quarantine}.
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, FleetConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet,
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy) {
|
||||
// No models: block to read from a plain profiles Map — fleetd #422's gate is wired to a
|
||||
// constant null (see enforceModelEnabled/models0), the pre-#422 behaviour for every caller
|
||||
// of this overload. Use the 9-arg overload below to test the gate against a LIVE supplier.
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet, quarantine,
|
||||
outagePolicy, constant(null));
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, plus a LIVE model-gate source (fleetd #422). The 8-arg overload above wires
|
||||
* {@code models} to a constant {@code null} because it has only a static {@code Map<String,
|
||||
* Profile>}, never a full {@code FleetConfig}, to read one from; this overload exists so a test
|
||||
* can prove {@link #enforceModelEnabled} and its candidate filter re-read {@link
|
||||
* FleetConfig.Models} on every call rather than a value captured once at construction — the
|
||||
* exact distinction fleetd #422 exists to get right (see the class doc's CB-559 note on {@link
|
||||
* #profileConfigs}, which this follows). Production wiring uses the
|
||||
* {@code Supplier<FleetConfig>} constructor below instead, which already threads a live
|
||||
* {@code config.get().models()} through.
|
||||
*
|
||||
* @param models required — pass a supplier returning {@code null} for a caller that has no
|
||||
* {@code models:} block to gate against, never a defaulting overload (the same
|
||||
* "explicit opt-out, never a silent default" rule {@code quarantine}/
|
||||
* {@code outagePolicy} already follow).
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, FleetConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet,
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy,
|
||||
Supplier<FleetConfig.Models> models) {
|
||||
// LinkedHashMap, not Map.copyOf: candidates() promises definition order and the weighted
|
||||
// policy breaks exact-weight ties on it, so a salted iteration order would make placement
|
||||
// differ from one JVM run to the next.
|
||||
this(delegates, defaultProfile,
|
||||
constant(Collections.unmodifiableMap(new LinkedHashMap<>(profileConfigs))),
|
||||
constant(placementPolicy), liveCount, constant(fleet), quarantine);
|
||||
constant(placementPolicy), liveCount, constant(fleet), quarantine, outagePolicy, models);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor that re-reads its placement inputs per spawn (CB-559), so a config
|
||||
* reload retargets the next member without a restart.
|
||||
* reload retargets the next member without a restart. No cool-off (fleetd #201 Unit 5); use the
|
||||
* 6-arg overload below to wire a real {@link BackendOutagePolicy}.
|
||||
*
|
||||
* @param config the live configuration — read at every spawn, never captured
|
||||
* @param quarantine required — CB-578 stage B; pass {@link BackendQuarantine#none()} to opt out
|
||||
@@ -186,12 +272,34 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
Supplier<FleetConfig> config,
|
||||
Function<String, Integer> liveCount,
|
||||
BackendQuarantine quarantine) {
|
||||
this(delegates, defaultProfile, config, liveCount, quarantine, NO_OUTAGE_POLICY);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor that re-reads its placement inputs per spawn (CB-559), with cool-off
|
||||
* (fleetd #201 Unit 5). This is what {@code Fleetd.main} actually wires up.
|
||||
*
|
||||
* @param config the live configuration — read at every spawn, never captured
|
||||
* @param quarantine required — CB-578 stage B; pass {@link BackendQuarantine#none()} to opt out
|
||||
* @param outagePolicy required — fleetd #201 Unit 5; pass a fresh, never-{@code record}-called
|
||||
* {@link BackendOutagePolicy} to opt out
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Supplier<FleetConfig> config,
|
||||
Function<String, Integer> liveCount,
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy) {
|
||||
this(delegates, defaultProfile,
|
||||
() -> config.get().profiles(),
|
||||
() -> PlacementPolicies.fromName(config.get().placement()),
|
||||
liveCount,
|
||||
() -> config.get().fleet(),
|
||||
quarantine);
|
||||
quarantine,
|
||||
outagePolicy,
|
||||
// fleetd #422: read live, same as profiles/placement/fleet above — a models.allow
|
||||
// edit (on/off or otherwise) is visible to the very next spawn, no restart needed.
|
||||
() -> config.get().models());
|
||||
}
|
||||
|
||||
/** The all-suppliers form every other constructor funnels into. */
|
||||
@@ -201,9 +309,12 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
Supplier<PlacementPolicy> placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
BackendQuarantine quarantine) {
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy,
|
||||
Supplier<FleetConfig.Models> models) {
|
||||
this.fleet = fleet;
|
||||
this.quarantine = Objects.requireNonNull(quarantine, "quarantine");
|
||||
this.outagePolicy = Objects.requireNonNull(outagePolicy, "outagePolicy");
|
||||
if (delegates.isEmpty()) {
|
||||
throw new IllegalArgumentException("at least one peer adapter must be configured");
|
||||
}
|
||||
@@ -212,6 +323,7 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
this.profileConfigs = profileConfigs;
|
||||
this.placementPolicy = placementPolicy;
|
||||
this.liveCount = liveCount;
|
||||
this.models = Objects.requireNonNull(models, "models");
|
||||
Map<String, HerdrPeerLauncher> index = new LinkedHashMap<>();
|
||||
for (HerdrPeerLauncher d : this.delegates) {
|
||||
for (String profile : d.profiles()) {
|
||||
@@ -242,6 +354,16 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return m == null ? Map.of() : m;
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-configured {@code models:} block, never null (fleetd #422). Read fresh on
|
||||
* every call, the same reason {@link #profiles0} is — a config reload's on/off edit must reach
|
||||
* the very next spawn.
|
||||
*/
|
||||
private FleetConfig.Models models0() {
|
||||
FleetConfig.Models m = models.get();
|
||||
return m == null ? NO_MODELS_CONFIGURED : m;
|
||||
}
|
||||
|
||||
/** The adapter owning {@code profileName} (null/blank → the default). Throws on an unknown profile. */
|
||||
private HerdrPeerLauncher route(String profileName) {
|
||||
String resolved = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
@@ -266,8 +388,12 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
// the charter makes explicit-profile spawns the normal path — so skipping the check
|
||||
// here would leave the cap dead config in real operation.
|
||||
HerdrPeerLauncher d = route(requestedProfile);
|
||||
// Checked in this order so exhaustion quarantine wins when both are active: quarantine
|
||||
// throws first and short-circuits before the cool-off check ever runs (fleetd #201 Unit 5).
|
||||
enforceNotQuarantined(requestedProfile);
|
||||
enforceNotCoolingOff(requestedProfile);
|
||||
enforceMaxLoad(requestedProfile);
|
||||
enforceModelEnabled(requestedProfile);
|
||||
PeerHandle handle = d.spawn(req);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
@@ -277,15 +403,10 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
// the whole profile list. An EXPLICIT profile (above) is left alone on purpose — it is the
|
||||
// operator overriding, and refusing it would break `fleet_spawn{profile:"opus"}`, which
|
||||
// carries no role and so would be judged against the dev pool it was never meant for.
|
||||
List<PlacementCandidate> candidates = candidates(req.role());
|
||||
String roleDefault = defaultProfileFor(req.role());
|
||||
Set<String> unreachable = new HashSet<>();
|
||||
// CB-578 stage B: computed once up front — a quarantine's expiry cannot pass within one spawn
|
||||
// call, so re-deriving it per retry would only cost work, never change the answer.
|
||||
Set<String> quarantined = quarantinedProfiles(candidates);
|
||||
PlacementContext ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable, quarantined);
|
||||
PlacementContext ctx = placementContextFor(req.role(), unreachable);
|
||||
|
||||
int maxAttempts = candidates.isEmpty() ? 1 : candidates.size();
|
||||
int maxAttempts = ctx.candidates().isEmpty() ? 1 : ctx.candidates().size();
|
||||
for (int attempt = 0; attempt < maxAttempts; attempt++) {
|
||||
// Deliberately uncaught: when no candidate is left (all at cap, or all unreachable) the
|
||||
// policy already throws a clear message. Catching it to rethrow a generic
|
||||
@@ -302,8 +423,7 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
// CB-547a: route the chosen profile but keep the caller's session identity — dropping it
|
||||
// here would silently sever the resume handle on every policy-routed spawn. CB-557: the
|
||||
// role rides along for the same reason, or a routed spawn would be labelled as a dev.
|
||||
SpawnRequest routedReq = new SpawnRequest(chosen.profile(), req.requestedCwd(), req.callerCwd(),
|
||||
req.sessionName(), req.resumeSessionId(), req.role());
|
||||
SpawnRequest routedReq = req.withProfile(chosen.profile());
|
||||
try {
|
||||
PeerHandle handle = d.spawn(routedReq);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
@@ -313,13 +433,16 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
chosen.profile(), e.getMessage());
|
||||
unreachable.add(chosen.profile());
|
||||
// Update the context for the next selection so the policy excludes this profile.
|
||||
ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable, quarantined);
|
||||
ctx = placementContextFor(req.role(), unreachable);
|
||||
}
|
||||
}
|
||||
|
||||
// unreachable.size() counts DISTINCT profiles, not attempts (a HashSet dedupes a profile
|
||||
// added twice) — say "distinct" so the count matches the sentence and the profile list that
|
||||
// follows, rather than reading as a count of attempts made (fleetd #315).
|
||||
throw new PeerUnreachableException(
|
||||
"no reachable worker profile available after trying " + unreachable.size()
|
||||
+ " candidate(s): " + String.join(", ", unreachable));
|
||||
+ " distinct candidate(s): " + String.join(", ", unreachable));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -379,6 +502,31 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse an explicit-profile spawn whose credential is cooling off after repeated backend errors
|
||||
* (fleetd #201 Unit 5 — {@link BackendOutagePolicy}): a SEPARATE, shorter-lived source from
|
||||
* {@link #enforceNotQuarantined}'s exhaustion quarantine. Checked after quarantine so exhaustion
|
||||
* wins when both are active — see the call site in {@link #spawn}.
|
||||
*
|
||||
* @throws PlacementException naming the profile, its credential, and the remaining cool-off
|
||||
*/
|
||||
private void enforceNotCoolingOff(String profile) {
|
||||
String credentialId = credentialIdFor(profile);
|
||||
outagePolicy.remainingCoolOffSeconds(credentialId).ifPresent(remaining -> {
|
||||
throw new PlacementException("worker profile '" + profile + "' credential '" + credentialId
|
||||
+ "' is cooling off after repeated backend errors; ~" + remaining
|
||||
+ "s remaining — refusing spawn");
|
||||
});
|
||||
}
|
||||
|
||||
/** The subset of {@code candidates} whose credential is currently cooling off (fleetd #201 Unit 5). */
|
||||
private Set<String> coolingOffProfiles(List<PlacementCandidate> candidates) {
|
||||
return candidates.stream()
|
||||
.map(PlacementCandidate::profile)
|
||||
.filter(p -> outagePolicy.remainingCoolOffSeconds(credentialIdFor(p)).isPresent())
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
private void enforceMaxLoad(String profile) {
|
||||
// Absent config, or a config whose maxLoad normalized to null (ABSENT ⇒ unlimited at load),
|
||||
// means no cap — never cap what wasn't configured. Note "non-positive ⇒ unlimited" was true
|
||||
@@ -397,6 +545,49 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse an explicit-profile spawn whose {@code model:} the operator has turned off in the
|
||||
* central {@code models.allow:} list (fleetd #422). Deliberately worded apart from {@link
|
||||
* #enforceNotQuarantined} and {@link #enforceNotCoolingOff}: those two report a BACKEND-reported
|
||||
* outage (exhaustion, repeated errors); this one reports an OPERATOR decision, so the message
|
||||
* says "turned off" and names the model, never "quarantined" or "cooling off". A FOURTH,
|
||||
* independent reason to refuse a spawn — never layered onto {@code BackendQuarantine} or {@code
|
||||
* BackendOutagePolicy}, which would misattribute an operator's own choice to the backend.
|
||||
*
|
||||
* <p>Reads {@link #models0()} fresh on every call — the same liveness {@link #profiles0()}
|
||||
* already has — so flipping {@code enabled: false} and reloading takes effect on the very next
|
||||
* spawn, no restart (criterion 4). A profile naming no model, or a model absent from {@code
|
||||
* models.allow:} entirely (nothing to gate against), is never refused here.
|
||||
*
|
||||
* @throws PlacementException naming the model and the profile, distinct from quarantine/cool-off
|
||||
*/
|
||||
private void enforceModelEnabled(String profile) {
|
||||
FleetConfig.Profile cfg = profiles0().get(profile);
|
||||
String model = (cfg == null) ? null : cfg.model();
|
||||
if (model == null) {
|
||||
return;
|
||||
}
|
||||
if (models0().offIds().contains(model)) {
|
||||
throw new PlacementException("worker profile '" + profile + "' names model '" + model
|
||||
+ "', which the operator has turned off in models.allow — refusing spawn");
|
||||
}
|
||||
}
|
||||
|
||||
/** The subset of {@code candidates} whose {@code model:} is currently turned off (fleetd #422). */
|
||||
private Set<String> modelOffProfiles(List<PlacementCandidate> candidates) {
|
||||
Set<String> off = models0().offIds();
|
||||
if (off.isEmpty()) {
|
||||
return Set.of();
|
||||
}
|
||||
return candidates.stream()
|
||||
.map(PlacementCandidate::profile)
|
||||
.filter(p -> {
|
||||
FleetConfig.Profile cfg = profiles0().get(p);
|
||||
return cfg != null && cfg.model() != null && off.contains(cfg.model());
|
||||
})
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* The profile names {@code role} may be placed on, in definition order.
|
||||
*
|
||||
@@ -413,8 +604,23 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return known.isEmpty() ? List.copyOf(configured.keySet()) : known;
|
||||
}
|
||||
|
||||
/** The profile an unqualified spawn for {@code role} falls back to under {@code fixed} placement. */
|
||||
private String defaultProfileFor(MemberRole role) {
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Live: reads {@link #poolFor}, which reads {@link #profileConfigs} and {@link #fleet} fresh
|
||||
* on every call, so a config reload is visible without a restart (fleetd #425) — unlike {@link
|
||||
* #defaultProfile}, the field captured once at construction, which this falls back to only when
|
||||
* {@link #poolFor} has nothing to offer at all (no profiles configured for this composite).
|
||||
*
|
||||
* <p>Exact only under the {@code fixed} placement policy — the one that reads this value
|
||||
* ({@code FixedPlacementPolicy}, package-private, hence not linked) as its first, preferred
|
||||
* candidate. {@code weighted}/{@code round-robin} placement can choose a different candidate
|
||||
* from {@code role}'s pool even on the very first spawn; this method does not simulate that
|
||||
* choice, matching what the {@code defaultProfile:}-derived reporting this replaces has always
|
||||
* done.
|
||||
*/
|
||||
@Override
|
||||
public String defaultProfileFor(MemberRole role) {
|
||||
List<String> pool = poolFor(role);
|
||||
return pool.isEmpty() ? defaultProfile : pool.getFirst();
|
||||
}
|
||||
@@ -431,6 +637,142 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the {@link PlacementContext} an unqualified spawn of {@code role} would be judged
|
||||
* against right now — the single source both {@link #spawn} and {@link #place} read, so the two
|
||||
* can never disagree about which conditions (quarantine, cool-off, model-off) apply to which
|
||||
* candidate (fleetd #425 rework: round 1 duplicated this into a second, blind resolver —
|
||||
* {@link #defaultProfileFor} — which is why it regressed; round 2 found that even a single
|
||||
* shared resolver is not enough on its own if the CALLER re-resolves through an explicit
|
||||
* profile afterwards — see {@link PlacementDecision}).
|
||||
*
|
||||
* @param unreachable the caller's mutable unreachable set; {@link #spawn} grows this across
|
||||
* retries and rebuilds the context from it, {@link #place} passes a fresh
|
||||
* empty one since it never retries
|
||||
*/
|
||||
private PlacementContext placementContextFor(MemberRole role, Set<String> unreachable) {
|
||||
List<PlacementCandidate> candidates = candidates(role);
|
||||
String roleDefault = defaultProfileFor(role);
|
||||
// CB-578 stage B: computed once up front — a quarantine's expiry cannot pass within one spawn
|
||||
// call, so re-deriving it per retry would only cost work, never change the answer.
|
||||
Set<String> quarantined = quarantinedProfiles(candidates);
|
||||
// fleetd #201 Unit 5: a distinct set from quarantined — see PlacementContext.coolingOff.
|
||||
Set<String> coolingOff = coolingOffProfiles(candidates);
|
||||
// fleetd #422: read live per spawn, same as quarantined/coolingOff above — a config reload
|
||||
// that flips a model's enabled state is visible to the very next unqualified spawn.
|
||||
Set<String> modelOff = modelOffProfiles(candidates);
|
||||
return new PlacementContext(roleDefault, candidates, liveCount, unreachable,
|
||||
quarantined, coolingOff, modelOff);
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>fleetd #425 rework, round 2: runs the exact same selection {@link #spawn} uses for a
|
||||
* blank-profile request — {@link #placementContextFor} plus one {@link PlacementPolicy#select}
|
||||
* — rather than {@link #defaultProfileFor}'s blind "pool's first entry", so a quarantined,
|
||||
* cooling-off, or model-off pool-first candidate is routed around here exactly as it would be
|
||||
* by a real spawn. Unlike {@link #spawn}, this never retries on {@link
|
||||
* PeerUnreachableException}: there is no spawn attempt to fail, so "unreachable" never grows
|
||||
* past the empty set it starts with, and a single {@link PlacementPolicy#select} call already
|
||||
* reflects the live quarantine/cool-off/model-off state.
|
||||
*
|
||||
* <p>Deliberately does <em>not</em> apply {@link #enforceMaxLoad} (or any of the other three
|
||||
* {@code enforce*} checks): those belong to {@link #spawn}'s EXPLICIT-profile branch, the
|
||||
* operator-override path, and this method answers a different question — "where would an
|
||||
* UNQUALIFIED spawn land". That is not the same as {@code select} ignoring these conditions —
|
||||
* every condition {@code select} filters on (quarantine, cooling off, {@code maxLoad} under
|
||||
* every placement policy including the default {@code fixed}, since fleetd #435, model-off,
|
||||
* unreachable, weight-0) is already reflected in the {@link PlacementDecision} this method
|
||||
* returns, because {@code select} walked past every excluded candidate to find it. What this
|
||||
* method's caller must not do is take that resolved name and hand it back to {@link
|
||||
* #spawn(SpawnRequest)} as an explicit profile: the explicit-profile branch treats the same
|
||||
* exclusion conditions as a reason to REFUSE, where {@code select} had already treated them as
|
||||
* a reason to fall through — round 1 of this fix did exactly that, turning a fall-through this
|
||||
* method had already resolved around into a refusal one call later. Round 2 fixes that at the
|
||||
* caller: {@link #spawn(SpawnRequest, PlacementDecision)} carries this exact decision to the
|
||||
* spawn without re-resolving or re-checking it, through the same routing path {@code select}
|
||||
* itself was consulted from.
|
||||
*
|
||||
* @throws PlacementException if no candidate in {@code role}'s pool is currently placeable
|
||||
* (mirrors what an actual unqualified spawn would throw)
|
||||
*/
|
||||
@Override
|
||||
public PlacementDecision place(MemberRole role) {
|
||||
PlacementContext ctx = placementContextFor(role, new HashSet<>());
|
||||
return new PlacementDecision(placementPolicy.get().select(ctx).profile());
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Delegates to {@link #place}, so the two can never disagree about the answer for the same
|
||||
* {@code role} at the same instant — kept as a convenience for a caller that only wants the
|
||||
* resolved name (a status report, a log line), never for a caller that will act on it by
|
||||
* spawning: that caller must hold the {@link PlacementDecision} itself and pass it to {@link
|
||||
* #spawn(SpawnRequest, PlacementDecision)} — see {@link PlacementDecision}'s javadoc for why
|
||||
* resolving here and spawning separately, with the name fed back in as an explicit profile,
|
||||
* regressed fleetd #425 twice.
|
||||
*/
|
||||
@Override
|
||||
public String routedProfileFor(MemberRole role) {
|
||||
return place(role).profile();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Routes {@code decision.profile()} directly to its owning delegate — the identical
|
||||
* {@code d.spawn(routedReq)} call {@link #spawn(SpawnRequest)}'s blank-profile branch makes for
|
||||
* its first pick — WITHOUT re-running {@link #enforceNotQuarantined}, {@link
|
||||
* #enforceNotCoolingOff}, {@link #enforceMaxLoad}, or {@link #enforceModelEnabled}: those are
|
||||
* the EXPLICIT-profile branch's checks, and {@code decision} did not come from an operator
|
||||
* naming a profile — it came from {@link #place}, which already applied whichever of these
|
||||
* conditions {@link PlacementPolicy#select} actually filters on (fleetd #425 rework, round 2).
|
||||
*
|
||||
* <p>The two branches disagree on purpose about what an excluded profile means, and that
|
||||
* disagreement is not what this method removes. The blank-profile routing branch (and
|
||||
* {@link #place}) treats a quarantined/cooling-off/at-cap/model-off/unreachable/weight-0 profile
|
||||
* as a reason to fall through to the next candidate; the EXPLICIT-profile branch treats naming
|
||||
* that same profile as a reason to refuse outright — someone who names a profile should get a
|
||||
* refusal, not a silent substitution onto a different backend. That is still correct after
|
||||
* fleetd #435. What round 1 got wrong, and what this method exists to stop happening again, is
|
||||
* turning a fall-through into a refusal by accident: resolving a name via {@link #place} and
|
||||
* then handing that same name back to {@link #spawn(SpawnRequest)} as an explicit profile takes
|
||||
* the refusing branch on a decision the routing branch had already approved by falling through
|
||||
* past everything else.
|
||||
*
|
||||
* <p>Before fleetd #435, this exact accident was reachable through {@code maxLoad} specifically:
|
||||
* {@code FixedPlacementPolicy} — the default policy — did not evaluate {@code maxLoad} at all
|
||||
* for automatic selection, so {@link #place} could approve an at-cap profile that {@link
|
||||
* #enforceMaxLoad} would then refuse one call later. fleetd #435 closed that: {@code
|
||||
* FixedPlacementPolicy} now walks past an at-cap candidate exactly like {@code weighted}/
|
||||
* {@code round-robin} already did, so {@link #place} can no longer return one, and this specific
|
||||
* failure — an approved placement dying at {@code enforceMaxLoad} — cannot happen any more.
|
||||
* What this method still buys, now that {@code maxLoad} can no longer cause it: it never
|
||||
* re-evaluates a condition {@link #place} already decided, and it closes the window between
|
||||
* that decision and the spawn in which the underlying state (another spawn landing on the same
|
||||
* profile, a config reload) could otherwise move and make a stale explicit re-check wrong.
|
||||
*
|
||||
* <p>Deliberately does not retry on {@link PeerUnreachableException} across candidates the way
|
||||
* {@link #spawn(SpawnRequest)}'s blank-profile branch does: retrying here would silently
|
||||
* re-place the caller onto a different profile than the one {@code decision} named, behind the
|
||||
* back of a caller that may already have provisioned something (a worktree's {@code repoRoot},
|
||||
* parity overlay) specifically for that name. A caller that wants the composite's own failover
|
||||
* should call {@link #spawn(SpawnRequest)} with a blank profile directly, not resolve through
|
||||
* {@link #place} first. Losing that retry on a resolve-then-spawn path is an accepted, unrelated
|
||||
* cost — see {@code SessionManager.acquireWithWorktree}'s own comment on it — never widened by
|
||||
* this round to include {@code maxLoad}, which is what round 1 actually lost.
|
||||
*/
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req, PlacementDecision decision) {
|
||||
HerdrPeerLauncher d = route(decision.profile());
|
||||
SpawnRequest routedReq = req.withProfile(decision.profile());
|
||||
PeerHandle handle = d.spawn(routedReq);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
return route(req.profileName()).effectiveCwd(req);
|
||||
@@ -441,6 +783,32 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return route(profileName).parityOverlay(profileName);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: the single live read that answers both "is the models.allow: gate
|
||||
* armed" and "which models are off", off the exact same accessor ({@link #models0()}) {@link
|
||||
* #enforceModelEnabled} and {@link #modelOffProfiles} read — so {@code fleet_profiles}/{@code
|
||||
* GET /profiles} (via {@link PeerLauncher#disabledModels()}, which now delegates here) can
|
||||
* never report a different answer than the gate enforces (the fleetd #404 lesson), and the
|
||||
* startup log line built from this can never disagree with either.
|
||||
*
|
||||
* <p>{@link #models0()} itself normalizes a {@code null} {@link #models} read to the shared
|
||||
* {@link #NO_MODELS_CONFIGURED} sentinel — deliberately the one object no config-supplied
|
||||
* {@code Models} instance can ever be identical to, since it is private to this class — so
|
||||
* comparing by reference here recovers exactly the fact {@code models0()}'s normalization
|
||||
* would otherwise erase: whether the live source was {@code null} (no {@code models:} block,
|
||||
* armed = false) or a real, config-supplied block (armed = true, even one whose {@code allow:}
|
||||
* is itself empty or absent — {@link FleetConfig.Models}'s "absent or empty allow: is off"
|
||||
* wording governs config-load validation, a distinct question from whether this gate is armed
|
||||
* for reporting).
|
||||
*/
|
||||
@Override
|
||||
public PeerLauncher.ModelGateState modelGateState() {
|
||||
FleetConfig.Models m = models0();
|
||||
return m == NO_MODELS_CONFIGURED
|
||||
? PeerLauncher.ModelGateState.notConfigured()
|
||||
: PeerLauncher.ModelGateState.armed(m.offIds());
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) {
|
||||
HerdrPeerLauncher d = spawnedBy.get(id);
|
||||
@@ -552,9 +920,21 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return byProfile.keySet();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>fleetd #425: reports the <em>live</em> {@code dev} pool's first entry — the same value
|
||||
* {@link #defaultProfileFor} computes for {@link MemberRole#DEV} — not the {@link
|
||||
* #defaultProfile} field captured at construction. An unqualified {@code fleet_spawn} defaults
|
||||
* to {@code MemberRole#DEV} (see {@link dev.ltms.fleet.peer.SpawnRequest}), so "the dev pool's
|
||||
* live first entry" is exactly the profile such a spawn actually lands on right now — the
|
||||
* question {@code fleet_profiles}' {@code "default"} field exists to answer. The frozen field is
|
||||
* a role-agnostic fallback used only when {@link #poolFor} has nothing to report at all (no
|
||||
* profiles configured), which {@link #defaultProfileFor} already handles.
|
||||
*/
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return defaultProfile;
|
||||
return defaultProfileFor(MemberRole.DEV);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -13,6 +13,7 @@ import java.nio.file.attribute.PosixFilePermissions;
|
||||
import java.time.Duration;
|
||||
import java.time.Instant;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
@@ -28,33 +29,85 @@ import java.util.stream.Stream;
|
||||
* control lacked: herdr applies that overlay BEFORE the shell starts, so any sourced file can undo
|
||||
* it — and did.
|
||||
*
|
||||
* <p><b>Which file is last depends on the platform, so the scrub runs from two of them.</b> zsh
|
||||
* <p><b>zsh reads its four startup files under three different conditions, so no single file is
|
||||
* guaranteed to run — the scrub has to cover the gap between them, not just the platforms.</b> zsh
|
||||
* reads {@code .zshenv} always, {@code .zprofile} and {@code .zlogin} only for a LOGIN shell, and
|
||||
* {@code .zshrc} only for an INTERACTIVE one. herdr does not open the same kind of shell
|
||||
* everywhere — measured on herdr 0.8.0: macOS panes run {@code -zsh} (login, so {@code .zlogin}
|
||||
* runs), Linux panes run a plain {@code /usr/bin/zsh} (interactive but NOT login, so
|
||||
* {@code .zlogin} never runs at all). A scrub in {@code .zlogin} alone is therefore a control that
|
||||
* silently does nothing on Linux — the exact failure this class exists to remove, one platform
|
||||
* over.
|
||||
* {@code .zshrc} only for an INTERACTIVE one. A pane shell that is at least one of login or
|
||||
* interactive is covered by sourcing the scrub from {@code .zshrc} and {@code .zlogin} (below), but
|
||||
* a pane shell that is NEITHER reads only {@code .zshenv} and stops — fleetd #388, measured: a herdr
|
||||
* pane can be neither login nor interactive, and such a pane read {@code .zshenv}, never reached
|
||||
* {@code scrub.zsh}, and left no report at all. A bare {@code argv[0]} of {@code /usr/bin/zsh}
|
||||
* proves the shell is NOT a login shell; it says nothing about whether it is interactive, so it
|
||||
* must never be read as "therefore interactive" — that wrong inference is what let #388 ship.
|
||||
*
|
||||
* <p>So both {@code .zshrc} and {@code .zlogin} source the same generated {@code scrub.zsh} after
|
||||
* sourcing their {@code $HOME} counterpart. On Linux only the first fires; on macOS both do, and
|
||||
* the second pass is deliberate rather than merely harmless — it re-scrubs anything the operator's
|
||||
* own {@code ~/.zlogin} exported after {@code .zshrc} had finished. Re-running is idempotent: a
|
||||
* name already blank is blanked again, and the report is rewritten with the same counts.
|
||||
* <p>So {@code .zshenv} carries a THIRD pass, guarded by the exact condition that defines the gap:
|
||||
* {@code [[ ! -o login && ! -o interactive ]]}. That guard is why this pass cannot double-scrub a
|
||||
* pane that {@code .zshrc} or {@code .zlogin} will also cover — one of {@code -o login}/
|
||||
* {@code -o interactive} is always true there, so the {@code .zshenv} pass never fires for them, and
|
||||
* their own unconditional sourcing is untouched. The guard also carries a sentinel
|
||||
* ({@value #SCRUB_SENTINEL}) so it fires once per PANE and not once per PROCESS: {@code .zshenv} is
|
||||
* read by every zsh a member's own tooling forks (a plain {@code zsh -c '...'} for a single
|
||||
* command is itself neither login nor interactive), and those children inherit variables their
|
||||
* parent deliberately set for them (git hooks get {@code GIT_DIR}, a venv gets
|
||||
* {@code VIRTUAL_ENV}, a build tool gets {@code NODE_OPTIONS} or {@code JAVA_TOOL_OPTIONS}).
|
||||
* Re-scrubbing every such child would blank all of that, and would also make the pane's own
|
||||
* {@code scrub-report.txt} — rewritten on every pass — describe whichever child exited last
|
||||
* instead of the pane. The sentinel is exported only AFTER {@code scrub.zsh} runs, so the pass
|
||||
* that sets it never sees it and cannot blank it; it must also be on the scrub's own allow-list
|
||||
* (see {@link #generate(Path, Set)}) so a later pass, in the same pane, cannot blank it back to
|
||||
* empty — an exported-but-empty sentinel reads as unset to the {@code -z} guard and would silently
|
||||
* re-enable scrubbing for every subsequent child of that pane.
|
||||
*
|
||||
* <p>So all three of {@code .zshenv} (gap only, guarded), {@code .zshrc}, and {@code .zlogin}
|
||||
* source the same generated {@code scrub.zsh} after sourcing their {@code $HOME} counterpart. A
|
||||
* login-and-interactive pane runs the {@code .zshrc} and {@code .zlogin} passes, and the second is
|
||||
* deliberate rather than merely harmless — it re-scrubs anything the operator's own
|
||||
* {@code ~/.zlogin} exported after {@code .zshrc} had finished. A pane that is neither runs only the
|
||||
* {@code .zshenv} pass. Re-running is idempotent: a name already blank is blanked again, and the
|
||||
* report is rewritten with the same counts.
|
||||
*
|
||||
* <p>Each generated file sources its {@code $HOME} counterpart FIRST, so {@code PATH} and every
|
||||
* toolchain binary still resolve exactly as the operator configured them; only afterwards does
|
||||
* {@code .zlogin} run the scrub: every EXPORTED variable not on the derived allow-list is re-exported
|
||||
* toolchain binary still resolve exactly as the operator configured them; only afterwards does the
|
||||
* scrub run: every EXPORTED variable not on the derived allow-list is re-exported
|
||||
* blank. Blank, not credential-shaped-pattern-filtered: a pattern list ({@code *TOKEN*}, …) is an
|
||||
* enumeration and misses what it did not think of — a username is the other half of a credential and
|
||||
* is shaped like none. Credential-SHAPED names among the blanked set go to the WARN log only,
|
||||
* never to the control.
|
||||
*
|
||||
* <p>The scrub also writes {@code scrub-report.txt} into its own directory: one {@code allowed N of
|
||||
* M} line (N = exports left untouched, M = exports present when the scrub ran), then the blanked
|
||||
* NAMES — never values. The launcher reads this back at teardown and logs it, because a blocked
|
||||
* count next to an unknown denominator is not a finding.
|
||||
* M failed F} line (N = exports left untouched, M = exports present when the scrub ran, F = names
|
||||
* the scrub attempted to blank but could not), then the NAMES — blanked ones bare, unblankable ones
|
||||
* {@code !}-prefixed — never values. The launcher reads this back at teardown and logs it, because a
|
||||
* blocked count next to an unknown denominator is not a finding.
|
||||
*
|
||||
* <p><b>fleetd #394:</b> plain {@code export "$n="} is a FATAL error for a zsh read-only or special
|
||||
* parameter (for example {@code UID}) — it aborts the whole sourced file, so every name still to
|
||||
* come is never blanked and the report above is never written at all. The blanking loop instead
|
||||
* routes each attempt through {@code eval}, which contains that error to the single iteration: the
|
||||
* loop always finishes, and a name that could not be blanked is counted as {@code failed} and
|
||||
* listed {@code !}-prefixed rather than silently disappearing. This is deliberately not a skip-list
|
||||
* of known-bad names — every enumerated name is still attempted, so a name nobody has thought of
|
||||
* yet still gets tried and, if it fails, still gets counted.
|
||||
*
|
||||
* <p>The blanking loop also re-asserts, on its own, the same {@code [A-Za-z_][A-Za-z0-9_]*} shape
|
||||
* check the enumeration loop already applied. Before {@code eval} was introduced a non-conforming
|
||||
* name reaching {@code export "$n="} was harmless either way — the quoting made it inert. With
|
||||
* {@code eval}, the name is spliced into a string and interpreted as shell syntax, so the enumeration
|
||||
* loop's check is no longer sufficient on its own to keep that call site safe — it is a guard on a
|
||||
* different loop, and the two must not silently drift apart. Re-checking right before the
|
||||
* {@code eval} keeps that call site safe by its own reading, independent of whatever the enumeration
|
||||
* loop does or stops doing in a later change.
|
||||
*
|
||||
* <p><b>fleetd #400:</b> {@code eval}'s exit status is not proof that the blank actually happened.
|
||||
* zsh coerces a bare {@code NAME=} assignment on an integer special parameter (measured on macOS zsh
|
||||
* 5.9: {@code SECONDS}, {@code RANDOM}, {@code SHLVL}, {@code HISTSIZE}, {@code COLUMNS},
|
||||
* {@code LINES}, {@code USERNAME}) to a number instead of failing — {@code eval} returns success,
|
||||
* the value is untouched, and a status-based classification reports it as blanked when it was not.
|
||||
* The fix classifies on the observed effect instead: after the attempt, the name's value is read
|
||||
* back with the {@code (P)} indirection flag and the decision is made from whether that is now
|
||||
* empty. This one check covers all three shapes a name can take at this point — a genuine blank, a
|
||||
* fatal read-only error {@code eval} merely contained, and this silent no-op — and the exit status
|
||||
* plays no part in the decision at all.
|
||||
*/
|
||||
public final class EnvAllowListScrub {
|
||||
|
||||
@@ -63,7 +116,11 @@ public final class EnvAllowListScrub {
|
||||
/** Name of the report file written into the generated directory by the scrub itself. */
|
||||
static final String REPORT_FILE = "scrub-report.txt";
|
||||
|
||||
/** The scrub body, generated once and sourced from both {@code .zshrc} and {@code .zlogin}. */
|
||||
/**
|
||||
* The scrub body, generated once and sourced from {@code .zshrc} and {@code .zlogin}
|
||||
* unconditionally, and from {@code .zshenv} when the pane shell is neither login nor
|
||||
* interactive (fleetd #388) — see the class javadoc.
|
||||
*/
|
||||
static final String SCRUB_FILE = "scrub.zsh";
|
||||
|
||||
/** Prefix of every generated directory — also what {@link #reapOrphans} matches on. */
|
||||
@@ -79,14 +136,42 @@ public final class EnvAllowListScrub {
|
||||
private static final String SOURCE_SCRUB =
|
||||
"source \"$ZDOTDIR/" + SCRUB_FILE + "\"\n";
|
||||
|
||||
/**
|
||||
* fleetd #388: marks a pane, not a process, as already scrubbed. Set only by the guarded
|
||||
* {@code .zshenv} pass (see {@link #NEITHER_LOGIN_NOR_INTERACTIVE_SCRUB}) after
|
||||
* {@code scrub.zsh} has run, so it must also be folded into that pass's own allow-list — see
|
||||
* the class javadoc's "must also be on the scrub's own allow-list" paragraph.
|
||||
*/
|
||||
static final String SCRUB_SENTINEL = "_CB633_SCRUBBED";
|
||||
|
||||
/**
|
||||
* Appended to {@code .zshenv}, after its {@code $HOME} source: the third pass, guarded on the
|
||||
* exact condition that defines the gap {@code .zshrc}/{@code .zlogin} do not cover — a shell
|
||||
* that is neither login nor interactive. The sentinel export happens only once the scrub has
|
||||
* already run, and only for as long as the current pane's environment has not been rebuilt from
|
||||
* scratch (a fresh {@code env -i} child would not inherit it — that is out of scope here, since
|
||||
* such a child is no longer running under the pane's own environment at all).
|
||||
*/
|
||||
private static final String NEITHER_LOGIN_NOR_INTERACTIVE_SCRUB =
|
||||
"if [[ ! -o login && ! -o interactive && -z \"${" + SCRUB_SENTINEL + ":-}\" ]]; then\n"
|
||||
+ " " + SOURCE_SCRUB
|
||||
+ " export " + SCRUB_SENTINEL + "=1\n"
|
||||
+ "fi\n";
|
||||
|
||||
private EnvAllowListScrub() {
|
||||
}
|
||||
|
||||
/**
|
||||
* A parsed {@code scrub-report.txt}: how many exported variables existed when the scrub ran,
|
||||
* how many were left untouched (allowed), and the NAMES that were blanked. Values never appear.
|
||||
* A parsed {@code scrub-report.txt}: how many exported variables existed when the scrub ran
|
||||
* ({@code total}), how many were left untouched ({@code allowed}), how many the scrub attempted
|
||||
* to blank but could not ({@code failed} — fleetd #394: a zsh read-only/special parameter such
|
||||
* as {@code UID} fatally errors on plain {@code export NAME=}, so those attempts go through
|
||||
* {@code eval} instead so the loop keeps going and the failure is counted rather than left
|
||||
* invisible), and the NAMES in each of the latter two categories. {@code allowed +
|
||||
* blanked.size() + unblankable.size() == total}, and {@code unblankable.size() == failed}.
|
||||
* Values never appear.
|
||||
*/
|
||||
record ScrubReport(int allowed, int total, List<String> blanked) {
|
||||
record ScrubReport(int allowed, int total, int failed, List<String> blanked, List<String> unblankable) {
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -105,11 +190,16 @@ public final class EnvAllowListScrub {
|
||||
reapOrphans(parentDir);
|
||||
Path dir = Files.createTempDirectory(parentDir, DIR_PREFIX);
|
||||
dir.toFile().deleteOnExit();
|
||||
// The report is written by zsh, after these hooks are registered, so register its path
|
||||
// too — otherwise the directory is non-empty at JVM exit and cannot be removed at all.
|
||||
dir.resolve(REPORT_FILE).toFile().deleteOnExit();
|
||||
write(dir, SCRUB_FILE, scrubScript(allowedNames));
|
||||
write(dir, ".zshenv", homeSourcingFile(".zshenv"));
|
||||
// zsh truncates this pre-created receipt after these hooks are registered. Register its
|
||||
// path too — otherwise the directory is non-empty at JVM exit and cannot be removed.
|
||||
Files.createFile(dir.resolve(REPORT_FILE)).toFile().deleteOnExit();
|
||||
// fleetd #388: scrub.zsh's OWN allow-list must also keep SCRUB_SENTINEL, or a later
|
||||
// pass in the same pane blanks it back to empty and the .zshenv guard below thinks it
|
||||
// was never scrubbed — see the class javadoc.
|
||||
Set<String> namesForScrubScript = new HashSet<>(allowedNames);
|
||||
namesForScrubScript.add(SCRUB_SENTINEL);
|
||||
write(dir, SCRUB_FILE, scrubScript(namesForScrubScript));
|
||||
write(dir, ".zshenv", homeSourcingFile(".zshenv") + NEITHER_LOGIN_NOR_INTERACTIVE_SCRUB);
|
||||
write(dir, ".zprofile", homeSourcingFile(".zprofile"));
|
||||
write(dir, ".zshrc", homeSourcingFile(".zshrc") + SOURCE_SCRUB);
|
||||
write(dir, ".zlogin", homeSourcingFile(".zlogin") + SOURCE_SCRUB);
|
||||
@@ -147,12 +237,11 @@ public final class EnvAllowListScrub {
|
||||
* into it: owner keeps full access, {@code group} gets traverse+read on the directory ({@code
|
||||
* rwxr-x---}, so a member process — a login shell reading it via {@code ZDOTDIR}, or another
|
||||
* process simply opening a file under it — running under that group can find and read the
|
||||
* files) and read-only on each file ({@code rw-r-----}) — deliberately no group WRITE anywhere,
|
||||
* since a member never needs to add or change what fleetd generated. (For the ZDOTDIR scrub
|
||||
* specifically, this also means the scrub script's own report write inside the pane fails
|
||||
* closed rather than open — see {@code scrub.zsh}'s trailing {@code 2>/dev/null} — which {@link
|
||||
* dev.ltms.fleet.member.HerdrPeerLauncher#releaseZdotdir} already treats as "cannot be
|
||||
* confirmed to have run" rather than success.)
|
||||
* files) and read-only on each file ({@code rw-r-----}), except the pre-created ZDOTDIR
|
||||
* {@code scrub-report.txt}. That receipt gets group write ({@code rw-rw----}), so
|
||||
* {@code scrub.zsh} can truncate and write it without granting group write on the directory.
|
||||
* If its optional permission change fails, the member cannot write a receipt and the launcher
|
||||
* keeps its existing WARN rather than failing the spawn.
|
||||
*
|
||||
* <p>Package-private and named generically on purpose: fleetd #213 built this for the ZDOTDIR
|
||||
* scrub directory, and fleetd #219 reuses it verbatim for {@link
|
||||
@@ -167,7 +256,15 @@ public final class EnvAllowListScrub {
|
||||
setGroupAndPermissions(dir, principal, "rwxr-x---");
|
||||
try (Stream<Path> entries = Files.list(dir)) {
|
||||
for (Path file : entries.toList()) {
|
||||
setGroupAndPermissions(file, principal, "rw-r-----");
|
||||
if (REPORT_FILE.equals(file.getFileName().toString())) {
|
||||
try {
|
||||
setGroupAndPermissions(file, principal, "rw-rw----");
|
||||
} catch (IOException | UnsupportedOperationException ignored) {
|
||||
// The receipt is optional. Its absence keeps the existing WARN path.
|
||||
}
|
||||
} else {
|
||||
setGroupAndPermissions(file, principal, "rw-r-----");
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (IOException e) {
|
||||
@@ -181,6 +278,33 @@ public final class EnvAllowListScrub {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #285: share ONE file with {@code group}, read-only ({@code rw-r-----}) — the same
|
||||
* per-file mode {@link #shareWithGroup} applies to a directory's entries, and the same error
|
||||
* shapes, but without touching a parent directory. Used for a file fleetd writes into a
|
||||
* directory it does NOT own — {@code configDir}'s own traversal permissions stay the
|
||||
* operator's setup — where the directory-wide {@link #shareWithGroup} would be wrong.
|
||||
*
|
||||
* @throws UncheckedIOException when {@code group} does not resolve on this host, the
|
||||
* filesystem has no POSIX group ownership, or a
|
||||
* group-ownership/permission call is refused
|
||||
*/
|
||||
static void shareFileWithGroup(Path file, String group) {
|
||||
try {
|
||||
GroupPrincipal principal = file.getFileSystem().getUserPrincipalLookupService()
|
||||
.lookupPrincipalByGroupName(group);
|
||||
setGroupAndPermissions(file, principal, "rw-r-----");
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot share generated file " + file + " with group '"
|
||||
+ group + "' — the group must exist, and the fleetd operator ("
|
||||
+ System.getProperty("user.name") + ") must be a member of it", e);
|
||||
} catch (UnsupportedOperationException e) {
|
||||
throw new UncheckedIOException("cannot share generated file " + file + " with group '"
|
||||
+ group + "' — this filesystem does not support POSIX group ownership",
|
||||
new IOException(e));
|
||||
}
|
||||
}
|
||||
|
||||
private static void setGroupAndPermissions(Path path, GroupPrincipal group, String perms) throws IOException {
|
||||
PosixFileAttributeView view = Files.getFileAttributeView(path, PosixFileAttributeView.class);
|
||||
if (view == null) {
|
||||
@@ -218,11 +342,13 @@ public final class EnvAllowListScrub {
|
||||
}
|
||||
return """
|
||||
# generated by fleetd (CB-633 memberCredentials policy=allow-list) — do not edit.
|
||||
# Sourced from .zshrc and again from .zlogin, each time AFTER that file has sourced
|
||||
# its $HOME counterpart — so this runs after everything the operator sourced, on a
|
||||
# login shell (macOS panes) and on a plain interactive one (Linux panes) alike.
|
||||
# Running twice is idempotent and deliberate: the second pass catches anything
|
||||
# ~/.zlogin exported after ~/.zshrc had finished.
|
||||
# Sourced unconditionally from .zshrc and again from .zlogin, each time AFTER that
|
||||
# file has sourced its $HOME counterpart — so this runs after everything the
|
||||
# operator sourced, on any pane that is login and/or interactive. Running twice is
|
||||
# idempotent and deliberate: the second pass catches anything ~/.zlogin exported
|
||||
# after ~/.zshrc had finished. Also sourced, once, from a guarded pass in .zshenv
|
||||
# (fleetd #388) when the pane shell is NEITHER login nor interactive — the one gap
|
||||
# those two files do not cover.
|
||||
|
||||
typeset -A _cb633_allowed
|
||||
for _cb633_n in %s; do _cb633_allowed[$_cb633_n]=1; done
|
||||
@@ -244,15 +370,57 @@ public final class EnvAllowListScrub {
|
||||
_cb633_blank+=("$_cb633_n")
|
||||
done
|
||||
|
||||
{ for _cb633_n in "${_cb633_blank[@]}"; do export "$_cb633_n="; done; } 2>/dev/null
|
||||
# fleetd #394: plain `export "$n="` is FATAL for a zsh read-only/special parameter
|
||||
# (e.g. UID) and aborts this whole sourced file — every name still to come is never
|
||||
# blanked, and the report below is never written, silently. `eval` contains that
|
||||
# error to the single iteration instead: it still fails for that one name, but the
|
||||
# loop continues and we can tell allowed / blanked / unblankable apart afterwards.
|
||||
# This is not a skip-list of known-bad names (that would miss the next one nobody
|
||||
# thought of) — every name in _cb633_blank is still attempted, unconditionally.
|
||||
# Every name reaching this loop already passed the identical identifier check in the
|
||||
# enumeration loop above — but that guard is 20 lines away in a different loop, and
|
||||
# this line is about to splice the name into a string handed to `eval`. Before this
|
||||
# fix the name only ever reached `export` quoted ("$n="), which is inert on a
|
||||
# non-identifier string either way; `eval` makes THIS line the only thing standing
|
||||
# between such a string and code execution in the member's pane, so it re-asserts the
|
||||
# same check on its own rather than trusting a guard it does not own. Under normal
|
||||
# operation this can never fire (the enumeration guard already filtered everything
|
||||
# reaching _cb633_blank), so a name caught here is counted as unblankable rather than
|
||||
# silently dropped — it is real evidence that the upstream guard was bypassed.
|
||||
#
|
||||
# fleetd #400: the attempt's own exit status is NOT proof of its effect. zsh coerces
|
||||
# a bare `NAME=` assignment on an integer special parameter (SECONDS, RANDOM, SHLVL,
|
||||
# HISTSIZE, COLUMNS, LINES, USERNAME on this host) to a number instead of failing —
|
||||
# `eval` returns 0, the value is untouched, and the old exit-status check reported it
|
||||
# as blanked when it was not. Classify on the observed effect instead: attempt the
|
||||
# export, then read the name's value back with the `(P)` indirection flag and decide
|
||||
# from whether it is now empty. One check then covers all three shapes a name can
|
||||
# take here — a genuine blank, a fatal read-only error `eval` merely contained, and
|
||||
# this silent no-op — without the exit status entering the decision at all.
|
||||
typeset -a _cb633_ok _cb633_unblankable
|
||||
_cb633_ok=()
|
||||
_cb633_unblankable=()
|
||||
for _cb633_n in "${_cb633_blank[@]}"; do
|
||||
if [[ ! "$_cb633_n" =~ ^[A-Za-z_][A-Za-z0-9_]*$ ]]; then
|
||||
_cb633_unblankable+=("$_cb633_n")
|
||||
continue
|
||||
fi
|
||||
eval "export ${_cb633_n}=" 2>/dev/null
|
||||
if [[ -z "${(P)_cb633_n}" ]]; then
|
||||
_cb633_ok+=("$_cb633_n")
|
||||
else
|
||||
_cb633_unblankable+=("$_cb633_n")
|
||||
fi
|
||||
done
|
||||
|
||||
integer _cb633_kept=$(( _cb633_total - ${#_cb633_blank} ))
|
||||
{
|
||||
print -r -- "allowed $_cb633_kept of $_cb633_total"
|
||||
for _cb633_n in "${_cb633_blank[@]}"; do print -r -- "$_cb633_n"; done
|
||||
print -r -- "allowed $_cb633_kept of $_cb633_total failed ${#_cb633_unblankable}"
|
||||
for _cb633_n in "${_cb633_ok[@]}"; do print -r -- "$_cb633_n"; done
|
||||
for _cb633_n in "${_cb633_unblankable[@]}"; do print -r -- "!$_cb633_n"; done
|
||||
} > "$ZDOTDIR/%s" 2>/dev/null
|
||||
|
||||
unset _cb633_allowed _cb633_names _cb633_blank _cb633_n _cb633_total _cb633_kept
|
||||
unset _cb633_allowed _cb633_names _cb633_blank _cb633_ok _cb633_unblankable _cb633_n _cb633_total _cb633_kept
|
||||
""".formatted(names, MemberEnvAllowList.zshCasePattern(), REPORT_FILE);
|
||||
}
|
||||
|
||||
@@ -266,6 +434,11 @@ public final class EnvAllowListScrub {
|
||||
* Read and parse {@link #REPORT_FILE} out of a generated ZDOTDIR directory. Returns {@code null}
|
||||
* when absent or unreadable (the pane may have been torn down before its login shell ever got to
|
||||
* the scrub) — callers treat that as "no measurement available", never as success.
|
||||
*
|
||||
* <p>First line is {@code "allowed <N> of <M> failed <F>"} (fleetd #394 added the trailing
|
||||
* {@code failed <F>} — a count of names the scrub attempted to blank but could not, e.g. a zsh
|
||||
* read-only/special parameter). Every following non-blank line is a name: a bare name was
|
||||
* blanked, a {@code !}-prefixed name was attempted and failed. Values never appear on either.
|
||||
*/
|
||||
static ScrubReport readReport(Path zdotdir) {
|
||||
Path report = zdotdir.resolve(REPORT_FILE);
|
||||
@@ -278,17 +451,24 @@ public final class EnvAllowListScrub {
|
||||
return null;
|
||||
}
|
||||
String[] parts = lines.getFirst().substring("allowed ".length()).trim().split("\\s+");
|
||||
if (parts.length != 3 || !"of".equals(parts[1])) {
|
||||
if (parts.length != 5 || !"of".equals(parts[1]) || !"failed".equals(parts[3])) {
|
||||
return null;
|
||||
}
|
||||
List<String> blanked = new ArrayList<>();
|
||||
List<String> unblankable = new ArrayList<>();
|
||||
for (int i = 1; i < lines.size(); i++) {
|
||||
if (!lines.get(i).isBlank()) {
|
||||
blanked.add(lines.get(i));
|
||||
String line = lines.get(i);
|
||||
if (line.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
if (line.startsWith("!")) {
|
||||
unblankable.add(line.substring(1));
|
||||
} else {
|
||||
blanked.add(line);
|
||||
}
|
||||
}
|
||||
return new ScrubReport(Integer.parseInt(parts[0]), Integer.parseInt(parts[2]),
|
||||
List.copyOf(blanked));
|
||||
Integer.parseInt(parts[4]), List.copyOf(blanked), List.copyOf(unblankable));
|
||||
} catch (IOException | NumberFormatException e) {
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -3,11 +3,13 @@ package dev.ltms.fleet.member;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.Tab;
|
||||
import dev.ltms.fleet.herdr.Workspace;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.StatusRefiner;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.CharterReceipt;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
@@ -15,6 +17,7 @@ import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.PlacementDecision;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -115,9 +118,11 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
*
|
||||
* <p>fleetd #185 stage 2: that mirroring assumption holds only while the member pane runs under
|
||||
* the SAME OS user as the daemon. When {@code memberHerdrSocket:} is configured, member panes
|
||||
* run on a second herdr owned by a different user — different {@code $HOME}, different {@code
|
||||
* secrets.sh}, different environment entirely — so this field's data no longer describes what a
|
||||
* member pane inherits. See {@link #logCredentialGap} for how that mode is handled.
|
||||
* are routed to a second herdr, and fleetd has no channel to confirm what OS user that herdr
|
||||
* runs as — it may be a different user with a different {@code $HOME} and {@code secrets.sh},
|
||||
* or the same one the daemon runs as. Either way this field's data can no longer be trusted to
|
||||
* describe what a member pane inherits. See {@link #logCredentialGap} for how that mode is
|
||||
* handled.
|
||||
*/
|
||||
private final Supplier<Set<String>> hostEnvNames;
|
||||
|
||||
@@ -151,6 +156,14 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
private final LongSupplier nowMillis; // monotonic clock (injectable for tests)
|
||||
private final Runnable sleeper; // sleep/wait hook (injectable for tests; never real-sleep in unit tests)
|
||||
|
||||
/**
|
||||
* fleetd #176 fix 2: the same UNKNOWN-refinement {@code dev.ltms.fleet.inject.StatusPoller}
|
||||
* uses, reused here for the spawn-readiness gate. Constructed once from {@link #agents} — see
|
||||
* {@link #refinedInjectable(String, Agent)} for the corroboration that keeps it from firing on
|
||||
* a dead pane's bare shell prompt.
|
||||
*/
|
||||
private final StatusRefiner statusRefiner;
|
||||
|
||||
// Per-process token mixed into each peer name so a fresh process (nameSeq back at 0) cannot
|
||||
// collide with same-profile peers that outlived a restart. See startUniquelyNamed.
|
||||
private final String nameNonce = String.format("%06x", new SecureRandom().nextInt(1 << 24));
|
||||
@@ -174,7 +187,11 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
*/
|
||||
private final ConcurrentMap<String, Path> zdotdirByPane = new ConcurrentHashMap<>();
|
||||
|
||||
/** Guards {@link #warnNonZsh} to one WARN per launcher instance, not one per spawn. */
|
||||
/**
|
||||
* Guards {@link #warnNonZsh} to one WARN per launcher instance, not one per spawn. fleetd #155:
|
||||
* only the {@code policy: deny-by-default} path still warns on a non-zsh shell — the {@code
|
||||
* policy: allow-list} path refuses the spawn instead (see {@link #applyEnvironmentAllowListPolicy}).
|
||||
*/
|
||||
private final AtomicBoolean nonZshShellWarned = new AtomicBoolean();
|
||||
/** Live config provides URI environment names that must never enter member panes. */
|
||||
private final Supplier<FleetConfig> config;
|
||||
@@ -272,6 +289,7 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
this.memberCredentials = memberCredentials;
|
||||
this.hostEnvNames = hostEnvNames != null ? hostEnvNames : () -> System.getenv().keySet();
|
||||
this.config = config;
|
||||
this.statusRefiner = new StatusRefiner(agents);
|
||||
}
|
||||
|
||||
// --- adapter seams -------------------------------------------------------------------------
|
||||
@@ -347,6 +365,42 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
return Files.isRegularFile(candidate) ? candidate : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code cwd} is a fleetd-provisioned git worktree — signalled the same way
|
||||
* {@code ClaudeCodeLauncher#writeIdeOverlay} already gates on: a {@code .git} that is a
|
||||
* <strong>regular file</strong> holding a {@code gitdir:} pointer, as opposed to a real
|
||||
* checkout's {@code .git} <strong>directory</strong>. {@code null}/blank never qualifies.
|
||||
*
|
||||
* <p>Shared by every write (and, since fleetd #249, every identity read) that must land only
|
||||
* in a worktree fleetd itself created for a member — never in a real checkout, an arbitrary
|
||||
* configured directory, or (see the incident below) the daemon's own fallback cwd. Package-
|
||||
* private (not {@code protected}) on purpose: {@link ClaudeCodeLauncher} and
|
||||
* {@link OpenCodeLauncher} both call it, and same-package visibility is enough — no subclass
|
||||
* outside this package needs it.
|
||||
*
|
||||
* <p><b>fleetd #149 incident.</b> {@code ClaudeCodeLauncher#seedTrustDialog} originally ran
|
||||
* unconditionally on any non-blank {@code cwd}. Most of that launcher's OWN tests spawn a
|
||||
* profile with no {@code cwd} configured, so the base class's {@code resolveCwd} falls
|
||||
* through to the real {@code user.dir} — and with no {@code configDir} either (also the
|
||||
* common case in that file's fixtures), the seed's target falls through the same way to the
|
||||
* real {@code ~/.claude.json}. Running this repo's own test suite corrupted the operator's
|
||||
* actual config file (it shrank from ~72 KB to a single seeded entry) the first time a
|
||||
* mutation happened to make the write non-additive. Gating both cwd-targeted writes on "this
|
||||
* is a worktree fleetd provisioned" — exactly the population fleetd #149 describes
|
||||
* ({@code worktree: true} always lands in a brand-new directory) — makes that class of write
|
||||
* impossible against a real checkout or an untouched fallback cwd, in production or in tests.
|
||||
*
|
||||
* <p><b>fleetd #249.</b> The same reasoning extends to a READ: {@code
|
||||
* OpenCodeSessionDiscovery#sessionIdForDirectory} keys on {@code directory}, a heuristic that
|
||||
* is only reliable when the directory is unique to this member — i.e., exactly the population
|
||||
* this gate identifies. {@link OpenCodeLauncher} uses it to withhold {@code agentSessionId()}
|
||||
* (report absence rather than a guess) and to refuse a {@code resumeSessionId} spawn that
|
||||
* cannot be resolved reliably going forward.
|
||||
*/
|
||||
static boolean isProvisionedWorktree(String cwd) {
|
||||
return cwd != null && !cwd.isBlank() && Files.isRegularFile(Path.of(cwd, ".git"));
|
||||
}
|
||||
|
||||
// --- profile surface -----------------------------------------------------------------------
|
||||
|
||||
/** The configured peer profile names (what {@code spawn(profile)} accepts). */
|
||||
@@ -541,6 +595,23 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
req.sessionName(), spawned.agentSessionId(), spawned.receipt());
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>fleetd #450: re-enters {@link #spawn(SpawnRequest)} with {@code decision}'s profile named
|
||||
* explicitly. This is the re-entering form the interface javadoc describes for a launcher with
|
||||
* no placement concept of its own — an instance of this class spawns a single adapter's own
|
||||
* profile set by explicit name only ({@link #place}/{@link #defaultProfileFor} are unoverridden
|
||||
* here and just wrap {@link #defaultProfile()}); it does no quarantine/cool-off/maxLoad/model-off
|
||||
* filtering of its own to re-apply. That filtering lives one layer up, in {@code
|
||||
* CompositePeerLauncher}, which is the launcher that routes across more than one profile and
|
||||
* therefore overrides this method with the routing form instead.
|
||||
*/
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req, PlacementDecision decision) {
|
||||
return spawn(req.withProfile(decision.profile()));
|
||||
}
|
||||
|
||||
/** The herdr daemon that owns this launcher's pane coordinates. */
|
||||
public HerdrClient herdr() {
|
||||
return agents.herdr();
|
||||
@@ -644,7 +715,20 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
if (paneId == null) {
|
||||
throw new IllegalStateException("pane.split returned no pane — cannot start a peer");
|
||||
}
|
||||
Agent peer = startUniquelyNamed(cfg, argv, paneId).agent();
|
||||
Agent peer;
|
||||
try {
|
||||
peer = startUniquelyNamed(cfg, argv, paneId).agent();
|
||||
} catch (RuntimeException e) {
|
||||
// The peer never started — don't leave the pane we just created orphaned.
|
||||
// Best-effort cleanup; never let it mask the real spawn failure.
|
||||
try {
|
||||
stop(paneId);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to close orphaned pane {} after spawn error: {}",
|
||||
paneId, cleanup.getMessage());
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
log.info("{} started pane={} terminal={}", namePrefix, peer.paneId(), peer.terminalId());
|
||||
return peer;
|
||||
}
|
||||
@@ -860,20 +944,31 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* {@link #reapOrphanWorkers() orphan-reap} and spawn-gate-timeout paths, plus any caller that
|
||||
* passes a pane directly, keep working without an owning id.
|
||||
*
|
||||
* <p>Resolves the tab from the pane <em>before</em> closing it. An already-gone pane/tab
|
||||
* (repeated DELETE, crashed peer) is treated as success; any other failure propagates so a
|
||||
* genuinely failed teardown is not reported as done.
|
||||
* <p>Resolves the tab from the pane <em>before</em> closing it. {@code agents.close} (the pane)
|
||||
* is the one step whose failure means the teardown itself may not have happened: an already-gone
|
||||
* pane (repeated DELETE, crashed peer) is treated as success, but any other failure propagates so
|
||||
* a genuinely failed teardown is not reported as done. {@code spaces.closeTab} (fleetd #293) is
|
||||
* different — by the time it runs the pane is already closed, so it is cosmetic workspace tidying
|
||||
* rather than a real teardown failure, and a failure there is logged and never propagates, so it
|
||||
* cannot mask the two cleanups below it ({@link #releaseZdotdir}, and the caller's worktree
|
||||
* removal in {@code SessionManager.release}).
|
||||
*/
|
||||
@Override
|
||||
public void stop(String idOrPane) {
|
||||
// Teardown knows only the pane, not which profile spawned it. Attempt tab cleanup when any
|
||||
// profile uses tab placement (so the bridge may have created a dedicated peer tab); the
|
||||
// single-occupant check below is what actually protects the user's shared tabs.
|
||||
// Teardown knows only the pane, not which profile spawned it — and, when this call is
|
||||
// routed here through CompositePeerLauncher's single-daemon stop() shortcut (fleetd #342),
|
||||
// not even which adapter's config actually governed the spawn: the shortcut can hand the
|
||||
// pane to a delegate that never spawned it, whose own profiles say nothing about how THIS
|
||||
// pane was placed. So the decision to look for a tab to clean up is made from the pane's
|
||||
// actual state, not from this delegate's static profile config: resolve the tab
|
||||
// unconditionally and let {@link WorkspaceControl#locatePane} tolerate "not found" (it
|
||||
// returns null rather than throwing); the single-occupant check below is what actually
|
||||
// protects the user's shared tabs, exactly as it always has.
|
||||
String paneId = paneByAgentId.remove(idOrPane);
|
||||
if (paneId == null) {
|
||||
paneId = idOrPane; // raw-pane fallback (reap, gate timeout, pane-addressed callers)
|
||||
}
|
||||
WorkspaceControl.PaneLocation loc = usesTabPlacement() ? spaces.locatePane(paneId) : null;
|
||||
WorkspaceControl.PaneLocation loc = spaces.locatePane(paneId);
|
||||
try {
|
||||
agents.close(paneId);
|
||||
} catch (HerdrException e) {
|
||||
@@ -881,7 +976,25 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
log.debug("pane.close({}) ignored — already gone: {}", paneId, e.getMessage());
|
||||
}
|
||||
if (loc != null && loc.tabPaneCount() == 1) {
|
||||
spaces.closeTab(loc.tabId());
|
||||
// fleetd #293: the pane above is already closed by this point, so a failing tab.close is
|
||||
// cosmetic workspace tidying, not a real teardown failure — it must not mask the two
|
||||
// cleanups below it (releaseZdotdir, and the caller's worktree removal). Unlike
|
||||
// agents.close above, this is not narrowed to "already gone": any failure here, whatever
|
||||
// its cause, is one we continue past, so we log it at WARN (not debug) with the tab id a
|
||||
// person can go close by hand.
|
||||
try {
|
||||
spaces.closeTab(loc.tabId());
|
||||
} catch (RuntimeException e) {
|
||||
// Caught as RuntimeException, not HerdrException, to match releaseZdotdir's own
|
||||
// guard five lines below. Today the two are the same set — HerdrCodec wraps every
|
||||
// encode/decode failure and UnixSocketHerdrClient wraps every IOException, so
|
||||
// HerdrException is all closeTab can actually throw. Narrowing to it anyway would
|
||||
// leave this step guarded against the expected failure and bare against any other,
|
||||
// which is the exact asymmetry fleetd #293 exists to remove. No behaviour change
|
||||
// today; it stops a later change inside WorkspaceControl.closeTab reopening it.
|
||||
log.warn("tab.close({}) failed — the pane is already torn down, so continuing; the "
|
||||
+ "tab may need manual cleanup: {}", loc.tabId(), e.getMessage());
|
||||
}
|
||||
} else if (loc != null) {
|
||||
log.debug("not closing tab {} — it holds {} panes (not a dedicated peer tab)",
|
||||
loc.tabId(), loc.tabPaneCount());
|
||||
@@ -896,11 +1009,6 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
/** Whether any configured profile places peers in their own tab (so tabs may need cleanup). */
|
||||
private boolean usesTabPlacement() {
|
||||
return profiles.values().stream().anyMatch(FleetConfig.Profile::tabPlacement);
|
||||
}
|
||||
|
||||
/** True when a herdr error means the target is already gone (safe to treat as done). */
|
||||
private static boolean isAlreadyGone(HerdrException e) {
|
||||
return e.code() != null && e.code().endsWith("_not_found");
|
||||
@@ -909,16 +1017,44 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
// --- spawn-readiness gate (CB-306) ---------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Poll {@link AgentControl#status} until the pane reports an injectable state or the configured
|
||||
* Poll {@link AgentControl#get} until the pane reports an injectable state or the configured
|
||||
* timeout elapses. On timeout, close the pane (self-reap) and throw.
|
||||
*
|
||||
* <p>fleetd #176 fix 1: {@code agents.get} was previously unguarded here, so a herdr
|
||||
* {@code *_not_found} answer — which is what happens when the backend process EXITED rather
|
||||
* than being slow — propagated as a raw {@link HerdrException} instead of the
|
||||
* {@link PeerUnreachableException} every other failure path of this gate produces, and skipped
|
||||
* teardown ({@link #stop}) entirely, leaking the pane/tab. {@link #failFastOnGoneBackend} closes
|
||||
* that gap: it stops waiting immediately (never burns the rest of the timeout), runs the same
|
||||
* teardown the timeout path below runs, and throws with a message that says the backend exited
|
||||
* rather than that the pane was slow. Any other {@link HerdrException} still propagates
|
||||
* unchanged — this gate does not interpret or recover from it, but it still closes the pane
|
||||
* it opened before handing the exception to its caller.
|
||||
*/
|
||||
private void waitUntilInjectableOrThrow(String paneId) {
|
||||
long deadline = nowMillis.getAsLong() + spawnReadyTimeoutMs;
|
||||
Object lastStatus = null;
|
||||
long start = nowMillis.getAsLong();
|
||||
long deadline = start + spawnReadyTimeoutMs;
|
||||
AgentStatus lastStatus = null;
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
var status = agents.status(paneId);
|
||||
lastStatus = status;
|
||||
if (status.injectable()) {
|
||||
Agent sample;
|
||||
try {
|
||||
sample = agents.get(paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (isAlreadyGone(e)) {
|
||||
failFastOnGoneBackend(paneId, e, nowMillis.getAsLong() - start);
|
||||
}
|
||||
// This gate must not interpret an unrelated herdr error, but the caller does not
|
||||
// receive paneId when spawn throws. Close the pane here before propagating e unchanged.
|
||||
try {
|
||||
stop(paneId);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to close orphaned pane {} after readiness-gate error: {}",
|
||||
paneId, cleanup.getMessage());
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
lastStatus = sample.status();
|
||||
if (lastStatus.injectable() || refinedInjectable(paneId, sample)) {
|
||||
log.debug("peer pane={} reached injectable state", paneId);
|
||||
return;
|
||||
}
|
||||
@@ -937,6 +1073,66 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
+ spawnReadyTimeoutMs + "ms");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176 fix 1: the backend process exited while the gate was still waiting — herdr
|
||||
* answered {@code *_not_found} instead of ever reporting an injectable status. Fails
|
||||
* immediately (never burns the rest of {@link #spawnReadyTimeoutMs}), runs the exact same
|
||||
* teardown {@link #waitUntilInjectableOrThrow}'s timeout path runs, and throws a
|
||||
* {@link PeerUnreachableException} whose message says the process exited rather than that the
|
||||
* pane was slow.
|
||||
*
|
||||
* @throws PeerUnreachableException always — this method never returns normally
|
||||
*/
|
||||
private void failFastOnGoneBackend(String paneId, HerdrException cause, long elapsedMs) {
|
||||
String tail = readPaneQuietly(paneId); // read before stop() closes the pane
|
||||
log.warn("peer pane={} backend process exited after {}ms while waiting for injectable "
|
||||
+ "state (herdr: {}) — closing. Pane tail:\n{}",
|
||||
paneId, elapsedMs, cause.getMessage(), tail);
|
||||
stop(paneId);
|
||||
throw new PeerUnreachableException(
|
||||
"worker pane " + paneId + " backend process exited after " + elapsedMs
|
||||
+ "ms while waiting to become injectable (herdr reported: " + cause.getMessage()
|
||||
+ "). Pane tail:\n" + tail);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176 fix 2: resolve a raw {@link AgentStatus#UNKNOWN} sample into a trustworthy
|
||||
* injectable state via pane content, the same refinement {@code StatusPoller} applies during a
|
||||
* peer's working life — but corroborated, because the trap this gate is exposed to that the
|
||||
* poller is not: a pane whose backend has already exited settles at a plain shell prompt, and
|
||||
* that prompt commonly contains the same {@code ❯} glyph {@link StatusRefiner#classify} treats
|
||||
* as "idle at the Claude Code TUI prompt". Naively wiring the refiner in would turn "the backend
|
||||
* died" into "ready to inject" — strictly worse than today's timeout.
|
||||
*
|
||||
* <p>Two guards, both required:
|
||||
* <ul>
|
||||
* <li>{@link StatusRefiner#classify} is written for the Claude Code TUI only (its own javadoc
|
||||
* says so), so refinement only ever runs for the {@code claude} adapter — never for
|
||||
* {@code opencode} or any future non-Claude backend, whatever its pane looks like.</li>
|
||||
* <li>The refined status is accepted only when the <em>same</em> {@code agents.get} sample
|
||||
* ({@code sample}, taken once by the caller) still reports a non-null
|
||||
* {@link Agent#agentType()} — herdr's own "a supported backend is still detected here"
|
||||
* signal, read from the very sample the raw status came from so the two can never
|
||||
* disagree. A bare shell prompt left by an exited backend reports no {@code agentType}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Called only when {@code sample.status() == UNKNOWN}, so a healthy spawn — which never sees
|
||||
* {@code UNKNOWN} — triggers zero extra herdr calls; only a persistently-{@code unknown} pane
|
||||
* pays the one extra {@code agent.read} {@link StatusRefiner#refine} performs.
|
||||
*/
|
||||
private boolean refinedInjectable(String paneId, Agent sample) {
|
||||
if (sample.status() != AgentStatus.UNKNOWN) {
|
||||
return false;
|
||||
}
|
||||
if (!"claude".equals(namePrefix)) {
|
||||
return false; // StatusRefiner.classify reads a Claude Code TUI prompt specifically
|
||||
}
|
||||
if (sample.agentType() == null) {
|
||||
return false; // no corroborating liveness signal — could be a dead pane's bare shell
|
||||
}
|
||||
return statusRefiner.refine(paneId, AgentStatus.UNKNOWN, agents).injectable();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #220: the pane's recent output, clipped, for the readiness-gate timeout log — or a
|
||||
* short note when it cannot be read. Best-effort by construction: this runs on a path that is
|
||||
@@ -1129,6 +1325,11 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
if (!creds.isAllowList()) {
|
||||
overlayBlockedCredentials(workerEnv, creds);
|
||||
logCredentialGap(creds, null);
|
||||
// fleetd #155: this overlay itself does not depend on the shell (it lands in the
|
||||
// pane-creation env map before any shell runs), so the spawn is never refused here —
|
||||
// only policy=allow-list's ZDOTDIR scrub needs a login shell to run at all. Still worth
|
||||
// telling the operator: the stronger post-shell control is unavailable on this shell.
|
||||
warnNonZsh(memberLoginShell());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1181,10 +1382,16 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* dev.ltms.fleet.session.Worktrees#shareWithGroup} already uses, reused rather than
|
||||
* inventing a second group key. Either one missing means the scrub cannot be guaranteed
|
||||
* reachable by the member, which is the same "cannot guarantee the scrub runs" case as a
|
||||
* non-zsh shell, so it gets the identical fallback.</li>
|
||||
* non-zsh shell — that one still gets the fallback (see fleetd #155's javadoc note below
|
||||
* on why a missing worktreeRoot/worktreeGroup is a different defect, out of that ticket's
|
||||
* scope).</li>
|
||||
* </ul>
|
||||
* In every branch: never refuse to spawn. A degraded credential control must not become an
|
||||
* outage for an opt-in feature.
|
||||
* fleetd #155: the one exception to "never refuse to spawn" is a non-zsh login shell, right
|
||||
* below. The operator asked for {@code policy: allow-list} specifically — its whole point is a
|
||||
* control a sourced file cannot undo — so silently degrading to the weaker overlay is exactly
|
||||
* the "silently does nothing" failure this ticket exists to remove. Every other branch in this
|
||||
* method (worktreeRoot/worktreeGroup missing) keeps the old "degrade, never refuse" behaviour;
|
||||
* that gap is real but is fleetd #213's scope, not this one.
|
||||
*/
|
||||
private Path applyEnvironmentAllowListPolicy(FleetConfig.Profile cfg, Launch launch) {
|
||||
FleetConfig.MemberCredentials creds = memberCredentials == null ? null : memberCredentials.get();
|
||||
@@ -1198,20 +1405,21 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
// must not even be called on this path, only the explicit memberLoginShell: config can
|
||||
// answer it. With memberHerdrSocket absent, nothing here changes: fleetd's own $SHELL is
|
||||
// still the input, exactly as before this fix.
|
||||
String loginShell = memberHerdrSocket ? configuredMemberLoginShell() : resolveEnv("SHELL");
|
||||
String loginShell = memberLoginShell();
|
||||
boolean zsh = isZshShell(loginShell);
|
||||
if (!zsh) {
|
||||
// A non-zsh login shell ignores ZDOTDIR entirely: NO scrub would run, so pretending
|
||||
// otherwise would be worse than saying so. Warn loudly and fall back to the CB-596
|
||||
// sentinel overlay over the enumerated known: names — weaker (a sourced file can undo
|
||||
// it), but strictly better than nothing. Deliberately no "allowed N of M" line here: the
|
||||
// scrub this count describes does not run on this path, so printing it would tell an
|
||||
// operator that a fraction of names were blocked when the real number blocked is zero.
|
||||
// logCredentialGap's WARN (below) is the only signal for this path.
|
||||
warnNonZsh(loginShell);
|
||||
overlayBlockedCredentials(launch.env(), creds);
|
||||
logCredentialGap(creds, null);
|
||||
return null;
|
||||
// fleetd #155: a non-zsh login shell ignores ZDOTDIR entirely — NO scrub would run. The
|
||||
// operator explicitly asked for policy=allow-list's blocking control, so degrading to
|
||||
// the weaker CB-596 overlay and spawning anyway would be the same "control silently does
|
||||
// nothing" defect this ticket exists to close. Refuse instead — the caller (FleetMcp.spawn)
|
||||
// catches IllegalArgumentException and surfaces the message to the operator.
|
||||
throw new IllegalArgumentException(
|
||||
"memberCredentials policy=allow-list requires the member's login shell to be "
|
||||
+ "zsh, so the ZDOTDIR scrub can run after it — refusing to spawn under "
|
||||
+ "login shell '" + (loginShell == null ? "<unset>" : loginShell)
|
||||
+ "'. Configure memberLoginShell: as a zsh path (only read when "
|
||||
+ "memberHerdrSocket is set), move the member's OS account onto zsh, or "
|
||||
+ "set memberCredentials.policy: deny-by-default instead.");
|
||||
}
|
||||
Path parentDir;
|
||||
String group = null;
|
||||
@@ -1266,6 +1474,20 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
return cfg == null ? null : cfg.memberLoginShell();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #155: the member pane's login shell, using the same {@code memberHerdrSocket} routing
|
||||
* {@link #applyEnvironmentAllowListPolicy} already used — fleetd's own {@code $SHELL} decides it
|
||||
* when {@code memberHerdrSocket} is absent (today's only mode); the configured {@code
|
||||
* memberLoginShell:} decides it otherwise, since fleetd's own {@code $SHELL} names a different
|
||||
* user's shell once member panes run under a different OS user. Shared by both {@link
|
||||
* #applyEnvironmentAllowListPolicy} (allow-list: refuses on non-zsh) and {@link
|
||||
* #applyMemberCredentialPolicy} (deny-by-default: warns on non-zsh) so the two policies agree on
|
||||
* what "the member's shell" means.
|
||||
*/
|
||||
private String memberLoginShell() {
|
||||
return memberHerdrSocketConfigured() ? configuredMemberLoginShell() : resolveEnv("SHELL");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213: {@code worktreeRoot}, as the ZDOTDIR scrub's parent directory under {@code
|
||||
* memberHerdrSocket}, or {@code null} when unconfigured — the same "cannot guarantee the scrub
|
||||
@@ -1315,9 +1537,9 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
Set<String> brokerUriEnvNames = brokerUriEnvNames();
|
||||
Set<String> allowed = new java.util.TreeSet<>(
|
||||
MemberEnvAllowList.derive(profiles.values(), creds.allowSet(), brokerUriEnvNames));
|
||||
if (creds.sshAuthSockAllowed()) {
|
||||
if (creds.sshAgentEnvInherited()) {
|
||||
allowed.add(SSH_AUTH_SOCK);
|
||||
} // blocked by default: absent from the set ⇒ blanked by the scrub like any other name
|
||||
} // omitted by default: absent from the set ⇒ blanked by the scrub like any other name
|
||||
allowed.addAll(launch.env().keySet());
|
||||
allowed.removeAll(brokerUriEnvNames);
|
||||
return allowed;
|
||||
@@ -1339,10 +1561,27 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* survive {@code allowed} (including the {@code LC_*} prefix rule). Neither number is a constant:
|
||||
* both come from the actual derived set and the actual environment this spawn sees. Never logs a
|
||||
* variable NAME or VALUE — only the counts.
|
||||
*
|
||||
* <p><strong>Under {@code memberHerdrSocket} the counts describe fleetd's own process, not the
|
||||
* member's</strong> (fleetd #269 follow-up), so the message says so rather than leaving the
|
||||
* reader to infer it from this javadoc, which the operator reading the log never sees.
|
||||
*/
|
||||
private void logAllowListCoverage(Set<String> allowed) {
|
||||
Set<String> hostNames = hostEnvNames.get();
|
||||
long kept = hostNames.stream().filter(name -> MemberEnvAllowList.keeps(allowed, name)).count();
|
||||
if (memberHerdrSocketConfigured()) {
|
||||
// fleetd #269 covered the sibling line below (logCredentialGap) and stopped there.
|
||||
// This line has the same problem: read plainly, "allowed 7 of 39" is a statement about
|
||||
// the member's pane, and under memberHerdrSocket it is not -- the pane is routed to a
|
||||
// second herdr whose environment fleetd cannot inspect. The counts stay useful, so
|
||||
// this is not a WARN and not a refusal; only the claim is narrowed to what is true.
|
||||
log.info("member credentials: allowed {} of {} names in fleetd's OWN environment — "
|
||||
+ "memberHerdrSocket is configured, so member panes are routed to a "
|
||||
+ "second herdr whose environment fleetd has no channel to inspect. "
|
||||
+ "These counts describe fleetd's process, NOT the member pane's.",
|
||||
kept, hostNames.size());
|
||||
return;
|
||||
}
|
||||
log.info("member credentials: allowed {} of {}", kept, hostNames.size());
|
||||
}
|
||||
|
||||
@@ -1350,15 +1589,23 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
private static final String SSH_AUTH_SOCK = MemberEnvAllowList.SSH_AUTH_SOCK;
|
||||
|
||||
/**
|
||||
* CB-633: a non-zsh login shell means the allow-list control CANNOT run — say so once per
|
||||
* launcher instance, naming the shell, instead of failing silently.
|
||||
* fleetd #155: called ONLY from the {@code policy: deny-by-default} branch of {@link
|
||||
* #applyMemberCredentialPolicy} — under {@code policy: allow-list}, a non-zsh shell now REFUSES
|
||||
* the spawn instead (see {@link #applyEnvironmentAllowListPolicy}), since that policy's whole
|
||||
* point is a control a sourced file cannot undo. deny-by-default's own overlay does not depend
|
||||
* on the shell (it lands in the pane-creation env map before any shell runs), so this is a
|
||||
* heads-up, not a defect report: say once per launcher instance, naming the shell, that the
|
||||
* stronger allow-list control is unavailable here — never silently.
|
||||
*/
|
||||
private void warnNonZsh(String shell) {
|
||||
if (isZshShell(shell)) {
|
||||
return;
|
||||
}
|
||||
if (nonZshShellWarned.compareAndSet(false, true)) {
|
||||
log.warn("memberCredentials policy=allow-list: member login shell '{}' is NOT zsh — "
|
||||
+ "ZDOTDIR scrubbing cannot run, so members' inherited environment is "
|
||||
+ "UNPROTECTED beyond the enumerated known: fallback. Move herdr onto a "
|
||||
+ "zsh account or switch policy back to deny-by-default.",
|
||||
log.warn("memberCredentials policy=deny-by-default: member login shell '{}' is NOT zsh — "
|
||||
+ "the pane-creation credential overlay still applies here (it does not "
|
||||
+ "depend on the shell), but policy=allow-list's stronger post-shell "
|
||||
+ "ZDOTDIR scrub is unavailable on this shell.",
|
||||
shell == null ? "<unset>" : shell);
|
||||
}
|
||||
}
|
||||
@@ -1369,21 +1616,24 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
/**
|
||||
* fleetd #213: {@code memberHerdrSocket} is configured and the member login shell IS zsh, but
|
||||
* {@code worktreeRoot} and/or {@code worktreeGroup} is missing, so the generated ZDOTDIR cannot
|
||||
* be placed anywhere the member's OS user can reach — {@code java.io.tmpdir} is fleetd's own
|
||||
* 0700 temp dir, unreadable by another uid, which is the exact gap this ticket exists to close.
|
||||
* Say so once per launcher instance, instead of either generating a directory nothing can read
|
||||
* (protection theatre) or refusing to spawn (turning a degraded credential control into an
|
||||
* outage for an opt-in feature).
|
||||
* be placed anywhere fleetd can be sure the member's OS user can reach — {@code java.io.tmpdir}
|
||||
* is fleetd's own 0700 temp dir, which is unreadable if the member pane runs as a different OS
|
||||
* user, and fleetd has no channel to confirm whether it does or not. Rather than gamble on that,
|
||||
* this treats memberHerdrSocket as reason enough to require an explicitly shared location, which
|
||||
* is the exact gap this ticket exists to close. Say so once per launcher instance, instead of
|
||||
* either generating a directory that might not be readable (protection theatre) or refusing to
|
||||
* spawn (turning a degraded credential control into an outage for an opt-in feature).
|
||||
*/
|
||||
private void warnCannotShareScrubDirectory() {
|
||||
if (cannotShareScrubDirWarned.compareAndSet(false, true)) {
|
||||
log.warn("memberCredentials policy=allow-list: memberHerdrSocket is configured and the "
|
||||
+ "member login shell is zsh, but worktreeRoot and/or worktreeGroup is not "
|
||||
+ "configured — the generated ZDOTDIR cannot be placed where the member's OS "
|
||||
+ "user can read it (java.io.tmpdir is fleetd's own, unreadable by another uid), "
|
||||
+ "so the scrub cannot be guaranteed to run. Falling back to the CB-596 sentinel "
|
||||
+ "overlay. Configure both worktreeRoot and worktreeGroup to enable the "
|
||||
+ "allow-list scrub under memberHerdrSocket.");
|
||||
+ "configured — the generated ZDOTDIR cannot be placed where fleetd can be sure "
|
||||
+ "the member's OS user can read it (java.io.tmpdir is fleetd's own 0700 dir, "
|
||||
+ "unreadable if the member runs as a different OS user — fleetd has no channel "
|
||||
+ "to confirm whether it does), so the scrub cannot be guaranteed to run. Falling "
|
||||
+ "back to the CB-596 sentinel overlay. Configure both worktreeRoot and "
|
||||
+ "worktreeGroup to enable the allow-list scrub under memberHerdrSocket.");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1411,11 +1661,19 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
log.warn("memberCredentials allow-list: pane {} left no scrub report in {} — the "
|
||||
+ "environment scrub cannot be confirmed to have run. Either the pane ended "
|
||||
+ "before its shell finished starting, or its shell never read our generated "
|
||||
+ "startup files, in which case that member saw the full host environment.",
|
||||
+ "startup files. Either way, we cannot tell from here whether the scrub ran, "
|
||||
+ "so we do not know what that member's environment contained.",
|
||||
paneId, dir);
|
||||
} else {
|
||||
log.info("memberCredentials allow-list: pane {} allowed {} of {} environment variables",
|
||||
paneId, report.allowed(), report.total());
|
||||
if (report.failed() > 0) {
|
||||
log.warn("memberCredentials allow-list: pane {} could not blank {} environment "
|
||||
+ "variable(s) — {} (likely a zsh read-only/special parameter) — those "
|
||||
+ "names were left in the member's environment. Confirm none of them is a "
|
||||
+ "credential.",
|
||||
paneId, report.failed(), report.unblankable());
|
||||
}
|
||||
List<String> shaped = report.blanked().stream()
|
||||
.filter(name -> CREDENTIAL_SHAPED_NAME.matcher(name).matches())
|
||||
.toList();
|
||||
@@ -1434,38 +1692,69 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
|
||||
/**
|
||||
* Guards the {@code effectiveAllowed == null} branch of {@link #logCredentialGap} — the
|
||||
* genuinely-unprotected report (deny-by-default, and the allow-list non-zsh fallback) — to one
|
||||
* WARN per launcher instance, not one per spawn.
|
||||
* genuinely-unprotected report (deny-by-default, and the allow-list non-zsh fallback) — AND
|
||||
* the {@code effectiveAllowed != null} / {@code keptByDerivedList} branch, the allow-list case
|
||||
* where a name is in the gap but the derived allow-list keeps it anyway. Both branches log the
|
||||
* same severity (WARN) about the same fact — a name genuinely reaching a member pane
|
||||
* unprotected — so they share this one guard, keyed per NAME rather than per launcher instance:
|
||||
* each credential-shaped name that is ever reported unprotected gets exactly one WARN, however
|
||||
* many spawns see it and whichever of the two branches first reports it.
|
||||
*
|
||||
* <p>fleetd #341: {@code memberCredentials} is a live, re-read-per-spawn supplier, so the
|
||||
* policy — and so the gap's actual member names — can change between two spawns on the same
|
||||
* launcher. Before this fix the guard was a single {@code AtomicBoolean} tripped by either
|
||||
* branch: spawn 1 could warn about name A and trip the flag, and a later spawn's gap containing
|
||||
* a different name B would never be reported, even though B is just as unprotected as A was.
|
||||
* {@code AtomicBoolean} could not express "once per distinct name" at all — only "once, ever,
|
||||
* for whichever name got there first" — so this is a {@code Set<String>} guard instead, the
|
||||
* same shape {@link OpenCodeLauncher#modelCheckSkippedWarned} already uses for its own
|
||||
* once-per-distinct-thing WARN. {@link #add}'s return value (true only the first time a name is
|
||||
* added) is what turns "log the whole gap" into "log only the names never warned about before".
|
||||
*
|
||||
* <p>Bounded by construction: every name added here first passed {@link
|
||||
* #CREDENTIAL_SHAPED_NAME}'s filter over {@link #hostEnvNames}, i.e. it is an actual
|
||||
* environment variable name from the daemon's own process — a small, OS-bounded set (the host
|
||||
* environment has, in practice, tens to a few hundred entries), not an attacker- or
|
||||
* request-controlled input. So this set cannot grow past "however many distinct credential-
|
||||
* shaped names this host's environment has ever held across this launcher's lifetime," which is
|
||||
* effectively fixed for the life of one daemon process — no separate cap is needed.
|
||||
*
|
||||
* <p>CB-633 follow-up (#192): kept SEPARATE from {@link #allowListGapLogged} on purpose.
|
||||
* {@code memberCredentials} is a live, re-read-per-spawn supplier, so the policy can change
|
||||
* between two spawns on the same launcher. A single shared flag would let a harmless allow-list
|
||||
* INFO on spawn 1 permanently suppress the real deny-by-default WARN a later spawn deserves —
|
||||
* the report that matters most getting hidden by the report that doesn't. Two flags mean each
|
||||
* report kind fires exactly once, independent of what the other kind already logged.
|
||||
* report kind (WARN vs. INFO) fires independently of what the other kind already logged; within
|
||||
* the WARN kind itself, the set above further separates by name, for the same reason.
|
||||
*/
|
||||
private final AtomicBoolean unprotectedGapLogged = new AtomicBoolean();
|
||||
private final Set<String> unprotectedGapNamesWarned = ConcurrentHashMap.newKeySet();
|
||||
|
||||
/**
|
||||
* Guards the {@code effectiveAllowed != null} branch of {@link #logCredentialGap} — the
|
||||
* allow-list-scrub-covered report — to one INFO per launcher instance. See {@link
|
||||
* #unprotectedGapLogged}'s javadoc for why this is a separate flag rather than a shared one.
|
||||
* #unprotectedGapNamesWarned}'s javadoc for why this is a separate flag rather than a shared
|
||||
* one; unlike that guard it stays a per-instance {@code AtomicBoolean}, not a per-name set —
|
||||
* fleetd #341 fixed the WARN-vs-WARN suppression, not this INFO's own one-shot shape, which was
|
||||
* not reported as broken and is out of that ticket's scope.
|
||||
*/
|
||||
private final AtomicBoolean allowListGapLogged = new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* fleetd #185 stage 2: guards {@link #warnUnknownMemberEnvironment} to one WARN per launcher
|
||||
* instance, not one per spawn — the same one-per-instance shape as {@link #unprotectedGapLogged}
|
||||
* and {@link #allowListGapLogged}, kept as its own flag for the same reason those two are split:
|
||||
* this mode is orthogonal to which of the other two branches would otherwise have fired.
|
||||
* instance, not one per spawn — the same one-shot shape {@link #unprotectedGapNamesWarned} and
|
||||
* {@link #allowListGapLogged} guard their own branches with, kept as its own flag for the same
|
||||
* reason those two are split: this mode is orthogonal to which of the other two branches would
|
||||
* otherwise have fired.
|
||||
*/
|
||||
private final AtomicBoolean unknownMemberEnvironmentWarned = new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* fleetd #185 stage 2: whether {@code memberHerdrSocket:} is configured, i.e. member panes run
|
||||
* on a second herdr owned by a different OS user than the daemon's own process. Re-read from the
|
||||
* live config on every call (same hot-reload shape as {@link #memberCredentials}), never cached,
|
||||
* so a config reload takes effect on the next spawn without a restart.
|
||||
* fleetd #185 stage 2: whether {@code memberHerdrSocket:} is configured, i.e. member panes are
|
||||
* routed to a second herdr. This tests only that the config key is set — fleetd has no channel
|
||||
* to confirm what OS user that second herdr runs as, so a {@code true} result means "member
|
||||
* panes may run under a different OS user," not that they do. Re-read from the live config on
|
||||
* every call (same hot-reload shape as {@link #memberCredentials}), never cached, so a config
|
||||
* reload takes effect on the next spawn without a restart.
|
||||
*
|
||||
* <p>{@link #config} is {@code null} on any call site that never threaded the full config
|
||||
* through (every production {@code HerdrPeerLauncher} does; a handful of older tests do not) —
|
||||
@@ -1488,14 +1777,15 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* fleetd #185 stage 2: the single replacement WARN for {@link #logCredentialGap}'s usual
|
||||
* conclusions when {@code memberHerdrSocket:} is configured. {@link #hostEnvNames} (and
|
||||
* everything derived from it — {@code known}/{@code allow} coverage, the allow-list scrub's
|
||||
* derived set) describes the DAEMON's own environment; under this config key member panes run as
|
||||
* a different OS user with a different environment entirely, so neither "every member pane
|
||||
* inherits them UNBLOCKED" nor "the scrub blanks them" is evidence-backed here — both would be
|
||||
* reporting on the wrong process. Logged once, names the config key, and states the honest
|
||||
* conclusion: the gap for member panes is UNKNOWN, not clean, so {@code memberCredentials} cannot
|
||||
* be verified from this daemon. The one count it does report is scoped explicitly to fleetd's own
|
||||
* environment, never presented as if it said anything about the member's — see {@link
|
||||
* #logCredentialGap}'s javadoc for why this branch exists.
|
||||
* derived set) describes the DAEMON's own environment; under this config key member panes are
|
||||
* routed to a second herdr, and fleetd has no channel to confirm what OS user that herdr runs
|
||||
* as or to read its environment, so neither "every member pane inherits them UNBLOCKED" nor
|
||||
* "the scrub blanks them" is evidence-backed here — both would be reporting on the wrong
|
||||
* process. Logged once, names the config key, and states the honest conclusion: the gap for
|
||||
* member panes is UNKNOWN, not clean, so {@code memberCredentials} cannot be verified from this
|
||||
* daemon. The one count it does report is scoped explicitly to fleetd's own environment, never
|
||||
* presented as if it said anything about the member's — see {@link #logCredentialGap}'s javadoc
|
||||
* for why this branch exists.
|
||||
*/
|
||||
private void warnUnknownMemberEnvironment(FleetConfig.MemberCredentials creds) {
|
||||
if (!unknownMemberEnvironmentWarned.compareAndSet(false, true)) {
|
||||
@@ -1508,13 +1798,13 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
.filter(name -> CREDENTIAL_SHAPED_NAME.matcher(name).matches())
|
||||
.filter(name -> !covered.contains(name))
|
||||
.count();
|
||||
log.warn("memberCredentials gap: memberHerdrSocket is configured, so member panes run under "
|
||||
+ "a different OS user than fleetd's own process, with a different environment "
|
||||
+ "entirely — fleetd has no channel to read that user's environment. {} of the "
|
||||
+ "{} names in fleetd's OWN environment are credential-shaped and not on "
|
||||
+ "known:/allow:, but that count describes fleetd's process, not the member "
|
||||
+ "herdr's. The credential gap for member panes is UNKNOWN, not clean, and "
|
||||
+ "memberCredentials cannot be verified from here.",
|
||||
log.warn("memberCredentials gap: memberHerdrSocket is configured, so member panes are routed "
|
||||
+ "to a second herdr — fleetd has no channel to confirm what OS user that herdr "
|
||||
+ "runs as, so it cannot tell whether those panes inherit its own environment or "
|
||||
+ "a different one entirely. {} of the {} names in fleetd's OWN environment are "
|
||||
+ "credential-shaped and not on known:/allow:, but that count describes fleetd's "
|
||||
+ "process, not the member herdr's. The credential gap for member panes is "
|
||||
+ "UNKNOWN, not clean, and memberCredentials cannot be verified from here.",
|
||||
gapInFleetdsOwnEnv, hostNames.size());
|
||||
}
|
||||
|
||||
@@ -1582,14 +1872,21 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
List<String> blankedByScrub = gap.stream()
|
||||
.filter(name -> !MemberEnvAllowList.keeps(effectiveAllowed, name))
|
||||
.toList();
|
||||
if (!keptByDerivedList.isEmpty() && unprotectedGapLogged.compareAndSet(false, true)) {
|
||||
// fleetd #341: filter to names this guard has never warned about before — not just
|
||||
// "isEmpty" on the whole branch — so a name this spawn's gap shares with an EARLIER
|
||||
// spawn's (already-warned) gap does not re-print, while a name unique to THIS gap still
|
||||
// does, whichever of the two WARN branches reported it first.
|
||||
List<String> newlyUnprotected = keptByDerivedList.stream()
|
||||
.filter(unprotectedGapNamesWarned::add)
|
||||
.toList();
|
||||
if (!newlyUnprotected.isEmpty()) {
|
||||
log.warn("memberCredentials gap: {} credential-shaped env var name(s) are on neither "
|
||||
+ "known: nor allow: — the derived allow-list keeps them anyway (a profile's "
|
||||
+ "gitTokenEnv/gitHostEnv/tokenEnv/env: names one, or this spawn injects it), "
|
||||
+ "so every member pane inherits them UNBLOCKED — {}. Add each to "
|
||||
+ "memberCredentials.known (or .allow if a member legitimately needs it), or "
|
||||
+ "remove it from whatever profile setting derives it in.",
|
||||
keptByDerivedList.size(), keptByDerivedList);
|
||||
newlyUnprotected.size(), newlyUnprotected);
|
||||
}
|
||||
if (!blankedByScrub.isEmpty() && allowListGapLogged.compareAndSet(false, true)) {
|
||||
log.info("memberCredentials gap: {} credential-shaped env var name(s) are on neither "
|
||||
@@ -1602,12 +1899,18 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
|
||||
/** The deny-by-default (and allow-list non-zsh fallback) WARN — unchanged byte-for-byte by #192. */
|
||||
private void warnGapUnprotected(List<String> gap) {
|
||||
if (unprotectedGapLogged.compareAndSet(false, true)) {
|
||||
// fleetd #341: same "only the names never warned before" filter as the sibling branch in
|
||||
// logCredentialGap above — see unprotectedGapNamesWarned's javadoc. Both branches share
|
||||
// this one guard because both report the exact same fact (a name reaching a member pane
|
||||
// unprotected) at the exact same severity; keying it by name is what lets a later spawn's
|
||||
// DIFFERENT name still get its own WARN after an earlier spawn's already fired.
|
||||
List<String> newlyUnprotected = gap.stream().filter(unprotectedGapNamesWarned::add).toList();
|
||||
if (!newlyUnprotected.isEmpty()) {
|
||||
log.warn("memberCredentials gap: {} credential-shaped env var name(s) are on neither "
|
||||
+ "known: nor allow: — every member pane inherits them UNBLOCKED — {}. "
|
||||
+ "Add each to memberCredentials.known (blocked by default) or .allow "
|
||||
+ "(if a member legitimately needs it).",
|
||||
gap.size(), gap);
|
||||
newlyUnprotected.size(), newlyUnprotected);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* fleetd #111 (CB-608): a single, testable read of the {@code memberCredentials:} policy — names
|
||||
* and counts only, never a value. The daemon never holds a credential's <em>value</em> in the
|
||||
* first place (only the names configured under {@code known:}/{@code allow:}), so there is
|
||||
* nothing here to redact by construction; the point of this class is that it is the ONE place
|
||||
* that turns a policy into names-and-counts, so nothing else hand-counts a second time.
|
||||
*
|
||||
* <p>Before this class, {@link dev.ltms.fleet.Fleetd#reportMemberCredentialsGap} computed these
|
||||
* same counts inline for the startup log line, and {@code scripts/probe-member-credentials.sh}
|
||||
* carried its own hardcoded {@code NAMES} array that the live policy could grow past silently
|
||||
* (#111) — the exact "hand-maintained second copy drifts" shape #114 fixed for the tool
|
||||
* catalogue. Both now read this class: the startup log via {@link
|
||||
* dev.ltms.fleet.Fleetd#reportMemberCredentialsGap}, and a live daemon via the {@code
|
||||
* GET /member-credentials} REST endpoint ({@link dev.ltms.fleet.rest.FleetApp}), which the probe
|
||||
* script fetches instead of carrying its own list.
|
||||
*
|
||||
* @param present policy configured with at least one {@code known} name. {@code false} for an
|
||||
* absent or empty {@code memberCredentials:} block — represented honestly as "no
|
||||
* policy", never as "nothing blocked" (an empty {@link #blocked} could otherwise be
|
||||
* misread as a clean bill of health).
|
||||
* @param policy the normalized policy mode ({@link FleetConfig.MemberCredentials#policy()}), or
|
||||
* {@code null} when {@link #present} is {@code false}.
|
||||
* @param known every name the policy declares, in configured order. Names only, never a value.
|
||||
* @param allowed the subset of {@link #known} explicitly let through. Names only.
|
||||
* @param blocked {@link #known} minus {@link #allowed} — the names an actual spawn shadows. Names
|
||||
* only.
|
||||
*/
|
||||
public record MemberCredentialPolicyView(boolean present, String policy, List<String> known,
|
||||
List<String> allowed, List<String> blocked) {
|
||||
|
||||
private static final MemberCredentialPolicyView ABSENT =
|
||||
new MemberCredentialPolicyView(false, null, List.of(), List.of(), List.of());
|
||||
|
||||
public MemberCredentialPolicyView {
|
||||
known = known == null ? List.of() : List.copyOf(known);
|
||||
allowed = allowed == null ? List.of() : List.copyOf(allowed);
|
||||
blocked = blocked == null ? List.of() : List.copyOf(blocked);
|
||||
}
|
||||
|
||||
/** The honest "no policy configured" view. */
|
||||
public static MemberCredentialPolicyView absent() {
|
||||
return ABSENT;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the view straight from the live config. {@code creds} may be {@code null} (no {@code
|
||||
* memberCredentials:} block at all) — treated the same as a present-but-empty block, exactly
|
||||
* like {@link dev.ltms.fleet.Fleetd#reportMemberCredentialsGap} already did.
|
||||
*/
|
||||
public static MemberCredentialPolicyView of(FleetConfig.MemberCredentials creds) {
|
||||
if (creds == null || creds.known().isEmpty()) {
|
||||
return ABSENT;
|
||||
}
|
||||
return new MemberCredentialPolicyView(true, creds.policy(), creds.known(), creds.allow(),
|
||||
List.copyOf(creds.blockedSet()));
|
||||
}
|
||||
|
||||
public int knownCount() {
|
||||
return known.size();
|
||||
}
|
||||
|
||||
public int allowedCount() {
|
||||
return allowed.size();
|
||||
}
|
||||
|
||||
public int blockedCount() {
|
||||
return blocked.size();
|
||||
}
|
||||
}
|
||||
@@ -34,7 +34,7 @@ import java.util.TreeSet;
|
||||
*
|
||||
* <p>{@code SSH_AUTH_SOCK} is deliberately NOT here. It is a handle to the operator's ssh-agent — a
|
||||
* member holding it can sign with the operator's keys — so keeping it is a config decision
|
||||
* ({@code memberCredentials.sshAuthSock: allow}), not a derivation default.
|
||||
* ({@code memberCredentials.sshAgentEnv: inherit}), not a derivation default.
|
||||
*
|
||||
* <p><b>CB-633 follow-up:</b> the union also includes {@code memberCredentials.allow:} — the
|
||||
* operator's own explicit list. Before this, {@code policy: allow-list} silently ignored every name
|
||||
@@ -42,7 +42,7 @@ import java.util.TreeSet;
|
||||
* turning the policy on could blank credentials working members already depended on. {@code
|
||||
* SSH_AUTH_SOCK} and configured broker URI environment names are exceptions: even when the operator
|
||||
* lists them under {@code allow:}, they are excluded here. {@code SSH_AUTH_SOCK} is added back ONLY
|
||||
* by the caller when {@code sshAuthSock: allow} is explicitly set
|
||||
* by the caller when {@code sshAgentEnv: inherit} is explicitly set
|
||||
* (see {@link #SSH_AUTH_SOCK}'s javadoc) — it is a live handle to the operator's own ssh-agent, not
|
||||
* a value, so treating it like any other allow-listed name would hand a member every key the
|
||||
* operator's agent holds the moment they typed the name under {@code allow:} for an unrelated
|
||||
@@ -53,7 +53,7 @@ public final class MemberEnvAllowList {
|
||||
/**
|
||||
* The operator's ssh-agent socket path. Deliberately excluded from {@link #derive}'s union of
|
||||
* {@code memberCredentials.allow:} — see the class javadoc's CB-633 follow-up note. Governed
|
||||
* ONLY by {@code memberCredentials.sshAuthSock}, never by appearing in {@code allow:}.
|
||||
* ONLY by {@code memberCredentials.sshAgentEnv}, never by appearing in {@code allow:}.
|
||||
*/
|
||||
public static final String SSH_AUTH_SOCK = "SSH_AUTH_SOCK";
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.CharterReceipt;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
@@ -20,7 +21,9 @@ import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.BooleanSupplier;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
@@ -73,6 +76,19 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
*/
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
|
||||
/**
|
||||
* fleetd #175: notified when {@link SessionAwareHandle} detects, on the same late-resolve
|
||||
* read that discovers the session id, that the live opencode session is running a DIFFERENT
|
||||
* model than the profile requested — opencode does not fail on an unknown {@code -m}, it
|
||||
* silently falls back to a default (potentially paid) model. Reused exactly as
|
||||
* {@code CompletionResolver}'s {@code BACKEND_EXHAUSTED} path uses it: this launcher supplies
|
||||
* only the herdr terminal id and a reason string; mapping target → session → profile →
|
||||
* credential stays entirely the sink's job (see {@code Fleetd.main}'s wiring). Defaults to
|
||||
* {@link ExhaustionSink#none()} for a caller (an older constructor, or a test not exercising
|
||||
* this) that opts out.
|
||||
*/
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so it
|
||||
* matches the legacy non-blocking spawn semantics. Config dirs are created under the JVM temp dir.
|
||||
@@ -132,9 +148,25 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs, spawnReadyPollMs,
|
||||
fleet, memberCredentials, config, ExhaustionSink.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor, plus the live config for URI environment exclusions and the fleetd
|
||||
* #175 model-mismatch quarantine sink.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet, memberCredentials, config);
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet, memberCredentials, config,
|
||||
exhaustionSink);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -182,6 +214,7 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
this.exhaustionSink = ExhaustionSink.none();
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -207,10 +240,29 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs, nowMillis, sleeper,
|
||||
configRoot, discoveryRoot, fleet, memberCredentials, config, ExhaustionSink.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the live config for URI environment exclusions and the
|
||||
* fleetd #175 model-mismatch quarantine sink — the constructor a test drives directly to
|
||||
* observe {@link ExhaustionSink#onExhausted} without going through {@code Fleetd.main}'s
|
||||
* wiring.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper, Path configRoot, Path discoveryRoot,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials, null, config);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
this.exhaustionSink = exhaustionSink == null ? ExhaustionSink.none() : exhaustionSink;
|
||||
}
|
||||
|
||||
private static Path defaultConfigRoot() {
|
||||
@@ -279,11 +331,22 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
+ "form — opencode's per-model context limit could not be applied for this profile",
|
||||
cfg.profile(), cfg.model());
|
||||
}
|
||||
// fleetd #393: memberSkills seeding (GitWorktrees#seedSkills) copies skill folders into
|
||||
// EVERY provisioned worktree's .claude/skills/ regardless of which kind ultimately spawns
|
||||
// into it — that copy step cannot know the kind, only the caller of GitWorktrees#add does
|
||||
// (see that method's own javadoc). .claude/skills/ is a Claude Code CLI convention the CLI
|
||||
// discovers on its own; opencode has no such discovery, so without this, a seeded skill
|
||||
// never reaches an opencode member even though GitWorktrees logged it as seeded. Read
|
||||
// whatever landed under <cwd>/.claude/skills/ here — the one place in this launcher that
|
||||
// knows both the kind (opencode, by construction: this IS OpenCodeLauncher) and the cwd.
|
||||
List<Path> skillInstructionFiles = skillInstructionFiles(spec.cwd());
|
||||
// A config file is needed for the bridge MCP mount, a member charter, the IDE MCP (+ its
|
||||
// guidance overlay, CB-634), a pinned endpoint (CB-508), or a resolvable autoCompactWindow.
|
||||
// guidance overlay, CB-634), a pinned endpoint (CB-508), a resolvable autoCompactWindow, or
|
||||
// at least one seeded skill to deliver via instructions[] (fleetd #393).
|
||||
if (cfg.hasMcp() || cfg.hasIdeMcp() || spec.charter() != null || hasCustomProvider(cfg)
|
||||
|| wantsContextLimit) {
|
||||
workerEnv.put("OPENCODE_CONFIG", writeConfig(cfg, spec.charter(), spec.cwd()).toString());
|
||||
|| wantsContextLimit || !skillInstructionFiles.isEmpty()) {
|
||||
workerEnv.put("OPENCODE_CONFIG",
|
||||
writeConfig(cfg, spec.charter(), spec.cwd(), skillInstructionFiles).toString());
|
||||
}
|
||||
applyGitToken(workerEnv, cfg);
|
||||
List<String> argv = argvWithResume(argvWithModel(argvWithAuto(cfg), cfg), spec.resumeSessionId());
|
||||
@@ -306,6 +369,65 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
return withAgent;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #393: the {@code SKILL.md} paths under {@code <cwd>/.claude/skills/} this launcher can
|
||||
* turn into {@code instructions[]} entries, plus the honest log this ticket asks for — emitted
|
||||
* here, at the one point a skill's fate for THIS spawn is actually known, rather than trusting
|
||||
* {@code GitWorktrees#seedSkills}'s kind-blind "N of M" line to mean "and it will be read."
|
||||
*
|
||||
* <p>Every non-hidden subdirectory of {@code .claude/skills/} is a candidate, whether it got
|
||||
* there via {@code memberSkills:} seeding or because the target repo ships its own copy — this
|
||||
* launcher does not care which; it only cares what it can find at spawn time. A candidate with
|
||||
* a {@code SKILL.md} at its top level (the same shape {@link #writeConfig} already requires for
|
||||
* the charter and IDE-rules instructions entries) is delivered; anything else is a directory
|
||||
* this launcher cannot turn into a flat instructions entry, named explicitly in the log rather
|
||||
* than silently dropped, so a caller sees a real "cannot consume" reason and not just a smaller
|
||||
* number than {@code GitWorktrees}' own count.
|
||||
*
|
||||
* <p>No candidates at all (directory absent or empty) logs nothing — the same
|
||||
* no-log-when-nothing-to-say shape {@link #hasCustomProvider} and friends already follow, and
|
||||
* the shape {@code GitWorktrees#seedSkills} itself uses when {@code memberSkills:} is unset.
|
||||
* A failure to even list the directory is logged and treated as "nothing delivered" — best
|
||||
* effort, must never fail the spawn, matching {@code GitWorktrees#seedSkills}'s own contract.
|
||||
*/
|
||||
private List<Path> skillInstructionFiles(String cwd) {
|
||||
if (cwd == null || cwd.isBlank()) {
|
||||
return List.of();
|
||||
}
|
||||
Path skillsDir = Path.of(cwd, ".claude", "skills");
|
||||
if (!Files.isDirectory(skillsDir)) {
|
||||
return List.of();
|
||||
}
|
||||
List<Path> candidates;
|
||||
try (var listing = Files.list(skillsDir)) {
|
||||
candidates = listing.filter(Files::isDirectory)
|
||||
.filter(p -> !p.getFileName().toString().startsWith("."))
|
||||
.sorted()
|
||||
.toList();
|
||||
} catch (IOException e) {
|
||||
log.warn("could not scan {} for skill folders to deliver to this opencode member: {}",
|
||||
skillsDir, e.getMessage());
|
||||
return List.of();
|
||||
}
|
||||
if (candidates.isEmpty()) {
|
||||
return List.of();
|
||||
}
|
||||
List<Path> delivered = candidates.stream()
|
||||
.map(dir -> dir.resolve("SKILL.md"))
|
||||
.filter(Files::isRegularFile)
|
||||
.toList();
|
||||
List<String> undeliverable = candidates.stream()
|
||||
.filter(dir -> !Files.isRegularFile(dir.resolve("SKILL.md")))
|
||||
.map(dir -> dir.getFileName().toString())
|
||||
.toList();
|
||||
log.info("skill delivery: {} of {} skill folder(s) under {} reached this opencode member via "
|
||||
+ "instructions[] (opencode does not read .claude/skills/ natively, unlike "
|
||||
+ "Claude Code){}",
|
||||
delivered.size(), candidates.size(), skillsDir,
|
||||
undeliverable.isEmpty() ? "" : "; no SKILL.md, could not be delivered: " + undeliverable);
|
||||
return delivered;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when this profile pins its own OpenAI-compatible endpoint (CB-508) rather than using
|
||||
* whatever provider opencode resolves by default.
|
||||
@@ -366,8 +488,15 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
* fresh per-spawn directory under {@link #configRoot}, and return the config file's path for
|
||||
* {@code OPENCODE_CONFIG}. The dir is unique per spawn so concurrent workers never race on it;
|
||||
* it is best-effort cleaned on JVM exit (worker config is disposable — regenerated every spawn).
|
||||
*
|
||||
* @param skillInstructionFiles fleetd #393: absolute {@code SKILL.md} paths from
|
||||
* {@link #skillInstructionFiles(String)}, appended to
|
||||
* {@code instructions[]} so a {@code memberSkills:}-seeded skill
|
||||
* reaches this opencode member the same way the charter and IDE
|
||||
* rules already do.
|
||||
*/
|
||||
private Path writeConfig(FleetConfig.Profile cfg, String charterText, String cwd) {
|
||||
private Path writeConfig(FleetConfig.Profile cfg, String charterText, String cwd,
|
||||
List<Path> skillInstructionFiles) {
|
||||
try {
|
||||
Path dir = Files.createTempDirectory(configParentDir(), "fleetd-opencode-");
|
||||
dir.toFile().deleteOnExit();
|
||||
@@ -395,7 +524,39 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
Files.writeString(charter, charterText);
|
||||
charter.toFile().deleteOnExit();
|
||||
|
||||
root.putArray("instructions").add(charter.toAbsolutePath().toString());
|
||||
// fleetd #393 follow-up: withArray, not putArray. putArray REPLACES whatever node
|
||||
// is already at "instructions"; withArray gets-or-creates. All three writers on
|
||||
// this array now use withArray so that write order stops being load-bearing for
|
||||
// whoever adds a fourth.
|
||||
//
|
||||
// Be precise about what this particular line is worth, because an earlier version
|
||||
// of this comment was wrong and claimed too much. THIS call is the one place where
|
||||
// the two idioms are equivalent, and no test can tell them apart: it runs first,
|
||||
// against a still-empty root, so there is never an existing node for putArray to
|
||||
// replace. Measured on the #393 merge: flipping this one call back to putArray
|
||||
// leaves the whole suite green (1618 tests), and always will. The edit is a
|
||||
// readability and future-proofing change with no test behind it, and that is not a
|
||||
// gap anyone can close.
|
||||
//
|
||||
// The ordering hazard is real, just not here. It is the LATER writers that can
|
||||
// destroy earlier entries. Measured on the same merge: flipping the skills writer
|
||||
// below to putArray deletes this charter entry and fails
|
||||
// OpenCodeLauncherTest.instructionsArrayHoldsCharterThenSkillsThenIdeRulesInOrder;
|
||||
// flipping the IDE-rules writer fails that test and
|
||||
// instructionsArrayHoldsCharterThenIdeRulesInOrder. Deleting this line altogether
|
||||
// fails instructionsArrayHoldsExactlyTheCharterWhenNothingElseWritesToIt — the
|
||||
// assertion that a role contract reaches an opencode member at all, which is the
|
||||
// hole that predates #393 and is what actually let the mutation hide.
|
||||
root.withArray("instructions").add(charter.toAbsolutePath().toString());
|
||||
}
|
||||
|
||||
// fleetd #393: each seeded skill's SKILL.md, delivered as a plain instructions[] entry
|
||||
// — the only mechanism opencode has for static guidance text. Unlike Claude Code's
|
||||
// Skill tool, opencode cannot load one of these on demand by name; the content is just
|
||||
// always part of the system prompt from spawn. That is a real difference in HOW the
|
||||
// content reaches the member, not a reason to withhold it.
|
||||
for (Path skillFile : skillInstructionFiles) {
|
||||
root.withArray("instructions").add(skillFile.toAbsolutePath().toString());
|
||||
}
|
||||
|
||||
if (cfg.hasMcp() || cfg.hasIdeMcp()) {
|
||||
@@ -618,12 +779,44 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
private final AtomicBoolean discoveryUnavailableWarned =
|
||||
new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* fleetd #267: one WARN per PROFILE (not per launcher instance — several profiles can each hit
|
||||
* this gap independently) for the model-mismatch check (fleetd #175) never getting to run
|
||||
* because the spawn was not given a fleetd-provisioned worktree (fleetd #249). Profile names
|
||||
* accumulate here for the life of this launcher instance and are never removed — the same
|
||||
* one-shot treatment {@link #discoveryUnavailableWarned} already gets, just keyed per profile
|
||||
* instead of globally.
|
||||
*/
|
||||
private final Set<String> modelCheckSkippedWarned = ConcurrentHashMap.newKeySet();
|
||||
|
||||
/** Add lazy on-disk session discovery to the base handle. */
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
String cwd = effectiveCwd(req);
|
||||
// fleetd #249: refuse rather than silently resume into unverifiable territory. opencode's
|
||||
// `-s <id>` flag itself resumes precisely — the resolved id is what fails, not the resume —
|
||||
// but resolvedSessionId() below can never confirm (or later re-report) this handle's own
|
||||
// identity without a fleetd-provisioned worktree (isProvisionedWorktree(cwd)), because the
|
||||
// directory is shared and sessionIdForDirectory's "most recently updated row" heuristic can
|
||||
// pick a sibling's session. Refusing here, before anything spawns, beats letting the member
|
||||
// start and only then discovering fleetd can never again verify who it actually is.
|
||||
if (req.resumeSessionId() != null && !req.resumeSessionId().isBlank()
|
||||
&& !isProvisionedWorktree(cwd)) {
|
||||
throw new IllegalArgumentException("resumeSessionId requires a fleetd-provisioned "
|
||||
+ "worktree for an opencode profile — without one, this member's cwd is shared "
|
||||
+ "with other sessions, so fleetd can never reliably confirm (now or later) which "
|
||||
+ "conversation it is actually running (fleetd #249). Pass fleet_spawn{worktree:"
|
||||
+ "<ticket-slug>} to resume this member.");
|
||||
}
|
||||
PeerHandle inner = super.spawn(req);
|
||||
return new SessionAwareHandle(inner, discovery, effectiveCwd(req),
|
||||
this::memberHerdrSocketConfigured, discoveryUnavailableWarned);
|
||||
// fleetd #175: the same profile config buildLaunch resolved for this spawn (requireProfile
|
||||
// is deterministic on req.profileName(), so re-resolving here costs a map lookup, not a
|
||||
// second decision) — SessionAwareHandle needs cfg.model() to know what THIS session should
|
||||
// be running.
|
||||
FleetConfig.Profile cfg = requireProfile(req.profileName());
|
||||
return new SessionAwareHandle(inner, discovery, cwd, cfg,
|
||||
this::memberHerdrSocketConfigured, discoveryUnavailableWarned,
|
||||
modelCheckSkippedWarned, exhaustionSink);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -633,22 +826,63 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
* are untouched — only the opencode-specific identity answer is added. {@code sessionName()}
|
||||
* stays null: opencode has no display-name seam, so the logical name lives only in the bridge's
|
||||
* roster (see the SESSION_NAME capability).
|
||||
*
|
||||
* <p>fleetd #175: {@link #agentSessionId()} is also where the model-mismatch check lives (see
|
||||
* {@link #checkModelMatch()}) — it is the one method the real late-resolve path
|
||||
* ({@code SessionManager.resolveAgentSessionId}, via {@code get}/{@code rosterResolved}/
|
||||
* {@code release}) actually calls, and only while the session id is still unknown. Putting the
|
||||
* check anywhere else risks repeating PR #203's mistake: a check that runs before opencode has
|
||||
* written the row it needs, and so never fires.
|
||||
*/
|
||||
private static final class SessionAwareHandle implements PeerHandle {
|
||||
private final PeerHandle delegate;
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
private final String cwd;
|
||||
private final FleetConfig.Profile cfg;
|
||||
private final BooleanSupplier discoveryUnavailable;
|
||||
private final AtomicBoolean discoveryUnavailableWarned;
|
||||
private final Set<String> modelCheckSkippedWarned;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
/** CAS'd true the first (and only) time a model mismatch is reported for this handle. */
|
||||
private final AtomicBoolean modelMismatchReported = new AtomicBoolean();
|
||||
/**
|
||||
* The session id, once {@link OpenCodeSessionDiscovery#sessionIdForDirectory} first
|
||||
* resolves a non-null answer for this handle (fleetd #234). Sticky on purpose: {@code
|
||||
* directory} is a shared-cwd heuristic (see {@link OpenCodeSessionDiscovery}'s class
|
||||
* javadoc) that can start returning a DIFFERENT row once another session shares the same
|
||||
* directory and writes a newer one — re-deriving it on every call would let this handle's
|
||||
* identity silently drift to a sibling's session. Once resolved, this IS the answer, and
|
||||
* {@link #checkModelMatch} reads only the row this id names, never "whatever is newest in
|
||||
* the directory right now."
|
||||
*/
|
||||
private final AtomicReference<String> resolvedSessionId = new AtomicReference<>();
|
||||
/**
|
||||
* fleetd #249: whether {@link #cwd} is a fleetd-provisioned git worktree
|
||||
* ({@link HerdrPeerLauncher#isProvisionedWorktree}), computed once at spawn time since
|
||||
* {@code cwd} never changes for this handle. When {@code false} the directory is shared
|
||||
* with other sessions (the default no-worktree spawn inherits the lead's own cwd), so
|
||||
* {@link OpenCodeSessionDiscovery#sessionIdForDirectory}'s "most recently updated row for
|
||||
* this directory" heuristic can and does pick another session's row — see that class's
|
||||
* javadoc. {@link #agentSessionId()} refuses to guess in that case: it reports absent
|
||||
* rather than a possibly-foreign id.
|
||||
*/
|
||||
private final boolean worktreeProvisioned;
|
||||
|
||||
SessionAwareHandle(PeerHandle delegate, OpenCodeSessionDiscovery discovery, String cwd,
|
||||
FleetConfig.Profile cfg,
|
||||
BooleanSupplier discoveryUnavailable,
|
||||
AtomicBoolean discoveryUnavailableWarned) {
|
||||
AtomicBoolean discoveryUnavailableWarned,
|
||||
Set<String> modelCheckSkippedWarned,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this.delegate = delegate;
|
||||
this.discovery = discovery;
|
||||
this.cwd = cwd;
|
||||
this.cfg = cfg;
|
||||
this.discoveryUnavailable = discoveryUnavailable;
|
||||
this.discoveryUnavailableWarned = discoveryUnavailableWarned;
|
||||
this.modelCheckSkippedWarned = modelCheckSkippedWarned;
|
||||
this.exhaustionSink = exhaustionSink;
|
||||
this.worktreeProvisioned = isProvisionedWorktree(cwd);
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -678,6 +912,9 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
// built from (see OpenCodeLauncher#defaultDiscoveryRoot's javadoc for the full
|
||||
// reasoning). Scanning fleetd's own $HOME under that config would only ever find "no
|
||||
// row" and read as "resume unsupported" — declare it unavailable instead, once, loudly.
|
||||
// Checked before the fleetd #249 worktree gate below: this OS-user mismatch makes
|
||||
// discovery unusable regardless of whether cwd happens to be a provisioned worktree, so
|
||||
// it earns the one-time WARN either way.
|
||||
if (discoveryUnavailable.getAsBoolean()) {
|
||||
if (discoveryUnavailableWarned.compareAndSet(false, true)) {
|
||||
log.warn("opencode session discovery unavailable: memberHerdrSocket is "
|
||||
@@ -689,11 +926,123 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
}
|
||||
return null;
|
||||
}
|
||||
// fleetd #249: cwd is shared with other sessions unless fleetd itself provisioned this
|
||||
// worktree, and sessionIdForDirectory's directory-keyed heuristic cannot tell this
|
||||
// member's row apart from a sibling's in that case (measured: a three-day-old row from
|
||||
// a different profile). Refuse to guess — absent is the honest answer, and it is what
|
||||
// this codebase already returns elsewhere for absent evidence (fleetd #175's UNKNOWN).
|
||||
// This IS the ordinary, expected shape of the large majority of spawns (no worktree
|
||||
// requested), not a configuration gap — but fleetd #267 found that same shape silently
|
||||
// switches off the fleetd #175 model-mismatch check for those spawns too, since
|
||||
// checkModelMatch's only call site is right below this gate. The check cannot be moved
|
||||
// off agentSessionId()'s resolved id: the id is the only safe way to key
|
||||
// actualModelForSessionId to THIS session's own row rather than "whatever is newest in
|
||||
// the shared directory" (fleetd #234) — re-deriving a second, independent answer via
|
||||
// `directory` here would reintroduce exactly the false-positive risk #234 fixed (a
|
||||
// sibling's differently-configured model looking like THIS profile's mismatch). So the
|
||||
// model genuinely is unknowable without a provisioned worktree, and unlike the silence
|
||||
// this branch used to keep, that gap now gets the same one-time, per-profile WARN
|
||||
// treatment discoveryUnavailable already gets above — but keyed by profile, since
|
||||
// several profiles can each hit this independently.
|
||||
if (!worktreeProvisioned) {
|
||||
if (cfg.model() != null && !cfg.model().isBlank()
|
||||
&& modelCheckSkippedWarned.add(cfg.profile())) {
|
||||
log.warn("opencode model-mismatch check (fleetd #175) cannot run for profile "
|
||||
+ "'{}': it was spawned without a fleetd-provisioned worktree (fleetd "
|
||||
+ "#249), so its cwd may be shared with other sessions and the actual "
|
||||
+ "model it is running cannot be safely told apart from a sibling's — "
|
||||
+ "spawn with worktree:true to enable the check for this profile.",
|
||||
cfg.profile());
|
||||
}
|
||||
return null;
|
||||
}
|
||||
// fleetd #234: once resolved, stay resolved. Re-deriving from `directory` on every call
|
||||
// would let this handle's identity drift to a sibling session that later shares the
|
||||
// same cwd and writes a newer row — see resolvedSessionId's javadoc.
|
||||
String cached = resolvedSessionId.get();
|
||||
if (cached != null) {
|
||||
return cached;
|
||||
}
|
||||
// Lazy + retried, never a spawn-time blocker: opencode writes the session record only
|
||||
// when the session is first persisted, so null here is the correct interim answer and
|
||||
// the caller re-calls later (each call re-scans, picking up a record that has since
|
||||
// appeared).
|
||||
return discovery.sessionIdForDirectory(cwd);
|
||||
String id = discovery.sessionIdForDirectory(cwd);
|
||||
if (id != null) {
|
||||
resolvedSessionId.compareAndSet(null, id);
|
||||
}
|
||||
// fleetd #175: check on the SAME tick — while the caller (SessionManager's late-resolve
|
||||
// step) is still re-polling because the id is unknown, the row this id came from (once
|
||||
// it exists) is exactly the row that also carries the actual model. Once id resolves,
|
||||
// the caller stops calling agentSessionId() for this session, so this is naturally a
|
||||
// once-only check that happens right when the row first appears.
|
||||
checkModelMatch(id);
|
||||
return id;
|
||||
}
|
||||
|
||||
/**
|
||||
* Verify the live opencode session is running the model {@link #cfg} requested (fleetd
|
||||
* #175) and, on a real mismatch, log an ERROR and quarantine through {@link
|
||||
* #exhaustionSink}. A no-op when there is nothing to compare against — no model configured,
|
||||
* already reported once for this handle, {@code sessionId} itself is not resolved yet
|
||||
* (fleetd #234: absent evidence, not a mismatch), or the actual model is still UNKNOWN (no
|
||||
* row yet, unreadable database, or unparseable evidence). UNKNOWN must never be treated as
|
||||
* a mismatch: that is the single most important safety rule here — a false positive would
|
||||
* quarantine a perfectly working profile's credential.
|
||||
*
|
||||
* @param sessionId the id {@link #agentSessionId()} just resolved (or had cached) for THIS
|
||||
* handle — the model is read back for this exact session (fleetd #234's
|
||||
* {@link OpenCodeSessionDiscovery#actualModelForSessionId}), never
|
||||
* re-derived from {@code directory}
|
||||
*/
|
||||
private void checkModelMatch(String sessionId) {
|
||||
if (modelMismatchReported.get() || cfg.model() == null || cfg.model().isBlank()) {
|
||||
return;
|
||||
}
|
||||
if (sessionId == null || sessionId.isBlank()) {
|
||||
return; // id not resolved yet — UNKNOWN, never a mismatch (fleetd #175's rule)
|
||||
}
|
||||
OpenCodeSessionDiscovery.ActualModel actual = discovery.actualModelForSessionId(sessionId);
|
||||
if (actual == null) {
|
||||
return; // UNKNOWN evidence — never a mismatch
|
||||
}
|
||||
String[] requestedParts = splitProviderModel(cfg.model());
|
||||
String requestedProvider = requestedParts == null ? null : requestedParts[0];
|
||||
String requestedId = requestedParts == null ? cfg.model() : requestedParts[1];
|
||||
boolean idMatches = requestedId.equals(actual.id());
|
||||
// Compare the provider ONLY when BOTH sides have one. requestedProvider == null covers
|
||||
// a bare profile model with no "/" — the profile never asked for a specific provider.
|
||||
// actual.provider() == null covers opencode's model JSON having an id but no providerID
|
||||
// (a real shape parseModel accepts) — that is missing evidence, not a contradiction, and
|
||||
// acceptance rule 4 says missing evidence is UNKNOWN, never a mismatch. Narrowing the
|
||||
// provider comparison this way keeps the id comparison (the part that actually caught the
|
||||
// xf bug) fully intact — a genuine id mismatch is still caught either way (fleetd #175
|
||||
// review round 2).
|
||||
boolean providerMatches = requestedProvider == null || actual.provider() == null
|
||||
|| requestedProvider.equals(actual.provider());
|
||||
if (idMatches && providerMatches) {
|
||||
return;
|
||||
}
|
||||
if (!modelMismatchReported.compareAndSet(false, true)) {
|
||||
return; // another thread already reported this exact mismatch
|
||||
}
|
||||
String actualDisplay = actual.provider() == null
|
||||
? actual.id() : actual.provider() + "/" + actual.id();
|
||||
log.error("opencode profile '{}' requested model '{}' but the live session is actually "
|
||||
+ "running '{}' — opencode does not fail on an unknown -m, it silently "
|
||||
+ "falls back to a default model, which may be a PAID credential "
|
||||
+ "(fleetd #175); quarantining this profile's credential",
|
||||
cfg.profile(), cfg.model(), actualDisplay);
|
||||
// fleetd #234: pass our OWN profile name too. This check fires from agentSessionId(),
|
||||
// called during SessionManager.acquire() BEFORE this session is registered in
|
||||
// sessions.roster() — a roster-only sink (Fleetd's target -> session -> profile lookup)
|
||||
// finds nothing at this point and silently no-ops (defect 2). We already know exactly
|
||||
// which profile to quarantine without the roster; the sink is passed it explicitly.
|
||||
exhaustionSink.onExhausted(delegate.terminalId(),
|
||||
"opencode model mismatch: profile '" + cfg.profile() + "' requested '"
|
||||
+ cfg.model() + "' but the live session is running '" + actualDisplay
|
||||
+ "' (fleetd #175)",
|
||||
cfg.profile());
|
||||
}
|
||||
|
||||
@Override
|
||||
|
||||
@@ -1,9 +1,12 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
import org.sqlite.SQLiteConfig;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.sql.Connection;
|
||||
@@ -29,11 +32,19 @@ import java.util.concurrent.atomic.AtomicBoolean;
|
||||
* here, so a layout change, or a switch to the HTTP server, changes exactly one class and nothing
|
||||
* in {@link OpenCodeLauncher}.
|
||||
*
|
||||
* <p>The determinism that makes this useful is structural, not a guess: every fleetd worker runs
|
||||
* in its own unique git worktree, so the row's {@code directory} (its project root) equals the
|
||||
* worker's cwd identifies <em>its</em> session unambiguously. We match on {@code directory} rather
|
||||
* than diffing {@code opencode session list} before/after — that races under concurrent spawns, and
|
||||
* the CLI listing does not even show the directory.
|
||||
* <p><strong>The {@code directory} match is a heuristic, not an identity — fleetd #234.</strong> A
|
||||
* worktree is opt-in: {@code fleet_spawn} only provisions one when the caller passes {@code
|
||||
* worktree:}; the default spawn inherits the lead's own cwd, which every other worker spawned the
|
||||
* same way (and every past session ever run there) shares. {@code directory} therefore does
|
||||
* <em>not</em> identify a session unambiguously in general — only in the special case of a fresh,
|
||||
* unique worktree does "most recently updated row for this directory" reliably mean "this worker's
|
||||
* own row." {@link #sessionIdForDirectory} still has to use this heuristic (the id has to come from
|
||||
* somewhere, and nothing else is available at this layer — see that method's javadoc), but a caller
|
||||
* that already holds a resolved id must never re-derive evidence about that same session via
|
||||
* {@code directory} again; see {@link #actualModelForSessionId}, which looks up by {@code id}
|
||||
* instead for exactly this reason. We match on {@code directory} rather than diffing
|
||||
* {@code opencode session list} before/after — that races under concurrent spawns, and the CLI
|
||||
* listing does not even show the directory.
|
||||
*
|
||||
* <p>All reads are best-effort and never throw: a missing or unreadable database, a query that
|
||||
* fails, or a directory with no row yet all yield {@code null}, and the caller (the session
|
||||
@@ -44,6 +55,18 @@ import java.util.concurrent.atomic.AtomicBoolean;
|
||||
final class OpenCodeSessionDiscovery {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(OpenCodeSessionDiscovery.class);
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
|
||||
/**
|
||||
* The model opencode actually ran a session on, parsed from the {@code session.model} JSON
|
||||
* column (fleetd #175). {@code id} is never null/blank on a non-null {@code ActualModel} —
|
||||
* {@link #parseModel} returns {@code null} instead when {@code id} cannot be determined, so a
|
||||
* caller only ever sees a fully-known record or {@code null} (UNKNOWN). {@code provider} may
|
||||
* still be {@code null} on its own when the profile that requested the session named no
|
||||
* provider prefix, or opencode's JSON omitted {@code providerID}.
|
||||
*/
|
||||
record ActualModel(String provider, String id) {
|
||||
}
|
||||
|
||||
private final Path storageRoot; // e.g. ~/.local/share/opencode (injectable for tests)
|
||||
private final Path databasePath;
|
||||
@@ -74,8 +97,16 @@ final class OpenCodeSessionDiscovery {
|
||||
/**
|
||||
* The opencode session id whose row references {@code directory} (the worker's cwd), or
|
||||
* {@code null} when no row matches yet. When several rows share the directory — e.g. repeated
|
||||
* spawns into the same worktree — the row with the highest {@code time_updated} wins: it is
|
||||
* the session the pane most likely corresponds to.
|
||||
* spawns into the same worktree, OR several workers sharing one cwd because none of them was
|
||||
* given a worktree (fleetd #234) — the row with the highest {@code time_updated} wins: it is
|
||||
* the session the pane most likely corresponds to. That "most likely" is a real caveat, not a
|
||||
* formality: when the directory is shared, this can and does pick another session's row (see
|
||||
* the class javadoc). This is the one place fleetd resolves an opencode session id at all —
|
||||
* nothing else is available at this layer to disambiguate further (no {@code opencode session
|
||||
* list} entry names the directory, and diffing before/after races under concurrent spawns) — so
|
||||
* the heuristic stays here unchanged. What must never happen is a SECOND, independent piece of
|
||||
* evidence about the same session being re-derived via {@code directory} once an id has already
|
||||
* come out of this method; see {@link #actualModelForSessionId}.
|
||||
*
|
||||
* <p>Never throws: a missing {@code opencode.db}, a locked/unreadable database, a query
|
||||
* failure, or a directory that has not been persisted yet all resolve to {@code null} rather
|
||||
@@ -117,4 +148,85 @@ final class OpenCodeSessionDiscovery {
|
||||
storageRoot, directory);
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The model opencode actually ran the session {@code sessionId} on (fleetd #175/#234), read by
|
||||
* primary-key lookup — the ONE row that id names, and no other. Deliberately keyed on
|
||||
* {@code id} rather than {@code directory}: two independent {@code WHERE directory = ? ORDER BY
|
||||
* time_updated DESC LIMIT 1} queries (one for the id, one for the model) can each pick a
|
||||
* DIFFERENT row once more than one session shares a directory (fleetd #234 — the default
|
||||
* no-worktree spawn shares the lead's cwd with every other worker and every past session ever
|
||||
* run there), silently comparing a profile's requested model against a session that is not even
|
||||
* the one whose id was returned. Keying on {@code id} instead makes that impossible: the model
|
||||
* read back is always the SAME session {@link #sessionIdForDirectory} (or a cached copy of its
|
||||
* answer) already resolved.
|
||||
*
|
||||
* <p>Kept as its own query and its own connection, independent from {@link
|
||||
* #sessionIdForDirectory}: a database whose schema predates the {@code model} column (or any
|
||||
* other read failure on this column alone) can never take id resolution down with it. That
|
||||
* would be a regression of the id-resolution feature #209 shipped; this method degrades on its
|
||||
* own.
|
||||
*
|
||||
* <p>Never throws, and every failure mode — no matching row, a missing/unreadable database, a
|
||||
* missing {@code model} column, a null/blank {@code model} value, or JSON that does not parse
|
||||
* into {@code {"id": "...", "providerID": "..."}} with a non-blank {@code id} — resolves to
|
||||
* {@code null}. That is UNKNOWN evidence, not a mismatch signal: the caller must never
|
||||
* quarantine a profile on the strength of a {@code null} here. A blank/null {@code sessionId}
|
||||
* (the id is not resolved yet) is UNKNOWN too, for the same reason — never call this with one.
|
||||
*
|
||||
* @param sessionId the session id already resolved by {@link #sessionIdForDirectory} for this
|
||||
* spawn — never re-derived from {@code directory} here
|
||||
* @return the actual model, or {@code null} when unknown
|
||||
*/
|
||||
ActualModel actualModelForSessionId(String sessionId) {
|
||||
if (sessionId == null || sessionId.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
if (!Files.isRegularFile(databasePath)) {
|
||||
// sessionIdForDirectory already WARNs once (shared warnedMissingDatabase) for this
|
||||
// exact condition — do not double-log it here.
|
||||
return null;
|
||||
}
|
||||
String sql = "SELECT model FROM session WHERE id = ?";
|
||||
try (Connection connection = openReadOnly();
|
||||
PreparedStatement statement = connection.prepareStatement(sql)) {
|
||||
statement.setString(1, sessionId);
|
||||
try (ResultSet rows = statement.executeQuery()) {
|
||||
if (rows.next()) {
|
||||
return parseModel(rows.getString("model"));
|
||||
}
|
||||
}
|
||||
} catch (SQLException e) {
|
||||
// Locked/corrupt database, or a `model` column this schema version does not have —
|
||||
// never fatal, and never a mismatch signal. See the class doc above.
|
||||
log.debug("opencode session model unreadable at {}: {}", databasePath, e.toString());
|
||||
return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse opencode's {@code model} column — {@code {"id":"...","providerID":"..."}} — into an
|
||||
* {@link ActualModel}, or {@code null} when {@code json} is null/blank, is not valid JSON, or
|
||||
* parses without a non-blank {@code id}. {@code providerID} may be absent; that alone does not
|
||||
* make the record unknown, since a caller comparing against a profile with no provider prefix
|
||||
* never looks at it.
|
||||
*/
|
||||
private static ActualModel parseModel(String json) {
|
||||
if (json == null || json.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
JsonNode node = MAPPER.readTree(json);
|
||||
String id = node.path("id").asText(null);
|
||||
if (id == null || id.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
String provider = node.path("providerID").asText(null);
|
||||
return new ActualModel(provider, id);
|
||||
} catch (IOException e) {
|
||||
log.debug("opencode session model JSON unparseable: {}", e.toString());
|
||||
return null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -56,10 +56,13 @@ public final class FleetMetrics {
|
||||
m.describe(REPLIES, "counter",
|
||||
"Worker replies by delivery path (rendezvous=resolved an open send, inbox=stranded and held).");
|
||||
m.describe(PUSH_NUDGES, "counter",
|
||||
"CB-307 push-loop nudges to the primary (delivered|exhausted).");
|
||||
"CB-307 push-loop nudges to the primary (sent|exhausted). fleetd #365: \"sent\" means "
|
||||
+ "the herdr paste-and-submit call succeeded, not that the pane read it — this "
|
||||
+ "layer has no read-receipt concept.");
|
||||
m.describe(HEARTBEAT_NUDGES, "counter",
|
||||
"CB-551 idle-lead heartbeat nudges (delivered|failed|exhausted). Quiet-cap exhaustion "
|
||||
+ "means the lead idled with nothing pending and was told to stand down.");
|
||||
"CB-551 idle-lead heartbeat nudges (sent|failed|exhausted). Quiet-cap exhaustion "
|
||||
+ "means the lead idled with nothing pending and was told to stand down. "
|
||||
+ "fleetd #365: \"sent\" means the herdr call succeeded, not that the lead read it.");
|
||||
m.describe(SPAWNS, "counter",
|
||||
"Worker spawn attempts by peer kind and outcome (ready|timeout|guard_rejected).");
|
||||
m.describe(HERDR_CALLS, "counter",
|
||||
|
||||
@@ -8,6 +8,7 @@ import com.rabbitmq.client.DeliverCallback;
|
||||
import com.rabbitmq.client.Recoverable;
|
||||
import com.rabbitmq.client.RecoveryListener;
|
||||
import com.rabbitmq.client.Return;
|
||||
import com.rabbitmq.client.impl.DefaultExceptionHandler;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -22,6 +23,7 @@ import java.util.concurrent.ConcurrentSkipListMap;
|
||||
import java.util.concurrent.ExecutionException;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.TimeoutException;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
/**
|
||||
* AMQP-backed {@link ReplyInbox} (CB-307 Stage 2): genuine cross-restart durability behind the same
|
||||
@@ -93,8 +95,23 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
private final Channel channel;
|
||||
/** All channel operations (publish/declare/ack/cancel) serialize on this — a Channel is not thread-safe. */
|
||||
private final Object channelLock = new Object();
|
||||
/** target → (msgId → held delivery). Per-target map is guarded by synchronizing on itself. */
|
||||
/**
|
||||
* target → (msgId → held delivery). Per-target map is guarded by synchronizing on itself.
|
||||
*
|
||||
* <p><strong>CB-318 tombstone.</strong> The value {@link #RELEASED} is a reserved sentinel: it
|
||||
* marks a target whose {@link #release} has already run, so {@link #deliverCallback} can tell a
|
||||
* delivery landing after release() apart from a fresh target it has never seen. See both methods'
|
||||
* javadoc for why a plain {@code held.remove(target)} is not enough.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, LinkedHashMap<String, Held>> held = new ConcurrentHashMap<>();
|
||||
|
||||
/**
|
||||
* CB-318 sentinel stored in {@link #held} for a target whose {@link #release} has already run.
|
||||
* Never mutated — every read site compares it by reference ({@code ==}) before touching it as a
|
||||
* map, because it is a single object shared across every released target and calling a mutator on
|
||||
* it would corrupt state for all of them.
|
||||
*/
|
||||
private static final LinkedHashMap<String, Held> RELEASED = new LinkedHashMap<>();
|
||||
/** Targets whose queue is declared and consumer is running, mapped to their broker consumer tag. */
|
||||
private final ConcurrentHashMap<String, String> consumerTags = new ConcurrentHashMap<>();
|
||||
|
||||
@@ -137,17 +154,22 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
/** As {@link #open(String)}, with an explicit consumer prefetch (CB-527: caps the held backlog per target). */
|
||||
public static AmqpReplyInbox open(String uri, int prefetch) {
|
||||
try {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares queues and re-attaches consumers.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
return new AmqpReplyInbox(factory.newConnection("fleetd-reply-inbox"), prefetch);
|
||||
return new AmqpReplyInbox(connectionFactory(uri).newConnection(AmqpConnectionFailureLogger.REPLY_INBOX), prefetch);
|
||||
} catch (Exception e) {
|
||||
throw new IllegalStateException("cannot connect to AMQP broker at " + uri, e);
|
||||
}
|
||||
}
|
||||
|
||||
static ConnectionFactory connectionFactory(String uri) throws Exception {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares queues and re-attaches consumers.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
factory.setExceptionHandler(new AmqpConnectionFailureLogger(AmqpConnectionFailureLogger.REPLY_INBOX, log));
|
||||
return factory;
|
||||
}
|
||||
|
||||
/** Wrap an already-open connection with {@link #DEFAULT_PREFETCH} (injection seam for the contract test). */
|
||||
AmqpReplyInbox(Connection connection) {
|
||||
this(connection, DEFAULT_PREFETCH);
|
||||
@@ -201,6 +223,12 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
channel.queueDeclare(queue, true, false, false, null); // durable, non-exclusive, keep on idle
|
||||
String tag = channel.basicConsume(queue, false, deliverCallback(target), _ -> { });
|
||||
consumerTags.put(target, tag);
|
||||
// CB-318: drop a stale RELEASED tombstone from a prior ownership of this same target
|
||||
// string, so a delivery under this fresh consumer is held normally instead of being
|
||||
// nacked forever by deliverCallback's RELEASED check. Safe to do here, still under
|
||||
// channelLock: no delivery for the consumer tag just registered above can reach
|
||||
// deliverCallback before this basicConsume call returns.
|
||||
held.remove(target, RELEASED);
|
||||
log.debug("AMQP inbox owns queue {} for target {}", queue, target);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot own queue " + queue, e);
|
||||
@@ -208,18 +236,103 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Release ownership of {@code target}: cancel its consumer, then nack-with-requeue every
|
||||
* delivery still held for it instead of just dropping the local record.
|
||||
*
|
||||
* <p><strong>Cancelling a consumer does not requeue its in-flight deliveries.</strong> In AMQP,
|
||||
* a delivery that was pushed to a consumer stays unacked, attached to the still-open
|
||||
* {@link #channel}, until that channel or the connection closes — {@code basicCancel} alone does
|
||||
* neither. So before this method existed with a requeue step, it dropped {@link #held}'s entries
|
||||
* for {@code target} while the broker still considered them outstanding: never acked, never
|
||||
* nacked, never requeued, and no longer reachable by {@link #peek} — permanently invisible. This
|
||||
* is unlike {@link #handleRecovery} and {@link #close()}, whose bare {@code held.clear()} is
|
||||
* correct because each has already made the broker requeue (a real connection drop, or
|
||||
* {@code channel.close()} respectively) before clearing local state.
|
||||
*
|
||||
* <p><strong>Order: cancel first, then nack.</strong> A delivery tag stays valid for
|
||||
* {@code basicNack} on this channel regardless of whether its consumer is still attached — only
|
||||
* a channel/connection close invalidates it — so cancelling {@code target}'s consumer first does
|
||||
* not risk the tags. Doing it the other way round does: nacking a delivery with {@code requeue}
|
||||
* while its consumer is still active hands the message straight back to that <em>same</em>
|
||||
* consumer the instant a prefetch slot frees up (confirmed against a real broker — see
|
||||
* {@code AmqpReplyInboxContractTest.releaseCancelsConsumerAndRequeuesHeldDeliveryForRecovery}),
|
||||
* which races this method's own {@code held.remove(target)}: the redelivery can land after the
|
||||
* clear and leave a stale entry behind, so {@link #peek} is no longer reliably empty right after
|
||||
* {@link #release}. Cancelling first closes that consumer, so the requeued message goes back to
|
||||
* the queue for whichever consumer picks it up next (a later {@link #own}), not this one.
|
||||
*
|
||||
* <p><strong>Failure of the requeue is best-effort, not fatal.</strong> {@link #release} runs
|
||||
* during teardown ({@code Fleetd} calls it right after {@code MessageService.abandon}), and a
|
||||
* throw here would abort cleanups the caller depends on — the same argument fleetd #293 settled
|
||||
* for {@code HerdrPeerLauncher.stop()}'s tab-close step. So a failed {@code basicNack} is logged
|
||||
* at WARN, naming the target and delivery tag that leaked, and release proceeds; the delivery
|
||||
* stays unacked on the broker rather than being silently dropped, so it is still recoverable by a
|
||||
* later connection drop even though this release did not manage to requeue it immediately. A
|
||||
* failed {@code basicCancel} still throws, unchanged from before this fix — that failure means
|
||||
* the consumer may still be attached, so best-effort requeue is not attempted underneath it.
|
||||
*
|
||||
* <p><strong>CB-318: {@code held.remove(target)} alone leaves a second window open.</strong> The
|
||||
* bullet above already explains why cancelling first does not save a tag from going stale — but
|
||||
* that only accounts for a delivery landing before this method starts touching {@link #held}.
|
||||
* {@code basicCancel} stops <em>new</em> dispatches; it does not flush one already handed to the
|
||||
* consumer work pool. So a delivery can still land on that pool's thread and reach
|
||||
* {@link #deliverCallback} at any point during, or after, this method's body — and a plain
|
||||
* {@code held.remove(target)} does nothing to stop it: {@code deliverCallback}'s
|
||||
* {@code computeIfAbsent} finds the key gone and happily creates a brand-new map under it, which
|
||||
* this method — already past its {@code remove} — never looks at again. That entry then sits
|
||||
* delivered-but-unacked on {@link #channel} until the whole inbox closes: never requeued, never
|
||||
* redelivered, and {@link #peek} is never called again for a target nothing owns any more.
|
||||
*
|
||||
* <p>The fix is {@link #held}{@code .compute(target, ...)} instead of {@code remove}: it takes
|
||||
* whatever was held (to nack, same as before) and, in the same atomic step, leaves the
|
||||
* {@link #RELEASED} tombstone behind instead of an absent key. {@code computeIfAbsent} and
|
||||
* {@code compute} calls for the same key are mutually exclusive in {@link ConcurrentHashMap} —
|
||||
* whichever of this call and a concurrent {@code deliverCallback} runs first is fully visible to
|
||||
* the other, with no gap between them. So a delivery that loses the race sees a real map here and
|
||||
* gets nacked by the loop below, same as always; a delivery that wins the race (runs first) is
|
||||
* itself nacked by that same loop, once it settles into {@code held}. A delivery that arrives once
|
||||
* this method has stored {@link #RELEASED} finds it via {@code computeIfAbsent} and refuses itself
|
||||
* — see {@link #deliverCallback}. Either way nothing is silently retained forever, satisfying the
|
||||
* ticket's invariant against dropping a message. This closes the window rather than merely
|
||||
* narrowing it — correctness does not depend on how much time elapses between the swap and this
|
||||
* method returning.
|
||||
*/
|
||||
@Override
|
||||
public void release(String target) {
|
||||
synchronized (channelLock) {
|
||||
String tag = consumerTags.remove(target);
|
||||
held.remove(target); // stale delivery tags must not survive release
|
||||
if (tag == null) {
|
||||
return;
|
||||
if (tag != null) {
|
||||
try {
|
||||
channel.basicCancel(tag);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot cancel consumer for " + target, e);
|
||||
}
|
||||
}
|
||||
try {
|
||||
channel.basicCancel(tag);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot cancel consumer for " + target, e);
|
||||
AtomicReference<LinkedHashMap<String, Held>> previouslyHeld = new AtomicReference<>();
|
||||
held.compute(target, (_, v) -> {
|
||||
previouslyHeld.set(v);
|
||||
return RELEASED;
|
||||
});
|
||||
var perTarget = previouslyHeld.get();
|
||||
if (perTarget != null && perTarget != RELEASED) {
|
||||
synchronized (perTarget) {
|
||||
for (Held h : perTarget.values()) {
|
||||
try {
|
||||
channel.basicNack(h.deliveryTag(), false, true); // requeue, don't drop
|
||||
} catch (IOException | RuntimeException e) {
|
||||
// Caught broadly (not just IOException) for the same reason #293 catches
|
||||
// RuntimeException in HerdrPeerLauncher.stop(): best-effort teardown must
|
||||
// not be guarded only against the expected failure and bare against any
|
||||
// other. The message stays unacked on the broker either way — not lost,
|
||||
// just not proactively requeued — until a connection drop frees it.
|
||||
log.warn("release({}): could not requeue held delivery (msgId={}, tag={})"
|
||||
+ " back to the broker — it stays unacked until a connection"
|
||||
+ " drop frees it: {}",
|
||||
target, h.message().msgId(), h.deliveryTag(), e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -272,7 +385,7 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
if (perTarget == null || perTarget == RELEASED) {
|
||||
return List.of();
|
||||
}
|
||||
synchronized (perTarget) {
|
||||
@@ -281,17 +394,17 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
public boolean ack(String target, String msgId) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
return;
|
||||
if (perTarget == null || perTarget == RELEASED) {
|
||||
return false;
|
||||
}
|
||||
Held h;
|
||||
synchronized (perTarget) {
|
||||
h = perTarget.remove(msgId);
|
||||
}
|
||||
if (h == null) {
|
||||
return; // never held (or already acked) — no-op
|
||||
return false; // never held (or already acked) — no-op
|
||||
}
|
||||
try {
|
||||
synchronized (channelLock) {
|
||||
@@ -305,6 +418,7 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
throw new IllegalStateException("cannot ack reply " + msgId + " on " + queueName(target), e);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
private DeliverCallback deliverCallback(String target) {
|
||||
@@ -316,6 +430,20 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
String content = new String(delivery.getBody(), StandardCharsets.UTF_8);
|
||||
var perTarget = held.computeIfAbsent(target, _ -> new LinkedHashMap<>());
|
||||
if (perTarget == RELEASED) {
|
||||
// CB-318: release() already ran for this target and left the RELEASED tombstone in
|
||||
// held (see release()'s javadoc) — computeIfAbsent() is guaranteed to see it rather
|
||||
// than recreate a fresh map, because ConcurrentHashMap serializes compute/
|
||||
// computeIfAbsent calls for the same key against each other. Refuse the delivery
|
||||
// instead of holding it somewhere release() will never look at again: requeue it, the
|
||||
// same way release() nacks its own held entries, so a later owner (or a connection
|
||||
// drop) can still recover it. This does not need channelLock across a broker round
|
||||
// trip — basicNack, like the duplicate-ack case just below, does not wait for one.
|
||||
synchronized (channelLock) {
|
||||
channel.basicNack(tag, false, true);
|
||||
}
|
||||
return;
|
||||
}
|
||||
boolean duplicate;
|
||||
synchronized (perTarget) {
|
||||
if (perTarget.containsKey(msgId)) {
|
||||
@@ -450,3 +578,44 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Keeps RabbitMQ's forgiving exception behaviour while adding the connection identity that its
|
||||
* default logger drops. Package-private so both AMQP connections use the same two names.
|
||||
*/
|
||||
final class AmqpConnectionFailureLogger extends DefaultExceptionHandler {
|
||||
|
||||
static final String REPLY_INBOX = "fleetd-reply-inbox";
|
||||
static final String LEAD_MAILBOX = "fleetd-lead-mailbox";
|
||||
|
||||
private final String connectionName;
|
||||
private final Logger logger;
|
||||
|
||||
AmqpConnectionFailureLogger(String connectionName, Logger logger) {
|
||||
this.connectionName = connectionName;
|
||||
this.logger = logger;
|
||||
}
|
||||
|
||||
String connectionName() {
|
||||
return connectionName;
|
||||
}
|
||||
|
||||
@Override
|
||||
protected void log(String message, Throwable cause) {
|
||||
if (isSocketClosedOrConnectionReset(cause)) {
|
||||
logger.warn("AMQP connection {}: {} (Exception message: {})", connectionName, message, cause.getMessage());
|
||||
} else {
|
||||
logger.error("AMQP connection {}: {}", connectionName, message, cause);
|
||||
}
|
||||
}
|
||||
|
||||
private static boolean isSocketClosedOrConnectionReset(Throwable cause) {
|
||||
// Deliberate copy of ForgivingExceptionHandler's private static helper; check it on amqp-client upgrades.
|
||||
if (!(cause instanceof IOException)) {
|
||||
return false;
|
||||
}
|
||||
return "Connection reset".equals(cause.getMessage())
|
||||
|| "Socket closed".equals(cause.getMessage())
|
||||
|| "Connection reset by peer".equals(cause.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -59,16 +59,17 @@ public final class InMemoryReplyInbox implements ReplyInbox {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
public boolean ack(String target, String msgId) {
|
||||
if (!owned.contains(target)) {
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
var perTarget = store.get(target);
|
||||
if (perTarget != null) {
|
||||
//noinspection SynchronizationOnLocalVariableOrMethodParameter
|
||||
synchronized (perTarget) {
|
||||
perTarget.remove(msgId);
|
||||
}
|
||||
if (perTarget == null) {
|
||||
return false;
|
||||
}
|
||||
//noinspection SynchronizationOnLocalVariableOrMethodParameter
|
||||
synchronized (perTarget) {
|
||||
return perTarget.remove(msgId) != null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,8 +4,8 @@ import java.util.List;
|
||||
|
||||
/**
|
||||
* The lead-to-lead message channel this daemon speaks, as its callers need it — one lead's own
|
||||
* mailbox: publish to a peer's coord-id, look at what has arrived for me, and ack what I have
|
||||
* delivered.
|
||||
* mailbox: publish to a peer's coord-id, look at what has arrived for me, ack what I have
|
||||
* delivered, and (fleetd #361) inspect any coord-id's mailbox from the outside without owning it.
|
||||
*
|
||||
* <p>Extracted from {@link LeadMailbox} purely as a seam. {@code LeadMailbox} is the one production
|
||||
* implementation and owns a live AMQP connection, so a test that wanted to exercise the routing in
|
||||
@@ -32,9 +32,99 @@ public interface LeadChannel {
|
||||
/** Non-destructive FIFO snapshot of the messages held for this daemon's own coord-id. */
|
||||
List<LeadMessage> peek();
|
||||
|
||||
/** Drop {@code msgId} from the held set and ack it on the broker. A no-op if it is not held. */
|
||||
/**
|
||||
* Drop {@code msgId} from the held set and ack it on the broker. A repeated ack that this
|
||||
* connection already completed may be a no-op. Any other unknown msgId must throw rather than
|
||||
* report an ack that did not reach the broker.
|
||||
*/
|
||||
void ack(String msgId);
|
||||
|
||||
/** This daemon's own lead coordination id — the mailbox it owns, and the {@code from} it sends as. */
|
||||
String selfCoordId();
|
||||
|
||||
/**
|
||||
* Whether a message sitting in {@link #peek}'s held set (fetched but not yet {@link #ack}ed) is
|
||||
* still safe if this daemon crashes or restarts right now — the conclusion of two independent
|
||||
* facts about how this channel owns its own queue: the queue was declared <em>durable</em>, and
|
||||
* the consumer that filled {@code held} uses <em>manual ack</em>, so an unacked delivery is still
|
||||
* owned by the broker rather than only in this process's memory. Both must hold for {@code true};
|
||||
* an implementation must derive this from what it actually did when it declared and consumed its
|
||||
* queue, never return a literal — fleetd #440 found {@code FleetMcp}'s {@code heldDurable} field
|
||||
* doing exactly that, unable to ever report {@code false} even after the fact stopped being true.
|
||||
*/
|
||||
boolean heldDurable();
|
||||
|
||||
/**
|
||||
* A non-destructive look at {@code coordId}'s mailbox — does it exist, how many messages are
|
||||
* waiting on it, and how many consumers are attached — without owning, consuming, or otherwise
|
||||
* changing it. {@code consumers == 0} on an existing mailbox is the observable form of "nobody
|
||||
* is reading this right now": a publish to it will sit queued rather than reach a pane.
|
||||
*
|
||||
* <p><strong>Never throws</strong> — this is a best-effort fact-finding call, not an operation a
|
||||
* caller must handle failing. But it must never turn "I could not check" into a false negative:
|
||||
* {@link MailboxState#absent(String)} means the broker positively confirmed there is no such
|
||||
* queue, and {@link MailboxState#unknown(String)} — a distinct value — means the look could not
|
||||
* be completed at all (broker unreachable, timed out, connection closed). A caller that
|
||||
* collapses those two into one, as fleetd #361 initially did, cannot tell "that peer is down"
|
||||
* from "I could not check", and a reader of {@code pending}/{@code consumers} cannot tell a
|
||||
* measured zero from a zero standing in for "not measured".
|
||||
*
|
||||
* <p><strong>Must never share fate with {@link #publish} or {@link #peek}/{@link #ack}.</strong>
|
||||
* fleetd #361: in AMQP 0-9-1 a passive queue declare of a queue that does not exist closes the
|
||||
* channel it was declared on with a 404. An implementation backed by a real broker connection
|
||||
* must inspect on a channel it can afford to lose — never the channel {@link #publish} or the
|
||||
* consume loop depends on — so that looking at a peer that happens to be down can never break
|
||||
* this daemon's own send or receive path.
|
||||
*/
|
||||
MailboxState inspect(String coordId);
|
||||
|
||||
/**
|
||||
* The result of {@link #inspect}. {@code presence} tells apart three states a caller must not
|
||||
* conflate: a confirmed-existing mailbox ({@link Presence#EXISTS}, the only case where
|
||||
* {@code pending}/{@code consumers} are measured facts), a confirmed-absent one
|
||||
* ({@link Presence#ABSENT} — the broker positively said "no such queue"), and one this call
|
||||
* simply could not determine ({@link Presence#UNKNOWN} — broker unreachable, timed out,
|
||||
* connection closed). {@code pending}/{@code consumers} are always {@code 0} and meaningless
|
||||
* outside {@link Presence#EXISTS}; a renderer must gate on {@link #exists()} (or {@code
|
||||
* presence} directly), never present them as measured otherwise.
|
||||
*
|
||||
* @param coordId the coord-id inspected
|
||||
* @param presence whether the mailbox is confirmed to exist, confirmed absent, or unknown
|
||||
* @param pending messages ready for delivery but not yet in a consumer's hands (0 unless EXISTS)
|
||||
* @param consumers how many consumers are attached (0 unless EXISTS)
|
||||
*/
|
||||
record MailboxState(String coordId, Presence presence, int pending, int consumers) {
|
||||
|
||||
/** Whether {@link #inspect} was able to reach a definite answer, of either kind. */
|
||||
public enum Presence { EXISTS, ABSENT, UNKNOWN }
|
||||
|
||||
/** {@code true} only when the broker confirmed this exact queue is currently declared. */
|
||||
public boolean exists() {
|
||||
return presence == Presence.EXISTS;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code true} when {@link #inspect} reached a definite answer (exists or confirmed
|
||||
* absent); {@code false} when it could not determine either way. A caller must never treat
|
||||
* {@code !known()} the same as a confirmed absence — the mailbox may well exist.
|
||||
*/
|
||||
public boolean known() {
|
||||
return presence != Presence.UNKNOWN;
|
||||
}
|
||||
|
||||
/** The broker confirmed this queue exists, with these measured counts. */
|
||||
public static MailboxState exists(String coordId, int pending, int consumers) {
|
||||
return new MailboxState(coordId, Presence.EXISTS, pending, consumers);
|
||||
}
|
||||
|
||||
/** The broker positively confirmed there is no such queue (e.g. a 404 on passive declare). */
|
||||
public static MailboxState absent(String coordId) {
|
||||
return new MailboxState(coordId, Presence.ABSENT, 0, 0);
|
||||
}
|
||||
|
||||
/** The look could not be completed — broker unreachable, timed out, or connection closed. */
|
||||
public static MailboxState unknown(String coordId) {
|
||||
return new MailboxState(coordId, Presence.UNKNOWN, 0, 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,6 +5,7 @@ import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
@@ -43,6 +44,9 @@ public final class LeadCoordLoop {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadCoordLoop.class);
|
||||
|
||||
/** A bounded window is enough: redelivery can only follow a recent failed ack or recovery. */
|
||||
private static final int RECENT_DELIVERY_LIMIT = 1_024;
|
||||
|
||||
/** How an arriving peer message is rendered into the lead's pane — the sender's coord-id, then its text. */
|
||||
static final String DELIVERY_FORMAT = "[lead %s] %s";
|
||||
|
||||
@@ -51,6 +55,8 @@ public final class LeadCoordLoop {
|
||||
private final Supplier<Map<String, String>> leads;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final long intervalMs;
|
||||
/** msgIds already written to the pane, so recovery redelivery is acked without another pane write. */
|
||||
private final LinkedHashMap<String, Boolean> delivered = new LinkedHashMap<>();
|
||||
|
||||
private volatile boolean running;
|
||||
|
||||
@@ -117,6 +123,13 @@ public final class LeadCoordLoop {
|
||||
if (held.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
LeadMessage msg = held.getFirst();
|
||||
if (wasDelivered(msg.msgId())) {
|
||||
// This lives here, rather than in LeadMailbox, because only this loop knows a pane write
|
||||
// happened. The mailbox only knows broker delivery tags and must still redeliver after a crash.
|
||||
ackDelivered(msg);
|
||||
return;
|
||||
}
|
||||
String lead = resolveLocalLead();
|
||||
if (lead == null) {
|
||||
// Left unacked on purpose: the broker keeps holding it until a lead pane exists.
|
||||
@@ -136,7 +149,6 @@ public final class LeadCoordLoop {
|
||||
lead, status, held.size());
|
||||
return;
|
||||
}
|
||||
LeadMessage msg = held.getFirst();
|
||||
try {
|
||||
agents.send(lead, DELIVERY_FORMAT.formatted(msg.from(), msg.content()));
|
||||
} catch (RuntimeException e) {
|
||||
@@ -145,6 +157,13 @@ public final class LeadCoordLoop {
|
||||
msg.msgId(), msg.from(), lead, e.toString());
|
||||
return;
|
||||
}
|
||||
rememberDelivered(msg.msgId());
|
||||
if (ackDelivered(msg)) {
|
||||
log.debug("lead coordination: delivered message {} from {} to lead {}", msg.msgId(), msg.from(), lead);
|
||||
}
|
||||
}
|
||||
|
||||
private boolean ackDelivered(LeadMessage msg) {
|
||||
try {
|
||||
channel.ack(msg.msgId());
|
||||
} catch (RuntimeException e) {
|
||||
@@ -152,9 +171,24 @@ public final class LeadCoordLoop {
|
||||
// deliberate direction of this trade.
|
||||
log.warn("lead coordination: delivered message {} but could not ack it: {}",
|
||||
msg.msgId(), e.toString());
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
private boolean wasDelivered(String msgId) {
|
||||
synchronized (delivered) {
|
||||
return delivered.containsKey(msgId);
|
||||
}
|
||||
}
|
||||
|
||||
private void rememberDelivered(String msgId) {
|
||||
synchronized (delivered) {
|
||||
delivered.put(msgId, Boolean.TRUE);
|
||||
if (delivered.size() > RECENT_DELIVERY_LIMIT) {
|
||||
delivered.remove(delivered.keySet().iterator().next());
|
||||
}
|
||||
}
|
||||
log.debug("lead coordination: delivered message {} from {} to lead {}", msg.msgId(), msg.from(), lead);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -96,7 +96,14 @@ public final class LeadHeartbeatLoop {
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Count one nudge outcome when a registry is wired; a no-op in unit tests. */
|
||||
/**
|
||||
* Count one nudge outcome when a registry is wired; a no-op in unit tests.
|
||||
*
|
||||
* <p>fleetd #365: the {@code "sent"} outcome (renamed from {@code "delivered"}) records only
|
||||
* that {@link #injectNudge} — a one-way herdr {@code agent.prompt} paste-and-submit — returned
|
||||
* without throwing, not that the lead's pane actually read or acted on the text. This layer has
|
||||
* no read-receipt concept, so "sent" is the honest word for what this call can ever establish.
|
||||
*/
|
||||
private void countNudge(String outcome) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(FleetMetrics.HEARTBEAT_NUDGES, "outcome", outcome);
|
||||
@@ -245,7 +252,7 @@ public final class LeadHeartbeatLoop {
|
||||
agents.send(leadTerminal, fleet.nudgeText());
|
||||
log.debug("idle-heartbeat: nudge sent to lead {} (quiet nudges so far in this stretch: {})",
|
||||
leadTerminal, quietCount);
|
||||
countNudge("delivered");
|
||||
countNudge("sent");
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("idle-heartbeat: failed to nudge lead {}: {}", leadTerminal, e.toString());
|
||||
countNudge("failed");
|
||||
|
||||
@@ -10,6 +10,7 @@ import com.rabbitmq.client.DeliverCallback;
|
||||
import com.rabbitmq.client.Recoverable;
|
||||
import com.rabbitmq.client.RecoveryListener;
|
||||
import com.rabbitmq.client.Return;
|
||||
import com.rabbitmq.client.ShutdownSignalException;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -58,7 +59,8 @@ import java.util.concurrent.TimeoutException;
|
||||
* <p><strong>Recovery.</strong> The connection is opened with automatic + topology recovery
|
||||
* enabled, mirroring {@code AmqpReplyInbox}: on reconnect the broker hands out fresh delivery tags,
|
||||
* so the held snapshot is cleared (dedup by {@code msgId} still prevents any double-queue on
|
||||
* redelivery) and any publish still awaiting its confirm is failed rather than left to idle out
|
||||
* redelivery). {@link LeadCoordLoop} separately deduplicates pane writes, since it alone knows
|
||||
* which messages reached a lead. Any publish still awaiting its confirm is failed rather than left to idle out
|
||||
* the confirm timeout against a sequence number that means nothing on the new channel.
|
||||
*/
|
||||
public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
@@ -84,6 +86,16 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
private final Object channelLock = new Object();
|
||||
/** msgId → held delivery, for this mailbox's own queue only (there is exactly one). */
|
||||
private final LinkedHashMap<String, Held> held = new LinkedHashMap<>();
|
||||
/**
|
||||
* fleetd #440: the answer to {@link #heldDurable()}, set once by {@link #own()} from the exact
|
||||
* booleans it passed to {@code queueDeclare}/{@code basicConsume} — never a separate literal that
|
||||
* could drift from what those calls actually did.
|
||||
*/
|
||||
private boolean heldDurable;
|
||||
/** Successful broker acks on this connection, retained only to make a repeated caller ack quiet. */
|
||||
private final LinkedHashMap<String, Boolean> recentlyAcked = new LinkedHashMap<>();
|
||||
/** Bounds {@link #recentlyAcked}: it is only an idempotency aid, never delivery state. */
|
||||
private static final int RECENT_ACK_LIMIT = 1_024;
|
||||
|
||||
/**
|
||||
* A dedicated channel for {@link #publish}, kept separate from {@link #channel} (consume + ack)
|
||||
@@ -122,17 +134,22 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
/** As {@link #open(String, String)}, with an explicit consumer prefetch. */
|
||||
public static LeadMailbox open(String uri, String selfCoordId, int prefetch) {
|
||||
try {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares the queue and re-attaches the consumer.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
return new LeadMailbox(factory.newConnection("fleetd-lead-mailbox"), selfCoordId, prefetch);
|
||||
return new LeadMailbox(connectionFactory(uri).newConnection(AmqpConnectionFailureLogger.LEAD_MAILBOX), selfCoordId, prefetch);
|
||||
} catch (Exception e) {
|
||||
throw new IllegalStateException("cannot connect to AMQP coordination broker at " + uri, e);
|
||||
}
|
||||
}
|
||||
|
||||
static ConnectionFactory connectionFactory(String uri) throws Exception {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares queues and re-attaches consumers.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
factory.setExceptionHandler(new AmqpConnectionFailureLogger(AmqpConnectionFailureLogger.LEAD_MAILBOX, log));
|
||||
return factory;
|
||||
}
|
||||
|
||||
/** Wrap an already-open connection with {@link #DEFAULT_PREFETCH} (injection seam for tests). */
|
||||
LeadMailbox(Connection connection, String selfCoordId) {
|
||||
this(connection, selfCoordId, DEFAULT_PREFETCH);
|
||||
@@ -156,16 +173,15 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
}
|
||||
// On automatic recovery the broker redelivers unacked messages with FRESH delivery-tags; the
|
||||
// tags we were holding are now stale. Drop the held snapshot so the re-attached consumer
|
||||
// repopulates it with valid tags (dedup by msgId still prevents any double-queue). Any publish
|
||||
// repopulates it with valid tags (dedup by msgId still prevents any double-queue). LeadCoordLoop
|
||||
// remembers successful pane writes separately, so that redelivery cannot write a pane twice. Any publish
|
||||
// confirm still in flight when the connection dropped is equally stale — fail it now rather
|
||||
// than let it silently ride out CONFIRM_TIMEOUT_MS.
|
||||
if (connection instanceof Recoverable recoverable) {
|
||||
recoverable.addRecoveryListener(new RecoveryListener() {
|
||||
@Override
|
||||
public void handleRecovery(Recoverable recoverable) {
|
||||
synchronized (held) {
|
||||
held.clear();
|
||||
}
|
||||
clearHeldForRecovery();
|
||||
failPendingPublishesOnRecovery();
|
||||
log.info("AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery");
|
||||
}
|
||||
@@ -181,13 +197,22 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
/** Declare + consume this daemon's own {@code lead.<selfCoordId>.inbox}. Called once, at construction. */
|
||||
private void own() throws IOException {
|
||||
String queue = queueName(selfCoordId);
|
||||
boolean durableQueue = true; // durable, non-exclusive, keep on idle
|
||||
boolean autoAck = false; // manual ack
|
||||
synchronized (channelLock) {
|
||||
channel.queueDeclare(queue, true, false, false, null); // durable, non-exclusive, keep on idle
|
||||
channel.basicConsume(queue, false, deliverCallback(), _ -> { }); // autoAck=false: manual ack
|
||||
channel.queueDeclare(queue, durableQueue, false, false, null);
|
||||
channel.basicConsume(queue, autoAck, deliverCallback(), _ -> { });
|
||||
}
|
||||
// fleetd #440: held mail is durable only while both hold — a durable queue AND manual ack.
|
||||
this.heldDurable = durableQueue && !autoAck;
|
||||
log.debug("lead mailbox owns queue {} for coord-id {}", queue, selfCoordId);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean heldDurable() {
|
||||
return heldDurable;
|
||||
}
|
||||
|
||||
/**
|
||||
* Publish {@code msg} to {@code toCoordId}'s mailbox and block until the broker's publisher
|
||||
* confirm for it lands. Does <em>not</em> imply owning or consuming {@code toCoordId}'s queue.
|
||||
@@ -254,6 +279,89 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361: look at {@code coordId}'s mailbox on a fresh, immediately-closed throwaway
|
||||
* channel — never {@link #channel} (consume/ack) or {@link #publishChannel} (publish). A
|
||||
* passive queue declare of a queue that does not exist closes the channel it was declared on
|
||||
* with a 404; using a disposable probe channel means that closure can never touch either
|
||||
* long-lived channel this instance depends on for {@link #publish} or the consume loop.
|
||||
*
|
||||
* <p>Classifies failures rather than collapsing them, both measured against a real broker in
|
||||
* {@code LeadMailboxTest} rather than assumed from the AMQP 0-9-1 spec text:
|
||||
* <ul>
|
||||
* <li>a genuine 404 — an {@link IOException} wrapping a {@link ShutdownSignalException} whose
|
||||
* {@link AMQP.Channel.Close#getReplyCode()} is {@code 404} — reports
|
||||
* {@link MailboxState#absent}; every other declare failure reports
|
||||
* {@link MailboxState#unknown} instead of quietly becoming the same "absent" value;
|
||||
* <li>{@code catch (RuntimeException e)} on both attempts matters as much as the checked
|
||||
* catches: a connection that is already closed makes {@link Connection#createChannel()}
|
||||
* throw {@link com.rabbitmq.client.AlreadyClosedException} (a {@link RuntimeException},
|
||||
* not an {@link IOException}) — an {@code inspect} that only caught {@code IOException}
|
||||
* would let that escape, breaking the "never throws" contract this method promises.
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>Honesty about which catch is measured and which is defensive:</strong> the
|
||||
* {@code createChannel()} catch above is exercised end-to-end against a real broker by
|
||||
* {@code LeadMailboxTest.inspectReportsUnknownRatherThanThrowingWhenTheConnectionIsAlreadyClosed}.
|
||||
* The second {@code catch (RuntimeException e)}, around the passive declare itself — for the
|
||||
* narrower race where the connection drops <em>between</em> {@code createChannel()} succeeding
|
||||
* and the declare landing — has no such test; reaching it needs a connection that dies at that
|
||||
* exact instant, which is not a scenario this suite drives on purpose. It stays purely
|
||||
* defensive: correct by the same reasoning as the first catch, but unproven the way the first
|
||||
* one is proven.
|
||||
*/
|
||||
@Override
|
||||
public MailboxState inspect(String coordId) {
|
||||
String queue = queueName(coordId);
|
||||
Channel probe;
|
||||
try {
|
||||
probe = connection.createChannel();
|
||||
} catch (IOException | RuntimeException e) {
|
||||
log.debug("lead mailbox inspect: cannot open a probe channel for {}: {}", coordId, e.toString());
|
||||
return MailboxState.unknown(coordId);
|
||||
}
|
||||
try {
|
||||
AMQP.Queue.DeclareOk declared = probe.queueDeclarePassive(queue);
|
||||
return MailboxState.exists(coordId, declared.getMessageCount(), declared.getConsumerCount());
|
||||
} catch (IOException e) {
|
||||
// The broker (or the client library) has already closed `probe` for us either way; only
|
||||
// a confirmed 404 means "no such queue" — anything else (a different declare failure) is
|
||||
// "could not determine", never silently reported as the same value as a genuine absence.
|
||||
return isMissingQueue(e) ? MailboxState.absent(coordId) : MailboxState.unknown(coordId);
|
||||
} catch (RuntimeException e) {
|
||||
// E.g. the connection dropped between createChannel() and the declare landing.
|
||||
log.debug("lead mailbox inspect: declare failed unexpectedly for {}: {}", coordId, e.toString());
|
||||
return MailboxState.unknown(coordId);
|
||||
} finally {
|
||||
try {
|
||||
if (probe.isOpen()) {
|
||||
probe.close();
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.debug("lead mailbox inspect: probe channel close for {}: {}", coordId, e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code true} only for the specific shape a missing-queue passive declare actually produces —
|
||||
* measured against a real broker, not assumed from the spec text (see {@code
|
||||
* LeadMailboxTest.passiveDeclareOfAMissingQueueThrowsAnIOExceptionWrappingA404ShutdownSignal}):
|
||||
* an {@link IOException} whose cause is a {@link ShutdownSignalException} carrying an
|
||||
* {@link AMQP.Channel.Close} reason with {@code replyCode == 404}. Any other shape (a different
|
||||
* reply code, a {@code ShutdownSignalException} cause whose reason is not a
|
||||
* {@code Channel.Close}, or no cause at all) is a declare failure of some other kind and must
|
||||
* not be read as "confirmed absent" — pinned hermetically, with no broker needed, by
|
||||
* {@code LeadMailboxIsMissingQueueTest} for exactly those three false shapes. Package-private
|
||||
* (not {@code private}) so that test can call it directly.
|
||||
*/
|
||||
static boolean isMissingQueue(IOException e) {
|
||||
if (!(e.getCause() instanceof ShutdownSignalException sse)) {
|
||||
return false;
|
||||
}
|
||||
return sse.getReason() instanceof AMQP.Channel.Close close && close.getReplyCode() == AMQP.NOT_FOUND;
|
||||
}
|
||||
|
||||
/** Convenience: {@link #peek} the current snapshot, then {@link #ack} every message in it. */
|
||||
public List<LeadMessage> drain() {
|
||||
List<LeadMessage> snapshot = peek();
|
||||
@@ -261,15 +369,26 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
return snapshot;
|
||||
}
|
||||
|
||||
/** Remove the held message {@code msgId} and ack it on the broker. No-op if not held. */
|
||||
/**
|
||||
* Remove the held message {@code msgId} and ack it on the broker.
|
||||
*
|
||||
* <p>A repeated ack that this connection already completed is a no-op, tracked in the bounded
|
||||
* {@link #recentlyAcked} set. Any other missing entry throws: recovery clears {@link #held} while
|
||||
* the broker still owns the unacked delivery, and quiet success there would hide a required retry.
|
||||
* The set is bounded because it only distinguishes a recent duplicate caller ack from an unknown
|
||||
* delivery; it is not a substitute for broker state across a reconnect.
|
||||
*/
|
||||
@Override
|
||||
public void ack(String msgId) {
|
||||
Held h;
|
||||
synchronized (held) {
|
||||
h = held.remove(msgId);
|
||||
if (h == null && recentlyAcked.containsKey(msgId)) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (h == null) {
|
||||
return; // never held (or already acked) — no-op
|
||||
throw new IllegalStateException("cannot ack lead message " + msgId + ": it is not held");
|
||||
}
|
||||
try {
|
||||
synchronized (channelLock) {
|
||||
@@ -283,6 +402,19 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
}
|
||||
throw new IllegalStateException("cannot ack lead message " + msgId, e);
|
||||
}
|
||||
synchronized (held) {
|
||||
recentlyAcked.put(msgId, Boolean.TRUE);
|
||||
if (recentlyAcked.size() > RECENT_ACK_LIMIT) {
|
||||
recentlyAcked.remove(recentlyAcked.keySet().iterator().next());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Clear stale delivery tags after recovery; package-private so the recovery contract test drives this exact path. */
|
||||
void clearHeldForRecovery() {
|
||||
synchronized (held) {
|
||||
held.clear();
|
||||
}
|
||||
}
|
||||
|
||||
private DeliverCallback deliverCallback() {
|
||||
|
||||
@@ -138,6 +138,53 @@ public final class MessageService {
|
||||
public record AskResult(AskOutcome outcome, String answer) {
|
||||
}
|
||||
|
||||
/**
|
||||
* How a worker's {@code fleet_reply} ({@link #reply(String, String)}) actually landed
|
||||
* (fleetd #365) — the two doors that expose it, {@code fleet_reply} and {@code POST
|
||||
* /sessions/{id}/reply}, both used to report the single word "delivered" whichever of these
|
||||
* happened, so a caller could not tell an active handoff from a reply merely held for later
|
||||
* drain. Both are successes; they are not the same fact.
|
||||
*/
|
||||
public enum ReplyOutcome {
|
||||
/** Resolved a {@code fleet_send}/{@code fleet_ask} that was actively waiting on this reply. */
|
||||
RESOLVED_SEND("resolved_send", true,
|
||||
"delivered — resolved the fleet_send that was waiting for it"),
|
||||
/**
|
||||
* No live waiter was open, but the reply completed a parked async ticket directly
|
||||
* ({@link #askAnsweredAsyncTasks}) — a {@code fleet_poll} caller sees it immediately.
|
||||
*/
|
||||
RESOLVED_ASYNC_TICKET("resolved_async_ticket", true,
|
||||
"delivered — resolved a pending async ticket (visible to fleet_poll)"),
|
||||
/** Nothing was waiting; the reply was queued in the inbox for a later drain (CB-307). */
|
||||
QUEUED("queued", false,
|
||||
"queued — no send or ticket was waiting; held in the inbox for a later drain");
|
||||
|
||||
private final String wireName;
|
||||
private final boolean delivered;
|
||||
private final String description;
|
||||
|
||||
ReplyOutcome(String wireName, boolean delivered, String description) {
|
||||
this.wireName = wireName;
|
||||
this.delivered = delivered;
|
||||
this.description = description;
|
||||
}
|
||||
|
||||
/** Stable machine-readable name for a JSON/metrics label (REST's {@code outcome} field). */
|
||||
public String wireName() {
|
||||
return wireName;
|
||||
}
|
||||
|
||||
/** Whether something was actively waiting and received this reply right now. */
|
||||
public boolean delivered() {
|
||||
return delivered;
|
||||
}
|
||||
|
||||
/** Shared human-readable text — the one place both {@code fleet_reply} and REST word this. */
|
||||
public String description() {
|
||||
return description;
|
||||
}
|
||||
}
|
||||
|
||||
/** Lifecycle phase of an async delegation ticket. */
|
||||
public enum Phase {
|
||||
/** Delegated and in flight — queued for the worker or being worked. */
|
||||
@@ -187,6 +234,24 @@ public final class MessageService {
|
||||
private volatile Long completedNanos;
|
||||
private volatile Reply question;
|
||||
private volatile String turnId;
|
||||
/**
|
||||
* Set when this task's {@code fleet_ask} lapsed with no answer (fleetd #307):
|
||||
* {@link #clearAsyncQuestion} then forgets {@link #turnId} (nulls it and drops the task from
|
||||
* {@code asyncTasksByTurn}) so {@link #hasAsyncQuestion} stops reporting the target BUSY — a
|
||||
* later {@code fleet_send} to it must be accepted, not refused. But the worker's turn is
|
||||
* still genuinely live: it resumed on its own and will eventually call its real
|
||||
* {@code fleet_reply}. Losing {@link #turnId} loses {@link #askAnsweredAsyncTasks}' only
|
||||
* signal that such a reply belongs to this task, so that reply used to fall straight to the
|
||||
* inbox and strand — {@code fleet_poll} stayed {@code PENDING} forever, later force-failed by
|
||||
* {@link #abandon} with the misleading "session released before it replied". This flag is a
|
||||
* second, independent signal that survives the forgetting: {@link #askAnsweredAsyncTasks}
|
||||
* accepts it in place of a live {@link #turnId}, without ever re-adding the task to
|
||||
* {@code asyncTasksByTurn} (so the BUSY release is untouched). Cleared implicitly once
|
||||
* {@link #future} resolves — every match in {@link #askAnsweredAsyncTasks} already requires
|
||||
* {@code !future.isDone()}, so a task that recovered (or was later failed by
|
||||
* {@link #abandon}) can never match again regardless of this flag's value.
|
||||
*/
|
||||
private volatile boolean askTimedOut;
|
||||
|
||||
private Task(String ticket, String target, LongSupplier nowNanos) {
|
||||
this.ticket = ticket;
|
||||
@@ -360,6 +425,18 @@ public final class MessageService {
|
||||
* {@link #ask} clears the ticket's question and returns it to {@code PENDING}, but {@link #send}
|
||||
* already closed the forward waiter the instant the question surfaced, so the target has
|
||||
* neither an accepted nor a queued delivery left to show for it.
|
||||
*
|
||||
* <p><strong>Deliberately still {@code question == null} only (fleetd #275).</strong> This
|
||||
* method must not also report a still-{@link Phase#ASKING} task as orphaned: the worker may
|
||||
* genuinely be waiting on a live primary that is about to (or already mid-{@link #answer})
|
||||
* answer it, and {@link dev.ltms.fleet.health.FleetHealthMonitor} would classify that as
|
||||
* {@code DELEGATION_ORPHANED} on nothing more than an active, healthy conversation. {@link
|
||||
* #abandon(String, String, boolean)}'s {@code sweepAsking} path fixes the actual reachable gap
|
||||
* (a target torn down for good while genuinely {@code ASKING}) at the point of teardown itself,
|
||||
* by completing the task's future right there — so by the time this method would ever see it,
|
||||
* {@code task.future.isDone()} is already {@code true} and it is excluded regardless of this
|
||||
* guard. Widening this check instead of that one would trade a real fix for false positives on
|
||||
* every ordinary in-flight question.
|
||||
*/
|
||||
public boolean hasOrphanedDelegation(String target) {
|
||||
if (target == null || hasAcceptedDelivery(target) || hasQueuedDelivery(target)) {
|
||||
@@ -383,45 +460,73 @@ public final class MessageService {
|
||||
* {@link Rendezvous#resolveQuestion} must keep today's {@code NO_WAITER} behaviour — questions
|
||||
* are interactive and must never be queued.
|
||||
*
|
||||
* <p><strong>Ambiguous match also falls to the inbox.</strong> {@link #askAnsweredAsyncTasks}
|
||||
* cannot actually return more than one entry today (see its own javadoc for why — in short,
|
||||
* {@link #hasAsyncQuestion} keeps a target BUSY, so no second task can reach this state, for as
|
||||
* long as an earlier one's {@code turnId} is still stamped). That is an emergent guarantee from
|
||||
* two other facts, not one this method enforces, so this branch stays in as defence in depth
|
||||
* rather than being removed as dead code: if it ever weakens, returning whichever candidate a
|
||||
* {@code ConcurrentHashMap} iteration reaches first would let a genuine reply complete the
|
||||
* <em>wrong</em> ticket — silently handing the lead something that reads like a correct answer to
|
||||
* a delegation the worker never touched, which is worse than a failure because the lead acts on
|
||||
* it. When more than one candidate exists, guessing is not safe: fall back to the inbox exactly
|
||||
* as the zero-candidate case does, and let {@link #abandon} apply the eventual recovery
|
||||
* deterministically instead.
|
||||
* <p><strong>Ambiguous match also falls to the inbox.</strong> {@link #askAnsweredAsyncTasks} can
|
||||
* return more than one entry — a reachable state, not a hypothetical one (see its own javadoc:
|
||||
* an {@code fleet_ask} that lapsed with no answer, fleetd #307, frees the target for a completely fresh
|
||||
* delegation, which can itself go on to ask-and-lapse before the first worker's real reply
|
||||
* arrives). Returning whichever candidate a {@code ConcurrentHashMap} iteration reaches first
|
||||
* would let a genuine reply complete the <em>wrong</em> ticket — silently handing the lead
|
||||
* something that reads like a correct answer to a delegation the worker never touched, which is
|
||||
* worse than a failure because the lead acts on it. When more than one candidate exists, guessing
|
||||
* is not safe: fall back to the inbox exactly as the zero-candidate case does, and let
|
||||
* {@link #abandon} apply the eventual recovery deterministically instead.
|
||||
*
|
||||
* @return always {@code true} — the reply resolved a live send, completed a parked ticket, or
|
||||
* was queued
|
||||
* <p><strong>{@code content} is required (fleetd #302).</strong> Both doors that reach this
|
||||
* method must reject a missing/blank reply the same way, so the check lives here rather than in
|
||||
* either caller: {@code FleetMcp.reply} already refuses a {@code null} content before it ever
|
||||
* calls this method (its own required-arg guard), and no test or production call site anywhere
|
||||
* in the codebase relies on replying with empty content — confirmed by searching every call site
|
||||
* of this method before adding the check, not assumed. Without this guard, a REST {@code
|
||||
* POST /sessions/{id}/reply} whose body omits {@code content} (or a client library that maps a
|
||||
* missing field to {@code ""}) used to reach {@link Rendezvous#resolve} with an empty string,
|
||||
* silently completing the lead's blocking wait with nothing — indistinguishable from a worker
|
||||
* that genuinely replied with nothing, which is worse than a loud failure because it destroys the
|
||||
* information that the reply never arrived.
|
||||
*
|
||||
* @throws IllegalArgumentException if {@code content} is {@code null} or blank — the caller must
|
||||
* report this as a client error (REST: 400 {@code bad_request}) rather than resolve
|
||||
* anything
|
||||
* @return which of the three ways (fleetd #365) the reply actually landed — never {@code null}
|
||||
*/
|
||||
public boolean reply(String session, String content) {
|
||||
public ReplyOutcome reply(String session, String content) {
|
||||
if (content == null || content.isBlank()) {
|
||||
throw new IllegalArgumentException("content is required");
|
||||
}
|
||||
if (rendezvous.resolve(session, content)) {
|
||||
count(FleetMetrics.REPLIES, "path", "rendezvous");
|
||||
return true; // a live send took it — unchanged fast path
|
||||
return ReplyOutcome.RESOLVED_SEND; // a live send took it — unchanged fast path
|
||||
}
|
||||
// #137: no live rendezvous waiter, but this may be the worker's real fleet_reply resuming a
|
||||
// turn that {@link #answer} already gave up waiting on. answer()'s own bounded wait (the
|
||||
// primary's fleet_send{turnId} call, capped well under a minute) can time out and close its
|
||||
// waiter long before the worker — now actually resuming real work — finishes and replies. That
|
||||
// reply used to have nowhere to land but the session inbox, leaving the async ticket's future
|
||||
// unresolved forever: fleet_poll{ticket} stayed PENDING until fleet_stop's abandon() forced it
|
||||
// FAILED with a misleading "session released before it replied" reason, even though the reply
|
||||
// had, in fact, arrived. Completing the matching ticket directly here means fleet_poll{ticket}
|
||||
// sees the real reply instead.
|
||||
// #137/fleetd #307: no live rendezvous waiter, but this may be the worker's real fleet_reply resuming
|
||||
// a turn that either answer() (#137) or ask() (fleetd #307) already gave up waiting on:
|
||||
// - answer()'s own bounded wait (the primary's fleet_send{turnId} call, capped well under a
|
||||
// minute) can time out and close its waiter long before the worker — now actually resuming
|
||||
// real work — finishes and replies.
|
||||
// - ask()'s own wait for the primary can time out first, with the worker resuming on its own
|
||||
// and finishing unanswered.
|
||||
// Either way that reply used to have nowhere to land but the session inbox, leaving the async
|
||||
// ticket's future unresolved forever: fleet_poll{ticket} stayed PENDING until fleet_stop's
|
||||
// abandon() forced it FAILED with a misleading "session released before it replied" reason,
|
||||
// even though the reply had, in fact, arrived. Completing the matching ticket directly here
|
||||
// means fleet_poll{ticket} sees the real reply instead.
|
||||
List<Task> candidates = askAnsweredAsyncTasks(session);
|
||||
if (candidates.size() == 1) {
|
||||
Task orphan = candidates.get(0);
|
||||
if (orphan.future.complete(new Reply(Outcome.REPLIED, content))) {
|
||||
if (orphan.turnId != null) {
|
||||
asyncTasksByTurn.remove(orphan.turnId, orphan);
|
||||
// fleetd #329 (F3): read orphan.turnId once. It used to be read twice, under no
|
||||
// lock — once for this null check, once as the removal key — the same double-read
|
||||
// shape fleetd #324 fixed in finishAsyncTask. clearAsyncQuestion's unlocked
|
||||
// forgetTurn=true path (ask()'s timeout cleanup) can null the field between the two
|
||||
// reads; capturing it once removes the torn read here too.
|
||||
String turnId = orphan.turnId;
|
||||
if (turnId != null) {
|
||||
if (replyOrphanTurnIdRaceHookForTest != null) {
|
||||
// Test-only (fleetd #329, F3): see the field's own javadoc.
|
||||
replyOrphanTurnIdRaceHookForTest.run();
|
||||
}
|
||||
asyncTasksByTurn.remove(turnId, orphan);
|
||||
}
|
||||
count(FleetMetrics.REPLIES, "path", "async-recovered");
|
||||
return true; // the ticket itself took it — no inbox stranding at all
|
||||
return ReplyOutcome.RESOLVED_ASYNC_TICKET; // the ticket itself took it — no inbox stranding
|
||||
}
|
||||
} else if (candidates.size() > 1) {
|
||||
List<String> tickets = candidates.stream().map(t -> t.ticket).toList();
|
||||
@@ -439,38 +544,46 @@ public final class MessageService {
|
||||
if (pushLoop != null) {
|
||||
pushLoop.onReplyQueued(session);
|
||||
}
|
||||
return true; // held, not lost
|
||||
return ReplyOutcome.QUEUED; // held, not lost — but not delivered either
|
||||
}
|
||||
|
||||
/**
|
||||
* Every still-open async task on {@code target} whose {@code fleet_ask} was already answered —
|
||||
* its {@link Task#turnId} is stamped but its {@link Task#question} was cleared by {@link #answer}
|
||||
* — yet whose future is not resolved yet (#137). Empty if no such task exists, including the
|
||||
* common case where {@code target}'s worker never used {@code fleet_ask} at all (a task that was
|
||||
* never asked has {@code turnId == null}, so it can never match here and only ever completes
|
||||
* through the ordinary rendezvous fast path in {@link #reply}).
|
||||
* Every still-open async task on {@code target} whose worker is genuinely expected to send a
|
||||
* real {@code fleet_reply} next with nothing left registered to catch it: either its
|
||||
* {@code fleet_ask} was already answered — {@link Task#turnId} is stamped but {@link
|
||||
* Task#question} was cleared by {@link #answer} — or its {@code fleet_ask} lapsed unanswered and
|
||||
* {@link Task#askTimedOut} marks that (fleetd #307; {@link Task#turnId} is {@code null} by then, forgotten
|
||||
* so the target is not left BUSY — see {@link Task#askTimedOut}'s own javadoc). Either way the
|
||||
* task's future is not resolved yet. Empty if no such task exists, including the common case
|
||||
* where {@code target}'s worker never used {@code fleet_ask} at all (a task that was never asked
|
||||
* has both {@code turnId == null} and {@code askTimedOut == false}, so it can never match here and
|
||||
* only ever completes through the ordinary rendezvous fast path in {@link #reply}).
|
||||
*
|
||||
* <p><strong>Returns at most one entry today — verified, not assumed.</strong> {@link #send}
|
||||
* refuses to open a waiter on {@code target} while {@link #hasAsyncQuestion} is true, and that
|
||||
* check matches ANY task whose {@code turnId} is still stamped in {@code asyncTasksByTurn} —
|
||||
* not only while its question is still open. {@link #answer} deliberately leaves that stamp in
|
||||
* place ({@code clearAsyncQuestion(turnId, false)}) until the resumed turn's own future actually
|
||||
* resolves, at which point {@link #finishAsyncTask} both removes the stamp AND completes that
|
||||
* task's future in the same call. So a second task can never reach "{@code turnId} stamped, future
|
||||
* still open" — the exact pair this method matches on — while a first one already holds it: by
|
||||
* the time the stamp is gone, so is the eligibility. This is an emergent property of those two
|
||||
* facts holding together, not something this method (or its callers) enforces on its own — flip
|
||||
* {@code forgetTurn} to {@code true} in that one {@link #answer} call and it silently stops being
|
||||
* true, with nothing left to fail loudly. The callers below still handle "more than one" as
|
||||
* defence in depth against exactly that, not because they exercise it today: {@link #reply}
|
||||
* treats it as unresolvable and falls back to the inbox; {@link #abandon} would pick the oldest
|
||||
* deterministically (its own {@code matching} list has no such guarantee — see its javadoc).
|
||||
* <p><strong>Can return more than one entry — reachable, not just defence in depth.</strong>
|
||||
* {@link #send} refuses to open a waiter on {@code target} while {@link #hasAsyncQuestion} is
|
||||
* true, and that check matches ANY task whose {@code turnId} is still stamped in
|
||||
* {@code asyncTasksByTurn}. While a task's {@code turnId} stays stamped — {@link #answer} leaves
|
||||
* it in place ({@code clearAsyncQuestion(turnId, false)}) until {@link #finishAsyncTask} removes
|
||||
* the stamp and completes the future in the same call — no second task on the same target can
|
||||
* reach an eligible state, because {@link #send} would refuse it as BUSY first. That single-task
|
||||
* guarantee holds only for the {@code turnId}-stamped half of this method's match: an
|
||||
* {@link Task#askTimedOut} task is, by construction, no longer stamped in {@code asyncTasksByTurn}
|
||||
* (that is the whole point of forgetting {@code turnId} in {@link #clearAsyncQuestion}), so the
|
||||
* target is free the moment one ask lapses. A fresh, independent {@code sendAsync} to the same
|
||||
* target can then be dispatched, itself pause on {@code fleet_ask}, and itself time out — landing
|
||||
* a second {@code askTimedOut} task on the very target the first one is still waiting to answer
|
||||
* for. Two (or more) genuinely open tasks on one target is therefore a real, reachable state
|
||||
* today, not a hypothetical: {@link #reply} treats it as unresolvable and falls back to the
|
||||
* inbox rather than guess which task a reply belongs to (guessing wrong would hand the lead a
|
||||
* plausible-looking answer to a delegation the worker never touched — worse than a failure,
|
||||
* because the lead acts on it); {@link #abandon} instead picks the oldest deterministically (its
|
||||
* own {@code matching} list has a different, wider match — see its javadoc).
|
||||
*/
|
||||
private List<Task> askAnsweredAsyncTasks(String target) {
|
||||
List<Task> candidates = new ArrayList<>();
|
||||
for (Task task : tasks.values()) {
|
||||
if (target.equals(task.target) && task.question == null && task.turnId != null
|
||||
&& !task.future.isDone()) {
|
||||
if (target.equals(task.target) && task.question == null && !task.future.isDone()
|
||||
&& (task.turnId != null || task.askTimedOut)) {
|
||||
candidates.add(task);
|
||||
}
|
||||
}
|
||||
@@ -580,6 +693,39 @@ public final class MessageService {
|
||||
* reply — see the note above)
|
||||
*/
|
||||
public boolean abandon(String target, String reason) {
|
||||
return abandon(target, reason, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #abandon(String, String)}, with control over whether a task still paused in
|
||||
* {@code fleet_ask} ({@link Phase#ASKING}) is swept too (fleetd #275).
|
||||
*
|
||||
* <p>{@code sweepAsking} must be {@code true} only when the caller has independent, certain
|
||||
* knowledge that {@code target} can never resume its turn — today that is only
|
||||
* {@code sessions.onRelease}'s teardown (an explicit {@code fleet_stop}, or the idle reaper):
|
||||
* the worker's pane is being stopped right now, so whatever it was mid-{@code fleet_ask} about
|
||||
* has no turn left to resume into. {@link dev.ltms.fleet.health.FleetHealthMonitor}'s
|
||||
* health-classification call keeps passing {@code false} (via {@link #abandon(String, String)}):
|
||||
* a GONE/NEVER_READY reading is the daemon's best guess from the live agent list, not a teardown
|
||||
* it performed itself, and {@code abandonDoesNotFailAnAsyncTicketWaitingForAnAnswer} documents
|
||||
* why an active ask must survive that guess — the primary may already be mid-{@link #answer} for
|
||||
* the very same turn, and completing it here first would preempt a real answer with a misleading
|
||||
* failure.
|
||||
*
|
||||
* <p><strong>Without {@code sweepAsking} on the release path, a target torn down while
|
||||
* genuinely {@code ASKING} was unrecoverable.</strong> {@link #resolveQuestion} had already
|
||||
* closed the forward waiter the instant the question surfaced (so the {@code waiter} branch
|
||||
* below finds nothing to fail), the {@code question == null} guard excluded the task from
|
||||
* {@code matching} (so the loop below skipped it too), and the worker's own {@code fleet_ask}
|
||||
* clears {@link Task#question} back to {@code null} only once it lapses (the reverse-rendezvous
|
||||
* window — up to {@code FleetMcp.ASK_DEFAULT_TIMEOUT_MS} / {@code FleetApp.MAX_ASK_TIMEOUT_MS},
|
||||
* 55–115s) — by which point the released session no longer appears in {@code sessions.roster()}
|
||||
* for {@link dev.ltms.fleet.health.FleetHealthMonitor} to ever re-observe, so nothing was ever
|
||||
* left to call {@link #abandon} on this target again. The ticket then sat in {@link #tasks}
|
||||
* forever: not terminal, so {@link #pruneTerminalTickets} never dropped it, and
|
||||
* {@code fleet_poll} reported it stuck at {@link Phase#PENDING} for good.
|
||||
*/
|
||||
public boolean abandon(String target, String reason, boolean sweepAsking) {
|
||||
boolean hadStrandedReply = hasStrandedReply(target);
|
||||
// CB-640: the session is gone — nothing will ever accept or deliver into it now.
|
||||
strandedReplies.remove(target);
|
||||
@@ -590,7 +736,8 @@ public final class MessageService {
|
||||
|
||||
List<Task> matching = new ArrayList<>();
|
||||
for (Task task : tasks.values()) {
|
||||
if (target.equals(task.target) && task.question == null && !task.future.isDone()) {
|
||||
if (target.equals(task.target) && (sweepAsking || task.question == null)
|
||||
&& !task.future.isDone()) {
|
||||
matching.add(task);
|
||||
}
|
||||
}
|
||||
@@ -607,19 +754,48 @@ public final class MessageService {
|
||||
for (Task task : matching) {
|
||||
boolean isRecovery = task == recoveryTask && recovered != null;
|
||||
Reply outcome = isRecovery ? recovered : new Reply(Outcome.WORKER_FAILED, reason);
|
||||
if (task.future.complete(outcome)) {
|
||||
if (outcome.outcome() == Outcome.WORKER_FAILED) {
|
||||
asyncFailed = true;
|
||||
} else if (task.turnId != null) {
|
||||
asyncTasksByTurn.remove(task.turnId, task);
|
||||
String turnId = task.turnId;
|
||||
// fleetd #335: completing THIS task's future must not depend on any other task's
|
||||
// cleanup succeeding — every task in `matching` is owed its own outcome regardless of
|
||||
// what happens below, so decide and record that before doing anything that can throw.
|
||||
boolean completedHere = task.future.complete(outcome);
|
||||
if (completedHere && outcome.outcome() == Outcome.WORKER_FAILED) {
|
||||
asyncFailed = true;
|
||||
}
|
||||
try {
|
||||
if (abandonCleanupHookForTest != null) {
|
||||
// Test-only (fleetd #335 site 1): see the field's own javadoc.
|
||||
abandonCleanupHookForTest.run();
|
||||
}
|
||||
} else if (isRecovery) {
|
||||
// The recovered reply was already drained out of the inbox, but this task resolved
|
||||
// through another path (e.g. a concurrent reply() or a second abandon() racing this
|
||||
// one) between us choosing it and completing it here. Put the reply back rather than
|
||||
// lose it silently — it may still belong to some other still-open task, or the next
|
||||
// caller that drains this target's inbox.
|
||||
inbox.publish(target, UUID.randomUUID().toString(), recovered.text());
|
||||
if (completedHere) {
|
||||
if (turnId != null) {
|
||||
// #275: whether this task was swept out of ASKING or was already answered and
|
||||
// only waiting on its resumed turn's real reply (#137), nothing will ever
|
||||
// complete this turnId now — drop it from this class's own bookkeeping AND the
|
||||
// reverse-rendezvous itself, so hasAsyncQuestion(target) stops reporting a turn
|
||||
// that is actually done, and a late answer() sees it as lapsed rather than
|
||||
// resolving a question nothing is listening for any more.
|
||||
asyncTasksByTurn.remove(turnId, task);
|
||||
rendezvous.closeAsk(turnId);
|
||||
}
|
||||
} else if (isRecovery) {
|
||||
// The recovered reply was already drained out of the inbox, but this task resolved
|
||||
// through another path (e.g. a concurrent reply() or a second abandon() racing this
|
||||
// one) between us choosing it and completing it here. Put the reply back rather than
|
||||
// lose it silently — it may still belong to some other still-open task, or the next
|
||||
// caller that drains this target's inbox.
|
||||
inbox.publish(target, UUID.randomUUID().toString(), recovered.text());
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
// fleetd #335: inbox.publish reaches a broker (AmqpReplyInbox throws
|
||||
// IllegalStateException on an unroutable/unconfirmed/interrupted publish) and this
|
||||
// loop has no other teardown path — a caller on the release path, or the health
|
||||
// monitor's GONE/NEVER_READY sweep. Losing this exception uncaught would abort the
|
||||
// loop and leave every task still to come in `matching` PENDING forever (fleetd
|
||||
// #335 site 1). completedHere is already recorded above, so only this task's
|
||||
// best-effort bookkeeping is lost — log it and let the loop reach the rest.
|
||||
log.error("abandon: per-task cleanup failed for ticket {} (target {}, turnId {})",
|
||||
task.ticket, target, turnId, e);
|
||||
}
|
||||
}
|
||||
if (failed) {
|
||||
@@ -650,9 +826,13 @@ public final class MessageService {
|
||||
/**
|
||||
* Acknowledge a specific reply by {@code msgId} for {@code target}. Removes it from the inbox
|
||||
* so that a subsequent drain or peek no longer returns it.
|
||||
*
|
||||
* @return {@code true} if an entry was actually removed, {@code false} if {@code msgId} was not
|
||||
* in {@code target}'s inbox (wrong id, wrong target, or already acked). The caller —
|
||||
* {@link dev.ltms.fleet.mcp.FleetMcp#ack} — must not report success on {@code false}.
|
||||
*/
|
||||
public void ackReply(String target, String msgId) {
|
||||
inbox.ack(target, msgId);
|
||||
public boolean ackReply(String target, String msgId) {
|
||||
return inbox.ack(target, msgId);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -746,23 +926,33 @@ public final class MessageService {
|
||||
if (task != null) {
|
||||
asyncTasksByWaiter.put(reply, task);
|
||||
}
|
||||
TurnToken token = new TurnToken(target, reply);
|
||||
TurnToken token = new TurnToken(target, reply, content);
|
||||
// The send has won the lock; the accepted-delivery hook records delegator ownership
|
||||
// here (CB-548). It runs BEFORE enqueue so a throwing hook — onAccepted is now a
|
||||
// public callback — fails the send without queuing a message that would orphan.
|
||||
if (onAccepted != null) {
|
||||
onAccepted.run();
|
||||
}
|
||||
CompletableFuture<Void> delivered = injector.enqueue(target, content, token);
|
||||
Injector.Delivery delivery = injector.enqueue(target, content, token);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return recorded(new Reply(outcomeOf(r.kind()), r.text(), r.turnId()));
|
||||
} catch (TimeoutException e) {
|
||||
boolean wasDelivered = delivered.isDone() && !delivered.isCompletedExceptionally();
|
||||
boolean wasDelivered = delivery.completion().isDone()
|
||||
&& !delivery.completion().isCompletedExceptionally();
|
||||
if (!wasDelivered) {
|
||||
if (timeoutCancellationRaceHookForTest != null) {
|
||||
// Test-only (fleetd #345): see the field's own javadoc.
|
||||
timeoutCancellationRaceHookForTest.run();
|
||||
}
|
||||
// The target monitor makes cancellation atomic with onStatus picking this
|
||||
// Pending up. If pickup won, report TIMED_OUT_WORKING because the text landed.
|
||||
wasDelivered = injector.cancel(delivery) == Injector.Cancellation.DELIVERED;
|
||||
}
|
||||
log.debug("send to {} timed out (delivered={})", target, wasDelivered);
|
||||
if (!wasDelivered) {
|
||||
// CB-640: still sitting in the injector's queue, waiting for the member to
|
||||
// go idle — record the fact for fleet health (see queuedDeliveries).
|
||||
// CB-640: record that delivery did not happen for fleet health (see
|
||||
// queuedDeliveries). The exact Pending was cancelled, so it cannot arrive later.
|
||||
queuedDeliveries.put(target, Boolean.TRUE);
|
||||
}
|
||||
return recorded(new Reply(
|
||||
@@ -825,7 +1015,41 @@ public final class MessageService {
|
||||
return new AskResult(AskOutcome.ANSWERED, answer);
|
||||
} catch (TimeoutException e) {
|
||||
log.debug("fleet_ask from {} went unanswered in {}ms", workerSession, timeoutMillis);
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
// Only the fresh owner tears down the shared turn (mirrors the finally block below).
|
||||
// A duplicate's own timeoutMillis says nothing about whether the SHARED ask is actually
|
||||
// done — it must leave the close/forget bookkeeping to the fresh owner, exactly as it
|
||||
// already leaves closeAsk to it.
|
||||
if (ticket.fresh()) {
|
||||
// fleetd #334: close the ask turn BEFORE forgetting this task's turnId mapping below.
|
||||
// Before this fix the order was reversed — the mapping was forgotten here first, and
|
||||
// rendezvous.closeAsk only ran afterward, in the shared finally. A primary's answer()
|
||||
// call racing this exact timeout could then find rendezvous.askSession(turnId) still
|
||||
// non-null (the ask still "answerable") after the Task mapping was already gone:
|
||||
// answer()'s own asyncTasksByTurn lookup returned null, its task != null guard skipped
|
||||
// the completion, and the async ticket sat at PENDING forever even though answer()
|
||||
// itself reported the worker's real reply. Closing here first removes that window:
|
||||
// any answer() call that still observes askSession(turnId) != null is necessarily
|
||||
// racing a point BEFORE the forgetting below runs (both happen on this one thread, in
|
||||
// this order, with nothing that yields in between), so the Task mapping is still there
|
||||
// for it to find; any call that observes askSession(turnId) == null now correctly
|
||||
// bails out STALE_TURN (see answer()'s own top check) before ever reaching
|
||||
// asyncTasksByTurn. rendezvous.closeAsk is idempotent — a no-op once the turn is
|
||||
// already removed, see its own javadoc — so the shared finally below re-running it
|
||||
// for this same fresh call is harmless.
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
if (askTimeoutRaceHookForTest != null) {
|
||||
// Test-only (fleetd #334): see the field's own javadoc.
|
||||
askTimeoutRaceHookForTest.run();
|
||||
}
|
||||
// fleetd #307: mark the task BEFORE clearAsyncQuestion(forgetTurn=true) below drops it
|
||||
// out of asyncTasksByTurn and nulls its turnId — that forgetting is deliberate and
|
||||
// stays (it is what keeps the target from staying BUSY forever), but it would
|
||||
// otherwise also erase askAnsweredAsyncTasks' only signal that the worker's eventual
|
||||
// real fleet_reply still belongs to this task, stranding it in the inbox with a false
|
||||
// "never replied" verdict.
|
||||
markAskTimedOut(ticket.turnId());
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
}
|
||||
return new AskResult(AskOutcome.TIMED_OUT, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
@@ -872,7 +1096,16 @@ public final class MessageService {
|
||||
}
|
||||
try {
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession);
|
||||
// #282: mirror send()'s registration (:802) so a SECOND fleet_ask inside this same
|
||||
// resumed turn can re-associate the async ticket with its new turnId via
|
||||
// markAsyncQuestion — without this, that second ask has no Task to attach to, and
|
||||
// markAsyncQuestion silently returns null.
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null) {
|
||||
asyncTasksByWaiter.put(reply, task);
|
||||
}
|
||||
if (!rendezvous.answerAsk(turnId, content)) {
|
||||
asyncTasksByWaiter.remove(reply);
|
||||
rendezvous.close(workerSession, reply);
|
||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||
}
|
||||
@@ -880,7 +1113,66 @@ public final class MessageService {
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
Reply result = new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
finishAsyncTask(turnId, result);
|
||||
// #282: this waiter can resolve with a FRESH question rather than a terminal reply —
|
||||
// the worker chained a second fleet_ask before replying. Mirror sendAsync's own guard
|
||||
// (:1000) and leave the ticket open (markAsyncQuestion above already re-armed it under
|
||||
// the new turnId) instead of completing it here with a QUESTION "reply".
|
||||
// Measured when #282 was merged: this guard is DEFENCE IN DEPTH, not the thing
|
||||
// that makes the chained ask work. ask() calls markAsyncQuestion (:860) before
|
||||
// resolveQuestion (:861), so by the time this thread wakes, the task has already
|
||||
// moved to the new turnId — this outcome check, not a Task lookup, is what tells the
|
||||
// two cases apart (see fleetd #329 below). Removing this guard alone leaves the test
|
||||
// green. Keep it anyway: it mirrors sendAsync's sibling guard, and that sibling's own
|
||||
// comment (:1017) warns the two orderings are not something to rely on. Do NOT delete
|
||||
// it as dead code without re-checking that ordering, and do not treat it as the sole
|
||||
// protection either.
|
||||
//
|
||||
// fleetd #329 (F1): complete the SAME Task object this method already looked up at
|
||||
// :991, rather than re-resolving it from turnId a second time. The old
|
||||
// finishAsyncTask(turnId, result) did its own asyncTasksByTurn.get(turnId) here, and
|
||||
// that second lookup races ask()'s own timeout path: ask()'s ticket.answer().get(...)
|
||||
// can time out (or lose that exact race) at essentially the same instant this method's
|
||||
// rendezvous.answerAsk(turnId, ...) above already succeeded, running
|
||||
// markAskTimedOut + clearAsyncQuestion(turnId, true) with no lock at all — which
|
||||
// forgets turnId (removes it from asyncTasksByTurn, nulls Task.turnId) before this
|
||||
// thread ever gets here. The worker's real reply then arrived, this method's own wait
|
||||
// woke up with it, and the by-turnId lookup found nothing: the async ticket's future
|
||||
// was never completed, so fleet_poll{ticket} stayed PENDING forever even though
|
||||
// answer() itself correctly returned REPLIED. Reusing the reference captured at :991
|
||||
// — before that race window opens — sidesteps the second lookup entirely: it is the
|
||||
// identical Task whatever asyncTasksByTurn or Task.turnId say by the time we reach
|
||||
// this point, so finishAsyncTask(Task, Reply) can still complete its future and detach
|
||||
// it using whatever turnId it now reads. The #282 chained-ask case is unaffected
|
||||
// because it is still gated purely by result.outcome() == QUESTION above, which does
|
||||
// not depend on this lookup — widening what "task" means here cannot complete a ticket
|
||||
// the chained ask deliberately left open.
|
||||
//
|
||||
// A null task is NOT only "this was never an async ticket". That reading was in this
|
||||
// comment when #329 merged and it is wrong; it is still not the whole story after
|
||||
// #334. A genuine async ticket can still land here with task == null — a blocking
|
||||
// (wait:true) send's fleet_ask never has a Task at all, so that case is expected and
|
||||
// fine. What #334 fixed was a SECOND, unintended way to get here with task == null:
|
||||
// ask()'s timeout path used to run clearAsyncQuestion(turnId, true) — which drops the
|
||||
// asyncTasksByTurn entry — in its catch block, while rendezvous.closeAsk(turnId) ran
|
||||
// later, in its finally. Between those two the ask was still answerable but the map
|
||||
// entry was already gone, so the lookup at :1053 returned null and this ticket was
|
||||
// never completed. Measured on 2026-09-04: a probe firing only that first half before
|
||||
// answer() ran printed "answer=REPLIED phase=PENDING reply=null" — the same stranded
|
||||
// ticket #329 set out to fix, one step earlier in the same race; the probe used
|
||||
// forgetTurnForTest, which omits ask()'s markAskTimedOut, and that omission does not
|
||||
// change the outcome, because askTimedOut is read only by askAnsweredAsyncTasks, and
|
||||
// reply() never reaches it while this method's own waiter is live. #334's fix
|
||||
// reorders ask()'s timeout catch to run closeAsk before the forgetting (see the
|
||||
// fresh-owner block there), which removes this path entirely rather than narrowing
|
||||
// it further: once closeAsk has run, rendezvous.askSession(turnId) is null and
|
||||
// answer() returns STALE_TURN from its own top check, before it ever reaches this
|
||||
// lookup — see aLateAnswerDuringAskTimeoutTeardownStillCompletesTheAsyncTicket in
|
||||
// MessageServiceTest, which pins the exact window with askTimeoutRaceHookForTest.
|
||||
// So by the time this line runs, task == null means only the ordinary blocking-send
|
||||
// case (or #329's own already-fixed race elsewhere) — not this one.
|
||||
if (result.outcome() != Outcome.QUESTION && task != null) {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
return result;
|
||||
} catch (TimeoutException e) {
|
||||
// The worker resumed but hasn't replied yet — no completion fallback arms an answered
|
||||
@@ -893,6 +1185,7 @@ public final class MessageService {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + workerSession, e);
|
||||
} finally {
|
||||
asyncTasksByWaiter.remove(reply);
|
||||
rendezvous.close(workerSession, reply);
|
||||
}
|
||||
} finally {
|
||||
@@ -929,14 +1222,25 @@ public final class MessageService {
|
||||
// this fires exactly once, from whichever path completes it: finishAsyncTask(task, result)
|
||||
// below on any non-QUESTION outcome of send() — a worker's fleet_reply, the CB-106
|
||||
// completion fallback, a CB-109 wedge, TIMED_OUT, BUSY, or BACKEND_EXHAUSTED — the same
|
||||
// finishAsyncTask reached via answer()'s finishAsyncTask(turnId, result) once a QUESTION
|
||||
// finishAsyncTask reached via answer()'s own finishAsyncTask(task, result) once a QUESTION
|
||||
// is resolved, completeExceptionally(t) just below when send() itself throws, or a CB-516
|
||||
// abandon() on teardown. Without this, MessageService.reply's rendezvous fast path (the
|
||||
// one an async ticket always takes) never told the push loop anything happened — see the
|
||||
// class javadoc on sendAsync/CB-107.
|
||||
task.future.whenComplete((reply, ex) -> {
|
||||
boolean failed = ex != null || reply == null || !reply.completed();
|
||||
pushLoop.onTicketTerminal(ticket, target, failed);
|
||||
try {
|
||||
pushLoop.onTicketTerminal(ticket, target, failed);
|
||||
} catch (Throwable t) {
|
||||
// fleetd #335 (site 2): this stage's own CompletableFuture is discarded, so an
|
||||
// uncaught throw here (e.g. a RejectedExecutionException from
|
||||
// ReplyPushLoop's scheduler, already shut down while this in-flight send's
|
||||
// whenComplete fires during the daemon's own shutdown sequence — messages.close()
|
||||
// only stops accepting NEW async work, it does not cancel a delivery already
|
||||
// running) vanishes with no log line and no metric, and the push loop never learns
|
||||
// the ticket went terminal — the exact thing this hook exists to tell it.
|
||||
log.error("push loop failed to learn ticket {} (target {}) went terminal", ticket, target, t);
|
||||
}
|
||||
});
|
||||
}
|
||||
asyncExecutor.submit(() -> {
|
||||
@@ -949,7 +1253,22 @@ public final class MessageService {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
} catch (Throwable t) {
|
||||
task.future.completeExceptionally(t);
|
||||
// fleetd #329 (F2): finishAsyncTask above already completes task.future — on its very
|
||||
// first line — before doing anything else, so anything that throws afterward (inside
|
||||
// finishAsyncTask's own cleanup, or from a future addition to this try block) lands
|
||||
// here with the future already resolved. completeExceptionally on an already-completed
|
||||
// future is a silent no-op: it returns false and does nothing, so without the check
|
||||
// below the exception simply vanished — no log, no metric, nothing. Measured (see the
|
||||
// ticket): temporarily reintroducing the fleetd #324 NPE reproduced 19 real exceptions
|
||||
// on the ordinary path, with 74/74 tests staying green and not one log line produced.
|
||||
// Do not stop completing the future first — finishAsyncTask completing before it
|
||||
// cleans up is what makes a late failure harmless to the ticket's own result — only
|
||||
// add the missing visibility for the case where that step, or whatever ran after it,
|
||||
// has already lost the race to report through the future.
|
||||
if (!task.future.completeExceptionally(t)) {
|
||||
log.error("async send {} -> {} threw after its ticket was already resolved",
|
||||
ticket, target, t);
|
||||
}
|
||||
}
|
||||
});
|
||||
pruneTerminalTickets();
|
||||
@@ -1002,6 +1321,27 @@ public final class MessageService {
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, detail, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test seam only — carries no production behaviour, and nothing in this class calls it;
|
||||
* {@link #pruneTerminalTickets} still reads {@link Task#completedNanos} directly.
|
||||
*
|
||||
* <p>Reports whether {@code ticket}'s completion hook (the {@code whenComplete} registered in
|
||||
* {@link Task}'s constructor) has actually run yet. Exists because {@link #poll} can report
|
||||
* {@link Phase#DONE} for a ticket before that hook fires: {@code CompletableFuture.complete()}
|
||||
* publishes its result and only afterwards runs dependent actions such as {@code whenComplete}
|
||||
* (fleetd #399), so a caller that observes the future done via {@link #poll} is not thereby
|
||||
* guaranteed to also observe {@link Task#completedNanos} stamped. A test that must order both
|
||||
* events — e.g. before advancing an injected clock past the TTL, to avoid stamping the
|
||||
* *advanced* time and masking a real eviction bug — waits on this instead of on
|
||||
* {@link Phase#DONE}.
|
||||
*
|
||||
* @return {@code false} for an unknown ticket or one whose completion hook has not run yet
|
||||
*/
|
||||
boolean isCompletionStampedForTest(String ticket) {
|
||||
Task task = tasks.get(ticket);
|
||||
return task != null && task.completedNanos != null;
|
||||
}
|
||||
|
||||
/** Best-effort live worker status for a pending poll; never throws (a lookup error is just noise). */
|
||||
private String liveStatus(String target) {
|
||||
try {
|
||||
@@ -1051,13 +1391,37 @@ public final class MessageService {
|
||||
private Task markAsyncQuestion(CompletableFuture<Rendezvous.Resolution> waiter, String text, String turnId) {
|
||||
Task task = waiter == null ? null : asyncTasksByWaiter.get(waiter);
|
||||
if (task != null) {
|
||||
String previousTurnId = task.turnId;
|
||||
task.question = new Reply(Outcome.QUESTION, text, turnId);
|
||||
task.turnId = turnId;
|
||||
asyncTasksByTurn.put(turnId, task);
|
||||
// #282: a second fleet_ask in the same resumed turn re-arms an already-answered task
|
||||
// (answer() re-registers it in asyncTasksByWaiter) under a FRESH turnId — drop the old
|
||||
// key so asyncTasksByTurn does not keep growing by one stale entry per chained ask.
|
||||
if (previousTurnId != null && !previousTurnId.equals(turnId)) {
|
||||
asyncTasksByTurn.remove(previousTurnId, task);
|
||||
}
|
||||
}
|
||||
return task;
|
||||
}
|
||||
|
||||
/**
|
||||
* Mark {@code turnId}'s task as having a {@code fleet_ask} that lapsed with no answer (fleetd #307), so
|
||||
* {@link #askAnsweredAsyncTasks} still recognizes the worker's eventual real {@code fleet_reply}
|
||||
* as belonging to it after {@link #clearAsyncQuestion}'s {@code forgetTurn=true} erases
|
||||
* {@link Task#turnId} — see {@link Task#askTimedOut}. Must be called before that forgetting, while
|
||||
* {@code turnId} can still resolve the task in {@code asyncTasksByTurn}; a lookup afterward would
|
||||
* find nothing. Only when it matches the task's current turn — same guard as
|
||||
* {@link #clearAsyncQuestion} — so a chained second {@code fleet_ask} (#282) that already moved
|
||||
* the task to a fresh {@code turnId} cannot mark it for a turn that is no longer its own.
|
||||
*/
|
||||
private void markAskTimedOut(String turnId) {
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null && turnId.equals(task.turnId)) {
|
||||
task.askTimedOut = true;
|
||||
}
|
||||
}
|
||||
|
||||
/** Clear an answered or lapsed question, but only when it matches the ticket's current turn. */
|
||||
private void clearAsyncQuestion(String turnId, boolean forgetTurn) {
|
||||
// CB-582: tell the push loop first — like ticketCollected, a removal for a turnId it never
|
||||
@@ -1076,20 +1440,174 @@ public final class MessageService {
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete and detach an async ticket after its worker's actual terminal reply. */
|
||||
/**
|
||||
* Complete and detach an async ticket after its worker's actual terminal reply.
|
||||
*
|
||||
* <p><strong>fleetd #324.</strong> {@code task.turnId} is read into {@code turnId} exactly once.
|
||||
* It used to be read twice — once for the null check, once as the removal key — and {@code
|
||||
* volatile} makes each of those reads individually fresh but does not make the pair atomic.
|
||||
* {@link #answer} calls this while holding {@code sessionLocks} for the target; {@link #ask}'s
|
||||
* own timeout path calls {@link #clearAsyncQuestion} (which nulls {@link Task#turnId}) under no
|
||||
* lock at all. When that unlocked null-out landed between the two reads here, the second read saw
|
||||
* {@code null} and {@code asyncTasksByTurn.remove(null, task)} threw {@code NullPointerException}
|
||||
* on the lead's own {@code answer()} call — even though {@code task.future.complete(result)} on
|
||||
* the line above had already run, so the answer was in fact delivered. Capturing the field once
|
||||
* removes the torn read; see the ticket for why the wider asymmetry between the locked and
|
||||
* unlocked sides is not fixed by this alone.
|
||||
*/
|
||||
private void finishAsyncTask(Task task, Reply result) {
|
||||
task.future.complete(result);
|
||||
if (task.turnId != null) {
|
||||
asyncTasksByTurn.remove(task.turnId, task);
|
||||
if (afterFinishAsyncTaskCompleteHookForTest != null) {
|
||||
// Test-only (fleetd #329, F2): see the field's own javadoc.
|
||||
afterFinishAsyncTaskCompleteHookForTest.run();
|
||||
}
|
||||
String turnId = task.turnId;
|
||||
if (turnId != null) {
|
||||
if (finishAsyncTaskRaceHook != null) {
|
||||
// Test-only (fleetd #324): see the field's own javadoc.
|
||||
finishAsyncTaskRaceHook.run();
|
||||
}
|
||||
asyncTasksByTurn.remove(turnId, task);
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete the async ticket correlated to a specific answered turn. */
|
||||
private void finishAsyncTask(String turnId, Reply result) {
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null) {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
/**
|
||||
* Null in production; test seam for fleetd #324 — invoked from {@link #finishAsyncTask(Task,
|
||||
* Reply)} right after {@code task.turnId}'s null-check passes and before the (now-local) value is
|
||||
* used for the removal. A test installs this to force, deterministically, the exact interleaving
|
||||
* that a real race between this method and {@link #ask}'s unlocked timeout cleanup can otherwise
|
||||
* only produce by chance: firing it here reproduces "the field went null between the check and the
|
||||
* use" against the pre-fix code, and demonstrates the fix tolerates it (the captured local is used
|
||||
* unconditionally, so a hook that nulls the field afterward cannot affect this call).
|
||||
*/
|
||||
private volatile Runnable finishAsyncTaskRaceHook;
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #324): install {@link #finishAsyncTaskRaceHook}. Package-private so the test,
|
||||
* in the same package, can reach it without widening any production API.
|
||||
*/
|
||||
void setFinishAsyncTaskRaceHookForTest(Runnable hook) {
|
||||
this.finishAsyncTaskRaceHook = hook;
|
||||
}
|
||||
|
||||
/**
|
||||
* Null in production; test seam for fleetd #329 (F2) — invoked from {@link #finishAsyncTask(Task,
|
||||
* Reply)} unconditionally, immediately after {@code task.future.complete(result)} runs (before
|
||||
* {@link Task#turnId} is even read, so it fires regardless of whether this task was ever asked).
|
||||
* A test installs this to force an exception into exactly the shape fleetd #329 identified:
|
||||
* something throws inside {@link #sendAsync}'s executor task after the async ticket's future is
|
||||
* already resolved, so the surrounding {@code catch (Throwable t)} can only report failure through
|
||||
* {@code completeExceptionally} — a silent no-op on an already-completed future. Engineering a
|
||||
* real exception to land in that exact post-completion window is what the ticket itself had to do
|
||||
* by temporarily deleting a production guard (fleetd #324's {@code turnId != null} check); this
|
||||
* hook drives the identical shape deterministically instead.
|
||||
*/
|
||||
private volatile Runnable afterFinishAsyncTaskCompleteHookForTest;
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #329, F2): install {@link #afterFinishAsyncTaskCompleteHookForTest}.
|
||||
* Package-private so the test, in the same package, can reach it without widening any production
|
||||
* API.
|
||||
*/
|
||||
void setAfterFinishAsyncTaskCompleteHookForTest(Runnable hook) {
|
||||
this.afterFinishAsyncTaskCompleteHookForTest = hook;
|
||||
}
|
||||
|
||||
/**
|
||||
* Null in production; test seam for fleetd #345 — invoked in {@link #send}'s timeout path after
|
||||
* {@link Injector.Delivery#completion()} reports incomplete and before {@link Injector#cancel}
|
||||
* takes the target monitor. A test installs this to make {@code onStatus} pick the exact queued
|
||||
* delivery up in that window, so {@code cancel} returns {@link Injector.Cancellation#DELIVERED}.
|
||||
* This deterministically covers the caller's need to use that result rather than relying on a
|
||||
* timing-sensitive real race.
|
||||
*/
|
||||
private volatile Runnable timeoutCancellationRaceHookForTest;
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #345): install {@link #timeoutCancellationRaceHookForTest}. Package-private
|
||||
* so the test, in the same package, can reach it without widening any production API.
|
||||
*/
|
||||
void setTimeoutCancellationRaceHookForTest(Runnable hook) {
|
||||
this.timeoutCancellationRaceHookForTest = hook;
|
||||
}
|
||||
|
||||
/**
|
||||
* Null in production; test seam for fleetd #329 (F3) — invoked from {@link #reply} right after
|
||||
* the single local read of {@code orphan.turnId} passes its null-check and before that (now-local)
|
||||
* value is used as the {@code asyncTasksByTurn} removal key. Mirrors {@link
|
||||
* #finishAsyncTaskRaceHook} exactly, for the structurally identical double-read fleetd #324 fixed
|
||||
* in {@link #finishAsyncTask}: a test installs this to force, deterministically, {@code
|
||||
* clearAsyncQuestion}'s unlocked {@code forgetTurn=true} path nulling {@link Task#turnId} in that
|
||||
* exact window, and to confirm the single-read fix tolerates it (the captured local is used
|
||||
* unconditionally, so a hook that nulls the field afterward cannot affect this call).
|
||||
*/
|
||||
private volatile Runnable replyOrphanTurnIdRaceHookForTest;
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #329, F3): install {@link #replyOrphanTurnIdRaceHookForTest}. Package-private
|
||||
* so the test, in the same package, can reach it without widening any production API.
|
||||
*/
|
||||
void setReplyOrphanTurnIdRaceHookForTest(Runnable hook) {
|
||||
this.replyOrphanTurnIdRaceHookForTest = hook;
|
||||
}
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #324): run the exact production cleanup {@link #ask}'s own timeout path runs
|
||||
* unlocked — {@link #clearAsyncQuestion(String, boolean)} with {@code forgetTurn=true} — so a test
|
||||
* can reproduce that specific mutation instead of hand-rolling an approximation of it.
|
||||
*/
|
||||
void forgetTurnForTest(String turnId) {
|
||||
clearAsyncQuestion(turnId, true);
|
||||
}
|
||||
|
||||
/**
|
||||
* Null in production; test seam for fleetd #335 (site 1) — invoked from {@link #abandon(String,
|
||||
* String, boolean)}'s per-task loop, once per task, right before that task's own cleanup
|
||||
* (turnId bookkeeping, or the stranded-reply put-back) runs. A test installs this to inject a
|
||||
* throw at that exact point deterministically.
|
||||
*
|
||||
* <p>The one production call there that can really throw is {@code inbox.publish} in the
|
||||
* put-back branch — {@link AmqpReplyInbox#publish} reaches a broker and throws {@link
|
||||
* IllegalStateException} on an unroutable, unconfirmed, or interrupted publish — but reaching
|
||||
* that branch requires a second completion of the very task {@code abandon} is about to
|
||||
* complete to win the race first (see the branch's own comment), and the #137 follow-up
|
||||
* investigation above already found the combination this needs (a stranded reply coinciding
|
||||
* with an open matching task) unreachable through the public API, not merely hard to time.
|
||||
* This hook reproduces the resulting shape — a per-task cleanup throw — directly, the same
|
||||
* technique {@link #finishAsyncTaskRaceHook} and {@link
|
||||
* #afterFinishAsyncTaskCompleteHookForTest} already use for their own hard-to-time races.
|
||||
*/
|
||||
private volatile Runnable abandonCleanupHookForTest;
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #335, site 1): install {@link #abandonCleanupHookForTest}. Package-private
|
||||
* so the test, in the same package, can reach it without widening any production API.
|
||||
*/
|
||||
void setAbandonCleanupHookForTest(Runnable hook) {
|
||||
this.abandonCleanupHookForTest = hook;
|
||||
}
|
||||
|
||||
/**
|
||||
* Null in production; test seam for fleetd #334 — invoked from {@link #ask}'s {@code
|
||||
* TimeoutException} catch, only for the fresh owner, right after {@code rendezvous.closeAsk}
|
||||
* has run and before {@link #markAskTimedOut} / {@link #clearAsyncQuestion} forget this task's
|
||||
* turnId mapping. A test installs this to call {@link #answer} for the very same {@code turnId}
|
||||
* synchronously from inside that exact window, deterministically reproducing the race a real
|
||||
* concurrent {@code answer()} call could otherwise only win by timing luck: with the ask already
|
||||
* closed, that call must see {@code rendezvous.askSession(turnId) == null} and return {@link
|
||||
* Outcome#STALE_TURN} immediately, never reaching {@code asyncTasksByTurn} at all — proving the
|
||||
* window fleetd #334 describes (mapping forgotten while the ask was still "answerable") is
|
||||
* closed, rather than merely narrowed the way fleetd #329 narrowed the sibling race in {@link
|
||||
* #finishAsyncTask}.
|
||||
*/
|
||||
private volatile Runnable askTimeoutRaceHookForTest;
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #334): install {@link #askTimeoutRaceHookForTest}. Package-private so the
|
||||
* test, in the same package, can reach it without widening any production API.
|
||||
*/
|
||||
void setAskTimeoutRaceHookForTest(Runnable hook) {
|
||||
this.askTimeoutRaceHookForTest = hook;
|
||||
}
|
||||
|
||||
/** A new send must not open a waiter while an async ticket owns this worker's paused turn. */
|
||||
|
||||
@@ -46,6 +46,13 @@ public interface ReplyInbox {
|
||||
/** Non-destructive snapshot of pending replies for {@code target} (FIFO), empty list if none. */
|
||||
List<InboxMessage> peek(String target);
|
||||
|
||||
/** Remove the reply {@code msgId} for {@code target} once the primary has taken it. No-op if absent. */
|
||||
void ack(String target, String msgId);
|
||||
/**
|
||||
* Remove the reply {@code msgId} for {@code target} once the primary has taken it.
|
||||
*
|
||||
* @return {@code true} if an entry was actually removed, {@code false} if there was nothing to
|
||||
* remove (unknown {@code target}, unowned {@code target}, or a {@code msgId} not held for
|
||||
* it). A {@code false} is not an error — acking a {@code target} this daemon does not own is
|
||||
* part of the normal contract, not a failure.
|
||||
*/
|
||||
boolean ack(String target, String msgId);
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@ package dev.ltms.fleet.msg;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
@@ -9,8 +10,11 @@ import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collection;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
@@ -68,6 +72,12 @@ public final class ReplyPushLoop {
|
||||
static final String QUESTIONS_NUDGE_FORMAT =
|
||||
"%d workers are paused on a question — run fleet_poll(ticket=...) for each, then answer "
|
||||
+ "with fleet_send(turnId=..., content=...): %s";
|
||||
static final String BACKEND_INCIDENT_NUDGE_FORMAT =
|
||||
"Backend credential %s is cooling for %d remaining seconds; affected profiles: %s; "
|
||||
+ "affected workers: %s. Run fleet_list to see more.";
|
||||
static final String BACKEND_TARGET_UNMAPPED_NUDGE_FORMAT =
|
||||
"A backend error on worker %s could not be mapped to a credential (%s) — no cool-off "
|
||||
+ "was applied. Run fleet_list to check that worker.";
|
||||
|
||||
private final PrimaryRegistry primaryRegistry;
|
||||
private final AgentControl agents;
|
||||
@@ -93,6 +103,13 @@ public final class ReplyPushLoop {
|
||||
* how depleted an older, still-open question's count is.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, PendingQuestion> pendingQuestions = new ConcurrentHashMap<>();
|
||||
/** Backend incidents awaiting one successful delivery, keyed by incident id and owning lead. */
|
||||
private final ConcurrentHashMap<IncidentLead, PendingIncident> pendingIncidents = new ConcurrentHashMap<>();
|
||||
/** Incident/lead pairs already delivered. They make repeated incident reports one-shot. */
|
||||
private final Set<IncidentLead> deliveredIncidents = ConcurrentHashMap.newKeySet();
|
||||
private final ConcurrentHashMap<UnmappedTargetLead, PendingUnmappedTarget> pendingUnmappedTargets =
|
||||
new ConcurrentHashMap<>();
|
||||
private final Set<UnmappedTargetLead> deliveredUnmappedTargets = ConcurrentHashMap.newKeySet();
|
||||
/** CB-590: leads with an active combined reminder schedule (replies and/or tickets and/or questions). */
|
||||
private final ConcurrentHashMap<String, Boolean> activeLeads = new ConcurrentHashMap<>();
|
||||
|
||||
@@ -115,7 +132,15 @@ public final class ReplyPushLoop {
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Count one nudge outcome when a registry is wired; a no-op in unit tests. */
|
||||
/**
|
||||
* Count one nudge outcome when a registry is wired; a no-op in unit tests.
|
||||
*
|
||||
* <p>fleetd #365: the {@code "sent"} outcome (renamed from {@code "delivered"}) records only
|
||||
* that {@code agents.send} — a one-way herdr {@code agent.prompt} paste-and-submit — returned
|
||||
* without throwing. Nothing in this loop, or anywhere downstream of it, confirms the pane
|
||||
* actually read or acted on the text; there is no read-receipt concept at this layer. "Sent"
|
||||
* says exactly that; "delivered" claimed more than this call can ever establish.
|
||||
*/
|
||||
private void countNudge(String outcome) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(FleetMetrics.PUSH_NUDGES, "outcome", outcome);
|
||||
@@ -178,7 +203,20 @@ public final class ReplyPushLoop {
|
||||
* tracked per question, not per lead per source).
|
||||
*/
|
||||
private record PendingQuestion(String turnId, String ticket, String target, String lead,
|
||||
String question, int nudgeCount) {
|
||||
String question, int nudgeCount) {
|
||||
}
|
||||
|
||||
private record IncidentLead(String incidentId, String lead) {
|
||||
}
|
||||
|
||||
private record PendingIncident(IncidentLead key, String credential, List<String> profiles,
|
||||
List<String> targets, int remainingCoolOffSeconds, int nudgeCount) {
|
||||
}
|
||||
|
||||
private record UnmappedTargetLead(String target, String reason, String lead) {
|
||||
}
|
||||
|
||||
private record PendingUnmappedTarget(UnmappedTargetLead key, int nudgeCount) {
|
||||
}
|
||||
|
||||
/** Questions still open for {@code lead}, snapshotted fresh for one tick. */
|
||||
@@ -192,6 +230,44 @@ public final class ReplyPushLoop {
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test seam only (fleetd #418): carries no production behaviour, and nothing in this class
|
||||
* calls it. Exposes {@link #pendingQuestionTurnIdsFor} — the exact state {@link #decide} reads
|
||||
* to decide whether a question keeps a lead's schedule alive.
|
||||
*
|
||||
* <p>{@code MessageService.ask()} does three things in order before a question is fully open to
|
||||
* this loop: it flips the ticket's {@code poll()} phase to {@code Phase.ASKING}, then resolves
|
||||
* the reverse-rendezvous waiter, then calls {@link #onQuestionOpened}, which is what actually
|
||||
* populates {@link #pendingQuestions}. A test that barriers on {@code Phase.ASKING} observes only
|
||||
* the first of those three steps — under load the asker thread can be descheduled between steps
|
||||
* one and three, so the barrier releases before this method's underlying map is populated, and
|
||||
* {@link #decide} correctly reports nothing pending yet. A test that must order itself after the
|
||||
* state {@link #decide} actually reads waits on this instead of on the phase.
|
||||
*
|
||||
* @return an unmodifiable snapshot; empty for a lead with no open questions
|
||||
*/
|
||||
Set<String> pendingQuestionTurnIdsForTest(String lead) {
|
||||
return pendingQuestionTurnIdsFor(lead);
|
||||
}
|
||||
|
||||
private List<PendingIncident> pendingIncidentsFor(String lead) {
|
||||
return pendingIncidents.values().stream().filter(i -> lead.equals(i.key().lead())).toList();
|
||||
}
|
||||
|
||||
private Set<IncidentLead> pendingIncidentKeysFor(String lead) {
|
||||
return pendingIncidentsFor(lead).stream().map(PendingIncident::key)
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
private List<PendingUnmappedTarget> pendingUnmappedTargetsFor(String lead) {
|
||||
return pendingUnmappedTargets.values().stream().filter(i -> lead.equals(i.key().lead())).toList();
|
||||
}
|
||||
|
||||
private Set<UnmappedTargetLead> pendingUnmappedTargetKeysFor(String lead) {
|
||||
return pendingUnmappedTargetsFor(lead).stream().map(PendingUnmappedTarget::key)
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* The reply-source reminder count {@link #decide} should see for {@code lead} on this tick:
|
||||
* the <em>minimum</em> nudge count among the reply targets currently pending for it (CB-598).
|
||||
@@ -236,6 +312,15 @@ public final class ReplyPushLoop {
|
||||
return min == Integer.MAX_VALUE ? 0 : min;
|
||||
}
|
||||
|
||||
private int minIncidentNudgeCountFor(String lead) {
|
||||
return pendingIncidentsFor(lead).stream().mapToInt(PendingIncident::nudgeCount).min().orElse(0);
|
||||
}
|
||||
|
||||
private int minUnmappedTargetNudgeCountFor(String lead) {
|
||||
return pendingUnmappedTargetsFor(lead).stream().mapToInt(PendingUnmappedTarget::nudgeCount)
|
||||
.min().orElse(0);
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure decision function: examine everything pending for {@code lead} — reply targets and
|
||||
* tickets alike — and return what the loop should do.
|
||||
@@ -272,17 +357,35 @@ public final class ReplyPushLoop {
|
||||
* @param questionReminderCount the lowest nudge count among questions open for this lead
|
||||
*/
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount) {
|
||||
return decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
minIncidentNudgeCountFor(lead));
|
||||
}
|
||||
|
||||
/** As above, with backend incidents as a fourth, independently bounded source. */
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount,
|
||||
int incidentReminderCount) {
|
||||
return decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
incidentReminderCount, minUnmappedTargetNudgeCountFor(lead));
|
||||
}
|
||||
|
||||
/** As above, with unmapped backend targets as a fifth, independently bounded source. */
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount,
|
||||
int incidentReminderCount, int unmappedTargetReminderCount) {
|
||||
boolean hasReplyWork = !pendingReplyTargetsFor(lead).isEmpty();
|
||||
boolean hasTicketWork = !pendingTicketIdsFor(lead).isEmpty();
|
||||
boolean hasQuestionWork = !pendingQuestionTurnIdsFor(lead).isEmpty();
|
||||
if (!hasReplyWork && !hasTicketWork && !hasQuestionWork) {
|
||||
boolean hasIncidentWork = !pendingIncidentKeysFor(lead).isEmpty();
|
||||
boolean hasUnmappedTargetWork = !pendingUnmappedTargetKeysFor(lead).isEmpty();
|
||||
if (!hasReplyWork && !hasTicketWork && !hasQuestionWork && !hasIncidentWork && !hasUnmappedTargetWork) {
|
||||
log.debug("push: nothing pending for lead {}, stopping reminder", lead);
|
||||
return Action.STOP;
|
||||
}
|
||||
boolean replyEligible = hasReplyWork && replyReminderCount < maxReminders;
|
||||
boolean ticketEligible = hasTicketWork && ticketReminderCount < maxReminders;
|
||||
boolean questionEligible = hasQuestionWork && questionReminderCount < maxReminders;
|
||||
if (!replyEligible && !ticketEligible && !questionEligible) {
|
||||
boolean incidentEligible = hasIncidentWork && incidentReminderCount < maxReminders;
|
||||
boolean unmappedTargetEligible = hasUnmappedTargetWork && unmappedTargetReminderCount < maxReminders;
|
||||
if (!replyEligible && !ticketEligible && !questionEligible && !incidentEligible && !unmappedTargetEligible) {
|
||||
log.debug("push: reminder cap ({}) reached for lead {} on every source with pending work, stopping",
|
||||
maxReminders, lead);
|
||||
countNudge("exhausted");
|
||||
@@ -302,6 +405,80 @@ public final class ReplyPushLoop {
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve who to nudge about {@code target}, the way every public entry point below wants it:
|
||||
* {@link PrimaryRegistry#nudgeTargetFor}, but only after checking the delegating lead it names
|
||||
* is still actually there (fleetd #368).
|
||||
*
|
||||
* <p><strong>The bug this closes.</strong> {@code PrimaryRegistry.forgetDelegation} is wired to
|
||||
* exactly one event — a worker's release — because that is the only teardown the daemon already
|
||||
* observes for a session in this map. Nothing removes a binding when the LEAD half goes away: a
|
||||
* lead that is closed, crashes, or is relaunched leaves {@code leadByTarget} entries pointing at
|
||||
* a terminal that no longer exists. {@code nudgeTargetFor} falls back to the single known
|
||||
* primary only when the map holds nothing for {@code target} — a stale non-null entry beats the
|
||||
* fallback every time, which is exactly backwards: the fallback's own javadoc argues it is safe
|
||||
* precisely in the case a stale entry now hides.
|
||||
*
|
||||
* <p><strong>The fix.</strong> Before trusting a recorded delegation, probe the lead the same
|
||||
* way {@link #decide} already does every tick ({@code agents.status}) — cheap, since it is a
|
||||
* local herdr round-trip, and it is the same signal {@code AgentControl.paneByTerminal} already
|
||||
* trusts to tell a genuinely dead target from a live one. A lead that fails the probe is treated
|
||||
* as if it had never been recorded: the stale entry is forgotten (self-healing, exactly like
|
||||
* {@code AgentControl.paneByTerminal} already does on {@code agent_not_found}) and resolution is
|
||||
* retried, which now reaches the fallback {@code nudgeTargetFor} was built to reach — the same
|
||||
* empty-map state its javadoc already argues is correct.
|
||||
*
|
||||
* <p><strong>fleetd #368 review — only a positive "gone" reading forgets the binding.</strong>
|
||||
* The first version of this method treated <em>any</em> {@code RuntimeException} from the probe
|
||||
* as death, which is the #359 mistake repeated: a transient socket blip or a codec error on a
|
||||
* perfectly live lead would silently and permanently unbind it, with no re-record ever coming.
|
||||
* That is destructive on one bad reading, exactly what #359 shipped a two-reading guard to avoid
|
||||
* for the analogous lead-tab-liveness question. {@link #isLive} now matches
|
||||
* {@code AgentControl.agentCall}'s own narrower rule (see its {@code agent_not_found} check): only
|
||||
* that specific, affirmative "herdr has no such agent" signal counts as gone. Every other failure
|
||||
* — timeout, transport error, a decode error — is treated as still live and the binding is left
|
||||
* alone, because guessing wrong here is unrecoverable while guessing "live" merely costs one more
|
||||
* retry on the next tick, which {@link #decide} already tolerates.
|
||||
*/
|
||||
private Optional<String> resolveLiveLead(String target) {
|
||||
Optional<String> lead = primaryRegistry.nudgeTargetFor(target);
|
||||
if (lead.isEmpty() || isLive(lead.get())) {
|
||||
return lead;
|
||||
}
|
||||
log.debug("push: lead {} delegated to for {} is no longer live, forgetting the stale binding "
|
||||
+ "and falling back", lead.get(), target);
|
||||
primaryRegistry.forgetDelegation(target);
|
||||
return primaryRegistry.nudgeTargetFor(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code lead} should still be trusted: {@code false} only when herdr affirmatively
|
||||
* reports the terminal gone ({@code agent_not_found}), never on a merely inconclusive failure.
|
||||
*
|
||||
* <p>fleetd #368 review: an earlier version returned {@code false} for any {@code RuntimeException},
|
||||
* which made a transient herdr hiccup on a live lead indistinguishable from the lead actually
|
||||
* being dead — and the caller's response to {@code false} ({@code forgetDelegation}) is
|
||||
* destructive and permanent. Narrowed to the one code {@code AgentControl.agentCall} itself
|
||||
* already treats as a genuine, resolvable absence (see its {@code agent_not_found} handling) —
|
||||
* every other {@code RuntimeException} is treated as "still live" and the binding survives to be
|
||||
* probed again next time, which costs nothing worse than one more retry.
|
||||
*/
|
||||
private boolean isLive(String lead) {
|
||||
try {
|
||||
agents.status(lead);
|
||||
return true;
|
||||
} catch (RuntimeException e) {
|
||||
boolean gone = e instanceof HerdrException he && "agent_not_found".equals(he.code());
|
||||
if (gone) {
|
||||
log.debug("push: lead {} no longer exists ({})", lead, e.toString());
|
||||
} else {
|
||||
log.debug("push: liveness check for lead {} was inconclusive ({}); treating as live "
|
||||
+ "rather than risk destroying a live binding", lead, e.toString());
|
||||
}
|
||||
return !gone;
|
||||
}
|
||||
}
|
||||
|
||||
// --- public entrypoints ----------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@@ -312,7 +489,7 @@ public final class ReplyPushLoop {
|
||||
* backstop until a lead is recorded.
|
||||
*/
|
||||
public void onReplyQueued(String target) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on {}, skipping reminder", target);
|
||||
return;
|
||||
@@ -340,7 +517,7 @@ public final class ReplyPushLoop {
|
||||
* @param failed whether the ticket ended in a failure phase rather than {@code DONE}
|
||||
*/
|
||||
public void onTicketTerminal(String ticket, String target, boolean failed) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on ticket {} (target {}), skipping nudge",
|
||||
ticket, target);
|
||||
@@ -375,7 +552,7 @@ public final class ReplyPushLoop {
|
||||
* @param question the question text
|
||||
*/
|
||||
public void onQuestionOpened(String ticket, String target, String turnId, String question) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on {}'s question (turnId {}), skipping nudge",
|
||||
target, turnId);
|
||||
@@ -394,6 +571,48 @@ public final class ReplyPushLoop {
|
||||
pendingQuestions.remove(turnId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Queue a one-shot backend credential outage notice for every distinct lead that owns an
|
||||
* affected worker. A successful injection records its {@code (incidentId, lead)} key, so a
|
||||
* repeat report never reminds that lead again.
|
||||
*/
|
||||
public void onBackendIncident(String incidentId, Collection<String> targets, String credential,
|
||||
Collection<String> profiles, int remainingCoolOffSeconds) {
|
||||
Map<String, List<String>> targetsByLead = new ConcurrentHashMap<>();
|
||||
for (String target : targets) {
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.warn("push: backend incident {} has no known lead for target {}", incidentId, target);
|
||||
continue;
|
||||
}
|
||||
targetsByLead.computeIfAbsent(lead.get(), _ -> new ArrayList<>()).add(target);
|
||||
}
|
||||
List<String> profileNames = profiles.stream().sorted().toList();
|
||||
for (var entry : targetsByLead.entrySet()) {
|
||||
IncidentLead key = new IncidentLead(incidentId, entry.getKey());
|
||||
if (deliveredIncidents.contains(key)) continue;
|
||||
pendingIncidents.putIfAbsent(key, new PendingIncident(key, credential, profileNames,
|
||||
entry.getValue().stream().sorted().toList(), remainingCoolOffSeconds, 0));
|
||||
startOrCoalesce(entry.getKey());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Tell the owning lead that a classified backend error could not be tied to a credential.
|
||||
* Without an owning lead, emit a warning because no control can act on the target.
|
||||
*/
|
||||
public void onBackendTargetUnmapped(String target, String reason) {
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.warn("push: backend target {} could not map to a credential: {}", target, reason);
|
||||
return;
|
||||
}
|
||||
UnmappedTargetLead key = new UnmappedTargetLead(target, reason, lead.get());
|
||||
if (deliveredUnmappedTargets.contains(key)) return;
|
||||
pendingUnmappedTargets.putIfAbsent(key, new PendingUnmappedTarget(key, 0));
|
||||
startOrCoalesce(lead.get());
|
||||
}
|
||||
|
||||
// --- the schedule ----------------------------------------------------------------------------
|
||||
|
||||
/** Start a reminder schedule for {@code lead}, or join the one already running. */
|
||||
@@ -425,18 +644,25 @@ public final class ReplyPushLoop {
|
||||
Set<String> repliesBefore = pendingReplyTargetsFor(lead);
|
||||
Set<String> ticketsBefore = pendingTicketIdsFor(lead);
|
||||
Set<String> questionsBefore = pendingQuestionTurnIdsFor(lead);
|
||||
Set<IncidentLead> incidentsBefore = pendingIncidentKeysFor(lead);
|
||||
Set<UnmappedTargetLead> unmappedTargetsBefore = pendingUnmappedTargetKeysFor(lead);
|
||||
int replyReminderCount = minReplyNudgeCountFor(lead);
|
||||
int ticketReminderCount = minTicketNudgeCountFor(lead);
|
||||
int questionReminderCount = minQuestionNudgeCountFor(lead);
|
||||
var action = decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount);
|
||||
int incidentReminderCount = minIncidentNudgeCountFor(lead);
|
||||
int unmappedTargetReminderCount = minUnmappedTargetNudgeCountFor(lead);
|
||||
var action = decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
incidentReminderCount, unmappedTargetReminderCount);
|
||||
switch (action) {
|
||||
case INJECT -> {
|
||||
injectNudge(lead, replyReminderCount, ticketReminderCount, questionReminderCount);
|
||||
injectNudge(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
incidentReminderCount, unmappedTargetReminderCount);
|
||||
scheduleNext(lead);
|
||||
}
|
||||
// Re-check after the configured backoff; the lead may become injectable soon.
|
||||
case WAIT_BUSY -> scheduleNext(lead);
|
||||
case STOP -> stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore);
|
||||
case STOP -> stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, incidentsBefore,
|
||||
unmappedTargetsBefore);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -480,11 +706,25 @@ public final class ReplyPushLoop {
|
||||
* decision-to-release window reclaims the schedule slot exactly like a raced-in reply or ticket.
|
||||
*/
|
||||
void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
|
||||
Set<String> questionsBefore) {
|
||||
Set<String> questionsBefore) {
|
||||
stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, pendingIncidentKeysFor(lead));
|
||||
}
|
||||
|
||||
private void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
|
||||
Set<String> questionsBefore, Set<IncidentLead> incidentsBefore) {
|
||||
stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, incidentsBefore,
|
||||
pendingUnmappedTargetKeysFor(lead));
|
||||
}
|
||||
|
||||
private void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
|
||||
Set<String> questionsBefore, Set<IncidentLead> incidentsBefore,
|
||||
Set<UnmappedTargetLead> unmappedTargetsBefore) {
|
||||
activeLeads.remove(lead);
|
||||
boolean racedIn = pendingReplyTargetsFor(lead).stream().anyMatch(t -> !repliesBefore.contains(t))
|
||||
|| pendingTicketIdsFor(lead).stream().anyMatch(t -> !ticketsBefore.contains(t))
|
||||
|| pendingQuestionTurnIdsFor(lead).stream().anyMatch(t -> !questionsBefore.contains(t));
|
||||
|| pendingQuestionTurnIdsFor(lead).stream().anyMatch(t -> !questionsBefore.contains(t))
|
||||
|| pendingIncidentKeysFor(lead).stream().anyMatch(i -> !incidentsBefore.contains(i))
|
||||
|| pendingUnmappedTargetKeysFor(lead).stream().anyMatch(i -> !unmappedTargetsBefore.contains(i));
|
||||
if (racedIn && activeLeads.putIfAbsent(lead, Boolean.TRUE) == null) {
|
||||
log.debug("push: new work for lead {} raced the reminder loop's stop — restarting", lead);
|
||||
scheduleNext(lead);
|
||||
@@ -495,18 +735,22 @@ public final class ReplyPushLoop {
|
||||
|
||||
/** Send one combined nudge covering everything currently pending for {@code lead}. */
|
||||
private void injectNudge(String lead, int replyReminderCount, int ticketReminderCount,
|
||||
int questionReminderCount) {
|
||||
int questionReminderCount, int incidentReminderCount,
|
||||
int unmappedTargetReminderCount) {
|
||||
// Re-read rather than threading it down from decide(): a reply can drain, a ticket be
|
||||
// collected, or a question be answered (or another arrive), between the decision and the
|
||||
// injection.
|
||||
Set<String> replyTargets = pendingReplyTargetsFor(lead);
|
||||
List<PendingTicket> tickets = pendingTicketsFor(lead);
|
||||
List<PendingQuestion> questions = pendingQuestionsFor(lead);
|
||||
if (replyTargets.isEmpty() && tickets.isEmpty() && questions.isEmpty()) {
|
||||
List<PendingIncident> incidents = pendingIncidentsFor(lead);
|
||||
List<PendingUnmappedTarget> unmappedTargets = pendingUnmappedTargetsFor(lead);
|
||||
if (replyTargets.isEmpty() && tickets.isEmpty() && questions.isEmpty() && incidents.isEmpty()
|
||||
&& unmappedTargets.isEmpty()) {
|
||||
log.debug("push: pending work for lead {} drained before the nudge could be sent", lead);
|
||||
return;
|
||||
}
|
||||
String nudge = formatNudge(replyTargets, tickets, questions);
|
||||
String nudge = formatNudge(replyTargets, tickets, questions, incidents, unmappedTargets);
|
||||
try {
|
||||
agents.send(lead, nudge);
|
||||
log.debug("push: nudge sent to lead {} (reply {}/{}, ticket {}/{}, question {}/{}; "
|
||||
@@ -514,7 +758,17 @@ public final class ReplyPushLoop {
|
||||
lead, replyReminderCount + 1, maxReminders, ticketReminderCount + 1, maxReminders,
|
||||
questionReminderCount + 1, maxReminders,
|
||||
replyTargets.size(), tickets.size(), questions.size());
|
||||
countNudge("delivered");
|
||||
countNudge("sent");
|
||||
for (PendingIncident incident : incidents) {
|
||||
if (pendingIncidents.remove(incident.key(), incident)) {
|
||||
deliveredIncidents.add(incident.key());
|
||||
}
|
||||
}
|
||||
for (PendingUnmappedTarget unmappedTarget : unmappedTargets) {
|
||||
if (pendingUnmappedTargets.remove(unmappedTarget.key(), unmappedTarget)) {
|
||||
deliveredUnmappedTargets.add(unmappedTarget.key());
|
||||
}
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("push: failed to nudge lead {} (reply {}/{}, ticket {}/{}, question {}/{}): {}",
|
||||
lead, replyReminderCount + 1, maxReminders, ticketReminderCount + 1, maxReminders,
|
||||
@@ -526,12 +780,13 @@ public final class ReplyPushLoop {
|
||||
// item already at or over the cap keeps riding along in the text (still pending, still
|
||||
// named) but its extra bumps here are inert: decide() already treats it as ineligible once
|
||||
// its count reaches maxReminders.
|
||||
bumpNudgeCounts(replyTargets, tickets, questions);
|
||||
bumpNudgeCounts(replyTargets, tickets, questions, incidents, unmappedTargets);
|
||||
}
|
||||
|
||||
/** Record that every one of these items was just named in a sent (or attempted) nudge. */
|
||||
private void bumpNudgeCounts(Set<String> replyTargets, List<PendingTicket> tickets,
|
||||
List<PendingQuestion> questions) {
|
||||
List<PendingQuestion> questions, List<PendingIncident> incidents,
|
||||
List<PendingUnmappedTarget> unmappedTargets) {
|
||||
for (String target : replyTargets) {
|
||||
pendingReplies.computeIfPresent(target, (t, e) -> new ReplyEntry(e.lead(), e.nudgeCount() + 1));
|
||||
}
|
||||
@@ -544,6 +799,14 @@ public final class ReplyPushLoop {
|
||||
new PendingQuestion(e.turnId(), e.ticket(), e.target(), e.lead(), e.question(),
|
||||
e.nudgeCount() + 1));
|
||||
}
|
||||
for (PendingIncident incident : incidents) {
|
||||
pendingIncidents.computeIfPresent(incident.key(), (id, e) -> new PendingIncident(e.key(),
|
||||
e.credential(), e.profiles(), e.targets(), e.remainingCoolOffSeconds(), e.nudgeCount() + 1));
|
||||
}
|
||||
for (PendingUnmappedTarget unmappedTarget : unmappedTargets) {
|
||||
pendingUnmappedTargets.computeIfPresent(unmappedTarget.key(), (id, e) ->
|
||||
new PendingUnmappedTarget(e.key(), e.nudgeCount() + 1));
|
||||
}
|
||||
}
|
||||
|
||||
/** Schedule the next tick on the scheduler thread pool. */
|
||||
@@ -556,7 +819,8 @@ public final class ReplyPushLoop {
|
||||
|
||||
/** Render everything pending for one lead as a single nudge line. */
|
||||
private static String formatNudge(Set<String> replyTargets, List<PendingTicket> tickets,
|
||||
List<PendingQuestion> questions) {
|
||||
List<PendingQuestion> questions, List<PendingIncident> incidents,
|
||||
List<PendingUnmappedTarget> unmappedTargets) {
|
||||
List<String> parts = new ArrayList<>();
|
||||
if (!replyTargets.isEmpty()) {
|
||||
parts.add(formatRepliesNudge(replyTargets));
|
||||
@@ -567,6 +831,12 @@ public final class ReplyPushLoop {
|
||||
if (!questions.isEmpty()) {
|
||||
parts.add(formatQuestionsNudge(questions));
|
||||
}
|
||||
if (!incidents.isEmpty()) {
|
||||
parts.addAll(incidents.stream().map(ReplyPushLoop::formatBackendIncidentNudge).toList());
|
||||
}
|
||||
if (!unmappedTargets.isEmpty()) {
|
||||
parts.addAll(unmappedTargets.stream().map(ReplyPushLoop::formatUnmappedTargetNudge).toList());
|
||||
}
|
||||
return String.join(" | ", parts);
|
||||
}
|
||||
|
||||
@@ -606,6 +876,16 @@ public final class ReplyPushLoop {
|
||||
return QUESTIONS_NUDGE_FORMAT.formatted(pending.size(), ids);
|
||||
}
|
||||
|
||||
private static String formatBackendIncidentNudge(PendingIncident incident) {
|
||||
return BACKEND_INCIDENT_NUDGE_FORMAT.formatted(incident.credential(), incident.remainingCoolOffSeconds(),
|
||||
String.join(", ", incident.profiles()), String.join(", ", incident.targets()));
|
||||
}
|
||||
|
||||
private static String formatUnmappedTargetNudge(PendingUnmappedTarget unmappedTarget) {
|
||||
return BACKEND_TARGET_UNMAPPED_NUDGE_FORMAT.formatted(unmappedTarget.key().target(),
|
||||
unmappedTarget.key().reason());
|
||||
}
|
||||
|
||||
// --- lifecycle -----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@@ -627,6 +907,10 @@ public final class ReplyPushLoop {
|
||||
pendingReplies.clear();
|
||||
pendingTickets.clear();
|
||||
pendingQuestions.clear();
|
||||
pendingIncidents.clear();
|
||||
deliveredIncidents.clear();
|
||||
pendingUnmappedTargets.clear();
|
||||
deliveredUnmappedTargets.clear();
|
||||
}
|
||||
|
||||
/** @see #stop() */
|
||||
|
||||
@@ -9,12 +9,19 @@ import java.util.concurrent.CompletableFuture;
|
||||
public final class TurnToken {
|
||||
private final String target;
|
||||
private final CompletableFuture<Rendezvous.Resolution> waiter;
|
||||
private final String injectedText;
|
||||
|
||||
public TurnToken(String target, CompletableFuture<Rendezvous.Resolution> waiter) {
|
||||
this(target, waiter, null);
|
||||
}
|
||||
|
||||
public TurnToken(String target, CompletableFuture<Rendezvous.Resolution> waiter, String injectedText) {
|
||||
this.target = target;
|
||||
this.waiter = waiter;
|
||||
this.injectedText = injectedText;
|
||||
}
|
||||
|
||||
public String target() { return target; }
|
||||
public CompletableFuture<Rendezvous.Resolution> waiter() { return waiter; }
|
||||
public String injectedText() { return injectedText; }
|
||||
}
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
package dev.ltms.fleet.peer;
|
||||
|
||||
import dev.ltms.fleet.placement.PlacementDecision;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
@@ -143,9 +145,160 @@ public interface PeerLauncher {
|
||||
|
||||
/**
|
||||
* The profile a no-argument {@link #spawn(SpawnRequest)} uses, or {@code null} if none is configured.
|
||||
*
|
||||
* <p>fleetd #425: for an implementation with role pools (a no-argument spawn is read as {@link
|
||||
* MemberRole#DEV}, see {@link SpawnRequest}), this must be the profile a live spawn of that role
|
||||
* would actually be placed on right now, not a value captured once at startup — a caller such as
|
||||
* {@code fleet_profiles} relies on this to report a live, not frozen, fact.
|
||||
*/
|
||||
String defaultProfile();
|
||||
|
||||
/**
|
||||
* The profile an unqualified spawn of {@code role} would resolve to right now — the role-aware,
|
||||
* live counterpart of {@link #defaultProfile()} (fleetd #425).
|
||||
*
|
||||
* <p>A caller that must provision something profile-specific (working directory, parity overlay
|
||||
* files) <em>before</em> the actual spawn — {@code SessionManager.acquireWithWorktree} is the one
|
||||
* that exists today — needs the exact profile that spawn will use, for the caller's real role,
|
||||
* not a role-agnostic guess. Calling {@link #defaultProfile()} for that purpose reads {@code
|
||||
* MemberRole#DEV}'s answer regardless of the caller's actual role, which is wrong for any other
|
||||
* role and can provision for a profile the spawn never lands on.
|
||||
*
|
||||
* <p>Default implementation returns {@link #defaultProfile()}, ignoring {@code role} — the right
|
||||
* answer for a launcher with no role-pool concept of its own (e.g. a single {@code
|
||||
* HerdrPeerLauncher} adapter, which is never reached this way in production: {@code
|
||||
* CompositePeerLauncher} always fronts it and resolves roles itself).
|
||||
*
|
||||
* <p>fleetd #453: this default is deliberately <em>not</em> abstract — unlike {@link
|
||||
* #spawn(SpawnRequest, PlacementDecision)} (fleetd #450), there is no live defect in inheriting
|
||||
* it today, and the only current single-adapter implementer ({@code HerdrPeerLauncher}) is
|
||||
* correct to do so. But it stays correct only as long as that holds: <strong>if a launcher ever
|
||||
* routes more than one profile per role, it MUST override this method</strong>, or every role
|
||||
* silently resolves to {@link #defaultProfile()} with no error and no log line. {@code
|
||||
* HerdrPeerLauncher.spawn(SpawnRequest, PlacementDecision)} — the override in {@code
|
||||
* dev.ltms.fleet.member}, not the declaration below — names this method and {@link #place}
|
||||
* explicitly as "unoverridden here" for exactly this reason. Read it before adding role-pool
|
||||
* routing to any {@code HerdrPeerLauncher} subclass.
|
||||
*
|
||||
* <p>Who is forced to read which paragraph, because it is not symmetric. A new class that
|
||||
* implements this interface directly must write a body for {@link #spawn(SpawnRequest,
|
||||
* PlacementDecision)}, which is abstract here, so it lands on this javadoc. A subclass of
|
||||
* {@code HerdrPeerLauncher} does not: that class already implements the method, and the
|
||||
* subclass inherits the body. So for a subclass this paragraph is advice, not a gate.
|
||||
*/
|
||||
default String defaultProfileFor(MemberRole role) {
|
||||
return defaultProfile();
|
||||
}
|
||||
|
||||
/**
|
||||
* The profile an <em>unqualified</em> spawn of {@code role} would actually be routed to right
|
||||
* now — the same candidate list, the same {@code quarantined}/{@code coolingOff}/{@code
|
||||
* modelOff} filtering, and the same {@code PlacementPolicy} that {@link #spawn} itself
|
||||
* consults for a blank-profile request (fleetd #425 rework).
|
||||
*
|
||||
* <p>This is <em>not</em> {@link #defaultProfileFor}: that method answers "what is first in
|
||||
* {@code role}'s pool", blind to quarantine, cool-off, and the model on/off gate — the right
|
||||
* answer for a role-agnostic, best-effort report ({@code fleet_profiles}' {@code "default"}
|
||||
* field), but the wrong one for a caller that needs the profile a spawn will actually land on.
|
||||
* A quarantined or model-off pool-first profile makes {@link #defaultProfileFor} return a name
|
||||
* an unqualified spawn will never be routed to.
|
||||
*
|
||||
* <p>Just the resolved name, not the full {@link PlacementDecision} — a caller that only wants
|
||||
* to know the answer (a status report, a log line) can call this; a caller that will later
|
||||
* <em>act</em> on the answer by spawning — provisioning a worktree for a specific profile
|
||||
* before the peer exists is the one that matters — must call {@link #place} and carry the
|
||||
* {@link PlacementDecision} itself through to {@link #spawn(SpawnRequest, PlacementDecision)}
|
||||
* instead of calling this method and feeding the string back in as an explicit profile. Doing
|
||||
* that re-enters {@link #spawn(SpawnRequest)}'s explicit-profile branch, which disagrees with
|
||||
* the routing branch on purpose about what an excluded profile means: the routing branch (and
|
||||
* {@link #place}) falls through a quarantined/cooling-off/at-cap/model-off/unreachable/weight-0
|
||||
* profile to the next candidate, while the explicit branch refuses outright — correct for an
|
||||
* operator who named that profile on purpose, wrong for a name that only ever came from placement
|
||||
* itself. That accidental refusal is exactly the regression fleetd #425 rework round 2 fixes:
|
||||
* the default implementation below delegates to {@link #place}, so the two can never drift apart,
|
||||
* but a caller that resolves through this method alone and spawns separately can still recreate
|
||||
* the round-1 defect for itself. (Before fleetd #435, this accident was also reachable through
|
||||
* {@code maxLoad} specifically, because {@code FixedPlacementPolicy} — the default policy — did
|
||||
* not evaluate it at all for automatic selection; #435 closed that gap, so a placement decision
|
||||
* can no longer be at cap in the first place. The refusal-vs-fall-through disagreement above is
|
||||
* the part that was never about {@code maxLoad} and is still real.)
|
||||
*
|
||||
* @throws RuntimeException (implementation-specific, typically a placement exception) if no
|
||||
* candidate in {@code role}'s pool is currently placeable
|
||||
*/
|
||||
default String routedProfileFor(MemberRole role) {
|
||||
return place(role).profile();
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve, <em>without spawning</em>, the {@link PlacementDecision} an unqualified spawn of
|
||||
* {@code role} would make right now — the same candidate list, the same {@code
|
||||
* quarantined}/{@code coolingOff}/{@code modelOff} filtering, and the same {@code
|
||||
* PlacementPolicy} {@link #spawn(SpawnRequest)}'s blank-profile branch itself consults (fleetd
|
||||
* #425 rework).
|
||||
*
|
||||
* <p>Pair this with {@link #spawn(SpawnRequest, PlacementDecision)}, never with {@link
|
||||
* #spawn(SpawnRequest)} fed the decision's profile as an explicit name — see {@link
|
||||
* PlacementDecision}'s own javadoc for why that second form regressed.
|
||||
*
|
||||
* <p>Default implementation wraps {@link #defaultProfile()}, ignoring {@code role} and every
|
||||
* placement condition — the right answer for a launcher with no pool or placement-policy
|
||||
* concept of its own, matching {@link #defaultProfileFor}'s own default.
|
||||
*
|
||||
* <p>fleetd #453: same reasoning as {@link #defaultProfileFor}'s own #453 note — this default
|
||||
* is deliberately not abstract (no live defect today, correct for the sole single-adapter
|
||||
* implementer), but <strong>a launcher that ever routes more than one profile per role MUST
|
||||
* override this method too</strong>, or placement silently ignores {@code role} for it. See
|
||||
* {@code HerdrPeerLauncher.spawn(SpawnRequest, PlacementDecision)}'s javadoc, which names this
|
||||
* method as "unoverridden here" and why that is correct only for a single-profile adapter.
|
||||
*
|
||||
* @throws RuntimeException (implementation-specific, typically a placement exception) if no
|
||||
* candidate in {@code role}'s pool is currently placeable
|
||||
*/
|
||||
default PlacementDecision place(MemberRole role) {
|
||||
return new PlacementDecision(defaultProfile());
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn against an already-resolved {@link PlacementDecision} from {@link #place}, honoring it
|
||||
* completely: none of the conditions {@link #place} already applied — quarantine, cooling off,
|
||||
* {@code maxLoad} (evaluated by every placement policy including the default {@code fixed},
|
||||
* since fleetd #435), model-off — are re-evaluated here; {@code decision} already reflects them.
|
||||
* This is not skipping a check {@code place} left undone; it is not repeating one {@code place}
|
||||
* already did, and not re-opening the window between that decision and this spawn in which the
|
||||
* underlying state could otherwise move. This is what lets a resolve-then-spawn caller
|
||||
* ({@code SessionManager.acquireWithWorktree}, which must know the profile before it can
|
||||
* provision a worktree for it) and a plain blank-profile {@link #spawn(SpawnRequest)} caller
|
||||
* land on the exact same outcome for the exact same placement state (fleetd #425 rework,
|
||||
* round 2).
|
||||
*
|
||||
* <p>{@code req}'s own {@link SpawnRequest#profileName()} is ignored in favor of {@code
|
||||
* decision.profile()} — the caller is expected to have built {@code req} with a blank or
|
||||
* matching profile; passing a request that names a <em>different</em>, explicit profile than
|
||||
* the decision it is paired with is a caller bug this method does not attempt to detect.
|
||||
*
|
||||
* <p>No default implementation (fleetd #450): the two correct bodies disagree on purpose, so an
|
||||
* implementer must choose one rather than silently inherit whichever this interface happened to
|
||||
* provide. An implementer with no placement concept of its own — spawns a single profile, e.g.
|
||||
* {@code HerdrPeerLauncher} — should delegate to {@link #spawn(SpawnRequest)} with the decision's
|
||||
* profile named explicitly, since there is no separate routing path to honor there: the
|
||||
* explicit-profile branch it re-enters and the routing branch {@link #place} would have used are
|
||||
* the same thing. <strong>A launcher that routes across more than one profile — the way {@code
|
||||
* CompositePeerLauncher} routes across every configured adapter — MUST NOT re-enter {@link
|
||||
* #spawn(SpawnRequest)}.</strong> Doing so re-applies that single-argument method's
|
||||
* explicit-profile checks ({@code enforceNotQuarantined}, {@code enforceNotCoolingOff}, {@code
|
||||
* enforceMaxLoad}, {@code enforceModelEnabled} in {@code CompositePeerLauncher}), which can
|
||||
* refuse the very profile {@link #place} just chose, if the underlying placement state moved in
|
||||
* the window between the {@link #place} call and this one — the exact window this method and
|
||||
* {@link PlacementDecision} exist to close (fleetd #444). Before #450 this was a {@code default}
|
||||
* method that only {@code CompositePeerLauncher} overrode; a future placement-doing launcher
|
||||
* could have inherited the re-entering body silently and never known. Making it abstract turns
|
||||
* that silent inheritance into a compile error.
|
||||
*
|
||||
* @throws IllegalArgumentException if the decision names an unknown profile
|
||||
*/
|
||||
PeerHandle spawn(SpawnRequest req, PlacementDecision decision);
|
||||
|
||||
/**
|
||||
* Resolve the effective working directory for a spawn {@code req} without actually spawning.
|
||||
* Resolution order: requestedCwd → profile cwd → callerCwd → daemon cwd.
|
||||
@@ -192,4 +345,66 @@ public interface PeerLauncher {
|
||||
* @return {@code true} when a reset was sent and its status transition must settle before reuse
|
||||
*/
|
||||
boolean clearContext(String id);
|
||||
|
||||
/**
|
||||
* Model ids the operator has currently turned off in the central {@code models.allow:} list
|
||||
* (fleetd #422) — empty for a launcher with nothing to gate against. {@code fleet_profiles}/
|
||||
* {@code GET /profiles} (via {@code FleetMcp.profilesView}) call this to report which models
|
||||
* are off, and MUST read this exact accessor rather than deriving their own answer: the fleetd
|
||||
* #404 lesson is that a status field reading a different source than the behaviour it describes
|
||||
* can drift from what the gate ({@code CompositePeerLauncher.enforceModelEnabled} and its
|
||||
* candidate filter) actually enforces. A default of {@code Set.of()} keeps every other {@link
|
||||
* PeerLauncher} implementer (the herdr adapters, and the two test-fake implementers) unchanged.
|
||||
*
|
||||
* <p>fleetd #422 follow-up: this alone cannot tell "no {@code models:} block at all" from "a
|
||||
* {@code models:} block where nothing is currently off" — both report an empty set here. Delegates
|
||||
* to {@link #modelGateState()} so the two facts always come from the one read {@link
|
||||
* #modelGateState()}'s implementer makes; do not override this method separately from that one.
|
||||
*/
|
||||
default Set<String> disabledModels() {
|
||||
return modelGateState().off();
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the central {@code models.allow:} gate (fleetd #422) is armed at all, together with
|
||||
* which model ids are currently off — fleetd #422 follow-up. {@link #disabledModels()} alone
|
||||
* cannot distinguish two states that both report an empty set: a host with no {@code models:}
|
||||
* block (nothing is gated, and nothing can be) and a host WITH a {@code models:} block where
|
||||
* nothing is currently turned off (the gate is armed and reporting zero). This method exists so
|
||||
* a caller — the startup log, {@code fleet_profiles}/{@code GET /profiles} — can tell the two
|
||||
* apart, the same reason {@code CompletionResolver.UnsetMeaning} exists: an accessor that can
|
||||
* legitimately report "empty" must never let a caller guess why.
|
||||
*
|
||||
* <p>Default {@link ModelGateState#notConfigured()} — every launcher without a {@code models:}
|
||||
* block to read from (the herdr adapters, and the two test-fake implementers), matching {@link
|
||||
* #disabledModels()}'s own default of an empty set.
|
||||
*/
|
||||
default ModelGateState modelGateState() {
|
||||
return ModelGateState.notConfigured();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: the result of {@link #modelGateState()} — see that method's javadoc
|
||||
* for why "armed" and "off" must be reported together from one read rather than as two
|
||||
* separately-derived facts that a reload landing between them could make disagree.
|
||||
*
|
||||
* @param configured {@code true} when a {@code models:} block exists at all (armed), regardless
|
||||
* of whether anything in it is currently turned off; {@code false} when there
|
||||
* is no block to gate against
|
||||
* @param off the model ids currently turned off; always empty when {@code configured} is
|
||||
* {@code false}
|
||||
*/
|
||||
record ModelGateState(boolean configured, Set<String> off) {
|
||||
public ModelGateState {
|
||||
off = Set.copyOf(off);
|
||||
}
|
||||
|
||||
public static ModelGateState notConfigured() {
|
||||
return new ModelGateState(false, Set.of());
|
||||
}
|
||||
|
||||
public static ModelGateState armed(Set<String> off) {
|
||||
return new ModelGateState(true, off);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,4 +29,9 @@ public record SpawnRequest(String profileName, String requestedCwd, String calle
|
||||
String sessionName, String resumeSessionId) {
|
||||
this(profileName, requestedCwd, callerCwd, sessionName, resumeSessionId, null);
|
||||
}
|
||||
|
||||
/** Return a copy of this request with {@code profileName} replaced by {@code profile}. */
|
||||
public SpawnRequest withProfile(String profile) {
|
||||
return new SpawnRequest(profile, requestedCwd, callerCwd, sessionName, resumeSessionId, role);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,200 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
* Fleetd #201 / #227: correlates classified backend errors (credential outage or provider 5xx,
|
||||
* produced elsewhere by the classifier that turns raw pane text into a typed event — this class
|
||||
* knows nothing about pane text, profiles, sessions, launchers, or leads) and decides, purely from
|
||||
* counts and timing, when a credential's backend is out.
|
||||
*
|
||||
* <p>Two classified errors on the <b>same credential</b> — never the profile name, never the error
|
||||
* text, see {@code FleetConfig.Profile#effectiveCredentialId()} like {@link BackendQuarantine} —
|
||||
* <b>from two distinct targets</b> inside a 60-second window is treated as an outage: {@link
|
||||
* #record} then returns an {@link Incident} and starts a 60-second cool-off for that credential. The
|
||||
* threshold counts distinct targets, not raw events, on purpose: the classifier is a heuristic and a
|
||||
* valid member report can quote an {@code API Error:} line, so one target repeating that line twice
|
||||
* must never remove fleet capacity by itself. Two different targets independently producing a
|
||||
* classified error is far less likely to be a coincidence, and a real backend outage hits every
|
||||
* target on that credential anyway, so this loses nothing against the case being protected against.
|
||||
* This is deliberately a different store from {@link BackendQuarantine}: that one holds a
|
||||
* 1800-second exhaustion cooldown for a spent credential, and reusing it here would both use the
|
||||
* wrong duration and report "backend exhausted" for what is a short transient fault.
|
||||
*
|
||||
* <p>Correlation and cool-off live in <b>one class</b> so that crossing the threshold and setting
|
||||
* the cool-off deadline happen as a single atomic update: every {@link #record} call goes through
|
||||
* {@link ConcurrentHashMap#compute}, which serializes on the credential's map bucket, so a
|
||||
* concurrent second and third event on the same credential can never both observe "about to cross"
|
||||
* and both mint an incident.
|
||||
*
|
||||
* <p>The clock is injected ({@link LongSupplier}, conventionally {@code System::nanoTime} like
|
||||
* {@link BackendQuarantine}), never read inline, so the window and cool-off are testable without a
|
||||
* real sleep.
|
||||
*/
|
||||
public final class BackendOutagePolicy {
|
||||
|
||||
/** Distinct targets a credential needs a classified error from to declare an outage. */
|
||||
public static final int THRESHOLD = 2;
|
||||
|
||||
/** How long a credential's evidence stays fresh, inclusive of both endpoints. */
|
||||
public static final long WINDOW_NANOS = 60_000_000_000L;
|
||||
|
||||
/** How long a credential sits out once the threshold is crossed. */
|
||||
public static final long COOLOFF_NANOS = 60_000_000_000L;
|
||||
|
||||
private final ConcurrentHashMap<String, CredentialState> states = new ConcurrentHashMap<>();
|
||||
private final LongSupplier nowNanos;
|
||||
private final AtomicLong incidentSequence = new AtomicLong();
|
||||
|
||||
public BackendOutagePolicy(LongSupplier nowNanos) {
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
}
|
||||
|
||||
/**
|
||||
* Records one classified backend error for {@code credentialId} against {@code target} (e.g. a
|
||||
* session or worker id — this class never interprets it, only collects it for the incident's
|
||||
* affected-targets list, and counts distinct targets toward the threshold) with {@code reason}
|
||||
* (the classifier's free-text reason, kept per event for whoever renders the eventual notice —
|
||||
* every event's reason is kept even when the same target repeats, so {@link Incident#reasons()}
|
||||
* can be longer than {@link Incident#evidenceCount()}).
|
||||
*
|
||||
* <p>Returns a populated {@link Incident} only at the exact moment a <b>second distinct target</b>
|
||||
* is seen for this credential inside the window — never before, and never again while the
|
||||
* resulting cool-off is active. A repeat error from a target already counted does not advance the
|
||||
* threshold. While a credential is cooling off, a fresh error is ignored outright: it neither
|
||||
* extends the deadline nor produces another incident. Once the cool-off has elapsed, the next
|
||||
* error clears the old evidence and starts a brand-new window — two fresh, distinct targets are
|
||||
* required to rearm.
|
||||
*/
|
||||
public Optional<Incident> record(String credentialId, String target, String reason) {
|
||||
Objects.requireNonNull(credentialId, "credentialId");
|
||||
Objects.requireNonNull(target, "target");
|
||||
Objects.requireNonNull(reason, "reason");
|
||||
long now = nowNanos.getAsLong();
|
||||
|
||||
AtomicReference<Incident> minted = new AtomicReference<>();
|
||||
states.compute(credentialId, (_, existing) -> {
|
||||
CredentialState state = existing;
|
||||
|
||||
if (state != null && state.inCoolOff()) {
|
||||
if (now < state.coolOffUntilNanos) {
|
||||
return state; // still cooling off: no extension, no incident
|
||||
}
|
||||
state = null; // cool-off elapsed: evidence is cleared, rearm from scratch
|
||||
}
|
||||
|
||||
if (state == null || now - state.firstEventNanos > WINDOW_NANOS) {
|
||||
return CredentialState.first(now, target, reason);
|
||||
}
|
||||
|
||||
CredentialState advanced = state.withAdditionalEvidence(target, reason);
|
||||
if (advanced.evidenceCount() < THRESHOLD) {
|
||||
return advanced;
|
||||
}
|
||||
|
||||
long coolOffUntilNanos = now + COOLOFF_NANOS;
|
||||
minted.set(new Incident(
|
||||
"outage-" + credentialId + "-" + incidentSequence.incrementAndGet(),
|
||||
credentialId,
|
||||
Set.copyOf(advanced.targets),
|
||||
advanced.evidenceCount(),
|
||||
toSecondsRoundedUp(WINDOW_NANOS),
|
||||
toSecondsRoundedUp(COOLOFF_NANOS),
|
||||
List.copyOf(advanced.reasons)));
|
||||
return advanced.enteringCoolOff(coolOffUntilNanos);
|
||||
});
|
||||
|
||||
return Optional.ofNullable(minted.get());
|
||||
}
|
||||
|
||||
/** Seconds left on {@code credentialId}'s cool-off, or empty when it is not cooling off. */
|
||||
public OptionalLong remainingCoolOffSeconds(String credentialId) {
|
||||
Objects.requireNonNull(credentialId, "credentialId");
|
||||
CredentialState state = states.get(credentialId);
|
||||
if (state == null || !state.inCoolOff()) {
|
||||
return OptionalLong.empty();
|
||||
}
|
||||
long remaining = state.coolOffUntilNanos - nowNanos.getAsLong();
|
||||
return remaining > 0 ? OptionalLong.of(toSecondsRoundedUp(remaining)) : OptionalLong.empty();
|
||||
}
|
||||
|
||||
private static long toSecondsRoundedUp(long nanos) {
|
||||
return (nanos + 999_999_999L) / 1_000_000_000L;
|
||||
}
|
||||
|
||||
/**
|
||||
* One credential crossing the outage threshold. {@code remainingCoolOffSeconds} is the cool-off
|
||||
* length as observed at the moment of minting — this incident is only ever produced right as the
|
||||
* cool-off starts, so it is always the full {@link #COOLOFF_NANOS} rounded up.
|
||||
*
|
||||
* <p>{@code evidenceCount} is {@code targets.size()} — the threshold is on distinct targets, not
|
||||
* raw events — while {@code reasons} keeps every event's reason, including repeats from a target
|
||||
* already counted. The two are deliberately different lengths: a single target hammering the same
|
||||
* classified error never grows {@code evidenceCount} past 1, but each occurrence still lands in
|
||||
* {@code reasons} for whoever renders the notice.
|
||||
*/
|
||||
public record Incident(
|
||||
String id,
|
||||
String credentialId,
|
||||
Set<String> targets,
|
||||
int evidenceCount,
|
||||
long windowSeconds,
|
||||
long remainingCoolOffSeconds,
|
||||
List<String> reasons) {
|
||||
}
|
||||
|
||||
/** Evidence accumulated for one credential since its window opened, or its active cool-off. */
|
||||
private static final class CredentialState {
|
||||
final long firstEventNanos;
|
||||
final Set<String> targets;
|
||||
final List<String> reasons;
|
||||
final long coolOffUntilNanos; // 0 means "not cooling off"
|
||||
|
||||
private CredentialState(long firstEventNanos, Set<String> targets, List<String> reasons,
|
||||
long coolOffUntilNanos) {
|
||||
this.firstEventNanos = firstEventNanos;
|
||||
this.targets = targets;
|
||||
this.reasons = reasons;
|
||||
this.coolOffUntilNanos = coolOffUntilNanos;
|
||||
}
|
||||
|
||||
static CredentialState first(long nowNanos, String target, String reason) {
|
||||
Set<String> targets = new LinkedHashSet<>();
|
||||
targets.add(target);
|
||||
List<String> reasons = new ArrayList<>();
|
||||
reasons.add(reason);
|
||||
return new CredentialState(nowNanos, targets, reasons, 0L);
|
||||
}
|
||||
|
||||
CredentialState withAdditionalEvidence(String target, String reason) {
|
||||
Set<String> newTargets = new LinkedHashSet<>(targets);
|
||||
newTargets.add(target);
|
||||
List<String> newReasons = new ArrayList<>(reasons);
|
||||
newReasons.add(reason);
|
||||
return new CredentialState(firstEventNanos, newTargets, newReasons, 0L);
|
||||
}
|
||||
|
||||
CredentialState enteringCoolOff(long coolOffUntilNanos) {
|
||||
return new CredentialState(firstEventNanos, targets, reasons, coolOffUntilNanos);
|
||||
}
|
||||
|
||||
/** Distinct targets seen so far — the threshold counts this, never {@code reasons.size()}. */
|
||||
int evidenceCount() {
|
||||
return targets.size();
|
||||
}
|
||||
|
||||
boolean inCoolOff() {
|
||||
return coolOffUntilNanos > 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3,6 +3,7 @@ package dev.ltms.fleet.placement;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.LongSupplier;
|
||||
@@ -21,46 +22,158 @@ import java.util.function.LongSupplier;
|
||||
* <p>The clock is injected ({@link LongSupplier}, conventionally {@code System::nanoTime} like
|
||||
* {@code FleetHealthMonitor}), never read inline, so a quarantine's expiry is testable without a
|
||||
* real sleep.
|
||||
*
|
||||
* <h2>Escalation (fleetd #466)</h2>
|
||||
* A flat cooldown does not fit every exhaustion. A backend that reports "out of capacity for the
|
||||
* rest of the hour" recovers in one cooldown; a weekly subscription limit does not — it keeps
|
||||
* reporting exhausted on every attempt made before the window resets, so a flat 30-minute cooldown
|
||||
* (the default {@code cooldownNanos}) means roughly 336 pointless spawn attempts across a week, one
|
||||
* every cooldown.
|
||||
*
|
||||
* <p><strong>This class only ever sees the exhaustion signal.</strong> Its only production caller is
|
||||
* {@code Fleetd.exhaustionSink}, wired to fire on a {@code BACKEND_EXHAUSTED} classification alone.
|
||||
* The daemon's other outage state — a credential "cooling off" after repeated non-exhaustion
|
||||
* backend errors (an HTTP 5xx storm, say) — is a separate mechanism, {@code BackendOutagePolicy},
|
||||
* with its own short fixed 60s cooldown and no repeat tracking. The two are never merged: escalating
|
||||
* on a cooling-off signal would turn a transient 5xx storm into a multi-hour backoff, which is
|
||||
* exactly the failure this ticket is not asking for. Confirmed by reading every call site of
|
||||
* {@link #quarantine} — {@code BackendOutagePolicy} has its own {@code coolOff} method and never
|
||||
* calls this one.
|
||||
*
|
||||
* <p><strong>Mechanism</strong> — the {@link #withEscalation} constructors track, per credential, how
|
||||
* many times in a row {@link #quarantine} has been called without an intervening "quiet" gap.
|
||||
* Each call computes {@code cooldownNanos * backoffMultiplier ^ (repeatCount - 1)}, capped at
|
||||
* {@code maxCooldownNanos}. A call counts as a continuation of the same streak — {@code repeatCount}
|
||||
* increments — when it arrives no more than one base {@code cooldownNanos} after the previous
|
||||
* quarantine's deadline (this covers both "still quarantined" and "quarantine just expired and it
|
||||
* was exhausted again immediately"); otherwise the streak resets and this call is treated as a fresh
|
||||
* first occurrence at the base cooldown.
|
||||
*
|
||||
* <p><strong>Reset, honestly stated.</strong> The ideal reset signal is "the cooldown expired and the
|
||||
* next attempt succeeded" — but nothing in this codebase reports a spawn success back to this class
|
||||
* (checked: {@code SessionManager} and {@code CompositePeerLauncher} never call any method here
|
||||
* except {@link #quarantine}/{@link #isQuarantined}/{@link #remainingSeconds}, none of which is a
|
||||
* success hook). Lacking that signal, the reset used here is a time-based proxy: a base-cooldown's
|
||||
* worth of quiet — no exhaustion report for that credential — since the last quarantine ended. It is
|
||||
* not proof the credential started working again, only the best available evidence without adding an
|
||||
* active probe, which is out of scope by the operator's own design constraint (no automatic probing
|
||||
* of a limited backend).
|
||||
*
|
||||
* <p><strong>Ceiling.</strong> {@code maxCooldownNanos} bounds the growth — an unbounded backoff is a
|
||||
* permanent, unrecoverable-without-a-restart outage, which would be worse than the flat-rate bug this
|
||||
* escalation fixes. {@link #withEscalation(LongSupplier, long)} defaults the ceiling to
|
||||
* {@value #DEFAULT_MAX_COOLDOWN_MULTIPLE}x the base cooldown (12x the 1800s default ≈ 6 hours), so a
|
||||
* chronically exhausted credential still gets re-tried roughly every 6 hours instead of every 30
|
||||
* minutes — about a dozen attempts a week instead of ~336.
|
||||
*
|
||||
* <p><strong>Backward compatibility.</strong> The original two-argument {@link #BackendQuarantine(
|
||||
* LongSupplier, long)} constructor is unchanged in behaviour: it is exactly {@code
|
||||
* withEscalation}'s mechanism with {@code backoffMultiplier = 1.0} and {@code maxCooldownNanos =
|
||||
* cooldownNanos}, which collapses the formula back to the original flat {@code now + cooldownNanos}
|
||||
* on every call regardless of history. Every existing call site (roughly 20 across the test suite,
|
||||
* plus {@link #none()}) keeps its current shape and behaviour unchanged.
|
||||
*/
|
||||
public final class BackendQuarantine {
|
||||
|
||||
private final ConcurrentHashMap<String, Long> quarantinedUntilNanos = new ConcurrentHashMap<>();
|
||||
/** Default growth per consecutive exhaustion streak — see the class doc's Mechanism section. */
|
||||
static final double DEFAULT_BACKOFF_MULTIPLIER = 2.0;
|
||||
/** Default ceiling, expressed as a multiple of the base cooldown — see the class doc's Ceiling section. */
|
||||
static final long DEFAULT_MAX_COOLDOWN_MULTIPLE = 12;
|
||||
|
||||
private final ConcurrentHashMap<String, QuarantineState> quarantines = new ConcurrentHashMap<>();
|
||||
private final LongSupplier nowNanos;
|
||||
private final long cooldownNanos;
|
||||
private final double backoffMultiplier;
|
||||
private final long maxCooldownNanos;
|
||||
/** True only for {@link #none()}. See {@link #quarantine} for why this exists. */
|
||||
private final boolean inert;
|
||||
|
||||
/** How many consecutive exhaustion reports a credential is on, and when the resulting cooldown ends. */
|
||||
private record QuarantineState(int repeatCount, long deadlineNanos) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Flat cooldown, unchanged from before fleetd #466 — every {@link #quarantine} call blocks the
|
||||
* credential for exactly {@code cooldownNanos}, regardless of how many times it was called
|
||||
* before. Equivalent to {@link #withEscalation} with no growth ({@code backoffMultiplier = 1.0})
|
||||
* and a ceiling equal to the base cooldown, so it degrades to the identical {@code now +
|
||||
* cooldownNanos} formula every call. Kept for the existing call sites that want a fixed cooldown
|
||||
* (and for tests exercising the fixed-cooldown shape in isolation); production wiring uses
|
||||
* {@link #withEscalation} instead.
|
||||
*
|
||||
* @param nowNanos monotonic clock, injected for testability
|
||||
* @param cooldownNanos how long a fresh {@link #quarantine} call blocks the credential for;
|
||||
* must be positive
|
||||
*/
|
||||
public BackendQuarantine(LongSupplier nowNanos, long cooldownNanos) {
|
||||
this(nowNanos, cooldownNanos, false);
|
||||
this(nowNanos, cooldownNanos, 1.0, cooldownNanos, false);
|
||||
}
|
||||
|
||||
private BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, boolean inert) {
|
||||
/**
|
||||
* Escalating cooldown (fleetd #466) — see the class doc's Mechanism/Reset/Ceiling sections.
|
||||
*
|
||||
* @param nowNanos monotonic clock, injected for testability
|
||||
* @param cooldownNanos base cooldown, applied to a fresh (non-streak) exhaustion; must be
|
||||
* positive
|
||||
* @param backoffMultiplier growth per consecutive exhaustion; must be {@code >= 1.0} ({@code 1.0}
|
||||
* disables growth and is exactly the flat two-argument constructor)
|
||||
* @param maxCooldownNanos ceiling on the escalated cooldown; must be {@code >= cooldownNanos}
|
||||
*/
|
||||
public BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, double backoffMultiplier,
|
||||
long maxCooldownNanos) {
|
||||
this(nowNanos, cooldownNanos, backoffMultiplier, maxCooldownNanos, false);
|
||||
}
|
||||
|
||||
private BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, double backoffMultiplier,
|
||||
long maxCooldownNanos, boolean inert) {
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
if (cooldownNanos <= 0) {
|
||||
throw new IllegalArgumentException("cooldownNanos must be positive: " + cooldownNanos);
|
||||
}
|
||||
if (backoffMultiplier < 1.0) {
|
||||
throw new IllegalArgumentException("backoffMultiplier must be >= 1.0: " + backoffMultiplier);
|
||||
}
|
||||
if (maxCooldownNanos < cooldownNanos) {
|
||||
throw new IllegalArgumentException(
|
||||
"maxCooldownNanos must be >= cooldownNanos: " + maxCooldownNanos + " < " + cooldownNanos);
|
||||
}
|
||||
this.cooldownNanos = cooldownNanos;
|
||||
this.backoffMultiplier = backoffMultiplier;
|
||||
this.maxCooldownNanos = maxCooldownNanos;
|
||||
this.inert = inert;
|
||||
}
|
||||
|
||||
/**
|
||||
* Escalating cooldown with the fleetd #466 default shape: cooldown doubles
|
||||
* ({@value #DEFAULT_BACKOFF_MULTIPLIER}x) per consecutive exhaustion streak, capped at
|
||||
* {@value #DEFAULT_MAX_COOLDOWN_MULTIPLE}x the base cooldown. This is what production wiring
|
||||
* ({@code Fleetd.main}) uses.
|
||||
*
|
||||
* @param nowNanos monotonic clock, injected for testability
|
||||
* @param cooldownNanos base cooldown, applied to a fresh (non-streak) exhaustion; must be positive
|
||||
*/
|
||||
public static BackendQuarantine withEscalation(LongSupplier nowNanos, long cooldownNanos) {
|
||||
return new BackendQuarantine(nowNanos, cooldownNanos, DEFAULT_BACKOFF_MULTIPLIER,
|
||||
cooldownNanos * DEFAULT_MAX_COOLDOWN_MULTIPLE, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* Inert quarantine — {@link #quarantine} does nothing on this instance, so nothing is ever
|
||||
* quarantined. The explicit stand-in a caller (or a test not exercising this feature) passes
|
||||
* instead of a defaulting overload, exactly like {@code ExhaustedPatternLookup.none()}.
|
||||
*/
|
||||
public static BackendQuarantine none() {
|
||||
return new BackendQuarantine(() -> 0L, 1, true);
|
||||
return new BackendQuarantine(() -> 0L, 1, 1.0, 1, true);
|
||||
}
|
||||
|
||||
/**
|
||||
* Quarantine {@code credentialId} for the configured cooldown, starting now. A repeat call while
|
||||
* already quarantined restarts the cooldown at full length — a fresh refusal is fresh evidence the
|
||||
* account is still exhausted, not a reason to let an earlier, shorter wait stand.
|
||||
* Quarantine {@code credentialId} starting now. On a flat instance (the two-argument
|
||||
* constructor) this always blocks for exactly {@code cooldownNanos}, restarting the cooldown at
|
||||
* full length on every call — a fresh refusal is fresh evidence the account is still exhausted,
|
||||
* not a reason to let an earlier, shorter wait stand. On an escalating instance ({@link
|
||||
* #withEscalation}) the cooldown grows with each call that arrives within one base cooldown of
|
||||
* the previous deadline, and resets to the base cooldown once a call arrives after a longer gap
|
||||
* — see the class doc.
|
||||
*
|
||||
* <p>On {@link #none()} this is a no-op. It has to be: that instance holds a clock frozen at 0,
|
||||
* so recording a deadline would produce a quarantine that never expires — a credential locked out
|
||||
@@ -73,7 +186,13 @@ public final class BackendQuarantine {
|
||||
if (inert) {
|
||||
return;
|
||||
}
|
||||
quarantinedUntilNanos.put(credentialId, nowNanos.getAsLong() + cooldownNanos);
|
||||
long now = nowNanos.getAsLong();
|
||||
quarantines.compute(credentialId, (id, prev) -> {
|
||||
int repeatCount = (prev == null || now - prev.deadlineNanos() > cooldownNanos)
|
||||
? 1
|
||||
: prev.repeatCount() + 1;
|
||||
return new QuarantineState(repeatCount, now + escalatedCooldownNanos(repeatCount));
|
||||
});
|
||||
}
|
||||
|
||||
/** Whether {@code credentialId} is quarantined right now. */
|
||||
@@ -87,6 +206,49 @@ public final class BackendQuarantine {
|
||||
return remaining > 0 ? OptionalLong.of(toSecondsRoundedUp(remaining)) : OptionalLong.empty();
|
||||
}
|
||||
|
||||
/**
|
||||
* Remaining seconds together with which consecutive exhaustion this is (fleetd #466 scope item
|
||||
* 2) — {@code repeatCount} 1 for a first occurrence, 2 for the second in a row, and so on; see
|
||||
* {@link #quarantine}'s class-doc Mechanism section for exactly when a call continues a streak
|
||||
* versus starts a fresh one.
|
||||
*
|
||||
* <p><strong>Read together, off the one {@link QuarantineState} entry {@link #quarantine} itself
|
||||
* wrote</strong> — a single {@code quarantines.get(credentialId)}, never a separate lookup or a
|
||||
* value re-derived from {@code remainingSeconds} (e.g. inverting {@link
|
||||
* #escalatedCooldownNanos}). That inversion is not just extra work to avoid: once a streak has
|
||||
* hit {@code maxCooldownNanos}, every further consecutive exhaustion reports the identical
|
||||
* cooldown, so a derivation that starts from the cooldown value cannot tell the 4th repeat from
|
||||
* the 9th — only the stored {@code repeatCount} can. This is the same rule {@code
|
||||
* CompositePeerLauncher.modelGateState()} documents for its own gate/report pair: the report
|
||||
* reads the exact accessor the behaviour reads, so it can never disagree with what actually
|
||||
* happened (the fleetd #404/#422 lesson). {@code fleet_profiles}/{@code fleet_list}/{@code GET
|
||||
* /profiles} all call this — never {@link #remainingSeconds} plus a second, independent count —
|
||||
* for exactly that reason.
|
||||
*
|
||||
* @return empty when {@code credentialId} is not currently quarantined (including on {@link
|
||||
* #none()}, which quarantines nothing)
|
||||
*/
|
||||
public Optional<Status> status(String credentialId) {
|
||||
QuarantineState state = quarantines.get(credentialId);
|
||||
if (state == null) {
|
||||
return Optional.empty();
|
||||
}
|
||||
long remaining = state.deadlineNanos() - nowNanos.getAsLong();
|
||||
return remaining > 0
|
||||
? Optional.of(new Status(toSecondsRoundedUp(remaining), state.repeatCount()))
|
||||
: Optional.empty();
|
||||
}
|
||||
|
||||
/**
|
||||
* @param remainingSeconds seconds left on the quarantine, identical to {@link
|
||||
* #remainingSeconds(String)}'s answer for the same credential at the
|
||||
* same instant
|
||||
* @param repeatCount 1 for a first occurrence, 2 for the second consecutive one, etc. —
|
||||
* see {@link #status(String)}
|
||||
*/
|
||||
public record Status(long remainingSeconds, int repeatCount) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Every currently-quarantined credential id and its remaining seconds (CB-578 stage B fleet
|
||||
* reporting) — expired entries are never included. Not pruned from the backing map here: it stays
|
||||
@@ -95,8 +257,8 @@ public final class BackendQuarantine {
|
||||
*/
|
||||
public Map<String, Long> activeRemainingSeconds() {
|
||||
Map<String, Long> out = new LinkedHashMap<>();
|
||||
quarantinedUntilNanos.forEach((credentialId, deadline) -> {
|
||||
long remaining = deadline - nowNanos.getAsLong();
|
||||
quarantines.forEach((credentialId, state) -> {
|
||||
long remaining = state.deadlineNanos() - nowNanos.getAsLong();
|
||||
if (remaining > 0) {
|
||||
out.put(credentialId, toSecondsRoundedUp(remaining));
|
||||
}
|
||||
@@ -105,8 +267,14 @@ public final class BackendQuarantine {
|
||||
}
|
||||
|
||||
private long remainingNanos(String credentialId) {
|
||||
Long deadline = quarantinedUntilNanos.get(credentialId);
|
||||
return deadline == null ? 0L : deadline - nowNanos.getAsLong();
|
||||
QuarantineState state = quarantines.get(credentialId);
|
||||
return state == null ? 0L : state.deadlineNanos() - nowNanos.getAsLong();
|
||||
}
|
||||
|
||||
/** {@code cooldownNanos * backoffMultiplier ^ (repeatCount - 1)}, capped at {@code maxCooldownNanos}. */
|
||||
private long escalatedCooldownNanos(int repeatCount) {
|
||||
double raw = cooldownNanos * Math.pow(backoffMultiplier, repeatCount - 1);
|
||||
return raw >= (double) maxCooldownNanos ? maxCooldownNanos : (long) raw;
|
||||
}
|
||||
|
||||
private static long toSecondsRoundedUp(long nanos) {
|
||||
|
||||
@@ -1,66 +1,151 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Backward-compatible placement: an unqualified spawn always resolves to the configured default
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. This ignores caps and
|
||||
* reachability so that a pre-existing config behaves identically after upgrade.
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. Reachability is a narrower
|
||||
* exception (fleetd #315, below): a profile is never checked for reachability up front, only
|
||||
* skipped once it has already failed in <em>this same</em> spawn call's retry loop — see the
|
||||
* unreachable case below.
|
||||
*
|
||||
* <p>Two exceptions walk past the default instead of returning it unconditionally:
|
||||
* <p>Six exceptions walk past the default instead of returning it unconditionally:
|
||||
* <ul>
|
||||
* <li>Quarantine (CB-578 stage B): a quarantined default is a credential that just refused on
|
||||
* a usage limit, not a transient capacity or reachability concern.
|
||||
* <li>Cooling off (fleetd #201 Unit 5): a credential cooling off after repeated backend errors
|
||||
* ({@code BackendOutagePolicy}) — a separate, shorter-lived source from quarantine. When a
|
||||
* profile is both quarantined and cooling off, only the quarantine reason is reported
|
||||
* (exhaustion takes priority), matching {@code CompositePeerLauncher}'s explicit-spawn order.
|
||||
* <li>At cap (fleetd #435): a profile whose live count has reached its {@code maxLoad}
|
||||
* ({@link PlacementPolicyUtil#atCap}) — a documented, unconditional capacity limit (see
|
||||
* {@code FleetConfig.Profile#maxLoad}), so {@code fixed} must gate on it exactly as {@code
|
||||
* weighted}/{@code round-robin} already do via {@link PlacementPolicyUtil#available}. Before
|
||||
* this fix {@code fixed} built its own {@link PlacementCandidate} for the default with {@code
|
||||
* maxLoad} forced to {@code null}, so a capped default was chosen anyway on every unqualified
|
||||
* spawn — the cap was advisory, not enforced, for the one placement policy every config uses
|
||||
* by default. Reported only when quarantine and cooling off are both absent, matching {@code
|
||||
* CompositePeerLauncher}'s explicit-spawn check order (quarantine, then cooling off, then max
|
||||
* load, then model-off).
|
||||
* <li>Model off (fleetd #422): a profile whose {@code model:} the operator has turned off in
|
||||
* {@code models.allow:} — an operator decision, never a backend-reported outage, so it is a
|
||||
* fifth, independent source from quarantine, cooling off, and at-cap (never merged with any
|
||||
* of them), exactly as {@code CompositePeerLauncher.enforceModelEnabled} and {@link
|
||||
* PlacementPolicyUtil#available} treat it. When a profile is model-off <em>and</em> quarantined,
|
||||
* cooling off, or at cap, only the higher-priority reason is reported, matching {@code
|
||||
* CompositePeerLauncher}'s explicit-spawn check order (quarantine, then cooling off, then max
|
||||
* load, then model-off).
|
||||
* <li>Unreachable (fleetd #315): {@code CompositePeerLauncher.spawn} retries a failed candidate
|
||||
* on the next one and rebuilds the {@link PlacementContext} so {@code ctx.unreachable()}
|
||||
* names every profile that already failed with {@code PeerUnreachableException} in this same
|
||||
* call. Without this check {@code select} kept handing back the same dead default forever —
|
||||
* the retry loop's own comment says "so the policy excludes this profile", and this is what
|
||||
* makes that true for {@code fixed} too, matching {@code weighted}/{@code round-robin}
|
||||
* (both filter on {@code ctx.unreachable()} via {@link PlacementPolicyUtil#available}).
|
||||
* <li>Weight 0 (CB-554): {@code fixed} is still automatic selection, so a profile the operator
|
||||
* marked "never auto-select me" ({@code weight <= 0}) must be skipped here exactly as
|
||||
* {@code weighted}/{@code round-robin} skip it — an explicit {@code fleet_spawn} naming
|
||||
* the profile is unaffected, only this automatic fallback walk.
|
||||
* </ul>
|
||||
* A fleet where nothing is ever quarantined or weight-0 never exercises either path, so today's
|
||||
* behaviour is unchanged.
|
||||
* A fleet where nothing is ever quarantined, cooling off, at cap, model-off, unreachable, or
|
||||
* weight-0 never exercises any of these paths, so today's behaviour is unchanged — in particular,
|
||||
* the very first selection of a spawn call always sees an empty {@code unreachable} set, so the
|
||||
* first choice is untouched.
|
||||
*/
|
||||
final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
|
||||
@Override
|
||||
public PlacementCandidate select(PlacementContext ctx) {
|
||||
String d = ctx.defaultProfile();
|
||||
if (d != null && !d.isBlank() && !ctx.quarantined().contains(d) && !weightExcluded(ctx, d)) {
|
||||
if (d != null && !d.isBlank() && !ctx.quarantined().contains(d) && !ctx.coolingOff().contains(d)
|
||||
&& !ctx.modelOff().contains(d) && !ctx.unreachable().contains(d) && !weightExcluded(ctx, d)
|
||||
&& !capExcluded(ctx, d)) {
|
||||
return new PlacementCandidate(d, null, 1.0f, null);
|
||||
}
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (!ctx.quarantined().contains(c.profile()) && !c.excluded()) {
|
||||
if (!ctx.quarantined().contains(c.profile()) && !ctx.coolingOff().contains(c.profile())
|
||||
&& !ctx.modelOff().contains(c.profile()) && !ctx.unreachable().contains(c.profile())
|
||||
&& !c.excluded() && !PlacementPolicyUtil.atCap(ctx, c)) {
|
||||
return new PlacementCandidate(c.profile(), null, c.weight(), c.maxLoad());
|
||||
}
|
||||
}
|
||||
if (d != null && !d.isBlank()) {
|
||||
boolean dQuarantined = ctx.quarantined().contains(d);
|
||||
// Exhaustion quarantine takes priority: reported only when quarantine is absent, so the
|
||||
// message never claims "cooling off" for a profile that is really backend-exhausted.
|
||||
boolean dCoolingOff = !dQuarantined && ctx.coolingOff().contains(d);
|
||||
// fleetd #435: at-cap sits between cooling off and model-off, matching
|
||||
// CompositePeerLauncher's explicit-spawn check order (quarantine, cooling off, max load,
|
||||
// then model-off) — reported only when quarantine/cooling-off are both absent.
|
||||
boolean dAtCap = !dQuarantined && !dCoolingOff && capExcluded(ctx, d);
|
||||
// fleetd #422: model-off is a fifth, independent source (an operator decision) — but
|
||||
// quarantine/cooling-off/at-cap still take priority when more than one applies, matching
|
||||
// CompositePeerLauncher's explicit-spawn check order (quarantine, cooling off, max load,
|
||||
// then model-off).
|
||||
boolean dModelOff = !dQuarantined && !dCoolingOff && !dAtCap && ctx.modelOff().contains(d);
|
||||
boolean dUnreachable = ctx.unreachable().contains(d);
|
||||
boolean dWeightExcluded = weightExcluded(ctx, d);
|
||||
if (dQuarantined && dWeightExcluded) {
|
||||
throw new PlacementException("worker profile '" + d + "' is quarantined (backend "
|
||||
+ "exhausted) and has weight 0 (excluded from automatic selection), and no "
|
||||
+ "available candidate remains");
|
||||
}
|
||||
if (dWeightExcluded) {
|
||||
throw new PlacementException("worker profile '" + d + "' has weight 0 (excluded "
|
||||
+ "from automatic selection) and no available candidate remains");
|
||||
}
|
||||
if (dQuarantined) {
|
||||
throw new PlacementException("worker profile '" + d + "' is quarantined (backend "
|
||||
+ "exhausted) and no un-quarantined candidate is available");
|
||||
if (dQuarantined || dCoolingOff || dAtCap || dModelOff || dUnreachable || dWeightExcluded) {
|
||||
List<String> reasons = new ArrayList<>();
|
||||
if (dQuarantined) {
|
||||
reasons.add("is quarantined (backend exhausted)");
|
||||
}
|
||||
if (dCoolingOff) {
|
||||
reasons.add("is cooling off after repeated backend errors");
|
||||
}
|
||||
if (dAtCap) {
|
||||
PlacementCandidate c = candidateFor(ctx, d);
|
||||
int live = ctx.liveCount().apply(d);
|
||||
reasons.add("is at maxLoad (" + live + " live >= " + c.maxLoad() + " cap)");
|
||||
}
|
||||
if (dModelOff) {
|
||||
reasons.add("names a model the operator has turned off in models.allow");
|
||||
}
|
||||
if (dUnreachable) {
|
||||
reasons.add("is unreachable");
|
||||
}
|
||||
if (dWeightExcluded) {
|
||||
reasons.add("has weight 0 (excluded from automatic selection)");
|
||||
}
|
||||
throw new PlacementException("worker profile '" + d + "' "
|
||||
+ String.join(" and ", reasons) + ", and no available candidate remains");
|
||||
}
|
||||
}
|
||||
if (!ctx.candidates().isEmpty()) {
|
||||
throw new PlacementException(
|
||||
"all worker profiles are excluded from automatic selection (quarantined or weight-0)");
|
||||
throw new PlacementException("all worker profiles are excluded from automatic "
|
||||
+ "selection (quarantined, cooling off, at cap, model-off, unreachable, or weight-0)");
|
||||
}
|
||||
throw new PlacementException("no worker profiles configured");
|
||||
}
|
||||
|
||||
/** Whether {@code profile} carries {@code weight <= 0} (CB-554) among {@code ctx}'s candidates. */
|
||||
private static boolean weightExcluded(PlacementContext ctx, String profile) {
|
||||
PlacementCandidate c = candidateFor(ctx, profile);
|
||||
return c != null && c.excluded();
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code profile} has reached its {@code maxLoad} cap (fleetd #435), using the shared
|
||||
* {@link PlacementPolicyUtil#atCap} definition — the same one {@code weighted}/{@code
|
||||
* round-robin} already consult via {@link PlacementPolicyUtil#available}. Looked up by name,
|
||||
* the same way {@link #weightExcluded} is: the default fast path above builds its own {@link
|
||||
* PlacementCandidate} with {@code maxLoad} forced to {@code null} (it carries no cap of its
|
||||
* own), so the candidate actually configured for {@code profile} has to be found in {@code
|
||||
* ctx.candidates()} first.
|
||||
*/
|
||||
private static boolean capExcluded(PlacementContext ctx, String profile) {
|
||||
PlacementCandidate c = candidateFor(ctx, profile);
|
||||
return c != null && PlacementPolicyUtil.atCap(ctx, c);
|
||||
}
|
||||
|
||||
/** The configured candidate named {@code profile} in {@code ctx}, or {@code null} if none. */
|
||||
private static PlacementCandidate candidateFor(PlacementContext ctx, String profile) {
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (c.profile().equals(profile)) {
|
||||
return c.excluded();
|
||||
return c;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -14,10 +14,37 @@ import java.util.function.Function;
|
||||
* @param quarantined profiles whose credential is currently quarantined (CB-578 stage B) — a
|
||||
* {@code BACKEND_EXHAUSTED} classification put it, or a profile it shares a
|
||||
* credential with, on cooldown. Filtered the same way as {@code unreachable}.
|
||||
* @param coolingOff profiles whose credential is currently cooling off after repeated backend
|
||||
* errors (fleetd #201 Unit 5 — {@code BackendOutagePolicy}), a SEPARATE,
|
||||
* shorter-lived source from {@code quarantined}: a credential outage cools off
|
||||
* even when no member was ever exhausted. Deliberately its own set rather than
|
||||
* merged into {@code quarantined} — {@link PlacementPolicyUtil} needs to tell
|
||||
* the two apart so its refusal message says "cooling off", not "exhausted",
|
||||
* when only this one is active. A profile can be in both sets at once; when it
|
||||
* is, exhaustion quarantine is reported (it takes priority).
|
||||
* @param modelOff profiles whose {@code model:} is currently turned off in {@code
|
||||
* models.allow:} (fleetd #422) — an operator decision, not a backend-reported
|
||||
* outage, so a SEPARATE, independent source from both {@code quarantined} and
|
||||
* {@code coolingOff}. A profile can be in this set together with either (or
|
||||
* both) of the others; {@link PlacementPolicyUtil} counts it into its own
|
||||
* bucket rather than merging it into theirs, the same reason
|
||||
* {@code coolingOff} is kept apart from {@code quarantined}.
|
||||
*/
|
||||
public record PlacementContext(String defaultProfile,
|
||||
List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount,
|
||||
Set<String> unreachable,
|
||||
Set<String> quarantined) {
|
||||
Set<String> quarantined,
|
||||
Set<String> coolingOff,
|
||||
Set<String> modelOff) {
|
||||
|
||||
/**
|
||||
* Back-compat form before the fleetd #422 model on/off gate was added — no candidate's model
|
||||
* is off. Keeps pre-#422 call sites (tests included) compiling and behaving identically.
|
||||
*/
|
||||
public PlacementContext(String defaultProfile, List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount, Set<String> unreachable,
|
||||
Set<String> quarantined, Set<String> coolingOff) {
|
||||
this(defaultProfile, candidates, liveCount, unreachable, quarantined, coolingOff, Set.of());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
|
||||
/**
|
||||
* An already-completed placement choice — the outcome of one {@link PeerLauncher#place} call,
|
||||
* carried forward so a later {@link PeerLauncher#spawn(SpawnRequest, PlacementDecision)} can honor
|
||||
* it directly instead of re-resolving the profile a second time (fleetd #425 rework, round 2).
|
||||
*
|
||||
* <p>The problem this exists to close: a caller that must know the profile <em>before</em> it can
|
||||
* spawn — {@code SessionManager.acquireWithWorktree} provisions a worktree's {@code repoRoot} and
|
||||
* parity overlay for a specific profile before the peer process exists — used to resolve that name
|
||||
* with {@code PeerLauncher.routedProfileFor(role)} and then hand the SAME string back to {@link
|
||||
* PeerLauncher#spawn(SpawnRequest)} as an EXPLICIT profile. That re-resolution is not free: naming
|
||||
* a profile explicitly makes {@code CompositePeerLauncher.spawn} take its THROWING branch
|
||||
* ({@code enforceNotQuarantined}/{@code enforceNotCoolingOff}/{@code enforceMaxLoad}/{@code
|
||||
* enforceModelEnabled}), while an unqualified spawn's ROUTING branch never runs those checks at
|
||||
* all — it instead FALLS THROUGH to the next candidate on exactly the same conditions the throwing
|
||||
* branch refuses on. That disagreement is deliberate: an operator who names a profile should get a
|
||||
* refusal, not a silent substitution. The bug is turning the fall-through into a refusal by
|
||||
* accident — resolving a name through the routing side and then re-entering the refusing side with
|
||||
* it, for a decision the routing side had already approved by walking past everything else.
|
||||
* Before fleetd #435, this accident was also reachable through {@code maxLoad} specifically: the
|
||||
* default {@code fixed} placement policy did not evaluate {@code maxLoad} at all for automatic
|
||||
* selection, so a profile placement itself just approved could still die at {@code enforceMaxLoad}
|
||||
* one call later, purely because the caller's route to the spawn passed through an explicit
|
||||
* profile name instead of the routing branch — a failure a worktree-less unqualified spawn would
|
||||
* never hit. fleetd #435 closed that specific gap ({@code fixed} now evaluates {@code maxLoad}
|
||||
* exactly like every other placement policy), so a {@link PlacementDecision} can no longer be
|
||||
* at-cap in the first place — but the refusal-vs-fall-through disagreement above was never about
|
||||
* {@code maxLoad}, and resolving a name and re-entering the refusing branch with it is still wrong
|
||||
* for every OTHER condition placement filters on.
|
||||
*
|
||||
* <p>{@link PeerLauncher#spawn(SpawnRequest, PlacementDecision)} closes that by spawning through
|
||||
* the identical code path the routing branch itself uses, keyed off the SAME decision {@link
|
||||
* PeerLauncher#place} returned — no re-checking of any condition placement already evaluated. A
|
||||
* resolve-then-spawn caller and a blank-profile {@link PeerLauncher#spawn(SpawnRequest)} caller can
|
||||
* then never disagree about which conditions apply to the same placement state, and neither one
|
||||
* re-opens the window between the placement decision and the spawn in which the underlying state
|
||||
* could otherwise move.
|
||||
*
|
||||
* @param profile the profile this decision resolved to (may be {@code null} only when no profile is
|
||||
* configured at all — the same corner case {@link PeerLauncher#defaultProfile()}
|
||||
* already tolerates)
|
||||
*/
|
||||
public record PlacementDecision(String profile) {
|
||||
}
|
||||
@@ -11,26 +11,40 @@ final class PlacementPolicyUtil {
|
||||
private PlacementPolicyUtil() {
|
||||
}
|
||||
|
||||
/**
|
||||
* True when {@code c} has reached its {@code maxLoad} cap: {@code liveCount(c.profile()) >=
|
||||
* c.maxLoad()}. A {@code null} maxLoad means unlimited, so it is never at cap.
|
||||
*
|
||||
* <p>Extracted as the single shared definition of "at cap" (fleetd #435): before this fix it
|
||||
* was computed inline in both {@link #available} and {@link #emptyException}, and {@code
|
||||
* FixedPlacementPolicy} — not a caller of either — quietly kept its own {@code select} free of
|
||||
* any cap check at all, so a capped default profile was chosen anyway under the default
|
||||
* placement policy. Every automatic policy must call this, not re-derive it.
|
||||
*/
|
||||
static boolean atCap(PlacementContext ctx, PlacementCandidate c) {
|
||||
Integer cap = c.maxLoad();
|
||||
return cap != null && ctx.liveCount().apply(c.profile()) >= cap;
|
||||
}
|
||||
|
||||
/**
|
||||
* Candidates that are not weight-excluded (CB-554: explicit {@code weight <= 0}, checked
|
||||
* first because it is a static config choice rather than transient state), not
|
||||
* known-unreachable, not quarantined (CB-578 stage B), and have not reached their maxLoad.
|
||||
* A {@code null} maxLoad means unlimited.
|
||||
* known-unreachable, not quarantined (CB-578 stage B), not cooling off after repeated backend
|
||||
* errors (fleetd #201 Unit 5 — a separate, shorter-lived source from quarantine), not naming a
|
||||
* model the operator has turned off (fleetd #422 — a third, independent source: an operator
|
||||
* decision, never a backend-reported outage), and have not reached their maxLoad (see {@link
|
||||
* #atCap}). A {@code null} maxLoad means unlimited.
|
||||
*/
|
||||
static List<PlacementCandidate> available(PlacementContext ctx) {
|
||||
List<PlacementCandidate> out = new ArrayList<>();
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (c.excluded() || ctx.unreachable().contains(c.profile())
|
||||
|| ctx.quarantined().contains(c.profile())) {
|
||||
|| ctx.quarantined().contains(c.profile())
|
||||
|| ctx.coolingOff().contains(c.profile())
|
||||
|| ctx.modelOff().contains(c.profile())
|
||||
|| atCap(ctx, c)) {
|
||||
continue;
|
||||
}
|
||||
Integer cap = c.maxLoad();
|
||||
if (cap != null) {
|
||||
int live = ctx.liveCount().apply(c.profile());
|
||||
if (live >= cap) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
out.add(c);
|
||||
}
|
||||
return out;
|
||||
@@ -38,24 +52,33 @@ final class PlacementPolicyUtil {
|
||||
|
||||
/**
|
||||
* Build a clear exception describing why every candidate was dropped: all weight-0, all
|
||||
* quarantined, all at capacity, all unreachable, or a mix. Each candidate is counted into
|
||||
* exactly one bucket (weight-excluded takes priority) so a candidate excluded for more than
|
||||
* one reason is never double-counted.
|
||||
* quarantined, all cooling off, all model-off, all at capacity, all unreachable, or a mix. Each
|
||||
* candidate is counted into exactly one bucket (weight-excluded first, then quarantined, then
|
||||
* cooling off, then model-off) so a candidate excluded for more than one reason is never
|
||||
* double-counted — a candidate that is both quarantined (CB-578 stage B, backend exhausted) and
|
||||
* cooling off (fleetd #201 Unit 5, repeated backend errors) counts only as quarantined, matching
|
||||
* {@code CompositePeerLauncher}'s explicit-spawn ordering: exhaustion quarantine takes priority
|
||||
* when more than one applies.
|
||||
*/
|
||||
static PlacementException emptyException(PlacementContext ctx) {
|
||||
int weightExcluded = 0;
|
||||
int atCap = 0;
|
||||
int unreachable = 0;
|
||||
int quarantined = 0;
|
||||
int coolingOff = 0;
|
||||
int modelOff = 0;
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
Integer cap = c.maxLoad();
|
||||
if (c.excluded()) {
|
||||
weightExcluded++;
|
||||
} else if (ctx.quarantined().contains(c.profile())) {
|
||||
quarantined++;
|
||||
} else if (ctx.coolingOff().contains(c.profile())) {
|
||||
coolingOff++;
|
||||
} else if (ctx.modelOff().contains(c.profile())) {
|
||||
modelOff++;
|
||||
} else if (ctx.unreachable().contains(c.profile())) {
|
||||
unreachable++;
|
||||
} else if (cap != null && ctx.liveCount().apply(c.profile()) >= cap) {
|
||||
} else if (atCap(ctx, c)) {
|
||||
atCap++;
|
||||
}
|
||||
}
|
||||
@@ -71,6 +94,14 @@ final class PlacementPolicyUtil {
|
||||
if (quarantined == total) {
|
||||
return new PlacementException("all worker profiles are quarantined (backend exhausted)");
|
||||
}
|
||||
if (coolingOff == total) {
|
||||
return new PlacementException(
|
||||
"all worker profiles are cooling off after repeated backend errors");
|
||||
}
|
||||
if (modelOff == total) {
|
||||
return new PlacementException(
|
||||
"all worker profiles name a model the operator has turned off");
|
||||
}
|
||||
if (atCap == total) {
|
||||
return new PlacementException("all worker profiles are at maxLoad");
|
||||
}
|
||||
@@ -79,7 +110,10 @@ final class PlacementPolicyUtil {
|
||||
}
|
||||
return new PlacementException("no worker profile available: " + atCap + " at maxLoad, "
|
||||
+ unreachable + " unreachable, " + quarantined + " quarantined, "
|
||||
+ coolingOff + " cooling off, "
|
||||
+ modelOff + " model-off, "
|
||||
+ weightExcluded + " weight-0, "
|
||||
+ (total - atCap - unreachable - quarantined - weightExcluded) + " remaining");
|
||||
+ (total - atCap - unreachable - quarantined - coolingOff - modelOff - weightExcluded)
|
||||
+ " remaining");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.Locale;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
|
||||
/**
|
||||
* Holds macOS idle sleep off by keeping a {@code caffeinate -i} child process alive for the life
|
||||
* of the returned {@link SleepAssertion}.
|
||||
*
|
||||
* <p>{@code -i} asserts only against <em>idle</em> sleep — it does not stop the lid closing or an
|
||||
* operator-requested sleep from taking effect. That is deliberate: this class exists to stop an
|
||||
* unattended host from sleeping out from under a member's long turn, never to override the
|
||||
* operator. {@code -s}/{@code -d} (which also block system/display sleep on demand) are
|
||||
* intentionally not used here.
|
||||
*
|
||||
* <p>{@link #acquire()} never throws. It returns {@code null} — a no-op — off macOS, and again if
|
||||
* starting the {@code caffeinate} child fails for any reason (binary missing, process table full,
|
||||
* …); either case is logged once at INFO, not on every occurrence, so a daemon that runs for
|
||||
* weeks with the tool unavailable does not fill its log.
|
||||
*/
|
||||
public final class CaffeinateSleepAssertionMechanism implements SleepAssertionMechanism {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(CaffeinateSleepAssertionMechanism.class);
|
||||
|
||||
private final AtomicBoolean loggedOnce = new AtomicBoolean(false);
|
||||
|
||||
/** {@code true} when running on macOS, the only platform {@code caffeinate} ships on. */
|
||||
public static boolean isSupportedPlatform() {
|
||||
return isSupportedPlatform(System.getProperty("os.name"));
|
||||
}
|
||||
|
||||
/** Package-visible so a test can drive the platform check without touching a real property. */
|
||||
static boolean isSupportedPlatform(String osName) {
|
||||
return osName != null && osName.toLowerCase(Locale.ROOT).contains("mac");
|
||||
}
|
||||
|
||||
@Override
|
||||
public SleepAssertion acquire() {
|
||||
if (!isSupportedPlatform()) {
|
||||
logOnce("not running on macOS (os.name={}); the idle-sleep guard is a no-op on this platform",
|
||||
System.getProperty("os.name"));
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
Process process = new ProcessBuilder("caffeinate", "-i")
|
||||
.redirectOutput(ProcessBuilder.Redirect.DISCARD)
|
||||
.redirectError(ProcessBuilder.Redirect.DISCARD)
|
||||
.start();
|
||||
return new CaffeinateAssertion(process);
|
||||
} catch (IOException | RuntimeException e) {
|
||||
logOnce("could not start 'caffeinate -i' ({}); the host may idle-sleep while members are live",
|
||||
e.toString());
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
private void logOnce(String format, Object arg) {
|
||||
if (loggedOnce.compareAndSet(false, true)) {
|
||||
log.info("idle-sleep guard: " + format, arg);
|
||||
}
|
||||
}
|
||||
|
||||
/** Wraps the live {@code caffeinate} child; {@link #close} force-destroys it, idempotently. */
|
||||
private static final class CaffeinateAssertion implements SleepAssertion {
|
||||
|
||||
private final Process process;
|
||||
|
||||
CaffeinateAssertion(Process process) {
|
||||
this.process = process;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
if (!process.isAlive()) {
|
||||
return;
|
||||
}
|
||||
process.destroy();
|
||||
try {
|
||||
if (!process.waitFor(2, TimeUnit.SECONDS)) {
|
||||
process.destroyForcibly();
|
||||
}
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
process.destroyForcibly();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,105 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.function.IntSupplier;
|
||||
|
||||
/**
|
||||
* Holds an OS-level assertion against idle sleep for exactly as long as at least one fleet
|
||||
* member is live.
|
||||
*
|
||||
* <p><strong>Why this exists:</strong> a fleetd host was measured idle-sleeping after as little
|
||||
* as one minute of inactivity (its {@code pmset -g custom} reports {@code sleep 1} on battery).
|
||||
* Overnight the daemon's AMQP link to the broker dropped 13 times, and cross-checking every drop
|
||||
* minute against {@code pmset -g log} found a sleep or wake event in the same minute or the one
|
||||
* before, every time. The AMQP churn is only the visible symptom — the real problem is that a
|
||||
* member mid-turn freezes with the host, and a long turn with nobody typing is exactly the case
|
||||
* that goes idle.
|
||||
*
|
||||
* <p><strong>How it tracks "live":</strong> this is driven by {@code SessionManager}'s existing
|
||||
* {@code onAcquire}/{@code onRelease} lifecycle hooks (added for CB-520/CB-516, previously wired
|
||||
* to nothing but the reply inbox) rather than a second member count kept in parallel. Wire it as:
|
||||
* <pre>{@code
|
||||
* IdleSleepGuard guard = new IdleSleepGuard(mechanism, sessions::size);
|
||||
* sessions.onAcquire(_ -> guard.recheck());
|
||||
* sessions.onRelease(_ -> guard.recheck());
|
||||
* }</pre>
|
||||
* Every acquire/release event re-reads {@code SessionManager#size()} — the same registry {@code
|
||||
* fleet_list}'s live/capacity numbers are themselves computed from — and only an actual 0→1 or
|
||||
* 1→0 crossing touches the OS. A listener exception is already caught and logged by {@code
|
||||
* SessionManager} itself (it must never let a listener failure block the acquire/release it is
|
||||
* reacting to), so {@link #recheck()} does not need its own top-level try/catch to honor that.
|
||||
*
|
||||
* <p><strong>Failure posture:</strong> every method here is safe to call whether or not {@link
|
||||
* SleepAssertionMechanism#acquire()} actually works. A mechanism that returns {@code null} (wrong
|
||||
* platform, missing tool, spawn failure) simply means this guard never holds anything — it never
|
||||
* throws and never blocks a spawn, a release, or shutdown.
|
||||
*/
|
||||
public final class IdleSleepGuard implements AutoCloseable {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(IdleSleepGuard.class);
|
||||
|
||||
private final SleepAssertionMechanism mechanism;
|
||||
private final IntSupplier liveCount;
|
||||
private final Object lock = new Object();
|
||||
private SleepAssertion held;
|
||||
|
||||
public IdleSleepGuard(SleepAssertionMechanism mechanism, IntSupplier liveCount) {
|
||||
this.mechanism = mechanism;
|
||||
this.liveCount = liveCount;
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-read the live count and acquire or release the held assertion to match: nothing held and
|
||||
* at least one member live ⇒ acquire; something held and no member live ⇒ release. A steady
|
||||
* count (still zero, still positive) is a no-op either way, so a single spawn or release only
|
||||
* ever touches the OS on the crossing, not on every call.
|
||||
*/
|
||||
public void recheck() {
|
||||
synchronized (lock) {
|
||||
int live = liveCount.getAsInt();
|
||||
if (live > 0 && held == null) {
|
||||
held = mechanism.acquire();
|
||||
if (held != null) {
|
||||
log.debug("idle-sleep guard armed: {} live member(s)", live);
|
||||
}
|
||||
} else if (live == 0 && held != null) {
|
||||
releaseHeldLocked();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** {@code true} while an assertion is actually held. Exposed for tests. */
|
||||
boolean isHeld() {
|
||||
synchronized (lock) {
|
||||
return held != null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Release whatever is held, if anything. Idempotent and safe to call at any time, including
|
||||
* repeatedly — a daemon shutdown hook calls this unconditionally so no assertion (and no
|
||||
* {@code caffeinate} child) survives the process, even if the drain that would otherwise have
|
||||
* driven the live count to zero was itself interrupted or threw.
|
||||
*/
|
||||
@Override
|
||||
public void close() {
|
||||
synchronized (lock) {
|
||||
if (held != null) {
|
||||
releaseHeldLocked();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Caller must hold {@link #lock}. */
|
||||
private void releaseHeldLocked() {
|
||||
try {
|
||||
held.close();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("idle-sleep guard: failed to release its assertion cleanly: {}", e.toString());
|
||||
} finally {
|
||||
held = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
/**
|
||||
* A held OS-level assertion against idle sleep. {@link #close} must be idempotent — safe to call
|
||||
* more than once — and must never throw, matching {@link IdleSleepGuard}'s "never break the
|
||||
* fleet" contract.
|
||||
*/
|
||||
public interface SleepAssertion extends AutoCloseable {
|
||||
@Override
|
||||
void close();
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
/**
|
||||
* The OS mechanism {@link IdleSleepGuard} uses to hold and release an idle-sleep assertion. This
|
||||
* is the seam a test exercises instead of the real effect (a live {@code caffeinate} child) — see
|
||||
* {@code IdleSleepGuardTest}.
|
||||
*
|
||||
* <p>Implementations must never throw. Every failure — wrong platform, missing tool, a spawn
|
||||
* error — must show up as {@link #acquire()} returning {@code null}, so a caller can treat "no
|
||||
* assertion held" and "the mechanism could not be used" identically and the fleet keeps running
|
||||
* either way.
|
||||
*/
|
||||
public interface SleepAssertionMechanism {
|
||||
|
||||
/**
|
||||
* Acquire a fresh assertion against idle sleep, or {@code null} when this mechanism is not
|
||||
* usable right now (wrong platform, the tool is missing, the child process could not start).
|
||||
* Never throws.
|
||||
*/
|
||||
SleepAssertion acquire();
|
||||
}
|
||||
@@ -9,14 +9,17 @@ import dev.ltms.fleet.auth.Principal;
|
||||
import dev.ltms.fleet.guard.GuardException;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.inject.MemberPresence;
|
||||
import dev.ltms.fleet.member.MemberCredentialPolicyView;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.session.ShuttingDownException;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.WorktreeRequest;
|
||||
@@ -32,6 +35,7 @@ import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
@@ -45,6 +49,22 @@ import java.util.stream.Collectors;
|
||||
*/
|
||||
public final class FleetApp {
|
||||
|
||||
/** The authorization action the matching route handler hands to {@link #allow}. */
|
||||
static Authz.Action routeAction(String route) {
|
||||
return switch (route) {
|
||||
case "GET /metrics" -> Authz.Action.METRICS;
|
||||
case "POST /members" -> Authz.Action.SPAWN;
|
||||
case "DELETE /members/{paneId}" -> Authz.Action.STOP;
|
||||
case "POST /sessions/{id}/message" -> Authz.Action.SEND;
|
||||
case "POST /sessions/{id}/reply" -> Authz.Action.REPLY;
|
||||
case "GET /sessions/{id}/replies" -> Authz.Action.DRAIN;
|
||||
case "POST /sessions/{id}/ask" -> Authz.Action.ASK;
|
||||
case "GET /sessions", "GET /agents", "GET /members", "GET /profiles",
|
||||
"GET /member-credentials", "GET /sessions/{id}/status", "GET /tasks/{ticket}" -> Authz.Action.READ;
|
||||
default -> throw new IllegalArgumentException("route has no authorization gate: " + route);
|
||||
};
|
||||
}
|
||||
|
||||
/** Default blocking window for a message; kept under typical HTTP idle timeouts. */
|
||||
private static final long DEFAULT_MESSAGE_TIMEOUT_MS = 25_000;
|
||||
private static final long MAX_MESSAGE_TIMEOUT_MS = 120_000;
|
||||
@@ -64,6 +84,16 @@ public final class FleetApp {
|
||||
private final HttpServlet mcpServlet; // MCP Streamable-HTTP endpoint, mounted at /mcp (nullable)
|
||||
private final CallerResolver auth; // CB-501: null → authz not enforced (legacy behaviour)
|
||||
private final Metrics metrics; // CB-502: null → /metrics not exposed
|
||||
// fleetd #111: re-read per request, same hot-reload shape as every other live config read —
|
||||
// absent() (the honest "no policy configured" view) for every constructor that does not wire
|
||||
// a real one, so existing legacy call sites keep building without knowing this field exists.
|
||||
private final Supplier<MemberCredentialPolicyView> memberCredentials;
|
||||
// fleetd #297: the SAME shared sources FleetMcp.profiles/fleet_profiles reads (BackendQuarantine
|
||||
// and BackendOutagePolicy are each one instance for the whole daemon — see Fleetd wiring) so
|
||||
// GET /profiles cannot drift from fleet_profiles about which profile is quarantined/cooling off.
|
||||
// .none() (the honest "feature not wired" view) for every constructor that does not pass one.
|
||||
private final FleetMcp.QuarantineSource quarantine;
|
||||
private final FleetMcp.OutageSource outage;
|
||||
private final ObjectMapper mapper = new ObjectMapper();
|
||||
|
||||
/**
|
||||
@@ -111,6 +141,36 @@ public final class FleetApp {
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable) {
|
||||
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics,
|
||||
deliverable, MemberCredentialPolicyView::absent);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param memberCredentials live {@code memberCredentials:} policy view (fleetd #111), re-read
|
||||
* per request for {@code GET /member-credentials}; production wiring
|
||||
* passes the same hot-reload shape as every other live config read
|
||||
*/
|
||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials) {
|
||||
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics,
|
||||
deliverable, memberCredentials, FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* @param quarantine the SAME {@link FleetMcp.QuarantineSource} instance passed to {@code
|
||||
* FleetMcp} (fleetd #297), so {@code GET /profiles} reports the identical
|
||||
* exhaustion-quarantine facts as {@code fleet_profiles} rather than a second,
|
||||
* independently-computed copy
|
||||
* @param outage the SAME {@link FleetMcp.OutageSource} instance passed to {@code FleetMcp} —
|
||||
* see {@code quarantine}; a SEPARATE check from it, never merged in
|
||||
*/
|
||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials,
|
||||
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage) {
|
||||
this.herdr = herdr;
|
||||
this.memberHerdr = memberHerdr != null ? memberHerdr : herdr;
|
||||
this.workers = workers;
|
||||
@@ -120,6 +180,9 @@ public final class FleetApp {
|
||||
this.mcpServlet = mcpServlet;
|
||||
this.auth = auth;
|
||||
this.metrics = metrics;
|
||||
this.memberCredentials = memberCredentials != null ? memberCredentials : MemberCredentialPolicyView::absent;
|
||||
this.quarantine = quarantine != null ? quarantine : FleetMcp.QuarantineSource.none();
|
||||
this.outage = outage != null ? outage : FleetMcp.OutageSource.none();
|
||||
}
|
||||
|
||||
/** Wire routes onto a fresh, unstarted Javalin instance. Caller starts it. */
|
||||
@@ -148,6 +211,7 @@ public final class FleetApp {
|
||||
app.get("/agents", this::agents);
|
||||
app.get("/members", this::listMembers); // CB-304: registry roster + live herdr status
|
||||
app.get("/profiles", this::profiles); // configured backend profiles
|
||||
app.get("/member-credentials", this::memberCredentials); // fleetd #111: policy names + counts, never a value
|
||||
app.post("/members", this::spawnMember); // optional ?role=&profile= or {"role":…,"profile":…}
|
||||
app.delete("/members/{paneId}", this::stopMember);
|
||||
app.post("/sessions/{id}/message", this::sendMessage); // fleet_send (primary; blocking, wait:false, or answer via turnId)
|
||||
@@ -201,7 +265,7 @@ public final class FleetApp {
|
||||
|
||||
/** Prometheus scrape endpoint (CB-502). */
|
||||
private void metrics(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.METRICS, null)) {
|
||||
if (!allow(ctx, routeAction("GET /metrics"), null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).contentType("text/plain; version=0.0.4; charset=utf-8").result(metrics.render());
|
||||
@@ -273,7 +337,7 @@ public final class FleetApp {
|
||||
* member workspace (they live on the member daemon only).
|
||||
*/
|
||||
private void sessions(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /sessions"), null)) {
|
||||
return;
|
||||
}
|
||||
List<Map<String, Object>> out = new ArrayList<>();
|
||||
@@ -298,52 +362,99 @@ public final class FleetApp {
|
||||
|
||||
/** Discovery: every agent herdr tracks, keyed by its Claude session UUID. */
|
||||
private void agents(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /agents"), null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).json(Map.of("agents",
|
||||
workers.list().stream().map(Agent.class::cast).map(FleetApp::view).toList()));
|
||||
try {
|
||||
ctx.status(200).json(Map.of("agents",
|
||||
workers.list().stream().map(Agent.class::cast).map(FleetApp::view).toList()));
|
||||
} catch (HerdrException e) {
|
||||
// fleetd #297: workers.list() reaches herdr — a transport failure must land in the same
|
||||
// {error, detail} envelope every other failure path here uses, not escape as a bare
|
||||
// exception and leave Javalin's default handling to respond outside the JSON contract.
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** CB-304: bridge-owned roster merged with live herdr status by paneId. */
|
||||
private void listMembers(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /members"), null)) {
|
||||
return;
|
||||
}
|
||||
// CB-519: the registry key is a host-unique id, not the pane coordinate — join on terminal.
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.map(Agent.class::cast)
|
||||
.filter(a -> a.terminalId() != null)
|
||||
.collect(Collectors.toMap(Agent::terminalId, Function.identity(), (_, b) -> b));
|
||||
// fleetd #209: this REST roster reports agentSessionId via SessionManager.rosterView, so it
|
||||
// uses the resolving roster read (caller-driven, not a timer) rather than the plain one.
|
||||
List<Map<String, Object>> out = sessions.rosterResolved().stream()
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.terminalId())))
|
||||
.toList();
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
// fleetd #199: the endpoint became /members in the CB-634 rename but the body key stayed
|
||||
// "workers", so a caller that read "members" saw an empty fleet and reported no members at
|
||||
// all. "members" is the canonical key; "workers" stays as a deprecated alias so an existing
|
||||
// REST consumer keeps working — the out-of-band path a lead falls back to when its MCP mount
|
||||
// drops reads this endpoint. Drop the alias once nothing reads it.
|
||||
body.put("members", out);
|
||||
body.put("workers", out);
|
||||
// CB-586: operator visibility for the refs/wip snapshot store without shelling into the
|
||||
// repo — how many snapshot refs exist and roughly what they cost. Present only once a
|
||||
// worktree session has established the repo, so a never-snapshotted fleet reports nothing.
|
||||
sessions.wipRefs().ifPresent(st -> body.put("wipRefs",
|
||||
Map.of("count", st.count(), "costBytes", st.costBytes())));
|
||||
ctx.status(200).json(body);
|
||||
try {
|
||||
// CB-519: the registry key is a host-unique id, not the pane coordinate — join on terminal.
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.map(Agent.class::cast)
|
||||
.filter(a -> a.terminalId() != null)
|
||||
.collect(Collectors.toMap(Agent::terminalId, Function.identity(), (_, b) -> b));
|
||||
// fleetd #209: this REST roster reports agentSessionId via SessionManager.rosterView, so it
|
||||
// uses the resolving roster read (caller-driven, not a timer) rather than the plain one.
|
||||
List<Map<String, Object>> out = sessions.rosterResolved().stream()
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.terminalId())))
|
||||
.toList();
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
// fleetd #199: the endpoint became /members in the CB-634 rename but the body key stayed
|
||||
// "workers", so a caller that read "members" saw an empty fleet and reported no members at
|
||||
// all. "members" is the canonical key; "workers" stays as a deprecated alias so an existing
|
||||
// REST consumer keeps working — the out-of-band path a lead falls back to when its MCP mount
|
||||
// drops reads this endpoint. Drop the alias once nothing reads it.
|
||||
body.put("members", out);
|
||||
body.put("workers", out);
|
||||
// CB-586: operator visibility for the refs/wip snapshot store without shelling into the
|
||||
// repo — how many snapshot refs exist and roughly what they cost. Present only once a
|
||||
// worktree session has established the repo, so a never-snapshotted fleet reports nothing.
|
||||
sessions.wipRefs().ifPresent(st -> body.put("wipRefs",
|
||||
Map.of("count", st.count(), "costBytes", st.costBytes())));
|
||||
ctx.status(200).json(body);
|
||||
} catch (HerdrException e) {
|
||||
// fleetd #297: same reasoning as agents() above — this is the out-of-band roster a lead
|
||||
// falls back to when its MCP mount drops, so it must stay inside the JSON error contract
|
||||
// exactly when herdr is briefly unreachable, not escape as a bare exception.
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** The configured worker profiles and which one a no-argument spawn uses. */
|
||||
/**
|
||||
* The configured worker profiles, which one a no-argument spawn uses, and (fleetd #297) the two
|
||||
* outage states {@code fleet_profiles} already reports: {@code quarantined} (CB-578 stage B —
|
||||
* the backend reported it out of capacity) and {@code coolingOff} (fleetd #201 Unit 5 — the
|
||||
* credential threw repeated non-exhaustion backend errors). Both are read from the SAME shared
|
||||
* {@link FleetMcp.QuarantineSource}/{@link FleetMcp.OutageSource} instances {@code FleetMcp}
|
||||
* reads, never recomputed, so the two doors cannot disagree about which profile is down and why.
|
||||
* Independent checks, so a profile can appear in both maps at once; each map is present only
|
||||
* when at least one profile is in that state.
|
||||
*/
|
||||
private void profiles(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /profiles"), null)) {
|
||||
return;
|
||||
}
|
||||
// fleetd #297: ONE body builder, shared with fleet_profiles. Handing both doors the same
|
||||
// QuarantineSource/OutageSource instances stops them reading different facts; rendering
|
||||
// through the same method stops them reporting those facts differently. Both are needed.
|
||||
ctx.status(200).json(FleetMcp.profilesView(workers, quarantine, outage));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #111 (CB-608): the live {@code memberCredentials:} policy as names and counts —
|
||||
* NEVER a value. The daemon does not hold a credential's value in the first place (only the
|
||||
* name it is configured under), so there is nothing to redact here beyond what {@link
|
||||
* MemberCredentialPolicyView} already omits by construction. This is the source
|
||||
* {@code scripts/probe-member-credentials.sh} reads instead of carrying its own hardcoded
|
||||
* name list, which is exactly what let the list drift silently behind the real policy.
|
||||
*/
|
||||
private void memberCredentials(Context ctx) {
|
||||
if (!allow(ctx, routeAction("GET /member-credentials"), null)) {
|
||||
return;
|
||||
}
|
||||
MemberCredentialPolicyView view = memberCredentials.get();
|
||||
ctx.status(200).json(Map.of(
|
||||
"profiles", workers.profiles(),
|
||||
"default", workers.defaultProfile() == null ? "" : workers.defaultProfile()));
|
||||
"present", view.present(),
|
||||
"policy", view.policy() == null ? "" : view.policy(),
|
||||
"known", view.known(),
|
||||
"allowed", view.allowed(),
|
||||
"knownCount", view.knownCount(),
|
||||
"allowedCount", view.allowedCount(),
|
||||
"blockedCount", view.blockedCount()));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -352,7 +463,7 @@ public final class FleetApp {
|
||||
* the subscription boundary, 400 for an unknown profile.
|
||||
*/
|
||||
private void spawnMember(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.SPAWN, null)) {
|
||||
if (!allow(ctx, routeAction("POST /members"), null)) {
|
||||
return;
|
||||
}
|
||||
String role = ctx.queryParam("role");
|
||||
@@ -391,6 +502,11 @@ public final class FleetApp {
|
||||
ctx.status(201).json(view(member));
|
||||
} catch (GuardException e) {
|
||||
ctx.status(403).json(Map.of("error", "subscription_boundary", "detail", e.getMessage()));
|
||||
} catch (ShuttingDownException e) {
|
||||
// fleetd #308: the daemon's shutdown drain has already started — 503, not a bare 500,
|
||||
// so this reads the same as PlacementException below: valid request, refused because
|
||||
// of a transient daemon state rather than a bad argument.
|
||||
ctx.status(503).json(Map.of("error", "shutting_down", "detail", e.getMessage()));
|
||||
} catch (PlacementException e) {
|
||||
// CB-599: no candidate had capacity (maxLoad, quarantine, or all-exhausted) — a benign,
|
||||
// likely-transient refusal, distinct from "profile does not exist" below. 503: the
|
||||
@@ -400,6 +516,12 @@ public final class FleetApp {
|
||||
ctx.status(400).json(Map.of("error", "unknown_profile", "detail", e.getMessage()));
|
||||
} catch (PeerUnreachableException e) {
|
||||
ctx.status(502).json(Map.of("error", "spawn_timeout", "detail", e.getMessage()));
|
||||
} catch (HerdrException e) {
|
||||
// fleetd #304: not every herdr failure on the spawn path is a readiness timeout, so
|
||||
// PeerUnreachableException above does not cover this. Without this catch the exception
|
||||
// escapes to Javalin's default 500, while fleet_spawn reports the same failure as a
|
||||
// clean named error (FleetMcp.spawn) — the #297 one-door-guarded shape.
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -420,13 +542,29 @@ public final class FleetApp {
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
}
|
||||
|
||||
/** Tear a worker down by pane id. */
|
||||
/**
|
||||
* Tear a worker down by pane id.
|
||||
*
|
||||
* <p>fleetd #304: the {@code HerdrException} catch is not cosmetic. {@code release} deregisters
|
||||
* the session, notifies the release listener and preserves a dirty worktree <em>before</em> it
|
||||
* calls {@code launcher.stop}, so a throw from that stop arrives after the teardown the caller
|
||||
* asked for has already happened. Letting it escape gave Javalin's default 500, which tells the
|
||||
* caller to retry — and the retry finds nothing in the registry, reaches the same stop, and
|
||||
* throws again, so it can never succeed. {@code herdrError} instead answers 404 ("the pane is
|
||||
* gone, stop retrying") or 502 ("herdr is upstream and broken, a retry may help"), matching what
|
||||
* {@code fleet_stop} reports for the same failure.
|
||||
*/
|
||||
private void stopMember(Context ctx) {
|
||||
String paneId = ctx.pathParam("paneId");
|
||||
if (!allow(ctx, Authz.Action.STOP, paneId)) {
|
||||
if (!allow(ctx, routeAction("DELETE /members/{paneId}"), paneId)) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
sessions.release(paneId);
|
||||
} catch (HerdrException e) {
|
||||
herdrError(ctx, e);
|
||||
return;
|
||||
}
|
||||
sessions.release(paneId);
|
||||
ctx.status(204);
|
||||
}
|
||||
|
||||
@@ -438,7 +576,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void sendMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.SEND, id)) {
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/message"), id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
@@ -525,7 +663,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void askMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.ASK, id)) {
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/ask"), id)) {
|
||||
return;
|
||||
}
|
||||
String question;
|
||||
@@ -558,13 +696,17 @@ public final class FleetApp {
|
||||
/**
|
||||
* The worker's structured reply ({@code fleet_reply}) — resolves the blocking send awaiting
|
||||
* on this session, or queues the reply in the inbox when no send is open (CB-307).
|
||||
*
|
||||
* <p>fleetd #365: the response body's {@code delivered} field used to be unconditionally
|
||||
* {@code true} for either case; it now reports whether a send/ticket was actually resolved,
|
||||
* with {@code outcome} naming which (see {@link MessageService.ReplyOutcome}).
|
||||
*/
|
||||
private void replyMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
// The rule that matters: a worker may reply only as itself. Over MCP this was already true
|
||||
// structurally (identity comes from the connection, never an argument); over REST the path
|
||||
// id was simply trusted, so this is where the invariant actually gets enforced.
|
||||
if (!allow(ctx, Authz.Action.REPLY, id)) {
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/reply"), id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
@@ -574,8 +716,26 @@ public final class FleetApp {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "body must be JSON"));
|
||||
return;
|
||||
}
|
||||
messages.reply(id, content);
|
||||
ctx.status(200).json(Map.of("sessionId", id, "delivered", true));
|
||||
// fleetd #302: content is required. `.path("content").asText("")` above turns a missing key
|
||||
// into "" rather than throwing, so without this check an empty/blank reply used to reach
|
||||
// messages.reply(...) and silently resolve the lead's waiter — the same class of bug as the
|
||||
// sibling "content is required" guards on sendMessage/askMessage below, except this one wrote
|
||||
// a WRONG value instead of failing loudly. The check lives in MessageService.reply so both
|
||||
// this door and FleetMcp.reply inherit the same rule; this catch only translates it into the
|
||||
// {error, detail} envelope this file uses everywhere else.
|
||||
MessageService.ReplyOutcome outcome;
|
||||
try {
|
||||
outcome = messages.reply(id, content);
|
||||
} catch (IllegalArgumentException e) {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", e.getMessage()));
|
||||
return;
|
||||
}
|
||||
// fleetd #365: "delivered": true used to be unconditional here, whether the reply resolved
|
||||
// a waiting send or was merely queued in the inbox for a later drain — the same gap
|
||||
// FleetMcp.reply had over MCP. `delivered` now reflects which actually happened, and
|
||||
// `outcome` names the specific case (see MessageService.ReplyOutcome).
|
||||
ctx.status(200).json(Map.of("sessionId", id, "delivered", outcome.delivered(),
|
||||
"outcome", outcome.wireName()));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -585,7 +745,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void drainReplies(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.DRAIN, id)) {
|
||||
if (!allow(ctx, routeAction("GET /sessions/{id}/replies"), id)) {
|
||||
return;
|
||||
}
|
||||
var replies = messages.drainReplies(id);
|
||||
@@ -603,7 +763,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void sessionStatus(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.READ, id)) {
|
||||
if (!allow(ctx, routeAction("GET /sessions/{id}/status"), id)) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
@@ -628,7 +788,7 @@ public final class FleetApp {
|
||||
|
||||
/** Poll an async (wait:false) delegation by ticket. 404 for an unknown/expired ticket. */
|
||||
private void taskStatus(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /tasks/{ticket}"), null)) {
|
||||
return;
|
||||
}
|
||||
MessageService.TaskView v = messages.poll(ctx.pathParam("ticket"));
|
||||
|
||||
@@ -13,6 +13,7 @@ import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.nio.file.attribute.PosixFilePermissions;
|
||||
import java.security.SecureRandom;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
@@ -91,7 +92,19 @@ public final class GitWorktrees implements Worktrees {
|
||||
private final String configuredRoot;
|
||||
/** OS group name for {@link #shareWithGroup} (fleetd #185 stage 3); {@code null} ⇒ feature off. */
|
||||
private final String group;
|
||||
/** Source directory of skill folders for {@link #seedSkills} (fleetd #362, {@code memberSkills:}
|
||||
* in config); {@code null} ⇒ feature off, no worktree is touched beyond today's behaviour. */
|
||||
private final String memberSkillsSource;
|
||||
/** Extra environment merged into every {@code git} subprocess this instance runs. Always {@code
|
||||
* Map.of()} from every production constructor. Test seam only (fleetd #362 review fix): lets
|
||||
* {@code GitWorktreesTest} point {@code GIT_CONFIG_GLOBAL} at an isolated temp file so it can
|
||||
* drive the real {@link #add} path against a controlled "operator's global git config" and
|
||||
* prove the excludesFile composition below without ever touching the real machine's config. */
|
||||
private final Map<String, String> gitEnv;
|
||||
private final Consumer<String> afterWorktreeAdded;
|
||||
/** How the initial {@code git worktree add} command runs. Package-private test seam for an
|
||||
* interrupted command after Git has made worktree state. */
|
||||
private final Function<String[], String> worktreeAddRunner;
|
||||
/** How {@link #shareWithGroup}'s processes (git config / chgrp / chmod / find) actually run.
|
||||
* Defaults to the real {@link #exec(String...)}. Package-private test seam so a unit test can
|
||||
* prove "no group configured ⇒ zero processes spawned" and inspect exactly what a configured
|
||||
@@ -121,7 +134,20 @@ public final class GitWorktrees implements Worktrees {
|
||||
* config); null/blank ⇒ {@link #shareWithGroup} is a no-op.
|
||||
*/
|
||||
public GitWorktrees(String configuredRoot, String group) {
|
||||
this(configuredRoot, group, _ -> {});
|
||||
this(configuredRoot, group, (String) null);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param configuredRoot nullable absolute or relative path; null/blank derives a sibling of
|
||||
* the repo root.
|
||||
* @param group optional OS group name (fleetd #185 stage 3, {@code worktreeGroup:}
|
||||
* in config); null/blank ⇒ {@link #shareWithGroup} is a no-op.
|
||||
* @param memberSkillsSource fleetd #362: optional directory of skill folders ({@code
|
||||
* memberSkills:} in config) copied into every provisioned worktree's
|
||||
* {@code .claude/skills/}; null/blank ⇒ {@link #seedSkills} is a no-op.
|
||||
*/
|
||||
public GitWorktrees(String configuredRoot, String group, String memberSkillsSource) {
|
||||
this(configuredRoot, group, _ -> {}, null, null, memberSkillsSource);
|
||||
}
|
||||
|
||||
/** Test seam for changing a real worktree between its creation and its security check. */
|
||||
@@ -131,7 +157,20 @@ public final class GitWorktrees implements Worktrees {
|
||||
|
||||
/** Test seam combining a configurable {@code group} with {@link #afterWorktreeAdded}. */
|
||||
GitWorktrees(String configuredRoot, String group, Consumer<String> afterWorktreeAdded) {
|
||||
this(configuredRoot, group, afterWorktreeAdded, null);
|
||||
this(configuredRoot, group, afterWorktreeAdded, null, null, null);
|
||||
}
|
||||
|
||||
/** Test seam for changing how {@link #shareWithGroup}'s processes run. */
|
||||
GitWorktrees(String configuredRoot, String group, Consumer<String> afterWorktreeAdded,
|
||||
Function<String[], String> shareGroupRunner) {
|
||||
this(configuredRoot, group, afterWorktreeAdded, shareGroupRunner, null, null);
|
||||
}
|
||||
|
||||
/** Test seam for changing how the initial {@code git worktree add} command runs, with no
|
||||
* {@code memberSkillsSource} configured. */
|
||||
GitWorktrees(String configuredRoot, String group, Consumer<String> afterWorktreeAdded,
|
||||
Function<String[], String> shareGroupRunner, Function<String[], String> worktreeAddRunner) {
|
||||
this(configuredRoot, group, afterWorktreeAdded, shareGroupRunner, worktreeAddRunner, null);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -140,13 +179,34 @@ public final class GitWorktrees implements Worktrees {
|
||||
* exactly what commands a configured group runs, without a real second OS user/group.
|
||||
*
|
||||
* @param shareGroupRunner {@code null} ⇒ the real {@link #exec(String...)}.
|
||||
* @param worktreeAddRunner {@code null} ⇒ the real {@link #exec(String...)}.
|
||||
* @param memberSkillsSource {@code null}/blank ⇒ {@link #seedSkills} is a no-op.
|
||||
*/
|
||||
GitWorktrees(String configuredRoot, String group, Consumer<String> afterWorktreeAdded,
|
||||
Function<String[], String> shareGroupRunner) {
|
||||
Function<String[], String> shareGroupRunner, Function<String[], String> worktreeAddRunner,
|
||||
String memberSkillsSource) {
|
||||
this(configuredRoot, group, afterWorktreeAdded, shareGroupRunner, worktreeAddRunner,
|
||||
memberSkillsSource, Map.of());
|
||||
}
|
||||
|
||||
/**
|
||||
* Full test seam, plus {@code gitEnv} (fleetd #362 review fix, verification only): extra
|
||||
* environment merged into every {@code git} subprocess this instance runs, so a test can isolate
|
||||
* something like {@code GIT_CONFIG_GLOBAL} from the real machine while still driving the real
|
||||
* {@link #add} path end to end. Every production constructor above delegates here with {@code
|
||||
* Map.of()}.
|
||||
*/
|
||||
GitWorktrees(String configuredRoot, String group, Consumer<String> afterWorktreeAdded,
|
||||
Function<String[], String> shareGroupRunner, Function<String[], String> worktreeAddRunner,
|
||||
String memberSkillsSource, Map<String, String> gitEnv) {
|
||||
this.configuredRoot = configuredRoot;
|
||||
this.group = (group == null || group.isBlank()) ? null : group;
|
||||
this.memberSkillsSource = (memberSkillsSource == null || memberSkillsSource.isBlank())
|
||||
? null : memberSkillsSource;
|
||||
this.gitEnv = gitEnv == null ? Map.of() : gitEnv;
|
||||
this.afterWorktreeAdded = afterWorktreeAdded == null ? _ -> {} : afterWorktreeAdded;
|
||||
this.shareGroupRunner = shareGroupRunner != null ? shareGroupRunner : this::exec;
|
||||
this.worktreeAddRunner = worktreeAddRunner != null ? worktreeAddRunner : this::exec;
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -161,18 +221,75 @@ public final class GitWorktrees implements Worktrees {
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot create worktree root " + root + ": " + e.getMessage(), e);
|
||||
}
|
||||
shareRootWithGroup(root);
|
||||
String wt = path.toAbsolutePath().toString();
|
||||
log.info("adding worktree branch={} path={} base={}", branch, wt, base);
|
||||
removeUserInfoFromHttpsOrigin(repoRoot);
|
||||
exec("git", "-C", repoRoot, "worktree", "add", wt, "-b", branch, base);
|
||||
afterWorktreeAdded.accept(wt);
|
||||
requireCredentialFreeHttpsOrigin(wt);
|
||||
configureEnvironmentCredentialHelper(repoRoot, wt);
|
||||
configureHttpsUrlRewriteForSshOrigin(repoRoot, wt);
|
||||
isolateToolSurface(wt);
|
||||
try {
|
||||
worktreeAddRunner.apply(new String[] {"git", "-C", repoRoot, "worktree", "add", wt, "-b", branch, base});
|
||||
afterWorktreeAdded.accept(wt);
|
||||
requireCredentialFreeHttpsOrigin(wt);
|
||||
configureEnvironmentCredentialHelper(repoRoot, wt);
|
||||
configureHttpsUrlRewriteForSshOrigin(repoRoot, wt);
|
||||
isolateToolSurface(wt);
|
||||
seedSkills(wt);
|
||||
} catch (RuntimeException e) {
|
||||
cleanupAfterAddFailure(repoRoot, wt, branch, e);
|
||||
throw e;
|
||||
}
|
||||
return wt;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code git worktree add} may have created the worktree and its branch by the time it, or any
|
||||
* later step through {@link #isolateToolSurface}, throws. This includes
|
||||
* {@link #requireCredentialFreeHttpsOrigin}, an intended security refusal, not only an IO
|
||||
* accident. A Git-reported {@code worktree add} failure usually creates nothing, but an
|
||||
* interrupted command can leave partial state. Without cleanup, {@code add()} never returns, so its caller
|
||||
* ({@code SessionManager#acquireWithWorktree}) never receives a path to register or clean up:
|
||||
* its local {@code path} stays null, the {@code if (path != null)} guard in its own catch block
|
||||
* never runs, and the worktree directory and branch leak on disk forever with nothing tracking
|
||||
* them (fleetd #274).
|
||||
*
|
||||
* <p>Reuses {@link #remove} — the same {@code git worktree remove --force} path every other
|
||||
* cleanup exit in this class already goes through — rather than a bespoke removal. It
|
||||
* additionally deletes {@code branch}: {@link #remove} alone deliberately leaves a released
|
||||
* session's branch behind (a worker's branch is expected to outlive its worktree, for PRs and
|
||||
* recovery), but a branch that never finished provisioning has no session, no PR, and nothing
|
||||
* else pointing at it, so leaving it behind would just trade one leak for a smaller one. Forced
|
||||
* (`-D`) because the branch is new and unmerged by construction. The worktree is removed first:
|
||||
* a branch checked out by a worktree cannot be deleted until the worktree that holds it is gone.
|
||||
*
|
||||
* <p>Cleanup failure must never mask {@code original} — that is the exception that explains
|
||||
* what actually went wrong — so a failure here is only logged, matching the pattern already
|
||||
* used in {@code SessionManager#acquireWithWorktree}'s own catch block.
|
||||
*/
|
||||
private void cleanupAfterAddFailure(String repoRoot, String worktreePath, String branch, RuntimeException original) {
|
||||
if (!Files.exists(Path.of(worktreePath))) {
|
||||
log.debug("provisioning failed before worktree {} existed; nothing to clean up", worktreePath);
|
||||
return;
|
||||
}
|
||||
log.warn("provisioning failed for branch={} path={}: {} — cleaning up before rethrowing",
|
||||
branch, worktreePath, original.getMessage());
|
||||
try {
|
||||
remove(repoRoot, worktreePath);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to remove leaked worktree {} after provisioning error: {}",
|
||||
worktreePath, cleanup.getMessage());
|
||||
}
|
||||
try {
|
||||
deleteBranch(repoRoot, branch);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to remove leaked branch {} after provisioning error: {}",
|
||||
branch, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void deleteBranch(String repoRoot, String branch) {
|
||||
exec("git", "-C", repoRoot, "branch", "-D", branch);
|
||||
}
|
||||
|
||||
/**
|
||||
* A linked worktree shares its primary checkout's git config. Remove HTTPS user info before
|
||||
* adding one, so a credential accidentally embedded in that config cannot reach the member.
|
||||
@@ -424,19 +541,75 @@ public final class GitWorktrees implements Worktrees {
|
||||
* neutralized copy from ever showing up as a local modification the worker might commit. A config
|
||||
* the repo does not carry is skipped silently — no stub is invented for a file the repo does not
|
||||
* have, and one missing file must never fail provisioning.
|
||||
*
|
||||
* <p>fleetd #134. The neutralization above is correct and stays unconditional — the defect was
|
||||
* that it was invisible on both sides. Neither the daemon's own log nor the worker sitting in the
|
||||
* worktree could tell a stub from the repo's real file: a real worker read a 3-byte {@code {}}
|
||||
* where the repo's {@code opencode.json} is 30+ lines, and reported — truthfully from what it
|
||||
* could see, and wrongly — that a mount key did not exist. Two fixes, aimed at two different
|
||||
* readers:
|
||||
* <ul>
|
||||
* <li>the daemon operator reads {@link #log}, so the summary below names the denominator, what
|
||||
* was neutralized, and why anything was not — the same shape {@code overlayParity} reports
|
||||
* its own copy in;
|
||||
* <li>the worker reads its own worktree, not the daemon's log, so the same fact is recorded a
|
||||
* second time in worktree-scoped git config ({@code fleet.neutralizedConfig} /
|
||||
* {@code fleet.neutralizedConfigNote}, readable with {@code git config --worktree --get-all
|
||||
* fleet.neutralizedConfig}) rather than as a file in the working tree. A working-tree file
|
||||
* would show up in {@code git status} for the worker to trip on or commit; worktree-scoped
|
||||
* config lives in {@code .git/worktrees/<nonce>/config.worktree} and can never appear there.
|
||||
* This reuses the exact mechanism {@link #configureEnvironmentCredentialHelper} and
|
||||
* {@link #configureHttpsUrlRewriteForSshOrigin} already use for other worktree-local state.
|
||||
* </ul>
|
||||
*/
|
||||
private void isolateToolSurface(String worktreePath) {
|
||||
Path root = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
List<String> neutralized = new ArrayList<>();
|
||||
List<String> skipped = new ArrayList<>();
|
||||
for (WorktreeHostileConfig cfg : WORKTREE_HOSTILE_CONFIGS) {
|
||||
neutralize(root, worktreePath, cfg);
|
||||
if (neutralize(root, worktreePath, cfg)) {
|
||||
neutralized.add(cfg.file());
|
||||
} else {
|
||||
skipped.add(cfg.file() + " absent");
|
||||
}
|
||||
}
|
||||
String detail = neutralized.isEmpty() ? String.join(", ", skipped)
|
||||
: skipped.isEmpty() ? String.join(", ", neutralized)
|
||||
: String.join(", ", neutralized) + " (" + String.join(", ", skipped) + ")";
|
||||
log.info("tool-surface isolation: neutralized {} of {} configs: {} — the worktree copy is a "
|
||||
+ "stub, not the repo's file; edit the real file in the primary checkout instead",
|
||||
neutralized.size(), WORKTREE_HOSTILE_CONFIGS.size(), detail);
|
||||
recordNeutralizedConfigForWorker(worktreePath, neutralized);
|
||||
}
|
||||
|
||||
private void neutralize(Path root, String worktreePath, WorktreeHostileConfig cfg) {
|
||||
/**
|
||||
* The worker-readable half of fleetd #134: record which files were neutralized where the worker
|
||||
* itself can read it, without a working-tree file that would show up in {@code git status}.
|
||||
* Worktree-scoped git config is per-worktree, lives under {@code .git/worktrees/<nonce>/} rather
|
||||
* than the working tree, and this repo already relies on the same mechanism (and the same
|
||||
* {@code extensions.worktreeConfig} enablement) for the credential helper and the SSH→HTTPS
|
||||
* rewrite — see {@link #configureEnvironmentCredentialHelper}.
|
||||
*/
|
||||
private void recordNeutralizedConfigForWorker(String worktreePath, List<String> neutralized) {
|
||||
if (neutralized.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
exec("git", "-C", worktreePath, "config", "extensions.worktreeConfig", "true");
|
||||
for (String file : neutralized) {
|
||||
exec("git", "-C", worktreePath, "config", "--worktree", "--add", "fleet.neutralizedConfig", file);
|
||||
}
|
||||
exec("git", "-C", worktreePath, "config", "--worktree", "fleet.neutralizedConfigNote",
|
||||
"the worktree copy of each fleet.neutralizedConfig path is a stub, not the repo's "
|
||||
+ "committed file; edit the real file from the primary checkout instead");
|
||||
}
|
||||
|
||||
/** @return true if {@code cfg} was neutralized (present, or created because {@link
|
||||
* WorktreeHostileConfig#createIfAbsent()}); false if the repo does not carry it and it was
|
||||
* left alone. */
|
||||
private boolean neutralize(Path root, String worktreePath, WorktreeHostileConfig cfg) {
|
||||
Path target = root.resolve(cfg.file());
|
||||
if (!Files.exists(target) && !cfg.createIfAbsent()) {
|
||||
log.debug("{} absent in the worktree — skipping (repo does not carry it)", cfg.file());
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
try {
|
||||
Files.writeString(target, cfg.stub());
|
||||
@@ -447,7 +620,323 @@ public final class GitWorktrees implements Worktrees {
|
||||
if (isTracked(root, cfg.file())) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", cfg.file());
|
||||
}
|
||||
log.debug("neutralized {} — worker tool surface is launcher-mounted only", cfg.file());
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #362: copy each skill folder from the configured {@link #memberSkillsSource} directory
|
||||
* into {@code <worktreePath>/.claude/skills/}, so a member spawned against ANY repo — not only
|
||||
* one that already ships its own {@code .claude/skills/} — can load a bridge skill such as
|
||||
* {@code implementer}. Every brief this fleet sends starts with {@code "Load the <name>
|
||||
* skill."}; outside a repo carrying its own copy that line was previously a no-op.
|
||||
*
|
||||
* <p><b>No-op — nothing read, nothing written, nothing logged</b> — when {@link
|
||||
* #memberSkillsSource} is null/blank (today's default), the same off-switch shape as
|
||||
* {@link #shareWithGroup}.
|
||||
*
|
||||
* <p><b>Invariant 1 — a repo's own skill wins.</b> A skill folder already present at
|
||||
* {@code <worktreePath>/.claude/skills/<name>} — because the just-checked-out branch commits its
|
||||
* own copy — is left completely untouched: never overwritten, and never even opened.
|
||||
*
|
||||
* <p><b>Invariant 2 — a seeded skill can never end up in a worker's commit.</b> Every path this
|
||||
* writes is untracked in the target repo (that is the whole reason it is being seeded), so
|
||||
* {@code git status} would otherwise show each one as a new, addable, committable path. The
|
||||
* repo-wide {@code .git/info/exclude} is NOT used for this: measured against a real linked
|
||||
* worktree, that file resolves to the repository's COMMON git dir even from a worktree (the
|
||||
* same file {@link dev.ltms.fleet.member.ClaudeCodeLauncher#writeIdeOverlay writeIdeOverlay}
|
||||
* appends {@code CLAUDE.local.md} to), so an entry written there would hide the seeded skill
|
||||
* from {@code git status} in the PRIMARY's own checkout and every sibling worktree too — not
|
||||
* only this one. Instead, {@link #excludeSeededSkillsFromGitStatus} points {@code
|
||||
* core.excludesFile} at a file scoped {@code --worktree} (the same {@code
|
||||
* extensions.worktreeConfig} mechanism {@link #configureEnvironmentCredentialHelper} already
|
||||
* relies on) that itself lives under this worktree's own private git dir ({@code
|
||||
* .git/worktrees/<nonce>/}, OUTSIDE the working tree) — invisible to this worktree's {@code git
|
||||
* status} and structurally impossible for this worktree to commit, with no effect on any other
|
||||
* worktree or the primary checkout. Proven with a real {@code git status --porcelain} in
|
||||
* {@code GitWorktreesTest}, not by reasoning.
|
||||
*
|
||||
* <p><b>Compose, don't replace.</b> {@code core.excludesFile} is single-valued: the first cut of
|
||||
* this method pointed it at fleetd's own file with {@code --replace-all}, which SHADOWS whatever
|
||||
* the operator's own (global, or repo-local) {@code core.excludesFile} was already resolving to
|
||||
* inside this worktree, rather than adding to it. Measured concretely: this repo's own {@code
|
||||
* .gitignore} does not ignore {@code target/} — only an operator's global excludesFile does — so
|
||||
* every worker's {@code mvn clean install} would otherwise make {@code target/} appear as
|
||||
* untracked, and {@link #hasUncommitted}'s deliberately-untracked-inclusive {@code git status
|
||||
* --porcelain} (CB-576) would then read every such worktree as dirty forever, so it is never
|
||||
* cleaned up. {@link #excludeSeededSkillsFromGitStatus} now reads whatever {@code
|
||||
* core.excludesFile} resolves to BEFORE writing anything (falling back to git's own documented
|
||||
* default, {@code $XDG_CONFIG_HOME/git/ignore} or {@code $HOME/.config/git/ignore}, when the key
|
||||
* is unset entirely — see {@code gitignore(5)}), and writes that content into its OWN exclude
|
||||
* file ahead of the seeded skill patterns, so every operator-configured pattern keeps applying
|
||||
* inside the seeded worktree exactly as it did before seeding ran.
|
||||
*
|
||||
* <p>Instead of using worktree-scoped-config as an add-then-append (a second key does not exist
|
||||
* for {@code core.excludesFile} — it takes exactly one value), an actual second exclude source
|
||||
* was ruled out because git resolves only ONE {@code core.excludesFile}; concatenating the prior
|
||||
* content into fleetd's own file is what "compose" reduces to for a single-valued key.
|
||||
*
|
||||
* <p>Proven the same way as invariant 2's own leak check: {@code
|
||||
* GitWorktreesTest#seedSkillsComposesWithAnAlreadyEffectiveGlobalExcludesFile} isolates a
|
||||
* synthetic "operator's global config" via {@code GIT_CONFIG_GLOBAL} (never the real machine's),
|
||||
* seeds a skill, and asserts {@code git status --porcelain} is still empty for a file matching
|
||||
* that global config's own ignore pattern.
|
||||
*
|
||||
* <p><b>Invariant 3 — best-effort.</b> A missing/unreadable {@link #memberSkillsSource}, or a
|
||||
* copy/exclude failure, is logged and skipped — it must never fail the spawn, the same contract
|
||||
* {@link #overlayParity} and {@link #isolateToolSurface} already hold.
|
||||
*
|
||||
* <p>Recorded for the worker itself the same way fleetd #134 records {@code
|
||||
* fleet.neutralizedConfig}: {@code fleet.seededSkills} (one value per seeded skill folder) and
|
||||
* {@code fleet.seededSkillsNote}, readable with {@code git config --worktree --get-all
|
||||
* fleet.seededSkills}.
|
||||
*
|
||||
* <p><b>Kind-blind by construction, not by a backend check here — this used to be a real gap
|
||||
* (fleetd #393).</b> This method only copies files; like {@link #isolateToolSurface} — which
|
||||
* neutralizes BOTH {@code .mcp.json} and {@code opencode.json} unconditionally — it runs the
|
||||
* same for every worktree regardless of which backend ultimately spawns into it, because no
|
||||
* caller of {@link #add} hands this class a kind to consult. Before fleetd #393, that made the
|
||||
* log line below a false claim of success for a {@code kind: opencode} member: opencode has no
|
||||
* built-in discovery of {@code .claude/skills/}, unlike the Claude Code CLI, so a seeded skill
|
||||
* never reached one. It now does — {@code OpenCodeLauncher#skillInstructionFiles} reads
|
||||
* whatever this method copied into {@code .claude/skills/} and appends each {@code SKILL.md} to
|
||||
* the generated {@code instructions[]} — but that delivery, and the log line that honestly
|
||||
* claims it (kind-aware, unlike this one), happens at the launcher, once the kind is actually
|
||||
* known, not here.
|
||||
*/
|
||||
private void seedSkills(String worktreePath) {
|
||||
if (memberSkillsSource == null) {
|
||||
return;
|
||||
}
|
||||
Path source = Path.of(memberSkillsSource).toAbsolutePath().normalize();
|
||||
if (!Files.isDirectory(source)) {
|
||||
log.warn("memberSkills source '{}' is not a directory — skipping skill seeding for worktree {}",
|
||||
source, worktreePath);
|
||||
return;
|
||||
}
|
||||
Path skillsRoot = Path.of(worktreePath).resolve(".claude").resolve("skills");
|
||||
List<String> seeded = new ArrayList<>();
|
||||
List<String> kept = new ArrayList<>();
|
||||
try (var candidates = Files.list(source)) {
|
||||
for (Path candidate : candidates
|
||||
.filter(Files::isDirectory)
|
||||
.filter(p -> !p.getFileName().toString().startsWith("."))
|
||||
.sorted()
|
||||
.toList()) {
|
||||
String name = candidate.getFileName().toString();
|
||||
Path dst = skillsRoot.resolve(name);
|
||||
if (Files.exists(dst)) {
|
||||
kept.add(name);
|
||||
continue;
|
||||
}
|
||||
copySkillDirectory(candidate, dst);
|
||||
seeded.add(name);
|
||||
}
|
||||
} catch (IOException | RuntimeException e) {
|
||||
log.warn("failed to seed skills into worktree {} from memberSkills source '{}': {}",
|
||||
worktreePath, source, e.getMessage());
|
||||
return;
|
||||
}
|
||||
String detail = seeded.isEmpty() ? "" : "seeded: " + String.join(", ", seeded);
|
||||
if (!kept.isEmpty()) {
|
||||
detail += (detail.isEmpty() ? "" : "; ") + "kept the repo's own copy of: " + String.join(", ", kept);
|
||||
}
|
||||
if (detail.isEmpty()) {
|
||||
detail = "no skill folders found under " + source;
|
||||
}
|
||||
// fleetd #393: this only claims the copy step, deliberately — it cannot know the member
|
||||
// kind that will spawn into this worktree (see this method's own javadoc), so it must not
|
||||
// read as "and the member will act on it." Whether that is true depends on the kind: the
|
||||
// Claude Code CLI discovers .claude/skills/ on its own; OpenCodeLauncher logs its own
|
||||
// "skill delivery" line, once the kind is known, naming what it could and could not turn
|
||||
// into instructions[].
|
||||
log.info("skill seeding: {} of {} candidate(s) from {} into {}/.claude/skills — {} "
|
||||
+ "(whether the spawned member can act on this depends on its kind — see "
|
||||
+ "the launcher's own log for that)",
|
||||
seeded.size(), seeded.size() + kept.size(), source, worktreePath, detail);
|
||||
if (seeded.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
excludeSeededSkillsFromGitStatus(worktreePath, seeded);
|
||||
recordSeededSkillsForWorker(worktreePath, seeded);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("seeded skill(s) {} into {} but could not hide them from git status: {} — "
|
||||
+ "they may show as untracked; never commit them", seeded, worktreePath, e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/** Recursively copy a skill folder ({@code src}) into a fresh destination ({@code dst}) that
|
||||
* {@link #seedSkills} has already confirmed does not exist, preserving the directory structure
|
||||
* (e.g. {@code implementer/SKILL.md}, {@code implementer/references/...}). */
|
||||
private static void copySkillDirectory(Path src, Path dst) {
|
||||
try (var walk = Files.walk(src)) {
|
||||
for (Path path : walk.sorted().toList()) {
|
||||
Path target = dst.resolve(src.relativize(path).toString());
|
||||
if (Files.isDirectory(path)) {
|
||||
Files.createDirectories(target);
|
||||
} else {
|
||||
Files.createDirectories(target.getParent());
|
||||
Files.copy(path, target, StandardCopyOption.COPY_ATTRIBUTES);
|
||||
}
|
||||
}
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot copy skill directory " + src + " -> " + dst + ": "
|
||||
+ e.getMessage(), e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Make every path in {@code seededSkillNames} (each a name under {@code .claude/skills/})
|
||||
* invisible to {@code git status} in THIS worktree only — see the invariant-2 discussion on
|
||||
* {@link #seedSkills}. Sets {@code core.excludesFile} scoped {@code --worktree} to a file
|
||||
* written under this worktree's own private git dir ({@code git rev-parse
|
||||
* --absolute-git-dir}), which lives outside the working tree, so the exclude file itself can
|
||||
* never be committed either.
|
||||
*
|
||||
* <p><b>Compose, don't replace.</b> {@code core.excludesFile} is single-valued, so pointing it at
|
||||
* fleetd's own file would otherwise SHADOW whatever excludesFile this worktree was already
|
||||
* resolving (an operator's global config, most commonly) rather than add to it — see the
|
||||
* "Compose, don't replace" discussion on {@link #seedSkills}. {@link
|
||||
* #previouslyEffectiveExcludesFileContent} is read BEFORE this method's own {@code --worktree}
|
||||
* write below, so it still sees whatever was effective beforehand; that content is written into
|
||||
* fleetd's own exclude file ahead of the seeded skill patterns, and the worktree-scoped override
|
||||
* then points at that combined file — so every pattern the operator's own configuration already
|
||||
* applied keeps applying, plus the seeded skill paths.
|
||||
*
|
||||
* <p><b>Assumes a fresh worktree — not idempotent.</b> {@link #seedSkills} only ever calls this
|
||||
* from {@link #add}, which always creates a brand-new worktree, so {@code core.excludesFile} is
|
||||
* never already worktree-scoped-set to fleetd's own file when this runs. A hypothetical second
|
||||
* call on the SAME worktree would read fleetd's own already-composed file back as "previously
|
||||
* effective" (worktree scope now wins) and append the seeded patterns a second time — harmless
|
||||
* to {@code git status} (duplicate exclude lines are a no-op), but not something to rely on. No
|
||||
* guard is added for this because the path does not exist today; if a future caller ever seeds
|
||||
* the same worktree twice, it will need one.
|
||||
*/
|
||||
private void excludeSeededSkillsFromGitStatus(String worktreePath, List<String> seededSkillNames) {
|
||||
exec("git", "-C", worktreePath, "config", "extensions.worktreeConfig", "true");
|
||||
String previouslyEffective = previouslyEffectiveExcludesFileContent(worktreePath);
|
||||
String gitDir = exec("git", "-C", worktreePath, "rev-parse", "--absolute-git-dir").trim();
|
||||
Path excludeFile = Path.of(gitDir, "fleet-seeded-skills-exclude");
|
||||
StringBuilder patterns = new StringBuilder();
|
||||
if (!previouslyEffective.isEmpty()) {
|
||||
patterns.append(previouslyEffective);
|
||||
}
|
||||
for (String name : seededSkillNames) {
|
||||
patterns.append("/.claude/skills/").append(name).append('/').append(System.lineSeparator());
|
||||
}
|
||||
try {
|
||||
Files.writeString(excludeFile, patterns.toString());
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot write skills exclude file " + excludeFile + ": "
|
||||
+ e.getMessage(), e);
|
||||
}
|
||||
exec("git", "-C", worktreePath, "config", "--worktree", "--replace-all", "core.excludesFile",
|
||||
excludeFile.toString());
|
||||
}
|
||||
|
||||
/**
|
||||
* The content of whatever {@code core.excludesFile} resolves to for {@code worktreePath} right
|
||||
* now — BEFORE {@link #excludeSeededSkillsFromGitStatus} points that key at fleetd's own file —
|
||||
* so it can be carried forward instead of shadowed. {@code --type=path} makes git itself perform
|
||||
* {@code ~}/{@code ~user} expansion the same way it would when actually reading the key to build
|
||||
* exclude rules, rather than handing back a raw, unexpanded config string.
|
||||
*
|
||||
* <p>When the key is unset entirely (exit code non-zero), falls back to git's own documented
|
||||
* default excludes file — {@code $XDG_CONFIG_HOME/git/ignore}, or {@code
|
||||
* $HOME/.config/git/ignore} when that variable is unset — per {@code gitignore(5)}: git applies
|
||||
* that file even with no {@code core.excludesFile} configured at all, so skipping it here would
|
||||
* silently drop patterns an operator never had to configure to get.
|
||||
*
|
||||
* <p>Never throws: a missing, unreadable, or unresolvable file is treated as "nothing to carry
|
||||
* forward" (empty string) — this is a best-effort read in service of {@link #seedSkills}'s own
|
||||
* invariant 3, not a new way for skill seeding to fail a spawn.
|
||||
*
|
||||
* <p><b>Review fix, finding 2.</b> The XDG-fallback branch below does not go through {@code git}
|
||||
* at all, so a first cut of it read {@code XDG_CONFIG_HOME}/{@code HOME} straight from the JVM's
|
||||
* own environment ({@link System#getenv} / {@code user.home}) — unlike every other value this
|
||||
* class resolves, which goes through a {@code git} subprocess and therefore already honours
|
||||
* {@link #gitEnv}. That meant no test could make this branch hermetic, and on any machine
|
||||
* carrying a real {@code ~/.config/git/ignore} (this repo's own dev machine does), every
|
||||
* skill-seeding test silently composed with that real file — correct in production, but
|
||||
* machine-dependent in the test suite, and a future broader pattern in that real file could
|
||||
* silently change what a seeded worktree's {@code git status} reports depending on whose home
|
||||
* directory ran the test. {@link #resolveEnv} now checks {@link #gitEnv} first for both
|
||||
* variables, falling back to the JVM's real environment only when the seam does not supply
|
||||
* them — production behaviour (empty {@link #gitEnv}) is unchanged, and a test can now isolate
|
||||
* this branch exactly as it already isolates every {@code git} subprocess call.
|
||||
*
|
||||
* <p><b>Snapshot, not a reference.</b> The content below is read once, at seeding time, and
|
||||
* copied into fleetd's own exclude file. If the operator edits their global excludesFile
|
||||
* afterward, an already-seeded worktree keeps the old copy — acceptable for a worktree's
|
||||
* expected lifetime, but worth knowing before reading a stale pattern as a bug.
|
||||
*
|
||||
* @return the file's content, trailing-newline-normalized, or {@code ""} when there is nothing
|
||||
* to compose with.
|
||||
*/
|
||||
private String previouslyEffectiveExcludesFileContent(String worktreePath) {
|
||||
String resolvedPath;
|
||||
if (exitCode("git", "-C", worktreePath, "config", "--get", "--type=path", "core.excludesFile") == 0) {
|
||||
resolvedPath = exec("git", "-C", worktreePath, "config", "--get", "--type=path",
|
||||
"core.excludesFile").trim();
|
||||
} else {
|
||||
String xdgConfigHome = resolveEnv("XDG_CONFIG_HOME");
|
||||
Path fallback = (xdgConfigHome != null && !xdgConfigHome.isBlank())
|
||||
? Path.of(xdgConfigHome, "git", "ignore")
|
||||
: Path.of(resolveHome(), ".config", "git", "ignore");
|
||||
resolvedPath = fallback.toString();
|
||||
}
|
||||
if (resolvedPath.isBlank()) {
|
||||
return "";
|
||||
}
|
||||
Path file = Path.of(resolvedPath);
|
||||
if (!Files.isRegularFile(file) || !Files.isReadable(file)) {
|
||||
return "";
|
||||
}
|
||||
try {
|
||||
String content = Files.readString(file);
|
||||
return content.isBlank() ? "" : content.stripTrailing() + System.lineSeparator();
|
||||
} catch (IOException e) {
|
||||
log.warn("could not read previously-effective excludesFile {} while seeding skills into "
|
||||
+ "{}: {} — its patterns will not carry forward into the seeded worktree",
|
||||
file, worktreePath, e.getMessage());
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve environment variable {@code name} for {@link #previouslyEffectiveExcludesFileContent}'s
|
||||
* XDG fallback, checking {@link #gitEnv} FIRST so a test can isolate this the same way it
|
||||
* already isolates every {@code git} subprocess this class runs, and falling back to the JVM's
|
||||
* real environment only when the seam does not supply it (always the case in production, where
|
||||
* {@link #gitEnv} is {@code Map.of()}).
|
||||
*/
|
||||
private String resolveEnv(String name) {
|
||||
String fromSeam = gitEnv.get(name);
|
||||
return fromSeam != null ? fromSeam : System.getenv(name);
|
||||
}
|
||||
|
||||
/** Same as {@link #resolveEnv(String)}, for {@code HOME} — falls back to {@code user.home}
|
||||
* (rather than {@code System.getenv("HOME")}) when the seam does not supply it, matching this
|
||||
* class's pre-existing behaviour for every other home-directory resolution. */
|
||||
private String resolveHome() {
|
||||
String fromSeam = gitEnv.get("HOME");
|
||||
return fromSeam != null ? fromSeam : System.getProperty("user.home");
|
||||
}
|
||||
|
||||
/**
|
||||
* The worker-readable half of fleetd #362, mirroring {@link #recordNeutralizedConfigForWorker}:
|
||||
* record which skill folders were seeded where the worker itself can read it, without a
|
||||
* working-tree file that would show up in {@code git status}.
|
||||
*/
|
||||
private void recordSeededSkillsForWorker(String worktreePath, List<String> seeded) {
|
||||
exec("git", "-C", worktreePath, "config", "extensions.worktreeConfig", "true");
|
||||
for (String name : seeded) {
|
||||
exec("git", "-C", worktreePath, "config", "--worktree", "--add", "fleet.seededSkills", name);
|
||||
}
|
||||
exec("git", "-C", worktreePath, "config", "--worktree", "fleet.seededSkillsNote",
|
||||
"each fleet.seededSkills value names a skill folder fleetd copied into "
|
||||
+ ".claude/skills/ because this repo did not already ship it; it is excluded "
|
||||
+ "from git status (core.excludesFile, worktree-scoped) and must never be committed");
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -484,25 +973,46 @@ public final class GitWorktrees implements Worktrees {
|
||||
}
|
||||
Path srcRoot = Path.of(repoRoot).toAbsolutePath().normalize();
|
||||
Path dstRoot = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
List<String> copied = new ArrayList<>();
|
||||
List<String> skipped = new ArrayList<>();
|
||||
List<String> neutralized = new ArrayList<>();
|
||||
for (String rel : overlay) {
|
||||
Path src = srcRoot.resolve(rel).normalize();
|
||||
if (!Files.exists(src)) {
|
||||
log.debug("parity overlay source missing — skipping {}", rel);
|
||||
skipped.add(rel + " absent");
|
||||
continue;
|
||||
}
|
||||
Path dst = dstRoot.resolve(rel).normalize();
|
||||
try {
|
||||
Files.createDirectories(dst.getParent());
|
||||
Files.copy(src, dst, StandardCopyOption.REPLACE_EXISTING, StandardCopyOption.COPY_ATTRIBUTES);
|
||||
log.debug("copied parity overlay {}", rel);
|
||||
copied.add(rel);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot copy overlay " + rel + ": " + e.getMessage(), e);
|
||||
}
|
||||
if (isTracked(dstRoot, rel)) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", rel);
|
||||
log.debug("marked overlay --skip-worktree {}", rel);
|
||||
neutralized.add(rel);
|
||||
}
|
||||
}
|
||||
// CB-148 point 3: a bare count ("copied 1") hides which candidates were even considered — the
|
||||
// same shape of under-reporting this repo has been bitten by before. Name the denominator
|
||||
// (every configured candidate), what was actually copied, and — for anything not copied —
|
||||
// why, so a spawn's overlay outcome is legible from the log alone, no filesystem dig required.
|
||||
String detail = copied.isEmpty() ? String.join(", ", skipped)
|
||||
: skipped.isEmpty() ? String.join(", ", copied)
|
||||
: String.join(", ", copied) + " (" + String.join(", ", skipped) + ")";
|
||||
log.info("parity overlay: copied {} of {} candidates: {}", copied.size(), overlay.size(), detail);
|
||||
// CB-134: a neutralized tracked file is otherwise a silent trap — a worker edits it, git
|
||||
// ignores the change with no error, and nothing anywhere said the file could not be
|
||||
// committed from this worktree. Name every file marked --skip-worktree here, with the
|
||||
// consequence stated in the message itself, rather than adding a worktree-local marker
|
||||
// file: acceptance criterion 1 requires the worktree hold exactly the configured overlay
|
||||
// set and nothing else, so an extra marker would itself violate the fix.
|
||||
if (!neutralized.isEmpty()) {
|
||||
log.info("parity overlay marked --skip-worktree (cannot be committed from this worktree): {}",
|
||||
String.join(", ", neutralized));
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -712,6 +1222,54 @@ public final class GitWorktrees implements Worktrees {
|
||||
group, repoRoot, worktreePath, touched);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #224: make {@code worktreeRoot} ITSELF group-traversable — established once, here,
|
||||
* where the root is created, never at a use site. {@link #shareWithGroup} shares each worktree
|
||||
* (and the repo's common git dir) with {@link #group}, but never the PARENT directory that
|
||||
* contains every worktree — and under {@code memberHerdrSocket:} the member pane runs as a
|
||||
* different OS user, which needs the execute bit on every ancestor directory to reach anything
|
||||
* underneath, no matter how carefully each child is shared. Without this, a member cannot read
|
||||
* the ephemeral {@code opencode.json} #219 places under this root, cannot reach its own
|
||||
* worktree, and cannot read #213's ZDOTDIR scrub when placed here either.
|
||||
*
|
||||
* <p>No-op — no process spawned — when {@link #group} is null/blank, so behaviour with
|
||||
* {@code worktreeGroup:} unset (today's only live mode) is unchanged. Only {@code root} itself
|
||||
* is touched (non-recursive): each child underneath is shared individually, either by
|
||||
* {@link #shareWithGroup} for a worktree or by the launcher that generates it (fleetd #213/#219)
|
||||
* for a scrub/config directory — sharing this level again would just duplicate that policy in
|
||||
* the wrong layer.
|
||||
*
|
||||
* <p>Fails loudly, naming {@code root}, its mode at the time of the attempt, and {@link #group}:
|
||||
* a member that starts and then cannot see its own checkout is worse than a refused spawn, since
|
||||
* nothing about that failure mode points at a directory's permission bits.
|
||||
*/
|
||||
private void shareRootWithGroup(Path root) {
|
||||
if (group == null) {
|
||||
return;
|
||||
}
|
||||
String mode = currentPosixMode(root);
|
||||
try {
|
||||
shareGroupRunner.apply(new String[]{"chgrp", group, root.toString()});
|
||||
shareGroupRunner.apply(new String[]{"chmod", "g+x", root.toString()});
|
||||
} catch (WorktreeException e) {
|
||||
throw new WorktreeException("cannot make worktree root " + root + " (mode " + mode
|
||||
+ ") group-traversable for group '" + group + "': " + e.getMessage()
|
||||
+ " — the group must exist, and the fleetd operator (" + System.getProperty("user.name")
|
||||
+ ") must be a member of it", e);
|
||||
}
|
||||
log.info("worktreeGroup={} made worktree root {} group-traversable (was mode {})", group, root, mode);
|
||||
}
|
||||
|
||||
/** {@code root}'s current POSIX permission string, or {@code "unknown"} on a filesystem that does
|
||||
* not support POSIX permissions — used only to name the mode in a refusal message. */
|
||||
private static String currentPosixMode(Path root) {
|
||||
try {
|
||||
return PosixFilePermissions.toString(Files.getPosixFilePermissions(root));
|
||||
} catch (IOException | UnsupportedOperationException e) {
|
||||
return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The repo's <em>common</em> git directory as an absolute path — where {@code objects},
|
||||
* {@code refs} and {@code worktrees} actually live. {@code git rev-parse --git-common-dir}
|
||||
@@ -891,6 +1449,9 @@ public final class GitWorktrees implements Worktrees {
|
||||
Process p;
|
||||
try {
|
||||
ProcessBuilder pb = new ProcessBuilder(command).redirectErrorStream(true);
|
||||
if (!gitEnv.isEmpty()) {
|
||||
pb.environment().putAll(gitEnv);
|
||||
}
|
||||
if (extraEnv != null && !extraEnv.isEmpty()) {
|
||||
pb.environment().putAll(extraEnv);
|
||||
}
|
||||
@@ -926,7 +1487,11 @@ public final class GitWorktrees implements Worktrees {
|
||||
private int exitCode(String... command) {
|
||||
Process p;
|
||||
try {
|
||||
p = new ProcessBuilder(command).redirectErrorStream(true).start();
|
||||
ProcessBuilder pb = new ProcessBuilder(command).redirectErrorStream(true);
|
||||
if (!gitEnv.isEmpty()) {
|
||||
pb.environment().putAll(gitEnv);
|
||||
}
|
||||
p = pb.start();
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("failed to start " + command[0] + ": " + e.getMessage(), e);
|
||||
}
|
||||
|
||||
@@ -46,7 +46,8 @@ public record MemberSession(
|
||||
String worktree,
|
||||
String branch,
|
||||
CharterReceipt charterReceipt,
|
||||
String agentSessionId) {
|
||||
String agentSessionId,
|
||||
String failureReason) {
|
||||
|
||||
/** One-shot worker lifecycle states. */
|
||||
public enum State {
|
||||
@@ -54,6 +55,7 @@ public record MemberSession(
|
||||
READY,
|
||||
BUSY,
|
||||
DONE,
|
||||
BACKEND_ERROR,
|
||||
FAILED,
|
||||
RELEASED
|
||||
}
|
||||
@@ -69,25 +71,35 @@ public record MemberSession(
|
||||
long lastActivityAtNanos, int turnCount, State state,
|
||||
String worktree, String branch) {
|
||||
this(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, null, null);
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, null, null, null);
|
||||
}
|
||||
|
||||
/** Backward-compatible shape without a backend failure reason. */
|
||||
public MemberSession(String paneId, String terminalId, String profile, MemberRole role,
|
||||
String cwd, String ownerTerminal, long spawnedAtNanos,
|
||||
long lastActivityAtNanos, int turnCount, State state,
|
||||
String worktree, String branch, CharterReceipt charterReceipt,
|
||||
String agentSessionId) {
|
||||
this(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, null);
|
||||
}
|
||||
|
||||
/** Return a copy of this session in {@code state}. */
|
||||
public MemberSession withState(State state) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId);
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
|
||||
/** Return a copy with {@code lastActivityAtNanos} updated to {@code nowNanos}. */
|
||||
public MemberSession withActivity(long nowNanos) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId);
|
||||
nowNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
|
||||
/** Return a copy with the turn count incremented and activity timestamped at {@code nowNanos}. */
|
||||
public MemberSession bumpTurn(long nowNanos) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount + 1, state, worktree, branch, charterReceipt, agentSessionId);
|
||||
nowNanos, turnCount + 1, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -98,6 +110,12 @@ public record MemberSession(
|
||||
*/
|
||||
public MemberSession withAgentSessionId(String agentSessionId) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId);
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
|
||||
/** Return a copy with the durable backend failure detail. */
|
||||
public MemberSession withFailureReason(String failureReason) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,6 +10,7 @@ import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.PlacementDecision;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -21,6 +22,7 @@ import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.LongSupplier;
|
||||
@@ -61,6 +63,8 @@ public final class SessionManager implements TurnListener {
|
||||
private final LongSupplier nowNanos;
|
||||
private final int contextCap;
|
||||
private final boolean clearAfterTurn;
|
||||
/** Null in production; test seam for the interval before an idle session's conditional release. */
|
||||
private final Consumer<MemberSession> beforeIdleRelease;
|
||||
private volatile MemberLifecycle memberLifecycle = MemberLifecycle.NONE;
|
||||
/**
|
||||
* CB-586: the repo root the fleet actually works in, remembered the first time a worktree
|
||||
@@ -71,6 +75,16 @@ public final class SessionManager implements TurnListener {
|
||||
*/
|
||||
private volatile String fleetRepoRoot;
|
||||
|
||||
/**
|
||||
* fleetd #308: flips true the instant {@link #drainAll} starts, before its registry snapshot
|
||||
* is even taken — so a spawn already in flight sees the refusal as early as a plain flag can
|
||||
* make it. This alone cannot close the race completely: a caller that read {@code false} just
|
||||
* before the flip can still land in the registry after the snapshot. {@link #drainAll}'s
|
||||
* post-loop sweep is what catches that straggler; the two mechanisms are deliberately paired,
|
||||
* see {@link #drainAll}'s javadoc.
|
||||
*/
|
||||
private final AtomicBoolean draining = new AtomicBoolean(false);
|
||||
|
||||
/** CB-520: notified with a terminalId on every acquire; no-op until wired. */
|
||||
private final List<Consumer<String>> acquireListeners = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
/** CB-516: notified with a {@link ReleaseDetail} on every release; no-op until wired. */
|
||||
@@ -102,13 +116,23 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap, boolean clearAfterTurn) {
|
||||
int contextCap, boolean clearAfterTurn) {
|
||||
this(launcher, worktrees, nowNanos, contextCap, clearAfterTurn, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Package-private constructor for a deterministic reap/delivery race test. Production callers
|
||||
* use the constructor above, whose null hook adds no callback or lock to an ordinary reap.
|
||||
*/
|
||||
SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap, boolean clearAfterTurn, Consumer<MemberSession> beforeIdleRelease) {
|
||||
this.launcher = launcher;
|
||||
this.worktrees = worktrees;
|
||||
this.presence = new PresenceFleet(this);
|
||||
this.nowNanos = nowNanos;
|
||||
this.contextCap = contextCap;
|
||||
this.clearAfterTurn = clearAfterTurn;
|
||||
this.beforeIdleRelease = beforeIdleRelease;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -181,48 +205,61 @@ public final class SessionManager implements TurnListener {
|
||||
public MemberSession acquire(String profile, MemberRole role, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt,
|
||||
String sessionName, String resumeSessionId) {
|
||||
// fleetd #308: refuse before anything else runs — no slot reservation, no launcher spawn —
|
||||
// so a caller learns the daemon is going down instead of getting a session drainAll will
|
||||
// never see again. Checked here because every other acquire(...) overload delegates to
|
||||
// this one, so this is the single point every spawn path passes through.
|
||||
if (draining.get()) {
|
||||
throw new ShuttingDownException("fleetd is shutting down; refusing to spawn a session "
|
||||
+ "the shutdown drain would never see");
|
||||
}
|
||||
MemberRole memberRole = (role == null) ? MemberRole.DEV : role;
|
||||
requireResumeCapability(profile, resumeSessionId);
|
||||
// CB-619 / fleetd #123: an explicit profile bypasses placement (CompositePeerLauncher only
|
||||
// constrains an UNQUALIFIED spawn to the role's pool), so it is the one path that can ask
|
||||
// for a role with no slot to bind it to. Refuse before anything spawns. A blank profile is
|
||||
// left to placement, which already restricts an unqualified spawn to the role's pool.
|
||||
if (profile != null && !profile.isBlank()) {
|
||||
memberLifecycle.requireSlotFor(memberRole, profile);
|
||||
}
|
||||
MemberLifecycle.SlotReservation reservation = memberLifecycle.reserve(memberRole, profile);
|
||||
String launchProfile = reservation == null ? profile : reservation.profile();
|
||||
if (wt == null) {
|
||||
// CB-557: the role must ride on the SpawnRequest, not stay a local. The launcher needs it
|
||||
// to pick the profile out of that role's pool and to label the tab; a role kept only on
|
||||
// the MemberSession is recorded after the spawn it was supposed to steer.
|
||||
SpawnRequest req = new SpawnRequest(profile, requestedCwd, callerCwd, sessionName, resumeSessionId, memberRole);
|
||||
SpawnRequest req = new SpawnRequest(launchProfile, requestedCwd, callerCwd, sessionName, resumeSessionId, memberRole);
|
||||
PeerHandle handle;
|
||||
boolean bound = false;
|
||||
try {
|
||||
handle = launcher.spawn(req);
|
||||
String resolvedProfile = resolveProfile(handle, launchProfile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, requestedCwd, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberRole actualRole = acquired(memberRole, resolvedProfile, handle.terminalId(), reservation);
|
||||
bound = reservation == null || actualRole == MemberRole.ARCHITECT;
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(), handle.terminalId(), resolvedProfile, actualRole, cwd, ownerTerminal, now, now, 0,
|
||||
MemberSession.State.SPAWNING, null, null, handle.charterReceipt(), handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
handles.put(handle.id(), handle);
|
||||
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("spawn failed for profile={} role={}: {}", profile, memberRole, e.getMessage());
|
||||
if (!bound) memberLifecycle.release(reservation);
|
||||
log.warn("spawn failed for profile={} role={}: {}", launchProfile, memberRole, e.getMessage());
|
||||
throw e;
|
||||
}
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, requestedCwd, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
memberRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
now,
|
||||
0,
|
||||
MemberSession.State.SPAWNING,
|
||||
null,
|
||||
null,
|
||||
handle.charterReceipt(),
|
||||
handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
handles.put(handle.id(), handle);
|
||||
memberLifecycle.acquired(session.role(), session.profile(), session.terminalId());
|
||||
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
return acquireWithWorktree(profile, memberRole, requestedCwd, callerCwd, ownerTerminal, wt,
|
||||
sessionName, resumeSessionId);
|
||||
try {
|
||||
return acquireWithWorktree(launchProfile, memberRole, requestedCwd, callerCwd, ownerTerminal, wt,
|
||||
sessionName, resumeSessionId, reservation);
|
||||
} catch (RuntimeException e) {
|
||||
memberLifecycle.release(reservation);
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -267,10 +304,32 @@ public final class SessionManager implements TurnListener {
|
||||
*/
|
||||
private void release(String paneId, ReleaseCause cause) {
|
||||
MemberSession removed = registry.remove(paneId);
|
||||
releaseRemoved(paneId, removed, handles.remove(paneId), cause);
|
||||
}
|
||||
|
||||
/**
|
||||
* Tear a session down only while {@code expected} is still its registry value. A lifecycle
|
||||
* transition replaces the immutable record, so this prevents a reap based on an old READY or
|
||||
* DONE record from stopping a worker that delivery has made BUSY.
|
||||
*/
|
||||
private boolean releaseIfCurrent(MemberSession expected, ReleaseCause cause) {
|
||||
if (!registry.remove(expected.paneId(), expected)) {
|
||||
// A lifecycle transition replaced the record between the caller's check and this remove.
|
||||
// Log it: this race is by definition unobservable otherwise, and a reaper that silently
|
||||
// declines to reap is the hardest kind of behaviour to diagnose after the fact.
|
||||
log.debug("skipping reap of pane={}: its registry record changed after the idle check "
|
||||
+ "(most likely a delivery made it BUSY)", expected.paneId());
|
||||
return false;
|
||||
}
|
||||
releaseRemoved(expected.paneId(), expected, handles.remove(expected.paneId()), cause);
|
||||
return true;
|
||||
}
|
||||
|
||||
private void releaseRemoved(String paneId, MemberSession removed, PeerHandle removedHandle,
|
||||
ReleaseCause cause) {
|
||||
// fleetd #209: remove right alongside the registry entry so a released session's handle is
|
||||
// never leaked — but keep the local reference below, so the id can still be resolved for
|
||||
// the ReleaseDetail this teardown notifies with.
|
||||
PeerHandle removedHandle = handles.remove(paneId);
|
||||
boolean preserveWorktree = cause == ReleaseCause.SHUTDOWN;
|
||||
String snapshotRef = null;
|
||||
if (removed != null) {
|
||||
@@ -329,7 +388,63 @@ public final class SessionManager implements TurnListener {
|
||||
// slot that no longer appears in the roster and can never be reclaimed.
|
||||
launcher.stop(paneId);
|
||||
if (removed != null && !preserveWorktree && removed.worktree() != null) {
|
||||
worktrees.remove(worktrees.repoRoot(removed.cwd()), removed.worktree());
|
||||
// fleetd #316: the `dirty` read above ran while the worker could still write to this
|
||||
// worktree, so a stale `false` must not be trusted to authorise the --force removal
|
||||
// below. Re-read the worktree's state one more time, right here — immediately before
|
||||
// the one step that would destroy it, and only on the path that is actually about to
|
||||
// do that (invariant 4: no second unconditional `git status` on a release that already
|
||||
// decided to preserve). By now `launcher.stop` has returned, so this read reflects
|
||||
// whatever the worker managed to write up to and including its teardown, not whatever
|
||||
// it had written at release-start time.
|
||||
if (dirtyImmediatelyBeforeRemoval(removed)) {
|
||||
preserveWorktree = true;
|
||||
// The pre-stop snapshot above never ran for this session (the pre-stop read said
|
||||
// clean), so this is the only chance to get the newly-discovered work into
|
||||
// refs/wip/* rather than leaving the on-disk preserve as the sole copy. Best-effort,
|
||||
// like every other snapshot attempt — trySnapshot logs and swallows its own failure.
|
||||
String lateSnapshotRef = trySnapshot(removed, cause);
|
||||
log.warn("release {} preserves worktree {} for pane={} terminal={}: it reported "
|
||||
+ "clean before the pane stopped but dirty immediately before removal — the "
|
||||
+ "worker wrote to it during teardown, and --force removing it now would "
|
||||
+ "have destroyed that work{}",
|
||||
cause, removed.worktree(), paneId, removed.terminalId(),
|
||||
lateSnapshotRef == null ? "" : " (snapshotted to refs/wip/" + removed.branch()
|
||||
+ " commit=" + lateSnapshotRef + ")");
|
||||
}
|
||||
}
|
||||
if (removed != null && !preserveWorktree && removed.worktree() != null) {
|
||||
// fleetd #283: this is the one cleanup step in this method that used to be bare. By the
|
||||
// time it runs, the registry entry, the retained handle, and the pane are all already
|
||||
// gone — so a throw here (a stale index lock, a slow filesystem, `remove`'s own 30s exec
|
||||
// timeout) must not escape release(): there is no retry path (a second stop on this
|
||||
// paneId is a no-op), and the caller would otherwise see a "failed stop" for a session
|
||||
// that is in fact fully torn down. Log and swallow, matching every sibling step above.
|
||||
try {
|
||||
worktrees.remove(worktrees.repoRoot(removed.cwd()), removed.worktree());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("failed to remove worktree {} for pane={} terminal={} after release: the "
|
||||
+ "pane is already stopped and the session already deregistered, so this is "
|
||||
+ "not retryable — the directory must be reclaimed manually: {}",
|
||||
removed.worktree(), paneId, removed.terminalId(), e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #316: the read that actually authorises {@code worktrees.remove}, taken with the
|
||||
* worker's pane already stopped. Fails toward preserving (returns {@code true}) on any
|
||||
* exception — the same rule the pre-stop check applies (CB-581): once we can no longer tell
|
||||
* whether the worktree is dirty, preserving costs disk while deleting on a guess can destroy
|
||||
* work that has no other copy.
|
||||
*/
|
||||
private boolean dirtyImmediatelyBeforeRemoval(MemberSession removed) {
|
||||
try {
|
||||
return worktrees.hasUncommitted(removed.worktree());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("release could not re-check worktree {} for pane={} terminal={} immediately "
|
||||
+ "before removal; preserving it rather than risk destroying unsaved work: {}",
|
||||
removed.worktree(), removed.paneId(), removed.terminalId(), e.toString());
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -468,9 +583,67 @@ public final class SessionManager implements TurnListener {
|
||||
|
||||
private MemberSession acquireWithWorktree(String profile, MemberRole memberRole, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt,
|
||||
String sessionName, String resumeSessionId) {
|
||||
String preResolvedProfile = (profile == null || profile.isBlank())
|
||||
? launcher.defaultProfile() : profile;
|
||||
String sessionName, String resumeSessionId,
|
||||
MemberLifecycle.SlotReservation reservation) {
|
||||
// fleetd #425 rework (round 2): resolved through launcher.place(memberRole) — the same
|
||||
// candidate list, quarantine/cool-off/model-off filtering, and PlacementPolicy an unqualified
|
||||
// spawn of this role is actually judged against right now — never launcher.defaultProfile()
|
||||
// (MemberRole.DEV only, wrong for any other role) and never launcher.defaultProfileFor()
|
||||
// (the role's pool FIRST entry, blind to quarantine/cool-off/model-off: a first-round fix
|
||||
// used exactly this and regressed fleetd #429's "the fleet keeps working when a model is
|
||||
// turned off" guarantee — a quarantined or model-off pool-first profile made this throw
|
||||
// instead of routing around it, which an unqualified spawn is supposed to do). This same
|
||||
// resolved decision is reused below for repoRoot, parityOverlay, AND the spawn itself so the
|
||||
// worktree is always provisioned for the profile the member actually runs on — the two could
|
||||
// disagree before fleetd #425: this name picked repoRoot/overlay, but the spawn below passed
|
||||
// the ORIGINAL (blank) profile through to placement, which re-resolves live and can pick a
|
||||
// different profile if the pool changed between the two reads, or a genuinely different one
|
||||
// under weighted/round-robin placement.
|
||||
//
|
||||
// Round 1 of this rework fed the resolved name back into launcher.spawn(SpawnRequest) as an
|
||||
// EXPLICIT profile. That was a mistake this round corrects, and the mistake is not that the
|
||||
// two branches apply different checks — they are SUPPOSED to disagree: the routing branch a
|
||||
// blank spawn takes treats a quarantined/cooling-off/at-cap/model-off/unreachable/weight-0
|
||||
// profile as a reason to fall through to the next candidate, while CompositePeerLauncher's
|
||||
// THROWING branch (enforceNotQuarantined/enforceNotCoolingOff/enforceMaxLoad/
|
||||
// enforceModelEnabled) treats naming that same profile explicitly as a reason to refuse
|
||||
// outright. That is correct: an operator who names a profile should get a refusal, not a
|
||||
// silent substitution onto a different backend. The mistake was turning a fall-through into
|
||||
// a refusal by accident — resolving a name via the routing side and then re-entering the
|
||||
// refusing side with it, for a placement the routing side had already approved by walking
|
||||
// past everything else.
|
||||
//
|
||||
// Before fleetd #435, this accident was reachable through maxLoad specifically: the default
|
||||
// `fixed` placement policy did not evaluate maxLoad at all for automatic selection, so an
|
||||
// at-cap pool-first profile that placement itself would have picked for a plain unqualified
|
||||
// spawn could die at enforceMaxLoad one call later, purely because this method's route to
|
||||
// the spawn passed through an explicit profile name — a failure a worktree-less unqualified
|
||||
// spawn never hit. fleetd #435 closed that gap (`fixed` now evaluates maxLoad exactly like
|
||||
// every other placement policy), so that specific failure can no longer happen — a
|
||||
// PlacementDecision this method resolves can no longer be at-cap in the first place. What
|
||||
// this round's fix still buys, now that maxLoad can no longer cause the accident: it keeps
|
||||
// the PlacementDecision from place() and hands it to launcher.spawn(SpawnRequest,
|
||||
// PlacementDecision) for an unqualified request, which spawns through the SAME routing
|
||||
// branch a blank spawn uses — no enforce* check is newly applied, and the window between the
|
||||
// placement decision and the spawn (in which the pool, a config reload, or another spawn
|
||||
// landing on the same profile could otherwise move the state) never reopens. An
|
||||
// explicitly-named profile still goes through launcher.spawn(SpawnRequest) and its throwing
|
||||
// branch, unchanged — that caller asked for one profile by name and still gets everything
|
||||
// enforceNotQuarantined/enforceNotCoolingOff/enforceMaxLoad/enforceModelEnabled decide about
|
||||
// it, refusal included.
|
||||
//
|
||||
// The one cost that remains, unchanged from round 1: an unqualified worktree-provisioned
|
||||
// spawn does not get CompositePeerLauncher's cross-candidate retry on a live
|
||||
// PeerUnreachableException raised by the backend itself at spawn time (a transport-level
|
||||
// failure placement cannot see in advance) — spawn(req, decision) commits to the one profile
|
||||
// place() already chose, the same way an explicit-profile spawn commits to its one name. That
|
||||
// trade is deliberate: a worktree provisioned for the wrong backend (the #425 hazard) is worse
|
||||
// than a spawn that fails cleanly and can be retried by the caller. Nothing else is lost:
|
||||
// maxLoad, quarantine, cool-off and model-off all behave identically whether or not a
|
||||
// worktree was requested — that agreement is the invariant this rework exists to hold.
|
||||
boolean unqualifiedProfile = profile == null || profile.isBlank();
|
||||
PlacementDecision decision = unqualifiedProfile ? launcher.place(memberRole) : new PlacementDecision(profile);
|
||||
String preResolvedProfile = decision.profile();
|
||||
// CB-507: resolve through the launcher's CB-112 chain (requested → profile cwd → caller →
|
||||
// daemon cwd → "."), never the raw args. A plain REST spawn supplies neither a requested
|
||||
// nor a caller cwd, so taking the first non-blank of those two yielded null and put
|
||||
@@ -493,27 +666,53 @@ public final class SessionManager implements TurnListener {
|
||||
// copies more files into the worktree after add() returns, so sharing the group any earlier
|
||||
// leaves those overlay files operator-owned and read-only for a different-uid member.
|
||||
worktrees.shareWithGroup(repoRoot, path);
|
||||
handle = launcher.spawn(new SpawnRequest(profile, path, callerCwd, sessionName, resumeSessionId, memberRole));
|
||||
// fleetd #425: preResolvedProfile, not the original (possibly blank) profile — see the
|
||||
// comment above where it is resolved. The overlay/repoRoot above and the spawn here must
|
||||
// name the same profile. An unqualified request stays unqualified here and is honored via
|
||||
// the PlacementDecision already captured above (spawn(req, decision) — the routing branch,
|
||||
// no enforce* re-check); an explicitly-named profile still goes through the single-arg
|
||||
// spawn(req) and its throwing branch, exactly as before this rework.
|
||||
SpawnRequest spawnReq = new SpawnRequest(unqualifiedProfile ? null : preResolvedProfile,
|
||||
path, callerCwd, sessionName, resumeSessionId, memberRole);
|
||||
handle = unqualifiedProfile ? launcher.spawn(spawnReq, decision) : launcher.spawn(spawnReq);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("spawn failed for profile={} role={} branch={} path={}: {}",
|
||||
preResolvedProfile, memberRole, branch, path, e.getMessage());
|
||||
if (path != null) {
|
||||
// fleetd #283: this catch covers every failure AFTER worktrees.add() returned —
|
||||
// overlayParity, shareWithGroup, launcher.spawn itself — so by this point `branch`
|
||||
// was actually created in git. #274 fixed the sibling failure INSIDE add() by having
|
||||
// GitWorktrees.cleanupAfterAddFailure delete both the worktree and the branch it
|
||||
// provisioned; this path removed only the worktree and left the branch orphaned. A
|
||||
// spawn failure here is routine (a quarantined credential, a backend refusal), so
|
||||
// every occurrence leaked a `worker/<slug>-<nonce>` branch nothing ever pointed at
|
||||
// again. Reuse the same Worktrees.deleteBranch GitWorktrees already has, rather than
|
||||
// a second copy of the git command. Best-effort and log-only, like the worktree
|
||||
// removal right above it — neither cleanup step may mask the original exception.
|
||||
try {
|
||||
worktrees.remove(repoRoot, path);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to clean up worktree {} after spawn error: {}", path, cleanup.getMessage());
|
||||
}
|
||||
try {
|
||||
worktrees.deleteBranch(repoRoot, branch);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to clean up branch {} after spawn error: {}", branch, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String resolvedProfile = resolveProfile(handle, preResolvedProfile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, path, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
// CB-619: see the no-worktree path above — bind before recording, and store the returned
|
||||
// actual role, so this session's role is never a lie about what it actually holds.
|
||||
MemberRole actualRole = acquired(memberRole, resolvedProfile, handle.terminalId(), reservation);
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
memberRole,
|
||||
actualRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
@@ -526,13 +725,24 @@ public final class SessionManager implements TurnListener {
|
||||
handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
handles.put(handle.id(), handle);
|
||||
memberLifecycle.acquired(session.role(), session.profile(), session.terminalId());
|
||||
log.debug("acquired worktree session id={} terminal={} profile={} branch={} path={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.branch(), session.worktree());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
|
||||
private MemberRole acquired(MemberRole role, String profile, String terminal,
|
||||
MemberLifecycle.SlotReservation reservation) {
|
||||
if (reservation == null) {
|
||||
return memberLifecycle.acquired(role, profile, terminal);
|
||||
}
|
||||
if (memberLifecycle.bind(reservation, terminal)) {
|
||||
return role;
|
||||
}
|
||||
memberLifecycle.release(reservation);
|
||||
return memberLifecycle.acquired(role, profile, terminal);
|
||||
}
|
||||
|
||||
private String slug(String raw) {
|
||||
return raw == null ? "ticket" : raw.toLowerCase().replaceAll("[^a-z0-9]+", "-").replaceAll("^-+|-+$", "");
|
||||
}
|
||||
@@ -638,6 +848,9 @@ public final class SessionManager implements TurnListener {
|
||||
m.put("profile", session.profile());
|
||||
m.put("role", session.role() == null ? "dev" : session.role().wireName());
|
||||
m.put("state", session.state().name().toLowerCase());
|
||||
if (session.failureReason() != null) {
|
||||
m.put("failureReason", session.failureReason());
|
||||
}
|
||||
if (session.worktree() != null) {
|
||||
m.put("worktree", session.worktree());
|
||||
}
|
||||
@@ -698,6 +911,36 @@ public final class SessionManager implements TurnListener {
|
||||
completeTurn(target, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* Record that a member backend failed after a delegated ticket expired. This CAS loop accepts
|
||||
* either side of the completion race: {@code BUSY -> BACKEND_ERROR} or
|
||||
* {@code DONE -> BACKEND_ERROR}. A released or unknown session is never recreated.
|
||||
*
|
||||
* @return {@code true} when the error is recorded on a live session
|
||||
*/
|
||||
public boolean onBackendError(String target, String reason) {
|
||||
while (true) {
|
||||
MemberSession current = findByTerminal(target);
|
||||
if (current == null) {
|
||||
log.warn("backend error for unknown member terminal={}", target);
|
||||
return false;
|
||||
}
|
||||
if (current.state() == MemberSession.State.RELEASED) {
|
||||
return false;
|
||||
}
|
||||
if (current.state() == MemberSession.State.BACKEND_ERROR) {
|
||||
return true;
|
||||
}
|
||||
MemberSession updated = current.withState(MemberSession.State.BACKEND_ERROR)
|
||||
.withFailureReason(reason).withActivity(nowNanos.getAsLong());
|
||||
if (replace(current, updated)) {
|
||||
log.warn("member terminal={} pane={} transitioned {} -> BACKEND_ERROR: {}",
|
||||
target, current.paneId(), current.state(), reason);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
if (!clearAfterTurn) return false;
|
||||
@@ -716,10 +959,9 @@ public final class SessionManager implements TurnListener {
|
||||
if (current == null || current.state() != MemberSession.State.BUSY) return false;
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession updated = current.withState(MemberSession.State.DONE).withActivity(now);
|
||||
if (replace(current, updated)) {
|
||||
log.debug("session transitioned terminal={} pane={} BUSY -> DONE turn={}",
|
||||
target, current.paneId(), updated.turnCount());
|
||||
}
|
||||
if (!replace(current, updated)) return false;
|
||||
log.debug("session transitioned terminal={} pane={} BUSY -> DONE turn={}",
|
||||
target, current.paneId(), updated.turnCount());
|
||||
if (contextCap > 0 && updated.turnCount() >= contextCap) {
|
||||
release(current.paneId());
|
||||
return false;
|
||||
@@ -766,14 +1008,18 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
long idleNanos = now - s.lastActivityAtNanos();
|
||||
if (idleNanos > idleTtlNanos) {
|
||||
log.debug("reaping idle session terminal={} pane={}: idle {}s exceeds the {}s ttl",
|
||||
s.terminalId(), s.paneId(), TimeUnit.NANOSECONDS.toSeconds(idleNanos),
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos));
|
||||
// CB-581: one session that fails to release must not abort the whole reaping pass —
|
||||
// match drainAll's per-session try/catch so the rest of the roster still gets reaped.
|
||||
try {
|
||||
release(s.paneId());
|
||||
reaped++;
|
||||
if (beforeIdleRelease != null) {
|
||||
beforeIdleRelease.accept(s);
|
||||
}
|
||||
if (releaseIfCurrent(s, ReleaseCause.COMPLETED)) {
|
||||
log.debug("reaping idle session terminal={} pane={}: idle {}s exceeds the {}s ttl",
|
||||
s.terminalId(), s.paneId(), TimeUnit.NANOSECONDS.toSeconds(idleNanos),
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos));
|
||||
reaped++;
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("reap failed for pane={} terminal={} worktree={}; continuing with "
|
||||
+ "remaining sessions", s.paneId(), s.terminalId(), s.worktree(), e);
|
||||
@@ -784,20 +1030,60 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
|
||||
/**
|
||||
* Gracefully drain all registered sessions on daemon shutdown. For each session that is
|
||||
* {@code BUSY}, poll up to {@code timeoutNanos} for it to leave {@code BUSY}, then release it
|
||||
* regardless. Non-busy sessions are released immediately. A failure releasing one session is
|
||||
* logged and does not abort the rest.
|
||||
* Gracefully drain all registered sessions on daemon shutdown. Non-busy sessions are released
|
||||
* immediately; a {@code BUSY} one is polled until it leaves {@code BUSY}, then released
|
||||
* regardless. A failure releasing one session is logged and does not abort the rest.
|
||||
*
|
||||
* <p>{@code timeoutNanos} is a budget for the WHOLE drain, not a grace period per session: the
|
||||
* deadline is taken once, before the loop. So the first BUSY session can spend all of it, and a
|
||||
* later BUSY one is then released with no wait at all. That is deliberate. This drain is only
|
||||
* one phase of shutdown — {@code Fleetd} closes the message service, the push loop, the
|
||||
* heartbeat, MCP and the router after it — and the whole sequence has to finish inside
|
||||
* launchd's exit window. A per-session grace would let N busy members drain for N * the
|
||||
* timeout, overrun that window, and get the daemon SIGKILLed part-way through; the members not
|
||||
* yet reached would then get no clean release, no preserved-worktree log, and no snapshot.
|
||||
* Cutting one turn short is the cheaper failure, and it is not silent: an abandoned BUSY
|
||||
* session is preserved, snapshotted, and logged at WARN by {@code logPreservedForShutdown}.
|
||||
*
|
||||
* <p>CB-544: this is a {@link ReleaseCause#SHUTDOWN} release — the worker's pane is stopped
|
||||
* (the process must end) but its worktree is preserved and its path logged. Shutdown is never
|
||||
* a reason to delete a worker's only copy of its uncommitted work. A session still {@code BUSY}
|
||||
* when the timeout expired is abandoned mid-turn and logged loudly so an operator can find its
|
||||
* kept worktree.
|
||||
*
|
||||
* <p>fleetd #308: {@code roster()} is a one-shot snapshot (see its javadoc), and nothing used
|
||||
* to stop a new session from registering after it was taken — {@link #acquire} stayed open for
|
||||
* as long as this drain waited on a {@code BUSY} session, up to the whole {@code timeoutNanos}
|
||||
* budget. Two things close that window, deliberately paired because neither alone is complete:
|
||||
* {@link #draining} is flipped true before the snapshot is even taken, so {@link #acquire}
|
||||
* refuses (invariant 3: loudly, via {@link ShuttingDownException}) as much of the window as a
|
||||
* plain flag can close; and the sweep below re-reads the registry once the initial snapshot has
|
||||
* fully drained and drains whatever a straggler — a caller that read the flag as {@code false}
|
||||
* a moment before it flipped — still managed to register. The sweep shares the same
|
||||
* {@code deadline} rather than getting its own: {@code timeoutNanos} is a budget for the WHOLE
|
||||
* drain (see above), and a straggler must not buy the drain more time than the flag it lost the
|
||||
* race against would have. In the ordinary case the sweep finds nothing and costs one empty
|
||||
* {@link #roster()} call.
|
||||
*/
|
||||
void drainAll(long timeoutNanos) {
|
||||
long deadline = System.nanoTime() + timeoutNanos;
|
||||
for (MemberSession s : roster()) {
|
||||
draining.set(true);
|
||||
drainSnapshot(roster(), deadline);
|
||||
List<MemberSession> stragglers = roster();
|
||||
if (!stragglers.isEmpty()) {
|
||||
log.warn("drain sweep found {} session(s) registered after the drain snapshot was "
|
||||
+ "taken (raced past the shutdown guard); draining them too", stragglers.size());
|
||||
drainSnapshot(stragglers, deadline);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Drain exactly the sessions in {@code snapshot}, waiting out a {@code BUSY} one against the
|
||||
* shared whole-drain {@code deadline} before releasing it. Shared by {@link #drainAll}'s main
|
||||
* pass and its post-loop straggler sweep (fleetd #308) so both honor the same one budget.
|
||||
*/
|
||||
private void drainSnapshot(List<MemberSession> snapshot, long deadline) {
|
||||
for (MemberSession s : snapshot) {
|
||||
try {
|
||||
if (s.state() == MemberSession.State.BUSY) {
|
||||
while (System.nanoTime() < deadline) {
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
package dev.ltms.fleet.session;
|
||||
|
||||
/**
|
||||
* Thrown by {@link SessionManager#acquire} when a spawn is requested after the daemon's shutdown
|
||||
* drain has already begun (fleetd #308).
|
||||
*
|
||||
* <p>{@link SessionManager#drainAll} snapshots the registry once and tears down exactly what is
|
||||
* in that snapshot. A session registered after the snapshot is invisible to the drain loop: its
|
||||
* pane is left running and its worktree is never preserved, and nothing else ever reclaims
|
||||
* either — the daemon's in-memory registry dies with the process. Refusing the spawn here,
|
||||
* loudly, is what stops that session from ever being created in the first place, rather than
|
||||
* silently handing the caller a session the daemon can no longer manage.
|
||||
*/
|
||||
public final class ShuttingDownException extends RuntimeException {
|
||||
public ShuttingDownException(String message) {
|
||||
super(message);
|
||||
}
|
||||
}
|
||||
@@ -11,6 +11,15 @@ public interface Worktrees {
|
||||
/** git -C <repoRoot> worktree remove --force <path>. Idempotent (already-gone tolerated). */
|
||||
void remove(String repoRoot, String worktreePath);
|
||||
|
||||
/**
|
||||
* git -C {@code repoRoot} branch -D {@code branch}. Force-deletes a branch that has no other
|
||||
* owner — used only on the failed-provisioning path (fleetd #274, #283), never on a normal
|
||||
* release: {@link SessionManager#release} deliberately leaves a released session's branch
|
||||
* behind so a lead can still recover the work, and this method must never be called from
|
||||
* that path.
|
||||
*/
|
||||
void deleteBranch(String repoRoot, String branch);
|
||||
|
||||
/**
|
||||
* True when the worktree holds uncommitted changes the bridge cannot see: tracked
|
||||
* modifications, staged files, or untracked files. {@code git status --porcelain} is the
|
||||
|
||||
@@ -0,0 +1,150 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #395: {@code exhaustedPattern} (see {@link FleetConfig.Profile#exhaustedPattern}) is
|
||||
* deliberately opt-in — an unset one leaves usage-limit detection silently OFF for that profile,
|
||||
* and nothing quarantines its credential. {@link Fleetd#reportExhaustedPatternGap} must say so at
|
||||
* startup, naming every unarmed profile, and must never fire when every profile is armed. Mirrors
|
||||
* {@link MemberCredentialsGapReportTest}'s pattern, capturing the real log via a
|
||||
* {@link ListAppender}.
|
||||
*
|
||||
* <p>The 8-profile shape in {@link #theLiveEightProfileShapeWarnsExactlyTheSixUnarmedProfiles} is
|
||||
* the live {@code fleetd.yaml} shape measured 2026-09-10 (fleetd #395's own ticket): 6 of 8
|
||||
* profiles unarmed, 2 of those 6 ({@code opus}, {@code sonnet}) running on the operator's Claude
|
||||
* subscription. {@code fleetd.yaml} itself is gitignored and unavailable to this test, so the
|
||||
* shape is reproduced as a throwaway config in a {@code @TempDir} rather than read off disk.
|
||||
*/
|
||||
class ExhaustedPatternGapReportTest {
|
||||
|
||||
private static FleetConfig load(Path dir, String yaml) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml);
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
private static ListAppender<ILoggingEvent> attach() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
||||
((Logger) LoggerFactory.getLogger(Fleetd.class)).detachAppender(appender);
|
||||
}
|
||||
|
||||
/** Every profile name mentioned by a WARN-level log line, across every WARN this call produced. */
|
||||
private static List<String> warnMessages(ListAppender<ILoggingEvent> appender) {
|
||||
return appender.list.stream()
|
||||
.filter(e -> e.getLevel() == Level.WARN)
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.toList();
|
||||
}
|
||||
|
||||
@Test
|
||||
void theLiveEightProfileShapeWarnsExactlyTheSixUnarmedProfiles(@TempDir Path dir) throws Exception {
|
||||
// Reproduces the live shape measured 2026-09-10: 8 profiles, 2 armed (sol, terra), 6
|
||||
// unarmed (local, local-direct, gx, opus, sonnet, xf) — 2 of the unarmed 6 (opus, sonnet)
|
||||
// are subscription: true.
|
||||
FleetConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
local-direct:
|
||||
baseUrl: http://gx01.gw:8000
|
||||
gx:
|
||||
kind: opencode
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
opus:
|
||||
subscription: true
|
||||
model: claude-opus-5
|
||||
sonnet:
|
||||
subscription: true
|
||||
model: claude-sonnet-5
|
||||
sol:
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
exhaustedPattern: "The usage limit has been reached"
|
||||
terra:
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
exhaustedPattern: "The usage limit has been reached"
|
||||
xf:
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
""");
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportExhaustedPatternGap(cfg);
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
List<String> warns = warnMessages(appender);
|
||||
assertFalse(warns.isEmpty(), "6 of 8 profiles are unarmed — at least one WARN must fire");
|
||||
|
||||
// Exactly one WARN aggregates every unarmed profile, naming all 6 and none of the 2 armed.
|
||||
String aggregate = warns.stream()
|
||||
.filter(m -> m.contains("no exhaustedPattern configured"))
|
||||
.findFirst()
|
||||
.orElseThrow(() -> new AssertionError("expected an aggregate unarmed-profiles WARN: " + warns));
|
||||
for (String unarmed : List.of("local", "local-direct", "gx", "opus", "sonnet", "xf")) {
|
||||
assertTrue(aggregate.contains(unarmed), "aggregate WARN must name '" + unarmed + "': " + aggregate);
|
||||
}
|
||||
for (String armed : List.of("sol", "terra")) {
|
||||
assertFalse(aggregate.contains(armed), "aggregate WARN must NOT name armed profile '" + armed + "': " + aggregate);
|
||||
}
|
||||
|
||||
// A second, louder WARN calls out the subscription profiles specifically.
|
||||
String subscriptionWarn = warns.stream()
|
||||
.filter(m -> m.contains("metered Claude plan"))
|
||||
.findFirst()
|
||||
.orElseThrow(() -> new AssertionError("expected a subscription-specific WARN: " + warns));
|
||||
assertTrue(subscriptionWarn.contains("opus"), subscriptionWarn);
|
||||
assertTrue(subscriptionWarn.contains("sonnet"), subscriptionWarn);
|
||||
assertFalse(subscriptionWarn.contains("local-direct"),
|
||||
"the subscription WARN must not name a non-subscription profile: " + subscriptionWarn);
|
||||
}
|
||||
|
||||
@Test
|
||||
void everyProfileArmedProducesNoWarningAtAll(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, """
|
||||
profiles:
|
||||
sol:
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
exhaustedPattern: "The usage limit has been reached"
|
||||
terra:
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
exhaustedPattern: "The usage limit has been reached"
|
||||
opus:
|
||||
subscription: true
|
||||
model: claude-opus-5
|
||||
exhaustedPattern: "5-hour limit reached"
|
||||
""");
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportExhaustedPatternGap(cfg);
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
assertTrue(warnMessages(appender).isEmpty(),
|
||||
"every profile is armed — a checker that warns anyway always fires: " + warnMessages(appender));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,212 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
import static org.junit.jupiter.api.Assertions.fail;
|
||||
|
||||
/**
|
||||
* fleetd #498: {@code Fleetd.awaitHerdr} used to return a bare {@code boolean}, collapsing "the
|
||||
* configured wait budget genuinely ran out" and "the waiting thread was interrupted, possibly
|
||||
* milliseconds in" onto the same {@code false} — and the caller's log line printed only the
|
||||
* configured budget, never how long the wait actually ran. This class covers both halves of the
|
||||
* fix:
|
||||
* <ul>
|
||||
* <li>the seam — {@link Fleetd#awaitHerdr} itself, driven with an injected clock and a stub
|
||||
* {@link HerdrClient}, one test per {@link Fleetd.HerdrWaitResult};</li>
|
||||
* <li>the call site — {@link Fleetd#logHerdrWaitOutcomeAndShouldReap}, the exact decision {@code
|
||||
* main} calls (extracted here because {@code main} itself boots the whole daemon and cannot
|
||||
* be driven from a unit test), pinning the three distinct log messages it emits.</li>
|
||||
* </ul>
|
||||
* Every expected message below is a plain literal, not built from {@code HERDR_WAIT_SECONDS} or
|
||||
* any other production constant — a test that derives its expectation the way the code does
|
||||
* cannot see a change to either (fleetd #496's identical trap).
|
||||
*/
|
||||
class FleetdAwaitHerdrTest {
|
||||
|
||||
// ---- the seam: Fleetd.awaitHerdr ----------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void answeredReturnsImmediatelyWithZeroElapsedAndNeverPolls() {
|
||||
HerdrStub herdr = new HerdrStub(0); // succeeds on the very first call
|
||||
LongSupplier clock = fixedClock(1_000L);
|
||||
AtomicBoolean polled = new AtomicBoolean(false);
|
||||
Runnable poller = () -> polled.set(true);
|
||||
|
||||
Fleetd.HerdrAwaitOutcome outcome = Fleetd.awaitHerdr(herdr, clock, poller);
|
||||
|
||||
assertEquals(Fleetd.HerdrWaitResult.ANSWERED, outcome.result());
|
||||
assertEquals(0L, outcome.elapsedNanos(), "a fixed clock must measure zero elapsed time");
|
||||
assertFalse(polled.get(), "herdr answering on the first try must never poll");
|
||||
}
|
||||
|
||||
@Test
|
||||
void deadlinePassedIsMeasuredNotAssumed() {
|
||||
HerdrStub herdr = new HerdrStub(-1); // never succeeds
|
||||
// call order inside awaitHerdr: start, then per failed attempt: deadline-check, elapsed-calc
|
||||
ScriptedClock clock = new ScriptedClock(0L, 30_500_000_000L, 30_500_000_000L);
|
||||
Runnable poller = () -> fail("the deadline was already exceeded on the first attempt — must not poll");
|
||||
|
||||
Fleetd.HerdrAwaitOutcome outcome = Fleetd.awaitHerdr(herdr, clock, poller);
|
||||
|
||||
assertEquals(Fleetd.HerdrWaitResult.DEADLINE_PASSED, outcome.result());
|
||||
assertEquals(30_500_000_000L, outcome.elapsedNanos(),
|
||||
"elapsed must be the MEASURED clock delta, not the configured budget");
|
||||
}
|
||||
|
||||
@Test
|
||||
void interruptedIsDistinctFromDeadlinePassedAndPreservesTheInterruptFlag() {
|
||||
HerdrStub herdr = new HerdrStub(-1); // never succeeds
|
||||
// start=0, deadline-check returns 500ms (well under the 30s budget) -> not deadline-passed,
|
||||
// then the poller interrupts, and the elapsed-calc call returns 750ms.
|
||||
ScriptedClock clock = new ScriptedClock(0L, 500_000_000L, 750_000_000L);
|
||||
Runnable poller = () -> Thread.currentThread().interrupt();
|
||||
|
||||
try {
|
||||
Fleetd.HerdrAwaitOutcome outcome = Fleetd.awaitHerdr(herdr, clock, poller);
|
||||
|
||||
assertEquals(Fleetd.HerdrWaitResult.INTERRUPTED, outcome.result());
|
||||
assertEquals(750_000_000L, outcome.elapsedNanos(),
|
||||
"elapsed must be measured even when the wait ends via interruption, not the deadline");
|
||||
assertTrue(Thread.currentThread().isInterrupted(),
|
||||
"the interrupt flag the old code re-set must still be set on return");
|
||||
} finally {
|
||||
Thread.interrupted(); // clear it so it cannot leak into another test on this thread
|
||||
}
|
||||
}
|
||||
|
||||
// ---- the call site: Fleetd.logHerdrWaitOutcomeAndShouldReap -------------------------------
|
||||
|
||||
@Test
|
||||
void answeredLogsNothingAndSaysReap() {
|
||||
ListAppender<ILoggingEvent> events = attach();
|
||||
try {
|
||||
boolean shouldReap = Fleetd.logHerdrWaitOutcomeAndShouldReap(
|
||||
new Fleetd.HerdrAwaitOutcome(Fleetd.HerdrWaitResult.ANSWERED, 0L));
|
||||
|
||||
assertTrue(shouldReap, "only ANSWERED should tell main to reap orphan workers");
|
||||
assertEquals(0, events.list.size(), "the answered path logs nothing itself");
|
||||
} finally {
|
||||
detach(events);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void deadlinePassedLogsConfiguredAndMeasuredElapsedTogether() {
|
||||
ListAppender<ILoggingEvent> events = attach();
|
||||
try {
|
||||
boolean shouldReap = Fleetd.logHerdrWaitOutcomeAndShouldReap(
|
||||
new Fleetd.HerdrAwaitOutcome(Fleetd.HerdrWaitResult.DEADLINE_PASSED, 30_500_000_000L));
|
||||
|
||||
assertFalse(shouldReap, "a deadline-passed wait must not tell main to reap");
|
||||
assertEquals(1, events.list.size());
|
||||
ILoggingEvent event = events.list.getFirst();
|
||||
assertEquals(Level.WARN, event.getLevel());
|
||||
assertEquals("herdr did not answer within the configured wait (configured=30s "
|
||||
+ "elapsed=30500ms) — starting anyway; /healthz will report degraded until it "
|
||||
+ "comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
event.getFormattedMessage());
|
||||
} finally {
|
||||
detach(events);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void interruptedLogsItsOwnMessageAndNeverClaimsTheBudgetElapsed() {
|
||||
ListAppender<ILoggingEvent> events = attach();
|
||||
try {
|
||||
// 3ms: the ticket's own example of "a few milliseconds in", not the 30s budget.
|
||||
boolean shouldReap = Fleetd.logHerdrWaitOutcomeAndShouldReap(
|
||||
new Fleetd.HerdrAwaitOutcome(Fleetd.HerdrWaitResult.INTERRUPTED, 3_000_000L));
|
||||
|
||||
assertFalse(shouldReap, "an interrupted wait must not tell main to reap");
|
||||
assertEquals(1, events.list.size());
|
||||
ILoggingEvent event = events.list.getFirst();
|
||||
assertEquals(Level.WARN, event.getLevel());
|
||||
String message = event.getFormattedMessage();
|
||||
assertEquals("herdr wait was interrupted before the configured wait ran out "
|
||||
+ "(configured=30s elapsed=3ms) — starting anyway; /healthz will report "
|
||||
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
message);
|
||||
assertFalse(message.contains("did not answer"),
|
||||
"an interrupted wait must not be reported as if herdr failed to answer within the budget");
|
||||
} finally {
|
||||
detach(events);
|
||||
}
|
||||
}
|
||||
|
||||
// ---- fixtures --------------------------------------------------------------------------
|
||||
|
||||
/** Always returns the same value, i.e. a clock that measures zero elapsed time. */
|
||||
private static LongSupplier fixedClock(long value) {
|
||||
return () -> value;
|
||||
}
|
||||
|
||||
/** Returns each value in order, then repeats the last one for any call beyond the list. */
|
||||
private static final class ScriptedClock implements LongSupplier {
|
||||
private final long[] values;
|
||||
private int index;
|
||||
|
||||
ScriptedClock(long... values) {
|
||||
this.values = values;
|
||||
}
|
||||
|
||||
@Override
|
||||
public long getAsLong() {
|
||||
long v = values[Math.min(index, values.length - 1)];
|
||||
if (index < values.length - 1) {
|
||||
index++;
|
||||
}
|
||||
return v;
|
||||
}
|
||||
}
|
||||
|
||||
/** Fails {@code failuresBeforeSuccess} times, then succeeds forever; {@code -1} never succeeds. */
|
||||
private static final class HerdrStub implements HerdrClient {
|
||||
private final int failuresBeforeSuccess;
|
||||
private int calls;
|
||||
|
||||
HerdrStub(int failuresBeforeSuccess) {
|
||||
this.failuresBeforeSuccess = failuresBeforeSuccess;
|
||||
}
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) throws HerdrException {
|
||||
calls++;
|
||||
if (failuresBeforeSuccess < 0 || calls <= failuresBeforeSuccess) {
|
||||
throw new HerdrException("herdr not up yet");
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
private static ListAppender<ILoggingEvent> attach() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
logger.setLevel(Level.DEBUG);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
||||
((Logger) LoggerFactory.getLogger(Fleetd.class)).detachAppender(appender);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.inject.BackendErrorPatternLookup;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
|
||||
/**
|
||||
* fleetd #248 / fleetd#201 Unit 5: {@link Fleetd#backendErrorPatternLookup} is the factory that
|
||||
* replaced the local lambda {@code Fleetd.main} used to build {@code backendErrorPatterns} — one
|
||||
* of the two arguments {@code CompletionResolver} lost cleanly (0 compile errors, every test still
|
||||
* green) when this ticket's measurement dropped it alongside {@code backendErrorSink}. This class
|
||||
* proves the factory's own behaviour; {@code FleetdCompletionResolverWiringTest} proves {@code
|
||||
* main} still passes its result into {@code CompletionResolver}.
|
||||
*/
|
||||
class FleetdBackendErrorPatternLookupTest {
|
||||
|
||||
private static MemberSession session(String terminal, String profile) {
|
||||
return new MemberSession("pane-" + terminal, terminal, profile, MemberRole.DEV,
|
||||
"/cwd", null, 0L, 0L, 0, MemberSession.State.READY, null, null);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a target on a profile with a configured pattern resolves to that pattern")
|
||||
void configuredProfileResolves() {
|
||||
Map<String, Pattern> byProfile = Map.of("terra", Pattern.compile("(?i)503"));
|
||||
BackendErrorPatternLookup lookup =
|
||||
Fleetd.backendErrorPatternLookup(() -> List.of(session("term1", "terra")), byProfile);
|
||||
|
||||
assertEquals("(?i)503", lookup.patternFor("term1").pattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a target on a profile with no configured pattern resolves to null")
|
||||
void unconfiguredProfileResolvesToNull() {
|
||||
Map<String, Pattern> byProfile = Map.of("terra", Pattern.compile("x"));
|
||||
BackendErrorPatternLookup lookup =
|
||||
Fleetd.backendErrorPatternLookup(() -> List.of(session("term1", "sol")), byProfile);
|
||||
|
||||
assertNull(lookup.patternFor("term1"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an unknown target resolves to null")
|
||||
void unknownTargetResolvesToNull() {
|
||||
BackendErrorPatternLookup lookup =
|
||||
Fleetd.backendErrorPatternLookup(List::of, Map.of("terra", Pattern.compile("x")));
|
||||
|
||||
assertNull(lookup.patternFor("term_stranger"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,275 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.BackendErrorSink;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementDecision;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #248 / fleetd#201 Unit 5: {@link Fleetd#backendErrorSink} is the factory that replaced
|
||||
* the local lambda {@code Fleetd.main} used to build {@code backendErrorSink} — the other half of
|
||||
* the pair this ticket's measurement dropped cleanly (0 compile errors, every test still green).
|
||||
*
|
||||
* <p>Before this ticket, the closest thing to coverage was {@code
|
||||
* dev.ltms.fleet.inject.BackendOutageFlowTest}, whose own class doc said it "mirrors {@code
|
||||
* Fleetd.main}'s {@code backendErrorSink} lambda line-for-line" — a hand-copy that proves itself,
|
||||
* never that {@code main} still wires the real thing. This class exercises the actual production
|
||||
* factory instead. {@code FleetdCompletionResolverWiringTest} proves {@code main} still passes its
|
||||
* result into {@code CompletionResolver}.
|
||||
*/
|
||||
class FleetdBackendErrorSinkTest {
|
||||
|
||||
private final List<ScheduledExecutorService> schedulers = new ArrayList<>();
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
schedulers.forEach(ScheduledExecutorService::shutdownNow);
|
||||
}
|
||||
|
||||
private static FleetConfig.Profile stubWorker(String profile, String credentialId) {
|
||||
return new FleetConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, null, null,
|
||||
null, null, credentialId, null);
|
||||
}
|
||||
|
||||
private static Map<String, FleetConfig.Profile> orderedProfiles() {
|
||||
Map<String, FleetConfig.Profile> m = new LinkedHashMap<>();
|
||||
m.put("terra", stubWorker("terra", "shared-openai"));
|
||||
m.put("sol", stubWorker("sol", "shared-openai"));
|
||||
return m;
|
||||
}
|
||||
|
||||
/** Minimal recording {@code HerdrClient} for the LEAD pane — mirrors ReplyPushLoopTest's own. */
|
||||
private static final class RecordingLeadClient implements HerdrClient {
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
private final List<Object> prompts = new CopyOnWriteArrayList<>();
|
||||
volatile CountDownLatch sendLatch = new CountDownLatch(1);
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
if ("agent.get".equals(method)) {
|
||||
return MAPPER.createObjectNode().set("agent", MAPPER.createObjectNode()
|
||||
.put("terminal_id", "term_primary").put("agent_status", "idle"));
|
||||
}
|
||||
if ("agent.prompt".equals(method)) {
|
||||
prompts.add(params);
|
||||
sendLatch.countDown();
|
||||
}
|
||||
return MAPPER.createObjectNode();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
|
||||
int sendCount() {
|
||||
return prompts.size();
|
||||
}
|
||||
}
|
||||
|
||||
/** A {@link PeerLauncher} that never actually spawns — enough to construct a bare {@link SessionManager}. */
|
||||
private static final class NeverSpawnsLauncher implements PeerLauncher {
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
return Set.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilitiesFor(String profileName) {
|
||||
return Set.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
throw new UnsupportedOperationException("not reachable — this test never acquires a session");
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req, PlacementDecision decision) {
|
||||
throw new UnsupportedOperationException("not reachable — this test never acquires a session");
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return Set.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
throw new UnsupportedOperationException("not reachable — this test never acquires a session");
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<?> list() {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a target with no resolvable profile logs and returns without recording an incident (never throws)")
|
||||
void unresolvableProfileDoesNotRecordOrThrow() {
|
||||
SessionManager sessions = new SessionManager(new NeverSpawnsLauncher());
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
RecordingLeadClient leadClient = new RecordingLeadClient();
|
||||
InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||
PrimaryRegistry registry = new PrimaryRegistry(null);
|
||||
ScheduledExecutorService scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
schedulers.add(scheduler);
|
||||
ReplyPushLoop pushLoop = new ReplyPushLoop(registry, new AgentControl(leadClient), inbox, scheduler, 3, 50);
|
||||
|
||||
BackendErrorSink sink = Fleetd.backendErrorSink(sessions, Map::of, outagePolicy, () -> pushLoop);
|
||||
sink.onBackendError("term_unmapped", "matched line", "503 Service Unavailable");
|
||||
|
||||
assertTrue(outagePolicy.remainingCoolOffSeconds("shared-openai").isEmpty(),
|
||||
"no credential is ever resolvable here, so nothing must be recorded");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("two distinct targets classified through the real factory start an incident and cool the credential")
|
||||
void twoDistinctTargetsStartAnIncident() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.readText("⏺ 503 Service Unavailable: upstream credential rejected\n❯ ");
|
||||
Map<String, FleetConfig.Profile> profiles = orderedProfiles();
|
||||
AtomicLong clockNanos = new AtomicLong(0L);
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(clockNanos::get);
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr),
|
||||
new WorkspaceControl(herdr), new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
profiles, "terra", _ -> "tok");
|
||||
CompositePeerLauncher workers = new CompositePeerLauncher(List.of(adapter), "terra", profiles,
|
||||
PlacementPolicies.weighted(), _ -> 0, null, BackendQuarantine.none(), outagePolicy);
|
||||
SessionManager sessions = new SessionManager(workers);
|
||||
MemberSession s1 = sessions.acquire("terra", null, null, null);
|
||||
MemberSession s2 = sessions.acquire("terra", null, null, null);
|
||||
|
||||
PrimaryRegistry registry = new PrimaryRegistry(null);
|
||||
registry.recordDelegation(s1.terminalId(), "term_primary");
|
||||
registry.recordDelegation(s2.terminalId(), "term_primary");
|
||||
InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||
inbox.own(s1.terminalId());
|
||||
inbox.own(s2.terminalId());
|
||||
RecordingLeadClient leadClient = new RecordingLeadClient();
|
||||
ScheduledExecutorService scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
schedulers.add(scheduler);
|
||||
ReplyPushLoop pushLoop = new ReplyPushLoop(registry, new AgentControl(leadClient), inbox, scheduler, 3, 50);
|
||||
AtomicReference<ReplyPushLoop> pushLoopRef = new AtomicReference<>(pushLoop);
|
||||
|
||||
// The exact object under test: Fleetd's real production factory, not a hand copy.
|
||||
BackendErrorSink sink = Fleetd.backendErrorSink(sessions, () -> profiles, outagePolicy, pushLoopRef::get);
|
||||
|
||||
sink.onBackendError(s1.terminalId(), "matched line", "503 Service Unavailable");
|
||||
assertTrue(outagePolicy.remainingCoolOffSeconds("shared-openai").isEmpty(),
|
||||
"one distinct target must not start a cool-off");
|
||||
|
||||
sink.onBackendError(s2.terminalId(), "matched line", "503 Service Unavailable");
|
||||
|
||||
assertTrue(leadClient.sendLatch.await(3, TimeUnit.SECONDS),
|
||||
"the second distinct target must cross the threshold and nudge the lead");
|
||||
var remaining = outagePolicy.remainingCoolOffSeconds("shared-openai");
|
||||
assertTrue(remaining.isPresent(), "two distinct targets must start a cool-off");
|
||||
assertEquals(1, leadClient.sendCount());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a backend-error session no longer blocks the real maxLoad spawn gate")
|
||||
void backendErrorSessionDoesNotBlockFreshSpawnAtMaxLoad() {
|
||||
SessionManager sessions = capacityLimitedSessions();
|
||||
MemberSession failed = sessions.acquire("terra", null, null, null);
|
||||
|
||||
assertTrue(sessions.onBackendError(failed.terminalId(), "backend exited"));
|
||||
|
||||
MemberSession fresh = sessions.acquire("terra", null, null, null);
|
||||
assertEquals("terra", fresh.profile(), "the real maxLoad gate grants a fresh spawn after a backend error");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a failed session no longer blocks the real maxLoad spawn gate")
|
||||
void failedSessionDoesNotBlockFreshSpawnAtMaxLoad() {
|
||||
SessionManager sessions = capacityLimitedSessions();
|
||||
MemberSession failed = sessions.acquire("terra", null, null, null);
|
||||
|
||||
sessions.onTurnFailed(failed.terminalId());
|
||||
|
||||
MemberSession fresh = sessions.acquire("terra", null, null, null);
|
||||
assertEquals("terra", fresh.profile(), "the real maxLoad gate grants a fresh spawn after a failed turn");
|
||||
}
|
||||
|
||||
private static SessionManager capacityLimitedSessions() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile profile = new FleetConfig.Profile("terra", "http://gx00.gw:8000", "coder",
|
||||
null, "FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, 1.0f, 1);
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("terra", profile);
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), profiles, "terra", _ -> "tok");
|
||||
AtomicReference<SessionManager> sessionsRef = new AtomicReference<>();
|
||||
CompositePeerLauncher workers = new CompositePeerLauncher(List.of(adapter), "terra", profiles,
|
||||
PlacementPolicies.fixed(), name -> Fleetd.liveSessionCount(sessionsRef.get().roster(), name));
|
||||
SessionManager sessions = new SessionManager(workers);
|
||||
sessionsRef.set(sessions);
|
||||
return sessions;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #466 follow-up: {@code Fleetd.main} builds the daemon's one {@code BackendQuarantine}
|
||||
* from {@link dev.ltms.fleet.placement.BackendQuarantine#withEscalation(java.util.function.LongSupplier,
|
||||
* long)} — the escalating factory — rather than the plain two-argument constructor, which is still a
|
||||
* flat cooldown (kept for backward compatibility, see that class's doc). {@code
|
||||
* BackendQuarantineTest} proves {@code withEscalation} itself escalates, is ceilinged, and resets;
|
||||
* it says nothing about which one {@code main} actually calls.
|
||||
*
|
||||
* <p>Measured directly: reverting {@code main} to {@code new BackendQuarantine(System::nanoTime,
|
||||
* TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()))} — the pre-#466 flat call — compiles
|
||||
* with 0 errors and leaves the entire 1608-test suite (including every {@code BackendQuarantineTest}
|
||||
* case) green, because no other test constructs its {@code BackendQuarantine} through {@code main};
|
||||
* every one of them builds its own instance directly. That silent regression is exactly the shape
|
||||
* {@link FleetdLeadSeatWiringTest} and {@link FleetdCompletionResolverWiringTest} already guard
|
||||
* against for their own constructor arguments — this is the same class of gap for fleetd #466's
|
||||
* factory choice, following their approach.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||
* BackendQuarantine} and never runs {@code main} — a green result here proves only that the exact
|
||||
* text {@code main} calls {@code BackendQuarantine.withEscalation(...)} rather than the flat
|
||||
* constructor. It does not prove that call actually executes at startup (no test here starts the
|
||||
* daemon), and it does not prove the escalation reaches a real backend or credential — only
|
||||
* {@code BackendQuarantineTest} proves the factory's own behaviour, and only a live daemon proves
|
||||
* the wiring runs.
|
||||
*/
|
||||
class FleetdBackendQuarantineWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] main's BackendQuarantine local is still built from BackendQuarantine.withEscalation(...)")
|
||||
void mainStillWiresTheEscalatingQuarantineFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"BackendQuarantine quarantine = BackendQuarantine.withEscalation(System::nanoTime,\n"
|
||||
+ " TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));"),
|
||||
"Fleetd.main's BackendQuarantine local must still be built from "
|
||||
+ "BackendQuarantine.withEscalation(System::nanoTime, "
|
||||
+ "TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds())). Reverting to the flat "
|
||||
+ "two-argument constructor (fleetd #466's measured regression) compiles with 0 errors "
|
||||
+ "and leaves the whole suite green, including every BackendQuarantineTest case that "
|
||||
+ "proves the escalation itself works — this source check is what must go red instead. "
|
||||
+ "A reverted daemon would go back to retrying a weekly subscription limit on every "
|
||||
+ "flat ~30-minute cooldown, about 336 times across the week.");
|
||||
|
||||
// Negative form of the same check: the pre-#466 flat call, if it ever reappears at this
|
||||
// declaration, must not be mistaken for the escalating one by a looser positive-only check.
|
||||
assertFalse(source.contains(
|
||||
"BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,\n"
|
||||
+ " TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));"),
|
||||
"main's BackendQuarantine local must never regress to the flat two-argument constructor");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* fleetd #416: {@code fleet_list}'s {@code CapacitySource.configuredProfiles} must enumerate the
|
||||
* <em>startup</em> profile set, not the live, hot-reloaded one.
|
||||
*
|
||||
* <p>{@code profiles} as a whole is a {@code DEFERRED} key ({@link ConfigRef#DEFERRED_KEYS}):
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} once at construction, so a profile
|
||||
* only added to the hot-reloaded map can never actually be spawned. Before this fix, {@code
|
||||
* Fleetd.main} built {@code CapacitySource} with {@code () -> config.get().profiles().keySet()} —
|
||||
* the live map — so {@code fleet_list} would report a freshly hot-reloaded profile as available
|
||||
* ({@code free > 0}) while {@code fleet_spawn} on that same profile failed with
|
||||
* {@code unknown worker profile}. Measured on another host: adding a throwaway profile and letting
|
||||
* it hot-reload gave {@code fleet_list} -> {@code free: 3} and {@code fleet_spawn} ->
|
||||
* {@code error: unknown worker profile}.
|
||||
*
|
||||
* <p>This test needs a reload, the same reason {@link FleetdExhaustionDetectionArmedWiringTest}
|
||||
* does: at startup the two snapshots agree, so a test of only a newly started daemon would not
|
||||
* detect a live {@code config.get()} lookup for the set.
|
||||
*
|
||||
* <p><b>maxLoad must stay hot.</b> It is read off {@code config.get()} exactly like
|
||||
* {@code credentialId} ({@link ConfigRef} documents both as hot, "read live off the config
|
||||
* supplier ... exactly like weight/maxLoad"), so a reload that only changes an existing profile's
|
||||
* {@code maxLoad} — no add/remove — must still change what {@code fleet_list} reports without a
|
||||
* restart. A fix that freezes the whole {@code CapacitySource} against {@code cfg} (rather than
|
||||
* only its {@code configuredProfiles} set) would trade this bug for its mirror image and is pinned
|
||||
* wrong by {@link #reloadedMaxLoadStillChangesWhatFleetListReports}.
|
||||
*/
|
||||
class FleetdCapacitySourceWiringTest {
|
||||
|
||||
private static final String STARTUP = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: terra
|
||||
maxLoad: 3
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
private static final String WITH_NEW_PROFILE = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: terra
|
||||
maxLoad: 3
|
||||
ghost404:
|
||||
baseUrl: http://gx00.gw:8001
|
||||
model: ghost404
|
||||
maxLoad: 3
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
private static final String WITH_CHANGED_MAX_LOAD = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: terra
|
||||
maxLoad: 9
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
@Test
|
||||
@DisplayName("a profile present only in the live (hot-reloaded) config is NOT listed")
|
||||
void liveOnlyProfileIsNotListed(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, STARTUP);
|
||||
FleetConfig cfg = FleetConfig.load(file);
|
||||
ConfigRef config = new ConfigRef(file, cfg);
|
||||
|
||||
Files.writeString(file, WITH_NEW_PROFILE);
|
||||
assertTrue(config.reload().applied());
|
||||
// The live snapshot now has the new profile — proves the reload really happened and this
|
||||
// test is not accidentally passing because nothing changed.
|
||||
assertTrue(config.get().profiles().containsKey("ghost404"));
|
||||
|
||||
FleetMcp.CapacitySource source = Fleetd.capacitySource(config, cfg, _ -> 0);
|
||||
|
||||
assertFalse(source.configuredProfiles().get().contains("ghost404"),
|
||||
"a profile added only to the hot-reloaded config must not be listed by fleet_list — "
|
||||
+ "HerdrPeerLauncher never learns about it until a restart, so fleet_spawn on it "
|
||||
+ "would fail with 'unknown worker profile' while fleet_list claimed it free");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a profile present in the startup set IS listed")
|
||||
void startupProfileIsListed(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, STARTUP);
|
||||
FleetConfig cfg = FleetConfig.load(file);
|
||||
ConfigRef config = new ConfigRef(file, cfg);
|
||||
|
||||
FleetMcp.CapacitySource source = Fleetd.capacitySource(config, cfg, _ -> 0);
|
||||
|
||||
// fleetd #416, both-directions requirement: a test that only ever passes an empty/absent
|
||||
// startup set (the case above) cannot tell a correct lookup from one that is permanently
|
||||
// empty (e.g. a mutation replacing the supplier with Set::of). This is the direction that
|
||||
// fails if the fix regresses to reporting nothing at all.
|
||||
assertTrue(source.configuredProfiles().get().contains("terra"),
|
||||
"a profile present in the startup snapshot must still be listed by fleet_list");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a hot maxLoad edit still changes what fleet_list reports")
|
||||
void reloadedMaxLoadStillChangesWhatFleetListReports(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, STARTUP);
|
||||
FleetConfig cfg = FleetConfig.load(file);
|
||||
ConfigRef config = new ConfigRef(file, cfg);
|
||||
|
||||
FleetMcp.CapacitySource source = Fleetd.capacitySource(config, cfg, _ -> 0);
|
||||
assertEquals(3, source.maxLoad().apply("terra"),
|
||||
"sanity: maxLoad reads 3 from the startup config before any reload");
|
||||
|
||||
Files.writeString(file, WITH_CHANGED_MAX_LOAD);
|
||||
assertTrue(config.reload().applied());
|
||||
|
||||
assertEquals(9, source.maxLoad().apply("terra"),
|
||||
"maxLoad must stay hot — the SAME CapacitySource instance must reflect a reloaded "
|
||||
+ "maxLoad without a restart, exactly like credentialId. Freezing the whole "
|
||||
+ "CapacitySource against the startup snapshot (rather than only its "
|
||||
+ "configuredProfiles set) would trade fleetd #416 for its mirror image.");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #248: this is the test that was actually missing. {@code Fleetd.main} builds its {@code
|
||||
* CompletionResolver} from an 8-argument constructor, and the ticket's own measurement proved two
|
||||
* ways to silently unwire it — both compiled with 0 errors and left every existing test green:
|
||||
*
|
||||
* <ul>
|
||||
* <li>replacing the worktree/branch argument (the 8th) with {@code _ -> null} — drops
|
||||
* fleetd#241's fallback-report location entirely;</li>
|
||||
* <li>replacing {@code backendErrorPatterns, backendErrorSink} (5th/6th) with {@code
|
||||
* BackendErrorPatternLookup.legacy(), BackendErrorSink.none()} — drops fleetd#201 Unit 5's
|
||||
* backend-error classification and cool-off entirely.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Neither mutation could be caught by any test that constructs its own {@code
|
||||
* CompletionResolver} (every test before this one did exactly that) or by a test of {@link
|
||||
* Fleetd#worktreeBranchLookup}, {@link Fleetd#backendErrorPatternLookup}, or {@link
|
||||
* Fleetd#backendErrorSink} in isolation (see {@code FleetdWorktreeBranchLookupTest}, {@code
|
||||
* FleetdBackendErrorPatternLookupTest}, {@code FleetdBackendErrorSinkTest}) — those prove the
|
||||
* factories work, never that {@code main} still calls them. This class is a plain source-text
|
||||
* assertion on {@code Fleetd.java} — crude, but honest about what it checks, and it turns red the
|
||||
* instant the wiring is dropped, mirroring the same fallback shape {@link
|
||||
* FleetdFleetAppConstructionTest} already uses for a different constructor argument.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||
* CompletionResolver} and never runs {@code main}.
|
||||
*/
|
||||
class FleetdCompletionResolverWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] CompletionResolver's construction call still names backendErrorPatterns and backendErrorSink")
|
||||
void backendErrorArgumentsAreStillNamedAtTheCallSite() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"exhaustionSink, backendErrorPatterns, backendErrorSink, System::nanoTime,"),
|
||||
"CompletionResolver's construction call must still pass backendErrorPatterns and "
|
||||
+ "backendErrorSink as its 5th/6th arguments. Replacing them with "
|
||||
+ "BackendErrorPatternLookup.legacy()/BackendErrorSink.none() (fleetd #248's measured "
|
||||
+ "mutation) compiles with 0 errors and leaves every behavioural test green — this "
|
||||
+ "source check is what must go red instead.");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] CompletionResolver's construction call still passes worktreeBranchLookup(sessions::roster)")
|
||||
void worktreeBranchLookupIsStillPassedAtTheCallSite() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains("worktreeBranchLookup(sessions::roster)"),
|
||||
"CompletionResolver's construction call must still pass worktreeBranchLookup(sessions::roster) "
|
||||
+ "as its 8th (last) argument. Replacing it with the inert `_ -> null` (fleetd #248's "
|
||||
+ "other measured mutation) compiles with 0 errors and leaves every behavioural test "
|
||||
+ "green — this source check is what must go red instead.");
|
||||
assertFalse(source.contains("System::nanoTime,\n _ -> null"),
|
||||
"the worktree/branch argument must never regress to the inert `_ -> null` literal");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] backendErrorPatterns is assigned from the extracted backendErrorPatternLookup(...) factory")
|
||||
void backendErrorPatternsComesFromTheFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"BackendErrorPatternLookup backendErrorPatterns = backendErrorPatternLookup(sessions::roster,"),
|
||||
"backendErrorPatterns must be assigned from Fleetd.backendErrorPatternLookup(...), not an "
|
||||
+ "inline lambda that a source check on the CompletionResolver call alone cannot see "
|
||||
+ "through");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] backendErrorSink is assigned from the extracted backendErrorSink(...) factory")
|
||||
void backendErrorSinkComesFromTheFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"BackendErrorSink backendErrorSink = backendErrorSink(sessions, () -> config.get().profiles(),"),
|
||||
"backendErrorSink must be assigned from Fleetd.backendErrorSink(...), not an inline lambda "
|
||||
+ "that a source check on the CompletionResolver call alone cannot see through");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,106 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertSame;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #474: proves the exact wiring {@code Fleetd.main} uses to construct its live {@code
|
||||
* ConfigRef} — {@code new ConfigRef(configPath, cfg, Fleetd::assertChartersNameOnlyRegisteredTools)}
|
||||
* — actually makes {@link ConfigRef#reload()} refuse a charter that names an MCP tool the server
|
||||
* does not register, the same way {@code Fleetd.main} itself refuses one at startup (see {@code
|
||||
* FleetdStartupValidationTest#mainRefusesACharterNamingAnUnregisteredTool}).
|
||||
*
|
||||
* <p>{@code dev.ltms.fleet.config.ConfigRefTest} pins the same behaviour through a locally-built
|
||||
* {@code Consumer<FleetConfig>} adapter that calls the same production {@code CharterToolSurface}
|
||||
* method, because that test lives in {@code dev.ltms.fleet.config} and cannot see {@code
|
||||
* Fleetd#assertChartersNameOnlyRegisteredTools} (package-private to {@code dev.ltms.fleet}). This
|
||||
* class is the companion proof that lives where the real method reference is visible, so the literal
|
||||
* expression {@code Fleetd::assertChartersNameOnlyRegisteredTools} — not just an equivalent — is
|
||||
* what gets exercised. {@code Fleetd.main} itself cannot be driven this far in a unit test: every
|
||||
* fixture in {@code FleetdStartupValidationTest} is deliberately invalid so {@code main} throws
|
||||
* before opening a socket, binding Javalin, or doing anything else with a real side effect, so a
|
||||
* test cannot get {@code main} far enough to hold a running daemon it could then reload — this test
|
||||
* builds the {@code ConfigRef} the same way {@code main} does and drives {@link ConfigRef#reload()}
|
||||
* directly instead, the same shape {@code FleetdExhaustionDetectionArmedWiringTest} and its
|
||||
* siblings already use for the rest of {@code Fleetd.main}'s wiring.
|
||||
*/
|
||||
class FleetdConfigRefCharterToolSurfaceWiringTest {
|
||||
|
||||
private static final String BASE = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
@Test
|
||||
void reloadRefusesACharterNamingAnUnregisteredToolThroughFleetdsOwnWiring(@TempDir Path dir)
|
||||
throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, BASE + """
|
||||
fleet:
|
||||
charters:
|
||||
dev: |
|
||||
Send the final handoff through fleet_reply.
|
||||
""");
|
||||
ConfigRef config = new ConfigRef(f, FleetConfig.load(f),
|
||||
Fleetd::assertChartersNameOnlyRegisteredTools);
|
||||
FleetConfig before = config.get();
|
||||
|
||||
Files.writeString(f, BASE + """
|
||||
fleet:
|
||||
charters:
|
||||
dev: |
|
||||
Send the final handoff through bridge_send.
|
||||
""");
|
||||
ConfigRef.Outcome out = config.reload();
|
||||
|
||||
assertFalse(out.applied());
|
||||
assertNotNull(out.error());
|
||||
assertTrue(out.error().contains("dev"), out.error());
|
||||
assertTrue(out.error().contains("bridge_send"), out.error());
|
||||
assertSame(before, config.get());
|
||||
}
|
||||
|
||||
@Test
|
||||
void reloadAcceptsACharterNamingOnlyRegisteredToolsThroughFleetdsOwnWiring(@TempDir Path dir)
|
||||
throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, BASE + """
|
||||
fleet:
|
||||
charters:
|
||||
dev: old charter
|
||||
""");
|
||||
ConfigRef config = new ConfigRef(f, FleetConfig.load(f),
|
||||
Fleetd::assertChartersNameOnlyRegisteredTools);
|
||||
|
||||
Files.writeString(f, BASE + """
|
||||
fleet:
|
||||
charters:
|
||||
dev: |
|
||||
Send the final handoff through fleet_reply.
|
||||
""");
|
||||
ConfigRef.Outcome out = config.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals("config reloaded", out.summary());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #474 follow-up: {@code Fleetd.main} builds its live {@code ConfigRef} from the
|
||||
* three-argument constructor, {@code new ConfigRef(configPath, cfg,
|
||||
* Fleetd::assertChartersNameOnlyRegisteredTools)}, so a reload runs the same charter tool-surface
|
||||
* gate startup does (see {@link dev.ltms.fleet.config.ConfigRef}'s class doc, "fleetd #474" bullet).
|
||||
* {@code ConfigRefTest} and {@code FleetdConfigRefCharterToolSurfaceWiringTest} prove the
|
||||
* three-argument constructor and the {@code Fleetd.assertChartersNameOnlyRegisteredTools} adapter
|
||||
* work correctly together — both build their OWN {@code ConfigRef} with that constructor, so neither
|
||||
* proves {@code main} still CHOOSES the three-argument form over the plain two-argument {@code new
|
||||
* ConfigRef(configPath, cfg)}.
|
||||
*
|
||||
* <p>Measured directly: reverting {@code Fleetd.java}'s {@code config} local to the two-argument
|
||||
* constructor compiles with 0 errors and leaves the entire 1633-test suite green — including every
|
||||
* {@code ConfigRefTest} and {@code FleetdConfigRefCharterToolSurfaceWiringTest} case — because
|
||||
* neither of those tests constructs its {@code ConfigRef} through {@code main}; both build their own
|
||||
* instance directly, wired with the check by hand. That silent regression is exactly the shape
|
||||
* {@link FleetdBackendQuarantineWiringTest}, {@link FleetdLeadSeatWiringTest} and {@link
|
||||
* FleetdCompletionResolverWiringTest} already guard against for their own constructor arguments —
|
||||
* this class is the same class of gap for fleetd #474's {@code extraValidation} argument, following
|
||||
* their approach.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||
* ConfigRef} and never runs {@code main} — a green result here proves only that the exact text
|
||||
* {@code main} contains is the three-argument construction with {@code
|
||||
* Fleetd::assertChartersNameOnlyRegisteredTools}. It does not prove that call actually executes at
|
||||
* startup (no test here starts the daemon), and it does not prove the reload gate itself works —
|
||||
* only {@code ConfigRefTest} and {@code FleetdConfigRefCharterToolSurfaceWiringTest} prove the
|
||||
* behaviour; only a live daemon proves the wiring runs.
|
||||
*/
|
||||
class FleetdConfigRefWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] main still builds config from the three-argument ConfigRef constructor")
|
||||
void mainStillWiresTheThreeArgumentConfigRefConstructor() throws Exception {
|
||||
String source = fleetdSource();
|
||||
|
||||
// A broken read (wrong working directory, wrong path, a file that came back empty) would
|
||||
// make the assertFalse below pass vacuously — a "clean" negative check that actually
|
||||
// checked nothing. Guard against that first, with an anchor that has nothing to do with
|
||||
// this mutation, so a bad read fails loudly here instead of silently proving nothing below.
|
||||
assertTrue(source.contains("public final class Fleetd"),
|
||||
"fleetdSource() did not read anything usable — src/main/java/dev/ltms/fleet/Fleetd.java "
|
||||
+ "did not come back containing its own class declaration. The assertFalse below "
|
||||
+ "would pass vacuously on a broken read; fix the read before trusting this test.");
|
||||
|
||||
assertTrue(source.contains(
|
||||
"ConfigRef config = new ConfigRef(configPath, cfg, "
|
||||
+ "Fleetd::assertChartersNameOnlyRegisteredTools);"),
|
||||
"Fleetd.main's config local must still be built from the three-argument ConfigRef "
|
||||
+ "constructor, with Fleetd::assertChartersNameOnlyRegisteredTools as "
|
||||
+ "extraValidation. Reverting to the plain two-argument constructor (fleetd #474's "
|
||||
+ "measured M2 regression) compiles with 0 errors and leaves the whole suite green — "
|
||||
+ "including ConfigRefTest and FleetdConfigRefCharterToolSurfaceWiringTest, because "
|
||||
+ "neither builds its ConfigRef through main — this source check is what must go "
|
||||
+ "red instead. A reverted daemon would accept, through a reload with no restart, "
|
||||
+ "exactly the charter that #469/#474 already refuse at startup.");
|
||||
|
||||
// Negative form of the same check: the pre-#474 two-argument call, if it ever reappears at
|
||||
// this declaration, must not be mistaken for the three-argument one by a looser
|
||||
// positive-only check — this is the M2 mutation this test exists to kill.
|
||||
assertFalse(source.contains("ConfigRef config = new ConfigRef(configPath, cfg);"),
|
||||
"main's config local must never regress to the plain two-argument ConfigRef "
|
||||
+ "constructor — that drops the reload-path charter check (fleetd #474's measured "
|
||||
+ "M2 mutation) with no other test catching it");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.inject.LiveExhaustedPatterns;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #446: {@code exhaustionDetectionArmed} must describe the LIVE config, not a startup
|
||||
* snapshot — the opposite of what fleetd #404 asked for. #404 pinned {@code exhaustedPattern} as
|
||||
* compiled once at startup (matching {@code CompletionResolver}'s then-frozen behaviour); #446's
|
||||
* whole point is that both sides — the classification {@code CompletionResolver} enforces and the
|
||||
* {@code exhaustionDetectionArmed} field this reports — now read the SAME live {@link
|
||||
* LiveExhaustedPatterns} instance, so a reload arms or disarms detection with no restart, and the
|
||||
* report can never disagree with the behaviour (the fleetd #404 rule, now upheld for real instead
|
||||
* of by freezing both sides).
|
||||
*
|
||||
* <p>This test needs a reload. At startup the two snapshots agree, so a test of only a newly
|
||||
* started daemon would not detect a live {@code config.get()} lookup in the report field.
|
||||
*
|
||||
* <p><b>The hotness proof criterion 1 demands:</b> run this against the fixed code (green) and
|
||||
* against the pre-#446 compile site — {@code Fleetd.main}'s {@code exhaustedPatternsByProfile}
|
||||
* compiled once into a {@code Map<String, Pattern>} at startup, handed to {@code quarantineSource}
|
||||
* — and show it fails there. See the class doc history above: that shape is exactly what
|
||||
* {@code reloadedPatternArmsDetectionWithNoRestart} below is written to catch, and it is the test
|
||||
* that used to assert the opposite (see git history for this file's pre-#446 version, which
|
||||
* asserted {@code assertFalse(...)} on the identical scenario this now asserts {@code assertTrue}
|
||||
* on).
|
||||
*/
|
||||
class FleetdExhaustionDetectionArmedWiringTest {
|
||||
|
||||
private static final String NO_PATTERN = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: terra
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
private static final String WITH_PATTERN = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: terra
|
||||
exhaustedPattern: "usage limit"
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
@Test
|
||||
@DisplayName("reloading a profile's exhaustedPattern IN arms exhaustionDetectionArmed, no restart")
|
||||
void reloadedPatternArmsDetectionWithNoRestart(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, NO_PATTERN);
|
||||
ConfigRef config = new ConfigRef(file, FleetConfig.load(file));
|
||||
|
||||
FleetMcp.QuarantineSource before = Fleetd.quarantineSource(config, BackendQuarantine.none(),
|
||||
new LiveExhaustedPatterns(() -> config.get().profiles()), Map.of());
|
||||
assertFalse(before.exhaustedPatternArmed().apply("terra"),
|
||||
"no exhaustedPattern configured yet — must report unarmed");
|
||||
|
||||
Files.writeString(file, WITH_PATTERN);
|
||||
assertTrue(config.reload().applied());
|
||||
assertTrue(config.get().profiles().get("terra").hasExhaustedPattern());
|
||||
|
||||
// Re-read exhaustedPatternArmed off the SAME QuarantineSource built BEFORE the reload — no
|
||||
// new object, no new wiring — to prove the field itself is live, not merely that a freshly
|
||||
// built source would be. This is the exact axis the pre-#446 code fails on: the same
|
||||
// Function<String,Boolean> lambda, called again after a reload, sees the new answer only if
|
||||
// it re-reads config.get() on every call rather than a value captured earlier.
|
||||
assertTrue(before.exhaustedPatternArmed().apply("terra"),
|
||||
"exhaustionDetectionArmed must read the LIVE config on every call — fleetd #446 made "
|
||||
+ "this hot; it must reflect a reload with no restart and no new QuarantineSource");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("reloading exhaustedPattern OUT disarms exhaustionDetectionArmed, no restart")
|
||||
void reloadedPatternRemovalDisarmsDetectionWithNoRestart(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, WITH_PATTERN);
|
||||
ConfigRef config = new ConfigRef(file, FleetConfig.load(file));
|
||||
|
||||
FleetMcp.QuarantineSource source = Fleetd.quarantineSource(config, BackendQuarantine.none(),
|
||||
new LiveExhaustedPatterns(() -> config.get().profiles()), Map.of());
|
||||
assertTrue(source.exhaustedPatternArmed().apply("terra"),
|
||||
"exhaustedPattern is configured from the start — must report armed");
|
||||
|
||||
Files.writeString(file, NO_PATTERN);
|
||||
assertTrue(config.reload().applied());
|
||||
|
||||
assertFalse(source.exhaustedPatternArmed().apply("terra"),
|
||||
"removing exhaustedPattern and reloading must disarm detection with no restart — an "
|
||||
+ "operator turning detection off (e.g. while debugging a false positive) "
|
||||
+ "must see that reflected immediately, exactly like arming it is");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a profile with no exhaustedPattern configured is reported unarmed")
|
||||
void aProfileWithNoPatternIsUnarmed(@TempDir Path dir) throws Exception {
|
||||
// fleetd #404's second direction, carried forward: this test alone fails if
|
||||
// exhaustedPatternArmed degenerates to a constant `profile -> true` — it needs at least one
|
||||
// profile that is genuinely unarmed to catch that.
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, WITH_PATTERN);
|
||||
ConfigRef config = new ConfigRef(file, FleetConfig.load(file));
|
||||
|
||||
FleetMcp.QuarantineSource source = Fleetd.quarantineSource(config, BackendQuarantine.none(),
|
||||
new LiveExhaustedPatterns(() -> config.get().profiles()), Map.of());
|
||||
|
||||
assertTrue(source.exhaustedPatternArmed().apply("terra"),
|
||||
"a profile whose exhaustedPattern is configured must report armed");
|
||||
assertFalse(source.exhaustedPatternArmed().apply("sonnet"),
|
||||
"a profile absent from config entirely must not report armed");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,175 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #446 follow-up (round 3): round 2's {@link FleetdUsageLimitFixWarningTest} calls {@link
|
||||
* Fleetd#usageLimitFixWarning}/{@link Fleetd#usageLimitFixWarningNoModel} directly, which pins
|
||||
* the two methods' TEXT but cannot see whether {@link Fleetd#exhaustionSink}'s {@code log.warn}
|
||||
* call actually invokes either one, or invokes the right one. A mutation battery run against
|
||||
* round 2's merge (1592 green tests) proved both gaps real:
|
||||
*
|
||||
* <ul>
|
||||
* <li><b>Cell A</b> — replacing the whole ternary result with a literal string
|
||||
* ({@code "usage-limit fix: MUTANT"}) left every test green;</li>
|
||||
* <li><b>Cell B</b> — swapping the ternary's two branches, so a profile WITH a {@code model:}
|
||||
* gets the no-model fallback message and vice versa, was measured unmeasured by the lead
|
||||
* but the existing test file shows no call that would notice it either.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>This class drives {@link Fleetd#exhaustionSink} — the actual factory {@code main} wires,
|
||||
* since round 3's extraction pulled it out of the inline lambda for exactly this reason — and
|
||||
* asserts on the REAL text a {@link ListAppender} attached to {@code Fleetd}'s own logger
|
||||
* captures, mirroring {@code ExhaustedPatternGapReportTest}'s idiom. Chosen over a source-text
|
||||
* assertion (the {@code FleetMcpAuthzTest} {@code everyRegisteredToolHasItsHandlerActionPinned}
|
||||
* idiom) because that route can prove Cell A (the ternary is still there, still built from the
|
||||
* two named methods) but not Cell B (which branch a given profile actually reaches) — this route
|
||||
* answers both from one mechanism, since it reads what the sink actually logged for each shape of
|
||||
* profile.
|
||||
*
|
||||
* <p>Each test filters for the WARN line starting {@code "usage-limit fix:"} specifically — {@link
|
||||
* Fleetd#exhaustionSink} also logs a separate {@code "credential '...' quarantined for..."} WARN
|
||||
* on every call, and asserting against the wrong one would pass or fail for the wrong reason. That
|
||||
* prefix survives even under the M5 mutation above (the mutant keeps the tag, only the body
|
||||
* becomes {@code "MUTANT"}), so the filter itself is not what either mutation defeats.
|
||||
*/
|
||||
class FleetdExhaustionSinkWarningTest {
|
||||
|
||||
private static final String YAML = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: claude-opus-5
|
||||
gx:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
private static SessionManager emptyRosterSessions() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
FleetConfig.Profile dummy = new FleetConfig.Profile(
|
||||
"dummy", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN", null,
|
||||
"tab", "fleetd-workers", "worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher launcher = new ClaudeCodeLauncher(new AgentControl(h), new WorkspaceControl(h),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(dummy.profile(), dummy), dummy.profile(), _ -> "tok");
|
||||
// Never acquires a session — exhaustionSink's roster lookup is expected to miss and fall
|
||||
// back to the profileHint argument, exactly like OpenCodeLauncher's real call site does
|
||||
// (fleetd #234) — so the sink under test never needs a populated roster.
|
||||
return new SessionManager(launcher);
|
||||
}
|
||||
|
||||
private static ListAppender<ILoggingEvent> attach() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
||||
((Logger) LoggerFactory.getLogger(Fleetd.class)).detachAppender(appender);
|
||||
}
|
||||
|
||||
/** The one WARN line {@code exhaustionSink} builds from the ternary under test, or {@code null}. */
|
||||
private static String fixWarning(ListAppender<ILoggingEvent> appender) {
|
||||
List<String> warns = appender.list.stream()
|
||||
.filter(e -> e.getLevel() == Level.WARN)
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.filter(m -> m.startsWith("usage-limit fix:"))
|
||||
.toList();
|
||||
assertTrue(warns.size() == 1,
|
||||
"expected exactly one 'usage-limit fix:' WARN per onExhausted call, got " + warns.size()
|
||||
+ ": " + warns);
|
||||
return warns.getFirst();
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a profile WITH a configured model gets the actionable models.allow fix, not the fallback")
|
||||
void profileWithModelGetsTheActionableFix(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, YAML);
|
||||
FleetConfig cfg = FleetConfig.load(file);
|
||||
ConfigRef config = ConfigRef.fixed(cfg);
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
Map<String, String> reasonByCredential = new HashMap<>();
|
||||
ExhaustionSink sink = Fleetd.exhaustionSink(emptyRosterSessions(), config, quarantine,
|
||||
reasonByCredential, cfg);
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
sink.onExhausted("term_x", "The usage limit has been reached", "terra");
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
String warning = fixWarning(appender);
|
||||
assertTrue(warning.contains("terra"), "must name the profile: " + warning);
|
||||
assertTrue(warning.contains("claude-opus-5"), "must name the model: " + warning);
|
||||
assertTrue(warning.contains("enabled: false"), "must name the action: " + warning);
|
||||
assertTrue(warning.contains("models.allow"), "must name where the action goes: " + warning);
|
||||
assertFalse(warning.contains("has no model: configured"),
|
||||
"a profile WITH a model must not get the no-model fallback text: " + warning);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a profile with NO configured model gets the fallback, never a fix that does not exist")
|
||||
void profileWithNoModelGetsTheFallbackNotAFix(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, YAML);
|
||||
FleetConfig cfg = FleetConfig.load(file);
|
||||
ConfigRef config = ConfigRef.fixed(cfg);
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
Map<String, String> reasonByCredential = new HashMap<>();
|
||||
ExhaustionSink sink = Fleetd.exhaustionSink(emptyRosterSessions(), config, quarantine,
|
||||
reasonByCredential, cfg);
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
sink.onExhausted("term_y", "The usage limit has been reached", "gx");
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
String warning = fixWarning(appender);
|
||||
assertTrue(warning.contains("gx"), "must name the profile: " + warning);
|
||||
assertTrue(warning.contains("has no model: configured"),
|
||||
"must say there is no model to gate by name: " + warning);
|
||||
assertFalse(warning.contains("enabled: false"),
|
||||
"a profile with no model: configured has no models.allow entry to flip — must "
|
||||
+ "not claim one exists: " + warning);
|
||||
}
|
||||
}
|
||||
@@ -80,7 +80,7 @@ class FleetdLeadMailboxSelectionTest {
|
||||
@Test
|
||||
void opensTheMailboxWhenAUriAndSelfIdAreConfigured() {
|
||||
var opener = new RecordingOpener();
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null);
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null, null);
|
||||
|
||||
Fleetd.openLeadMailbox(coordinator, Map.of(), opener);
|
||||
|
||||
@@ -94,7 +94,7 @@ class FleetdLeadMailboxSelectionTest {
|
||||
void honoursUriEnvOverALiteralUri() {
|
||||
var opener = new RecordingOpener();
|
||||
var coordinator = new FleetConfig.Coordinator("amqp://stale:stale@old:5672/x", "COORD_URI",
|
||||
"mac-opus", 8);
|
||||
"mac-opus", 8, null);
|
||||
|
||||
Fleetd.openLeadMailbox(coordinator, Map.of("COORD_URI", RESOLVED_URI), opener);
|
||||
|
||||
@@ -105,7 +105,7 @@ class FleetdLeadMailboxSelectionTest {
|
||||
@Test
|
||||
void turnsOffWhenUriEnvDoesNotResolve() {
|
||||
var opener = new RecordingOpener();
|
||||
var coordinator = new FleetConfig.Coordinator(null, "COORD_URI", "mac-opus", null);
|
||||
var coordinator = new FleetConfig.Coordinator(null, "COORD_URI", "mac-opus", null, null);
|
||||
|
||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener));
|
||||
|
||||
@@ -116,7 +116,7 @@ class FleetdLeadMailboxSelectionTest {
|
||||
void warnsAndStaysOffWhenSelfIdIsMissing() {
|
||||
var appender = captureFleetdLogs();
|
||||
var opener = new RecordingOpener();
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, null, null);
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, null, null, null);
|
||||
|
||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener));
|
||||
|
||||
@@ -131,7 +131,7 @@ class FleetdLeadMailboxSelectionTest {
|
||||
var appender = captureFleetdLogs();
|
||||
var opener = new RecordingOpener();
|
||||
opener.unreachable = true;
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null);
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null, null);
|
||||
|
||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener),
|
||||
"a down coordination broker turns the feature off; it must never take the daemon down");
|
||||
|
||||
@@ -0,0 +1,89 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #480 Unit A, hard requirement 6: pin {@code Fleetd.main}'s construction of {@link
|
||||
* dev.ltms.fleet.lead.LeadRollover} with a source-text assertion, mirroring {@code
|
||||
* FleetdCompletionResolverWiringTest}'s pattern — five log-only reporters in {@code Fleetd.main}
|
||||
* already survived mutation batteries this exact way (fleetd #415's extraction antidote note).
|
||||
*
|
||||
* <p>What this class still covers, and what it never claimed to. {@code LeadRolloverTest}
|
||||
* constructs its own {@code LeadRollover} directly (as every prior test of an extracted factory
|
||||
* does) with a hand-built lookup, so a mutation that deletes the {@code leadRollover(...)} call
|
||||
* from {@code main} — or replaces one of its arguments with something that still compiles, e.g.
|
||||
* {@code router.leadAgents()} swapped for {@code null}, or the whole assignment swapped for a bare
|
||||
* {@code null} literal — leaves every behavioural test green. This is a plain string read, guarded
|
||||
* by an unrelated anchor assertion so a broken or empty file read cannot pass as a real change.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||
* LeadRollover} and never runs {@code main}. It pins the {@code leadRollover(...)} CALL SITE's
|
||||
* argument list — that {@code main} still passes {@code leads} at all — never what the factory
|
||||
* DOES with that argument once inside its own body.
|
||||
*
|
||||
* <p><b>Correction (fleetd #480 relative-handover-path follow-up): that gap used to be real, and
|
||||
* now is not — but not here.</b> This class's javadoc previously claimed "no behavioural test can
|
||||
* catch this wiring dropping out" for the whole factory, including the lambda {@code
|
||||
* leadRollover(...)} builds internally (terminal → lead name → {@code Leader.cwd()}). That claim
|
||||
* was proven true at the time — mutating that lambda's body to {@code String leadName = null;}
|
||||
* (always "no lead found", which silently reintroduces the daemon-cwd bug this ticket fixes) left
|
||||
* the full suite green, {@code Tests run: 1669, Failures: 0}. It is no longer true: {@code
|
||||
* FleetdLeadRolloverWorkspaceLookupTest} now calls {@code Fleetd.leadRollover(...)} directly with a
|
||||
* real {@link dev.ltms.fleet.config.ConfigRef} built from a temp {@code fleetd.yaml}, and fails
|
||||
* against that exact one-line mutation. So: THIS class still covers only the call site's argument
|
||||
* list; {@code FleetdLeadRolloverWorkspaceLookupTest} is what now covers the lambda's body. Neither
|
||||
* one subsumes the other — keep both.
|
||||
*/
|
||||
class FleetdLeadRolloverWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] unrelated anchor: Fleetd.java still declares the Fleetd class")
|
||||
void unrelatedAnchorStillPresent() throws Exception {
|
||||
// Guards the two assertions below: without this, a bad read (empty string, wrong file,
|
||||
// truncated file) could vacuously fail to contain the leadRollover(...) call too, and a
|
||||
// test that only asserts "contains X" would report a false pass for the wrong reason if X
|
||||
// happened to match. Asserting an unrelated, structurally distant string first proves the
|
||||
// read actually pulled real file content.
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains("public final class Fleetd"),
|
||||
"sanity anchor failed — the file read did not return real Fleetd.java source; the "
|
||||
+ "leadRollover(...) wiring assertions below cannot be trusted until this passes");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] main still constructs LeadRollover via the leadRollover(...) factory, exactly as heartbeat is constructed")
|
||||
void mainStillCallsTheLeadRolloverFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"LeadRollover leadRollover = leadRollover(cfg, router.leadAgents(), config, leads);"),
|
||||
"Fleetd.main must still assign `LeadRollover leadRollover = leadRollover(cfg, "
|
||||
+ "router.leadAgents(), config, leads);`. Dropping this call, or swapping one of "
|
||||
+ "its arguments for something that still compiles (e.g. null in place of "
|
||||
+ "router.leadAgents()), leaves every behavioural test green — this source check is "
|
||||
+ "what must go red instead. fleetd #480 correction 2 deliberately dropped "
|
||||
+ "primaryRegistry from this call — see LeadRollover's class javadoc for why a "
|
||||
+ "single-slot lookup was wrong here. The fleetd #480 relative-handover-path "
|
||||
+ "follow-up added `leads` (terminal → lead name) so the factory can resolve a "
|
||||
+ "relative handoverPath against the calling lead's own workspace.");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] the leadRollover(...) factory itself gates construction on cfg.leadRollover() != null")
|
||||
void factoryGatesOnConfigPresence() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains("if (cfg.leadRollover() == null) {"),
|
||||
"Fleetd.leadRollover(...) must refuse to construct a LeadRollover when the "
|
||||
+ "leadRollover: block is absent — an upgraded daemon must never silently acquire "
|
||||
+ "the ability to clear the lead's own pane. See LeadHeartbeatLoop's construction "
|
||||
+ "gate (cfg.leadHeartbeat() != null) for the pattern this mirrors.");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,163 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.lead.LeadRollover;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.HashMap;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
|
||||
/**
|
||||
* fleetd #480 relative-handover-path follow-up, correction round: a BEHAVIOURAL test of {@link
|
||||
* Fleetd#leadRollover}'s own body — the terminal → lead-name → {@code Leader.cwd()} lookup it
|
||||
* builds — not another source-text pin.
|
||||
*
|
||||
* <p>{@code FleetdLeadRolloverWiringTest} (a plain string read) still earns its keep: it pins the
|
||||
* call site's ARGUMENT LIST, so a mutation that drops {@code leads} back out of the call, or
|
||||
* swaps it for {@code Map::of}, still goes red there. But nothing before this class exercised the
|
||||
* LAMBDA BODY {@code leadRollover(...)} builds — the terminal-to-workspace {@code
|
||||
* Function<String,String>} that {@link LeadRollover#open} actually calls. Every prior test either
|
||||
* exercised {@code LeadRollover} directly with a hand-built lookup ({@code LeadRolloverTest}), or
|
||||
* read source text without ever calling the factory ({@code FleetdLeadRolloverWiringTest}) — so a
|
||||
* mutation that breaks the lookup ITSELF (e.g. always resolving no lead, or reading a snapshot
|
||||
* instead of the live supplier) left every existing test green while the daemon's real wiring
|
||||
* silently reintroduced the exact bug this whole ticket fixes: a relative {@code handoverPath}
|
||||
* resolving against the daemon's own {@code cwd} instead of the calling lead's.
|
||||
*
|
||||
* <p>{@code Fleetd.leadRollover(...)} is package-private and {@code static}, so this test — living
|
||||
* in the same {@code dev.ltms.fleet} package — calls it directly, exactly the way {@code
|
||||
* FleetdExhaustionDetectionArmedWiringTest} and {@code FleetdCapacitySourceWiringTest} already call
|
||||
* other package-private startup factories with a real {@link ConfigRef} built from a temp {@code
|
||||
* fleetd.yaml} (never the gitignored live one).
|
||||
*
|
||||
* <p><b>Proved against the mutation it exists to catch.</b> Before this class was added, mutating
|
||||
* {@code leadRollover(...)}'s lambda body to {@code String leadName = null;} (always "no lead
|
||||
* found", which forces every relative {@code handoverPath} onto the {@code user.dir} fallback —
|
||||
* i.e. the original bug) left the full suite green: {@code Tests run: 1669, Failures: 0}. With
|
||||
* {@link #relativeHandoverPathResolvesAgainstTheLeadsConfiguredCwd()} added, the same one-line
|
||||
* mutation now fails that test (it asserts the resolved path equals the configured lead's {@code
|
||||
* cwd}, which the mutant can never produce) — see this ticket's fleet_reply history for both runs.
|
||||
*/
|
||||
class FleetdLeadRolloverWorkspaceLookupTest {
|
||||
|
||||
private static AgentControl fakeAgents() {
|
||||
return new AgentControl(new FakeHerdr());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] Fleetd.leadRollover(...) resolves a relative handoverPath against "
|
||||
+ "the CALLING lead's configured cwd, not the daemon's own working directory")
|
||||
void relativeHandoverPathResolvesAgainstTheLeadsConfiguredCwd(@TempDir Path dir) throws Exception {
|
||||
Path leadCwd = dir.resolve("lead-workspace");
|
||||
Files.createDirectories(leadCwd);
|
||||
Path yaml = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
fleet:
|
||||
leaders:
|
||||
opus:
|
||||
tab: "lead: opus"
|
||||
cwd: "%s"
|
||||
leadRollover:
|
||||
handoverPath: handover.md
|
||||
""".formatted(leadCwd.toString()));
|
||||
ConfigRef config = new ConfigRef(yaml, FleetConfig.load(yaml));
|
||||
|
||||
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(), config,
|
||||
() -> Map.of("term_opus", "opus"));
|
||||
assertNotNull(rollover, "leadRollover: is present in the loaded config, so the factory "
|
||||
+ "must construct an object");
|
||||
|
||||
LeadRollover.PendingRollover pending = rollover.open("term_opus", "test");
|
||||
|
||||
String expected = leadCwd.resolve("handover.md").normalize().toString();
|
||||
assertEquals(expected, pending.handoverPath(),
|
||||
"a relative handoverPath must resolve against the CALLING lead's configured cwd "
|
||||
+ "through the REAL Fleetd.leadRollover(...) wiring — not the daemon's own "
|
||||
+ "working directory. This is the exact axis that was proven uncovered: "
|
||||
+ "mutating the factory's lambda body to always report \"no lead found\" "
|
||||
+ "left every prior test green.");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] a terminal not present in the live lead-terminal map falls back "
|
||||
+ "to the daemon's own user.dir")
|
||||
void terminalNotInLiveMapFallsBackToUserDir(@TempDir Path dir) throws Exception {
|
||||
Path yaml = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
leadRollover:
|
||||
handoverPath: handover.md
|
||||
""");
|
||||
ConfigRef config = new ConfigRef(yaml, FleetConfig.load(yaml));
|
||||
|
||||
// No lead has been discovered yet — exactly the real shape of a lead the live tab scan
|
||||
// has not yet scanned, or one with no fleet.leaders entry at all.
|
||||
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(), config, Map::of);
|
||||
assertNotNull(rollover);
|
||||
|
||||
LeadRollover.PendingRollover pending = rollover.open("term_unknown", "test");
|
||||
|
||||
String expected = Path.of(System.getProperty("user.dir")).resolve("handover.md")
|
||||
.normalize().toString();
|
||||
assertEquals(expected, pending.handoverPath(),
|
||||
"a lead not present in the live terminal→name map must fall back to the daemon's "
|
||||
+ "own user.dir — the same fallback LeadLauncher#launch already uses for a "
|
||||
+ "lead with no configured cwd. This is deliberate, pinned behaviour, not an "
|
||||
+ "accident of the null-check chain.");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the terminal→lead-name lookup is read LIVE on every open() call, "
|
||||
+ "never snapshotted at Fleetd.leadRollover(...) construction time")
|
||||
void workspaceLookupIsReadLiveNotSnapshotted(@TempDir Path dir) throws Exception {
|
||||
Path leadCwd = dir.resolve("lead-workspace");
|
||||
Files.createDirectories(leadCwd);
|
||||
Path yaml = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
fleet:
|
||||
leaders:
|
||||
opus:
|
||||
tab: "lead: opus"
|
||||
cwd: "%s"
|
||||
leadRollover:
|
||||
handoverPath: handover.md
|
||||
""".formatted(leadCwd.toString()));
|
||||
ConfigRef config = new ConfigRef(yaml, FleetConfig.load(yaml));
|
||||
|
||||
// EMPTY at the moment leadRollover(...) is called. A lookup captured (snapshotted) here
|
||||
// instead of read live through the supplier on every call would never see the entry added
|
||||
// below — exactly the natural mistake to make, since leads are discovered by a live tab
|
||||
// scan that runs AFTER this factory is constructed at startup.
|
||||
Map<String, String> liveLeadTerminals = new HashMap<>();
|
||||
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(), config,
|
||||
() -> liveLeadTerminals);
|
||||
assertNotNull(rollover);
|
||||
|
||||
// The lead is "discovered" only now — mutate the SAME backing map the supplier reads from.
|
||||
liveLeadTerminals.put("term_opus", "opus");
|
||||
|
||||
LeadRollover.PendingRollover pending = rollover.open("term_opus", "test");
|
||||
|
||||
String expected = leadCwd.resolve("handover.md").normalize().toString();
|
||||
assertEquals(expected, pending.handoverPath(),
|
||||
"the terminal→lead-name lookup must be read LIVE on every open() call — a lead "
|
||||
+ "discovered by the tab scan AFTER Fleetd.leadRollover(...) was constructed "
|
||||
+ "must still resolve correctly, not only one that was already live at "
|
||||
+ "construction time");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,156 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* fleetd #176: {@link Fleetd#leadSeatLookup} is the factory {@code Fleetd.main} wires into {@code
|
||||
* FleetMcp.LeadSeatSource} so {@code fleet_list}'s {@code free} can subtract the seat(s) a
|
||||
* {@code subscription: true} profile's own live LEAD session holds on that same account —
|
||||
* {@code maxLoad} never counted the lead, only members. {@code FleetdLeadSeatWiringTest} proves
|
||||
* {@code main} still passes this factory's result in; this class proves the factory's own matching
|
||||
* logic: subscription-only, credential-matched, and counting only CURRENTLY LIVE leads.
|
||||
*/
|
||||
class FleetdLeadSeatLookupTest {
|
||||
|
||||
private static FleetConfig.Profile subscriptionProfile(String name, String credentialId) {
|
||||
return new FleetConfig.Profile(name, null, "claude-sonnet-5", null, null, null,
|
||||
"tab", "fleet", "w #{n}", null, null, null, null, null, null, null,
|
||||
null, 3, true, null, credentialId, null);
|
||||
}
|
||||
|
||||
private static FleetConfig.Profile offSubscriptionProfile(String name, String credentialId) {
|
||||
return new FleetConfig.Profile(name, "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||
null, "tab", "fleet", "w #{n}", null, null, null, null, null, null, null,
|
||||
null, 2, false, null, credentialId, null);
|
||||
}
|
||||
|
||||
private static FleetConfig.Leader leadOnProfile(String profile) {
|
||||
return new FleetConfig.Leader(profile, "lead: primary", 1, "lead:", 10, "claude", "claude-sonnet-5");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a live lead sharing the target profile's credential counts as one seat")
|
||||
void liveLeadSharingCredentialCountsAsOneSeat() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("sonnet", subscriptionProfile("sonnet", null));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("sonnet"));
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders,
|
||||
() -> Map.of("term_primary", "primary"));
|
||||
|
||||
assertEquals(1, lookup.apply("sonnet"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("no live lead names this profile ⇒ zero seats, exactly as before this ticket")
|
||||
void noLiveLeadOnTheProfileCountsAsZero() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("sonnet", subscriptionProfile("sonnet", null));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("sonnet"));
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders, Map::of);
|
||||
|
||||
assertEquals(0, lookup.apply("sonnet"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a lead entry with no `profile:` (recognise-only) contributes no seats — cannot be derived")
|
||||
void recogniseOnlyLeadWithNoProfileContributesNothing() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("sonnet", subscriptionProfile("sonnet", null));
|
||||
FleetConfig.Leader recogniseOnly = new FleetConfig.Leader(null, "lead: primary", 1, "lead:", 10,
|
||||
"claude", "claude-sonnet-5");
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", recogniseOnly);
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders,
|
||||
() -> Map.of("term_primary", "primary"));
|
||||
|
||||
assertEquals(0, lookup.apply("sonnet"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a non-subscription profile never has a lead seat subtracted, whatever the credential match")
|
||||
void nonSubscriptionProfileIsNeverAdjusted() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(
|
||||
"terra", offSubscriptionProfile("terra", "shared-openai"),
|
||||
"sonnet", subscriptionProfile("sonnet", "shared-openai"));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("sonnet"));
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders,
|
||||
() -> Map.of("term_primary", "primary"));
|
||||
|
||||
assertEquals(0, lookup.apply("terra"), "terra is not subscription:true, so it must never be adjusted");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("explicit, different credentialIds still separate two subscription profiles (post fleetd #176 "
|
||||
+ "stage 2 sentinel)")
|
||||
void differentCredentialIsNotCounted() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(
|
||||
"sonnet", subscriptionProfile("sonnet", "claude-account-a"),
|
||||
"opus", subscriptionProfile("opus", "claude-account-b"));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders,
|
||||
() -> Map.of("term_primary", "primary"));
|
||||
|
||||
assertEquals(0, lookup.apply("sonnet"), "different accounts must never be conflated into one seat count "
|
||||
+ "— an explicit credentialId on both sides must still win over the subscription sentinel, so an "
|
||||
+ "operator with two separate Claude logins on one host can keep them apart");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176 stage 2 — the exact live shape that shipped inert: a lead on subscription profile
|
||||
* {@code opus}, members on a DIFFERENTLY NAMED subscription profile {@code sonnet}, same Claude
|
||||
* login, and NEITHER profile sets {@code credentialId}. Every other test in this class puts the
|
||||
* lead on the SAME profile name as the target, which happened to keep working even with the old
|
||||
* fall-back-to-profile-name {@code effectiveCredentialId()} — this is the one that did not, and
|
||||
* its absence is what let the stage-1 fix ship without ever catching the bug it was filed for.
|
||||
*/
|
||||
@Test
|
||||
@DisplayName("[LIVE SHAPE] lead on a DIFFERENT subscription profile, same account, neither sets "
|
||||
+ "credentialId ⇒ still counts as a seat")
|
||||
void leadOnADifferentSubscriptionProfileSameAccountStillCountsAsASeat() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(
|
||||
"opus", subscriptionProfile("opus", null),
|
||||
"sonnet", subscriptionProfile("sonnet", null));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders,
|
||||
() -> Map.of("term_primary", "primary"));
|
||||
|
||||
assertEquals(1, lookup.apply("sonnet"), "opus and sonnet are both subscription:true with no explicit "
|
||||
+ "credentialId, so they share one Claude login and the lead's live seat on opus must be charged "
|
||||
+ "against sonnet too — this is the live host's actual shape (fleetd #176 stage 2)");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("two live instances of the same lead count as two seats")
|
||||
void twoLiveInstancesOfTheSameLeadCountAsTwoSeats() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("sonnet", subscriptionProfile("sonnet", null));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("sonnet"));
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders,
|
||||
() -> Map.of("term_a", "primary", "term_b", "primary"));
|
||||
|
||||
assertEquals(2, lookup.apply("sonnet"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an unconfigured target profile resolves to zero, not a thrown exception")
|
||||
void unconfiguredTargetProfileIsZero() {
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(Map::of, Map.of(), Map::of);
|
||||
|
||||
assertEquals(0, lookup.apply("ghost"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("live leads are read through the supplier on every call, not snapshotted")
|
||||
void liveLeadsAreReadThroughOnEveryCall() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("sonnet", subscriptionProfile("sonnet", null));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("sonnet"));
|
||||
java.util.Map<String, String> live = new java.util.HashMap<>();
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders, () -> live);
|
||||
|
||||
assertEquals(0, lookup.apply("sonnet"));
|
||||
live.put("term_primary", "primary");
|
||||
assertEquals(1, lookup.apply("sonnet"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,43 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #176: {@code Fleetd.main} builds its {@code FleetMcp} from a 14-argument constructor whose
|
||||
* last argument is a {@code FleetMcp.LeadSeatSource} wrapping {@link Fleetd#leadSeatLookup}. That
|
||||
* argument is exactly the kind of wiring fleetd #248 warned about: dropping it (or swapping it for
|
||||
* the inert {@code FleetMcp.LeadSeatSource.none()}) compiles with 0 errors and leaves every test
|
||||
* that builds its own {@code FleetMcp}/{@code CapacitySource} directly — every test that predates
|
||||
* this ticket — green, because none of them go through {@code main} at all.
|
||||
*
|
||||
* <p>{@link FleetdLeadSeatLookupTest} proves the factory's own matching logic; this class is the
|
||||
* plain source-text assertion that proves {@code main} still passes its result in, mirroring
|
||||
* {@code FleetdCompletionResolverWiringTest}'s approach for the same class of gap.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a
|
||||
* {@code FleetMcp} and never runs {@code main}.
|
||||
*/
|
||||
class FleetdLeadSeatWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] FleetMcp's construction call still passes a LeadSeatSource built from leadSeatLookup(...)")
|
||||
void fleetMcpConstructionStillWiresLeadSeatLookup() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains("new FleetMcp.LeadSeatSource(leadSeatLookup(() -> config.get().profiles(), "
|
||||
+ "leaders, leads))"),
|
||||
"FleetMcp's construction call must still pass a LeadSeatSource built from "
|
||||
+ "Fleetd.leadSeatLookup(...). Dropping it or swapping in "
|
||||
+ "FleetMcp.LeadSeatSource.none() (fleetd #176's would-be silent regression, the same "
|
||||
+ "shape as fleetd #248's measured mutations) compiles with 0 errors and leaves every "
|
||||
+ "existing behavioural test green — this source check is what must go red instead.");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: {@code Fleetd.modelGateCoverageLine} is the startup-log counterpart of
|
||||
* {@code exhaustedPatternCoverageLine}/{@code errorPatternCoverageLine} — see {@code
|
||||
* FleetdPatternCoverageLineTest} for the identical shape this follows — except here there is no
|
||||
* {@code UnsetMeaning} choice for a caller to get backwards: {@link
|
||||
* PeerLauncher.ModelGateState#configured()} already states, unambiguously, whether an empty
|
||||
* {@link PeerLauncher.ModelGateState#off()} means "no {@code models:} block to gate with at all"
|
||||
* or "a block armed and currently reporting zero off". This class proves {@code
|
||||
* modelGateCoverageLine} words those two states — plus the third, N off — distinctly, so a
|
||||
* mutation that made it ignore {@code configured()} either way is caught here.
|
||||
*/
|
||||
class FleetdModelGateCoverageLineTest {
|
||||
|
||||
@Test
|
||||
@DisplayName("no models: block reports not configured, distinct from armed-with-zero")
|
||||
void noModelsBlockReportsNotConfigured() {
|
||||
String line = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.notConfigured());
|
||||
assertEquals("not configured (no models: block — nothing is gated, and nothing can be)", line);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a models: block armed with nothing off reports armed, distinct from not configured")
|
||||
void armedWithNothingOffReportsArmed() {
|
||||
String line = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.armed(Set.of()));
|
||||
assertEquals("armed (models: block present; 0 models currently turned off)", line);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a models: block with N off names the off models")
|
||||
void armedWithModelsOffNamesThem() {
|
||||
String line = Fleetd.modelGateCoverageLine(
|
||||
PeerLauncher.ModelGateState.armed(Set.of("deepseek-v4-flash", "claude-opus-9000")));
|
||||
assertEquals("armed (2 model(s) turned off: [claude-opus-9000, deepseek-v4-flash])", line);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("the three states produce pairwise-distinct wording for the same empty-looking input")
|
||||
void theThreeStatesProduceDistinctWording() {
|
||||
String notConfigured = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.notConfigured());
|
||||
String armedZero = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.armed(Set.of()));
|
||||
String armedOne = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.armed(Set.of("x")));
|
||||
|
||||
// Pinned individually above; restated here so this test alone still catches a regression
|
||||
// even if one of the three tests above were ever deleted — the exact FleetdPatternCoverageLineTest
|
||||
// pattern, adapted from "two keys" to "three states of one gate".
|
||||
assertNotEquals(notConfigured, armedZero,
|
||||
"collapsing 'no models: block' into 'armed, zero off' is fleetd #422 follow-up's exact defect");
|
||||
assertNotEquals(armedZero, armedOne);
|
||||
assertNotEquals(notConfigured, armedOne);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
||||
|
||||
/**
|
||||
* fleetd #415 (review follow-up): {@code CompletionResolverTest} proves {@code coverage()} words
|
||||
* {@code UnsetMeaning.OFF} and {@code UnsetMeaning.BUILT_IN_DEFAULT} correctly — but every one of
|
||||
* those tests supplies the meaning itself. That proves the enum's wording, never that {@code
|
||||
* Fleetd} pairs the right meaning with the right pattern key. That pairing is #415's actual
|
||||
* defect: {@code coverage()} had no way to know what unset meant for its key, so the fix moved
|
||||
* the fact to the caller — and nothing yet proved the caller states it correctly.
|
||||
*
|
||||
* <p><b>Measured or it didn't happen:</b> swapping the two {@code UnsetMeaning} arguments at
|
||||
* {@code Fleetd}'s two coverage call sites — giving {@code exhaustedPattern} the built-in-default
|
||||
* wording and {@code errorPattern} the off wording, #415's exact defect with the keys exchanged —
|
||||
* compiled with 0 errors and left all 1506 existing tests green. This class exists to turn that
|
||||
* swap red.
|
||||
*
|
||||
* <p>It calls {@link Fleetd#exhaustedPatternCoverageLine} and {@link Fleetd#errorPatternCoverageLine}
|
||||
* directly rather than reading {@code Fleetd.java} as source text (the shape {@code
|
||||
* FleetdCompletionResolverWiringTest} uses for a different wiring gap): those two methods are the
|
||||
* extracted call sites {@code main} actually invokes, following the same {@code static} factory +
|
||||
* dedicated-test pattern as {@link Fleetd#capacitySource} and {@link Fleetd#worktreeBranchLookup}.
|
||||
*/
|
||||
class FleetdPatternCoverageLineTest {
|
||||
|
||||
private static final Set<String> PROFILES = Set.of("terra", "gx10");
|
||||
|
||||
@Test
|
||||
@DisplayName("exhaustedPatternCoverageLine says off when no profile configures exhaustedPattern")
|
||||
void exhaustedPatternCoverageLineSaysOffWhenNoProfileConfiguresIt() {
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [gx10, terra])",
|
||||
Fleetd.exhaustedPatternCoverageLine(PROFILES, Set.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("errorPatternCoverageLine says built-in default when no profile configures errorPattern")
|
||||
void errorPatternCoverageLineSaysBuiltInDefaultWhenNoProfileConfiguresIt() {
|
||||
assertEquals("built-in default for all profiles (no profile customises errorPattern; "
|
||||
+ "profiles: [gx10, terra])",
|
||||
Fleetd.errorPatternCoverageLine(PROFILES, Set.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("the two keys produce different wording for the identical empty-coverage input")
|
||||
void theTwoKeysProduceDifferentWordingForTheSameEmptyInput() {
|
||||
String exhaustedLine = Fleetd.exhaustedPatternCoverageLine(PROFILES, Set.of());
|
||||
String errorLine = Fleetd.errorPatternCoverageLine(PROFILES, Set.of());
|
||||
|
||||
// Pinned individually above; restated here so this test alone still catches a swap even
|
||||
// if one of the two tests above were ever deleted.
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [gx10, terra])",
|
||||
exhaustedLine);
|
||||
assertEquals("built-in default for all profiles (no profile customises errorPattern; "
|
||||
+ "profiles: [gx10, terra])", errorLine);
|
||||
assertNotEquals(exhaustedLine, errorLine,
|
||||
"swapping which UnsetMeaning pairs with which pattern key at Fleetd's call sites "
|
||||
+ "must be caught here — that pairing, not coverage()'s own wording in isolation, "
|
||||
+ "is fleetd #415's actual defect");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* Proves {@link Fleetd#main(String[])} calls every startup report before validation aborts startup.
|
||||
* The invalid non-loopback bind makes {@link FleetConfig#validateAll()} throw before {@code main}
|
||||
* can open the herdr socket or bind a port. The fixture also triggers every report, so removing any
|
||||
* one call from {@code main} leaves its expected log line absent.
|
||||
*/
|
||||
class FleetdStartupReportTest {
|
||||
|
||||
private static Level originalLevel;
|
||||
|
||||
private static ListAppender<ILoggingEvent> attach() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
originalLevel = logger.getLevel();
|
||||
logger.setLevel(Level.INFO);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(originalLevel);
|
||||
}
|
||||
|
||||
private static boolean contains(ListAppender<ILoggingEvent> appender, String fragment) {
|
||||
return appender.list.stream()
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.anyMatch(message -> message.contains(fragment));
|
||||
}
|
||||
|
||||
@Test
|
||||
void mainReportsEveryStartupGapBeforeValidationAborts(@TempDir Path dir) throws Exception {
|
||||
Path config = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(config, """
|
||||
bind:
|
||||
host: 0.0.0.0
|
||||
port: 8765
|
||||
profiles:
|
||||
worker:
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
gitTokenEnv: GITEA_TOKEN
|
||||
""");
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
assertThrows(IllegalStateException.class, () -> Fleetd.main(new String[]{config.toString()}));
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
assertTrue(contains(appender, "startup git host GITEA_HOST:"),
|
||||
"Fleetd.main must report the git host shape");
|
||||
assertTrue(contains(appender, "member trust model: members run as the same OS user"),
|
||||
"Fleetd.main must report the member trust model");
|
||||
assertTrue(contains(appender, "memberCredentials: absent or empty"),
|
||||
"Fleetd.main must report an absent memberCredentials policy");
|
||||
assertTrue(contains(appender, "exhaustedPattern: profile(s) [worker] have no exhaustedPattern configured"),
|
||||
"Fleetd.main must report profiles without exhaustedPattern");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,154 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* Proves the one thing no test proved before this ticket's follow-up: that {@code Fleetd.main}
|
||||
* ITSELF — not a copy of its logic, not the validator called directly — still refuses to start on
|
||||
* a bad config. Mutation testing found that deleting {@code cfg.validateAll();} (née six
|
||||
* individual {@code cfg.validateXxx();} calls) from {@code Fleetd.main} left the full 1478-test
|
||||
* suite green; every existing test called a validator directly and none exercised {@code
|
||||
* Fleetd.main} as the caller. See {@code FleetConfigValidateAllTest} for why the fix collapses
|
||||
* those six calls into one reflective {@link FleetConfig#validateAll()}, and {@code
|
||||
* ConfigRefTest} for the equivalent proof on the {@link
|
||||
* dev.ltms.fleet.config.ConfigRef#reload()} path.
|
||||
*
|
||||
* <p>This is deliberately the real {@code static void main(String[] args)} — package-private, so
|
||||
* only a test in this package can call it, which is exactly what makes this proof strong: it is
|
||||
* not a helper extracted for testability, it is the literal method {@code java -jar fleetd.jar}
|
||||
* invokes. Every fixture below is otherwise-valid and fails exactly one validator, and — because
|
||||
* {@link FleetConfig#validateAll()} runs immediately after {@code SubscriptionGuard.
|
||||
* assertPrimaryClean}, before {@code main} opens the herdr socket, binds Javalin, or touches
|
||||
* anything else with a real side effect — calling {@code Fleetd.main} with one of these configs is
|
||||
* safe: it is guaranteed to throw before reaching any of that, precisely because the config is
|
||||
* deliberately invalid.
|
||||
*/
|
||||
class FleetdStartupValidationTest {
|
||||
|
||||
private static void assertMainRefuses(Path dir, String fileName, String yaml,
|
||||
String mustContain) throws Exception {
|
||||
Path f = dir.resolve(fileName);
|
||||
Files.writeString(f, yaml);
|
||||
IllegalStateException e = assertThrows(IllegalStateException.class,
|
||||
() -> Fleetd.main(new String[]{f.toString()}),
|
||||
fileName + ": Fleetd.main must refuse this config before doing anything else");
|
||||
assertTrue(e.getMessage().contains(mustContain),
|
||||
fileName + ": expected message to contain \"" + mustContain + "\" but was: "
|
||||
+ e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void mainRefusesANonLoopbackBindWithoutTokenMode(@TempDir Path dir) throws Exception {
|
||||
assertMainRefuses(dir, "auth-exposure.yaml", """
|
||||
bind:
|
||||
host: 0.0.0.0
|
||||
port: 8765
|
||||
""", "auth.mode: token");
|
||||
}
|
||||
|
||||
@Test
|
||||
void mainRefusesALeadTabPrefixCollision(@TempDir Path dir) throws Exception {
|
||||
assertMainRefuses(dir, "lead-tab-prefixes.yaml", """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
fleet:
|
||||
tabLabel: "lead: {role} {profile}"
|
||||
leaders:
|
||||
opus:
|
||||
tab: "lead: opus"
|
||||
""", "fleet.tabLabel");
|
||||
}
|
||||
|
||||
@Test
|
||||
void mainRefusesASubscriptionProfileThatReseatsAnthropicBaseUrl(@TempDir Path dir)
|
||||
throws Exception {
|
||||
assertMainRefuses(dir, "subscription-profiles.yaml", """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
profiles:
|
||||
sonnet:
|
||||
subscription: true
|
||||
argv: ["ccs", "sonnet"]
|
||||
env:
|
||||
ANTHROPIC_BASE_URL: http://anything-not-on-the-allowlist
|
||||
""", "ANTHROPIC_BASE_URL");
|
||||
}
|
||||
|
||||
@Test
|
||||
void mainRefusesAnUnknownCharterKey(@TempDir Path dir) throws Exception {
|
||||
assertMainRefuses(dir, "charters.yaml", """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
fleet:
|
||||
charters:
|
||||
architetc: text
|
||||
""", "architetc");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #469: closes the gap left by #464's {@code CharterToolSurfaceTest}, which wrote its
|
||||
* own charter into a {@code @TempDir} fixture and so could never fail on anything anyone wrote
|
||||
* into the live {@code fleetd.yaml}. {@code bridge_send} is CB-634's own motivating example — a
|
||||
* tool name the rename removed — and the charter key ({@code dev}) is a real role wire name, so
|
||||
* this fixture passes {@code cfg.validateAll()}'s charter check (key valid, text non-blank) and
|
||||
* is refused only by the new {@code CharterToolSurface} call right after it. The failure message
|
||||
* must name both the charter key and the unknown tool.
|
||||
*/
|
||||
@Test
|
||||
void mainRefusesACharterNamingAnUnregisteredTool(@TempDir Path dir) throws Exception {
|
||||
assertMainRefuses(dir, "charter-tool-surface.yaml", """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
fleet:
|
||||
charters:
|
||||
dev: |
|
||||
Send the final handoff through bridge_send.
|
||||
""", "bridge_send");
|
||||
}
|
||||
|
||||
@Test
|
||||
void mainRefusesAnArchitectSlotNamingAnUnconfiguredProfile(@TempDir Path dir) throws Exception {
|
||||
assertMainRefuses(dir, "members.yaml", """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
profiles:
|
||||
gx10:
|
||||
baseUrl: http://gx10.gw:8000
|
||||
fleet:
|
||||
architects:
|
||||
lead-designer:
|
||||
profile: sonnet
|
||||
""", "lead-designer");
|
||||
}
|
||||
|
||||
@Test
|
||||
void mainRefusesAProfileNamingAModelOutsideTheAllowList(@TempDir Path dir) throws Exception {
|
||||
assertMainRefuses(dir, "models.yaml", """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx10.gw:8000
|
||||
model: claude-sonnet-5
|
||||
rogue:
|
||||
baseUrl: http://gx11.gw:8000
|
||||
model: claude-opus-9000
|
||||
models:
|
||||
allow:
|
||||
- model: claude-sonnet-5
|
||||
""", "rogue");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,75 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #446 follow-up (criterion 2): {@code exhaustionSink} used to build the "name the fix"
|
||||
* WARNING inline with two SLF4J {@code {}}-placeholder {@code log.warn} calls. Nothing pinned
|
||||
* that text — a mutation battery run against the merged PR #457 proved it: renaming the leading
|
||||
* {@code "usage-limit fix:"} tag to {@code "MUTANT no fix named:"} at both call sites left all
|
||||
* 1586 existing tests green. This class is what makes that mutation red.
|
||||
*
|
||||
* <p>{@link Fleetd#usageLimitFixWarning} and {@link Fleetd#usageLimitFixWarningNoModel} are the
|
||||
* extracted call sites {@code exhaustionSink} actually invokes — {@code log.warn} is called with
|
||||
* each method's return value as a single, already-formatted argument, so the string these tests
|
||||
* assert on is byte-identical to what {@code fleetd.out} receives. Same extracted-static-method +
|
||||
* dedicated-test idiom as {@link Fleetd#exhaustedPatternCoverageLine} /
|
||||
* {@link FleetdPatternCoverageLineTest}, chosen over the {@link ch.qos.logback.core.read.ListAppender}
|
||||
* idiom {@code ExhaustedPatternGapReportTest} uses — this warning is built once, at one call site,
|
||||
* from a single method with no branching inside the message itself, so there is no real log
|
||||
* plumbing left to prove; capturing an appender would only add setup/teardown around the same
|
||||
* assertion.
|
||||
*
|
||||
* <p>Each assertion below checks for a specific substring — the leading {@code "usage-limit fix:"}
|
||||
* tag an operator would grep the log for, the profile name, the model name, and the literal
|
||||
* action ({@code enabled: false} under {@code models.allow}) — rather than the whole sentence.
|
||||
* Criterion 2 promised those facts, not a particular wording of the prose that carries them; a
|
||||
* whole-sentence {@code assertEquals} breaks on every copy-edit of that surrounding prose and
|
||||
* teaches the next person to delete the test rather than read why it failed. The leading tag is
|
||||
* pinned on purpose, unlike the rest of the sentence: it is the identifying label the fleetd #446
|
||||
* follow-up's own mutation battery renamed ({@code "usage-limit fix:"} → {@code "MUTANT no fix
|
||||
* named:"}) to prove this class did not yet catch a tag rename — only the three-fact substrings
|
||||
* above would have missed it, since none of them mention the tag itself.
|
||||
*/
|
||||
class FleetdUsageLimitFixWarningTest {
|
||||
|
||||
@Test
|
||||
@DisplayName("usageLimitFixWarning names the profile, the model, and the models.allow fix")
|
||||
void usageLimitFixWarningNamesProfileModelAndFix() {
|
||||
String warning = Fleetd.usageLimitFixWarning("terra", "claude-opus-5");
|
||||
|
||||
assertTrue(warning.startsWith("usage-limit fix:"),
|
||||
"must carry the grep-able identifying tag: " + warning);
|
||||
assertTrue(warning.contains("terra"), "must name the profile: " + warning);
|
||||
assertTrue(warning.contains("claude-opus-5"), "must name the model: " + warning);
|
||||
assertTrue(warning.contains("enabled: false"), "must name the action: " + warning);
|
||||
assertTrue(warning.contains("models.allow"), "must name where the action goes: " + warning);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("usageLimitFixWarning does not cross-name a different profile or model")
|
||||
void usageLimitFixWarningDoesNotCrossName() {
|
||||
String warning = Fleetd.usageLimitFixWarning("terra", "claude-opus-5");
|
||||
|
||||
// Guards against the mutation of swapping the two profile/model arguments at the call
|
||||
// site — a test that only checks "some name appears" cannot catch that.
|
||||
assertTrue(!warning.contains("sonnet"), "must not name an unrelated model: " + warning);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("usageLimitFixWarningNoModel names the profile but no fix, since none exists")
|
||||
void usageLimitFixWarningNoModelNamesProfileNotAFix() {
|
||||
String warning = Fleetd.usageLimitFixWarningNoModel("gx", 1800);
|
||||
|
||||
assertTrue(warning.startsWith("usage-limit fix:"),
|
||||
"must carry the grep-able identifying tag: " + warning);
|
||||
assertTrue(warning.contains("gx"), "must name the profile: " + warning);
|
||||
assertTrue(warning.contains("1800"), "must name the quarantine duration it falls back to: " + warning);
|
||||
assertTrue(!warning.contains("enabled: false"),
|
||||
"a profile with no model: configured has no models.allow entry to flip — the "
|
||||
+ "fallback message must not claim one exists: " + warning);
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user