Compare commits
212 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| d678783af7 | |||
| 799014e99d | |||
| 69e09b10fa | |||
| b9d09e044e | |||
| fd8650cda4 | |||
| 11050e24ed | |||
| 769f282408 | |||
| 6f71f40047 | |||
| fb36c5238f | |||
| cc9cdc938b | |||
| 5fede82468 | |||
| 1515025804 | |||
| 7754f53662 | |||
| a507f7b31b | |||
| 71c322f104 | |||
| fde2c15627 | |||
| 2830735644 | |||
| 4e3ac91a22 | |||
| 29d3f0b41f | |||
| cbc732444f | |||
| 6f828b8c38 | |||
| 145a8c8862 | |||
| 7057291739 | |||
| 2302b3bc11 | |||
| 3a004dc1b3 | |||
| a9a3c12232 | |||
| 94ec77a1bc | |||
| 49f285cfda | |||
| 5a12ae7930 | |||
| 127e6832a9 | |||
| 6442a583ae | |||
| 24b96d29ae | |||
| 4721771052 | |||
| 2f71a30bd7 | |||
| 22cdebbdbe | |||
| 154971c2b8 | |||
| fef287c346 | |||
| a37acd5ee3 | |||
| d6ef0c8013 | |||
| dfb70871b4 | |||
| 4ee7b16929 | |||
| 8557289dc0 | |||
| dd2efd8541 | |||
| 5af786d135 | |||
| 380eb63277 | |||
| 6c61355f8f | |||
| 96c406b968 | |||
| a5d81c3f70 | |||
| 92c0f164f1 | |||
| 9e813ec179 | |||
| 395b3b5c46 | |||
| 6e9e464d62 | |||
| 105c065615 | |||
| 01492059d4 | |||
| f84824ee29 | |||
| c4d40fbc2b | |||
| 7c684e40d3 | |||
| 6938f52155 | |||
| 0df34f3220 | |||
| 457458437f | |||
| 3759c41f99 | |||
| 09159f2857 | |||
| 29cd1194c2 | |||
| 815e8f8b23 | |||
| 1e60ac0745 | |||
| 650a4c146b | |||
| 23f299e105 | |||
| dbf6fef0e9 | |||
| d292522d00 | |||
| 0241e0d3a8 | |||
| e4973eb8a4 | |||
| b6b88c5f1c | |||
| 86dddfe240 | |||
| 0d5944af63 | |||
| 4a5030a5c6 | |||
| c1c8794c48 | |||
| 65a78932c1 | |||
| f429ca1a50 | |||
| 73aab3f83e | |||
| 887aca0183 | |||
| 3fd23ecafa | |||
| f379847942 | |||
| ea12107497 | |||
| 591df91de1 | |||
| 6a814176f0 | |||
| d11d1d157c | |||
| d057d56156 | |||
| d703ce1313 | |||
| b32a30fd47 | |||
| 464dbc0930 | |||
| a8cadd9150 | |||
| 57b8c0b56d | |||
| 147f50c19e | |||
| eee4d576a2 | |||
| b4f9d7f53a | |||
| 3aca53b967 | |||
| 4aa1fae296 | |||
| 0c865032f9 | |||
| ea41bbf6b9 | |||
| 7b918c51ff | |||
| 554395b104 | |||
| 823976c1b5 | |||
| c8388a7f92 | |||
| 02e6aef98c | |||
| 6d493bc7bb | |||
| e5cb51a90e | |||
| e545c08082 | |||
| efa0deb9b2 | |||
| ca47e90c01 | |||
| fa1f49675b | |||
| 8426c3528f | |||
| 65f98ba910 | |||
| d05205d1eb | |||
| 667254df47 | |||
| 2926cd1784 | |||
| c801851c66 | |||
| de70aa38f1 | |||
| 53a533afb4 | |||
| 77ad88631b | |||
| d88017807b | |||
| 2159a5a94a | |||
| b9c2cf69f4 | |||
| 8beae50fe7 | |||
| b2a58cb966 | |||
| b8b25cf74c | |||
| f159ca7d27 | |||
| 83f2aea60f | |||
| 002329adb5 | |||
| a49671ceb9 | |||
| 76672ff016 | |||
| 9379f92c23 | |||
| 21ff63b11d | |||
| 21844b54d7 | |||
| 4769481515 | |||
| bbbb4c1eb3 | |||
| 38c248e617 | |||
| 9debc0de27 | |||
| f34361b263 | |||
| be5ba22c75 | |||
| 85c90d440a | |||
| e97502d550 | |||
| aef14ff46e | |||
| cba516bda4 | |||
| ba51e0c6cc | |||
| 086c59848e | |||
| 0c10079755 | |||
| ece2091b53 | |||
| b3f917e6f5 | |||
| fa39a5f55e | |||
| d5128a1d35 | |||
| 61097e5cf0 | |||
| ef507bcd12 | |||
| 94f50e507a | |||
| 75b15086b0 | |||
| dab9645906 | |||
| e93b5f6512 | |||
| bfabe13e8f | |||
| 7e49c6eca2 | |||
| 11cbfa79b4 | |||
| c2c2746922 | |||
| 6ed70700a0 | |||
| 0087645da4 | |||
| 19cacf5b62 | |||
| 4dd12083ab | |||
| 30e21adec7 | |||
| 719b79f892 | |||
| 66e5247b6d | |||
| 1e41bd63b4 | |||
| 4887d03d88 | |||
| 7d5434455d | |||
| f71ee4926e | |||
| 282a2fc2b8 | |||
| f04e934b94 | |||
| 3fae35c357 | |||
| 18aecbfe67 | |||
| 5d75f72473 | |||
| e028a0ae54 | |||
| 2fa673d4c0 | |||
| 9020d01b40 | |||
| 1006805027 | |||
| 27aefbf9a0 | |||
| d42c2bc204 | |||
| 3916adc372 | |||
| fa97f598dd | |||
| ea9aa4fd77 | |||
| a2b8caf6b5 | |||
| b5ddbe5757 | |||
| d223a93039 | |||
| 96d8191149 | |||
| f0e7ac73d6 | |||
| e2fe861b4d | |||
| 279d6f5fbd | |||
| 34480cebef | |||
| 9d0bf14c46 | |||
| 5a3ab5764c | |||
| 3bad9f5785 | |||
| 8308c0b68f | |||
| 3b3063eb2b | |||
| df9086263d | |||
| eb568ff451 | |||
| ac790e4cce | |||
| 6b5f3f472f | |||
| 38dec72152 | |||
| 0e8bfb74fc | |||
| 51f7b0a3ca | |||
| 21c539f22e | |||
| bd2774b5f1 | |||
| 01a840cc14 | |||
| c4deef08be | |||
| e897e5257b | |||
| 2afa3652bb | |||
| 80092ff359 |
@@ -1,15 +1,15 @@
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"name": "fleetd",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the fleetd MCP gateway.",
|
||||
"owner": {
|
||||
"name": "LTMS"
|
||||
},
|
||||
"plugins": [
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"name": "fleet",
|
||||
"source": "./plugin",
|
||||
"description": "Make a project bridge-ready: mount the fleetd MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.1.0",
|
||||
"description": "Mount the fleetd MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.2.0",
|
||||
"author": {
|
||||
"name": "LTMS"
|
||||
}
|
||||
|
||||
@@ -0,0 +1,102 @@
|
||||
---
|
||||
name: hunter
|
||||
description: Defect-hunt procedure for a fleetd worker — sweep an assigned package for real bugs and report several ranked findings without fixing anything. Load this when the lead asks you to hunt or audit a scope rather than review one diff. Do NOT load `reviewer` for this; the two want different output.
|
||||
---
|
||||
|
||||
# Hunter worker — procedure
|
||||
|
||||
The turn contract (one `fleet_reply`, `fleet_ask` for the lead's decisions, honest reporting,
|
||||
never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and already applies.
|
||||
|
||||
**This skill is not `reviewer`.** `reviewer` judges one diff and reports the *single* most
|
||||
important issue in about 90 words. A hunt sweeps a whole package and reports *several* findings
|
||||
in a long structured form. Loading both gives you two contradictory output contracts, and the
|
||||
usual result is a worker that writes a good report into its terminal and ends the turn without
|
||||
sending it. Load exactly one.
|
||||
|
||||
## 0. Read this before you read code: how the report gets home
|
||||
|
||||
Your terminal reaches nobody. The lead sees **only** the text inside your `fleet_reply` call.
|
||||
|
||||
A long report is exactly the case where this goes wrong, so plan for it:
|
||||
|
||||
- **Write the report into the `fleet_reply` argument itself.** Do not compose it in your terminal
|
||||
and then summarise it into the call.
|
||||
- If the report is long, **send it anyway** — one `fleet_reply` with everything.
|
||||
- If you end the turn without replying, the bridge scrapes your pane instead. That scrape carries
|
||||
at most the last 4000 characters, and on a hunt it usually captures the tail of the lead's own
|
||||
brief rather than your findings. The lead then has nothing and has to ask you again.
|
||||
|
||||
## 1. Change nothing
|
||||
|
||||
A hunt is read-only. Do not edit a production file, do not "quickly fix" what you find, and do
|
||||
not run a formatter. You may run the build and tests to *check* a claim, and you should say so
|
||||
when you did.
|
||||
|
||||
## 2. Read the whole scope first
|
||||
|
||||
Read every file in the assigned package before you judge any of it. A defect that a caller
|
||||
elsewhere in the same package makes unreachable is not a defect, and you cannot know that from
|
||||
one file.
|
||||
|
||||
Stay inside the scope. If a defect there depends on a class outside it, read that class to
|
||||
confirm — but the defect itself must live in the scope you were given.
|
||||
|
||||
## 3. The bar — this matters more than the count
|
||||
|
||||
**Name the path into the bad state.** Say which caller, in which state, reaches it. A defect on
|
||||
paper is not a reachable defect. If you cannot name that path, keep the finding but mark it
|
||||
`unproven` and say exactly what you could not check. Do not drop it, and do not dress it up.
|
||||
|
||||
**Say which direction the harm goes.** Data loss, privilege escalation and silent wrong answers
|
||||
are worth reporting even when the window is narrow. A finding whose worst outcome is a worse log
|
||||
line is not worth a block.
|
||||
|
||||
Two workers once ran the same scope: the one that applied the direction-of-harm filter found ten
|
||||
real defects, the one that did not found none. Fewer findings the lead can act on beat many the
|
||||
lead has to triage.
|
||||
|
||||
## 4. Shapes that have produced real merged fixes here
|
||||
|
||||
Read for these first:
|
||||
|
||||
1. **A one-way gate.** A guard added after an incident closes only the direction that incident
|
||||
came from. Do not only ask what closes the gate — ask **which states still open it**.
|
||||
2. **A value read once, then used later to authorise something destructive**, after something
|
||||
else has had a chance to change it.
|
||||
3. **A failure downgraded to a value that looks like a legitimate result** — `-1`, `null`, an
|
||||
empty list, `false` — which a caller then trusts.
|
||||
4. **A lock held for one half of a read-modify-write and not the other**, or two collections
|
||||
updated under different locks.
|
||||
5. **A comment or javadoc stating an invariant the code no longer keeps.** Comments are
|
||||
load-bearing in this repo; a stale one has already caused a bug.
|
||||
|
||||
## 5. What you cannot check, and must not claim you did
|
||||
|
||||
- `fleetd/fleetd.yaml` is gitignored and **absent from your worktree**. You cannot read it. If a
|
||||
finding depends on live configuration, name the key and say you could not check it.
|
||||
- `.mcp.json`, `opencode.json` and `.autoenv` in your worktree are neutralised stubs, not the
|
||||
repo's real files.
|
||||
- The `wiki/` submodule pointer is months old. Do not cite it.
|
||||
|
||||
Reporting a fact you took from the lead's brief as something you measured yourself is a false
|
||||
report, even when the fact is correct. Say where each fact came from.
|
||||
|
||||
## 6. The report — what goes in `fleet_reply`
|
||||
|
||||
One block per finding, most severe first:
|
||||
|
||||
```
|
||||
FINDING N — <one line>
|
||||
file:line
|
||||
Path in: <which caller, in which state, reaches this>
|
||||
Direction: <data loss | escalation | silent wrong answer | outage | ...>
|
||||
Window/trigger: <when it actually happens>
|
||||
Confidence: <confirmed by reading | unproven — say what you could not check>
|
||||
Why nothing else catches it: <the guard or test you checked, and why it misses>
|
||||
```
|
||||
|
||||
End with one line naming every file you read, so the lead knows the denominator.
|
||||
|
||||
**Nothing clears the bar?** Reply `NO FINDINGS`, name the files you read, and say what you ruled
|
||||
out. A clean sweep is a valid result; an invented defect is worse than none.
|
||||
@@ -39,6 +39,19 @@ a worker made all 59 of its edits in the primary's tree and never noticed.
|
||||
test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-toplevel)"
|
||||
```
|
||||
|
||||
**Never run `git stash` (or `git stash pop`/`apply`/`drop`).** Your worktree is isolated, but the
|
||||
stash is **not**: `refs/stash` is one stack shared by the primary's checkout and every other
|
||||
worker's worktree of this repo. Measured on 2026-09-04 — `git stash list` from a worker's worktree
|
||||
and from the primary's tree returned byte-identical output. So a `git stash` you run can be popped
|
||||
into someone else's tree, and a `git stash pop` you run can drop **another worker's** uncommitted
|
||||
edits on top of yours. This has already happened here: two workers were running in parallel and one
|
||||
of them had its in-progress edit silently overwritten by the other's stash.
|
||||
|
||||
The branch is your isolation, so use it instead. To set work aside, commit it on your own branch
|
||||
(`git commit -m "wip: ..."`) and carry on; to try something and back out, use
|
||||
`git diff > /tmp/<your-branch>.patch` then `git checkout -- <file>`. Both stay inside your worktree.
|
||||
If you find a stash entry you did not create, leave it alone and say so in your report.
|
||||
|
||||
## 2. Implement
|
||||
|
||||
- Implement exactly the scope the lead named. Keep the diff focused; note anything out of scope
|
||||
|
||||
@@ -10,6 +10,11 @@ never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and alrea
|
||||
skill is only the *review procedure*: how to work the scope, and the exact shape of what you
|
||||
send back.
|
||||
|
||||
**Wrong skill for a sweep.** This one reviews *one* diff or scope and reports the *single* most
|
||||
important issue. If the lead asked you to hunt or audit a whole package for several defects, load
|
||||
`hunter` instead and ignore this file — the two want different output, and following both is how a
|
||||
worker ends its turn with a good report that never gets sent.
|
||||
|
||||
## 1. Read the whole scope before you judge
|
||||
|
||||
The delegation names your scope — a file, a diff, a PR, a function. **Read all of it first.**
|
||||
|
||||
@@ -7,6 +7,14 @@
|
||||
> wiki ([Use Cases](https://git.ltms.dev/fleet/fleetd/wiki/7-Use-Cases) → *The portable
|
||||
> CLAUDE.md block*); improvements go to the template first, then out to each project. Anything
|
||||
> specific to *this* repo lives under §Project addendum below, never inline above it.
|
||||
>
|
||||
> **Anything you measure in an addendum is perishable.** Date it, give the command that
|
||||
> re-measures it and what each outcome means, and tell the reader to delete the section once
|
||||
> it stops reproducing. The four parts work together: deciding what would falsify a claim is
|
||||
> the expensive step, and a reader in the middle of another task will not pay it, so a bare
|
||||
> "verify before relying on this" costs the same space and does nothing. The case this is for
|
||||
> is a note that goes stale as a live restriction — it will tell a future session it cannot do
|
||||
> the thing at the moment doing it becomes the job.
|
||||
|
||||
If no `fleet_*` MCP tools are mounted in this session, this section does not apply — skip it.
|
||||
|
||||
@@ -71,8 +79,11 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
the final judgment call, verification, merges, and anything that depends on context only you
|
||||
hold. Nothing else is yours by default.
|
||||
3. **Spawn every delegated unit first** — `fleet_spawn{profile, worktree:true, ticket}`, one per
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model and cost, not in
|
||||
tier, so the default is rarely what you want.
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model, cost and
|
||||
LIVENESS, not in tier, so the default is rarely what you want. The default is whatever the
|
||||
daemon reports, and on a host where it sits on an exhausted or withdrawn credential every
|
||||
unqualified spawn fails — sometimes loudly, sometimes as a member that spawns fine and then
|
||||
produces nothing. `fleet_profiles` reports the default; check it once per session.
|
||||
4. **Then send them all** — `fleet_send{sessionId, content, wait:false}`. Line 1 of every brief is
|
||||
`Load the <name> skill.` naming the worker's playbook; those skills are opt-in and that line is
|
||||
what makes them reliable. Where the project ships no such skill, spell the procedure out in the
|
||||
@@ -94,6 +105,18 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
8. **Adjudicate, merge, tear down — yours alone.** Read the diff yourself: fully if it is small,
|
||||
targeted at the reported findings and the risky paths if it is large. Reviewer findings direct
|
||||
your attention; they never substitute for it. Then merge, then `fleet_stop{paneId}`.
|
||||
**If the forge refuses you the merge** — a protected branch, a token without the grant — the
|
||||
adjudication is still yours. Read the diff, decide, and hand the operator a merge-ready queue
|
||||
with the refusal quoted. Never report a PR as merged, and never call one "ready to merge"
|
||||
without having read the diff yourself. A refusal is exactly when that shortcut is tempting,
|
||||
because no action is left that forces you to look, and taking it turns this step into
|
||||
forwarding a reviewer's verdict — which is delegating the merge by proxy, two lines above.
|
||||
**Test a refusal; do not read it off a permissions field.** A protected branch holds its merge
|
||||
rights separately from the repository permissions, so that field can say yes while the merge is
|
||||
refused, and still say no after a grant makes it work. Probe instead, with a request that cannot
|
||||
succeed on its merits, so a rejection can only mean the refusal. Treat a transport failure as a
|
||||
third answer that proves nothing: a timeout, a DNS error or a bad URL is not a refusal, and
|
||||
counting it as one makes you sure of something you never measured.
|
||||
|
||||
**Steps 3 and 4 are separate on purpose** — spawning and sending in one loop is how parallel work
|
||||
silently becomes serial, and it is the most common way this layer is wasted. For the same reason,
|
||||
@@ -140,6 +163,11 @@ The traffic between leads is coordination and nothing else:
|
||||
3. **Verify a peer exactly as you verify yourself.** Peer status buys nothing: check the claim
|
||||
against the code, and re-run the build. A peer's correction gets the same treatment — right or
|
||||
wrong on the evidence, not on who said it. Neither of you merges the other's work unreviewed.
|
||||
4. **Ask a peer to read your project addendum.** Your addendum is instruction surface: every future
|
||||
session on your host obeys it, and a wrong one is obeyed just as faithfully as a right one. The
|
||||
author is the worst reader of their own qualifier placement — measured here, one addendum carried
|
||||
two defects and a non-author found both. If you have no peer, at least re-read it asking "which
|
||||
sentence goes false first, and would a reader reach the caveat before acting?"
|
||||
|
||||
Being messaged by a peer does not make you its worker: answer with `fleet_reply`, and push back on
|
||||
the substance if it is wrong. A peer that simply complies has thrown away the reason there are two of
|
||||
@@ -200,11 +228,27 @@ must obey belongs in the charter, not here.
|
||||
adapter, with a message naming the credential and the remaining seconds ("cooling off after
|
||||
repeated backend errors") — distinct wording from a quarantine refusal, so don't conflate the
|
||||
two when reading a spawn failure.
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR) and
|
||||
`reviewer` (scoped review → one structured finding). Name one in every delegation.
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR),
|
||||
`reviewer` (one diff → one structured finding) and `hunter` (sweep a package → several ranked
|
||||
findings, change nothing). Name exactly one in every delegation. **`reviewer` and `hunter` are
|
||||
not interchangeable** — `reviewer` caps the answer at one finding in about 90 words, so naming
|
||||
it for a multi-finding sweep hands the worker two contradictory output contracts. That has
|
||||
already cost three workers' turns: each wrote a good report to its terminal and ended the turn
|
||||
with no `fleet_reply`, and the scrape returned the tail of the brief instead.
|
||||
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace) and
|
||||
`fleets-status` (report every fleet that shares one LavinMQ instance).
|
||||
- **This repo is also a Claude Code marketplace, and ships a plugin.** `.claude-plugin/marketplace.json`
|
||||
points at `plugin/`, which carries the MCP mount and the `setup` skill
|
||||
(`/claude-bridge:setup` — make any project bridge-ready). It was added in CB-527 and then went
|
||||
unmentioned by every instruction file, so it drifted and a later session planned it from scratch
|
||||
(#362). **Read `plugin/` before designing anything about onboarding a project.** Two limits are
|
||||
structural, not bugs: a plugin cannot carry the role agent files, because
|
||||
`ClaudeCodeLauncher.java:371` requires `<cwd>/.claude/agents/<role>.md` in the member's own
|
||||
worktree; and a plugin cannot deliver anything to members at all, because
|
||||
`ClaudeCodeLauncher.java:285` exports `CLAUDE_CONFIG_DIR` and every Claude profile here sets it,
|
||||
so a member never reads the operator's plugin store. **The plugin is the lead-side surface;
|
||||
member-facing assets travel in the worktree.**
|
||||
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
||||
(a submodule with its own remote).
|
||||
- **A provisioned worktree neutralizes `.mcp.json`, `opencode.json` and `.autoenv`** — the repo's
|
||||
@@ -215,11 +259,14 @@ must obey belongs in the charter, not here.
|
||||
reads the list with `git config --worktree --get-all fleet.neutralizedConfig`, and the
|
||||
consequence with `git config --worktree --get fleet.neutralizedConfigNote`. Never brief a worker
|
||||
to edit one of these files: the edit cannot be committed, and it will not tell you so.
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, turn-done fallback —
|
||||
are diagrammed in `docs/MCP-Contract.md` **§6 only**. The rest of that page is a pre-build design
|
||||
doc whose tool names, parameter names and REST paths never caught up with the code, so do not use
|
||||
it as the tool reference (CB-609). Section 6 is kept out of this file because this file loads into
|
||||
every session's context.
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, the turn-done
|
||||
fallback and status gating — are diagrammed in `docs/MCP-Contract.md`. That page is now flows
|
||||
only: its pre-build tool catalogue, parameter tables and REST paths were deleted rather than
|
||||
corrected, because a hand-maintained second copy of the tool surface is what drifted for a month
|
||||
while this line pointed every session at it (CB-609 / #114). **The live MCP schema is the tool
|
||||
reference**, with the intent→tool table above as the short form. `McpContractDocTest` fails if
|
||||
that page names a `fleet_*` tool the server does not register. The flows are kept out of this
|
||||
file because this file loads into every session's context.
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
|
||||
+54
-30
@@ -1,56 +1,80 @@
|
||||
# CB-504 — systemd unit for fleetd (Linux).
|
||||
#
|
||||
# The macOS launchd agent (deploy/dev.ltms.fleetd.plist) is the supervision target for the
|
||||
# current single-host deployment. This unit exists for the per-host gateways CB-308 introduces,
|
||||
# which will run on Linux.
|
||||
# fleetd #360: the previous version of this file started clean and broke the daemon in three ways
|
||||
# that nothing logs (see the DO NOT block and the ExecStart/PrivateTmp comments below for what and
|
||||
# why). The unit below, plus its companion deploy/herdr.service, is the version that has actually
|
||||
# run on fleet01 without those failures. Do not "improve" it back toward the old shape without
|
||||
# re-reading why each line is the way it is.
|
||||
#
|
||||
# Install (user service — fleetd drives the user's herdr, not a system daemon):
|
||||
# mkdir -p ~/.config/systemd/user
|
||||
# cp deploy/fleetd.service ~/.config/systemd/user/
|
||||
# # edit ExecStart / WorkingDirectory / Environment below, then:
|
||||
# cp deploy/fleetd.service deploy/herdr.service ~/.config/systemd/user/
|
||||
# # edit WorkingDirectory / ExecStart below for your host's paths and java location
|
||||
# systemctl --user daemon-reload
|
||||
# systemctl --user enable --now fleetd
|
||||
# systemctl --user enable --now herdr fleetd
|
||||
# loginctl enable-linger $USER # REQUIRED -- see below
|
||||
# journalctl --user -u fleetd -f
|
||||
#
|
||||
# `loginctl enable-linger` is not optional and is easy to miss, because leaving it out looks like
|
||||
# success: `systemctl --user enable` reports "enabled" and both units run for as long as you stay
|
||||
# logged in. A user manager without lingering starts at your first login and stops at your last
|
||||
# logout, so the fleet simply does not come back after a reboot -- which is the whole reason to
|
||||
# use systemd here rather than the setsid scripts these units replaced. Check it with
|
||||
# `loginctl show-user $USER -p Linger`; the answer must be `Linger=yes`.
|
||||
#
|
||||
# Secrets (AI_GATEWAY_TOKEN, WORKER_GITEA_TOKEN, LAVINMQ_URI, COORD_AMQP_URI, ...) are not set
|
||||
# here and need no systemd drop-in: ExecStart runs a login shell, so they come from wherever your
|
||||
# login shell already sources them (this host: ~/.fleet/secrets.sh via ~/.zprofile). If a token is
|
||||
# missing there, fleetd still starts — the daemon reports every secret a configured profile
|
||||
# references, by name, never by value:
|
||||
# journalctl --user -u fleetd | grep 'startup secret'
|
||||
# A resolved one logs "startup secret NAME: set (profile 'x' tokenEnv)"; a missing one logs
|
||||
# "startup secret NAME: MISSING" at WARN and the daemon starts anyway — the first visible symptom
|
||||
# is a member that cannot open a pull request, hours later and in a different component.
|
||||
|
||||
[Unit]
|
||||
Description=fleetd — claude-bridge message server
|
||||
Description=fleetd — fleet message server
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
# Ordering only: herdr is a user process and its socket may appear after us. This is advisory —
|
||||
# fleetd retries the herdr socket rather than exiting, which is what actually makes a late
|
||||
# socket survivable. Do NOT add Requires=: a herdr restart must not take fleetd down with it.
|
||||
# Ordering only. fleetd retries the herdr socket rather than exiting, which is what actually makes
|
||||
# a late socket survivable. Do NOT add Requires=: a herdr restart must not take fleetd down too.
|
||||
After=herdr.service
|
||||
Wants=herdr.service
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
WorkingDirectory=%h/src/claude-bridge/fleetd
|
||||
ExecStart=/usr/lib/jvm/temurin-25-jdk/bin/java -jar target/fleetd.jar fleetd.yaml
|
||||
WorkingDirectory=%h/LTMS/fleetd/fleetd
|
||||
|
||||
Environment=HERDR_SOCKET_PATH=%h/.config/herdr/herdr.sock
|
||||
# PATH matters more than it looks (CB-511): fleetd propagates its own PATH to every worker it
|
||||
# spawns, so this line decides whether the fleet can run a build at all. systemd does not source a
|
||||
# login shell, so without it the daemon — and every worker — gets a bare default with no JDK/Maven.
|
||||
Environment=PATH=/usr/lib/jvm/temurin-25-jdk/bin:/usr/share/maven/bin:/usr/local/bin:/usr/bin:/bin
|
||||
# Secrets are NOT set here — this file is committed. Put the API/worker tokens in a private
|
||||
# drop-in that systemd reads with restrictive permissions:
|
||||
# systemctl --user edit fleetd → [Service] / Environment=FLEETD_API_TOKEN=...
|
||||
# or point EnvironmentFile at a 0600 file:
|
||||
# EnvironmentFile=%h/.config/fleetd/env
|
||||
# A LOGIN shell, not java directly. Every secret this daemon needs (AI_GATEWAY_TOKEN,
|
||||
# WORKER_GITEA_TOKEN, LAVINMQ_URI, COORD_AMQP_URI) lives in ~/.fleet/secrets.sh, which only
|
||||
# ~/.zprofile sources. systemd runs no login shell. Started any other way the daemon boots fine
|
||||
# and looks healthy, and the failure appears hours later as a member that cannot open a pull
|
||||
# request. exec keeps it one process, so systemd tracks the right PID.
|
||||
# This also avoids a SECOND copy of the secrets in a systemd drop-in: one source of truth.
|
||||
ExecStart=/bin/zsh -lc "exec java -jar target/fleetd.jar fleetd.yaml"
|
||||
|
||||
# PrivateTmp MUST stay false -- see herdr.service. fleetd creates the member ZDOTDIR scrub dir and
|
||||
# the opencode config dir under java.io.tmpdir, and the member pane (a herdr child, a different
|
||||
# unit) has to read them. A private /tmp turns the credential scrub into a silent no-op.
|
||||
PrivateTmp=false
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=10s
|
||||
# A bad config (e.g. a non-loopback bind without token auth) makes fleetd fail fast by design.
|
||||
# Give up rather than restart-loop on a permanent error.
|
||||
# A bad config makes fleetd fail fast by design. Give up rather than restart-loop forever.
|
||||
StartLimitBurst=5
|
||||
StartLimitIntervalSec=120
|
||||
|
||||
# The daemon reads the repo, writes worktrees, and talks to a Unix socket — it needs no more.
|
||||
# DO NOT add ProtectSystem=, ProtectHome=, ProtectKernelTunables= or ProtectControlGroups=.
|
||||
# Measured on fleet01 2026-09-05: each of those gives the unit its own mount namespace, and
|
||||
# fleetd resolves a caller role by running lsof to find the loopback peer PID
|
||||
# (mcp/LsofPeerPidLookup). Inside such a namespace lsof returns nothing, every caller falls back
|
||||
# to ANONYMOUS, and the primary is refused every orchestration call with
|
||||
# "unauthenticated: anonymous may not SPAWN".
|
||||
# The daemon still starts, healthz still returns ok and the secrets still resolve - the only
|
||||
# symptom is that the fleet cannot be driven at all. Verified by bisecting the directives:
|
||||
# no sandbox 3 lsof lines | ProtectSystem=strict 0 | ProtectHome=read-only 0
|
||||
# ProtectKernelTunables 0 | ProtectControlGroups 0 | RestrictSUIDSGID 3 | NoNewPrivileges 3
|
||||
# The two below add no mount namespace and are safe.
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectSystem=strict
|
||||
ProtectHome=read-write
|
||||
ProtectKernelTunables=true
|
||||
ProtectControlGroups=true
|
||||
RestrictSUIDSGID=true
|
||||
|
||||
StandardOutput=journal
|
||||
|
||||
Executable
+19
@@ -0,0 +1,19 @@
|
||||
#!/bin/zsh
|
||||
# fleetd #360 — template for the script deploy/herdr.service's ExecStart wraps in a pty.
|
||||
#
|
||||
# `script -qfec <this> /dev/null` needs a real command to run, and that command has to be a LOGIN
|
||||
# shell script: herdr itself needs the same secrets fleetd.service's login shell picks up (this
|
||||
# host: ~/.fleet/secrets.sh via ~/.zprofile), because members it spawns inherit its environment.
|
||||
# systemd's own Environment= lines in herdr.service are not enough for that -- they set TERM and a
|
||||
# bare PATH so the pty starts at all, nothing more.
|
||||
#
|
||||
# Copy this file to the path deploy/herdr.service's ExecStart names
|
||||
# (%h/LTMS/fleetd/fleetd-run/herdr-inner.sh by default) and `chmod +x` it. Not committed under
|
||||
# that path itself because the session name below is host-specific.
|
||||
|
||||
# A 0x0 pty makes every pane spawn fail with "ghostty error -2" (see herdr-multi-instance-facts /
|
||||
# fleet01-headless-herdr-standup) -- give it a real size before herdr ever touches it.
|
||||
stty rows 50 cols 200
|
||||
|
||||
# -l: login shell, so herdr and everything it spawns gets the real secrets and PATH.
|
||||
exec zsh -lc 'exec herdr --session <name>'
|
||||
@@ -0,0 +1,40 @@
|
||||
# fleetd #360 — systemd unit for herdr (Linux), the terminal multiplexer fleetd drives.
|
||||
#
|
||||
# This is fleetd.service's companion: fleetd.service's After=/Wants=herdr.service assumes this
|
||||
# unit exists. Before this ticket it did not, so on a fresh host fleetd started against a herdr
|
||||
# that systemd never supervised at all.
|
||||
#
|
||||
# Install: see deploy/fleetd.service's header comment (both units install the same way).
|
||||
#
|
||||
# ExecStart below runs deploy/herdr-inner.sh (copy the template of that name from this directory
|
||||
# to the path in ExecStart, or point ExecStart at wherever you keep it, and make it executable).
|
||||
# It is a separate file rather than an inline command because it must itself be a login shell (see
|
||||
# its own header for why) and systemd's ExecStart does not run one.
|
||||
|
||||
[Unit]
|
||||
Description=herdr terminal multiplexer (fleet session)
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
# script(1) gives herdr a real pty. Without it the client reports a 0x0 window and every pane
|
||||
# spawn fails with "ghostty error -2" -- which surfaces as a fleetd spawn failure, not a herdr one.
|
||||
ExecStart=/usr/bin/script -qfec %h/LTMS/fleetd/fleetd-run/herdr-inner.sh /dev/null
|
||||
StandardInput=null
|
||||
Environment=TERM=xterm-256color
|
||||
Environment=PATH=%h/.local/bin:/usr/local/bin:/usr/bin:/bin
|
||||
|
||||
# PrivateTmp MUST stay false. fleetd writes the member ZDOTDIR scrub dir and the opencode config
|
||||
# dir under its own java.io.tmpdir, and the member pane -- a child of THIS process -- has to read
|
||||
# them. A private /tmp here silently breaks the credential scrub instead of failing loudly.
|
||||
PrivateTmp=false
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=5s
|
||||
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=herdr
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -192,7 +192,7 @@ Deliberately small; every one maps to a failure mode we have actually hit.
|
||||
| `fleet_send_duration_seconds` | histogram | delegated turn latency |
|
||||
| `fleet_replies_total{path}` | counter | path ∈ rendezvous\|inbox — how often a reply strands (CB-307's whole reason to exist) |
|
||||
| `fleet_inbox_depth{target}` | gauge | undrained replies; steady-state should be 0 |
|
||||
| `fleet_push_nudges_total{outcome}` | counter | outcome ∈ delivered\|exhausted — a rising `exhausted` means the primary is not draining |
|
||||
| `fleet_push_nudges_total{outcome}` | counter | outcome ∈ sent\|exhausted — a rising `exhausted` means the primary is not draining. `sent` was called `delivered` until fleetd #365; it counts the herdr paste-and-submit call returning, never a confirmation the pane read it |
|
||||
| `fleet_spawns_total{kind,outcome}` | counter | outcome ∈ ready\|timeout\|guard_rejected; per peer kind (CB-402) |
|
||||
| `fleet_sessions{state}` | gauge | SPAWNING/READY/BUSY/DONE census |
|
||||
| `fleet_herdr_calls_total{method,outcome}` | counter | socket health — the dependency everything rests on |
|
||||
|
||||
+123
-324
@@ -1,311 +1,162 @@
|
||||
# MCP Contract — `fleetd`'s unified gateway
|
||||
# MCP flows and error model — `fleetd`
|
||||
|
||||
> **Status: 🔴 HISTORICAL DESIGN — do NOT use as the tool reference.** Written 2026-07-14, before
|
||||
> any MCP code existed. The system shipped and this page never caught up, so **its tool names,
|
||||
> parameter names and REST paths are wrong today**. Audited 2026-08-17; the specific drift:
|
||||
> **What this page is.** The **flows**: how a delegation, a clarification, a detached task and a
|
||||
> silent member each travel through `fleetd`. These shapes are what shipped, and they are hard to
|
||||
> read off the code because they span the MCP face, the rendezvous registry, the `Injector` and
|
||||
> herdr.
|
||||
>
|
||||
> - **Tools it names that do not exist:** `fleet_read`, `fleet_cancel`.
|
||||
> - **Shipped tools it omits:** `fleet_poll`, `fleet_ack`, `fleet_profiles`, `fleet_whoami`.
|
||||
> - **Parameter names are wrong nearly everywhere** — it says `message`/`target`/`timeout_seconds`/
|
||||
> `block` where the code takes `content`/`sessionId`/`timeoutMs`/`wait`; `text` where
|
||||
> `fleet_reply` takes `content`; `target` where `fleet_stop` takes `paneId`.
|
||||
> - **REST paths are wrong:** it says `POST /workers` and `DELETE /workers/{paneId}`; the daemon
|
||||
> serves `POST /members` and `DELETE /members/{paneId}`.
|
||||
> **What this page is NOT: a tool reference.** It deliberately holds no tool catalogue, no
|
||||
> parameter tables and no REST paths. **The live MCP schema is the authority** — each tool's own
|
||||
> description and parameters, as mounted — with the intent→tool table in `CLAUDE.md` as the short
|
||||
> form.
|
||||
>
|
||||
> **The authoritative tool surface is the live MCP schema** (each tool's own description and
|
||||
> parameters, as mounted), with the intent→tool table in `CLAUDE.md` as the short form. Both were
|
||||
> checked against `mcp/FleetMcp.java` on 2026-08-17 and are accurate.
|
||||
> That absence is the fix for fleetd #114 (CB-609), and it is worth stating why. This page used to
|
||||
> carry a full tool catalogue written in July 2026, before any MCP code existed. The code shipped;
|
||||
> the page did not follow. By August it named two tools that do not exist, omitted five that do,
|
||||
> had the wrong name for nearly every parameter, pointed at REST paths the daemon does not serve,
|
||||
> and — worst — still described an identity model (*"any connection that does not map to a known
|
||||
> worker is treated as a primary"*) that was a real privilege bug, fixed since by the ancestry
|
||||
> walk in fleetd #161. Every one of those errors is the same error: **a second, hand-maintained
|
||||
> copy of something the code already states**. So the second copy is gone rather than corrected.
|
||||
> Only the flows remain, because a flow is a shape rather than a name, and shapes are what this
|
||||
> page was ever good for.
|
||||
>
|
||||
> What is still worth reading here is **§6 — the flows and the error model** (rendezvous,
|
||||
> `fleet_ask`, detached delivery, the turn-done fallback). The shapes it describes are the ones
|
||||
> that shipped; only the names around them drifted. Rewriting this page is tracked as **CB-609**.
|
||||
|
||||
`fleetd` is the **sole communication gateway** for every Claude session in the bridge. Both
|
||||
the **primary** (Opus, on subscription) and every **worker** (off-subscription Claude Code)
|
||||
mount the *same* MCP server with a single `claude mcp add` line, and talk only through its
|
||||
tools. No Claude session ever addresses a broker, a peer, or the network directly.
|
||||
|
||||
This document defines every MCP tool that face must expose, who may call it, its blocking
|
||||
semantics, and how it maps onto the code already in the tree.
|
||||
> The names that do appear below are checked by `McpContractDocTest`, which fails if this page
|
||||
> names a `fleet_*` tool the server does not register. That test is the whole reason it is safe to
|
||||
> write a tool name here at all.
|
||||
|
||||
---
|
||||
|
||||
## 1. Design constraints (non-negotiable)
|
||||
## 1. Rendezvous flows
|
||||
|
||||
These come from the project's core invariants and bound every decision below.
|
||||
### 1.1 Delegation — happy path
|
||||
|
||||
1. **One server, both roles.** The primary and all workers mount an identical server. The
|
||||
catalog must serve both, and `fleetd` must decide *who is calling* from the connection —
|
||||
never from a caller-supplied argument that could be spoofed.
|
||||
2. **Subscription-safe by construction.** No MCP tool ever reads, sets, or forwards
|
||||
`ANTHROPIC_BASE_URL`. Mounting the bridge cannot move a session off subscription.
|
||||
Enforced today by [`SubscriptionGuard`](1-Architecture).
|
||||
3. **Blocking rendezvous, no busy-poll.** The primary consumes a worker's reply through a
|
||||
*single* MCP call that `fleetd` holds open — never a cross-turn poll loop that would burn
|
||||
subscription quota.
|
||||
4. **Status-gated delivery.** Anything that puts text into a worker flows through the existing
|
||||
[`Injector`](1-Architecture): delivered only when the worker is `idle`/`blocked`, at most
|
||||
one message per turn.
|
||||
5. **`fleetd` owns policy; herdr owns PTYs.** MCP tools express *intent*; `fleetd`
|
||||
translates it into guard checks, rendezvous bookkeeping, and herdr `agent.*` calls.
|
||||
|
||||
---
|
||||
|
||||
## 2. Topology
|
||||
|
||||
Both faces live in the one daemon. The **north face** is MCP (this document); the **south
|
||||
face** is the herdr Unix socket. REST/SSE remains only for non-Claude clients and dashboards.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
OPUS["Opus — primary<br/>(Claude Code, env CLEAN)<br/>MCP client"]
|
||||
subgraph BD["fleetd — standalone daemon"]
|
||||
MCP["MCP server (north face)<br/>fleet_send · fleet_reply<br/>fleet_ask · fleet_status · lifecycle"]
|
||||
RDV["rendezvous registry<br/>(blocking-call waiters)"]
|
||||
INJ["Injector + StatusPoller<br/>(status-gated writer)"]
|
||||
SOCK["herdr socket client (south face)"]
|
||||
MCP --> RDV
|
||||
RDV --> INJ
|
||||
INJ --> SOCK
|
||||
MCP --> SOCK
|
||||
end
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
|
||||
OPUS -->|"fleet_send (blocks)"| MCP
|
||||
W -.->|"fleet_reply / fleet_ask"| MCP
|
||||
SOCK -->|"agent.start · agent.send<br/>agent.get · pane.close"| HERDR
|
||||
HERDR -->|"drives PTY"| W
|
||||
|
||||
classDef ext fill:#2b6cb0,stroke:#1a365d,color:#ffffff;
|
||||
classDef core fill:#2f855a,stroke:#22543d,color:#ffffff;
|
||||
class OPUS,W ext
|
||||
class MCP,RDV,INJ,SOCK core
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. Identity & addressing
|
||||
|
||||
Because the same server is mounted by everyone, `fleetd` resolves the caller's role on every
|
||||
request — this is the linchpin of the whole contract and has no code yet.
|
||||
|
||||
- **Workers are known.** `fleetd` spawns every worker
|
||||
([`WorkerService`](1-Architecture)) and records its herdr session UUID / `terminal_id` on
|
||||
the returned [`Agent`]. When a call arrives on a connection that maps to a known worker,
|
||||
the caller is *that* worker — so **workers never pass a target**; routing is implicit.
|
||||
- **The primary is "not a worker".** Any connection that does not map to a known worker is
|
||||
treated as a primary. It addresses workers **explicitly** by `target` — a session UUID,
|
||||
a `terminal_id`, or a friendly `profile` name.
|
||||
- **Turn correlation.** A blocking `fleet_send` registers a *waiter* keyed by worker
|
||||
identity. A worker's later `fleet_reply` / `fleet_ask` on the same identity resolves that
|
||||
waiter. A `turn_id` is minted per exchange so a clarification round-trip
|
||||
(§6.2) rejoins the right turn.
|
||||
|
||||
---
|
||||
|
||||
## 4. Transport
|
||||
|
||||
`fleetd` is a long-lived daemon serving **multiple** concurrent clients (one primary + N
|
||||
workers), so a per-client stdio child is the wrong shape. The recommended transport is
|
||||
**streamable-HTTP / SSE** on the same bind as the REST face:
|
||||
|
||||
```bash
|
||||
# identical on primary and every worker
|
||||
claude mcp add --transport http fleetd http://127.0.0.1:8080/mcp
|
||||
```
|
||||
|
||||
This adds an MCP-server dependency the pom does not yet carry. See [Open decisions](#10-open-decisions).
|
||||
|
||||
---
|
||||
|
||||
## 5. Tool catalog
|
||||
|
||||
| Tool | Caller | Blocks? | Backing (exists today?) |
|
||||
|---|---|---|---|
|
||||
| [`fleet_send`](#fleet_send) | primary | yes (default) | `Injector.enqueue` ✅ · rendezvous registry ❌ (CB-104) |
|
||||
| [`fleet_reply`](#fleet_reply) | worker | no | rendezvous ❌ · pane injection via `Injector` ✅ |
|
||||
| [`fleet_ask`](#fleet_ask) | worker | yes | reverse rendezvous ❌ |
|
||||
| [`fleet_status`](#fleet_status) | either | no | `AgentControl.status` ✅ · `Injector.activeTargets` ✅ |
|
||||
| [`fleet_spawn`](#lifecycle) | primary | no | `WorkerService.spawn` ✅ (`POST /workers`) |
|
||||
| [`fleet_list`](#lifecycle) | either | no | `WorkerService.list` ✅ (`/agents`) |
|
||||
| [`fleet_stop`](#lifecycle) | primary | no | `WorkerService.stop` ✅ (`DELETE /workers/{paneId}`) |
|
||||
| [`fleet_read`](#fleet_read) | primary | no | `AgentControl.read` ✅ |
|
||||
| [`fleet_cancel`](#fleet_cancel) | primary | no | — ❌ (future) |
|
||||
|
||||
### Core: delegation & rendezvous
|
||||
|
||||
#### `fleet_send`
|
||||
*(primary → worker — the headline tool, CB-104)*
|
||||
|
||||
- **Params:** `message` (required); `target` (optional — defaults to the sole worker / default
|
||||
profile); `timeout_seconds` (default 600); `block` (default `true`); `auto_spawn`
|
||||
(default `true`); `turn_id` (optional — supplied when answering a worker's `fleet_ask`).
|
||||
- **Blocking (`block:true`):** enqueue `message` via the `Injector`, then hold the call open
|
||||
until exactly one of:
|
||||
- worker calls `fleet_reply` → `{ outcome:"reply", text }`
|
||||
- worker calls `fleet_ask` → `{ outcome:"question", text, turn_id }`
|
||||
- worker's `agent_status` reaches done/idle with no reply → `{ outcome:"turn_done", text:<terminal tail> }`
|
||||
- deadline elapses → `{ outcome:"timeout" }`
|
||||
- worker gone → error `worker_gone`
|
||||
- **Detached (`block:false`):** enqueue and return `{ outcome:"dispatched", dispatch_id }`
|
||||
immediately. The eventual reply is injected into the primary's idle pane (§6.3), or drained
|
||||
via `fleet_status` on a split-host primary.
|
||||
|
||||
#### `fleet_reply`
|
||||
*(worker → primary)*
|
||||
|
||||
- **Params:** `text` (required); `final` (default `true`).
|
||||
- **Behavior:** resolve the primary waiter registered against this worker with `text`. If no
|
||||
waiter exists (detached delegation), `fleetd` **injects the primary's idle pane** instead.
|
||||
Returns `{ delivered:true, mode:"resolved"|"injected" }`. No `target` — identity is implicit.
|
||||
|
||||
#### `fleet_ask`
|
||||
*(worker → primary — the reverse rendezvous)*
|
||||
|
||||
- **Params:** `question` (required); `timeout_seconds`.
|
||||
- **Behavior:** blocks the *worker's* call. Surfaces the question to the primary (resolving its
|
||||
open `fleet_send` with `outcome:"question"`, or injecting its pane). When the primary
|
||||
answers — a `fleet_send` carrying the matching `turn_id` — that unblocks this call and
|
||||
returns `{ answer }` to the worker, which continues **in the same turn**.
|
||||
|
||||
### Worker lifecycle
|
||||
<a id="lifecycle"></a>
|
||||
Thin adapters over [`WorkerService`](1-Architecture) — parity with the existing REST routes.
|
||||
|
||||
- **`fleet_spawn`** — `{ profile? }` → worker view (`sessionId`, `terminalId`, `paneId`,
|
||||
`status`). Guard-checked; a boundary breach returns error `subscription_boundary` (the
|
||||
REST `403`).
|
||||
- **`fleet_list`** — no params → all workers + `agent_status`. Read-only, either role.
|
||||
- **`fleet_stop`** — `{ target }` → tears down the pane and its dedicated tab. Idempotent.
|
||||
|
||||
### Observability
|
||||
|
||||
#### `fleet_status`
|
||||
*(either role — the README's 4th named tool)*
|
||||
|
||||
- **Params:** `target?`.
|
||||
- **Behavior:** per-worker `agent_status`, queue depth (`Injector.activeTargets`), whether a
|
||||
rendezvous is open, and ids. For the *calling* session it also reports/drains **pending
|
||||
messages addressed to me** — the path a split-host primary's `Stop`-hook uses to wake and
|
||||
collect replies without being injectable. Read-only, non-blocking.
|
||||
|
||||
#### `fleet_read`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`; `source` ∈ `visible | recent | recent_unwrapped | detection`.
|
||||
- **Behavior:** returns the worker's terminal text so the primary can peek at a *detached*
|
||||
worker's progress. Adapter over `AgentControl.read`.
|
||||
|
||||
### Control (future)
|
||||
|
||||
#### `fleet_cancel`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`. Interrupt the worker's current turn / abandon the rendezvous. No
|
||||
backing code yet.
|
||||
|
||||
---
|
||||
|
||||
## 6. Rendezvous flows
|
||||
|
||||
### 6.1 Delegation — happy path
|
||||
|
||||
One blocking call, zero polls.
|
||||
One blocking call, zero polls. The lead's call is held open by `fleetd` until the member answers.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary (Opus)
|
||||
participant B as fleetd (MCP + Injector)
|
||||
participant P as "Lead (primary)"
|
||||
participant B as "fleetd (MCP + Injector)"
|
||||
participant H as herdr
|
||||
participant W as Worker (Claude)
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B->>B: register waiter(w)
|
||||
B->>H: agent.send(w, "do X") (idle window)
|
||||
H-->>W: prompt injected
|
||||
W->>W: works the turn
|
||||
W->>B: fleet_reply("result")
|
||||
B->>B: resolve waiter(w)
|
||||
B-->>P: { outcome:"reply", text:"result" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B->>B: "register waiter(sessionId)"
|
||||
B->>H: "agent.send — only in an injectable window"
|
||||
H-->>W: "prompt injected"
|
||||
W->>W: "works the turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B->>B: "resolve waiter"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.2 Clarification — reverse rendezvous (`fleet_ask`)
|
||||
**The cap that matters:** a blocking `fleet_send` is bounded by the *caller's own* MCP client
|
||||
timeout, about 60 seconds — not by the task. Anything slower than that must use the detached flow
|
||||
in §1.3, or the lead's call returns while the member is still working.
|
||||
|
||||
The worker pauses mid-turn to ask; the primary answers; the worker resumes in the same turn.
|
||||
### 1.2 Clarification — reverse rendezvous
|
||||
|
||||
The member pauses mid-turn to ask, the lead answers, and the member resumes **the same turn** with
|
||||
its context intact.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>B: fleet_ask("which config?") — worker blocks
|
||||
B-->>P: { outcome:"question", text:"which config?", turn_id }
|
||||
P->>B: fleet_send("config.yaml", target=w, turn_id) — blocks again
|
||||
B-->>W: resolve fleet_ask → { answer:"config.yaml" }
|
||||
W->>W: resumes same turn
|
||||
W->>B: fleet_reply("done")
|
||||
B-->>P: { outcome:"reply", text:"done" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B-->>W: "content injected"
|
||||
W->>B: "fleet_ask{question} — member blocks"
|
||||
B-->>P: "{ outcome: question, turnId }"
|
||||
P->>B: "fleet_send{turnId, content} — answers THIS turn"
|
||||
B-->>W: "fleet_ask returns the answer"
|
||||
W->>W: "resumes the same turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.3 Detached delegation — pane injection
|
||||
**Answer with `turnId`, never `sessionId`.** A `sessionId` send starts a new turn; it does not
|
||||
resolve the waiting `fleet_ask`.
|
||||
|
||||
The primary does not block; the reply arrives later in its idle pane.
|
||||
**The window is about 55 seconds and no nudge extends it.** So never brief a member to "ask me":
|
||||
decide before delegating, or give the member an explicit default to fall back on.
|
||||
|
||||
### 1.3 Detached delegation — the lead does not block
|
||||
|
||||
The lead gets a ticket immediately and collects the answer later. This is the flow for any real
|
||||
task, because of the ~60s cap in §1.1.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w, block=false)
|
||||
B-->>P: { outcome:"dispatched", dispatch_id }
|
||||
P->>P: continues its own work
|
||||
W->>B: fleet_reply("result")
|
||||
Note over B: no waiter → detached path
|
||||
B->>B: Injector.enqueue(primary_pane, "result")
|
||||
B-->>P: injected into idle pane (status-gated)
|
||||
P->>B: "fleet_send{sessionId, content, wait:false}"
|
||||
B-->>P: "accepted — ticket"
|
||||
P->>P: "continues its own work"
|
||||
W->>B: "fleet_reply{content}"
|
||||
Note over B: "no waiter is blocked — the reply is held"
|
||||
B->>B: "nudge the lead's own pane (status-gated)"
|
||||
P->>B: "fleet_poll{ticket}"
|
||||
B-->>P: "the member's report"
|
||||
P->>B: "fleet_ack{target, msgId}"
|
||||
```
|
||||
|
||||
### 6.4 Uncooperative worker — turn-done fallback
|
||||
A terminal ticket nudges the lead's pane by itself, so a detached task does not need watching. The
|
||||
nudge needs an injectable lead pane and is capped, so it is a convenience rather than a guarantee.
|
||||
|
||||
A worker that never calls `fleet_reply` still returns a result: `fleetd` reads its terminal
|
||||
tail when the turn completes.
|
||||
### 1.4 The member never replies — turn-done fallback
|
||||
|
||||
A member that ends its turn without `fleet_reply` still produces something: `fleetd` reads its
|
||||
pane tail. This is a **fallback, not a channel** — it is lossy in three separate ways, and every
|
||||
one of them has produced a wrong answer in practice.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>W: works, never calls fleet_reply
|
||||
B->>B: StatusPoller sees agent_status → idle/done
|
||||
B->>B: AgentControl.read(w, "recent")
|
||||
B-->>P: { outcome:"turn_done", text:<terminal tail> }
|
||||
P->>B: "fleet_send — blocks or detaches"
|
||||
B-->>W: "content injected"
|
||||
W->>W: "works, never calls fleet_reply"
|
||||
B->>B: "StatusPoller sees the turn end"
|
||||
B->>B: "read the pane tail"
|
||||
B->>B: "classify: exhausted? echoed brief? real report?"
|
||||
B-->>P: "{ outcome: turn_done } or a named failure"
|
||||
```
|
||||
|
||||
The three ways it goes wrong, and what each looks like now:
|
||||
|
||||
| What happened | What the lead used to get | What it gets today |
|
||||
|---|---|---|
|
||||
| The report is longer than the scrape window | The **end** silently cut off | Still clipped, but marked partial |
|
||||
| The member never started — spent credential | The lead's **own brief** echoed back as a report | A named failure: backend exhausted |
|
||||
| The member is simply slow | A tail of work in progress | Unchanged — read it as a hint, not a result |
|
||||
|
||||
The echoed-brief case is the one to remember: it reads as a long, on-topic report with nothing in
|
||||
it from the member. It is suppressed now, but the general rule stands — **check the member's
|
||||
worktree with `git log` before believing a report you did not watch arrive.**
|
||||
|
||||
---
|
||||
|
||||
## 7. Status gating
|
||||
## 2. Status gating
|
||||
|
||||
Delivery only happens in a safe window. This is the state machine the `Injector` already
|
||||
enforces via `AgentStatus.injectable()`; MCP `fleet_send` is simply its producer.
|
||||
Delivery only happens in a safe window. `fleet_send` is a producer for the `Injector`, which
|
||||
already enforces this through `AgentStatus.injectable()`.
|
||||
|
||||
```mermaid
|
||||
stateDiagram-v2
|
||||
[*] --> IDLE
|
||||
IDLE --> WORKING: message delivered / picks up
|
||||
WORKING --> IDLE: turn done
|
||||
WORKING --> BLOCKED: awaits input
|
||||
BLOCKED --> WORKING: input delivered
|
||||
IDLE --> UNKNOWN: detection glitch
|
||||
BLOCKED --> UNKNOWN: detection glitch
|
||||
UNKNOWN --> IDLE: re-detected
|
||||
IDLE --> WORKING: "message delivered, picked up"
|
||||
WORKING --> IDLE: "turn done"
|
||||
WORKING --> BLOCKED: "awaits input"
|
||||
BLOCKED --> WORKING: "input delivered"
|
||||
IDLE --> UNKNOWN: "detection glitch"
|
||||
BLOCKED --> UNKNOWN: "detection glitch"
|
||||
UNKNOWN --> IDLE: "re-detected"
|
||||
|
||||
note right of IDLE
|
||||
injectable — deliver head of FIFO
|
||||
@@ -321,69 +172,17 @@ stateDiagram-v2
|
||||
end note
|
||||
```
|
||||
|
||||
At most one message is delivered per turn: after a send the `Injector` waits for a `WORKING`
|
||||
pickup before delivering the next, with a `PICKUP_GRACE_POLLS` fallback for turns faster than
|
||||
the poll interval. A herdr `events.subscribe` stream can later replace the sampling without
|
||||
touching this state machine.
|
||||
**At most one message per turn.** After a send, the `Injector` waits for a `WORKING` pickup before
|
||||
delivering the next, with a grace-poll fallback for turns that finish faster than the poll
|
||||
interval.
|
||||
|
||||
---
|
||||
Two consequences a lead feels directly:
|
||||
|
||||
## 8. Error model
|
||||
- **A second send to a busy member never lands.** It reports as queued and times out. The member
|
||||
is fine; the message simply waits, and then restarts the member when it next goes idle.
|
||||
- **A spawned member is not deliverable until it has mounted the MCP.** Until then a send waits on
|
||||
that gate for about 60 seconds and then fails without ever reaching the pane.
|
||||
|
||||
| Condition | `fleet_send` result | Notes |
|
||||
|---|---|---|
|
||||
| Worker replies | `{ outcome:"reply" }` | normal |
|
||||
| Worker asks | `{ outcome:"question", turn_id }` | answer with `fleet_send(turn_id)` |
|
||||
| Turn ends, no reply | `{ outcome:"turn_done" }` | terminal tail as text |
|
||||
| Deadline elapsed | `{ outcome:"timeout" }` | message may still be queued/delivered |
|
||||
| Worker vanished | error `worker_gone` | `Injector.drop` fails the queued future |
|
||||
| Guard breach on spawn | error `subscription_boundary` | REST `403` parity |
|
||||
| Delivery failed at herdr | error, message dropped | poisoned message not left blocking the FIFO |
|
||||
|
||||
`fleet_reply` from a worker with no open waiter is **not** an error — it falls through to
|
||||
detached pane injection (§6.3).
|
||||
|
||||
---
|
||||
|
||||
## 9. Mapping to existing code
|
||||
|
||||
The MCP face is a thin adapter layer; nearly every capability already exists behind the REST
|
||||
seam. Only the **rendezvous registry** and the **caller-identity resolver** are new.
|
||||
|
||||
| MCP tool | Existing collaborator | New work |
|
||||
|---|---|---|
|
||||
| `fleet_send` | `Injector.enqueue`, `AgentControl.send` | waiter registry, timeout, outcome mux (CB-104) |
|
||||
| `fleet_reply` / `fleet_ask` | `Injector` (pane injection) | reverse rendezvous, identity resolver |
|
||||
| `fleet_status` | `AgentControl.status`, `Injector.activeTargets` | pending-drain projection |
|
||||
| `fleet_spawn` / `list` / `stop` | `WorkerService.{spawn,list,stop}` | MCP adapter only |
|
||||
| `fleet_read` | `AgentControl.read` | MCP adapter only |
|
||||
|
||||
Because the REST routes in `FleetApp` already exercise the collaborators, MCP tools are
|
||||
validated by **parity** against those routes, not by re-testing behavior.
|
||||
|
||||
---
|
||||
|
||||
## 10. Open decisions
|
||||
|
||||
1. **`fleet_ask` direction.** This page defines it as *worker-asks-primary* (a genuine reverse
|
||||
channel, matching the "inject the primary's pane" language). The alternative — a synonym for
|
||||
a blocking primary→worker send — is weaker and produces different plumbing. **Recommend
|
||||
worker-asks-primary.**
|
||||
2. **Detached delivery shape.** A `block:false` param on `fleet_send` (keeps the catalog
|
||||
small) vs. a separate `fleet_dispatch` tool. **Recommend the param.**
|
||||
3. **Auto-spawn on send.** `fleet_send` provisions a worker per profile when none exists
|
||||
(simplest primary UX) vs. requiring an explicit `fleet_spawn` first. **Recommend
|
||||
auto-spawn, defaulting on.**
|
||||
4. **Transport & SDK.** Streamable-HTTP/SSE co-located with the REST bind (recommended) vs.
|
||||
stdio. Requires choosing a Java MCP server SDK and adding it to the pom.
|
||||
|
||||
---
|
||||
|
||||
## 11. Implementation staging
|
||||
|
||||
- **CB-104** — blocking `fleet_send` + rendezvous registry + caller-identity resolver
|
||||
(the producer that finally drives the inert `StatusPoller`).
|
||||
- **CB-1xx** — `fleet_reply` / `fleet_ask` reverse rendezvous + detached pane injection.
|
||||
- **CB-1xx** — lifecycle + observability adapters (`fleet_spawn/list/stop/status/read`).
|
||||
- **CB-1xx** — transport wiring + `claude mcp add` docs; parity tests vs. REST.
|
||||
- **Later** — `fleet_cancel`; swap `StatusPoller` for herdr `events.subscribe`.
|
||||
`UNKNOWN` is deliberately neither injectable nor a pickup. A pane whose status cannot be read is
|
||||
not a pane that is safe to write to — see fleetd #176 for what happens when a gate treats an
|
||||
unreadable pane as a ready one.
|
||||
|
||||
+79
-29
@@ -110,6 +110,14 @@ bind:
|
||||
# notifications:
|
||||
# mode: disabled
|
||||
|
||||
# Idle-sleep guard: while at least one member is live, hold an OS-level assertion against idle
|
||||
# sleep (macOS only — a `caffeinate -i` child; a no-op elsewhere or if caffeinate is missing), so
|
||||
# an unattended host does not idle-sleep out from under a member's long turn. Unlike health/
|
||||
# configReload above, this is ON BY DEFAULT — omitting the block entirely leaves it enabled, the
|
||||
# same as `enabled: true`. Uncomment only to turn it off:
|
||||
# idleSleepGuard:
|
||||
# enabled: false
|
||||
|
||||
# herdr Unix socket. Omit to use the client default
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
@@ -307,16 +315,22 @@ profiles:
|
||||
# and no refusal on cost; the cap on live members is the single thing standing between a
|
||||
# fan-out and your monthly limit. Set it deliberately and keep it small.
|
||||
#
|
||||
# GOTCHA 3 (fleetd #176) — `maxLoad` counts members, never the lead itself. The lead is a live
|
||||
# `claude` session on this SAME account (a lead is never moved off-subscription, whatever its
|
||||
# own profile says), so it already holds one seat before any member spawns. If a lead's
|
||||
# `fleet.leaders.<name>.profile` names THIS profile — or ANY OTHER `subscription: true`
|
||||
# profile that shares this one's account (see THE SENTINEL, just below, next to
|
||||
# `credentialId:`) — `fleet_list`'s `free` for this profile subtracts that lead's live
|
||||
# seat(s) automatically; see `profile:` under THE FLEET below. If no lead entry names a
|
||||
# profile sharing this account, fleetd has no way to know a lead holds a seat here, and `free`
|
||||
# will overstate what a fresh `fleet_spawn` actually gets by exactly the seats the lead is
|
||||
# quietly holding.
|
||||
# GOTCHA 3 (fleetd #176, corrected by fleetd #257) — `maxLoad` counts members, never the lead
|
||||
# itself. The lead is a live `claude` session on this SAME account (a lead is never moved
|
||||
# off-subscription, whatever its own profile says), so it already holds one seat before any
|
||||
# member spawns. If a lead's `fleet.leaders.<name>.profile` names THIS profile — or ANY OTHER
|
||||
# `subscription: true` profile that shares this one's account (see THE SENTINEL, just below,
|
||||
# next to `credentialId:`) — `fleet_list` reports that seat count under `leadSeats`; see
|
||||
# `profile:` under THE FLEET below. `free` itself is NEVER reduced by `leadSeats`: `free` means
|
||||
# "what the real placement gate (`CompositePeerLauncher#enforceMaxLoad`) will actually grant a
|
||||
# fresh `fleet_spawn` right now", and that gate only ever compares live members against
|
||||
# `maxLoad` — it has no notion of the lead's own seat. An earlier cut of this feature
|
||||
# subtracted `leadSeats` from `free` on the theory it made `free` describe the true ceiling on
|
||||
# the account, but no backend seat ceiling shared with the lead has ever actually been
|
||||
# measured, and the subtraction just made `free` disagree with the one thing it is supposed to
|
||||
# describe — the fleetd #257 fix. `maxLoad: 3` means 3 member slots, full stop; a lead sharing
|
||||
# the account is a fact you can see in `leadSeats`, not a reason `free` undercounts spawns that
|
||||
# will, in practice, succeed.
|
||||
#
|
||||
# THE SENTINEL (fleetd #176 stage 2, correcting an inert stage 1 fix): every `subscription:
|
||||
# true` profile that leaves `credentialId` unset shares ONE implicit account-wide credential
|
||||
@@ -536,17 +550,18 @@ fleet:
|
||||
# auto-launched: it is also how fleetd learns which account this lead's own session shares. A
|
||||
# `subscription: true` profile bills the operator's Claude account, and the lead itself is always
|
||||
# a live `claude` session on that same account — `maxLoad` never counted that seat. If a lead
|
||||
# entry here names a profile that shares a worker profile's account, `fleet_list`'s `free` for
|
||||
# that worker profile subtracts the lead's live seat(s) automatically. "Shares the account" is
|
||||
# decided by matching `effectiveCredentialId()`, which (fleetd #176 stage 2 — see THE SENTINEL,
|
||||
# next to `credentialId:`, in THE WORKERS above) means: an explicit, matching `credentialId:` on
|
||||
# both, OR — the common case, needing NO extra config — both being `subscription: true` with
|
||||
# `credentialId` left unset, since those all share one implicit account-wide id. A lead on `opus`
|
||||
# and workers on `sonnet` link automatically this way; they do NOT need the same profile name.
|
||||
# Setting `profile:` on an already-running, recognise-only lead is safe — the daemon only launches
|
||||
# the SHORTFALL below `instances`, so naming a profile here does not, by itself, start anything.
|
||||
# Omit it and fleetd has no way to derive the sharing — there is no other reliable signal on the
|
||||
# daemon's side — so that lead's seat goes uncounted, exactly as before this ticket.
|
||||
# entry here names a profile that shares a worker profile's account, `fleet_list` reports the
|
||||
# lead's live seat(s) on that worker profile under `leadSeats` — informational only, as of fleetd
|
||||
# #257 it is NEVER subtracted from `free` (see GOTCHA 3, next to `maxLoad:`, in THE WORKERS above,
|
||||
# for why). "Shares the account" is decided by matching `effectiveCredentialId()`, which (fleetd
|
||||
# #176 stage 2 — see THE SENTINEL, next to `credentialId:`, in THE WORKERS above) means: an
|
||||
# explicit, matching `credentialId:` on both, OR — the common case, needing NO extra config — both
|
||||
# being `subscription: true` with `credentialId` left unset, since those all share one implicit
|
||||
# account-wide id. A lead on `opus` and workers on `sonnet` link automatically this way; they do
|
||||
# NOT need the same profile name. Setting `profile:` on an already-running, recognise-only lead is
|
||||
# safe — the daemon only launches the SHORTFALL below `instances`, so naming a profile here does
|
||||
# not, by itself, start anything. Omit it and fleetd has no way to derive the sharing — there is
|
||||
# no other reliable signal on the daemon's side — so that lead's seat never appears in `leadSeats`.
|
||||
#
|
||||
# `tab:` (CB-579) is REQUIRED and is the only field identity depends on — the exact label of the
|
||||
# tab hosting the lead, matched case-insensitively. Label the tab yourself and put that same
|
||||
@@ -669,20 +684,35 @@ guard:
|
||||
# every name here NOT also in `allow` is overlaid with a non-secret sentinel value before
|
||||
# the pane's login shell runs — real protection only for names that shell does not itself
|
||||
# re-export (see the ROUND-2 CORRECTION note above). Under allow-list: reporting only.
|
||||
# sshAuthSock → whether SSH_AUTH_SOCK may pass through under allow-list ("allow") or must be
|
||||
# blanked like any other non-derived name ("block", the default). This is a decision you
|
||||
# have to make explicitly: SSH_AUTH_SOCK is a handle to YOUR ssh-agent, and a member
|
||||
# holding it can sign with your keys — it sits in no secret file and looks like no
|
||||
# credential, which is why it slipped past three earlier tickets (gitea #110). Blocking
|
||||
# it breaks git over SSH inside members (push/fetch authenticate as you); use HTTPS
|
||||
# remotes or scoped deploy keys instead of allowing it lightly.
|
||||
# sshAgentEnv → whether SSH_AUTH_SOCK may pass through under allow-list ("inherit") or is omitted
|
||||
# from the member environment ("omit", the default). Omitting it only omits the
|
||||
# inherited ssh-agent path. It discourages automatic use of the operator's agent.
|
||||
# It does not deny same-user access to that socket. It also does not block SSH keys that
|
||||
# are readable on disk. Git over SSH may still work from inside a member. Keep the block:
|
||||
# it is correct and costs nothing, but it is not a control. A member runs as the same OS
|
||||
# user as the lead. Inside one uid, ordinary Unix permissions provide no meaningful
|
||||
# confidentiality boundary. A real boundary needs a different OS user or OS-level
|
||||
# confinement, such as a container or VM. That is the open question in fleetd #184.
|
||||
#
|
||||
# Still do not set this to "inherit" casually. SSH_AUTH_SOCK is a live handle to YOUR
|
||||
# ssh-agent, so a member holding it can sign with EVERY key the agent holds. It sits in
|
||||
# no secret file and looks like no credential, which is why it slipped past three
|
||||
# earlier tickets (gitea #110). Blocking it does not contain a member, but allowing it
|
||||
# hands one a signing capability for no gain — the block costs nothing, so keep it.
|
||||
#
|
||||
# Both halves of this are measured, not argued. 2026-08-28: a member with
|
||||
# SSH_AUTH_SOCK blanked pushed to the forge over SSH successfully, because `ssh -G`
|
||||
# resolves an IdentityFile outside ~/.ssh that is readable and has no passphrase. An
|
||||
# earlier version of this comment claimed blocking the socket BREAKS git over SSH. It
|
||||
# does not. That claim came from looking only in ~/.ssh, which holds nothing but four
|
||||
# `Include` lines — looking in one place and concluding about the whole host.
|
||||
#
|
||||
# HOT-RELOADABLE the same way `fleet:` is (CB-559): read fresh on every spawn, so editing this list
|
||||
# and reloading config (or restarting) changes what the NEXT spawn inherits; already-running members
|
||||
# are unaffected either way.
|
||||
# memberCredentials:
|
||||
# policy: deny-by-default # or "deny-list", or "allow-list" (CB-633) — see above
|
||||
# sshAuthSock: block # allow-list only; see the sshAuthSock note above
|
||||
# sshAgentEnv: omit # allow-list only; see the sshAgentEnv note above
|
||||
# allow:
|
||||
# - AI_GATEWAY_TOKEN # named in a profile's tokenEnv (local/gx) — a member reaching the
|
||||
# # gateway is by design, not a leak
|
||||
@@ -752,6 +782,19 @@ guard:
|
||||
# so fleetd falls back to the weaker CB-596 sentinel overlay instead (a WARN names the gap).
|
||||
# worktreeGroup: fleet-workers
|
||||
|
||||
# fleetd #362: a directory of skill folders (each a subdirectory holding a SKILL.md, the same
|
||||
# shape as this repo's own .claude/skills/) copied into every PROVISIONED worktree's
|
||||
# .claude/skills/, so a member spawned against ANY repo — not only one that already ships its own
|
||||
# copy — can load a bridge skill (e.g. implementer). Unset (the default): no worktree is touched
|
||||
# beyond today's behaviour. A skill folder the target repo already carries under
|
||||
# .claude/skills/<name> is never overwritten — the repo's own copy always wins. Claude Code
|
||||
# members only; an opencode member reads a different path (.opencode/agent) this key does not
|
||||
# touch. Best-effort like worktreeGroup above: a missing/unreadable directory here is logged and
|
||||
# skipped, never a failed spawn. Every non-hidden subdirectory of this directory is copied
|
||||
# wholesale, with no per-file allowlist — don't park scratch files or drafts alongside the real
|
||||
# skill folders, they will be copied into every provisioned worktree too.
|
||||
# memberSkills: /path/to/fleetd/checkout/.claude/skills
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
# the feature disabled. By default the daemon never reaps, caps, or drains sessions.
|
||||
# idleTtlSeconds → reap READY/DONE sessions idle longer than this (never BUSY/SPAWNING)
|
||||
@@ -799,10 +842,17 @@ guard:
|
||||
# across every daemon sharing this vhost.
|
||||
# prefetch → consumer basicQos, capping how many unacked messages the mailbox holds in-heap.
|
||||
# Default 32 when omitted.
|
||||
# peers → fleetd #361: the coord-ids of the OTHER daemons on this vhost, declared by the
|
||||
# operator (the daemon never guesses). fleet_list reports each one's live reachability
|
||||
# (a passive queue check, never a presence protocol) alongside this daemon's own
|
||||
# mailbox state. Omit, or leave empty, for a daemon with no known peers yet — an
|
||||
# undeclared peer can still reach you and be reached by fleet_send, it just will not
|
||||
# show up as a row in fleet_list.
|
||||
# coordinator:
|
||||
# uriEnv: LEAD_COORD_URI
|
||||
# selfId: mac-opus
|
||||
# prefetch: 32
|
||||
# peers: [fleet01-lead]
|
||||
|
||||
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open fleet_send,
|
||||
# the ReplyPushLoop injects a *drain nudge* (never the payload) into the primary's own herdr
|
||||
|
||||
@@ -29,6 +29,7 @@
|
||||
<commons-compress.version>1.27.1</commons-compress.version>
|
||||
<commons-lang3.version>3.18.0</commons-lang3.version>
|
||||
<sqlite-jdbc.version>3.53.4.0</sqlite-jdbc.version>
|
||||
<archunit.version>1.5.0</archunit.version>
|
||||
</properties>
|
||||
|
||||
<!--
|
||||
@@ -176,6 +177,14 @@
|
||||
<version>${testcontainers.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
|
||||
<!-- fleetd #131: package-boundary and cycle enforcement (PackageCyclesTest). -->
|
||||
<dependency>
|
||||
<groupId>com.tngtech.archunit</groupId>
|
||||
<artifactId>archunit-junit5</artifactId>
|
||||
<version>${archunit.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
|
||||
<build>
|
||||
|
||||
@@ -51,9 +51,12 @@ import dev.ltms.fleet.session.SessionReaper;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.member.HerdrPeerLauncher;
|
||||
import dev.ltms.fleet.member.MemberCredentialPolicyView;
|
||||
import dev.ltms.fleet.member.OpenCodeLauncher;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.power.CaffeinateSleepAssertionMechanism;
|
||||
import dev.ltms.fleet.power.IdleSleepGuard;
|
||||
import io.javalin.Javalin;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
@@ -120,6 +123,12 @@ public final class Fleetd {
|
||||
// else can fail on a silently-empty one. A daemon started without a login shell (launchd)
|
||||
// boots fine either way — this is the only thing that says so out loud.
|
||||
reportRequiredSecrets(cfg);
|
||||
// fleetd #377: the git host value a member receives as GITEA_HOST is often a full URL
|
||||
// (scheme and trailing slash), not a host name. A member that assumes a bare host then
|
||||
// builds https://https://... and the request never leaves the machine. Report the shape
|
||||
// next to the secret report — shape only, never the value.
|
||||
reportGitHostShape(cfg);
|
||||
reportMemberTrustModel(cfg);
|
||||
// CB-596: an absent (or empty) memberCredentials: block blocks NOTHING — no credential
|
||||
// name is hardcoded any more to fall back on. Say so loudly, the same way a missing
|
||||
// secret is reported above, so upgrading past this commit never silently drops CB-592's
|
||||
@@ -246,11 +255,30 @@ public final class Fleetd {
|
||||
contextCap = cfg.lifecycle().contextCap();
|
||||
}
|
||||
boolean clearAfterTurn = cfg.lifecycle() != null && cfg.lifecycle().clearAfterTurn();
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(cfg.worktreeRoot(), cfg.worktreeGroup()),
|
||||
SessionManager sessions = new SessionManager(workers,
|
||||
new GitWorktrees(cfg.worktreeRoot(), cfg.worktreeGroup(), cfg.memberSkills()),
|
||||
System::nanoTime, contextCap, clearAfterTurn);
|
||||
liveCountRef.set(profileName -> (int) sessions.roster().stream()
|
||||
.filter(s -> profileName.equals(s.profile()))
|
||||
.count());
|
||||
liveCountRef.set(profileName -> liveSessionCount(sessions.roster(), profileName));
|
||||
|
||||
// Idle-sleep guard: hold an OS-level assertion against idle sleep while at least one
|
||||
// member is live, so an unattended host does not idle-sleep out from under a member's
|
||||
// long turn (see FleetConfig.IdleSleepGuard / dev.ltms.fleet.power.IdleSleepGuard for the
|
||||
// measurement that motivated this). Opt-out via idleSleepGuard.enabled: false; on by
|
||||
// default. Hangs off SessionManager's own onAcquire/onRelease hooks (CB-520/CB-516,
|
||||
// previously wired only to the reply inbox) and SessionManager#size() — the exact registry
|
||||
// fleet_list's live/capacity numbers are themselves computed from — rather than tracking
|
||||
// members a second way. No-op (never constructed) off macOS or when idleSleepGuard.enabled
|
||||
// is explicitly false; the mechanism itself is additionally a no-op if 'caffeinate' cannot
|
||||
// be started, so this can never fail a spawn, a release, or startup.
|
||||
boolean idleSleepGuardEnabled = cfg.idleSleepGuard() == null || cfg.idleSleepGuard().isEnabled();
|
||||
final IdleSleepGuard idleSleepGuard;
|
||||
if (idleSleepGuardEnabled) {
|
||||
idleSleepGuard = new IdleSleepGuard(new CaffeinateSleepAssertionMechanism(), sessions::size);
|
||||
sessions.onAcquire(_ -> idleSleepGuard.recheck());
|
||||
sessions.onRelease(_ -> idleSleepGuard.recheck());
|
||||
} else {
|
||||
idleSleepGuard = null;
|
||||
}
|
||||
|
||||
// CB-303 part 1: idle-ttl reaper — only when configured, defaults to disabled.
|
||||
final SessionReaper reaper;
|
||||
@@ -556,8 +584,12 @@ public final class Fleetd {
|
||||
if (cfg.health() != null && cfg.health().isEnabled()) {
|
||||
// CB-580: a member found GONE/NEVER_READY must fail whatever ticket is waiting on it,
|
||||
// through the same idempotent target-wide operation CB-516 already uses on release.
|
||||
// fleetd #386: System::nanoTime freezes across a macOS sleep, so the stall check also
|
||||
// gets a wall-clock source to detect and correct for that freeze. Every other decision
|
||||
// in FleetHealthMonitor stays on the monotonic clock, unchanged.
|
||||
healthMonitor = new FleetHealthMonitor(agents, sessions::roster, messages, healthScheduler,
|
||||
System::nanoTime, cfg.health().intervalOrDefault(),
|
||||
System::nanoTime, () -> TimeUnit.MILLISECONDS.toNanos(System.currentTimeMillis()),
|
||||
cfg.health().intervalOrDefault(),
|
||||
cfg.health().workingSuspectAfterOrDefault(), messages::abandon);
|
||||
String coverage = FleetHealthMonitor.coverage(true,
|
||||
cfg.health().notifications() != null && cfg.health().notifications().configured());
|
||||
@@ -591,7 +623,11 @@ public final class Fleetd {
|
||||
if (detail.agentSessionId() != null) {
|
||||
reason += " agentSessionId=" + detail.agentSessionId();
|
||||
}
|
||||
messages.abandon(detail.terminalId(), reason);
|
||||
// fleetd #275: this is an explicit teardown (fleet_stop, or the idle reaper) — the
|
||||
// worker's pane is being stopped right now, so an open fleet_ask has no turn left to
|
||||
// resume into. Sweep it too, unlike FleetHealthMonitor's health-classification call
|
||||
// (see MessageService.abandon's javadoc for why those two must differ).
|
||||
messages.abandon(detail.terminalId(), reason, true);
|
||||
replyInbox.release(detail.terminalId());
|
||||
primaryRegistry.forgetDelegation(detail.terminalId()); // CB-532: don't leak the lead binding
|
||||
});
|
||||
@@ -621,6 +657,19 @@ public final class Fleetd {
|
||||
log.info("auth: loopback-trust (any loopback non-worker caller is the primary)");
|
||||
}
|
||||
|
||||
// fleetd #297: named once and reused verbatim below for FleetApp's GET /profiles, rather than
|
||||
// built a second time — two independently-constructed sources reading the SAME BackendQuarantine
|
||||
// / BackendOutagePolicy would still be able to drift (e.g. a future edit to the credentialIdFor
|
||||
// closure in only one of the two places), exactly the shape #284 was.
|
||||
FleetMcp.QuarantineSource quarantineSource = new FleetMcp.QuarantineSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, quarantine);
|
||||
FleetMcp.OutageSource outageSource = new FleetMcp.OutageSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, outagePolicy);
|
||||
|
||||
FleetMcp mcp = new FleetMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, metrics, new FleetMcp.CapacitySource(profile -> liveCountRef.get().apply(profile),
|
||||
profile -> {
|
||||
@@ -632,16 +681,15 @@ public final class Fleetd {
|
||||
return FleetHealthMonitor.coverage(health != null && health.isEnabled(),
|
||||
health != null && health.notifications() != null && health.notifications().configured());
|
||||
}),
|
||||
new FleetMcp.QuarantineSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, quarantine),
|
||||
quarantineSource,
|
||||
leadMailbox,
|
||||
new FleetMcp.OutageSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, outagePolicy),
|
||||
new FleetMcp.LeadSeatSource(leadSeatLookup(() -> config.get().profiles(), leaders, leads)));
|
||||
outageSource,
|
||||
new FleetMcp.LeadSeatSource(leadSeatLookup(() -> config.get().profiles(), leaders, leads)),
|
||||
// fleetd #361: the operator-declared peers this daemon's fleet_list should try to
|
||||
// reach. Read from the SAME snapshot leadMailbox itself opened from (cfg.coordinator()),
|
||||
// not the live config.get() — coordinator wiring is already boot-time-fixed (see
|
||||
// leadMailbox above), so peers follows the same rule rather than half hot-reloading.
|
||||
cfg.coordinator() == null ? List.of() : cfg.coordinator().peers());
|
||||
|
||||
// CB-637: the receive half. Only constructed when a lead mailbox actually opened — with no
|
||||
// coordinator (or an unreachable one) there is nothing to deliver, so no scheduler is
|
||||
@@ -687,6 +735,11 @@ public final class Fleetd {
|
||||
if (configWatcher != null) configWatcher.stop(); // CB-559: stop polling the config file
|
||||
mcp.close();
|
||||
if (reaper != null) reaper.stop();
|
||||
// Idle-sleep guard: release unconditionally, even though sessions.close() above already
|
||||
// drained every session (and each release already drove the live count to 0, which
|
||||
// releases the guard's assertion on its own) — this is the backstop for a drain that was
|
||||
// itself interrupted or threw, so no caffeinate child ever outlives the daemon.
|
||||
if (idleSleepGuard != null) idleSleepGuard.close();
|
||||
// Release the broker connection last among message resources (no-op for the in-memory inbox).
|
||||
if (replyInbox instanceof AutoCloseable closeable) {
|
||||
try {
|
||||
@@ -709,8 +762,14 @@ public final class Fleetd {
|
||||
|
||||
// CB-185: give FleetApp both daemons — /healthz must require both to answer and
|
||||
// GET /sessions must merge across both, or a down/unpolled member daemon is invisible.
|
||||
// fleetd #111: live (re-read-per-request) memberCredentials view for GET /member-credentials —
|
||||
// same hot-reload shape as the memberCredentials supplier passed to ClaudeCodeLauncher above.
|
||||
// fleetd #297: quarantineSource/outageSource are the SAME instances passed to FleetMcp above —
|
||||
// GET /profiles must report the identical quarantine/cool-off facts as fleet_profiles.
|
||||
Javalin app = new FleetApp(herdr, memberHerdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics, deliverable).build();
|
||||
callers, metrics, deliverable,
|
||||
() -> MemberCredentialPolicyView.of(config.get().memberCredentials()),
|
||||
quarantineSource, outageSource).build();
|
||||
app.start(cfg.bind().host(), cfg.bind().port());
|
||||
log.info("fleetd listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
@@ -850,6 +909,19 @@ public final class Fleetd {
|
||||
.orElse(null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Count sessions that occupy a profile's spawn capacity. A {@code BACKEND_ERROR} or
|
||||
* {@code FAILED} session stays in the roster so {@code fleet_list} can show its failure, but a
|
||||
* member that cannot accept another delivery does not use a seat.
|
||||
*/
|
||||
static int liveSessionCount(List<MemberSession> roster, String profileName) {
|
||||
return (int) roster.stream()
|
||||
.filter(session -> profileName.equals(session.profile()))
|
||||
.filter(session -> session.state() != MemberSession.State.BACKEND_ERROR)
|
||||
.filter(session -> session.state() != MemberSession.State.FAILED)
|
||||
.count();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #248 / fleetd#201 Unit 5: factory for the production {@link BackendErrorSink} — the
|
||||
* collaborator {@link CompletionResolver} notifies when a pane-scrape classification actually
|
||||
@@ -1132,6 +1204,94 @@ public final class Fleetd {
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #377: the env var names holding the git host value that members receive as
|
||||
* {@code GITEA_HOST}. {@code HerdrPeerLauncher.applyGitToken} injects {@code GITEA_HOST}
|
||||
* only for profiles that opted in via {@code gitTokenEnv} (CB-302), reading the value from
|
||||
* that profile's {@code gitHostEnv} (default {@code GITEA_HOST}), so the shape only matters
|
||||
* where a git token is opted in. A var used by more than one profile is one entry naming
|
||||
* every profile that reads it, the same shape as {@link #requiredSecretEnvVars}.
|
||||
*
|
||||
* <p>Package-private and pure (no I/O, no logging) so the derivation is unit-testable
|
||||
* without capturing log output; {@link #reportGitHostShape(FleetConfig)} is the logging caller.
|
||||
*/
|
||||
static Map<String, List<String>> gitHostEnvVars(FleetConfig cfg) {
|
||||
Map<String, List<String>> hostsBy = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (profile.hasGitToken()) {
|
||||
hostsBy.computeIfAbsent(profile.gitHostEnv(), _ -> new ArrayList<>())
|
||||
.add("profile '" + name + "' gitHostEnv");
|
||||
}
|
||||
});
|
||||
return hostsBy;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the value already starts with a URI scheme ({@code https://...}, {@code
|
||||
* http://...}). A bare host name and a host:port must both report {@code false} — the shape
|
||||
* this ticket exists for is a value that <em>looks like</em> a host but is a full URL, and
|
||||
* confusing those in the report would move the failure to the log instead of the network.
|
||||
*/
|
||||
static boolean startsWithScheme(String value) {
|
||||
return value.matches("[A-Za-z][A-Za-z0-9+.-]*://.*");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #377: log, on the same startup path as {@link #reportRequiredSecrets}, the SHAPE of
|
||||
* each git host value that members receive as {@code GITEA_HOST}: set or unset, its length,
|
||||
* whether it starts with a scheme, whether it ends with a slash. Never the value itself — the
|
||||
* same discipline as {@link #reportRequiredSecrets}, which logs by name only. Either shape is
|
||||
* legitimate: the value is passed to members unchanged, and a line that quietly rewrites it
|
||||
* would change what works on one host and breaks on another. The shape line only tells the
|
||||
* operator which URL form to expect from a member that builds on {@code GITEA_HOST}. An
|
||||
* unset variable is logged at INFO — useful information, not an error — and the daemon
|
||||
* starts on either way.
|
||||
*
|
||||
* <p>The env read lives in the overload below so a test can drive the line with a known
|
||||
* value and prove that value never reaches the log.
|
||||
*/
|
||||
static void reportGitHostShape(FleetConfig cfg) {
|
||||
reportGitHostShape(cfg, System.getenv());
|
||||
}
|
||||
|
||||
static void reportGitHostShape(FleetConfig cfg, Map<String, String> env) {
|
||||
Map<String, List<String>> hostsBy = gitHostEnvVars(cfg);
|
||||
if (hostsBy.isEmpty()) {
|
||||
log.info("startup git host: no profile sets a gitTokenEnv — nothing to check");
|
||||
return;
|
||||
}
|
||||
hostsBy.forEach((varName, sources) -> {
|
||||
String value = env.get(varName);
|
||||
if (value == null || value.isBlank()) {
|
||||
log.info("startup git host {}: unset ({}) — a member gets GITEA_TOKEN but no "
|
||||
+ "GITEA_HOST value", varName, String.join(", ", sources));
|
||||
} else {
|
||||
log.info("startup git host {}: set ({}) — length={}, startsWithScheme={}, "
|
||||
+ "trailingSlash={}",
|
||||
varName, String.join(", ", sources),
|
||||
value.length(), startsWithScheme(value), value.endsWith("/"));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #184: state the member trust model at startup. Environment controls and worktrees do
|
||||
* not make a sandbox when fleetd and its members use the same OS user. A separate herdr may
|
||||
* provide that boundary, but fleetd cannot inspect the uid at the other end of its socket.
|
||||
*/
|
||||
static void reportMemberTrustModel(FleetConfig cfg) {
|
||||
if (cfg.memberHerdrSocket() != null && !cfg.memberHerdrSocket().isBlank()) {
|
||||
log.info("member trust model: members are routed to a separate herdr through "
|
||||
+ "memberHerdrSocket. fleetd cannot see that herdr's uid, so confirm it runs "
|
||||
+ "as a different OS user before treating it as a boundary.");
|
||||
return;
|
||||
}
|
||||
log.info("member trust model: members run as the same OS user as fleetd, not in a sandbox. "
|
||||
+ "A member can read any file this user can read, including SSH keys and credential "
|
||||
+ "stores, whatever memberCredentials says. To add a real boundary, route members to "
|
||||
+ "a second herdr under a different OS user with memberHerdrSocket.");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-596: {@code known:} empty (block absent entirely, or present but empty) means {@link
|
||||
* FleetConfig.MemberCredentials#blockedSet()} is empty too — every member pane inherits the
|
||||
@@ -1145,12 +1305,14 @@ public final class Fleetd {
|
||||
* #requiredSecretEnvVars} is exposed for {@link #reportRequiredSecrets}'s own test.
|
||||
*/
|
||||
static void reportMemberCredentialsGap(FleetConfig cfg) {
|
||||
FleetConfig.MemberCredentials creds = cfg.memberCredentials();
|
||||
if (creds != null && !creds.known().isEmpty()) {
|
||||
// fleetd #111: the counts below come from MemberCredentialPolicyView, the same class the
|
||||
// live GET /member-credentials endpoint reads — one place computes them, not two.
|
||||
MemberCredentialPolicyView view = MemberCredentialPolicyView.of(cfg.memberCredentials());
|
||||
if (view.present()) {
|
||||
log.info("memberCredentials: policy={}, {} known name(s), {} allowed — blocking {} on "
|
||||
+ "every spawn{}",
|
||||
creds.policy(), creds.known().size(), creds.allow().size(), creds.blockedSet().size(),
|
||||
creds.isAllowList()
|
||||
view.policy(), view.knownCount(), view.allowedCount(), view.blockedCount(),
|
||||
cfg.memberCredentials().isAllowList()
|
||||
? " (allow-list: known/allow are reporting only — the control is the derived ZDOTDIR scrub)"
|
||||
: "");
|
||||
return;
|
||||
|
||||
@@ -236,7 +236,15 @@ public final class CallerResolver {
|
||||
// loopback-trust: same-host callers that are not workers are the primary. A non-loopback
|
||||
// caller is anonymous even here — and startup refuses that combination anyway
|
||||
// (FleetConfig.validateAuthExposure), so this is defence in depth, not the control.
|
||||
return isLoopback(remoteAddr) ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
//
|
||||
// fleetd #317: "not a worker" must not be conflated with "identity unresolved". The real
|
||||
// primary is a real process — its pid resolves (c.resolved()), it just owns no herdr pane.
|
||||
// A caller whose peer-PID lookup failed (LsofPeerPidLookup's -1 sentinel — on any failure,
|
||||
// silently including "lsof found no match") has no such pid, and PaneLocator's own javadoc
|
||||
// already names what happens if that case is handed the primary role: a worker→primary
|
||||
// escalation. So an unresolved caller is refused (ANONYMOUS — the same clean, already-tested
|
||||
// "authenticated as nothing" outcome used everywhere else in this method), never promoted.
|
||||
return isLoopback(remoteAddr) && c.resolved() ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
}
|
||||
|
||||
private boolean presentedTokenMatches(String authorizationHeader) {
|
||||
@@ -262,11 +270,15 @@ public final class CallerResolver {
|
||||
return token.isEmpty() ? null : token;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #305: delegates to {@link ConnectionIdentity#isLoopback}. This used to be a second,
|
||||
* independent copy of the same rule, and the two drifted: this one accepted all of
|
||||
* {@code 127.0.0.0/8}, {@code ConnectionIdentity}'s accepted only {@code 127.0.0.1}. A caller
|
||||
* from {@code 127.0.0.2} therefore had its identity skipped (so it had no terminal) and was
|
||||
* then read as loopback here — which under loopback-trust is the primary. Sharing the inputs
|
||||
* would not have prevented that; only sharing the computation does.
|
||||
*/
|
||||
private static boolean isLoopback(String remoteAddr) {
|
||||
if (remoteAddr == null) {
|
||||
return false;
|
||||
}
|
||||
return remoteAddr.equals("127.0.0.1") || remoteAddr.equals("::1")
|
||||
|| remoteAddr.equals("0:0:0:0:0:0:0:1") || remoteAddr.startsWith("127.");
|
||||
return ConnectionIdentity.isLoopback(remoteAddr);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -22,46 +22,159 @@ import java.util.function.Supplier;
|
||||
* choice rather than an accident of where the field was initialised.
|
||||
*
|
||||
* <h2>Not every key can change under a running daemon</h2>
|
||||
* Keys fall into three classes, and the difference is about what already exists when the reload
|
||||
* Keys fall into four classes, and the difference is about what already exists when the reload
|
||||
* happens — not about how important the key is.
|
||||
*
|
||||
* <ul>
|
||||
* <li><strong>Hot</strong> — re-read per use, so a reload takes effect on the next spawn:
|
||||
* {@code fleet:} (every role pool, {@code charters}, and {@code tabLabel}),
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Those
|
||||
* three are read through a supplier on {@code CompositePeerLauncher}, which is what makes
|
||||
* them hot — not the fact that they are config. <strong>This does NOT include
|
||||
* {@code fleet.leaders}</strong>: {@code Fleetd.main} reads {@code cfg.fleet().leaders()}
|
||||
* once at startup to build the {@code LeadTabScanner} and the {@code LeadLauncher}, and
|
||||
* neither is reconstructed on reload — so a lead added, removed, or re-{@code tab}'d under
|
||||
* {@code fleet.leaders} needs a restart, the same as any deferred key below.</li>
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Both are
|
||||
* read through a supplier on {@code CompositePeerLauncher}, which is what makes them hot —
|
||||
* not the fact that they are config. Most of {@code fleet:} — every role pool
|
||||
* ({@code architects}/{@code developers}/{@code reviewers}), {@code charters}, and
|
||||
* {@code tabLabel} — is read the same live way, through the same supplier
|
||||
* ({@code () -> config.get().fleet()}). <strong>But {@code fleet:} as a whole is NOT in this
|
||||
* class</strong>: {@code fleet.leaders} inside the same key is frozen, which is exactly what
|
||||
* makes {@code fleet:} split rather than hot — see below.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code idleSleepGuard:} ({@code Fleetd.java} reads it once, at startup, to decide whether
|
||||
* to construct an {@code IdleSleepGuard} and wire {@code SessionManager}'s
|
||||
* {@code onAcquire}/{@code onRelease} hooks to it — neither is rebuilt on reload, so a
|
||||
* running daemon keeps whatever this was at startup regardless of a later edit),
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code quarantineCooldownSeconds}
|
||||
* (CB-578 stage B — baked once into the {@code BackendQuarantine} built at startup),
|
||||
* {@code guard:}, {@code worktreeRoot:}, adding or removing a profile (a new backend needs its own launcher,
|
||||
* {@code guard:}, {@code worktreeRoot:}, {@code worktreeGroup:} and {@code memberSkills:}
|
||||
* (all three of the latter baked once into the {@code GitWorktrees} built at
|
||||
* {@code Fleetd.java:251} and never rebuilt — fleetd #323 instance 2 found
|
||||
* {@code worktreeGroup} missing from this list and from {@link #changedDeferredKeys};
|
||||
* {@code memberSkills} (fleetd #362) followed the same shape), {@code primary:} (fleetd #326 — {@code Fleetd.java:506, 519,
|
||||
* 520} read {@code cfg.primary()} only off the startup snapshot to build {@code
|
||||
* PrimaryRegistry} and size {@code ReplyPushLoop}'s reminder cap/backoff, and neither is
|
||||
* rebuilt on reload. Say the consequence exactly: {@code primary.terminal} is DEPRECATED
|
||||
* (CB-532, and {@code Fleetd.java:511} warns about it at startup) — a lead's identity comes
|
||||
* from {@code leaders:}/{@code leadScan:}, so changing this pin does not demote or promote a
|
||||
* lead that uses those. What a changed pin still does not take effect on until a restart is
|
||||
* the fallback nudge destination the pin remains, the deprecated identity path for an operator
|
||||
* who still relies on it, and {@code pushReminders}/{@code pushBackoffMs}), {@code configReload:} (fleetd #326 — {@code
|
||||
* Fleetd.java:679-680} read it only at startup to decide whether to build a {@code
|
||||
* ConfigWatcher} at all and with what interval; the watcher that would apply a later change is
|
||||
* itself built once, so a running watcher keeps polling on its original enabled flag and
|
||||
* interval regardless of what a reload changes it to, the same shape as {@code lifecycle} —
|
||||
* not cold, because no already-open resource goes inconsistent with the new value, the watcher
|
||||
* (if any) simply keeps its old settings), adding or removing a profile (a new backend needs its own launcher,
|
||||
* which is constructed once), <em>and an existing profile's launch settings</em> —
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl},
|
||||
* {@code exhaustedPattern} (CB-578 stage A — compiled once into {@code Fleetd.main}'s
|
||||
* pattern map at startup), {@code errorPattern} (fleetd #201 Unit 5 — compiled once into
|
||||
* {@code Fleetd.main}'s backend-error pattern map at startup, the same way), and the rest.
|
||||
* {@code Fleetd.main}'s backend-error pattern map at startup, the same way),
|
||||
* {@code ideProjectDir} / {@code ideOpenCommand} / {@code autoCompactWindow} (fleetd #323
|
||||
* instance 1 — all three are read at spawn off the same frozen profile map and were missing
|
||||
* from {@link #sameLaunchSettings}), and the rest of {@link #sameLaunchSettings}.
|
||||
* {@code credentialId} (CB-578 stage B) is NOT on
|
||||
* this list — it is read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so it is hot instead.
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} at construction and resolves
|
||||
* each spawn out of that copy, so those never reach a launch until the daemon restarts. A
|
||||
* reload logs these rather than pretending they applied.</li>
|
||||
* <li><strong>Cold</strong> — cannot change at all under a running daemon: {@code bind:},
|
||||
* {@code herdrSocket:}, {@code broker:} and {@code auth:}. The socket is bound, the broker
|
||||
* <li><strong>Split</strong> (fleetd #330; extended to a third key by fleetd #333) — read
|
||||
* <em>both</em> ways at different sites, so the key does not fit any class above as a whole:
|
||||
* {@code health:}, {@code coordinator:} and {@code fleet:}. Each is read off the startup
|
||||
* snapshot to build a long-lived object, and read live off {@link #get()} at a different,
|
||||
* unrelated site — so half of a reload's effect already applies while the other half waits
|
||||
* for a restart, and a bare "config reloaded" would under-claim by exactly that half.
|
||||
* <ul>
|
||||
* <li>{@code health:} — the monitor itself ({@code enabled}, {@code intervalSeconds},
|
||||
* {@code workingSuspectAfterSeconds}) is built once at {@code Fleetd.java:556-563}
|
||||
* and never rebuilt, so a changed value needs a restart to actually start, stop, or
|
||||
* retime it. The coverage string {@code fleet_profiles} reports
|
||||
* ({@code Fleetd.java:648-650}) is read live off {@link #get()} on every call, so it
|
||||
* already reflects the new value.</li>
|
||||
* <li>{@code coordinator:} — the {@code LeadMailbox} connection ({@code uri},
|
||||
* {@code uriEnv}, {@code selfId}, {@code prefetch}) is opened once at
|
||||
* {@code Fleetd.java:502} and never reopened, so a changed value needs a restart —
|
||||
* {@code selfId} in particular names this daemon's own AMQP inbox queue, and a peer
|
||||
* lead that learned the old name would not discover a new one on its own. The broker
|
||||
* URI env-var <em>name</em> that {@code MemberEnvAllowList} keeps out of a member's
|
||||
* environment is read live off {@link #get()} on every spawn
|
||||
* ({@code HerdrPeerLauncher.java:1530}), so it already applies.</li>
|
||||
* <li>{@code fleet:} (fleetd #333) — {@code fleet.leaders} is the frozen half:
|
||||
* {@code Fleetd.java:281} reads {@code cfg.fleet().leaders()} off the startup snapshot
|
||||
* for two long-lived objects built right after it and never rebuilt — the
|
||||
* {@code LeadTabScanner}'s {@code tab label → lead name} map ({@code Fleetd.java:301},
|
||||
* wired into {@code CallerResolver.withLeadsAndMembers} at {@code Fleetd.java:620/624},
|
||||
* which is how a caller's pane is recognised as a lead at all) and, when herdr answered
|
||||
* at startup, {@code LeadLauncher(...).ensureLeads()} ({@code Fleetd.java:315}), which
|
||||
* auto-launches each declared lead up to its {@code instances} count. So a lead added,
|
||||
* removed, or given a new {@code tab:} label under {@code fleet.leaders} needs a
|
||||
* restart — until then it is invisible to identity resolution, and this is exactly the
|
||||
* scenario fleetd #333 named: an operator edits a lead's {@code tab:} to match a
|
||||
* renamed pane, sees "config reloaded", and the pane keeps resolving as a worker,
|
||||
* because {@code CallerResolver} is still matching against the old label. The live
|
||||
* half is the rest of {@code fleet:} — {@code architects}/{@code developers}/
|
||||
* {@code reviewers}, {@code charters}, {@code tabLabel} — read live through the same
|
||||
* {@code CompositePeerLauncher} supplier the Hot bullet above names, so a reload that
|
||||
* only touches those already applies with nothing to report. Because the hot and frozen
|
||||
* halves of {@code fleet:} are disjoint sub-fields rather than the same fields read two
|
||||
* ways (contrast {@code coordinator.uriEnv} above), {@link #changedSplitKeys} compares
|
||||
* {@code fleet.leaders} alone, not the whole {@code Fleet} record — comparing the whole
|
||||
* record would report "split" for a {@code tabLabel}-only change that is actually fully
|
||||
* hot, over-claiming in exactly the direction this class exists to avoid under-claiming
|
||||
* in.</li>
|
||||
* </ul>
|
||||
* A split change is still accepted — {@link Outcome#applied()} stays {@code true}, the same
|
||||
* as a deferred change — because the live half genuinely took effect; refusing the whole
|
||||
* reload would leave the operator worse off than today. {@link Outcome#split()} names the
|
||||
* key and says which half is which each time, rather than trying to score "how changed" a
|
||||
* mixed key is or handle "both halves changed in one reload" as a special case.</li>
|
||||
* <li><strong>Cold</strong> — cannot change at all under a running daemon. All five of
|
||||
* {@link #COLD_KEYS}: {@code bind:}, {@code herdrSocket:}, {@code memberHerdrSocket:},
|
||||
* {@code broker:} and {@code auth:}. The sockets are already connected, the broker
|
||||
* connection is open, and the auth mode decides who may reach the port that is already
|
||||
* listening.</li>
|
||||
* listening. This bullet omitted {@code memberHerdrSocket:} until fleetd #333 — say "all
|
||||
* five of COLD_KEYS" rather than re-listing them, so prose and set cannot drift again.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>The denominator, measured on 2026-09-04 (fleetd #330; recounted for fleetd #333);
|
||||
* recounted again for fleetd #362, and again after {@code idleSleepGuard:} was added.</strong>
|
||||
* {@code FleetConfig} has 24 top-level record components: 5 cold, 13 deferred, 3 split, 3
|
||||
* hot-excluded. Three of them are named nowhere in this file, and the reason is the same for all
|
||||
* three: {@code placement}, {@code memberCredentials} and {@code memberLoginShell} are
|
||||
* <strong>hot</strong> and correctly absent — all three are read live off {@code config.get()}
|
||||
* (placement through the {@code CompositePeerLauncher} supplier the Hot bullet names;
|
||||
* {@code memberCredentials}/{@code memberLoginShell} at spawn time, {@code Fleetd.java:198, 205, 729}
|
||||
* and {@code HerdrPeerLauncher#configuredMemberLoginShell}), so a reload takes effect on the next
|
||||
* spawn with no entry needed here.
|
||||
* {@code health} and {@code coordinator} used to be a third kind — <strong>undecided</strong>, not
|
||||
* hot — until fleetd #330 added the <strong>split</strong> class above and gave them a home. A
|
||||
* reload touching either used to report a bare "config reloaded", which under-claimed; now it names
|
||||
* the key and says which half is which. {@code fleet} was the same story in reverse: fleetd #330's
|
||||
* own fact-find named {@code health}/{@code coordinator} as "the complete set of split-shaped keys"
|
||||
* and filed {@code fleet.leaders}'s restart requirement as a documented caveat sitting in the
|
||||
* <strong>hot-excluded</strong> escape hatch instead — correctly documented, but in the one bucket
|
||||
* this file's own coverage test cannot check the truth of (see that test's javadoc). fleetd #333
|
||||
* moved it into <strong>split</strong>, where {@link #changedSplitKeys} actually reports it.
|
||||
* <p>The point of writing the count down: "not mentioned in this file" looks identical for a key
|
||||
* that is correctly hot and for a key nobody triaged. Three times now — {@code worktreeGroup} (#323),
|
||||
* {@code primary}/{@code configReload} (#326), and {@code fleet.leaders} sitting in the escape hatch
|
||||
* (#333) — the second kind hid among the first. A top-level coverage checker in the
|
||||
* {@link ConfigRefProfileCoverageTest} shape (one level up, over {@code FleetConfig} itself rather
|
||||
* than {@code FleetConfig.Profile}) proves this file's four classes exhaust the record's components
|
||||
* — see {@code ConfigRefTopLevelCoverageTest}. That test proves the record's <em>shape</em> is fully
|
||||
* triaged; it does NOT prove a {@code SPLIT_KEYS}/{@code COLD_KEYS}/{@code DEFERRED_KEYS} member has
|
||||
* any reporting code behind it at all — {@code ConfigRefTopLevelReportingCoverageTest} is what
|
||||
* fleetd #333 added for that, after measuring that a {@code SPLIT_KEYS} entry with its reporting
|
||||
* branch deleted passes both this file's own "kept in step" assert and
|
||||
* {@code ConfigRefTopLevelCoverageTest} unchanged. fleetd #337 extended it to {@code DEFERRED_KEYS}
|
||||
* after measuring the same one-way gap there directly: dropping {@code guard}'s branch out of
|
||||
* {@link #changedDeferredKeys} while {@code "guard"} stayed in the set left the whole suite green.
|
||||
*
|
||||
* <p><strong>A cold change refuses the whole reload.</strong> Not the hot half applied and the cold
|
||||
* half warned about: that would leave the running daemon in a state matching no file on disk, which
|
||||
* is the worst thing a reload can do to an operator debugging one. Refusing keeps the invariant that
|
||||
* the live config is always some version of the file, and the message names the keys that must
|
||||
* change through a restart.
|
||||
* change through a restart. A split change does <em>not</em> refuse, for a different reason than a
|
||||
* deferred change does not: its live half genuinely took effect, so refusing would throw that away
|
||||
* and leave the operator worse off than the partial-but-honest report {@link Outcome#split()} gives.
|
||||
*
|
||||
* <p>A reload that fails to parse or fails validation is also refused, and the previous config keeps
|
||||
* running. A config file being edited is normally read once mid-save; degrading a working daemon
|
||||
@@ -71,10 +184,43 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigRef.class);
|
||||
|
||||
/** Keys that cannot change under a running daemon — see the class doc. */
|
||||
private static final Set<String> COLD_KEYS =
|
||||
/**
|
||||
* Keys that cannot change under a running daemon — see the class doc.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelCoverageTest} can fold it
|
||||
* into the top-level triage it checks, the same way it reads {@link #SPLIT_KEYS}.
|
||||
*/
|
||||
static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "memberHerdrSocket", "broker", "auth");
|
||||
|
||||
/**
|
||||
* Keys read BOTH off the startup snapshot and live off {@link #get()} at different sites, so
|
||||
* neither the hot, deferred nor cold class fits them as a whole — see the class doc's Split
|
||||
* bullet (fleetd #330). A changed split key is accepted ({@link Outcome#applied()} stays
|
||||
* {@code true}) and reported by name, with a message naming which half is live and which needs
|
||||
* a restart.
|
||||
*/
|
||||
static final Set<String> SPLIT_KEYS = Set.of("health", "coordinator", "fleet");
|
||||
|
||||
/**
|
||||
* Top-level keys {@link #changedDeferredKeys} compares — see the class doc's Deferred bullet.
|
||||
* Promoted here from a test-side copy in {@code ConfigRefTopLevelCoverageTest} by fleetd #337,
|
||||
* the same reason {@link #COLD_KEYS} and {@link #SPLIT_KEYS} live here rather than in a test: a
|
||||
* second, hand-maintained copy of this set is exactly the kind of thing that silently drifts
|
||||
* from the method it is supposed to describe. {@code spawnReadyTimeoutMs} and
|
||||
* {@code spawnReadyPollMs} are compared together in one branch and reported under the combined
|
||||
* label {@code "spawnReady*"}; {@code profiles} is compared twice over (added/removed names,
|
||||
* then an existing profile's launch settings) — see {@link #changedDeferredKeys}.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelCoverageTest} and
|
||||
* {@code ConfigRefTopLevelReportingCoverageTest} can both read it, the same way they already
|
||||
* read {@link #COLD_KEYS} and {@link #SPLIT_KEYS}.
|
||||
*/
|
||||
static final Set<String> DEFERRED_KEYS = Set.of(
|
||||
"guard", "worktreeRoot", "worktreeGroup", "memberSkills", "primary", "configReload",
|
||||
"leadHeartbeat", "lifecycle", "spawnReadyTimeoutMs", "spawnReadyPollMs",
|
||||
"quarantineCooldownSeconds", "profiles", "idleSleepGuard");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<FleetConfig> current;
|
||||
|
||||
@@ -102,25 +248,37 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
/**
|
||||
* What a reload attempt did.
|
||||
*
|
||||
* <p>{@code split} is a separate field from {@code deferred} rather than a differently-worded
|
||||
* entry inside it, because the two carry different guarantees for any caller that branches on
|
||||
* them rather than just printing {@link #summary()}: every {@code deferred} entry means "this
|
||||
* key's whole change waits for a restart", while every {@code split} entry means "part of this
|
||||
* key's change already applied, and the message says which part" — collapsing them would force
|
||||
* a caller to re-parse the message to tell those apart. See the class doc's Split bullet
|
||||
* (fleetd #330) for why the key needs this at all.
|
||||
*
|
||||
* @param applied true when the new config is now live
|
||||
* @param coldKeys cold keys whose value changed, which is why an unapplied reload was refused
|
||||
* @param deferred keys that changed and were accepted, but whose effect waits for a restart
|
||||
* @param deferred keys that changed and were accepted, but whose effect waits entirely on a
|
||||
* restart
|
||||
* @param split split keys that changed and were accepted, each named with which half of it
|
||||
* is already live and which half waits for a restart
|
||||
* @param error the parse or validation failure that refused the reload, else {@code null}
|
||||
*/
|
||||
public record Outcome(boolean applied, List<String> coldKeys, List<String> deferred,
|
||||
String error) {
|
||||
List<String> split, String error) {
|
||||
|
||||
public Outcome {
|
||||
coldKeys = List.copyOf(coldKeys);
|
||||
deferred = List.copyOf(deferred);
|
||||
split = List.copyOf(split);
|
||||
}
|
||||
|
||||
static Outcome refusedCold(List<String> keys) {
|
||||
return new Outcome(false, keys, List.of(), null);
|
||||
return new Outcome(false, keys, List.of(), List.of(), null);
|
||||
}
|
||||
|
||||
static Outcome failed(String error) {
|
||||
return new Outcome(false, List.of(), List.of(), error);
|
||||
return new Outcome(false, List.of(), List.of(), List.of(), error);
|
||||
}
|
||||
|
||||
/** A one-line summary for the operator — the reason, not just the verdict. */
|
||||
@@ -132,11 +290,18 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
return "config reload refused — these keys cannot change under a running daemon: "
|
||||
+ String.join(", ", coldKeys) + ". Restart fleetd to apply them.";
|
||||
}
|
||||
if (!deferred.isEmpty()) {
|
||||
return "config reloaded; these changes need a restart to take effect: "
|
||||
+ String.join(", ", deferred);
|
||||
if (deferred.isEmpty() && split.isEmpty()) {
|
||||
return "config reloaded";
|
||||
}
|
||||
return "config reloaded";
|
||||
StringBuilder out = new StringBuilder("config reloaded");
|
||||
if (!deferred.isEmpty()) {
|
||||
out.append("; these changes need a restart to take effect: ")
|
||||
.append(String.join(", ", deferred));
|
||||
}
|
||||
if (!split.isEmpty()) {
|
||||
out.append("; partially live — ").append(String.join(" | ", split));
|
||||
}
|
||||
return out.toString();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -177,14 +342,21 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
}
|
||||
|
||||
List<String> deferred = changedDeferredKeys(old, fresh);
|
||||
List<String> split = changedSplitKeys(old, fresh);
|
||||
current.set(fresh);
|
||||
Outcome out = new Outcome(true, List.of(), deferred, null);
|
||||
Outcome out = new Outcome(true, List.of(), deferred, split, null);
|
||||
log.info(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Cold keys whose value differs between the running config and the candidate. */
|
||||
private static List<String> changedColdKeys(FleetConfig old, FleetConfig fresh) {
|
||||
/**
|
||||
* Cold keys whose value differs between the running config and the candidate.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #sameLaunchSettings} is package-private — see that test's class doc (fleetd #333).
|
||||
*/
|
||||
static List<String> changedColdKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.bind(), fresh.bind())) {
|
||||
changed.add("bind");
|
||||
@@ -206,8 +378,16 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
return changed;
|
||||
}
|
||||
|
||||
/** Changed keys that were accepted but whose effect waits for a restart. */
|
||||
private static List<String> changedDeferredKeys(FleetConfig old, FleetConfig fresh) {
|
||||
/**
|
||||
* Changed keys that were accepted but whose effect waits for a restart.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #changedColdKeys} and {@link #changedSplitKeys} already are (fleetd #333, extended to
|
||||
* this method by fleetd #337 — membership in {@link #DEFERRED_KEYS} proved nothing about this
|
||||
* method on its own until then; see that test's class doc).
|
||||
*/
|
||||
static List<String> changedDeferredKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.lifecycle(), fresh.lifecycle())) {
|
||||
changed.add("lifecycle");
|
||||
@@ -221,6 +401,44 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
if (!Objects.equals(old.worktreeRoot(), fresh.worktreeRoot())) {
|
||||
changed.add("worktreeRoot");
|
||||
}
|
||||
// Baked into the same GitWorktrees as worktreeRoot (Fleetd.java:251) and never rebuilt
|
||||
// either — see the class doc. Missing this check was fleetd #323 instance 2: a reload
|
||||
// that changed only worktreeGroup reported "config reloaded" with nothing deferred, and
|
||||
// newly provisioned worktrees kept the old sharing behaviour.
|
||||
if (!Objects.equals(old.worktreeGroup(), fresh.worktreeGroup())) {
|
||||
changed.add("worktreeGroup");
|
||||
}
|
||||
// fleetd #362: baked into the same GitWorktrees as worktreeRoot/worktreeGroup
|
||||
// (Fleetd.java:251) and never rebuilt either — a reload that changes only memberSkills
|
||||
// must be reported the same way, or a newly provisioned worktree keeps seeding from (or
|
||||
// skipping) the old source directory with nothing telling the operator why.
|
||||
if (!Objects.equals(old.memberSkills(), fresh.memberSkills())) {
|
||||
changed.add("memberSkills");
|
||||
}
|
||||
// fleetd #326: Fleetd.java:506, 519, 520 read cfg.primary() only off the startup snapshot
|
||||
// (PrimaryRegistry's pinned terminal, ReplyPushLoop's reminder cap and backoff) — neither is
|
||||
// rebuilt on reload, so a changed value needs a restart. Note what it does NOT mean:
|
||||
// primary.terminal is deprecated (CB-532), identity comes from leaders:/leadScan:, so a lead
|
||||
// using those is unaffected by this pin either way. See the class doc for the exact scope.
|
||||
if (!Objects.equals(old.primary(), fresh.primary())) {
|
||||
changed.add("primary");
|
||||
}
|
||||
// fleetd #326: Fleetd.java:679-680 read cfg.configReload() only at startup to decide whether
|
||||
// to build a ConfigWatcher at all and with what interval — the watcher that would apply a
|
||||
// later change is itself built once, so a running watcher keeps its original enabled flag and
|
||||
// interval regardless of what a reload changes it to. Not cold: no already-open resource goes
|
||||
// inconsistent with the new value, a watcher (if any) simply keeps polling on the old settings.
|
||||
if (!Objects.equals(old.configReload(), fresh.configReload())) {
|
||||
changed.add("configReload");
|
||||
}
|
||||
// Fleetd.java reads cfg.idleSleepGuard() once, at startup, to decide whether to construct
|
||||
// an IdleSleepGuard at all and wire SessionManager's onAcquire/onRelease hooks to it —
|
||||
// neither is rebuilt on reload, so a running daemon keeps whatever this was at startup
|
||||
// (armed or not) regardless of a later edit here. Not cold: nothing already-open goes
|
||||
// inconsistent with the new value, an armed-or-not guard just keeps its original answer.
|
||||
if (!Objects.equals(old.idleSleepGuard(), fresh.idleSleepGuard())) {
|
||||
changed.add("idleSleepGuard");
|
||||
}
|
||||
if (!Objects.equals(old.spawnReadyTimeoutMs(), fresh.spawnReadyTimeoutMs())
|
||||
|| !Objects.equals(old.spawnReadyPollMs(), fresh.spawnReadyPollMs())) {
|
||||
changed.add("spawnReady*");
|
||||
@@ -265,14 +483,101 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically. Compares every component
|
||||
* the launcher reads at spawn; {@code weight}, {@code maxLoad} and {@code credentialId} are
|
||||
* excluded because those are read live (by the placement policy and, for credentialId, by
|
||||
* {@code CompositePeerLauncher}/the CB-578 stage B exhaustion sink) and really do take effect on
|
||||
* the next spawn.
|
||||
* Split keys whose value differs between the running config and the candidate — see the class
|
||||
* doc's Split bullet (fleetd #330, extended for {@code fleet:} by fleetd #333). Unlike
|
||||
* {@link #changedDeferredKeys}, this does not try to tell which sub-field moved for {@code
|
||||
* health:} or {@code coordinator:}: any change to either gets the same fixed message, because
|
||||
* the message already names both halves every time, so there is no "which half changed"
|
||||
* question left for the caller to answer. {@code fleet:} is different on purpose — see below.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #sameLaunchSettings} is package-private — see that test's class doc (fleetd #333). That
|
||||
* test exists because membership in {@link #SPLIT_KEYS} proves nothing about this method on its
|
||||
* own: fleetd #333 measured that dropping the {@code coordinator} branch out of this method
|
||||
* while leaving {@code "coordinator"} in {@code SPLIT_KEYS} left the whole suite green except a
|
||||
* hand-written {@code ConfigRefTest} case — neither {@code ConfigRefTopLevelCoverageTest} (it
|
||||
* only reads the set) nor the "kept in step" assert below (it only checks the reported keys are
|
||||
* a SUBSET of {@code SPLIT_KEYS}, never that every {@code SPLIT_KEYS} member has a branch here)
|
||||
* would have caught it.
|
||||
*/
|
||||
private static boolean sameLaunchSettings(FleetConfig.Profile a, FleetConfig.Profile b) {
|
||||
return Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
static List<String> changedSplitKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.health(), fresh.health())) {
|
||||
changed.add("health: the monitor itself (enabled, interval, workingSuspectAfter) is "
|
||||
+ "frozen at startup and needs a restart; the coverage status fleet_profiles "
|
||||
+ "reports is read live and already applied");
|
||||
}
|
||||
if (!Objects.equals(old.coordinator(), fresh.coordinator())) {
|
||||
changed.add("coordinator: the LeadMailbox connection (uri, uriEnv, selfId, prefetch) is "
|
||||
+ "opened once and needs a restart; the broker URI env-var name kept out of a "
|
||||
+ "member's environment is read live on every spawn and already applied");
|
||||
}
|
||||
// fleetd #333: unlike health/coordinator above, most of `fleet:` (architects, developers,
|
||||
// reviewers, charters, tabLabel) is genuinely hot — ConfigRefTest.aHotChangeIsAppliedAndRead-
|
||||
// ThroughGet and aCharterChangeIsHotAndReachesTheLiveConfig prove it reaches the live config
|
||||
// with no restart note. Only fleet.leaders is frozen (Fleetd.java:281 reads
|
||||
// cfg.fleet().leaders() off the startup snapshot to build both the LeadTabScanner's
|
||||
// tab-label-to-name map, wired into CallerResolver.withLeadsAndMembers at Fleetd.java:620/624,
|
||||
// and — when herdr answered — LeadLauncher(...).ensureLeads() at Fleetd.java:315, which
|
||||
// auto-launches each lead up to its `instances` count; neither is rebuilt on reload). So this
|
||||
// compares fleet.leaders alone, not the whole Fleet record: comparing the whole record would
|
||||
// report "split" for a tabLabel-only or charters-only change that is actually fully hot,
|
||||
// which is the over-claim mirror of the under-claim bug this class exists to prevent.
|
||||
if (!Objects.equals(leadersOf(old), leadersOf(fresh))) {
|
||||
changed.add("fleet: fleet.leaders (each lead's tab, workspace, cwd, profile and "
|
||||
+ "instances count) is read once at startup to build the LeadTabScanner's "
|
||||
+ "identity map and to auto-launch leads, and neither is rebuilt on reload, so a "
|
||||
+ "lead added, removed, or given a new tab: label needs a restart — until then it "
|
||||
+ "stays unrecognised, and a caller from its new tab resolves as a worker, not a "
|
||||
+ "lead; the rest of fleet: (architects, developers, reviewers, charters, "
|
||||
+ "tabLabel) is read live through the supplier on CompositePeerLauncher and "
|
||||
+ "already applied");
|
||||
}
|
||||
// Kept in step with SPLIT_KEYS the same way changedColdKeys is kept in step with COLD_KEYS —
|
||||
// every message here must be traceable to one of the split keys the class doc documents.
|
||||
// NOTE what this does NOT prove, per the javadoc above: it does not catch a SPLIT_KEYS
|
||||
// member with no branch above at all, only a branch whose message is mis-worded relative to
|
||||
// the set. ConfigRefTopLevelReportingCoverageTest is what proves the former.
|
||||
assert changed.stream().allMatch(m -> SPLIT_KEYS.stream().anyMatch(k -> m.startsWith(k + ":")))
|
||||
: "a split entry was reported that does not start with a SPLIT_KEYS name: " + changed;
|
||||
return changed;
|
||||
}
|
||||
|
||||
/** {@code cfg.fleet().leaders()}, defensively, in case a caller hands in a non-defaulted config. */
|
||||
private static Map<String, FleetConfig.Leader> leadersOf(FleetConfig cfg) {
|
||||
return cfg.fleet() == null ? Map.of() : cfg.fleet().leaders();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@link FleetConfig.Profile} record components deliberately left out of
|
||||
* {@link #sameLaunchSettings} because they are read <em>live</em>, not baked in at spawn — see
|
||||
* the class doc's <em>Hot</em> bullet. {@code weight} and {@code maxLoad} are read live by the
|
||||
* placement policy on every spawn; {@code credentialId} is read live by
|
||||
* {@code CompositePeerLauncher} and the CB-578 stage B exhaustion sink. Nothing else is
|
||||
* excluded — see {@code sameLaunchSettingsComparesEveryProfileComponentOrExcludesIt} in
|
||||
* {@code ConfigRefProfileCoverageTest}, which enumerates every {@code Profile} record component
|
||||
* by reflection and fails the build if one is neither compared below nor named here.
|
||||
*/
|
||||
static final Set<String> LAUNCH_SETTINGS_EXCLUDED = Set.of("weight", "maxLoad", "credentialId");
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically.
|
||||
*
|
||||
* <p>This must compare every {@link FleetConfig.Profile} record component except the three in
|
||||
* {@link #LAUNCH_SETTINGS_EXCLUDED}. That is not a claim this javadoc can make good on by
|
||||
* itself — a javadoc saying "compares every component" is exactly what fleetd #323 found to be
|
||||
* false for three fields (and a sibling method's field list, for a fourth). The actual
|
||||
* guarantee comes from {@code ConfigRefProfileCoverageTest}: it enumerates every record
|
||||
* component of {@code FleetConfig.Profile} by reflection, mutates each one not in
|
||||
* {@code LAUNCH_SETTINGS_EXCLUDED} on a base profile, and asserts this method reports a
|
||||
* difference — so a new component that is neither compared here nor added to
|
||||
* {@code LAUNCH_SETTINGS_EXCLUDED} (with a reason) fails that test by name, rather than
|
||||
* silently reporting "config reloaded" for a value the daemon never picked up.
|
||||
*/
|
||||
static boolean sameLaunchSettings(FleetConfig.Profile a, FleetConfig.Profile b) {
|
||||
return Objects.equals(a.profile(), b.profile())
|
||||
&& Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
&& Objects.equals(a.model(), b.model())
|
||||
&& Objects.equals(a.configDir(), b.configDir())
|
||||
&& Objects.equals(a.tokenEnv(), b.tokenEnv())
|
||||
@@ -298,6 +603,16 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
// fleetd #201 Unit 5: errorPattern is compiled once into Fleetd.main's backend-error
|
||||
// pattern map at startup (see BackendErrorPatternLookup wiring), the same way
|
||||
// exhaustedPattern is — a reload never re-reads it either.
|
||||
&& Objects.equals(a.errorPattern(), b.errorPattern());
|
||||
&& Objects.equals(a.errorPattern(), b.errorPattern())
|
||||
// fleetd #323 instance 1: ideProjectDir and ideOpenCommand are read at spawn off the
|
||||
// same frozen profile map as ideMcpUrl above (ClaudeCodeLauncher.java:267/269,
|
||||
// OpenCodeLauncher.java:474/480/486) and were missing from this comparison.
|
||||
&& Objects.equals(a.ideProjectDir(), b.ideProjectDir())
|
||||
&& Objects.equals(a.ideOpenCommand(), b.ideOpenCommand())
|
||||
// fleetd #323 instance 1: autoCompactWindow is read at spawn the same way
|
||||
// (ClaudeCodeLauncher.java:926, OpenCodeLauncher.java:650). Comparing it here only
|
||||
// makes the reload REPORT that a restart is needed — it deliberately does not make
|
||||
// autoCompactWindow take effect live, which is a separate, larger change.
|
||||
&& Objects.equals(a.autoCompactWindow(), b.autoCompactWindow());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import com.fasterxml.jackson.annotation.JsonCreator;
|
||||
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
|
||||
import com.fasterxml.jackson.annotation.JsonProperty;
|
||||
import com.fasterxml.jackson.core.JsonParser;
|
||||
import com.fasterxml.jackson.core.JsonToken;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
@@ -92,12 +94,36 @@ import java.util.regex.PatternSyntaxException;
|
||||
* fleetd's own process, so fleetd's own {@code $SHELL} says nothing about what
|
||||
* that pane runs. There is no channel to ask herdr for another user's shell, so
|
||||
* this must be told, never guessed. {@code null}/blank (or a value not ending
|
||||
* in {@code zsh}) is treated the same as "not zsh": the {@code
|
||||
* memberCredentials.policy: allow-list} ZDOTDIR scrub is skipped in favour of
|
||||
* the CB-596 sentinel overlay — a degraded control, never a refusal to spawn.
|
||||
* in {@code zsh}) is treated the same as "not zsh". fleetd #155: what that means
|
||||
* now depends on {@code memberCredentials.policy}. Under {@code deny-by-default}
|
||||
* it stays a degraded control, never a refusal to spawn — the pane-creation
|
||||
* overlay is unaffected by shell type, so the spawn proceeds with a WARN naming
|
||||
* the shell. Under {@code allow-list} the spawn is REFUSED instead: that policy's
|
||||
* whole point is a control a sourced file cannot undo, so silently falling back
|
||||
* to the weaker overlay would be the same "control silently does nothing" defect
|
||||
* #155 exists to remove — configure this field (or move the account to zsh, or
|
||||
* switch policy back to {@code deny-by-default}) to unblock the spawn.
|
||||
* When {@code memberHerdrSocket} is NOT configured this field is never
|
||||
* consulted at all; fleetd keeps reading its own {@code $SHELL}, exactly as
|
||||
* before this field existed.
|
||||
* @param memberSkills fleetd #362: nullable directory of skill folders (each a subdirectory
|
||||
* holding a {@code SKILL.md}, the same shape as this repo's own {@code
|
||||
* .claude/skills/}) copied into every provisioned worktree's {@code
|
||||
* .claude/skills/}, so a member spawned against ANY repo — not only one that
|
||||
* already ships its own copy — can load a bridge skill such as {@code
|
||||
* implementer}. {@code null}/blank ⇒ off: no worktree is touched beyond
|
||||
* today's behaviour. A skill folder the target repo already carries is never
|
||||
* overwritten — see {@link dev.ltms.fleet.session.GitWorktrees}. Claude Code
|
||||
* members only; an opencode member's equivalent lives under a different path
|
||||
* ({@code .opencode/agent}) and is not covered by this key. Every non-hidden
|
||||
* subdirectory of this directory is copied wholesale, with no per-file
|
||||
* allowlist — do not park scratch files or drafts alongside the real skill
|
||||
* folders, they will be copied into every provisioned worktree too.
|
||||
* @param idleSleepGuard opt-in-by-default: hold an OS-level assertion against idle sleep while at
|
||||
* least one member is live, so an unattended host does not sleep out from
|
||||
* under a member's long turn. {@code null} (the block omitted) behaves the
|
||||
* same as an explicit {@code enabled: true}; set {@code enabled: false} to
|
||||
* turn it off. See {@link dev.ltms.fleet.power.IdleSleepGuard}.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record FleetConfig(
|
||||
@@ -122,7 +148,37 @@ public record FleetConfig(
|
||||
MemberCredentials memberCredentials,
|
||||
Coordinator coordinator,
|
||||
String worktreeGroup,
|
||||
String memberLoginShell) {
|
||||
String memberLoginShell,
|
||||
String memberSkills,
|
||||
IdleSleepGuard idleSleepGuard) {
|
||||
|
||||
/** Back-compat form before the {@code idleSleepGuard:} block was added. */
|
||||
public FleetConfig(Bind bind, String herdrSocket, String memberHerdrSocket, Map<String, Profile> profiles,
|
||||
Guard guard, String worktreeRoot, Lifecycle lifecycle, Integer spawnReadyTimeoutMs,
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, Health health, String placement, Auth auth,
|
||||
ConfigReload configReload, Integer quarantineCooldownSeconds,
|
||||
MemberCredentials memberCredentials, Coordinator coordinator, String worktreeGroup,
|
||||
String memberLoginShell, String memberSkills) {
|
||||
this(bind, herdrSocket, memberHerdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, coordinator, worktreeGroup,
|
||||
memberLoginShell, memberSkills, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the {@code memberSkills} key was added. */
|
||||
public FleetConfig(Bind bind, String herdrSocket, String memberHerdrSocket, Map<String, Profile> profiles,
|
||||
Guard guard, String worktreeRoot, Lifecycle lifecycle, Integer spawnReadyTimeoutMs,
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, Health health, String placement, Auth auth,
|
||||
ConfigReload configReload, Integer quarantineCooldownSeconds,
|
||||
MemberCredentials memberCredentials, Coordinator coordinator, String worktreeGroup,
|
||||
String memberLoginShell) {
|
||||
this(bind, herdrSocket, memberHerdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, coordinator, worktreeGroup,
|
||||
memberLoginShell, null, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the {@code memberLoginShell} key was added. */
|
||||
public FleetConfig(Bind bind, String herdrSocket, String memberHerdrSocket, Map<String, Profile> profiles,
|
||||
@@ -133,7 +189,7 @@ public record FleetConfig(
|
||||
MemberCredentials memberCredentials, Coordinator coordinator, String worktreeGroup) {
|
||||
this(bind, herdrSocket, memberHerdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, coordinator, worktreeGroup, null);
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, coordinator, worktreeGroup, null, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the {@code worktreeGroup} key was added. */
|
||||
@@ -314,8 +370,9 @@ public record FleetConfig(
|
||||
* statement that does not stop being true just because the profile was
|
||||
* named directly. A negative value has no sane meaning (there is no
|
||||
* "excluded" to degrade to below zero) and is refused at config load
|
||||
* instead, naming the profile and the key. Live means any session the
|
||||
* registry still owns (acquired and not yet released), in any state.
|
||||
* instead, naming the profile and the key. Live means a session that can
|
||||
* receive another delivery. The roster keeps terminal {@code BACKEND_ERROR}
|
||||
* and {@code FAILED} sessions for diagnostics, but they do not use capacity.
|
||||
* @param kind which peer launcher spawns this profile: {@code "claude-code"} (default —
|
||||
* the {@link dev.ltms.fleet.member.ClaudeCodeLauncher}) or {@code "opencode"}.
|
||||
* The {@code CompositePeerLauncher} routes {@code spawn}/reap by this value, so
|
||||
@@ -879,12 +936,18 @@ public record FleetConfig(
|
||||
* {@code null} ⇒ kept as {@code null} (no self id configured).
|
||||
* @param prefetch the consumer's {@code basicQos} prefetch count. {@code null}/non-positive ⇒
|
||||
* {@link LeadMailbox#DEFAULT_PREFETCH}.
|
||||
* @param peers fleetd #361: the coord-ids the operator declares as this daemon's peers — the
|
||||
* daemon never guesses who else exists. {@code fleet_list} reports each one's
|
||||
* live reachability. Blank entries are dropped; {@code null} ⇒ an empty list, so
|
||||
* a config written before this field existed still parses unchanged.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record Coordinator(String uri, String uriEnv, String selfId, Integer prefetch) {
|
||||
public record Coordinator(String uri, String uriEnv, String selfId, Integer prefetch, List<String> peers) {
|
||||
|
||||
public Coordinator {
|
||||
selfId = (selfId == null || selfId.isBlank()) ? null : selfId;
|
||||
peers = peers == null ? List.of()
|
||||
: peers.stream().filter(p -> p != null && !p.isBlank()).toList();
|
||||
}
|
||||
|
||||
/** True when a {@code uriEnv} is configured by name, whether or not its variable resolves. */
|
||||
@@ -1236,6 +1299,25 @@ public record FleetConfig(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Hold an OS-level assertion against idle sleep while at least one member is live (see
|
||||
* {@link dev.ltms.fleet.power.IdleSleepGuard}).
|
||||
*
|
||||
* <p>Unlike most opt-in blocks in this file, this one defaults to <em>on</em>: an unattended
|
||||
* host idle-sleeping mid-turn is a correctness problem (a dropped AMQP link, a frozen member),
|
||||
* not a convenience, so the safer default is armed. An operator who wants the previous
|
||||
* behaviour (no assertion held, ever) sets {@code enabled: false} explicitly.
|
||||
*
|
||||
* @param enabled {@code false} turns the guard off; {@code null} (the block omitted
|
||||
* entirely) or {@code true} leaves it on
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record IdleSleepGuard(Boolean enabled) {
|
||||
public boolean isEnabled() {
|
||||
return !Boolean.FALSE.equals(enabled);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The terminal → lead-name map seeded from the legacy singular {@code primary:} pin (CB-530).
|
||||
*
|
||||
@@ -1351,19 +1433,19 @@ public record FleetConfig(
|
||||
* deny-list/deny-by-default every name here that is NOT also in {@link #allow} is
|
||||
* overlaid with a non-secret sentinel value. Under allow-list this list is
|
||||
* reporting only.
|
||||
* @param sshAuthSock whether the member may inherit {@code SSH_AUTH_SOCK} under the allow-list
|
||||
* policy ({@code "allow"}) or must have it blanked ({@code "block"}, the default).
|
||||
* @param sshAgentEnv whether the member may inherit {@code SSH_AUTH_SOCK} under the allow-list
|
||||
* policy ({@code "inherit"}) or must have it omitted ({@code "omit"}, the default).
|
||||
* This is a DECISION, never a default: {@code SSH_AUTH_SOCK} is a handle to the
|
||||
* operator's ssh-agent, and a member holding it can sign with the operator's own
|
||||
* keys — but it appears in no secret file and is credential-shaped like nothing on
|
||||
* any list, which is why three earlier tickets missed it (gitea #110 / CB-607).
|
||||
* Blocking it breaks git over SSH inside the member; allow it only when members do
|
||||
* Omitting it does not prevent git over SSH inside the member; inherit it only when members do
|
||||
* not need to authenticate as the operator over SSH. Ignored under deny-list /
|
||||
* deny-by-default, which never touch the name.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record MemberCredentials(String policy, List<String> allow, List<String> known,
|
||||
String sshAuthSock) {
|
||||
String sshAgentEnv) {
|
||||
|
||||
/** Default policy: block every {@code known} name not in {@code allow}, via the env overlay. */
|
||||
public static final String POLICY_DENY_BY_DEFAULT = "deny-by-default";
|
||||
@@ -1380,11 +1462,25 @@ public record FleetConfig(
|
||||
*/
|
||||
public static final String POLICY_ALLOW_LIST = "allow-list";
|
||||
|
||||
/** The pre-CB-633 three-field form — {@code sshAuthSock} defaults to blocked. */
|
||||
/** The pre-CB-633 three-field form — {@code sshAgentEnv} defaults to omitted. */
|
||||
public MemberCredentials(String policy, List<String> allow, List<String> known) {
|
||||
this(policy, allow, known, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Reads both the current {@code sshAgentEnv} key and the compatible {@code sshAuthSock} key.
|
||||
* When both keys are present, {@code sshAgentEnv} wins, even if its value is unrecognised.
|
||||
*/
|
||||
@JsonCreator
|
||||
public static MemberCredentials fromYaml(@JsonProperty("policy") String policy,
|
||||
@JsonProperty("allow") List<String> allow,
|
||||
@JsonProperty("known") List<String> known,
|
||||
@JsonProperty("sshAgentEnv") String sshAgentEnv,
|
||||
@JsonProperty("sshAuthSock") String sshAuthSock) {
|
||||
return new MemberCredentials(policy, allow, known,
|
||||
sshAgentEnv != null ? sshAgentEnv : sshAuthSock);
|
||||
}
|
||||
|
||||
public MemberCredentials {
|
||||
String normalizedPolicy = (policy == null || policy.isBlank())
|
||||
? POLICY_DENY_BY_DEFAULT : policy.toLowerCase(java.util.Locale.ROOT);
|
||||
@@ -1393,8 +1489,9 @@ public record FleetConfig(
|
||||
policy = POLICY_DENY_LIST.equals(normalizedPolicy) ? POLICY_DENY_BY_DEFAULT : normalizedPolicy;
|
||||
allow = allow == null ? List.of() : List.copyOf(allow);
|
||||
known = known == null ? List.of() : List.copyOf(known);
|
||||
sshAuthSock = (sshAuthSock != null && "allow".equalsIgnoreCase(sshAuthSock.trim()))
|
||||
? "allow" : "block";
|
||||
sshAgentEnv = (sshAgentEnv != null
|
||||
&& ("inherit".equalsIgnoreCase(sshAgentEnv.trim())
|
||||
|| "allow".equalsIgnoreCase(sshAgentEnv.trim()))) ? "inherit" : "omit";
|
||||
}
|
||||
|
||||
/** True when this block selects the CB-633 derived-allow-list policy. */
|
||||
@@ -1403,8 +1500,8 @@ public record FleetConfig(
|
||||
}
|
||||
|
||||
/** True when {@code SSH_AUTH_SOCK} may pass through under the allow-list policy. Default: no. */
|
||||
public boolean sshAuthSockAllowed() {
|
||||
return "allow".equals(sshAuthSock);
|
||||
public boolean sshAgentEnvInherited() {
|
||||
return "inherit".equals(sshAgentEnv);
|
||||
}
|
||||
|
||||
/** {@link #allow} as a set, for membership checks. */
|
||||
@@ -1471,7 +1568,8 @@ public record FleetConfig(
|
||||
"bind", "herdrSocket", "memberHerdrSocket", "profiles", "guard", "worktreeRoot",
|
||||
"lifecycle", "spawnReadyTimeoutMs", "spawnReadyPollMs", "broker", "primary", "fleet",
|
||||
"leadHeartbeat", "health", "placement", "auth", "configReload", "quarantineCooldownSeconds",
|
||||
"memberCredentials", "coordinator", "worktreeGroup", "memberLoginShell");
|
||||
"memberCredentials", "coordinator", "worktreeGroup", "memberLoginShell", "memberSkills",
|
||||
"idleSleepGuard");
|
||||
|
||||
/** Load and validate config from {@code path}. */
|
||||
public static FleetConfig load(Path path) {
|
||||
@@ -1483,7 +1581,7 @@ public record FleetConfig(
|
||||
rejectDuplicateMemberSlots(yaml);
|
||||
rejectNegativeMaxLoad(yaml);
|
||||
rejectAutoCompactWindowOutOfRange(yaml);
|
||||
rejectMalformedErrorPattern(yaml);
|
||||
rejectMalformedProfilePatterns(yaml);
|
||||
rejectUnknownKind(yaml);
|
||||
rejectUnknownAuthMode(yaml);
|
||||
rejectUnknownPlacement(yaml);
|
||||
@@ -1833,20 +1931,26 @@ public record FleetConfig(
|
||||
}
|
||||
|
||||
/**
|
||||
* Reject a profile whose {@code errorPattern} (fleetd #201 Unit 5) is not a valid Java regex,
|
||||
* naming the profile, the key, and the parser's own message.
|
||||
* Reject a profile whose {@code errorPattern} (fleetd #201 Unit 5) or {@code exhaustedPattern}
|
||||
* (CB-578 stage A) is not a valid Java regex, naming the profile, the key, and the parser's own
|
||||
* message.
|
||||
*
|
||||
* <p>Unset/{@code null} means "use {@code CompletionResolver}'s built-in {@code (?i)\bAPI
|
||||
* Error\s*:} compatibility pattern" and passes silently. A profile that DOES set the key gets it
|
||||
* compiled once at daemon startup ({@code Fleetd.main}, mirroring {@code exhaustedPattern}) — an
|
||||
* uncaught {@link java.util.regex.PatternSyntaxException} there crashes startup without naming
|
||||
* which profile or key is at fault. Validate eagerly here instead, at config load, the same
|
||||
* "fail loud at load, not lazily later" reasoning as {@link #rejectAutoCompactWindowOutOfRange}.
|
||||
* <p>Unset/{@code null} means, for {@code errorPattern}, "use {@code CompletionResolver}'s
|
||||
* built-in {@code (?i)\bAPI Error\s*:} compatibility pattern", and for {@code exhaustedPattern},
|
||||
* "opt out of that classification" — either way it passes silently. A profile that DOES set
|
||||
* either key gets it compiled once at daemon startup ({@code Fleetd.main}) — an uncaught
|
||||
* {@link java.util.regex.PatternSyntaxException} there crashes startup without naming which
|
||||
* profile or key is at fault (fleetd #273: this happened for {@code exhaustedPattern}, which had
|
||||
* no validator here even though its sibling {@code errorPattern} did). Validate eagerly here
|
||||
* instead, at config load, the same "fail loud at load, not lazily later" reasoning as
|
||||
* {@link #rejectAutoCompactWindowOutOfRange}. Both keys are checked from a single load, and any
|
||||
* failures from either are collected together into one message.
|
||||
*
|
||||
* @param yaml the raw config text
|
||||
* @throws IllegalStateException when any profile's {@code errorPattern} fails to compile
|
||||
* @throws IllegalStateException when any profile's {@code errorPattern} or
|
||||
* {@code exhaustedPattern} fails to compile
|
||||
*/
|
||||
static void rejectMalformedErrorPattern(String yaml) {
|
||||
static void rejectMalformedProfilePatterns(String yaml) {
|
||||
Map<?, ?> raw;
|
||||
try {
|
||||
raw = YAML.readValue(yaml, Map.class);
|
||||
@@ -1861,18 +1965,20 @@ public record FleetConfig(
|
||||
if (!(e.getValue() instanceof Map<?, ?> p)) {
|
||||
continue;
|
||||
}
|
||||
if (!(p.get("errorPattern") instanceof String pattern) || pattern.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
try {
|
||||
Pattern.compile(pattern);
|
||||
} catch (PatternSyntaxException ex) {
|
||||
bad.add("profiles." + e.getKey() + ".errorPattern (\"" + pattern + "\"): " + ex.getMessage());
|
||||
for (String key : List.of("errorPattern", "exhaustedPattern")) {
|
||||
if (!(p.get(key) instanceof String pattern) || pattern.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
try {
|
||||
Pattern.compile(pattern);
|
||||
} catch (PatternSyntaxException ex) {
|
||||
bad.add("profiles." + e.getKey() + "." + key + " (\"" + pattern + "\"): " + ex.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
bad.sort(String::compareTo);
|
||||
if (!bad.isEmpty()) {
|
||||
throw new IllegalStateException("refusing to start: malformed errorPattern — "
|
||||
throw new IllegalStateException("refusing to start: malformed pattern — "
|
||||
+ String.join("; ", bad));
|
||||
}
|
||||
}
|
||||
@@ -2140,9 +2246,18 @@ public record FleetConfig(
|
||||
// memberLoginShell is left as-is (fleetd #213), like worktreeGroup: null/blank is "not
|
||||
// configured", and there is no sane non-null default — a member's login shell is
|
||||
// operator-specific and only meaningful when memberHerdrSocket is also set.
|
||||
// memberSkills is left as-is (fleetd #362), like worktreeGroup/memberLoginShell: null/blank
|
||||
// is "off", and there is no sane non-null default — the daemon may not even run from a
|
||||
// checkout that ships its own .claude/skills/.
|
||||
// idleSleepGuard is left as-is, like leadHeartbeat/configReload above, but for the opposite
|
||||
// reason: it is on by default already (its own isEnabled() treats null the same as
|
||||
// enabled: true — see its javadoc), so defaulting the block here would change nothing a
|
||||
// reader observes and would only obscure that "block omitted" and "block present and
|
||||
// enabled" are deliberately the same outcome.
|
||||
return new FleetConfig(b, herdrSocket, memberHerdrSocket, profiles, g, worktreeRoot, l, timeout, pollMs,
|
||||
broker, primary, f, leadHeartbeat, health, placementOrDefault, a, configReload,
|
||||
quarantineCooldown, mc, coordinator, worktreeGroup, memberLoginShell);
|
||||
quarantineCooldown, mc, coordinator, worktreeGroup, memberLoginShell, memberSkills,
|
||||
idleSleepGuard);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -14,6 +14,7 @@ import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiConsumer;
|
||||
@@ -28,17 +29,65 @@ public final class FleetHealthMonitor {
|
||||
static final int MAX_FAIL_TARGET_ATTEMPTS = 3;
|
||||
// CB-641: Match the injector's 60s readiness gate so health allows a full first boot.
|
||||
static final long READINESS_GRACE_NANOS = TimeUnit.SECONDS.toNanos(60);
|
||||
/**
|
||||
* fleetd #280: how long after a terminal transition to wait before the one bounded re-check
|
||||
* fires. Must exceed the worst-case reverse-rendezvous {@code fleet_ask} window (55-115s, see
|
||||
* {@code FleetMcp.ASK_DEFAULT_TIMEOUT_MS} / {@code FleetApp.MAX_ASK_TIMEOUT_MS}) so that, if the
|
||||
* target was genuinely {@code ASKING} when {@code state} was first observed, its own ask has had
|
||||
* time to lapse (clearing {@code Task#question} back to {@code null}) before this fires.
|
||||
*/
|
||||
static final long ASK_LAPSE_RECHECK_DELAY_SECONDS = 120;
|
||||
|
||||
/**
|
||||
* fleetd #386: {@code System.nanoTime()} (or whatever {@link #clock} is) does not advance while
|
||||
* macOS sleeps, so a raw {@code nowNanos - lastActivityAtNanos} comparison freezes with the
|
||||
* host and can never cross {@link #workingSuspectAfterNanos}. This is a second, wall-clock
|
||||
* source used ONLY inside the stall check ({@link #stallElapsedNanos}) to detect and correct
|
||||
* for that freeze. Nothing else in this class reads it — every other decision (readiness grace,
|
||||
* the fault classification itself) stays exactly on {@link #clock}, as the ticket requires.
|
||||
*/
|
||||
private static final LongSupplier DEFAULT_REALTIME_CLOCK =
|
||||
() -> TimeUnit.MILLISECONDS.toNanos(System.currentTimeMillis());
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final MessageService messages;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final LongSupplier realtimeClock;
|
||||
private final long intervalSeconds;
|
||||
private final long tickIntervalNanos;
|
||||
private final long workingSuspectAfterNanos;
|
||||
private final BiConsumer<String, String> failTarget;
|
||||
private final Map<String, HealthPrior> priors = new HashMap<>();
|
||||
private final Map<String, HealthState> states = new HashMap<>();
|
||||
/**
|
||||
* fleetd #386 clock-drift bookkeeping. {@code haveClockBaseline}/{@code lastTickMonoNanos}/
|
||||
* {@code lastTickRealNanos} track the previous tick's pair of readings so each new tick can
|
||||
* measure how far the two clocks moved apart since then. {@code accumulatedDriftNanos} is the
|
||||
* running total of every such divergence observed since this monitor started (never decreases —
|
||||
* the monotonic clock can only lag real time, never lead it). {@code busyDriftBaselineNanos}/
|
||||
* {@code busyBaselineActivityNanos} record, per target, the value of {@code accumulatedDriftNanos}
|
||||
* at the moment this monitor first saw that target's CURRENT {@code lastActivityAtNanos} while
|
||||
* BUSY — so {@link #stallElapsedNanos} adds back only the drift observed DURING this BUSY span,
|
||||
* never drift from a sleep that happened before the member went busy. All five fields are touched
|
||||
* only from {@code tick()}, like {@link #priors}.
|
||||
*/
|
||||
private boolean haveClockBaseline = false;
|
||||
private long lastTickMonoNanos;
|
||||
private long lastTickRealNanos;
|
||||
private long accumulatedDriftNanos = 0;
|
||||
private final Map<String, Long> busyDriftBaselineNanos = new HashMap<>();
|
||||
private final Map<String, Long> busyBaselineActivityNanos = new HashMap<>();
|
||||
/**
|
||||
* The live classification per member, and the only one of this class's three maps that more
|
||||
* than one scheduler task touches. {@code tick} writes it (and prunes it to the roster);
|
||||
* fleetd #280's delayed {@link #recheckTerminalTarget} reads it from its own separate scheduled
|
||||
* task. Both run on the single-threaded scheduler {@code Fleetd} passes in today, so they are
|
||||
* serialised — but nothing in this class enforces that, and an unsynchronised {@link HashMap}
|
||||
* read racing a resize can spin a CPU forever rather than fail visibly. {@code priors} and
|
||||
* {@code orphanStreaks} stay plain maps because {@code tick} is still their only toucher.
|
||||
*/
|
||||
private final Map<String, HealthState> states = new ConcurrentHashMap<>();
|
||||
/**
|
||||
* CB-643: consecutive ticks on which a target looked like an orphaned delegation. The fact
|
||||
* {@link MessageService#hasOrphanedDelegation} reports is a true snapshot, but it can read true
|
||||
@@ -70,12 +119,29 @@ public final class FleetHealthMonitor {
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, long intervalSeconds,
|
||||
long workingSuspectAfterSeconds, BiConsumer<String, String> failTarget) {
|
||||
this(agents, roster, messages, scheduler, clock, DEFAULT_REALTIME_CLOCK, intervalSeconds,
|
||||
workingSuspectAfterSeconds, failTarget);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param realtimeClock fleetd #386: a wall-clock nanosecond source (e.g.
|
||||
* {@code System.currentTimeMillis()} converted to nanos) that keeps
|
||||
* advancing while {@code clock} is frozen by a host sleep. Used only to
|
||||
* correct the stall check — see the class-level javadoc on the
|
||||
* clock-drift fields.
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, LongSupplier realtimeClock,
|
||||
long intervalSeconds, long workingSuspectAfterSeconds,
|
||||
BiConsumer<String, String> failTarget) {
|
||||
this.agents = agents;
|
||||
this.roster = roster;
|
||||
this.messages = messages;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.realtimeClock = Objects.requireNonNull(realtimeClock, "realtimeClock");
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.tickIntervalNanos = TimeUnit.SECONDS.toNanos(intervalSeconds);
|
||||
this.workingSuspectAfterNanos = TimeUnit.SECONDS.toNanos(workingSuspectAfterSeconds);
|
||||
this.failTarget = Objects.requireNonNull(failTarget, "failTarget");
|
||||
}
|
||||
@@ -105,6 +171,7 @@ public final class FleetHealthMonitor {
|
||||
for (Agent agent : agentsNow) live.put(agent.terminalId(), agent);
|
||||
HashSet<String> current = new HashSet<>();
|
||||
long nowNanos = clock.getAsLong();
|
||||
long driftBeforeThisTick = observeClockDrift(nowNanos);
|
||||
for (MemberSession session : rosterNow) {
|
||||
current.add(session.terminalId());
|
||||
Agent agent = live.get(session.terminalId());
|
||||
@@ -115,7 +182,7 @@ public final class FleetHealthMonitor {
|
||||
&& session.state() != MemberSession.State.SPAWNING;
|
||||
boolean readinessGraceElapsed = nowNanos - session.spawnedAtNanos() >= READINESS_GRACE_NANOS;
|
||||
boolean stalled = session.state() == MemberSession.State.BUSY
|
||||
&& nowNanos - session.lastActivityAtNanos() >= workingSuspectAfterNanos;
|
||||
&& stallElapsedNanos(session, nowNanos, driftBeforeThisTick) >= workingSuspectAfterNanos;
|
||||
// CB-643: the three message-layer facts CB-640 published. Read them here rather than
|
||||
// leaving them false — that constant is what made 8 of the 9 fault states dead.
|
||||
boolean queuedDelivery = messages.hasQueuedDelivery(session.terminalId());
|
||||
@@ -133,6 +200,8 @@ public final class FleetHealthMonitor {
|
||||
priors.keySet().retainAll(current);
|
||||
states.keySet().retainAll(current);
|
||||
orphanStreaks.keySet().retainAll(current);
|
||||
busyDriftBaselineNanos.keySet().retainAll(current);
|
||||
busyBaselineActivityNanos.keySet().retainAll(current);
|
||||
} catch (Throwable error) {
|
||||
// Any unclassified collection failure must never kill the monitor's only scheduler task.
|
||||
log.warn("fleet health collection failed; will retry next tick", error);
|
||||
@@ -157,6 +226,65 @@ public final class FleetHealthMonitor {
|
||||
return streak >= ORPHAN_CONFIRM_TICKS;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #386: compare this tick's monotonic and real-time readings against the previous
|
||||
* tick's, and fold any positive divergence into {@link #accumulatedDriftNanos} (a ratchet — it
|
||||
* never decreases, since the monotonic clock can only fall behind real time, never ahead of
|
||||
* it). Logs once, at WARN, when that single tick's divergence exceeds one full tick interval —
|
||||
* the signature of a host that slept between the two ticks (a tick literally cannot run while
|
||||
* the process itself is suspended, so the whole sleep duration lands inside one tick's gap).
|
||||
*
|
||||
* @return {@link #accumulatedDriftNanos} as it stood BEFORE this tick's divergence was folded
|
||||
* in — the baseline {@link #stallElapsedNanos} needs when a target is observed BUSY
|
||||
* for the first time this tick, so a sleep that happened before this member went busy
|
||||
* is not attributed to it.
|
||||
*/
|
||||
private long observeClockDrift(long nowNanos) {
|
||||
long nowRealNanos = realtimeClock.getAsLong();
|
||||
long driftBeforeThisTick = accumulatedDriftNanos;
|
||||
if (haveClockBaseline) {
|
||||
long monoDelta = nowNanos - lastTickMonoNanos;
|
||||
long realDelta = nowRealNanos - lastTickRealNanos;
|
||||
long tickDrift = realDelta - monoDelta;
|
||||
if (tickDrift > tickIntervalNanos) {
|
||||
log.warn("fleet health: the monotonic clock did not advance for about {}s that the "
|
||||
+ "real clock did since the last tick (host likely slept); the stall "
|
||||
+ "detector could not see that time", TimeUnit.NANOSECONDS.toSeconds(tickDrift));
|
||||
}
|
||||
if (tickDrift > 0) {
|
||||
accumulatedDriftNanos = driftBeforeThisTick + tickDrift;
|
||||
}
|
||||
}
|
||||
lastTickMonoNanos = nowNanos;
|
||||
lastTickRealNanos = nowRealNanos;
|
||||
haveClockBaseline = true;
|
||||
return driftBeforeThisTick;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #386: {@code nowNanos - lastActivityAtNanos} alone freezes across a host sleep, since
|
||||
* both come from the monotonic {@link #clock}. This adds back the real-time drift observed
|
||||
* since this BUSY span started — not the monitor's whole lifetime, so a sleep that happened
|
||||
* before this member went busy never leaks into its stall reading (see the class-level javadoc
|
||||
* on the drift fields). The baseline resets whenever {@code lastActivityAtNanos} changes (a new
|
||||
* turn) or the member is not currently BUSY.
|
||||
*/
|
||||
private long stallElapsedNanos(MemberSession session, long nowNanos, long driftBeforeThisTick) {
|
||||
String target = session.terminalId();
|
||||
if (session.state() != MemberSession.State.BUSY) {
|
||||
busyDriftBaselineNanos.remove(target);
|
||||
busyBaselineActivityNanos.remove(target);
|
||||
return nowNanos - session.lastActivityAtNanos();
|
||||
}
|
||||
Long baselineActivity = busyBaselineActivityNanos.get(target);
|
||||
if (baselineActivity == null || baselineActivity != session.lastActivityAtNanos()) {
|
||||
busyBaselineActivityNanos.put(target, session.lastActivityAtNanos());
|
||||
busyDriftBaselineNanos.put(target, driftBeforeThisTick);
|
||||
}
|
||||
long driftSinceBusyStart = accumulatedDriftNanos - busyDriftBaselineNanos.get(target);
|
||||
return (nowNanos - session.lastActivityAtNanos()) + driftSinceBusyStart;
|
||||
}
|
||||
|
||||
void reportTransition(String target, HealthState next) {
|
||||
HealthState previous = states.put(target, next);
|
||||
if (previous == next) return;
|
||||
@@ -171,6 +299,7 @@ public final class FleetHealthMonitor {
|
||||
// member stayed terminal.
|
||||
if (terminal(next)) {
|
||||
failTerminalTarget(target, next);
|
||||
scheduleTerminalRecheck(target, next);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -191,6 +320,48 @@ public final class FleetHealthMonitor {
|
||||
target, state, MAX_FAIL_TARGET_ATTEMPTS, last);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #280: schedule the one bounded, delayed follow-up for a terminal transition — never a
|
||||
* per-tick retry (CB-580 rejected that shape; {@link #reportTransition} still fires
|
||||
* {@link #failTerminalTarget} exactly once per transition, unconditionally on the tick loop).
|
||||
* This is a single one-shot task, scheduled once per transition into GONE/NEVER_READY, so a
|
||||
* member stuck terminal for the rest of its life gets exactly one extra attempt, not one per
|
||||
* tick. See {@link #recheckTerminalTarget} for why the extra attempt is safe.
|
||||
*/
|
||||
private void scheduleTerminalRecheck(String target, HealthState state) {
|
||||
if (scheduler.isShutdown()) return;
|
||||
try {
|
||||
scheduler.schedule(() -> recheckTerminalTarget(target, state),
|
||||
ASK_LAPSE_RECHECK_DELAY_SECONDS, TimeUnit.SECONDS);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("fleet health: could not schedule terminal re-check for member={} state={}",
|
||||
target, state, e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #280: the delayed re-check {@link #scheduleTerminalRecheck} scheduled for one terminal
|
||||
* transition. By now, a {@code fleet_ask} that was still open when {@code state} was first
|
||||
* observed has had time to lapse on its own (see {@link #ASK_LAPSE_RECHECK_DELAY_SECONDS}),
|
||||
* clearing {@code Task#question} back to {@code null} — which is exactly what
|
||||
* {@link MessageService#abandon(String, String, boolean)}'s {@code sweepAsking=false} filter
|
||||
* needs to finally match it. Calling {@link #failTerminalTarget} again is safe only because
|
||||
* {@code sweepAsking} stays {@code false}: a task genuinely still {@code ASKING} is skipped
|
||||
* exactly as it was on the very first attempt — this never fails a ticket whose ask has not yet
|
||||
* lapsed.
|
||||
*
|
||||
* <p><strong>Guarded on "target is still classified {@code state}."</strong> Without this guard,
|
||||
* a member that recovered (or was released and dropped from the roster) between the transition
|
||||
* and this re-check would still take a blind {@code failTarget} call — reaching into whatever
|
||||
* brand-new, unrelated turn it has since picked up and failing it too. {@link #states} already
|
||||
* carries the live classification (updated every tick, pruned to the current roster on release),
|
||||
* so a stale or recovered target simply reads as a mismatch here and this is a no-op.
|
||||
*/
|
||||
void recheckTerminalTarget(String target, HealthState state) {
|
||||
if (states.get(target) != state) return;
|
||||
failTerminalTarget(target, state);
|
||||
}
|
||||
|
||||
private static boolean terminal(HealthState state) {
|
||||
return state == HealthState.GONE || state == HealthState.NEVER_READY;
|
||||
}
|
||||
|
||||
@@ -5,6 +5,7 @@ import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
@@ -53,17 +54,40 @@ import java.util.function.Supplier;
|
||||
* daemon and found again by this scan. The trust direction above is unaffected — fleetd writing a
|
||||
* name for a lead it just started is not a pane promoting itself — but <em>staleness</em> becomes
|
||||
* real: a label left behind by a session that has since died would read as a live lead forever.
|
||||
* This scanner does not solve that (its job is naming, and a stale name costs nothing here); the
|
||||
* launcher does, by requiring a running agent in the tab before it counts the lead as live. If you
|
||||
* ever make a decision that <em>removes</em> something based on this map, add the same check.
|
||||
* The remaining hazard is an <em>operator</em> one — a worker {@code tabLabel} template that
|
||||
* happens to start with the same prefix would promote the whole fleet — and that is refused at
|
||||
* startup by {@code FleetConfig.validateLeadTabPrefixes} rather than documented here.
|
||||
*
|
||||
* <p><strong>fleetd #359 — the staleness check this class used to skip.</strong> This used to say
|
||||
* "a stale name costs nothing here" and leave liveness to {@code LeadLauncher}, on the theory that
|
||||
* naming and removing are different decisions. That was wrong: {@code dev.ltms.fleet.msg.LeadCoordLoop}
|
||||
* makes exactly the kind of removal decision the old javadoc warned about, by reading this map to
|
||||
* pick which pane a peer message goes into — and a stale entry there is not free. On a host where
|
||||
* {@code fleetd} had restarted more than once, a labelled-but-dead tab from a previous life was
|
||||
* reported right alongside the live one; {@code LeadCoordLoop.resolveLocalLead()} saw more than one
|
||||
* candidate and refused to guess (safe), but the fix the daemon's own WARN suggests — name a lead
|
||||
* after {@code coordinator.selfId} — stops being safe once two tabs can share a label: step 1 of
|
||||
* that resolution picks whichever matching entry it finds first, which can be the dead one, and
|
||||
* typing a peer's message into a dead shell does not fail — it is silently gone instead of merely
|
||||
* held. {@link #scan()} now cross-checks every labelled tab against {@code agent.list} (the same
|
||||
* signal {@code LeadLauncher.countLeads} already trusts for the same purpose) and drops any tab
|
||||
* with no agent running in it, so a dead tab is never in the map for a caller to pick at all.
|
||||
*
|
||||
* <p><strong>Caching.</strong> {@link #get()} is on the request path (every resolve), so the scan
|
||||
* is TTL-cached and a stale-but-valid map is preferred to a herdr round-trip. A failed scan keeps
|
||||
* the previous answer instead of emptying it — a herdr hiccup must not silently demote a live lead
|
||||
* mid-session.
|
||||
* is TTL-cached and a stale-but-valid map is preferred to a herdr round-trip. A failed scan (herdr
|
||||
* throws) keeps the previous answer instead of emptying it — a herdr hiccup must not silently
|
||||
* demote a live lead mid-session.
|
||||
*
|
||||
* <p><strong>fleetd #359 review, finding 2 — a successful-but-wrong scan is the same hazard.</strong>
|
||||
* The catch above only fires when a call throws. It does nothing for a call that returns 200 with an
|
||||
* incomplete answer — exactly what the ticket's own evidence showed {@code agent.list} can do. Once
|
||||
* this class started trusting that signal, an empty read would otherwise get cached as fact and
|
||||
* silently drop a lead {@code CallerResolver} had, until then, correctly resolved — turning it into a
|
||||
* {@code Role.WORKER}, which refuses every orchestration call. So a terminal this class already
|
||||
* reported as live is not dropped the first time {@code agent.list} loses it: {@link #scan()} grants
|
||||
* it one grace scan (see {@code gracedTerminals}) and only drops it if a <em>later</em> scan still
|
||||
* finds no agent. A terminal never reported live before gets no grace — that would weaken the
|
||||
* original #359 fix itself, which this class's own test suite already pins.
|
||||
*/
|
||||
public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
|
||||
@@ -79,6 +103,15 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
private long scannedAtNanos;
|
||||
private boolean everScanned;
|
||||
|
||||
/**
|
||||
* Terminals currently on their one grace scan: {@code cached} reported them live, the most
|
||||
* recent {@link #scan()} found no agent for them, and they were re-included anyway. Cleared for
|
||||
* a terminal the instant it is seen live again; a terminal still here on the <em>next</em> scan
|
||||
* is finally dropped. Scoped separately from {@link #cached} so a graced terminal cannot renew
|
||||
* its own grace forever just by staying in the exposed map (fleetd #359 review, finding 2).
|
||||
*/
|
||||
private Set<String> gracedTerminals = Set.of();
|
||||
|
||||
/**
|
||||
* @param herdr the herdr client to query ({@code workspace.list},
|
||||
* {@code tab.list}, {@code pane.list} — all read-only)
|
||||
@@ -141,7 +174,7 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
return cached;
|
||||
}
|
||||
|
||||
/** One full pass: labelled tabs → their panes → those panes' terminals. */
|
||||
/** One full pass: labelled tabs → live agents in them → those panes' terminals. */
|
||||
private Map<String, String> scan() {
|
||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
||||
for (JsonNode w : herdr.call("workspace.list").path("workspaces")) {
|
||||
@@ -158,17 +191,47 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
}
|
||||
}
|
||||
|
||||
Map<String, String> byTerminal = new LinkedHashMap<>();
|
||||
if (!nameByTab.isEmpty()) {
|
||||
// One pane.list for every tab: panes carry tab_id, so the join is local.
|
||||
for (JsonNode p : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String name = nameByTab.get(p.path("tab_id").asText(null));
|
||||
String terminal = p.path("terminal_id").asText(null);
|
||||
if (name != null && terminal != null && !terminal.isBlank()) {
|
||||
byTerminal.put(terminal, name);
|
||||
}
|
||||
if (nameByTab.isEmpty()) {
|
||||
gracedTerminals = Set.of();
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
// fleetd #359: a labelled tab is only a lead when herdr also reports a running agent in
|
||||
// it — the same liveness signal LeadLauncher.countLeads trusts for the identical purpose.
|
||||
// Without this, a tab left behind by a session that has since died reads as live forever.
|
||||
Set<String> tabsWithAgent = new HashSet<>();
|
||||
for (JsonNode a : herdr.call("agent.list").path("agents")) {
|
||||
String tabId = a.path("tab_id").asText(null);
|
||||
if (tabId != null) {
|
||||
tabsWithAgent.add(tabId);
|
||||
}
|
||||
}
|
||||
|
||||
Map<String, String> byTerminal = new LinkedHashMap<>();
|
||||
Set<String> stillGraced = new HashSet<>();
|
||||
// One pane.list for every tab: panes carry tab_id, so the join is local.
|
||||
for (JsonNode p : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String tabId = p.path("tab_id").asText(null);
|
||||
String name = nameByTab.get(tabId);
|
||||
String terminal = p.path("terminal_id").asText(null);
|
||||
if (name == null || terminal == null || terminal.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
if (tabsWithAgent.contains(tabId)) {
|
||||
byTerminal.put(terminal, name);
|
||||
continue;
|
||||
}
|
||||
// No agent reported for this tab, but its tab/pane are still here — this is the
|
||||
// ambiguous case review finding 2 named: a successful agent.list that came back short
|
||||
// does not prove the lead is dead. Grant one grace scan to a terminal we had already
|
||||
// reported as live; a terminal we never reported live gets none, so the original #359
|
||||
// fix (a genuinely dead tab is never reported) is unaffected for the common case.
|
||||
if (cached.containsKey(terminal) && !gracedTerminals.contains(terminal)) {
|
||||
byTerminal.put(terminal, name);
|
||||
stillGraced.add(terminal);
|
||||
}
|
||||
}
|
||||
gracedTerminals = stillGraced;
|
||||
return Collections.unmodifiableMap(byTerminal);
|
||||
}
|
||||
|
||||
@@ -177,12 +240,14 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
*
|
||||
* <p>Exact match (case-insensitive, ends stripped) against {@link #tabToName} — no prefix
|
||||
* stripping, so an operator's {@code "lead: something-else"} tab is never mistaken for a
|
||||
* configured lead just because it shares a prefix.
|
||||
* configured lead just because it shares a prefix. The match strips a trailing
|
||||
* {@link PendingCloseMarker} first, so a tab {@code LeadLauncher} has flagged as maybe-dead but
|
||||
* not yet closed keeps resolving normally while that reconcile is pending.
|
||||
*/
|
||||
private String leadNameOf(String label) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
return tabToName.get(label.strip().toLowerCase(Locale.ROOT));
|
||||
return tabToName.get(PendingCloseMarker.strip(label).toLowerCase(Locale.ROOT));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
/**
|
||||
* The suffix {@code dev.ltms.fleet.lead.LeadLauncher} appends to a lead tab's label the first time a
|
||||
* reconcile finds no running agent in it, before it is sure enough to close the tab outright.
|
||||
*
|
||||
* <p><strong>fleetd #359 review, finding 1.</strong> The daemon's own evidence showed
|
||||
* {@code agent.list} can read "no agent" for a tab that genuinely has one running — so a single such
|
||||
* reading must never be treated as proof a tab is dead. {@code LeadLauncher} now writes this marker
|
||||
* on the first miss, and only closes the tab if a <em>later</em>, independent reconcile still finds
|
||||
* it dead while the marker is still there. Two consecutive misses, one restart apart, is a much
|
||||
* stronger claim than one.
|
||||
*
|
||||
* <p>{@link LeadTabScanner} strips the same suffix before matching a label against a configured
|
||||
* lead's {@code tab}, so a flagged-but-actually-still-live tab keeps resolving normally — the marker
|
||||
* changes nothing about which pane {@code LeadCoordLoop} can reach while the flag is pending. Both
|
||||
* classes must use exactly this suffix, which is why it lives here rather than as a private constant
|
||||
* on either.
|
||||
*/
|
||||
public final class PendingCloseMarker {
|
||||
|
||||
public static final String SUFFIX = " [fleetd:pending-close]";
|
||||
|
||||
private PendingCloseMarker() {
|
||||
}
|
||||
|
||||
/** The label with any trailing pending-close marker removed, for name matching. */
|
||||
public static String strip(String label) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
String stripped = label.strip();
|
||||
return stripped.endsWith(SUFFIX)
|
||||
? stripped.substring(0, stripped.length() - SUFFIX.length()).strip()
|
||||
: stripped;
|
||||
}
|
||||
|
||||
/** Whether a label currently carries the marker. */
|
||||
public static boolean isFlagged(String label) {
|
||||
return label != null && label.strip().endsWith(SUFFIX);
|
||||
}
|
||||
}
|
||||
@@ -1,10 +1,12 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
/**
|
||||
* Notified when {@link CompletionResolver} actually delivers a typed backend-error classification
|
||||
* to a waiting send (fleetd#201 / #227) — never on a race that lost. {@link CompletionResolver}
|
||||
* calls this only after {@code Rendezvous.resolveFailure} returns {@code true} for that exact
|
||||
* waiter, mirroring the win-only race rule {@link ExhaustionSink} already uses.
|
||||
* Notified when {@link CompletionResolver} has a backend-error match at the start of a pane line,
|
||||
* or both a match and its too-fast crash signature, for a waiting send (fleetd#201 / #227). A text
|
||||
* match inside ordinary pane prose can be a member's report about an error, so it fails the send
|
||||
* without notifying this sink.
|
||||
* {@link CompletionResolver} calls this only after {@code Rendezvous.resolveFailure} returns
|
||||
* {@code true} for that exact waiter, mirroring the win-only race rule {@link ExhaustionSink} uses.
|
||||
*
|
||||
* <p>The public send result is unchanged by this classification — it is still a failed send
|
||||
* ({@code Rendezvous.Kind#FAILED}); this sink is the internal seam a later stage (fleetd#201 Unit
|
||||
|
||||
@@ -380,16 +380,19 @@ public final class CompletionResolver implements TurnListener {
|
||||
+ "matched the profile's exhausted pattern): {}", target, reason);
|
||||
// CB-578 stage B: only on the resolution that actually won the race — a late
|
||||
// duplicate must never quarantine a credential twice for one refusal.
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
if (startsWithExhaustion(matchedLine, exhausted)) {
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
// fleetd#164 (part 2) / fleetd#201: a scrape that read cleanly and produced content still
|
||||
// isn't a real reply when that content is the backend's own rejection (e.g. an HTTP 400
|
||||
// before the worker did any work). Classify it as a failure naming the member, rather than
|
||||
// handing the caller a scrape that reads like a completed answer, and — only on the
|
||||
// resolution that actually wins the race, mirroring the exhaustion sink above — notify the
|
||||
// typed backend-error sink so a later stage can act on repeated failures.
|
||||
// handing the caller a scrape that reads like a completed answer. A text match alone is not
|
||||
// enough to notify the typed backend-error sink: this assistant block can be a member's
|
||||
// normal prose about an error. A line that starts with the error match is stronger evidence;
|
||||
// the too-fast path below also has its crash signature before it records a credential failure.
|
||||
String backendError = firstMatchingLine(assistantBlock, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
// Carry the whole scrape, not just the matched line. The pattern is a heuristic: a member
|
||||
@@ -401,9 +404,9 @@ public final class CompletionResolver implements TurnListener {
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
// fleetd#201 Unit 1: only on the resolution that actually won the race — a late
|
||||
// duplicate must never double-count one backend failure.
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
if (startsWithBackendError(backendError, backendErrorPatternOrFallback(target))) {
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -477,7 +480,9 @@ public final class CompletionResolver implements TurnListener {
|
||||
+ "usable assistant block; no fleet_reply): {}", target, reason);
|
||||
// CB-578 stage B: only on the resolution that actually won the race — a late
|
||||
// duplicate must never quarantine a credential twice for one refusal.
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
if (startsWithExhaustion(matchedLine, exhausted)) {
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -492,8 +497,9 @@ public final class CompletionResolver implements TurnListener {
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback from the raw scrape: {}", target, reason);
|
||||
// fleetd#201 Unit 1: only on the resolution that actually won the race.
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
if (startsWithBackendError(backendError, backendErrorPatternOrFallback(target))) {
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
@@ -540,10 +546,21 @@ public final class CompletionResolver implements TurnListener {
|
||||
* {@link #MIN_TURN_NANOS} — a crash signature (e.g. a backend HTTP 400 before the worker did
|
||||
* anything) that a bare {@code BUSY -> DONE} transition cannot be told apart from a genuinely
|
||||
* fast completion. Runs the same backend-error classification the normal and raw-scrape paths
|
||||
* apply, against whatever is on screen right now: a match is a typed failure that notifies
|
||||
* {@link #backendErrorSink} (only on the resolution that wins the race); a non-match stays the
|
||||
* original generic too-fast failure, naming the member and both timings, with whatever the pane
|
||||
* shows appended so the caller sees the cause, not just "it failed".
|
||||
* apply, against whatever is on screen right now: a match together with the too-fast crash
|
||||
* signature notifies {@link #backendErrorSink} (only on the resolution that wins the race). A
|
||||
* non-match stays the generic too-fast failure, naming the member and both timings, with
|
||||
* whatever the pane shows appended so the caller sees the evidence, not just "it failed".
|
||||
*
|
||||
* <p>fleetd#376: <strong>this path must never resolve a completion.</strong> A fast backend can
|
||||
* genuinely answer inside the floor, so the failure is sometimes wrong — but it is wrong in the
|
||||
* loud direction, and the fix for that is honest wording, not a guess at the pane's meaning.
|
||||
* Reclassifying from the scrape was tried and rejected: there is no reliable positive marker for
|
||||
* "this is a real reply" across backends. {@link #lastAssistantBlock} falls back to the entire
|
||||
* pane when it finds no {@code ⏺} marker, so on a crash the candidate "reply" is the whole
|
||||
* screen; and {@code ⏺} itself is a Claude Code marker that an opencode pane never carries — the
|
||||
* very backend whose speed raised this ticket. Any weaker test (non-blank, or "contains sentence
|
||||
* punctuation") passes on almost every crash, because a pane holding a file path, a version
|
||||
* number or a hostname contains a full stop. That trades a loud wrong answer for a silent one.
|
||||
*/
|
||||
private void failTooFast(String target, InFlight turn, CompletableFuture<Rendezvous.Resolution> waiter,
|
||||
long elapsedNanos) {
|
||||
@@ -554,13 +571,13 @@ public final class CompletionResolver implements TurnListener {
|
||||
scrape = "";
|
||||
}
|
||||
String clippedScrape = clip(scrape);
|
||||
String baseReason = String.format(
|
||||
"member %s went BUSY -> DONE in %dms (floor %dms) — too fast to be real work, most "
|
||||
+ "likely a backend error before any work started",
|
||||
String timing = String.format(
|
||||
"member %s went BUSY -> DONE in %dms (floor %dms)",
|
||||
target, elapsedNanos / 1_000_000, MIN_TURN_NANOS / 1_000_000);
|
||||
String backendError = firstMatchingLine(scrape, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
String reason = baseReason + ": " + clippedScrape;
|
||||
String reason = timing + " — too fast to be real work, and the pane carries a backend "
|
||||
+ "error: " + clippedScrape;
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
@@ -569,7 +586,14 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
return;
|
||||
}
|
||||
fail(target, turn, clippedScrape.isBlank() ? baseReason : baseReason + ": " + clippedScrape);
|
||||
// fleetd#376: no pattern matched, so the cause is genuinely unknown. Say that, rather than
|
||||
// asserting a backend error the way this message used to — a fast backend really can finish
|
||||
// inside the floor, and a reader who trusts a wrong cause stops looking at the pane.
|
||||
String reason = timing + " — inside the floor. That is usually a backend error before any "
|
||||
+ "work started, but a fast backend can answer inside it too, and nothing here can "
|
||||
+ "tell those apart, so the turn is reported failed rather than guessed. Read the "
|
||||
+ "pane below before deciding which it was";
|
||||
fail(target, turn, clippedScrape.isBlank() ? reason : reason + ": " + clippedScrape);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -611,6 +635,68 @@ public final class CompletionResolver implements TurnListener {
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when nothing before the match on this pane line ends a sentence — that is, the match is
|
||||
* still inside the line's first sentence rather than inside prose a member wrote about it.
|
||||
* Used to decide whether an exhaustion match may quarantine a credential (fleetd #348).
|
||||
*
|
||||
* <p><strong>Why this is looser than {@link #startsWithBackendError}.</strong> An
|
||||
* {@code exhaustedPattern} is written per profile and may name only the decisive words of a
|
||||
* provider message — {@code "usage limit has been reached"} without its leading {@code "The"}.
|
||||
* A start-of-line check would then reject the genuine refusal. That is the false negative
|
||||
* fleetd #348's invariant 1 calls the worse direction: an unrecorded exhaustion leaves the
|
||||
* fleet spawning into a credential with no capacity, and a quarantine runs 1800s against the
|
||||
* backend-error cooldown's fixed 60s.
|
||||
*
|
||||
* <p>This rule accepts a superset of what a start-of-line check accepts: if the match begins
|
||||
* right after the chrome, there is nothing in front of it, so there is no sentence ending
|
||||
* either. So moving to it cannot add a false negative.
|
||||
*
|
||||
* <p><strong>No chrome skipping here, deliberately.</strong> The first version of this method
|
||||
* copied {@code startsWithBackendError}'s leading-chrome loop. Measured on merge: deleting that
|
||||
* loop left all 1369 tests green, and it must — the scan only looks for {@code . ! ?}, and no
|
||||
* terminal chrome character is one of those. A step that cannot change the result is worse than
|
||||
* no step, because the next reader takes it as evidence that chrome was handled.
|
||||
*
|
||||
* <p>It stays a heuristic. Prose whose <em>first</em> sentence carries the pattern still
|
||||
* notifies the sink, and a genuine refusal behind an earlier full stop (a hostname, a version
|
||||
* number) still does not. Both are known and neither is fixed here.
|
||||
*/
|
||||
private static boolean startsWithExhaustion(String line, Pattern pattern) {
|
||||
var matcher = pattern.matcher(line);
|
||||
if (!matcher.find()) {
|
||||
return false;
|
||||
}
|
||||
for (int prefix = 0; prefix < matcher.start(); prefix++) {
|
||||
if (".!?".indexOf(line.charAt(prefix)) >= 0) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the error pattern begins the matched pane line, rather than appearing in prose.
|
||||
*
|
||||
* <p>Leading terminal chrome is skipped first — box-drawing characters, bullets, gutter bars and
|
||||
* spaces. #339 introduced this check with a bare {@code lookingAt}, and that rejected a genuine
|
||||
* error line rendered as {@code "| 503 Service Unavailable: ..."}: the send still failed, but the
|
||||
* credential outage was never recorded. That is the false negative #339's own invariant 3 called
|
||||
* worse than the false positive it set out to fix — measured with a throwaway probe on the
|
||||
* raw-scrape path, which is exactly the path whose comment says to expect leading chrome.
|
||||
*
|
||||
* <p>Skipping only a leading run of non-letter, non-digit characters keeps the fix's intent. A
|
||||
* member's prose ({@code "I checked the retry path. An API Error: makes it back off."}) still
|
||||
* does not match, because there the pattern sits after words, not after chrome.
|
||||
*/
|
||||
private static boolean startsWithBackendError(String line, Pattern pattern) {
|
||||
int i = 0;
|
||||
while (i < line.length() && !Character.isLetterOrDigit(line.charAt(i))) {
|
||||
i++;
|
||||
}
|
||||
return pattern.matcher(line.substring(i)).lookingAt();
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for the CB-578 stage A exhausted-pattern classification, logged at startup
|
||||
* the way {@link dev.ltms.fleet.health.FleetHealthMonitor#coverage} is — so an operator can
|
||||
|
||||
@@ -145,8 +145,58 @@ public final class Injector {
|
||||
return router != null ? router.agentsFor(target) : agents;
|
||||
}
|
||||
|
||||
/** The result of trying to remove an undelivered message from the injector. */
|
||||
public enum Cancellation {
|
||||
CANCELLED,
|
||||
DELIVERED,
|
||||
NOT_DELIVERED
|
||||
}
|
||||
|
||||
/**
|
||||
* An identity handle for one queued delivery. It is the only value accepted by
|
||||
* {@link #cancel(Delivery)}, so a caller cannot cancel a different message with the same target
|
||||
* or text.
|
||||
*/
|
||||
public static final class Delivery {
|
||||
private final Pending pending;
|
||||
|
||||
private Delivery(Pending pending) {
|
||||
this.pending = pending;
|
||||
}
|
||||
|
||||
public CompletableFuture<Void> completion() {
|
||||
return pending.delivered;
|
||||
}
|
||||
}
|
||||
|
||||
/** A pending message and the future that completes when it has been delivered. */
|
||||
private record Pending(String text, TurnToken token, CompletableFuture<Void> delivered) {
|
||||
private static final class Pending {
|
||||
enum State { QUEUED, DELIVERED, NOT_DELIVERED, CANCELLED }
|
||||
|
||||
final String target;
|
||||
final String text;
|
||||
final TurnToken token;
|
||||
final CompletableFuture<Void> delivered;
|
||||
volatile State state = State.QUEUED; // written under the owning Target monitor
|
||||
|
||||
Pending(String target, String text, TurnToken token, CompletableFuture<Void> delivered) {
|
||||
this.target = target;
|
||||
this.text = text;
|
||||
this.token = token;
|
||||
this.delivered = delivered;
|
||||
}
|
||||
|
||||
String text() {
|
||||
return text;
|
||||
}
|
||||
|
||||
TurnToken token() {
|
||||
return token;
|
||||
}
|
||||
|
||||
CompletableFuture<Void> delivered() {
|
||||
return delivered;
|
||||
}
|
||||
}
|
||||
|
||||
/** Per-worker delivery state, guarded by its own monitor (single writer per worker). */
|
||||
@@ -157,6 +207,7 @@ public final class Injector {
|
||||
boolean awaitingCompletion; // a delivered message's turn is not yet known-complete
|
||||
boolean turnObserved; // saw a real `working` sample since that delivery (turn ran)
|
||||
int unknownSinceTurn; // consecutive `unknown` samples while a delegation is outstanding (CB-109)
|
||||
int unknownSincePostTurn; // the same, for the post-turn housekeeping phase (fleetd #306)
|
||||
int notReadySincePoll; // consecutive injectable samples a queued message waited on the readiness gate (CB-114)
|
||||
boolean postTurnPending; // completion observed; adapter housekeeping has not started yet
|
||||
boolean awaitingPostTurnPickup;
|
||||
@@ -176,15 +227,47 @@ public final class Injector {
|
||||
* <p>Uses an atomic map update so a concurrent {@link #drop} cannot slip between "find the
|
||||
* target" and "queue the message" and orphan it in a target it just removed.
|
||||
*/
|
||||
public CompletableFuture<Void> enqueue(String target, String text, TurnToken token) {
|
||||
public Delivery enqueue(String target, String text, TurnToken token) {
|
||||
CompletableFuture<Void> delivered = new CompletableFuture<>();
|
||||
Pending p = new Pending(text, token, delivered);
|
||||
Pending p = new Pending(target, text, token, delivered);
|
||||
targets.compute(target, (_, existing) -> {
|
||||
Target t = (existing != null) ? existing : new Target();
|
||||
t.add(p); // synchronized on the Target monitor — atomic with a concurrent drop
|
||||
return t;
|
||||
});
|
||||
return delivered;
|
||||
return new Delivery(p);
|
||||
}
|
||||
|
||||
/**
|
||||
* Cancel this exact queued delivery. The target monitor serializes this operation with
|
||||
* {@link #onStatus}: if delivery wins that race, this returns {@link Cancellation#DELIVERED}
|
||||
* rather than claiming the message remained queued.
|
||||
*/
|
||||
public Cancellation cancel(Delivery delivery) {
|
||||
Pending p = delivery.pending;
|
||||
Target t = targets.get(p.target);
|
||||
if (t == null) {
|
||||
return cancellationOf(p);
|
||||
}
|
||||
synchronized (t) {
|
||||
if (p.state != Pending.State.QUEUED || !t.queue.remove(p)) {
|
||||
return cancellationOf(p);
|
||||
}
|
||||
p.state = Pending.State.CANCELLED;
|
||||
if (isQuiescent(t)) {
|
||||
targets.remove(p.target, t);
|
||||
}
|
||||
return Cancellation.CANCELLED;
|
||||
}
|
||||
}
|
||||
|
||||
private static Cancellation cancellationOf(Pending p) {
|
||||
return p.state == Pending.State.DELIVERED ? Cancellation.DELIVERED : Cancellation.NOT_DELIVERED;
|
||||
}
|
||||
|
||||
private static boolean isQuiescent(Target t) {
|
||||
return t.queue.isEmpty() && !t.awaitingPickup && !t.awaitingCompletion
|
||||
&& !t.postTurnPending && !t.awaitingPostTurnPickup && !t.postTurnObserved;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -217,10 +300,12 @@ public final class Injector {
|
||||
t.awaitingPickup = false;
|
||||
t.injectableSincePickup = 0;
|
||||
t.unknownSinceTurn = 0;
|
||||
t.unknownSincePostTurn = 0;
|
||||
t.notReadySincePoll = 0;
|
||||
if (t.awaitingCompletion) t.turnObserved = true;
|
||||
} else if (status.injectable()) { // IDLE or BLOCKED
|
||||
t.unknownSinceTurn = 0;
|
||||
t.unknownSincePostTurn = 0;
|
||||
if (t.awaitingPostTurnPickup) {
|
||||
if (++t.injectableSincePostTurnPickup >= PICKUP_GRACE_POLLS) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
@@ -271,6 +356,7 @@ public final class Injector {
|
||||
try {
|
||||
agentsFor(target).send(target, p.text());
|
||||
t.queue.poll();
|
||||
p.state = Pending.State.DELIVERED;
|
||||
t.awaitingPickup = true;
|
||||
t.awaitingCompletion = true;
|
||||
t.turnObserved = false;
|
||||
@@ -280,6 +366,7 @@ public final class Injector {
|
||||
// Delivery failed at herdr; drop the poisoned message and surface it
|
||||
// rather than blocking the queue behind it.
|
||||
t.queue.poll();
|
||||
p.state = Pending.State.NOT_DELIVERED;
|
||||
sent = p;
|
||||
sendError = e;
|
||||
}
|
||||
@@ -290,6 +377,9 @@ public final class Injector {
|
||||
// fail every queued message and release the target (CB-114) instead of
|
||||
// polling it indefinitely with the caller's future never completing.
|
||||
notReady = new ArrayList<>(t.queue);
|
||||
for (Pending pending : notReady) {
|
||||
pending.state = Pending.State.NOT_DELIVERED;
|
||||
}
|
||||
log.warn("readiness grace for {} expired after {} polls ({}s): target never "
|
||||
+ "became deliverable, so failing {} queued message(s) that never "
|
||||
+ "reached its pane",
|
||||
@@ -315,12 +405,29 @@ public final class Injector {
|
||||
t.unknownSinceTurn = 0;
|
||||
turnFailed = true;
|
||||
}
|
||||
// fleetd #306: the same escape for the post-turn housekeeping phase. Four latches
|
||||
// gate delivery (awaitingCompletion, postTurnPending, awaitingPostTurnPickup,
|
||||
// postTurnObserved) and only the first had a way out of a sustained unknown streak —
|
||||
// a gate that closed one direction only. The other two below are released here as
|
||||
// well; postTurnPending needs no escape because it is cleared unconditionally on the
|
||||
// line after the listener call that sets it.
|
||||
//
|
||||
// This does NOT set turnFailed. The delegated turn already completed and its waiter
|
||||
// already resolved — what is outstanding is adapter housekeeping (the `/clear`).
|
||||
// Reporting a turn failure here would drive SessionManager.onFailed on a session
|
||||
// that genuinely finished its work, which is a worse lie than the wedge.
|
||||
if ((t.awaitingPostTurnPickup || t.postTurnObserved)
|
||||
&& ++t.unknownSincePostTurn >= TURN_STALL_GRACE_POLLS) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
t.postTurnObserved = false;
|
||||
t.injectableSincePostTurnPickup = 0;
|
||||
t.unknownSincePostTurn = 0;
|
||||
}
|
||||
}
|
||||
|
||||
// Reclaim the entry once the worker is fully quiescent (nothing queued, no pickup or
|
||||
// completion awaited), so the map cannot grow without bound across short-lived workers.
|
||||
if (t.queue.isEmpty() && !t.awaitingPickup && !t.awaitingCompletion
|
||||
&& !t.postTurnPending && !t.awaitingPostTurnPickup && !t.postTurnObserved) {
|
||||
if (isQuiescent(t)) {
|
||||
targets.remove(target, t);
|
||||
}
|
||||
}
|
||||
@@ -411,6 +518,9 @@ public final class Injector {
|
||||
boolean hadDeliveredTurn;
|
||||
synchronized (t) {
|
||||
pending = new ArrayList<>(t.queue);
|
||||
for (Pending p : pending) {
|
||||
p.state = Pending.State.NOT_DELIVERED;
|
||||
}
|
||||
t.queue.clear();
|
||||
hadDeliveredTurn = t.awaitingCompletion;
|
||||
t.awaitingCompletion = false;
|
||||
|
||||
@@ -4,6 +4,7 @@ import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.PendingCloseMarker;
|
||||
import dev.ltms.fleet.herdr.Tab;
|
||||
import dev.ltms.fleet.herdr.Workspace;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
@@ -12,9 +13,11 @@ import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
|
||||
/**
|
||||
@@ -44,8 +47,33 @@ import dev.ltms.fleet.peer.PeerLauncher;
|
||||
* privilege escalation (the tab label never granted anything a pane could take for itself; see that
|
||||
* class's javadoc), but <em>staleness</em>: a label left behind by a crashed session would otherwise
|
||||
* read as a live lead forever, and the lead would never be relaunched. So a lead counts as live only
|
||||
* when herdr also reports a <em>running agent</em> in that tab — see {@link #liveLeads}. A labelled
|
||||
* when herdr also reports a <em>running agent</em> in that tab — see {@link #countLeads}. A labelled
|
||||
* tab with no agent in it is not a lead.
|
||||
*
|
||||
* <p><strong>fleetd #359 — a stale label used to pile up, not just mislead.</strong> Finding "not
|
||||
* live" here used to mean only one thing: launch another. The old label was left exactly where it
|
||||
* was, so a daemon that restarted enough times — or hit one herdr read that missed a genuinely
|
||||
* running agent — accumulated one more identically-labelled dead tab per occurrence, and
|
||||
* {@code LeadTabScanner} (before its own #359 fix) reported every one of them as a lead.
|
||||
*
|
||||
* <p><strong>Review finding 1 — closing on one reading is worse than the bug.</strong> The first
|
||||
* version of this fix closed a name's dead tabs the moment a single {@link #countLeads} reading
|
||||
* called them dead. The ticket's own live evidence rules that out: on a real host, {@code
|
||||
* agent.list} was seen reporting "0 live" for a tab that a plain {@code ps} confirmed was running a
|
||||
* real session. Closing on that reading would have destroyed the operator's actual lead — a worse
|
||||
* failure than the extra tab it replaces. So {@link #ensureLeads()} now needs the same dead reading
|
||||
* <em>twice</em>, one restart apart, before it closes anything: the first time a labelled tab reads
|
||||
* dead, it is only flagged ({@link PendingCloseMarker}), left running, untouched; it is closed only
|
||||
* if a <em>later</em>, independently-connected reconcile still finds it dead while the flag is
|
||||
* still there. A transient miss self-heals — the next reconcile sees the agent again and clears the
|
||||
* flag (see {@code toUnflag} below) — so the worst case for a single bad reading is one extra tab
|
||||
* surviving one more restart, never a live session destroyed. Two alternatives were considered and
|
||||
* rejected: corroborating {@code agent.list} against a second, truly independent signal was dropped
|
||||
* because nothing else herdr exposes proves "is a process attached to this pane" any better — a
|
||||
* second call to the same unreliable source is not independent evidence; capping the close to "all
|
||||
* but the most recent dead tab" was dropped because "most recent" has no reliable ordering across
|
||||
* tab ids and would leave the true failure mode (a name that is <em>never</em> reconfirmed) growing
|
||||
* by one tab per bad reading forever, which is the exact defect this ticket exists to fix.
|
||||
*/
|
||||
public final class LeadLauncher {
|
||||
|
||||
@@ -79,9 +107,9 @@ public final class LeadLauncher {
|
||||
return 0;
|
||||
}
|
||||
|
||||
Map<String, Integer> live;
|
||||
Map<String, LeadCount> live;
|
||||
try {
|
||||
live = liveLeads(leaders);
|
||||
live = countLeads(leaders);
|
||||
} catch (HerdrException e) {
|
||||
// Counting is the whole safety mechanism against double-spawning. If we cannot count, we
|
||||
// must not guess — spawning a second orchestrator is worse than starting none.
|
||||
@@ -93,9 +121,53 @@ public final class LeadLauncher {
|
||||
for (Map.Entry<String, FleetConfig.Leader> e : leaders.entrySet()) {
|
||||
String name = e.getKey();
|
||||
FleetConfig.Leader lead = e.getValue();
|
||||
int running = live.getOrDefault(name, 0);
|
||||
LeadCount state = live.getOrDefault(name, LeadCount.NONE);
|
||||
int running = state.running();
|
||||
int wanted = lead.instances();
|
||||
|
||||
// fleetd #359 review finding 1: a tab already flagged pending-close, still labelled for
|
||||
// this lead, and STILL hosting no agent on this separate reconcile — two independent
|
||||
// readings agree, so close it. A tab found dead for the first time is only flagged below,
|
||||
// never closed on the spot.
|
||||
for (String tabId : state.toClose()) {
|
||||
log.info("lead '{}': closing tab {} — flagged pending-close on a previous reconcile "
|
||||
+ "and still no agent running in it", name, tabId);
|
||||
try {
|
||||
spaces.closeTab(tabId);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("could not close stale tab {} for lead '{}': {}",
|
||||
tabId, name, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
// A tab labelled for this lead, with no agent running in it, seen dead for the first
|
||||
// time — flag it rather than closing it. One reading of `agent.list` is not enough
|
||||
// evidence to destroy a tab that might genuinely be live (see the class javadoc).
|
||||
for (String tabId : state.toFlag()) {
|
||||
String flagged = lead.tabLabel() + PendingCloseMarker.SUFFIX;
|
||||
log.info("lead '{}': tab {} has no agent running in it this reconcile — flagging it "
|
||||
+ "'{}' rather than closing; it is only closed if a later reconcile still "
|
||||
+ "finds it dead", name, tabId, flagged);
|
||||
try {
|
||||
spaces.renameTab(tabId, flagged);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("could not flag stale tab {} for lead '{}': {}",
|
||||
tabId, name, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
// A previously-flagged tab that is running an agent again — the miss that flagged it was
|
||||
// transient. Clear the flag so a future, unrelated miss starts its own two-reading count
|
||||
// rather than closing on the strength of this one's already-spent flag.
|
||||
for (String tabId : state.toUnflag()) {
|
||||
log.info("lead '{}': tab {} is running an agent again — clearing its pending-close flag",
|
||||
name, tabId);
|
||||
try {
|
||||
spaces.renameTab(tabId, lead.tabLabel());
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("could not clear the pending-close flag on tab {} for lead '{}': {}",
|
||||
tabId, name, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
if (running >= wanted) {
|
||||
log.info("lead '{}': {} live, {} wanted — nothing to start", name, running, wanted);
|
||||
continue;
|
||||
@@ -126,18 +198,29 @@ public final class LeadLauncher {
|
||||
}
|
||||
|
||||
/**
|
||||
* How many live leads exist per configured name: a running agent in a tab labelled with that
|
||||
* lead's exact {@code tab} (CB-579). Member workspaces are excluded, exactly as the scanner
|
||||
* excludes them: a member must not be counted as a lead because it happens to sit in a matching
|
||||
* tab.
|
||||
* How many live leads exist per configured name, and which of that name's labelled tabs are
|
||||
* <em>not</em> live: a running agent in a tab labelled with that lead's exact {@code tab}
|
||||
* (CB-579). Member workspaces are excluded, exactly as the scanner excludes them: a member must
|
||||
* not be counted as a lead because it happens to sit in a matching tab.
|
||||
*
|
||||
* <p>There used to be a second path here — a running agent on the terminal a
|
||||
* {@code fleet.leaders.<name>.terminal} pin named, for a lead opened and pinned by hand. That
|
||||
* pin is retired: {@code tab} is now the only field identity depends on, and {@link Agent}
|
||||
* already carries {@link Agent#tabId()} directly, so a hand-opened lead is found the same way an
|
||||
* auto-launched one is — by labelling its tab to match.
|
||||
*
|
||||
* <p>fleetd #359 review finding 1: a labelled tab with nothing running in it is split into
|
||||
* {@code toClose} (already flagged pending-close by a previous reconcile, and still dead — two
|
||||
* independent readings agree) and {@code toFlag} (dead for the first time — not enough evidence
|
||||
* to close yet). {@code toUnflag} is the reverse: a tab flagged pending-close that is running an
|
||||
* agent again, so the flag it carries no longer means anything and {@link #ensureLeads()} clears
|
||||
* it.
|
||||
*/
|
||||
private Map<String, Integer> liveLeads(Map<String, FleetConfig.Leader> leaders) {
|
||||
private record LeadCount(int running, List<String> toClose, List<String> toFlag, List<String> toUnflag) {
|
||||
static final LeadCount NONE = new LeadCount(0, List.of(), List.of(), List.of());
|
||||
}
|
||||
|
||||
private Map<String, LeadCount> countLeads(Map<String, FleetConfig.Leader> leaders) {
|
||||
// A lead and the members share ONE workspace now (the operator asked for a single "session"
|
||||
// with many tabs), so a workspace can no longer be excluded wholesale — the lead lives in the
|
||||
// member workspace by design. The sole discriminator is the exact tab label: a lead carries
|
||||
@@ -145,6 +228,7 @@ public final class LeadLauncher {
|
||||
// profile's `worker: {profile} #{n}` template. These never collide, so an exact-label match
|
||||
// separates them without needing to know which workspace anyone is in.
|
||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
||||
Set<String> flaggedTabIds = new LinkedHashSet<>();
|
||||
for (Workspace ws : spaces.listWorkspaces()) {
|
||||
if (ws.workspaceId() == null) {
|
||||
continue;
|
||||
@@ -153,31 +237,65 @@ public final class LeadLauncher {
|
||||
String declared = leadNameOf(tab.label(), leaders);
|
||||
if (declared != null && tab.tabId() != null) {
|
||||
nameByTab.put(tab.tabId(), declared);
|
||||
if (PendingCloseMarker.isFlagged(tab.label())) {
|
||||
flaggedTabIds.add(tab.tabId());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Set<String> liveTabIds = new LinkedHashSet<>();
|
||||
Map<String, Integer> counts = new LinkedHashMap<>();
|
||||
for (Agent a : agents.list()) {
|
||||
String name = nameByTab.get(a.tabId());
|
||||
if (name != null) {
|
||||
counts.merge(name, 1, Integer::sum);
|
||||
liveTabIds.add(a.tabId());
|
||||
}
|
||||
}
|
||||
return counts;
|
||||
|
||||
Map<String, List<String>> toCloseByName = new LinkedHashMap<>();
|
||||
Map<String, List<String>> toFlagByName = new LinkedHashMap<>();
|
||||
Map<String, List<String>> toUnflagByName = new LinkedHashMap<>();
|
||||
nameByTab.forEach((tabId, name) -> {
|
||||
boolean live = liveTabIds.contains(tabId);
|
||||
boolean flagged = flaggedTabIds.contains(tabId);
|
||||
if (live) {
|
||||
if (flagged) {
|
||||
toUnflagByName.computeIfAbsent(name, k -> new ArrayList<>()).add(tabId);
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (flagged) {
|
||||
toCloseByName.computeIfAbsent(name, k -> new ArrayList<>()).add(tabId);
|
||||
} else {
|
||||
toFlagByName.computeIfAbsent(name, k -> new ArrayList<>()).add(tabId);
|
||||
}
|
||||
});
|
||||
|
||||
Map<String, LeadCount> out = new LinkedHashMap<>();
|
||||
for (String name : leaders.keySet()) {
|
||||
out.put(name, new LeadCount(counts.getOrDefault(name, 0),
|
||||
toCloseByName.getOrDefault(name, List.of()),
|
||||
toFlagByName.getOrDefault(name, List.of()),
|
||||
toUnflagByName.getOrDefault(name, List.of())));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* The configured lead a tab label names, or {@code null} for a label that names none.
|
||||
*
|
||||
* <p>Matched exactly (case-insensitively) against each lead's configured {@code tab}, so an
|
||||
* operator's {@code "lead: something-else"} tab is not mistaken for a configured lead.
|
||||
* operator's {@code "lead: something-else"} tab is not mistaken for a configured lead. A
|
||||
* trailing {@link PendingCloseMarker} is stripped first, so a tab this class flagged on a
|
||||
* previous reconcile is still recognised as the same lead's tab on this one.
|
||||
*/
|
||||
private String leadNameOf(String label, Map<String, FleetConfig.Leader> leaders) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
String l = label.strip();
|
||||
String l = PendingCloseMarker.strip(label);
|
||||
for (Map.Entry<String, FleetConfig.Leader> e : leaders.entrySet()) {
|
||||
String tab = e.getValue().tabLabel();
|
||||
if (tab != null && l.equalsIgnoreCase(tab.strip())) {
|
||||
|
||||
@@ -36,6 +36,25 @@ public final class ConnectionIdentity {
|
||||
* primary / an off-host client) and its {@code pid} (or {@code -1} if not resolvable).
|
||||
*/
|
||||
public record Caller(String terminal, long pid) {
|
||||
|
||||
/**
|
||||
* Whether the OS peer-PID lookup actually succeeded — {@code false} means {@code pid} is
|
||||
* the {@code -1} sentinel, not a real process id, so this caller's identity could not be
|
||||
* established at all. That is a different fact from a real pid that simply owns no worker
|
||||
* pane (the primary's own connection): the primary is {@code resolved()} and has a
|
||||
* {@code null terminal}; an unresolvable caller is {@code !resolved()} and also has a
|
||||
* {@code null terminal}. The two look identical through {@link #terminal} alone, which is
|
||||
* exactly how fleetd #317 happened — a failed {@code lsof} lookup and a genuine primary both
|
||||
* fell through to {@code Principal.primary(...)}.
|
||||
*
|
||||
* <p>Centralised here, next to the sentinel it tests, for the same reason
|
||||
* {@link ConnectionIdentity#isLoopback} is centralised rather than left for each caller to
|
||||
* reimplement: a raw {@code pid > 0} check duplicated at every call site is precisely the
|
||||
* "one rule, two copies" shape that let #305 drift.
|
||||
*/
|
||||
public boolean resolved() {
|
||||
return pid > 0;
|
||||
}
|
||||
}
|
||||
|
||||
/** Resolve the caller's terminal and PID from one peer-PID lookup. */
|
||||
@@ -60,7 +79,44 @@ public final class ConnectionIdentity {
|
||||
return pid > 0 ? cwds.cwdForPid(pid) : null;
|
||||
}
|
||||
|
||||
private static boolean isLoopback(String addr) {
|
||||
return "127.0.0.1".equals(addr) || "::1".equals(addr) || "0:0:0:0:0:0:0:1".equals(addr);
|
||||
/**
|
||||
* Whether {@code addr} is a same-host address, and therefore one whose peer PID is worth
|
||||
* looking up. <strong>This is the one definition of loopback in the daemon</strong> —
|
||||
* {@code CallerResolver} calls it rather than keeping its own, because the two used to differ
|
||||
* and that difference was a privilege escalation (fleetd #305).
|
||||
*
|
||||
* <p>The whole of {@code 127.0.0.0/8} counts, not just {@code 127.0.0.1}. On Linux every
|
||||
* address in that range is bound to {@code lo} by default, so a process can connect to
|
||||
* {@code 127.0.0.1:8765} with a source address of {@code 127.0.0.2} — measured on the Linux
|
||||
* fleet host, where binding that source succeeds.
|
||||
*
|
||||
* <p><strong>What excluding an address costs, stated as it is today.</strong> This paragraph
|
||||
* used to say that narrowing this range turned a worker into the lead, and that widening the
|
||||
* check was what closed the hole. That was true only while there were <em>two</em> definitions
|
||||
* that disagreed: {@code ConnectionIdentity} skipped the identity lookup for {@code 127.0.0.2}
|
||||
* while {@code CallerResolver} read the same address as loopback and granted the primary role.
|
||||
* #305 removed the second copy, and with one shared definition the old sentence no longer holds.
|
||||
*
|
||||
* <p>Measured on 2026-09-04 by narrowing this method back to exactly {@code 127.0.0.1} and
|
||||
* running {@code CallerResolverTest} and {@code ConnectionIdentityTest}: a caller from
|
||||
* {@code 127.0.0.2} then resolves to {@code ANONYMOUS}, not {@code PRIMARY} — for a worker
|
||||
* ({@code aWorkerOnAnyLoopbackSourceAddressIsStillAWorkerNotThePrimary}) and for a non-worker
|
||||
* ({@code aNonWorkerOnAnyLoopbackSourceAddressIsStillThePrimary}) alike. Excluding an address
|
||||
* now <em>refuses</em> its caller; it does not promote one.
|
||||
*
|
||||
* <p>So keep the whole range, but for the plain reason: a genuine worker or primary that
|
||||
* connects from {@code 127.0.0.2} must be identifiable at all, and narrowing this predicate
|
||||
* locks it out. That is an outage, and an outage is the direction to fail in — which is exactly
|
||||
* why the range must not be narrowed casually and also why doing so is no longer a security
|
||||
* hole. This predicate still does not decide whether a caller is trusted; it decides whether the
|
||||
* caller's identity is <em>resolved at all</em>. What makes an unresolved caller safe is
|
||||
* {@link Caller#resolved()} (#317), not this method.
|
||||
*/
|
||||
public static boolean isLoopback(String addr) {
|
||||
if (addr == null) {
|
||||
return false;
|
||||
}
|
||||
String a = addr.startsWith("::ffff:") ? addr.substring(7) : addr; // IPv4-mapped IPv6
|
||||
return a.startsWith("127.") || "::1".equals(a) || "0:0:0:0:0:0:0:1".equals(a);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -22,6 +22,7 @@ import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.ShuttingDownException;
|
||||
import dev.ltms.fleet.session.WorktreeRequest;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
@@ -43,6 +44,7 @@ import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiFunction;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
@@ -100,6 +102,8 @@ public final class FleetMcp {
|
||||
private final LeadSeatSource leadSeats;
|
||||
/** CB-637: this daemon's lead-to-lead channel; {@code null} when no coordinator is configured. */
|
||||
private final LeadChannel leadChannel;
|
||||
/** fleetd #361: {@code coordinator.peers} — see {@link CoordinationSource}. Empty when unset. */
|
||||
private final List<String> peers;
|
||||
|
||||
/** Capacity facts used by {@code fleet_list}; production must supply the placement live count. */
|
||||
public record CapacitySource(Function<String, Integer> liveCount, Function<String, Integer> maxLoad,
|
||||
@@ -139,22 +143,56 @@ public final class FleetMcp {
|
||||
|
||||
/**
|
||||
* fleetd #176: the seats a profile's own live LEAD session(s) hold on the same Claude
|
||||
* subscription — the third reason (alongside {@link QuarantineSource} and {@link OutageSource})
|
||||
* {@code free} can overstate what a fresh {@code fleet_spawn} would actually get.
|
||||
* subscription — a fact {@code fleet_list} reports alongside {@code free} via the
|
||||
* {@code leadSeats} key.
|
||||
*
|
||||
* <p>{@code maxLoad} counts only <em>members</em>, never the lead itself. But a
|
||||
* {@code subscription: true} profile bills the operator's own Claude account, and the lead is
|
||||
* always a live {@code claude} session on that same account (it is never moved off-subscription
|
||||
* — see {@code LeadLauncher}). So a fan-out that fills every member slot still leaves the lead's
|
||||
* own seat unaccounted for, and the daemon reports a slot that was never really free. See
|
||||
* {@code Fleetd.leadSeatLookup} for how the count is derived — from {@code fleet.leaders.<name>
|
||||
* .profile} and each profile's {@code effectiveCredentialId()}, never a hardcoded constant.
|
||||
* — see {@code LeadLauncher}). See {@code Fleetd.leadSeatLookup} for how the count is derived —
|
||||
* from {@code fleet.leaders.<name>.profile} and each profile's {@code effectiveCredentialId()},
|
||||
* never a hardcoded constant.
|
||||
*
|
||||
* <p>fleetd #257: this count is reported, never subtracted from {@code free}. An earlier cut of
|
||||
* this feature subtracted it, on the theory that it made {@code free} describe the real ceiling
|
||||
* on the account — but the real spawn gate ({@code CompositePeerLauncher#enforceMaxLoad}) never
|
||||
* read this count at all, so the subtraction made {@code free} disagree with the one thing it is
|
||||
* supposed to describe: what a fresh {@code fleet_spawn} will actually get. No backend seat
|
||||
* ceiling shared with the lead has been measured either — see {@code fleetd.example.yaml}'s
|
||||
* {@code maxLoad} docs. {@code free} now always equals {@code max(0, maxLoad - live)}, and
|
||||
* {@code leadSeats} is reported purely as a fact the caller may act on however it likes.
|
||||
*/
|
||||
public record LeadSeatSource(Function<String, Integer> seatsFor) {
|
||||
/** Inert source — no profile is ever reported as sharing a seat with a lead. */
|
||||
public static LeadSeatSource none() { return new LeadSeatSource(_ -> 0); }
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361: peer-visibility facts for {@code fleet_list}'s {@code coordinator} row — this
|
||||
* daemon's own {@link LeadChannel} (for its self mailbox state and held messages) plus the
|
||||
* coord-ids the operator has declared as peers ({@code coordinator.peers}). Bundled as its own
|
||||
* Source, the same idiom as {@link OutageSource}/{@link QuarantineSource}/{@link LeadSeatSource},
|
||||
* so {@code listFleet}'s already-long overload chain gains exactly one new required parameter
|
||||
* instead of a further bare positional argument.
|
||||
*
|
||||
* <p>Every mailbox look this triggers goes through {@link LeadChannel#inspect}, which is
|
||||
* specified to run on its own disposable channel — never the channel {@link LeadChannel#publish}
|
||||
* or the consume loop depends on — so a peer that happens to be down, or the coordination broker
|
||||
* itself being unreachable, can never take {@code fleet_send}/{@code LeadCoordLoop}'s own path
|
||||
* down with it. See {@code FleetMcp.probe} for the additional timeout bound on top of that.
|
||||
*
|
||||
* @param leadChannel this daemon's own channel, or {@code null} when no coordinator is configured
|
||||
* @param peers the coord-ids declared under {@code coordinator.peers}, or empty
|
||||
*/
|
||||
public record CoordinationSource(LeadChannel leadChannel, List<String> peers) {
|
||||
public CoordinationSource {
|
||||
peers = peers == null ? List.of() : List.copyOf(peers);
|
||||
}
|
||||
|
||||
/** Inert source — no coordinator row is ever reported. */
|
||||
public static CoordinationSource none() { return new CoordinationSource(null, List.of()); }
|
||||
}
|
||||
|
||||
/**
|
||||
* @param callers resolves each call's {@link Principal}; {@code null} disables authorization.
|
||||
* This surface needs its own enforcement: {@code /mcp} is a raw servlet on
|
||||
@@ -202,19 +240,34 @@ public final class FleetMcp {
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, with fleetd #176 lead-seat facts (see {@link LeadSeatSource}). This is what
|
||||
* {@code Fleetd.main} actually wires up.
|
||||
*
|
||||
* @param leadSeats required — pass {@link LeadSeatSource#none()} for a caller that does not want
|
||||
* the feature, never a defaulting overload (the same rule {@code quarantine} and
|
||||
* {@code outage} follow).
|
||||
* As above, with fleetd #176 lead-seat facts (see {@link LeadSeatSource}).
|
||||
*/
|
||||
public FleetMcp(MessageService messages, PeerLauncher workers, SessionManager sessions,
|
||||
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||
CallerResolver callers, Metrics metrics, CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, LeadChannel leadChannel, OutageSource outage,
|
||||
LeadSeatSource leadSeats) {
|
||||
this(messages, workers, sessions, identity, presence, primaryRegistry, callers, metrics, capacity,
|
||||
healthCoverage, quarantine, leadChannel, outage, leadSeats, List.of());
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, with fleetd #361 {@code coordinator.peers} (see {@link CoordinationSource}). This is
|
||||
* what {@code Fleetd.main} actually wires up.
|
||||
*
|
||||
* @param leadSeats required — pass {@link LeadSeatSource#none()} for a caller that does not want
|
||||
* the feature, never a defaulting overload (the same rule {@code quarantine} and
|
||||
* {@code outage} follow).
|
||||
* @param peers the coord-ids declared under {@code coordinator.peers}; empty when unset or
|
||||
* when {@code leadChannel} is {@code null}.
|
||||
*/
|
||||
public FleetMcp(MessageService messages, PeerLauncher workers, SessionManager sessions,
|
||||
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||
CallerResolver callers, Metrics metrics, CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, LeadChannel leadChannel, OutageSource outage,
|
||||
LeadSeatSource leadSeats, List<String> peers) {
|
||||
this.leadChannel = leadChannel;
|
||||
this.peers = peers == null ? List.of() : List.copyOf(peers);
|
||||
this.capacity = capacity;
|
||||
this.quarantine = Objects.requireNonNull(quarantine, "quarantine");
|
||||
this.outage = Objects.requireNonNull(outage, "outage");
|
||||
@@ -249,7 +302,7 @@ public final class FleetMcp {
|
||||
// Each handler is built once and wired to its fleet_* tool below.
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> sendHandler =
|
||||
(exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.SEND,
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_send", req.arguments()),
|
||||
str(req.arguments(), "sessionId"));
|
||||
if (denied != null) return denied;
|
||||
String caller = callerTerminal(exchange);
|
||||
@@ -291,7 +344,7 @@ public final class FleetMcp {
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> replyHandler =
|
||||
(exchange, req) -> {
|
||||
String self = callerTerminal(exchange);
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.REPLY, self);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_reply", req.arguments()), self);
|
||||
if (denied != null) return denied;
|
||||
return reply(messages, self, str(req.arguments(), "content"));
|
||||
};
|
||||
@@ -299,36 +352,38 @@ public final class FleetMcp {
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> askHandler =
|
||||
(exchange, req) -> {
|
||||
String self = callerTerminal(exchange);
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.ASK, self);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_ask", req.arguments()), self);
|
||||
if (denied != null) return denied;
|
||||
return ask(messages, self, str(req.arguments(), "question"), timeoutMs(req.arguments()));
|
||||
};
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> statusHandler =
|
||||
(exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_status", req.arguments()), null);
|
||||
if (denied != null) return denied;
|
||||
return status(messages, str(req.arguments(), "sessionId"));
|
||||
};
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> pollHandler =
|
||||
(exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
Map<String, Object> a = req.arguments();
|
||||
return poll(messages, str(a, "ticket"), str(a, "target"));
|
||||
String target = str(a, "target");
|
||||
// The action depends on the ARGUMENTS, not on the tool name -- see pollAction.
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_poll", a), target);
|
||||
if (denied != null) return denied;
|
||||
return poll(messages, str(a, "ticket"), target);
|
||||
};
|
||||
// CB-307 Increment 3: per-msgId ack (not needed in v1 but supported by the inbox).
|
||||
// Acking removes a reply from the inbox, so it is a drain, not a read.
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> ackHandler =
|
||||
(exchange, req) -> {
|
||||
Map<String, Object> a = req.arguments();
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.DRAIN, str(a, "target"));
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_ack", a), str(a, "target"));
|
||||
if (denied != null) return denied;
|
||||
return ack(messages, str(a, "target"), str(a, "msgId"));
|
||||
};
|
||||
// Fleet management (CB-108): spawn/list/stop over ClaudeCodeLauncher.
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> spawnHandler =
|
||||
(exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.SPAWN, null);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_spawn", req.arguments()), null);
|
||||
if (denied != null) return denied;
|
||||
String caller = callerTerminal(exchange);
|
||||
// SPAWN is already auth-gated to PRIMARY (architects can never call it), but
|
||||
@@ -345,29 +400,29 @@ public final class FleetMcp {
|
||||
};
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> listHandler =
|
||||
(exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_list", Map.of()), null);
|
||||
if (denied != null) return denied;
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
||||
leadSeats, callers == null ? Map.of() : callers.leads(),
|
||||
callerTerminal(exchange),
|
||||
leadChannel == null ? null : leadChannel.selfCoordId());
|
||||
new CoordinationSource(leadChannel, peers));
|
||||
};
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> stopHandler =
|
||||
(exchange, req) -> {
|
||||
String paneId = str(req.arguments(), "paneId");
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.STOP, paneId);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_stop", req.arguments()), paneId);
|
||||
if (denied != null) return denied;
|
||||
return stop(sessions, paneId);
|
||||
};
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> profilesHandler =
|
||||
(exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_profiles", Map.of()), null);
|
||||
if (denied != null) return denied;
|
||||
return profiles(workers, quarantine, outage);
|
||||
};
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> whoamiHandler =
|
||||
(exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_whoami", Map.of()), null);
|
||||
if (denied != null) return denied;
|
||||
return whoami(principal(exchange), sessions);
|
||||
};
|
||||
@@ -686,6 +741,15 @@ public final class FleetMcp {
|
||||
* turned into a tool error naming the coord-id. It is never allowed to escape as a crash: an
|
||||
* unreachable peer is an ordinary outcome of addressing a fleet you do not control.
|
||||
*
|
||||
* <p><strong>fleetd #361: the success text is honest about what "durably confirmed" does and
|
||||
* does not mean.</strong> The broker's publisher confirm proves the message is durably queued —
|
||||
* it says nothing about whether the peer's pane has, or ever will, receive it. After a
|
||||
* successful publish this looks at the target mailbox's consumer count (via
|
||||
* {@link LeadChannel#inspect}, bounded and never allowed to fail the call — see {@link #probe})
|
||||
* and appends a warning when it is zero: that is the observable form of "nobody is reading this
|
||||
* right now". A zero-consumer publish is still reported as a SUCCESS, never an error — the
|
||||
* message is safely queued and will be read once a daemon owning that coord-id connects.
|
||||
*
|
||||
* @param leadChannel this daemon's channel, or {@code null} when no coordinator is configured
|
||||
*/
|
||||
static McpSchema.CallToolResult sendToLead(LeadChannel leadChannel, String coordId, String content,
|
||||
@@ -713,7 +777,63 @@ public final class FleetMcp {
|
||||
+ ". Check that a daemon is running with coordinator.selfId=\"" + coordId
|
||||
+ "\" and is connected to the same coordination broker.");
|
||||
}
|
||||
return text("delivered to peer lead " + coordId + " (msgId " + msg.msgId() + ")");
|
||||
String result = "published to peer lead \"" + coordId + "\"'s mailbox and durably confirmed "
|
||||
+ "by the broker (msgId " + msg.msgId() + ").";
|
||||
LeadChannel.MailboxState state = probe(leadChannel, coordId);
|
||||
if (state.exists() && state.consumers() == 0) {
|
||||
result += " Warning: that mailbox currently has NO consumers attached — nobody is reading "
|
||||
+ "it right now. The message is safely queued and will be delivered once a daemon "
|
||||
+ "with coordinator.selfId=\"" + coordId + "\" is running and connected; until then "
|
||||
+ "it will not reach that lead's pane.";
|
||||
}
|
||||
return text(result);
|
||||
}
|
||||
|
||||
/**
|
||||
* Which authorization action a {@code fleet_poll} call needs, decided by its arguments
|
||||
* (fleetd #272).
|
||||
*
|
||||
* <p>{@code fleet_poll} is <strong>two operations behind one tool name</strong>. With {@code
|
||||
* ticket} it observes an async delegation and changes nothing, which is a {@link
|
||||
* Authz.Action#READ}. With {@code target} it calls {@link MessageService#drainReplies} on that
|
||||
* session -- the replies are removed from the inbox and a second call returns nothing -- so it
|
||||
* is a {@link Authz.Action#DRAIN}, the same gate {@code fleet_ack} already uses for removing a
|
||||
* single message, and the same one the REST path uses at {@code FleetApp.drainReplies}.
|
||||
*
|
||||
* <p>Until this method existed the handler passed a constant {@code READ} for both branches.
|
||||
* {@code READ} is open to every authenticated role, so any worker could read a peer's id out of
|
||||
* {@code fleet_list} and destroy the replies that peer had queued for the primary. The gate
|
||||
* failed open, and it did so because the required action is a function of the arguments while
|
||||
* the handler chose it before looking at them.
|
||||
*
|
||||
* <p>The choice lives in this method, and not inline in the handler, so that a test can assert
|
||||
* the mapping the handler actually uses. {@code FleetMcpAuthzTest} already checked every
|
||||
* {@link Authz.Action} against every {@link Role} and passed throughout -- it tested the policy
|
||||
* table, which was correct, while the defect was in which action the caller handed it.
|
||||
*
|
||||
* @param target the {@code target} argument of the call, or {@code null}/blank when absent
|
||||
*/
|
||||
static Authz.Action pollAction(String target) {
|
||||
return isBlank(target) ? Authz.Action.READ : Authz.Action.DRAIN;
|
||||
}
|
||||
|
||||
/**
|
||||
* The action a registered tool handler actually hands to the authorization gate.
|
||||
* Keeping this choice beside the registered-tool inventory makes a new tool fail the coverage
|
||||
* test until its action is pinned.
|
||||
*/
|
||||
static Authz.Action toolAction(String toolName, Map<String, Object> arguments) {
|
||||
return switch (toolName) {
|
||||
case "fleet_send" -> Authz.Action.SEND;
|
||||
case "fleet_reply" -> Authz.Action.REPLY;
|
||||
case "fleet_ask" -> Authz.Action.ASK;
|
||||
case "fleet_status", "fleet_list", "fleet_profiles", "fleet_whoami" -> Authz.Action.READ;
|
||||
case "fleet_poll" -> pollAction(str(arguments, "target"));
|
||||
case "fleet_ack" -> Authz.Action.DRAIN;
|
||||
case "fleet_spawn" -> Authz.Action.SPAWN;
|
||||
case "fleet_stop" -> Authz.Action.STOP;
|
||||
default -> throw new IllegalArgumentException("unregistered tool: " + toolName);
|
||||
};
|
||||
}
|
||||
|
||||
/** {@code fleet_poll}: check an async delegation by ticket, or drain a worker's inbox by target. */
|
||||
@@ -750,17 +870,27 @@ public final class FleetMcp {
|
||||
* or — when no send is open — queueing the reply in the inbox for later drain (CB-307).
|
||||
* {@code callerTerminal} is resolved from the connection (never an argument); a {@code null}
|
||||
* means the caller is not a known worker (e.g. the primary called it by mistake).
|
||||
*
|
||||
* <p>fleetd #365: the result text names which of those actually happened
|
||||
* ({@link MessageService.ReplyOutcome#description()}) instead of the single word "delivered"
|
||||
* for both — a queued reply is a real success, but it is not the same fact as one that resolved
|
||||
* a live waiter, and the caller could not previously tell them apart.
|
||||
*/
|
||||
static McpSchema.CallToolResult reply(MessageService messages, String callerTerminal, String content) {
|
||||
if (callerTerminal == null) {
|
||||
return error("fleet_reply is for workers only — could not identify the calling worker "
|
||||
+ "from the connection");
|
||||
}
|
||||
if (content == null) {
|
||||
// fleetd #302: isBlank, not == null, to match fleet_send's own guard above. MessageService
|
||||
// .reply now REJECTS blank content, and this handler is a bare BiFunction with no try/catch
|
||||
// around it — so a whitespace-only fleet_reply would leave here as an uncaught
|
||||
// IllegalArgumentException instead of this clean tool error. Null and whitespace are the
|
||||
// same mistake by the caller and must get the same answer.
|
||||
if (isBlank(content)) {
|
||||
return error("content is required");
|
||||
}
|
||||
messages.reply(callerTerminal, content);
|
||||
return text("delivered");
|
||||
MessageService.ReplyOutcome outcome = messages.reply(callerTerminal, content);
|
||||
return text(outcome.description());
|
||||
}
|
||||
|
||||
/** {@code fleet_ack}: acknowledge (remove) a specific reply from the inbox. */
|
||||
@@ -900,6 +1030,10 @@ public final class FleetMcp {
|
||||
return text(json(memberView(member)));
|
||||
} catch (GuardException e) {
|
||||
return error("subscription boundary: " + e.getMessage());
|
||||
} catch (ShuttingDownException e) {
|
||||
// fleetd #308: the daemon's shutdown drain has already started — refuse loudly rather
|
||||
// than register a session drainAll will never see again.
|
||||
return error("shutting down: " + e.getMessage());
|
||||
} catch (PlacementException e) {
|
||||
// CB-599: no candidate had capacity (maxLoad, quarantine, or all-exhausted) — distinct
|
||||
// from "profile does not exist" below.
|
||||
@@ -958,6 +1092,25 @@ public final class FleetMcp {
|
||||
* both maps at once when it is both exhaustion-quarantined AND cooling off.
|
||||
*/
|
||||
static McpSchema.CallToolResult profiles(PeerLauncher workers, QuarantineSource quarantine, OutageSource outage) {
|
||||
return text(json(profilesView(workers, quarantine, outage)));
|
||||
}
|
||||
|
||||
/**
|
||||
* The body both front doors answer {@code profiles} with: the configured profile names, the
|
||||
* default, and the two independent outage states — {@code quarantined} (the backend reported it
|
||||
* out of capacity) and {@code coolingOff} (the credential threw repeated non-exhaustion backend
|
||||
* errors). Each map is present only when at least one profile is in that state, and a profile
|
||||
* can appear in both at once, because the two checks are separate.
|
||||
*
|
||||
* <p>fleetd #297: extracted so {@code fleet_profiles} and {@code GET /profiles} render from ONE
|
||||
* body builder rather than two copies. Passing both doors the same {@link QuarantineSource} and
|
||||
* {@link OutageSource} instances is necessary but not sufficient: with the loop written out
|
||||
* twice, a later edit to the row shape — a renamed key, an added field — lands on one door and
|
||||
* not the other, and the two then disagree about a live outage. That is exactly what fleetd
|
||||
* #284 was, where one rule computed in two places was widened in only one and a single response
|
||||
* contradicted itself. Shared inputs do not make duplicated computation safe.
|
||||
*/
|
||||
public static Map<String, Object> profilesView(PeerLauncher workers, QuarantineSource quarantine, OutageSource outage) {
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
result.put("profiles", workers.profiles());
|
||||
result.put("default", workers.defaultProfile() == null ? "" : workers.defaultProfile());
|
||||
@@ -989,7 +1142,7 @@ public final class FleetMcp {
|
||||
if (!coolingOff.isEmpty()) {
|
||||
result.put("coolingOff", coolingOff);
|
||||
}
|
||||
return text(json(result));
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1024,7 +1177,8 @@ public final class FleetMcp {
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, Map<String, String> leads, String selfTerm) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, leads, selfTerm, null);
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, leads, selfTerm,
|
||||
CoordinationSource.none());
|
||||
}
|
||||
|
||||
/** As above, plus fleetd #201 Unit 5 cool-off facts (see {@link OutageSource}). */
|
||||
@@ -1033,33 +1187,33 @@ public final class FleetMcp {
|
||||
QuarantineSource quarantine, OutageSource outage,
|
||||
Map<String, String> leads, String selfTerm) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
||||
LeadSeatSource.none(), leads, selfTerm, null);
|
||||
LeadSeatSource.none(), leads, selfTerm, CoordinationSource.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, additionally reporting this daemon's own lead coordination id (CB-637) when one is
|
||||
* configured and its channel opened. There is no peer-discovery surface yet — a lead addresses a
|
||||
* peer by a coord-id it was told — so this row exists to answer the one question the operator
|
||||
* cannot answer any other way: what is MY coord-id, the one a peer must use to reach me. It is
|
||||
* omitted entirely when no coordinator is configured, so an ordinary fleet's output is unchanged.
|
||||
* As above, additionally reporting this daemon's own lead coordination state (CB-637, fleetd
|
||||
* #361) when a coordinator is configured and its channel opened — see {@link #coordinatorView}
|
||||
* for the shape. Omitted entirely when no coordinator is configured, so an ordinary fleet's
|
||||
* output is unchanged.
|
||||
*
|
||||
* @param selfCoordId this daemon's coord-id, or {@code null} when lead coordination is off
|
||||
* @param coordination this daemon's lead channel plus its declared peers, or
|
||||
* {@link CoordinationSource#none()} when lead coordination is off
|
||||
*/
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, Map<String, String> leads, String selfTerm,
|
||||
String selfCoordId) {
|
||||
CoordinationSource coordination) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, OutageSource.none(),
|
||||
LeadSeatSource.none(), leads, selfTerm, selfCoordId);
|
||||
LeadSeatSource.none(), leads, selfTerm, coordination);
|
||||
}
|
||||
|
||||
/** As above, plus fleetd #201 Unit 5 cool-off facts (see {@link OutageSource}). */
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, OutageSource outage,
|
||||
Map<String, String> leads, String selfTerm, String selfCoordId) {
|
||||
Map<String, String> leads, String selfTerm, CoordinationSource coordination) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
||||
LeadSeatSource.none(), leads, selfTerm, selfCoordId);
|
||||
LeadSeatSource.none(), leads, selfTerm, coordination);
|
||||
}
|
||||
|
||||
/** As above, plus fleetd #176 lead-seat facts (see {@link LeadSeatSource}). */
|
||||
@@ -1067,7 +1221,7 @@ public final class FleetMcp {
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, OutageSource outage,
|
||||
LeadSeatSource leadSeats, Map<String, String> leads, String selfTerm,
|
||||
String selfCoordId) {
|
||||
CoordinationSource coordination) {
|
||||
try {
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.map(Agent.class::cast)
|
||||
@@ -1089,8 +1243,9 @@ public final class FleetMcp {
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
result.put("leads", leadRows); result.put("members", out);
|
||||
result.put("healthCoverage", healthCoverage.value().get());
|
||||
if (selfCoordId != null && !selfCoordId.isBlank()) {
|
||||
result.put("coordinator", Map.of("selfId", selfCoordId, "configured", true));
|
||||
Map<String, Object> coordinatorRow = coordinatorView(coordination);
|
||||
if (coordinatorRow != null) {
|
||||
result.put("coordinator", coordinatorRow);
|
||||
}
|
||||
if (capacity.available()) result.put("capacity", profiles.stream()
|
||||
.map(profile -> capacityView(profile, capacity.liveCount(), capacity.maxLoad(), roster, messages,
|
||||
@@ -1101,6 +1256,140 @@ public final class FleetMcp {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361: the {@code coordinator} row — this daemon's own coord-id and mailbox state, the
|
||||
* messages currently held for it, and the live reachability of every operator-declared peer.
|
||||
* {@code null} (the row is then omitted entirely) when lead coordination is off, so an ordinary
|
||||
* fleet's {@code fleet_list} output is byte-identical to before this feature existed.
|
||||
*
|
||||
* <p>Every peer/self mailbox look goes through {@link #probe}, which bounds each
|
||||
* {@link LeadChannel#inspect} call to {@link #PEER_PROBE_TIMEOUT_MS} and never lets it throw —
|
||||
* a coordination broker that is down or slow degrades this row toward "unreachable"/"unknown"
|
||||
* counts, it can never make {@code fleet_list} itself slow or fail. {@code held} comes from
|
||||
* {@link LeadChannel#peek}, a pure in-memory read with no broker round trip, so it is never
|
||||
* subject to that bound.
|
||||
*/
|
||||
private static Map<String, Object> coordinatorView(CoordinationSource coordination) {
|
||||
LeadChannel channel = coordination.leadChannel();
|
||||
if (channel == null) {
|
||||
return null;
|
||||
}
|
||||
String selfId = channel.selfCoordId();
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("selfId", selfId);
|
||||
row.put("configured", true);
|
||||
row.put("mailbox", mailboxView(probe(channel, selfId)));
|
||||
row.put("held", channel.peek().stream().map(FleetMcp::heldView).toList());
|
||||
row.put("peers", coordination.peers().stream().map(p -> peerView(channel, p)).toList());
|
||||
return row;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361: render a {@link LeadChannel.MailboxState} without ever presenting an unmeasured
|
||||
* fact as a measured one. {@code status} is the tri-state itself — {@code "exists"},
|
||||
* {@code "absent"} (the broker positively confirmed no such queue), or {@code "unknown"} (the
|
||||
* probe could not determine either way: down, unreachable, or timed out). {@code pending}/
|
||||
* {@code consumers} are included ONLY when {@code status == "exists"} — a reader must never see
|
||||
* them default to {@code 0} for a mailbox this call never actually measured. This is the fix for
|
||||
* the review finding that a collapsed {@code absent()} rendered a self-probe timeout as
|
||||
* "pending: 0, consumers: 0", indistinguishable from an actually-empty, actually-unread mailbox.
|
||||
*/
|
||||
private static Map<String, Object> mailboxView(LeadChannel.MailboxState state) {
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("status", state.exists() ? "exists" : state.known() ? "absent" : "unknown");
|
||||
if (state.exists()) {
|
||||
row.put("pending", state.pending());
|
||||
row.put("consumers", state.consumers());
|
||||
}
|
||||
return row;
|
||||
}
|
||||
|
||||
/** One held-for-me message: enough to identify it and see roughly what it says, never the whole body. */
|
||||
private static Map<String, Object> heldView(LeadMessage m) {
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("msgId", m.msgId());
|
||||
row.put("from", m.from());
|
||||
row.put("preview", preview(m.content()));
|
||||
return row;
|
||||
}
|
||||
|
||||
/** Cap a held message's content to a short preview — {@code fleet_list} must never dump a full body. */
|
||||
private static final int HELD_PREVIEW_MAX_CHARS = 80;
|
||||
|
||||
private static String preview(String content) {
|
||||
if (content == null) {
|
||||
return "";
|
||||
}
|
||||
return content.length() <= HELD_PREVIEW_MAX_CHARS
|
||||
? content
|
||||
: content.substring(0, HELD_PREVIEW_MAX_CHARS) + "…";
|
||||
}
|
||||
|
||||
/**
|
||||
* One declared peer's row: its coord-id, then the same tri-state {@link #mailboxView} shape.
|
||||
* Deliberately no boolean "reachable" field — that collapsed "confirmed gone" and "could not
|
||||
* check" into the same {@code false}, which is exactly the review finding this row now avoids:
|
||||
* an operator reading {@code status} can tell "fleet01 is down" (a {@code coordinator.selfId}
|
||||
* nobody has ever run) apart from "my own broker is slow or unreachable right now".
|
||||
*/
|
||||
private static Map<String, Object> peerView(LeadChannel channel, String coordId) {
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("coordId", coordId);
|
||||
row.putAll(mailboxView(probe(channel, coordId)));
|
||||
return row;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361: how long {@code fleet_list} waits on any single {@link LeadChannel#inspect} call
|
||||
* before giving up on it — see {@link #probe}.
|
||||
*/
|
||||
private static final long PEER_PROBE_TIMEOUT_MS = 1_500L;
|
||||
|
||||
/**
|
||||
* Dedicated pool for {@link LeadChannel#inspect} calls so a slow one blocks only its own virtual
|
||||
* thread, never the MCP request thread calling {@code fleet_list}. Not a <em>bounded</em> pool —
|
||||
* {@code newThreadPerTaskExecutor} starts a fresh virtual thread per call with no cap on how many
|
||||
* run at once; virtual threads make that cheap, not bounded. What actually keeps a hung probe
|
||||
* from accumulating forever is the {@link Future#cancel} in {@link #probe}, not a pool limit.
|
||||
*/
|
||||
private static final java.util.concurrent.ExecutorService PEER_PROBE_POOL =
|
||||
java.util.concurrent.Executors.newThreadPerTaskExecutor(Thread.ofVirtual().name("fleet-peer-probe-", 0).factory());
|
||||
|
||||
/**
|
||||
* fleetd #361: {@link LeadChannel#inspect}, bounded to {@link #PEER_PROBE_TIMEOUT_MS} and never
|
||||
* allowed to throw or hang the caller — a coordination broker that is unreachable or slow
|
||||
* degrades to {@link LeadChannel.MailboxState#unknown} (never {@code absent}: a timeout proves
|
||||
* nothing about whether the mailbox exists) rather than making {@code fleet_list} slow or
|
||||
* failing it. {@code inspect} itself is already specified to never throw, but this is the seam
|
||||
* that also survives an implementation that does, or one that blocks indefinitely on a dead
|
||||
* connection.
|
||||
*
|
||||
* <p><strong>A timeout cancels the orphaned task</strong> rather than abandoning it. Before this,
|
||||
* {@code get(timeout)} on a hung {@code inspect} left the submitted task running forever on its
|
||||
* own virtual thread, holding the AMQP channel it had already opened — against a broker that
|
||||
* hangs rather than fails fast, every {@code fleet_list} call would orphan one more channel until
|
||||
* the connection's channel-max (2047 by default) was exhausted, which would break {@link
|
||||
* LeadChannel#publish} too. {@link Future#cancel(boolean) cancel(true)} interrupts the orphaned
|
||||
* task's thread; {@link LeadMailbox#inspect} has no interruptible wait of its own to catch that,
|
||||
* but the underlying AMQP RPC continuation does block on one, so the interrupt reaches it and the
|
||||
* task's {@code finally} still closes the probe channel it opened rather than leaking it forever.
|
||||
*/
|
||||
private static LeadChannel.MailboxState probe(LeadChannel channel, String coordId) {
|
||||
return probe(channel, coordId, PEER_PROBE_TIMEOUT_MS);
|
||||
}
|
||||
|
||||
/** As {@link #probe(LeadChannel, String)}, with an explicit timeout — a seam for tests. */
|
||||
static LeadChannel.MailboxState probe(LeadChannel channel, String coordId, long timeoutMs) {
|
||||
java.util.concurrent.Future<LeadChannel.MailboxState> future =
|
||||
PEER_PROBE_POOL.submit(() -> channel.inspect(coordId));
|
||||
try {
|
||||
return future.get(timeoutMs, TimeUnit.MILLISECONDS);
|
||||
} catch (Exception e) {
|
||||
future.cancel(true); // best-effort: don't leave a hung probe (and its channel) running forever
|
||||
return LeadChannel.MailboxState.unknown(coordId);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Capacity is advisory only. {@code reclaimable} says there is no bridge work, not that fleetd
|
||||
* may stop the member: the bridge has capacity facts but no work list, and choosing work needs
|
||||
@@ -1110,15 +1399,35 @@ public final class FleetMcp {
|
||||
private static Map<String, Object> memberCapacityView(MemberSession session, Agent live,
|
||||
MessageService messages, long nowNanos) {
|
||||
Map<String, Object> row = SessionManager.rosterView(session, live);
|
||||
boolean open = messages != null && messages.hasAcceptedDelivery(session.terminalId());
|
||||
boolean inbox = messages != null && messages.hasInboxMessage(session.terminalId());
|
||||
boolean reclaimable = (session.state() == MemberSession.State.READY || session.state() == MemberSession.State.DONE)
|
||||
&& !open && !inbox;
|
||||
boolean reclaimable = reclaimable(session, messages);
|
||||
row.put("reclaimable", reclaimable);
|
||||
row.put("idleForSeconds", reclaimable ? Math.max(0, (nowNanos - session.lastActivityAtNanos()) / 1_000_000_000L) : null);
|
||||
return row;
|
||||
}
|
||||
|
||||
/**
|
||||
* The one definition of {@code reclaimable}: this member holds a spawn seat, and has no open
|
||||
* bridge work, so stopping it gives the seat back. Both views in a single {@code fleet_list}
|
||||
* response call it — the per-member flag in {@link #memberCapacityView} and the per-profile
|
||||
* count in {@link #capacityView} — because two copies of this rule in one response is how the
|
||||
* two numbers come to disagree.
|
||||
*
|
||||
* <p>fleetd #284: {@code BACKEND_ERROR} and {@code FAILED} are deliberately NOT reclaimable.
|
||||
* The ticket asked for them to be, and that half of the ticket was wrong. Once
|
||||
* {@code Fleetd.liveSessionCount} stopped counting a terminal session as live, that seat is
|
||||
* ALREADY in {@code free}; counting it here too reports the same seat twice, and
|
||||
* {@code free + reclaimable} then reads as more capacity than {@code maxLoad} allows. The dead
|
||||
* session stays visible either way: its roster row still carries {@code state:
|
||||
* "backend_error"} or {@code "failed"}, which is what tells the lead to stop it.
|
||||
*/
|
||||
static boolean reclaimable(MemberSession session, MessageService messages) {
|
||||
boolean holdsSeat = session.state() == MemberSession.State.READY
|
||||
|| session.state() == MemberSession.State.DONE;
|
||||
return holdsSeat && (messages == null
|
||||
|| (!messages.hasAcceptedDelivery(session.terminalId())
|
||||
&& !messages.hasInboxMessage(session.terminalId())));
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-583: {@code free} alone cannot tell a lead "busy, will free up" from "refusing, and
|
||||
* nothing changes for N seconds" — those need different decisions. So a quarantined profile
|
||||
@@ -1137,10 +1446,17 @@ public final class FleetMcp {
|
||||
* <p>fleetd #176: {@code maxLoad} counts panes, not subscription seats — it never counted the
|
||||
* lead's own seat on a {@code subscription: true} profile's account. {@link LeadSeatSource}
|
||||
* reports that count (0 for a non-subscription profile, or when no live lead shares its
|
||||
* credential), and it is subtracted from {@code free} the same way {@code live} already is —
|
||||
* {@code maxLoad} itself is left untouched, so the row still reports the configured cap. The
|
||||
* {@code leadSeats} key is added only when the count is positive, for the same
|
||||
* byte-identical-when-unused reason as the quarantine/cool-off keys above.
|
||||
* credential) via the {@code leadSeats} key, added only when the count is positive, for the
|
||||
* same byte-identical-when-unused reason as the quarantine/cool-off keys above.
|
||||
*
|
||||
* <p>fleetd #257: {@code leadSeatCount} is reported, never subtracted from {@code free}.
|
||||
* {@code free} means "what a fresh {@code fleet_spawn} on this profile will actually get", and
|
||||
* the real gate ({@code CompositePeerLauncher#enforceMaxLoad}) only ever compares {@code live}
|
||||
* against {@code maxLoad} — it has no notion of a lead's own seat. Subtracting
|
||||
* {@code leadSeatCount} here made {@code free} disagree with the gate it is supposed to
|
||||
* describe: it could report {@code free: 0} while a spawn on that exact profile still
|
||||
* succeeded. {@code leadSeats} stays in the row as a fact the caller can act on however it
|
||||
* likes, but it no longer changes what {@code free} means.
|
||||
*/
|
||||
private static Map<String, Object> capacityView(String profile, Function<String, Integer> liveCount,
|
||||
Function<String, Integer> maxLoad, List<MemberSession> roster,
|
||||
@@ -1150,12 +1466,16 @@ public final class FleetMcp {
|
||||
int live = liveCount.apply(profile);
|
||||
int leadSeatCount = leadSeats.seatsFor().apply(profile);
|
||||
int reclaimable = (int) roster.stream().filter(s -> profile.equals(s.profile()))
|
||||
.filter(s -> (s.state() == MemberSession.State.READY || s.state() == MemberSession.State.DONE))
|
||||
.filter(s -> messages == null || (!messages.hasAcceptedDelivery(s.terminalId()) && !messages.hasInboxMessage(s.terminalId())))
|
||||
.filter(s -> reclaimable(s, messages))
|
||||
.count();
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("profile", profile); row.put("maxLoad", cap); row.put("live", live);
|
||||
row.put("free", cap == null ? null : Math.max(0, cap - live - leadSeatCount));
|
||||
// fleetd #257: free must report what the real spawn gate (CompositePeerLauncher#enforceMaxLoad)
|
||||
// will actually grant, and that gate never reads leadSeatCount — only maxLoad and live. Not
|
||||
// subtracting the lead's seat here used to make free UNDERSTATE what a fresh fleet_spawn would
|
||||
// get, so a lead believing free:0 gave up on a profile the gate would still spawn onto.
|
||||
// leadSeatCount is still reported via the leadSeats key below, just never subtracted from free.
|
||||
row.put("free", cap == null ? null : Math.max(0, cap - live));
|
||||
row.put("reclaimable", reclaimable);
|
||||
if (leadSeatCount > 0) {
|
||||
row.put("leadSeats", leadSeatCount);
|
||||
@@ -1260,8 +1580,12 @@ public final class FleetMcp {
|
||||
+ "routes your answer back into the same turn (omit for a normal delegation)"),
|
||||
"coordId", stringProp("A peer LEAD's coordination id — delivers content to that "
|
||||
+ "lead's durable mailbox on the shared coordination broker, which works "
|
||||
+ "across hosts. Mutually exclusive with sessionId and turnId. Your own "
|
||||
+ "coordId is reported by fleet_list.")),
|
||||
+ "across hosts. Mutually exclusive with sessionId and turnId. Success means "
|
||||
+ "the message is durably queued and confirmed by the broker, with a warning "
|
||||
+ "if that mailbox has no consumers attached right now (queued, but nobody is "
|
||||
+ "reading it yet) — it does not mean the peer's pane has seen it. Your own "
|
||||
+ "coordId, this daemon's mailbox state, and every coordinator.peers entry's "
|
||||
+ "live reachability are reported by fleet_list's coordinator row.")),
|
||||
List.of("content")));
|
||||
}
|
||||
|
||||
@@ -1315,7 +1639,11 @@ public final class FleetMcp {
|
||||
+ "worktree:<ticket-slug> to provision an isolated git worktree. Pass resumeSessionId "
|
||||
+ "to relaunch onto a prior conversation instead of starting cold — this requires an "
|
||||
+ "explicit profile whose backend supports it (fleet_list shows agentSessionId for "
|
||||
+ "resumable members), and is refused otherwise rather than silently starting fresh. "
|
||||
+ "resumable members; it is absent for a member fleetd cannot reliably re-identify, "
|
||||
+ "e.g. an opencode member spawned without a worktree), and is refused otherwise "
|
||||
+ "rather than silently starting fresh. For an opencode profile, resumeSessionId "
|
||||
+ "itself also requires worktree:true/<slug> on THIS spawn — without one fleetd can "
|
||||
+ "never re-verify which conversation it actually resumed (fleetd #249). "
|
||||
+ "sessionName gives the member a display name in its own UI when the backend supports "
|
||||
+ "one. Returns the member's sessionId (use with fleet_send) and paneId (use with "
|
||||
+ "fleet_stop).",
|
||||
@@ -1326,7 +1654,7 @@ public final class FleetMcp {
|
||||
"worktree", Map.of("type", "string", "description", "'true' or a ticket slug — requests an isolated git worktree"),
|
||||
"ticket", stringProp("Ticket slug when worktree:true"),
|
||||
"sessionName", stringProp("Logical display name for the member's own session, when its backend supports one"),
|
||||
"resumeSessionId", stringProp("A prior member's agentSessionId (from fleet_list) to resume — requires an explicit profile that supports it")),
|
||||
"resumeSessionId", stringProp("A prior member's agentSessionId (from fleet_list) to resume — requires an explicit profile that supports it, and (for opencode) a worktree on this spawn too")),
|
||||
List.of()));
|
||||
}
|
||||
|
||||
@@ -1347,15 +1675,31 @@ public final class FleetMcp {
|
||||
+ "discover a peer lead without being told its address. 'members' are the "
|
||||
+ "sessions delegated to — each with sessionId, paneId, role (architect/dev/"
|
||||
+ "reviewer), profile (the backend it runs on), state, optional "
|
||||
+ "worktree/branch/owner/agentSessionId (the id to pass as fleet_spawn's "
|
||||
+ "resumeSessionId to relaunch onto that same conversation, when the backend "
|
||||
+ "supports it), and live herdr status. An empty 'members' "
|
||||
+ "worktree/branch/owner/agentSessionId, and live herdr status. agentSessionId, "
|
||||
+ "when present, is the id to pass as fleet_spawn's resumeSessionId to relaunch "
|
||||
+ "onto that same conversation. It is ABSENT — not a guess — for a member fleetd "
|
||||
+ "cannot reliably re-identify: some backends (e.g. opencode) resolve it from the "
|
||||
+ "member's working directory, which only uniquely identifies a member when it "
|
||||
+ "was spawned into its own fleetd-provisioned worktree (worktree:true/<slug>); a "
|
||||
+ "member spawned without one shares its directory with others and never reports "
|
||||
+ "an id, however long it runs (fleetd #249). An empty 'members' "
|
||||
+ "means no members are spawned; it says nothing about peers. When capacity "
|
||||
+ "facts are configured, a 'capacity' row per profile also reports free: 0 for "
|
||||
+ "a quarantined profile's credential (see fleet_profiles), whatever its "
|
||||
+ "facts are configured, a 'capacity' row per profile reports 'free' — the "
|
||||
+ "slots a fresh fleet_spawn on that profile will actually be granted right "
|
||||
+ "now (max(0, maxLoad - live)), the same check the spawn gate itself runs. A "
|
||||
+ "'leadSeats' key, when present, reports how many of those live slots are a "
|
||||
+ "lead session sharing this profile's subscription — informational only, "
|
||||
+ "already NOT subtracted from 'free' (fleetd #257). It also reports free: 0 "
|
||||
+ "for a quarantined profile's credential (see fleet_profiles), whatever its "
|
||||
+ "maxLoad/live — with credentialId and quarantinedForSeconds naming the "
|
||||
+ "quarantine, so 'free: 0, busy' can be told apart from 'free: 0, refusing "
|
||||
+ "for N seconds'.",
|
||||
+ "for N seconds'. When lead-to-lead coordination is configured, a 'coordinator' "
|
||||
+ "object reports this daemon's own coord-id ('selfId') and mailbox state "
|
||||
+ "('mailbox': pending/consumers), the messages currently held for it ('held': "
|
||||
+ "msgId/from/preview, never the full body), and one row per coordinator.peers "
|
||||
+ "coord-id ('peers': coordId/reachable, plus pending/consumers when reachable) — "
|
||||
+ "this is peer DISCOVERY for cross-host leads, distinct from the local 'leads' "
|
||||
+ "array above. It is omitted entirely when no coordinator is configured.",
|
||||
objectSchema(Map.of(), List.of()));
|
||||
}
|
||||
|
||||
|
||||
@@ -41,6 +41,14 @@ public final class LsofPeerPidLookup implements PeerPidLookup {
|
||||
if (!p.waitFor(2, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
}
|
||||
if (found < 0) {
|
||||
// fleetd #317: this is the silent path — lsof ran clean and simply reported no
|
||||
// matching process (e.g. queried before the OS socket table settles). Previously
|
||||
// this logged nothing at all, which is exactly why the escalation went unnoticed;
|
||||
// the exception path below already logs. A caller now refused because of this is
|
||||
// still refused (never promoted) — this line only makes the refusal diagnosable.
|
||||
log.debug("lsof peer-pid lookup for port {} found no matching process", port);
|
||||
}
|
||||
return found;
|
||||
} catch (Exception e) {
|
||||
log.debug("lsof peer-pid lookup for port {} failed: {}", port, e.getMessage());
|
||||
|
||||
@@ -19,6 +19,7 @@ import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.nio.file.attribute.PosixFileAttributeView;
|
||||
import java.util.Arrays;
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
@@ -495,7 +496,7 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
* <p><b>Additive, not a rewrite.</b> {@code .claude.json} is large (tens of KB, dozens of
|
||||
* projects) and Claude Code itself rewrites it while running, so this reads the file as a JSON
|
||||
* tree (missing or unreadable → treated as an empty object) and changes only
|
||||
* {@code projects.<cwd>.hasTrustDialogAccepted} / {@code .hasCompletedProjectOnboarding} —
|
||||
* {@code projects.<cwd>.hasTrustDialogAccepted} (that key alone — see fleetd #247) —
|
||||
* every other top-level key and every other project entry is written back untouched. Only the
|
||||
* one project entry for {@code cwd} is replaced/created; an existing entry for a DIFFERENT cwd
|
||||
* (or the operator's own project history) is never touched.
|
||||
@@ -505,7 +506,7 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
* peer that starts without the seed still starts; it just may hit the dialog fleetd #149
|
||||
* describes.
|
||||
*
|
||||
* <p><b>Gated to a provisioned worktree</b> ({@link #isProvisionedWorktree}) — see that
|
||||
* <p><b>Gated to a provisioned worktree</b> ({@link HerdrPeerLauncher#isProvisionedWorktree}) — see that
|
||||
* method's javadoc for the incident that made this gate mandatory, not optional: this must
|
||||
* never run against a real checkout or an un-configured fallback cwd, only the exact
|
||||
* always-fresh-directory population fleetd #149 describes.
|
||||
@@ -516,57 +517,215 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
* sibling-temp-file + {@code ATOMIC_MOVE}, never a truncate-in-place) so a crash mid-write or a
|
||||
* concurrent reader never observes a half-written file, and through {@link #TRUST_JSON_LOCK} so
|
||||
* two concurrent spawns' entries both survive instead of the second write silently discarding
|
||||
* the first. Both exist because of a real incident: see {@link #isProvisionedWorktree}'s javadoc
|
||||
* the first. Both exist because of a real incident: see {@link HerdrPeerLauncher#isProvisionedWorktree}'s javadoc
|
||||
* and {@link #writeAtomically}'s javadoc.
|
||||
*
|
||||
* <p><b>Compare-and-swap against a writer the lock cannot reach (fleetd #247).</b>
|
||||
* {@code TRUST_JSON_LOCK} only serialises calls this launcher itself makes inside this one JVM.
|
||||
* It does nothing about a writer outside it — and on a host where the profile's
|
||||
* {@code configDir} is the operator's own {@code CLAUDE_CONFIG_DIR}, the file this method writes
|
||||
* IS the operator's own live Claude Code session's config file, being read and written by that
|
||||
* session while it runs. Measured 2026-09-03: its mtime moved minutes after a spawn while that
|
||||
* session was active. A plain read-modify-write there is a routine lost update, not a rare one:
|
||||
* fleetd reads v1, the operator's session reads v1 and writes v2 with their own change, fleetd's
|
||||
* {@code ATOMIC_MOVE} then lands v3 built from v1 — atomic, but v2's change is gone. So before
|
||||
* the move this method re-reads {@code target}'s exact bytes and compares them with the bytes it
|
||||
* built its update from; a mismatch means someone else wrote in between, and it discards its
|
||||
* work and rebuilds from the fresh bytes, up to {@link #MAX_TRUST_JSON_CAS_ATTEMPTS} times.
|
||||
* <b>Exhausting the retries writes nothing</b> — see the WARN at the end of the loop for why
|
||||
* that, not a last write-anyway, is the safe failure: the member shows the trust dialog and
|
||||
* fails to reach an injectable state, which is visible, logged and recoverable; overwriting the
|
||||
* operator's live config with a stale copy is neither. This narrows the lost-update window, it
|
||||
* does not close it — a write landing between the final re-read and the {@code ATOMIC_MOVE}
|
||||
* itself is still lost, because there is no OS-level compare-and-swap on a plain file, only this
|
||||
* cooperative narrowing of the gap.
|
||||
*
|
||||
* <p><b>fleetd #285: refuses under {@code memberHerdrSocket} rather than writing somewhere the
|
||||
* member cannot read.</b> Under {@code memberHerdrSocket:} the member pane runs as a
|
||||
* <em>different OS user with its own {@code $HOME}</em> — the same reason {@link
|
||||
* #writeCharterFile} routes the role/reply charter under {@code worktreeRoot} instead of
|
||||
* {@code java.io.tmpdir} and refuses the spawn when it cannot. This method has no equivalent
|
||||
* relocation available: unlike the charter (fleetd's own content, free to place anywhere and
|
||||
* hand to the peer via an argv flag), {@code .claude.json} is a file Claude Code looks up for
|
||||
* ITSELF at a fixed location — {@code CLAUDE_CONFIG_DIR/.claude.json}, or else the member OS
|
||||
* user's own {@code ~/.claude.json}, a path fleetd has no channel to learn. So when {@code
|
||||
* configDir} is unset, there is no member-readable target to seed at all — writing the
|
||||
* unqualified default would land in <em>fleetd's own</em> {@code ~/.claude.json} instead, the
|
||||
* exact defect this fix closes, not a workable fallback. And even with {@code configDir} set,
|
||||
* the file this method itself just wrote is {@code 0600} (owner-only — see {@link
|
||||
* #copyPosixPermissionsIfPresent}), unreadable by a different-uid member unless shared with
|
||||
* {@code worktreeGroup}, the same group {@link EnvAllowListScrub#shareWithGroup} already uses
|
||||
* for the ZDOTDIR scrub (fleetd #213) and the charter file (fleetd #219/#222). So under {@code
|
||||
* memberHerdrSocket} this method requires BOTH {@code configDir} and {@code worktreeGroup}
|
||||
* before it ever touches a file, and refuses the spawn — naming exactly which one is missing —
|
||||
* rather than silently corrupt fleetd's own home or hand the member an unreadable path. This
|
||||
* mirrors {@link #writeCharterFile}'s "refuse, don't degrade" decision: a member spawned without
|
||||
* a readable trust seed is not degraded, it sits on the interactive dialog forever and never
|
||||
* calls {@code fleet_reply} — exactly the failure fleetd #149 exists to prevent, so trading it
|
||||
* for "spawn something" is not worth it. When {@code configDir} and {@code worktreeGroup} are
|
||||
* both present, the write proceeds exactly as below and the resulting file is additionally
|
||||
* chgrp'd/chmod'd group-readable ({@code rw-r-----}) via {@link
|
||||
* EnvAllowListScrub#shareFileWithGroup(Path, String)} so the member's OS user can actually
|
||||
* open it — the
|
||||
* directory itself (unlike {@code worktreeRoot} or the charter's per-spawn directory) is not
|
||||
* fleetd-managed, so its own traversal permissions remain the operator's setup, same as they
|
||||
* already must be for the member to read anything else fleetd points {@code CLAUDE_CONFIG_DIR}
|
||||
* at. With {@code memberHerdrSocket} ABSENT (today's only live mode) every branch below is
|
||||
* byte-identical to before this fix.
|
||||
*
|
||||
* @param configDir the profile's {@code CLAUDE_CONFIG_DIR} ({@code cfg.configDir()}), or
|
||||
* {@code null}/blank to target the default {@code ~/.claude.json}
|
||||
* {@code null}/blank to target the default {@code ~/.claude.json} — refused
|
||||
* outright when {@code memberHerdrSocket} is configured, see above
|
||||
* @param cwd the spawn's resolved working directory — the exact key Claude Code will look
|
||||
* up for itself once it starts there
|
||||
* @throws IllegalStateException when {@code memberHerdrSocket} is configured but {@code
|
||||
* configDir} and/or {@code worktreeGroup} is not — the same
|
||||
* refusal shape as {@link #writeCharterFile}
|
||||
*/
|
||||
private static void seedTrustDialog(String configDir, String cwd) {
|
||||
private void seedTrustDialog(String configDir, String cwd) {
|
||||
if (!isProvisionedWorktree(cwd)) {
|
||||
return;
|
||||
}
|
||||
Path target = (configDir == null || configDir.isBlank())
|
||||
boolean unsetConfigDir = configDir == null || configDir.isBlank();
|
||||
boolean memberHerdrSocket = memberHerdrSocketConfigured();
|
||||
String group = memberHerdrSocket ? memberGroup() : null;
|
||||
if (memberHerdrSocket && (unsetConfigDir || group == null)) {
|
||||
throw new IllegalStateException("memberHerdrSocket is configured, so the workspace-trust "
|
||||
+ "seed (.claude.json, which gates Claude Code's interactive trust dialog) must be "
|
||||
+ "placed where the member's OS user can read it — configDir, shared via "
|
||||
+ "worktreeGroup — but " + (unsetConfigDir ? "configDir" : "worktreeGroup")
|
||||
+ " is not configured. Refusing to spawn rather than write fleetd's own default "
|
||||
+ "'~/.claude.json' or hand the member a config file it cannot read: that member "
|
||||
+ "would sit on the interactive trust dialog forever and never reach an "
|
||||
+ "injectable state. Configure configDir on this profile and worktreeGroup on the "
|
||||
+ "fleet to enable claude-code member spawns under memberHerdrSocket.");
|
||||
}
|
||||
Path target = unsetConfigDir
|
||||
? Path.of(System.getProperty("user.home"), ".claude.json")
|
||||
: Path.of(configDir, ".claude.json");
|
||||
if (unsetConfigDir) {
|
||||
// fleetd #247: configDir unset is the ONLY path that targets ~/.claude.json — the
|
||||
// operator's own home file, not a per-profile one — and it is the default, so a
|
||||
// profile that simply forgot to set configDir gets no signal at all short of the
|
||||
// operator noticing their own file changing. Say so loudly, every time it is about to
|
||||
// happen, rather than only once ever: each occurrence is a live write to a real
|
||||
// person's home config and deserves its own log line. (Reached only when
|
||||
// memberHerdrSocket is absent — the block above already refused otherwise.)
|
||||
log.warn("seedTrustDialog: profile has no configDir set, so the workspace-trust seed "
|
||||
+ "for cwd '{}' is about to write the operator's own default '{}' — set "
|
||||
+ "configDir on this profile to target a per-member config file instead",
|
||||
cwd, target);
|
||||
}
|
||||
synchronized (TRUST_JSON_LOCK) {
|
||||
boolean written = false;
|
||||
try {
|
||||
if (target.getParent() != null) {
|
||||
Files.createDirectories(target.getParent());
|
||||
}
|
||||
ObjectNode root = null;
|
||||
if (Files.isRegularFile(target)) {
|
||||
JsonNode existing = TRUST_JSON.readTree(target.toFile());
|
||||
if (existing instanceof ObjectNode existingObject) {
|
||||
root = existingObject;
|
||||
for (int attempt = 1; attempt <= MAX_TRUST_JSON_CAS_ATTEMPTS && !written; attempt++) {
|
||||
byte[] before = Files.isRegularFile(target) ? Files.readAllBytes(target) : null;
|
||||
ObjectNode root = parseTrustJsonOrEmpty(before);
|
||||
JsonNode projectsNode = root.get("projects");
|
||||
ObjectNode projects = projectsNode instanceof ObjectNode projectsObject
|
||||
? projectsObject : TRUST_JSON.createObjectNode();
|
||||
if (!(projectsNode instanceof ObjectNode)) {
|
||||
root.set("projects", projects);
|
||||
}
|
||||
JsonNode projectNode = projects.get(cwd);
|
||||
ObjectNode project = projectNode instanceof ObjectNode projectObject
|
||||
? projectObject : TRUST_JSON.createObjectNode();
|
||||
if (!(projectNode instanceof ObjectNode)) {
|
||||
projects.set(cwd, project);
|
||||
}
|
||||
// fleetd #247: ONLY hasTrustDialogAccepted. We used to write
|
||||
// hasCompletedProjectOnboarding beside it; do not put it back. Measured on
|
||||
// 2026-09-03, minutes after a live spawn seeded this file: 28 of 28 project
|
||||
// entries carried hasTrustDialogAccepted and 0 of 28 carried the onboarding key
|
||||
// — including the 27 entries Claude Code wrote for itself. Claude Code
|
||||
// normalises the whole file when it saves and drops that key every time, so
|
||||
// writing it achieved nothing except making the next reader think it mattered.
|
||||
// The member reached idle with the trust flag alone, which is the only outcome
|
||||
// this seed exists for. If a future Claude Code needs the second flag the
|
||||
// symptom returns as the trust dialog fleetd #149 describes — re-measure then,
|
||||
// do not restore it on a guess.
|
||||
project.put("hasTrustDialogAccepted", true);
|
||||
String newContent = TRUST_JSON.writerWithDefaultPrettyPrinter().writeValueAsString(root);
|
||||
|
||||
trustJsonCasTestHook.run();
|
||||
|
||||
// fleetd #247 CAS: re-read immediately before the move and compare with what
|
||||
// this attempt built its update from. A mismatch means another writer (most
|
||||
// plausibly the operator's own live Claude Code — see this method's javadoc)
|
||||
// landed a change in between; discard this attempt's work and rebuild from the
|
||||
// fresh bytes rather than blindly overwriting it.
|
||||
byte[] atMove = Files.isRegularFile(target) ? Files.readAllBytes(target) : null;
|
||||
if (!Arrays.equals(before, atMove)) {
|
||||
continue;
|
||||
}
|
||||
writeAtomically(target, newContent);
|
||||
written = true;
|
||||
}
|
||||
if (root == null) {
|
||||
root = TRUST_JSON.createObjectNode();
|
||||
if (!written) {
|
||||
// fleetd #247: deliberately do NOT write here. A member that starts without the
|
||||
// seed still starts — it may hit the trust dialog fleetd #149 describes and fail
|
||||
// to reach an injectable state, but that failure is visible (herdr reports it,
|
||||
// the spawn-readiness gate times out) and recoverable (retry the spawn). Writing
|
||||
// our stale copy over whatever the other writer left would be silent and, if that
|
||||
// other writer is the operator's own live session, could destroy real
|
||||
// configuration — fail toward the recoverable outcome, not the silent one.
|
||||
log.warn("seedTrustDialog: gave up seeding workspace-trust for cwd '{}' into '{}' "
|
||||
+ "after {} attempts — another writer (most plausibly the operator's own "
|
||||
+ "live Claude Code sharing this file) kept changing it faster than we "
|
||||
+ "could re-read it, so nothing was written; the member may show the "
|
||||
+ "trust dialog instead", cwd, target, MAX_TRUST_JSON_CAS_ATTEMPTS);
|
||||
}
|
||||
JsonNode projectsNode = root.get("projects");
|
||||
ObjectNode projects = projectsNode instanceof ObjectNode projectsObject
|
||||
? projectsObject : TRUST_JSON.createObjectNode();
|
||||
if (!(projectsNode instanceof ObjectNode)) {
|
||||
root.set("projects", projects);
|
||||
}
|
||||
JsonNode projectNode = projects.get(cwd);
|
||||
ObjectNode project = projectNode instanceof ObjectNode projectObject
|
||||
? projectObject : TRUST_JSON.createObjectNode();
|
||||
if (!(projectNode instanceof ObjectNode)) {
|
||||
projects.set(cwd, project);
|
||||
}
|
||||
project.put("hasTrustDialogAccepted", true);
|
||||
project.put("hasCompletedProjectOnboarding", true);
|
||||
writeAtomically(target, TRUST_JSON.writerWithDefaultPrettyPrinter().writeValueAsString(root));
|
||||
} catch (Exception e) {
|
||||
log.debug("cannot seed workspace-trust entry for cwd '{}' into '{}'", cwd, target, e);
|
||||
return;
|
||||
}
|
||||
// fleetd #285: the write above lands as fleetd's own OS user; under memberHerdrSocket
|
||||
// that is NOT the member's OS user, so without this the member still cannot read the
|
||||
// file it exists to seed — a silent readiness timeout with the write looking "done".
|
||||
// Deliberately OUTSIDE the swallow-all catch above: a group that fails to resolve here
|
||||
// means the seed is unreadable despite a successful write, which must fail as loudly as
|
||||
// writeCharterFile's own EnvAllowListScrub.shareWithGroup call already does.
|
||||
if (written && memberHerdrSocket) {
|
||||
EnvAllowListScrub.shareFileWithGroup(target, group);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/** Bound on {@link #seedTrustDialog}'s fleetd #247 compare-and-swap retry loop. */
|
||||
private static final int MAX_TRUST_JSON_CAS_ATTEMPTS = 5;
|
||||
|
||||
/**
|
||||
* Test-only seam for fleetd #247: invoked once per CAS attempt inside {@link #seedTrustDialog}'s
|
||||
* retry loop, after that attempt has read the target's bytes and built its replacement content,
|
||||
* but immediately before the final re-read/compare that decides whether to write. A no-op in
|
||||
* production. Package-visible (not {@code private}) so {@code ClaudeCodeLauncherTest} can install
|
||||
* a hook here that writes to the target file, deterministically simulating a writer racing
|
||||
* fleetd's own read-modify-write at the exact instant the CAS is meant to catch — the same race
|
||||
* an external process (most plausibly the operator's own live Claude Code) creates, without
|
||||
* depending on real thread scheduling to land the interleaving. A test that sets this MUST
|
||||
* restore it to the no-op default in a {@code finally} block — it is shared, static state.
|
||||
*/
|
||||
static Runnable trustJsonCasTestHook = () -> {};
|
||||
|
||||
/**
|
||||
* Parse {@code bytes} as a {@code .claude.json} tree, or hand back a fresh empty object when
|
||||
* {@code bytes} is {@code null} (no file yet) or does not parse to a JSON object — the same
|
||||
* missing-or-unreadable-is-empty fallback {@link #seedTrustDialog} always used, factored out so
|
||||
* the fleetd #247 CAS loop can call it once per attempt.
|
||||
*/
|
||||
private static ObjectNode parseTrustJsonOrEmpty(byte[] bytes) throws IOException {
|
||||
if (bytes == null) {
|
||||
return TRUST_JSON.createObjectNode();
|
||||
}
|
||||
JsonNode existing = TRUST_JSON.readTree(bytes);
|
||||
return existing instanceof ObjectNode existingObject ? existingObject : TRUST_JSON.createObjectNode();
|
||||
}
|
||||
|
||||
/**
|
||||
* Write {@code content} to {@code target} atomically: serialise to a sibling temp file in the
|
||||
* <strong>same directory</strong> as {@code target} (an atomic move is only guaranteed within
|
||||
@@ -578,7 +737,7 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
* <p><b>fleetd #149 incident.</b> The original implementation used
|
||||
* {@code Files.writeString(target, content)} directly, which truncates {@code target} in place
|
||||
* before writing the replacement bytes. Combined with an ungated {@code cwd} (see
|
||||
* {@link #isProvisionedWorktree}'s javadoc), a mutation-testing run hit that truncation window
|
||||
* {@link HerdrPeerLauncher#isProvisionedWorktree}'s javadoc), a mutation-testing run hit that truncation window
|
||||
* against the operator's real {@code ~/.claude.json} and left it at 178 bytes. The gate closes
|
||||
* <em>which file</em> this can ever target; this closes <em>how</em> the target is written, so
|
||||
* that even a legitimate write against a real, live, concurrently-read {@code .claude.json}
|
||||
@@ -628,32 +787,6 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code cwd} is a fleetd-provisioned git worktree — signalled the same way
|
||||
* {@link #writeIdeOverlay} already gates on: a {@code .git} that is a <strong>regular
|
||||
* file</strong> holding a {@code gitdir:} pointer, as opposed to a real checkout's {@code .git}
|
||||
* <strong>directory</strong>. {@code null}/blank never qualifies.
|
||||
*
|
||||
* <p>Shared by every write that must land only in a worktree fleetd itself created for a
|
||||
* member — never in a real checkout, an arbitrary configured directory, or (see the incident
|
||||
* below) the daemon's own fallback cwd.
|
||||
*
|
||||
* <p><b>fleetd #149 incident.</b> {@link #seedTrustDialog} originally ran unconditionally on
|
||||
* any non-blank {@code cwd}. Most of this launcher's OWN tests spawn a profile with no
|
||||
* {@code cwd} configured, so the base class's {@code resolveCwd} falls through to the real
|
||||
* {@code user.dir} — and with no {@code configDir} either (also the common case in this
|
||||
* file's fixtures), the seed's target falls through the same way to the real
|
||||
* {@code ~/.claude.json}. Running this repo's own test suite corrupted the operator's actual
|
||||
* config file (it shrank from ~72 KB to a single seeded entry) the first time a mutation
|
||||
* happened to make the write non-additive. Gating both cwd-targeted writes on "this is a
|
||||
* worktree fleetd provisioned" — exactly the population fleetd #149 describes
|
||||
* ({@code worktree: true} always lands in a brand-new directory) — makes that class of write
|
||||
* impossible against a real checkout or an untouched fallback cwd, in production or in tests.
|
||||
*/
|
||||
private static boolean isProvisionedWorktree(String cwd) {
|
||||
return cwd != null && !cwd.isBlank() && Files.isRegularFile(Path.of(cwd, ".git"));
|
||||
}
|
||||
|
||||
/** {@code s}, or {@code null} when {@code s} is null/blank — the charter-presence test used above. */
|
||||
private static String nonBlank(String s) {
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
|
||||
@@ -369,8 +369,7 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
// CB-547a: route the chosen profile but keep the caller's session identity — dropping it
|
||||
// here would silently sever the resume handle on every policy-routed spawn. CB-557: the
|
||||
// role rides along for the same reason, or a routed spawn would be labelled as a dev.
|
||||
SpawnRequest routedReq = new SpawnRequest(chosen.profile(), req.requestedCwd(), req.callerCwd(),
|
||||
req.sessionName(), req.resumeSessionId(), req.role());
|
||||
SpawnRequest routedReq = req.withProfile(chosen.profile());
|
||||
try {
|
||||
PeerHandle handle = d.spawn(routedReq);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
@@ -385,9 +384,12 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
// unreachable.size() counts DISTINCT profiles, not attempts (a HashSet dedupes a profile
|
||||
// added twice) — say "distinct" so the count matches the sentence and the profile list that
|
||||
// follows, rather than reading as a count of attempts made (fleetd #315).
|
||||
throw new PeerUnreachableException(
|
||||
"no reachable worker profile available after trying " + unreachable.size()
|
||||
+ " candidate(s): " + String.join(", ", unreachable));
|
||||
+ " distinct candidate(s): " + String.join(", ", unreachable));
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -13,6 +13,7 @@ import java.nio.file.attribute.PosixFilePermissions;
|
||||
import java.time.Duration;
|
||||
import java.time.Instant;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
@@ -28,24 +29,46 @@ import java.util.stream.Stream;
|
||||
* control lacked: herdr applies that overlay BEFORE the shell starts, so any sourced file can undo
|
||||
* it — and did.
|
||||
*
|
||||
* <p><b>Which file is last depends on the platform, so the scrub runs from two of them.</b> zsh
|
||||
* <p><b>zsh reads its four startup files under three different conditions, so no single file is
|
||||
* guaranteed to run — the scrub has to cover the gap between them, not just the platforms.</b> zsh
|
||||
* reads {@code .zshenv} always, {@code .zprofile} and {@code .zlogin} only for a LOGIN shell, and
|
||||
* {@code .zshrc} only for an INTERACTIVE one. herdr does not open the same kind of shell
|
||||
* everywhere — measured on herdr 0.8.0: macOS panes run {@code -zsh} (login, so {@code .zlogin}
|
||||
* runs), Linux panes run a plain {@code /usr/bin/zsh} (interactive but NOT login, so
|
||||
* {@code .zlogin} never runs at all). A scrub in {@code .zlogin} alone is therefore a control that
|
||||
* silently does nothing on Linux — the exact failure this class exists to remove, one platform
|
||||
* over.
|
||||
* {@code .zshrc} only for an INTERACTIVE one. A pane shell that is at least one of login or
|
||||
* interactive is covered by sourcing the scrub from {@code .zshrc} and {@code .zlogin} (below), but
|
||||
* a pane shell that is NEITHER reads only {@code .zshenv} and stops — fleetd #388, measured: a herdr
|
||||
* pane can be neither login nor interactive, and such a pane read {@code .zshenv}, never reached
|
||||
* {@code scrub.zsh}, and left no report at all. A bare {@code argv[0]} of {@code /usr/bin/zsh}
|
||||
* proves the shell is NOT a login shell; it says nothing about whether it is interactive, so it
|
||||
* must never be read as "therefore interactive" — that wrong inference is what let #388 ship.
|
||||
*
|
||||
* <p>So both {@code .zshrc} and {@code .zlogin} source the same generated {@code scrub.zsh} after
|
||||
* sourcing their {@code $HOME} counterpart. On Linux only the first fires; on macOS both do, and
|
||||
* the second pass is deliberate rather than merely harmless — it re-scrubs anything the operator's
|
||||
* own {@code ~/.zlogin} exported after {@code .zshrc} had finished. Re-running is idempotent: a
|
||||
* name already blank is blanked again, and the report is rewritten with the same counts.
|
||||
* <p>So {@code .zshenv} carries a THIRD pass, guarded by the exact condition that defines the gap:
|
||||
* {@code [[ ! -o login && ! -o interactive ]]}. That guard is why this pass cannot double-scrub a
|
||||
* pane that {@code .zshrc} or {@code .zlogin} will also cover — one of {@code -o login}/
|
||||
* {@code -o interactive} is always true there, so the {@code .zshenv} pass never fires for them, and
|
||||
* their own unconditional sourcing is untouched. The guard also carries a sentinel
|
||||
* ({@value #SCRUB_SENTINEL}) so it fires once per PANE and not once per PROCESS: {@code .zshenv} is
|
||||
* read by every zsh a member's own tooling forks (a plain {@code zsh -c '...'} for a single
|
||||
* command is itself neither login nor interactive), and those children inherit variables their
|
||||
* parent deliberately set for them (git hooks get {@code GIT_DIR}, a venv gets
|
||||
* {@code VIRTUAL_ENV}, a build tool gets {@code NODE_OPTIONS} or {@code JAVA_TOOL_OPTIONS}).
|
||||
* Re-scrubbing every such child would blank all of that, and would also make the pane's own
|
||||
* {@code scrub-report.txt} — rewritten on every pass — describe whichever child exited last
|
||||
* instead of the pane. The sentinel is exported only AFTER {@code scrub.zsh} runs, so the pass
|
||||
* that sets it never sees it and cannot blank it; it must also be on the scrub's own allow-list
|
||||
* (see {@link #generate(Path, Set)}) so a later pass, in the same pane, cannot blank it back to
|
||||
* empty — an exported-but-empty sentinel reads as unset to the {@code -z} guard and would silently
|
||||
* re-enable scrubbing for every subsequent child of that pane.
|
||||
*
|
||||
* <p>So all three of {@code .zshenv} (gap only, guarded), {@code .zshrc}, and {@code .zlogin}
|
||||
* source the same generated {@code scrub.zsh} after sourcing their {@code $HOME} counterpart. A
|
||||
* login-and-interactive pane runs the {@code .zshrc} and {@code .zlogin} passes, and the second is
|
||||
* deliberate rather than merely harmless — it re-scrubs anything the operator's own
|
||||
* {@code ~/.zlogin} exported after {@code .zshrc} had finished. A pane that is neither runs only the
|
||||
* {@code .zshenv} pass. Re-running is idempotent: a name already blank is blanked again, and the
|
||||
* report is rewritten with the same counts.
|
||||
*
|
||||
* <p>Each generated file sources its {@code $HOME} counterpart FIRST, so {@code PATH} and every
|
||||
* toolchain binary still resolve exactly as the operator configured them; only afterwards does
|
||||
* {@code .zlogin} run the scrub: every EXPORTED variable not on the derived allow-list is re-exported
|
||||
* toolchain binary still resolve exactly as the operator configured them; only afterwards does the
|
||||
* scrub run: every EXPORTED variable not on the derived allow-list is re-exported
|
||||
* blank. Blank, not credential-shaped-pattern-filtered: a pattern list ({@code *TOKEN*}, …) is an
|
||||
* enumeration and misses what it did not think of — a username is the other half of a credential and
|
||||
* is shaped like none. Credential-SHAPED names among the blanked set go to the WARN log only,
|
||||
@@ -63,7 +86,11 @@ public final class EnvAllowListScrub {
|
||||
/** Name of the report file written into the generated directory by the scrub itself. */
|
||||
static final String REPORT_FILE = "scrub-report.txt";
|
||||
|
||||
/** The scrub body, generated once and sourced from both {@code .zshrc} and {@code .zlogin}. */
|
||||
/**
|
||||
* The scrub body, generated once and sourced from {@code .zshrc} and {@code .zlogin}
|
||||
* unconditionally, and from {@code .zshenv} when the pane shell is neither login nor
|
||||
* interactive (fleetd #388) — see the class javadoc.
|
||||
*/
|
||||
static final String SCRUB_FILE = "scrub.zsh";
|
||||
|
||||
/** Prefix of every generated directory — also what {@link #reapOrphans} matches on. */
|
||||
@@ -79,6 +106,28 @@ public final class EnvAllowListScrub {
|
||||
private static final String SOURCE_SCRUB =
|
||||
"source \"$ZDOTDIR/" + SCRUB_FILE + "\"\n";
|
||||
|
||||
/**
|
||||
* fleetd #388: marks a pane, not a process, as already scrubbed. Set only by the guarded
|
||||
* {@code .zshenv} pass (see {@link #NEITHER_LOGIN_NOR_INTERACTIVE_SCRUB}) after
|
||||
* {@code scrub.zsh} has run, so it must also be folded into that pass's own allow-list — see
|
||||
* the class javadoc's "must also be on the scrub's own allow-list" paragraph.
|
||||
*/
|
||||
static final String SCRUB_SENTINEL = "_CB633_SCRUBBED";
|
||||
|
||||
/**
|
||||
* Appended to {@code .zshenv}, after its {@code $HOME} source: the third pass, guarded on the
|
||||
* exact condition that defines the gap {@code .zshrc}/{@code .zlogin} do not cover — a shell
|
||||
* that is neither login nor interactive. The sentinel export happens only once the scrub has
|
||||
* already run, and only for as long as the current pane's environment has not been rebuilt from
|
||||
* scratch (a fresh {@code env -i} child would not inherit it — that is out of scope here, since
|
||||
* such a child is no longer running under the pane's own environment at all).
|
||||
*/
|
||||
private static final String NEITHER_LOGIN_NOR_INTERACTIVE_SCRUB =
|
||||
"if [[ ! -o login && ! -o interactive && -z \"${" + SCRUB_SENTINEL + ":-}\" ]]; then\n"
|
||||
+ " " + SOURCE_SCRUB
|
||||
+ " export " + SCRUB_SENTINEL + "=1\n"
|
||||
+ "fi\n";
|
||||
|
||||
private EnvAllowListScrub() {
|
||||
}
|
||||
|
||||
@@ -105,11 +154,16 @@ public final class EnvAllowListScrub {
|
||||
reapOrphans(parentDir);
|
||||
Path dir = Files.createTempDirectory(parentDir, DIR_PREFIX);
|
||||
dir.toFile().deleteOnExit();
|
||||
// The report is written by zsh, after these hooks are registered, so register its path
|
||||
// too — otherwise the directory is non-empty at JVM exit and cannot be removed at all.
|
||||
dir.resolve(REPORT_FILE).toFile().deleteOnExit();
|
||||
write(dir, SCRUB_FILE, scrubScript(allowedNames));
|
||||
write(dir, ".zshenv", homeSourcingFile(".zshenv"));
|
||||
// zsh truncates this pre-created receipt after these hooks are registered. Register its
|
||||
// path too — otherwise the directory is non-empty at JVM exit and cannot be removed.
|
||||
Files.createFile(dir.resolve(REPORT_FILE)).toFile().deleteOnExit();
|
||||
// fleetd #388: scrub.zsh's OWN allow-list must also keep SCRUB_SENTINEL, or a later
|
||||
// pass in the same pane blanks it back to empty and the .zshenv guard below thinks it
|
||||
// was never scrubbed — see the class javadoc.
|
||||
Set<String> namesForScrubScript = new HashSet<>(allowedNames);
|
||||
namesForScrubScript.add(SCRUB_SENTINEL);
|
||||
write(dir, SCRUB_FILE, scrubScript(namesForScrubScript));
|
||||
write(dir, ".zshenv", homeSourcingFile(".zshenv") + NEITHER_LOGIN_NOR_INTERACTIVE_SCRUB);
|
||||
write(dir, ".zprofile", homeSourcingFile(".zprofile"));
|
||||
write(dir, ".zshrc", homeSourcingFile(".zshrc") + SOURCE_SCRUB);
|
||||
write(dir, ".zlogin", homeSourcingFile(".zlogin") + SOURCE_SCRUB);
|
||||
@@ -147,12 +201,11 @@ public final class EnvAllowListScrub {
|
||||
* into it: owner keeps full access, {@code group} gets traverse+read on the directory ({@code
|
||||
* rwxr-x---}, so a member process — a login shell reading it via {@code ZDOTDIR}, or another
|
||||
* process simply opening a file under it — running under that group can find and read the
|
||||
* files) and read-only on each file ({@code rw-r-----}) — deliberately no group WRITE anywhere,
|
||||
* since a member never needs to add or change what fleetd generated. (For the ZDOTDIR scrub
|
||||
* specifically, this also means the scrub script's own report write inside the pane fails
|
||||
* closed rather than open — see {@code scrub.zsh}'s trailing {@code 2>/dev/null} — which {@link
|
||||
* dev.ltms.fleet.member.HerdrPeerLauncher#releaseZdotdir} already treats as "cannot be
|
||||
* confirmed to have run" rather than success.)
|
||||
* files) and read-only on each file ({@code rw-r-----}), except the pre-created ZDOTDIR
|
||||
* {@code scrub-report.txt}. That receipt gets group write ({@code rw-rw----}), so
|
||||
* {@code scrub.zsh} can truncate and write it without granting group write on the directory.
|
||||
* If its optional permission change fails, the member cannot write a receipt and the launcher
|
||||
* keeps its existing WARN rather than failing the spawn.
|
||||
*
|
||||
* <p>Package-private and named generically on purpose: fleetd #213 built this for the ZDOTDIR
|
||||
* scrub directory, and fleetd #219 reuses it verbatim for {@link
|
||||
@@ -167,7 +220,15 @@ public final class EnvAllowListScrub {
|
||||
setGroupAndPermissions(dir, principal, "rwxr-x---");
|
||||
try (Stream<Path> entries = Files.list(dir)) {
|
||||
for (Path file : entries.toList()) {
|
||||
setGroupAndPermissions(file, principal, "rw-r-----");
|
||||
if (REPORT_FILE.equals(file.getFileName().toString())) {
|
||||
try {
|
||||
setGroupAndPermissions(file, principal, "rw-rw----");
|
||||
} catch (IOException | UnsupportedOperationException ignored) {
|
||||
// The receipt is optional. Its absence keeps the existing WARN path.
|
||||
}
|
||||
} else {
|
||||
setGroupAndPermissions(file, principal, "rw-r-----");
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (IOException e) {
|
||||
@@ -181,6 +242,33 @@ public final class EnvAllowListScrub {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #285: share ONE file with {@code group}, read-only ({@code rw-r-----}) — the same
|
||||
* per-file mode {@link #shareWithGroup} applies to a directory's entries, and the same error
|
||||
* shapes, but without touching a parent directory. Used for a file fleetd writes into a
|
||||
* directory it does NOT own — {@code configDir}'s own traversal permissions stay the
|
||||
* operator's setup — where the directory-wide {@link #shareWithGroup} would be wrong.
|
||||
*
|
||||
* @throws UncheckedIOException when {@code group} does not resolve on this host, the
|
||||
* filesystem has no POSIX group ownership, or a
|
||||
* group-ownership/permission call is refused
|
||||
*/
|
||||
static void shareFileWithGroup(Path file, String group) {
|
||||
try {
|
||||
GroupPrincipal principal = file.getFileSystem().getUserPrincipalLookupService()
|
||||
.lookupPrincipalByGroupName(group);
|
||||
setGroupAndPermissions(file, principal, "rw-r-----");
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot share generated file " + file + " with group '"
|
||||
+ group + "' — the group must exist, and the fleetd operator ("
|
||||
+ System.getProperty("user.name") + ") must be a member of it", e);
|
||||
} catch (UnsupportedOperationException e) {
|
||||
throw new UncheckedIOException("cannot share generated file " + file + " with group '"
|
||||
+ group + "' — this filesystem does not support POSIX group ownership",
|
||||
new IOException(e));
|
||||
}
|
||||
}
|
||||
|
||||
private static void setGroupAndPermissions(Path path, GroupPrincipal group, String perms) throws IOException {
|
||||
PosixFileAttributeView view = Files.getFileAttributeView(path, PosixFileAttributeView.class);
|
||||
if (view == null) {
|
||||
@@ -218,11 +306,13 @@ public final class EnvAllowListScrub {
|
||||
}
|
||||
return """
|
||||
# generated by fleetd (CB-633 memberCredentials policy=allow-list) — do not edit.
|
||||
# Sourced from .zshrc and again from .zlogin, each time AFTER that file has sourced
|
||||
# its $HOME counterpart — so this runs after everything the operator sourced, on a
|
||||
# login shell (macOS panes) and on a plain interactive one (Linux panes) alike.
|
||||
# Running twice is idempotent and deliberate: the second pass catches anything
|
||||
# ~/.zlogin exported after ~/.zshrc had finished.
|
||||
# Sourced unconditionally from .zshrc and again from .zlogin, each time AFTER that
|
||||
# file has sourced its $HOME counterpart — so this runs after everything the
|
||||
# operator sourced, on any pane that is login and/or interactive. Running twice is
|
||||
# idempotent and deliberate: the second pass catches anything ~/.zlogin exported
|
||||
# after ~/.zshrc had finished. Also sourced, once, from a guarded pass in .zshenv
|
||||
# (fleetd #388) when the pane shell is NEITHER login nor interactive — the one gap
|
||||
# those two files do not cover.
|
||||
|
||||
typeset -A _cb633_allowed
|
||||
for _cb633_n in %s; do _cb633_allowed[$_cb633_n]=1; done
|
||||
@@ -244,7 +334,30 @@ public final class EnvAllowListScrub {
|
||||
_cb633_blank+=("$_cb633_n")
|
||||
done
|
||||
|
||||
{ for _cb633_n in "${_cb633_blank[@]}"; do export "$_cb633_n="; done; } 2>/dev/null
|
||||
# `export UID=` is not a failed command: it is a FATAL zsh parameter error
|
||||
# ("failed to change user ID") that aborts this whole sourced file mid-loop,
|
||||
# leaving every later name unscrubbed and the report below unwritten — silently,
|
||||
# because of the 2>/dev/null. Neither `|| true` nor a `${(t)n}` type guard
|
||||
# contains it; only `eval` does. `eval` is safe here precisely because the loop
|
||||
# above already rejected every name that is not [A-Za-z_][A-Za-z0-9_]*, so
|
||||
# nothing but a bare identifier can reach it.
|
||||
#
|
||||
# Enumerating the special names instead (UID|EUID|GID|EGID|PPID|LINENO) also
|
||||
# works, but only for the ones enumerated: a special that turns up exported on
|
||||
# some other host brings the abort straight back. `eval` contains all of them.
|
||||
#
|
||||
# Then VERIFY. A contained failure is still a failure, so a name that did not
|
||||
# actually blank must not be reported as blanked. It currently falls into the
|
||||
# "allowed" count, which is imprecise in the safe direction; the honest third
|
||||
# count ("tried and could not blank") needs a report-format change and belongs
|
||||
# with fleetd #394, not here.
|
||||
typeset -a _cb633_done
|
||||
_cb633_done=()
|
||||
for _cb633_n in "${_cb633_blank[@]}"; do
|
||||
eval "export ${_cb633_n}=" 2>/dev/null
|
||||
[[ -z "${(P)_cb633_n}" ]] && _cb633_done+=("$_cb633_n")
|
||||
done
|
||||
_cb633_blank=("${_cb633_done[@]}")
|
||||
|
||||
integer _cb633_kept=$(( _cb633_total - ${#_cb633_blank} ))
|
||||
{
|
||||
@@ -252,7 +365,7 @@ public final class EnvAllowListScrub {
|
||||
for _cb633_n in "${_cb633_blank[@]}"; do print -r -- "$_cb633_n"; done
|
||||
} > "$ZDOTDIR/%s" 2>/dev/null
|
||||
|
||||
unset _cb633_allowed _cb633_names _cb633_blank _cb633_n _cb633_total _cb633_kept
|
||||
unset _cb633_done _cb633_allowed _cb633_names _cb633_blank _cb633_n _cb633_total _cb633_kept
|
||||
""".formatted(names, MemberEnvAllowList.zshCasePattern(), REPORT_FILE);
|
||||
}
|
||||
|
||||
|
||||
@@ -117,9 +117,11 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
*
|
||||
* <p>fleetd #185 stage 2: that mirroring assumption holds only while the member pane runs under
|
||||
* the SAME OS user as the daemon. When {@code memberHerdrSocket:} is configured, member panes
|
||||
* run on a second herdr owned by a different user — different {@code $HOME}, different {@code
|
||||
* secrets.sh}, different environment entirely — so this field's data no longer describes what a
|
||||
* member pane inherits. See {@link #logCredentialGap} for how that mode is handled.
|
||||
* are routed to a second herdr, and fleetd has no channel to confirm what OS user that herdr
|
||||
* runs as — it may be a different user with a different {@code $HOME} and {@code secrets.sh},
|
||||
* or the same one the daemon runs as. Either way this field's data can no longer be trusted to
|
||||
* describe what a member pane inherits. See {@link #logCredentialGap} for how that mode is
|
||||
* handled.
|
||||
*/
|
||||
private final Supplier<Set<String>> hostEnvNames;
|
||||
|
||||
@@ -184,7 +186,11 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
*/
|
||||
private final ConcurrentMap<String, Path> zdotdirByPane = new ConcurrentHashMap<>();
|
||||
|
||||
/** Guards {@link #warnNonZsh} to one WARN per launcher instance, not one per spawn. */
|
||||
/**
|
||||
* Guards {@link #warnNonZsh} to one WARN per launcher instance, not one per spawn. fleetd #155:
|
||||
* only the {@code policy: deny-by-default} path still warns on a non-zsh shell — the {@code
|
||||
* policy: allow-list} path refuses the spawn instead (see {@link #applyEnvironmentAllowListPolicy}).
|
||||
*/
|
||||
private final AtomicBoolean nonZshShellWarned = new AtomicBoolean();
|
||||
/** Live config provides URI environment names that must never enter member panes. */
|
||||
private final Supplier<FleetConfig> config;
|
||||
@@ -358,6 +364,42 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
return Files.isRegularFile(candidate) ? candidate : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code cwd} is a fleetd-provisioned git worktree — signalled the same way
|
||||
* {@code ClaudeCodeLauncher#writeIdeOverlay} already gates on: a {@code .git} that is a
|
||||
* <strong>regular file</strong> holding a {@code gitdir:} pointer, as opposed to a real
|
||||
* checkout's {@code .git} <strong>directory</strong>. {@code null}/blank never qualifies.
|
||||
*
|
||||
* <p>Shared by every write (and, since fleetd #249, every identity read) that must land only
|
||||
* in a worktree fleetd itself created for a member — never in a real checkout, an arbitrary
|
||||
* configured directory, or (see the incident below) the daemon's own fallback cwd. Package-
|
||||
* private (not {@code protected}) on purpose: {@link ClaudeCodeLauncher} and
|
||||
* {@link OpenCodeLauncher} both call it, and same-package visibility is enough — no subclass
|
||||
* outside this package needs it.
|
||||
*
|
||||
* <p><b>fleetd #149 incident.</b> {@code ClaudeCodeLauncher#seedTrustDialog} originally ran
|
||||
* unconditionally on any non-blank {@code cwd}. Most of that launcher's OWN tests spawn a
|
||||
* profile with no {@code cwd} configured, so the base class's {@code resolveCwd} falls
|
||||
* through to the real {@code user.dir} — and with no {@code configDir} either (also the
|
||||
* common case in that file's fixtures), the seed's target falls through the same way to the
|
||||
* real {@code ~/.claude.json}. Running this repo's own test suite corrupted the operator's
|
||||
* actual config file (it shrank from ~72 KB to a single seeded entry) the first time a
|
||||
* mutation happened to make the write non-additive. Gating both cwd-targeted writes on "this
|
||||
* is a worktree fleetd provisioned" — exactly the population fleetd #149 describes
|
||||
* ({@code worktree: true} always lands in a brand-new directory) — makes that class of write
|
||||
* impossible against a real checkout or an untouched fallback cwd, in production or in tests.
|
||||
*
|
||||
* <p><b>fleetd #249.</b> The same reasoning extends to a READ: {@code
|
||||
* OpenCodeSessionDiscovery#sessionIdForDirectory} keys on {@code directory}, a heuristic that
|
||||
* is only reliable when the directory is unique to this member — i.e., exactly the population
|
||||
* this gate identifies. {@link OpenCodeLauncher} uses it to withhold {@code agentSessionId()}
|
||||
* (report absence rather than a guess) and to refuse a {@code resumeSessionId} spawn that
|
||||
* cannot be resolved reliably going forward.
|
||||
*/
|
||||
static boolean isProvisionedWorktree(String cwd) {
|
||||
return cwd != null && !cwd.isBlank() && Files.isRegularFile(Path.of(cwd, ".git"));
|
||||
}
|
||||
|
||||
// --- profile surface -----------------------------------------------------------------------
|
||||
|
||||
/** The configured peer profile names (what {@code spawn(profile)} accepts). */
|
||||
@@ -655,7 +697,20 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
if (paneId == null) {
|
||||
throw new IllegalStateException("pane.split returned no pane — cannot start a peer");
|
||||
}
|
||||
Agent peer = startUniquelyNamed(cfg, argv, paneId).agent();
|
||||
Agent peer;
|
||||
try {
|
||||
peer = startUniquelyNamed(cfg, argv, paneId).agent();
|
||||
} catch (RuntimeException e) {
|
||||
// The peer never started — don't leave the pane we just created orphaned.
|
||||
// Best-effort cleanup; never let it mask the real spawn failure.
|
||||
try {
|
||||
stop(paneId);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to close orphaned pane {} after spawn error: {}",
|
||||
paneId, cleanup.getMessage());
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
log.info("{} started pane={} terminal={}", namePrefix, peer.paneId(), peer.terminalId());
|
||||
return peer;
|
||||
}
|
||||
@@ -871,20 +926,31 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* {@link #reapOrphanWorkers() orphan-reap} and spawn-gate-timeout paths, plus any caller that
|
||||
* passes a pane directly, keep working without an owning id.
|
||||
*
|
||||
* <p>Resolves the tab from the pane <em>before</em> closing it. An already-gone pane/tab
|
||||
* (repeated DELETE, crashed peer) is treated as success; any other failure propagates so a
|
||||
* genuinely failed teardown is not reported as done.
|
||||
* <p>Resolves the tab from the pane <em>before</em> closing it. {@code agents.close} (the pane)
|
||||
* is the one step whose failure means the teardown itself may not have happened: an already-gone
|
||||
* pane (repeated DELETE, crashed peer) is treated as success, but any other failure propagates so
|
||||
* a genuinely failed teardown is not reported as done. {@code spaces.closeTab} (fleetd #293) is
|
||||
* different — by the time it runs the pane is already closed, so it is cosmetic workspace tidying
|
||||
* rather than a real teardown failure, and a failure there is logged and never propagates, so it
|
||||
* cannot mask the two cleanups below it ({@link #releaseZdotdir}, and the caller's worktree
|
||||
* removal in {@code SessionManager.release}).
|
||||
*/
|
||||
@Override
|
||||
public void stop(String idOrPane) {
|
||||
// Teardown knows only the pane, not which profile spawned it. Attempt tab cleanup when any
|
||||
// profile uses tab placement (so the bridge may have created a dedicated peer tab); the
|
||||
// single-occupant check below is what actually protects the user's shared tabs.
|
||||
// Teardown knows only the pane, not which profile spawned it — and, when this call is
|
||||
// routed here through CompositePeerLauncher's single-daemon stop() shortcut (fleetd #342),
|
||||
// not even which adapter's config actually governed the spawn: the shortcut can hand the
|
||||
// pane to a delegate that never spawned it, whose own profiles say nothing about how THIS
|
||||
// pane was placed. So the decision to look for a tab to clean up is made from the pane's
|
||||
// actual state, not from this delegate's static profile config: resolve the tab
|
||||
// unconditionally and let {@link WorkspaceControl#locatePane} tolerate "not found" (it
|
||||
// returns null rather than throwing); the single-occupant check below is what actually
|
||||
// protects the user's shared tabs, exactly as it always has.
|
||||
String paneId = paneByAgentId.remove(idOrPane);
|
||||
if (paneId == null) {
|
||||
paneId = idOrPane; // raw-pane fallback (reap, gate timeout, pane-addressed callers)
|
||||
}
|
||||
WorkspaceControl.PaneLocation loc = usesTabPlacement() ? spaces.locatePane(paneId) : null;
|
||||
WorkspaceControl.PaneLocation loc = spaces.locatePane(paneId);
|
||||
try {
|
||||
agents.close(paneId);
|
||||
} catch (HerdrException e) {
|
||||
@@ -892,7 +958,25 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
log.debug("pane.close({}) ignored — already gone: {}", paneId, e.getMessage());
|
||||
}
|
||||
if (loc != null && loc.tabPaneCount() == 1) {
|
||||
spaces.closeTab(loc.tabId());
|
||||
// fleetd #293: the pane above is already closed by this point, so a failing tab.close is
|
||||
// cosmetic workspace tidying, not a real teardown failure — it must not mask the two
|
||||
// cleanups below it (releaseZdotdir, and the caller's worktree removal). Unlike
|
||||
// agents.close above, this is not narrowed to "already gone": any failure here, whatever
|
||||
// its cause, is one we continue past, so we log it at WARN (not debug) with the tab id a
|
||||
// person can go close by hand.
|
||||
try {
|
||||
spaces.closeTab(loc.tabId());
|
||||
} catch (RuntimeException e) {
|
||||
// Caught as RuntimeException, not HerdrException, to match releaseZdotdir's own
|
||||
// guard five lines below. Today the two are the same set — HerdrCodec wraps every
|
||||
// encode/decode failure and UnixSocketHerdrClient wraps every IOException, so
|
||||
// HerdrException is all closeTab can actually throw. Narrowing to it anyway would
|
||||
// leave this step guarded against the expected failure and bare against any other,
|
||||
// which is the exact asymmetry fleetd #293 exists to remove. No behaviour change
|
||||
// today; it stops a later change inside WorkspaceControl.closeTab reopening it.
|
||||
log.warn("tab.close({}) failed — the pane is already torn down, so continuing; the "
|
||||
+ "tab may need manual cleanup: {}", loc.tabId(), e.getMessage());
|
||||
}
|
||||
} else if (loc != null) {
|
||||
log.debug("not closing tab {} — it holds {} panes (not a dedicated peer tab)",
|
||||
loc.tabId(), loc.tabPaneCount());
|
||||
@@ -907,11 +991,6 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
/** Whether any configured profile places peers in their own tab (so tabs may need cleanup). */
|
||||
private boolean usesTabPlacement() {
|
||||
return profiles.values().stream().anyMatch(FleetConfig.Profile::tabPlacement);
|
||||
}
|
||||
|
||||
/** True when a herdr error means the target is already gone (safe to treat as done). */
|
||||
private static boolean isAlreadyGone(HerdrException e) {
|
||||
return e.code() != null && e.code().endsWith("_not_found");
|
||||
@@ -931,7 +1010,8 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* that gap: it stops waiting immediately (never burns the rest of the timeout), runs the same
|
||||
* teardown the timeout path below runs, and throws with a message that says the backend exited
|
||||
* rather than that the pane was slow. Any other {@link HerdrException} still propagates
|
||||
* unchanged — this gate does not know how to recover from it.
|
||||
* unchanged — this gate does not interpret or recover from it, but it still closes the pane
|
||||
* it opened before handing the exception to its caller.
|
||||
*/
|
||||
private void waitUntilInjectableOrThrow(String paneId) {
|
||||
long start = nowMillis.getAsLong();
|
||||
@@ -945,7 +1025,15 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
if (isAlreadyGone(e)) {
|
||||
failFastOnGoneBackend(paneId, e, nowMillis.getAsLong() - start);
|
||||
}
|
||||
throw e; // any other herdr failure is not ours to interpret — let it propagate
|
||||
// This gate must not interpret an unrelated herdr error, but the caller does not
|
||||
// receive paneId when spawn throws. Close the pane here before propagating e unchanged.
|
||||
try {
|
||||
stop(paneId);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to close orphaned pane {} after readiness-gate error: {}",
|
||||
paneId, cleanup.getMessage());
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
lastStatus = sample.status();
|
||||
if (lastStatus.injectable() || refinedInjectable(paneId, sample)) {
|
||||
@@ -1219,6 +1307,11 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
if (!creds.isAllowList()) {
|
||||
overlayBlockedCredentials(workerEnv, creds);
|
||||
logCredentialGap(creds, null);
|
||||
// fleetd #155: this overlay itself does not depend on the shell (it lands in the
|
||||
// pane-creation env map before any shell runs), so the spawn is never refused here —
|
||||
// only policy=allow-list's ZDOTDIR scrub needs a login shell to run at all. Still worth
|
||||
// telling the operator: the stronger post-shell control is unavailable on this shell.
|
||||
warnNonZsh(memberLoginShell());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1271,10 +1364,16 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* dev.ltms.fleet.session.Worktrees#shareWithGroup} already uses, reused rather than
|
||||
* inventing a second group key. Either one missing means the scrub cannot be guaranteed
|
||||
* reachable by the member, which is the same "cannot guarantee the scrub runs" case as a
|
||||
* non-zsh shell, so it gets the identical fallback.</li>
|
||||
* non-zsh shell — that one still gets the fallback (see fleetd #155's javadoc note below
|
||||
* on why a missing worktreeRoot/worktreeGroup is a different defect, out of that ticket's
|
||||
* scope).</li>
|
||||
* </ul>
|
||||
* In every branch: never refuse to spawn. A degraded credential control must not become an
|
||||
* outage for an opt-in feature.
|
||||
* fleetd #155: the one exception to "never refuse to spawn" is a non-zsh login shell, right
|
||||
* below. The operator asked for {@code policy: allow-list} specifically — its whole point is a
|
||||
* control a sourced file cannot undo — so silently degrading to the weaker overlay is exactly
|
||||
* the "silently does nothing" failure this ticket exists to remove. Every other branch in this
|
||||
* method (worktreeRoot/worktreeGroup missing) keeps the old "degrade, never refuse" behaviour;
|
||||
* that gap is real but is fleetd #213's scope, not this one.
|
||||
*/
|
||||
private Path applyEnvironmentAllowListPolicy(FleetConfig.Profile cfg, Launch launch) {
|
||||
FleetConfig.MemberCredentials creds = memberCredentials == null ? null : memberCredentials.get();
|
||||
@@ -1288,20 +1387,21 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
// must not even be called on this path, only the explicit memberLoginShell: config can
|
||||
// answer it. With memberHerdrSocket absent, nothing here changes: fleetd's own $SHELL is
|
||||
// still the input, exactly as before this fix.
|
||||
String loginShell = memberHerdrSocket ? configuredMemberLoginShell() : resolveEnv("SHELL");
|
||||
String loginShell = memberLoginShell();
|
||||
boolean zsh = isZshShell(loginShell);
|
||||
if (!zsh) {
|
||||
// A non-zsh login shell ignores ZDOTDIR entirely: NO scrub would run, so pretending
|
||||
// otherwise would be worse than saying so. Warn loudly and fall back to the CB-596
|
||||
// sentinel overlay over the enumerated known: names — weaker (a sourced file can undo
|
||||
// it), but strictly better than nothing. Deliberately no "allowed N of M" line here: the
|
||||
// scrub this count describes does not run on this path, so printing it would tell an
|
||||
// operator that a fraction of names were blocked when the real number blocked is zero.
|
||||
// logCredentialGap's WARN (below) is the only signal for this path.
|
||||
warnNonZsh(loginShell);
|
||||
overlayBlockedCredentials(launch.env(), creds);
|
||||
logCredentialGap(creds, null);
|
||||
return null;
|
||||
// fleetd #155: a non-zsh login shell ignores ZDOTDIR entirely — NO scrub would run. The
|
||||
// operator explicitly asked for policy=allow-list's blocking control, so degrading to
|
||||
// the weaker CB-596 overlay and spawning anyway would be the same "control silently does
|
||||
// nothing" defect this ticket exists to close. Refuse instead — the caller (FleetMcp.spawn)
|
||||
// catches IllegalArgumentException and surfaces the message to the operator.
|
||||
throw new IllegalArgumentException(
|
||||
"memberCredentials policy=allow-list requires the member's login shell to be "
|
||||
+ "zsh, so the ZDOTDIR scrub can run after it — refusing to spawn under "
|
||||
+ "login shell '" + (loginShell == null ? "<unset>" : loginShell)
|
||||
+ "'. Configure memberLoginShell: as a zsh path (only read when "
|
||||
+ "memberHerdrSocket is set), move the member's OS account onto zsh, or "
|
||||
+ "set memberCredentials.policy: deny-by-default instead.");
|
||||
}
|
||||
Path parentDir;
|
||||
String group = null;
|
||||
@@ -1356,6 +1456,20 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
return cfg == null ? null : cfg.memberLoginShell();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #155: the member pane's login shell, using the same {@code memberHerdrSocket} routing
|
||||
* {@link #applyEnvironmentAllowListPolicy} already used — fleetd's own {@code $SHELL} decides it
|
||||
* when {@code memberHerdrSocket} is absent (today's only mode); the configured {@code
|
||||
* memberLoginShell:} decides it otherwise, since fleetd's own {@code $SHELL} names a different
|
||||
* user's shell once member panes run under a different OS user. Shared by both {@link
|
||||
* #applyEnvironmentAllowListPolicy} (allow-list: refuses on non-zsh) and {@link
|
||||
* #applyMemberCredentialPolicy} (deny-by-default: warns on non-zsh) so the two policies agree on
|
||||
* what "the member's shell" means.
|
||||
*/
|
||||
private String memberLoginShell() {
|
||||
return memberHerdrSocketConfigured() ? configuredMemberLoginShell() : resolveEnv("SHELL");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213: {@code worktreeRoot}, as the ZDOTDIR scrub's parent directory under {@code
|
||||
* memberHerdrSocket}, or {@code null} when unconfigured — the same "cannot guarantee the scrub
|
||||
@@ -1405,9 +1519,9 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
Set<String> brokerUriEnvNames = brokerUriEnvNames();
|
||||
Set<String> allowed = new java.util.TreeSet<>(
|
||||
MemberEnvAllowList.derive(profiles.values(), creds.allowSet(), brokerUriEnvNames));
|
||||
if (creds.sshAuthSockAllowed()) {
|
||||
if (creds.sshAgentEnvInherited()) {
|
||||
allowed.add(SSH_AUTH_SOCK);
|
||||
} // blocked by default: absent from the set ⇒ blanked by the scrub like any other name
|
||||
} // omitted by default: absent from the set ⇒ blanked by the scrub like any other name
|
||||
allowed.addAll(launch.env().keySet());
|
||||
allowed.removeAll(brokerUriEnvNames);
|
||||
return allowed;
|
||||
@@ -1429,10 +1543,27 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* survive {@code allowed} (including the {@code LC_*} prefix rule). Neither number is a constant:
|
||||
* both come from the actual derived set and the actual environment this spawn sees. Never logs a
|
||||
* variable NAME or VALUE — only the counts.
|
||||
*
|
||||
* <p><strong>Under {@code memberHerdrSocket} the counts describe fleetd's own process, not the
|
||||
* member's</strong> (fleetd #269 follow-up), so the message says so rather than leaving the
|
||||
* reader to infer it from this javadoc, which the operator reading the log never sees.
|
||||
*/
|
||||
private void logAllowListCoverage(Set<String> allowed) {
|
||||
Set<String> hostNames = hostEnvNames.get();
|
||||
long kept = hostNames.stream().filter(name -> MemberEnvAllowList.keeps(allowed, name)).count();
|
||||
if (memberHerdrSocketConfigured()) {
|
||||
// fleetd #269 covered the sibling line below (logCredentialGap) and stopped there.
|
||||
// This line has the same problem: read plainly, "allowed 7 of 39" is a statement about
|
||||
// the member's pane, and under memberHerdrSocket it is not -- the pane is routed to a
|
||||
// second herdr whose environment fleetd cannot inspect. The counts stay useful, so
|
||||
// this is not a WARN and not a refusal; only the claim is narrowed to what is true.
|
||||
log.info("member credentials: allowed {} of {} names in fleetd's OWN environment — "
|
||||
+ "memberHerdrSocket is configured, so member panes are routed to a "
|
||||
+ "second herdr whose environment fleetd has no channel to inspect. "
|
||||
+ "These counts describe fleetd's process, NOT the member pane's.",
|
||||
kept, hostNames.size());
|
||||
return;
|
||||
}
|
||||
log.info("member credentials: allowed {} of {}", kept, hostNames.size());
|
||||
}
|
||||
|
||||
@@ -1440,15 +1571,23 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
private static final String SSH_AUTH_SOCK = MemberEnvAllowList.SSH_AUTH_SOCK;
|
||||
|
||||
/**
|
||||
* CB-633: a non-zsh login shell means the allow-list control CANNOT run — say so once per
|
||||
* launcher instance, naming the shell, instead of failing silently.
|
||||
* fleetd #155: called ONLY from the {@code policy: deny-by-default} branch of {@link
|
||||
* #applyMemberCredentialPolicy} — under {@code policy: allow-list}, a non-zsh shell now REFUSES
|
||||
* the spawn instead (see {@link #applyEnvironmentAllowListPolicy}), since that policy's whole
|
||||
* point is a control a sourced file cannot undo. deny-by-default's own overlay does not depend
|
||||
* on the shell (it lands in the pane-creation env map before any shell runs), so this is a
|
||||
* heads-up, not a defect report: say once per launcher instance, naming the shell, that the
|
||||
* stronger allow-list control is unavailable here — never silently.
|
||||
*/
|
||||
private void warnNonZsh(String shell) {
|
||||
if (isZshShell(shell)) {
|
||||
return;
|
||||
}
|
||||
if (nonZshShellWarned.compareAndSet(false, true)) {
|
||||
log.warn("memberCredentials policy=allow-list: member login shell '{}' is NOT zsh — "
|
||||
+ "ZDOTDIR scrubbing cannot run, so members' inherited environment is "
|
||||
+ "UNPROTECTED beyond the enumerated known: fallback. Move herdr onto a "
|
||||
+ "zsh account or switch policy back to deny-by-default.",
|
||||
log.warn("memberCredentials policy=deny-by-default: member login shell '{}' is NOT zsh — "
|
||||
+ "the pane-creation credential overlay still applies here (it does not "
|
||||
+ "depend on the shell), but policy=allow-list's stronger post-shell "
|
||||
+ "ZDOTDIR scrub is unavailable on this shell.",
|
||||
shell == null ? "<unset>" : shell);
|
||||
}
|
||||
}
|
||||
@@ -1459,21 +1598,24 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
/**
|
||||
* fleetd #213: {@code memberHerdrSocket} is configured and the member login shell IS zsh, but
|
||||
* {@code worktreeRoot} and/or {@code worktreeGroup} is missing, so the generated ZDOTDIR cannot
|
||||
* be placed anywhere the member's OS user can reach — {@code java.io.tmpdir} is fleetd's own
|
||||
* 0700 temp dir, unreadable by another uid, which is the exact gap this ticket exists to close.
|
||||
* Say so once per launcher instance, instead of either generating a directory nothing can read
|
||||
* (protection theatre) or refusing to spawn (turning a degraded credential control into an
|
||||
* outage for an opt-in feature).
|
||||
* be placed anywhere fleetd can be sure the member's OS user can reach — {@code java.io.tmpdir}
|
||||
* is fleetd's own 0700 temp dir, which is unreadable if the member pane runs as a different OS
|
||||
* user, and fleetd has no channel to confirm whether it does or not. Rather than gamble on that,
|
||||
* this treats memberHerdrSocket as reason enough to require an explicitly shared location, which
|
||||
* is the exact gap this ticket exists to close. Say so once per launcher instance, instead of
|
||||
* either generating a directory that might not be readable (protection theatre) or refusing to
|
||||
* spawn (turning a degraded credential control into an outage for an opt-in feature).
|
||||
*/
|
||||
private void warnCannotShareScrubDirectory() {
|
||||
if (cannotShareScrubDirWarned.compareAndSet(false, true)) {
|
||||
log.warn("memberCredentials policy=allow-list: memberHerdrSocket is configured and the "
|
||||
+ "member login shell is zsh, but worktreeRoot and/or worktreeGroup is not "
|
||||
+ "configured — the generated ZDOTDIR cannot be placed where the member's OS "
|
||||
+ "user can read it (java.io.tmpdir is fleetd's own, unreadable by another uid), "
|
||||
+ "so the scrub cannot be guaranteed to run. Falling back to the CB-596 sentinel "
|
||||
+ "overlay. Configure both worktreeRoot and worktreeGroup to enable the "
|
||||
+ "allow-list scrub under memberHerdrSocket.");
|
||||
+ "configured — the generated ZDOTDIR cannot be placed where fleetd can be sure "
|
||||
+ "the member's OS user can read it (java.io.tmpdir is fleetd's own 0700 dir, "
|
||||
+ "unreadable if the member runs as a different OS user — fleetd has no channel "
|
||||
+ "to confirm whether it does), so the scrub cannot be guaranteed to run. Falling "
|
||||
+ "back to the CB-596 sentinel overlay. Configure both worktreeRoot and "
|
||||
+ "worktreeGroup to enable the allow-list scrub under memberHerdrSocket.");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1524,38 +1666,69 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
|
||||
/**
|
||||
* Guards the {@code effectiveAllowed == null} branch of {@link #logCredentialGap} — the
|
||||
* genuinely-unprotected report (deny-by-default, and the allow-list non-zsh fallback) — to one
|
||||
* WARN per launcher instance, not one per spawn.
|
||||
* genuinely-unprotected report (deny-by-default, and the allow-list non-zsh fallback) — AND
|
||||
* the {@code effectiveAllowed != null} / {@code keptByDerivedList} branch, the allow-list case
|
||||
* where a name is in the gap but the derived allow-list keeps it anyway. Both branches log the
|
||||
* same severity (WARN) about the same fact — a name genuinely reaching a member pane
|
||||
* unprotected — so they share this one guard, keyed per NAME rather than per launcher instance:
|
||||
* each credential-shaped name that is ever reported unprotected gets exactly one WARN, however
|
||||
* many spawns see it and whichever of the two branches first reports it.
|
||||
*
|
||||
* <p>fleetd #341: {@code memberCredentials} is a live, re-read-per-spawn supplier, so the
|
||||
* policy — and so the gap's actual member names — can change between two spawns on the same
|
||||
* launcher. Before this fix the guard was a single {@code AtomicBoolean} tripped by either
|
||||
* branch: spawn 1 could warn about name A and trip the flag, and a later spawn's gap containing
|
||||
* a different name B would never be reported, even though B is just as unprotected as A was.
|
||||
* {@code AtomicBoolean} could not express "once per distinct name" at all — only "once, ever,
|
||||
* for whichever name got there first" — so this is a {@code Set<String>} guard instead, the
|
||||
* same shape {@link OpenCodeLauncher#modelCheckSkippedWarned} already uses for its own
|
||||
* once-per-distinct-thing WARN. {@link #add}'s return value (true only the first time a name is
|
||||
* added) is what turns "log the whole gap" into "log only the names never warned about before".
|
||||
*
|
||||
* <p>Bounded by construction: every name added here first passed {@link
|
||||
* #CREDENTIAL_SHAPED_NAME}'s filter over {@link #hostEnvNames}, i.e. it is an actual
|
||||
* environment variable name from the daemon's own process — a small, OS-bounded set (the host
|
||||
* environment has, in practice, tens to a few hundred entries), not an attacker- or
|
||||
* request-controlled input. So this set cannot grow past "however many distinct credential-
|
||||
* shaped names this host's environment has ever held across this launcher's lifetime," which is
|
||||
* effectively fixed for the life of one daemon process — no separate cap is needed.
|
||||
*
|
||||
* <p>CB-633 follow-up (#192): kept SEPARATE from {@link #allowListGapLogged} on purpose.
|
||||
* {@code memberCredentials} is a live, re-read-per-spawn supplier, so the policy can change
|
||||
* between two spawns on the same launcher. A single shared flag would let a harmless allow-list
|
||||
* INFO on spawn 1 permanently suppress the real deny-by-default WARN a later spawn deserves —
|
||||
* the report that matters most getting hidden by the report that doesn't. Two flags mean each
|
||||
* report kind fires exactly once, independent of what the other kind already logged.
|
||||
* report kind (WARN vs. INFO) fires independently of what the other kind already logged; within
|
||||
* the WARN kind itself, the set above further separates by name, for the same reason.
|
||||
*/
|
||||
private final AtomicBoolean unprotectedGapLogged = new AtomicBoolean();
|
||||
private final Set<String> unprotectedGapNamesWarned = ConcurrentHashMap.newKeySet();
|
||||
|
||||
/**
|
||||
* Guards the {@code effectiveAllowed != null} branch of {@link #logCredentialGap} — the
|
||||
* allow-list-scrub-covered report — to one INFO per launcher instance. See {@link
|
||||
* #unprotectedGapLogged}'s javadoc for why this is a separate flag rather than a shared one.
|
||||
* #unprotectedGapNamesWarned}'s javadoc for why this is a separate flag rather than a shared
|
||||
* one; unlike that guard it stays a per-instance {@code AtomicBoolean}, not a per-name set —
|
||||
* fleetd #341 fixed the WARN-vs-WARN suppression, not this INFO's own one-shot shape, which was
|
||||
* not reported as broken and is out of that ticket's scope.
|
||||
*/
|
||||
private final AtomicBoolean allowListGapLogged = new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* fleetd #185 stage 2: guards {@link #warnUnknownMemberEnvironment} to one WARN per launcher
|
||||
* instance, not one per spawn — the same one-per-instance shape as {@link #unprotectedGapLogged}
|
||||
* and {@link #allowListGapLogged}, kept as its own flag for the same reason those two are split:
|
||||
* this mode is orthogonal to which of the other two branches would otherwise have fired.
|
||||
* instance, not one per spawn — the same one-shot shape {@link #unprotectedGapNamesWarned} and
|
||||
* {@link #allowListGapLogged} guard their own branches with, kept as its own flag for the same
|
||||
* reason those two are split: this mode is orthogonal to which of the other two branches would
|
||||
* otherwise have fired.
|
||||
*/
|
||||
private final AtomicBoolean unknownMemberEnvironmentWarned = new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* fleetd #185 stage 2: whether {@code memberHerdrSocket:} is configured, i.e. member panes run
|
||||
* on a second herdr owned by a different OS user than the daemon's own process. Re-read from the
|
||||
* live config on every call (same hot-reload shape as {@link #memberCredentials}), never cached,
|
||||
* so a config reload takes effect on the next spawn without a restart.
|
||||
* fleetd #185 stage 2: whether {@code memberHerdrSocket:} is configured, i.e. member panes are
|
||||
* routed to a second herdr. This tests only that the config key is set — fleetd has no channel
|
||||
* to confirm what OS user that second herdr runs as, so a {@code true} result means "member
|
||||
* panes may run under a different OS user," not that they do. Re-read from the live config on
|
||||
* every call (same hot-reload shape as {@link #memberCredentials}), never cached, so a config
|
||||
* reload takes effect on the next spawn without a restart.
|
||||
*
|
||||
* <p>{@link #config} is {@code null} on any call site that never threaded the full config
|
||||
* through (every production {@code HerdrPeerLauncher} does; a handful of older tests do not) —
|
||||
@@ -1578,14 +1751,15 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* fleetd #185 stage 2: the single replacement WARN for {@link #logCredentialGap}'s usual
|
||||
* conclusions when {@code memberHerdrSocket:} is configured. {@link #hostEnvNames} (and
|
||||
* everything derived from it — {@code known}/{@code allow} coverage, the allow-list scrub's
|
||||
* derived set) describes the DAEMON's own environment; under this config key member panes run as
|
||||
* a different OS user with a different environment entirely, so neither "every member pane
|
||||
* inherits them UNBLOCKED" nor "the scrub blanks them" is evidence-backed here — both would be
|
||||
* reporting on the wrong process. Logged once, names the config key, and states the honest
|
||||
* conclusion: the gap for member panes is UNKNOWN, not clean, so {@code memberCredentials} cannot
|
||||
* be verified from this daemon. The one count it does report is scoped explicitly to fleetd's own
|
||||
* environment, never presented as if it said anything about the member's — see {@link
|
||||
* #logCredentialGap}'s javadoc for why this branch exists.
|
||||
* derived set) describes the DAEMON's own environment; under this config key member panes are
|
||||
* routed to a second herdr, and fleetd has no channel to confirm what OS user that herdr runs
|
||||
* as or to read its environment, so neither "every member pane inherits them UNBLOCKED" nor
|
||||
* "the scrub blanks them" is evidence-backed here — both would be reporting on the wrong
|
||||
* process. Logged once, names the config key, and states the honest conclusion: the gap for
|
||||
* member panes is UNKNOWN, not clean, so {@code memberCredentials} cannot be verified from this
|
||||
* daemon. The one count it does report is scoped explicitly to fleetd's own environment, never
|
||||
* presented as if it said anything about the member's — see {@link #logCredentialGap}'s javadoc
|
||||
* for why this branch exists.
|
||||
*/
|
||||
private void warnUnknownMemberEnvironment(FleetConfig.MemberCredentials creds) {
|
||||
if (!unknownMemberEnvironmentWarned.compareAndSet(false, true)) {
|
||||
@@ -1598,13 +1772,13 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
.filter(name -> CREDENTIAL_SHAPED_NAME.matcher(name).matches())
|
||||
.filter(name -> !covered.contains(name))
|
||||
.count();
|
||||
log.warn("memberCredentials gap: memberHerdrSocket is configured, so member panes run under "
|
||||
+ "a different OS user than fleetd's own process, with a different environment "
|
||||
+ "entirely — fleetd has no channel to read that user's environment. {} of the "
|
||||
+ "{} names in fleetd's OWN environment are credential-shaped and not on "
|
||||
+ "known:/allow:, but that count describes fleetd's process, not the member "
|
||||
+ "herdr's. The credential gap for member panes is UNKNOWN, not clean, and "
|
||||
+ "memberCredentials cannot be verified from here.",
|
||||
log.warn("memberCredentials gap: memberHerdrSocket is configured, so member panes are routed "
|
||||
+ "to a second herdr — fleetd has no channel to confirm what OS user that herdr "
|
||||
+ "runs as, so it cannot tell whether those panes inherit its own environment or "
|
||||
+ "a different one entirely. {} of the {} names in fleetd's OWN environment are "
|
||||
+ "credential-shaped and not on known:/allow:, but that count describes fleetd's "
|
||||
+ "process, not the member herdr's. The credential gap for member panes is "
|
||||
+ "UNKNOWN, not clean, and memberCredentials cannot be verified from here.",
|
||||
gapInFleetdsOwnEnv, hostNames.size());
|
||||
}
|
||||
|
||||
@@ -1672,14 +1846,21 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
List<String> blankedByScrub = gap.stream()
|
||||
.filter(name -> !MemberEnvAllowList.keeps(effectiveAllowed, name))
|
||||
.toList();
|
||||
if (!keptByDerivedList.isEmpty() && unprotectedGapLogged.compareAndSet(false, true)) {
|
||||
// fleetd #341: filter to names this guard has never warned about before — not just
|
||||
// "isEmpty" on the whole branch — so a name this spawn's gap shares with an EARLIER
|
||||
// spawn's (already-warned) gap does not re-print, while a name unique to THIS gap still
|
||||
// does, whichever of the two WARN branches reported it first.
|
||||
List<String> newlyUnprotected = keptByDerivedList.stream()
|
||||
.filter(unprotectedGapNamesWarned::add)
|
||||
.toList();
|
||||
if (!newlyUnprotected.isEmpty()) {
|
||||
log.warn("memberCredentials gap: {} credential-shaped env var name(s) are on neither "
|
||||
+ "known: nor allow: — the derived allow-list keeps them anyway (a profile's "
|
||||
+ "gitTokenEnv/gitHostEnv/tokenEnv/env: names one, or this spawn injects it), "
|
||||
+ "so every member pane inherits them UNBLOCKED — {}. Add each to "
|
||||
+ "memberCredentials.known (or .allow if a member legitimately needs it), or "
|
||||
+ "remove it from whatever profile setting derives it in.",
|
||||
keptByDerivedList.size(), keptByDerivedList);
|
||||
newlyUnprotected.size(), newlyUnprotected);
|
||||
}
|
||||
if (!blankedByScrub.isEmpty() && allowListGapLogged.compareAndSet(false, true)) {
|
||||
log.info("memberCredentials gap: {} credential-shaped env var name(s) are on neither "
|
||||
@@ -1692,12 +1873,18 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
|
||||
/** The deny-by-default (and allow-list non-zsh fallback) WARN — unchanged byte-for-byte by #192. */
|
||||
private void warnGapUnprotected(List<String> gap) {
|
||||
if (unprotectedGapLogged.compareAndSet(false, true)) {
|
||||
// fleetd #341: same "only the names never warned before" filter as the sibling branch in
|
||||
// logCredentialGap above — see unprotectedGapNamesWarned's javadoc. Both branches share
|
||||
// this one guard because both report the exact same fact (a name reaching a member pane
|
||||
// unprotected) at the exact same severity; keying it by name is what lets a later spawn's
|
||||
// DIFFERENT name still get its own WARN after an earlier spawn's already fired.
|
||||
List<String> newlyUnprotected = gap.stream().filter(unprotectedGapNamesWarned::add).toList();
|
||||
if (!newlyUnprotected.isEmpty()) {
|
||||
log.warn("memberCredentials gap: {} credential-shaped env var name(s) are on neither "
|
||||
+ "known: nor allow: — every member pane inherits them UNBLOCKED — {}. "
|
||||
+ "Add each to memberCredentials.known (blocked by default) or .allow "
|
||||
+ "(if a member legitimately needs it).",
|
||||
gap.size(), gap);
|
||||
newlyUnprotected.size(), newlyUnprotected);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* fleetd #111 (CB-608): a single, testable read of the {@code memberCredentials:} policy — names
|
||||
* and counts only, never a value. The daemon never holds a credential's <em>value</em> in the
|
||||
* first place (only the names configured under {@code known:}/{@code allow:}), so there is
|
||||
* nothing here to redact by construction; the point of this class is that it is the ONE place
|
||||
* that turns a policy into names-and-counts, so nothing else hand-counts a second time.
|
||||
*
|
||||
* <p>Before this class, {@link dev.ltms.fleet.Fleetd#reportMemberCredentialsGap} computed these
|
||||
* same counts inline for the startup log line, and {@code scripts/probe-member-credentials.sh}
|
||||
* carried its own hardcoded {@code NAMES} array that the live policy could grow past silently
|
||||
* (#111) — the exact "hand-maintained second copy drifts" shape #114 fixed for the tool
|
||||
* catalogue. Both now read this class: the startup log via {@link
|
||||
* dev.ltms.fleet.Fleetd#reportMemberCredentialsGap}, and a live daemon via the {@code
|
||||
* GET /member-credentials} REST endpoint ({@link dev.ltms.fleet.rest.FleetApp}), which the probe
|
||||
* script fetches instead of carrying its own list.
|
||||
*
|
||||
* @param present policy configured with at least one {@code known} name. {@code false} for an
|
||||
* absent or empty {@code memberCredentials:} block — represented honestly as "no
|
||||
* policy", never as "nothing blocked" (an empty {@link #blocked} could otherwise be
|
||||
* misread as a clean bill of health).
|
||||
* @param policy the normalized policy mode ({@link FleetConfig.MemberCredentials#policy()}), or
|
||||
* {@code null} when {@link #present} is {@code false}.
|
||||
* @param known every name the policy declares, in configured order. Names only, never a value.
|
||||
* @param allowed the subset of {@link #known} explicitly let through. Names only.
|
||||
* @param blocked {@link #known} minus {@link #allowed} — the names an actual spawn shadows. Names
|
||||
* only.
|
||||
*/
|
||||
public record MemberCredentialPolicyView(boolean present, String policy, List<String> known,
|
||||
List<String> allowed, List<String> blocked) {
|
||||
|
||||
private static final MemberCredentialPolicyView ABSENT =
|
||||
new MemberCredentialPolicyView(false, null, List.of(), List.of(), List.of());
|
||||
|
||||
public MemberCredentialPolicyView {
|
||||
known = known == null ? List.of() : List.copyOf(known);
|
||||
allowed = allowed == null ? List.of() : List.copyOf(allowed);
|
||||
blocked = blocked == null ? List.of() : List.copyOf(blocked);
|
||||
}
|
||||
|
||||
/** The honest "no policy configured" view. */
|
||||
public static MemberCredentialPolicyView absent() {
|
||||
return ABSENT;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the view straight from the live config. {@code creds} may be {@code null} (no {@code
|
||||
* memberCredentials:} block at all) — treated the same as a present-but-empty block, exactly
|
||||
* like {@link dev.ltms.fleet.Fleetd#reportMemberCredentialsGap} already did.
|
||||
*/
|
||||
public static MemberCredentialPolicyView of(FleetConfig.MemberCredentials creds) {
|
||||
if (creds == null || creds.known().isEmpty()) {
|
||||
return ABSENT;
|
||||
}
|
||||
return new MemberCredentialPolicyView(true, creds.policy(), creds.known(), creds.allow(),
|
||||
List.copyOf(creds.blockedSet()));
|
||||
}
|
||||
|
||||
public int knownCount() {
|
||||
return known.size();
|
||||
}
|
||||
|
||||
public int allowedCount() {
|
||||
return allowed.size();
|
||||
}
|
||||
|
||||
public int blockedCount() {
|
||||
return blocked.size();
|
||||
}
|
||||
}
|
||||
@@ -34,7 +34,7 @@ import java.util.TreeSet;
|
||||
*
|
||||
* <p>{@code SSH_AUTH_SOCK} is deliberately NOT here. It is a handle to the operator's ssh-agent — a
|
||||
* member holding it can sign with the operator's keys — so keeping it is a config decision
|
||||
* ({@code memberCredentials.sshAuthSock: allow}), not a derivation default.
|
||||
* ({@code memberCredentials.sshAgentEnv: inherit}), not a derivation default.
|
||||
*
|
||||
* <p><b>CB-633 follow-up:</b> the union also includes {@code memberCredentials.allow:} — the
|
||||
* operator's own explicit list. Before this, {@code policy: allow-list} silently ignored every name
|
||||
@@ -42,7 +42,7 @@ import java.util.TreeSet;
|
||||
* turning the policy on could blank credentials working members already depended on. {@code
|
||||
* SSH_AUTH_SOCK} and configured broker URI environment names are exceptions: even when the operator
|
||||
* lists them under {@code allow:}, they are excluded here. {@code SSH_AUTH_SOCK} is added back ONLY
|
||||
* by the caller when {@code sshAuthSock: allow} is explicitly set
|
||||
* by the caller when {@code sshAgentEnv: inherit} is explicitly set
|
||||
* (see {@link #SSH_AUTH_SOCK}'s javadoc) — it is a live handle to the operator's own ssh-agent, not
|
||||
* a value, so treating it like any other allow-listed name would hand a member every key the
|
||||
* operator's agent holds the moment they typed the name under {@code allow:} for an unrelated
|
||||
@@ -53,7 +53,7 @@ public final class MemberEnvAllowList {
|
||||
/**
|
||||
* The operator's ssh-agent socket path. Deliberately excluded from {@link #derive}'s union of
|
||||
* {@code memberCredentials.allow:} — see the class javadoc's CB-633 follow-up note. Governed
|
||||
* ONLY by {@code memberCredentials.sshAuthSock}, never by appearing in {@code allow:}.
|
||||
* ONLY by {@code memberCredentials.sshAgentEnv}, never by appearing in {@code allow:}.
|
||||
*/
|
||||
public static final String SSH_AUTH_SOCK = "SSH_AUTH_SOCK";
|
||||
|
||||
|
||||
@@ -21,6 +21,7 @@ import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.BooleanSupplier;
|
||||
@@ -669,17 +670,44 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
private final AtomicBoolean discoveryUnavailableWarned =
|
||||
new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* fleetd #267: one WARN per PROFILE (not per launcher instance — several profiles can each hit
|
||||
* this gap independently) for the model-mismatch check (fleetd #175) never getting to run
|
||||
* because the spawn was not given a fleetd-provisioned worktree (fleetd #249). Profile names
|
||||
* accumulate here for the life of this launcher instance and are never removed — the same
|
||||
* one-shot treatment {@link #discoveryUnavailableWarned} already gets, just keyed per profile
|
||||
* instead of globally.
|
||||
*/
|
||||
private final Set<String> modelCheckSkippedWarned = ConcurrentHashMap.newKeySet();
|
||||
|
||||
/** Add lazy on-disk session discovery to the base handle. */
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
String cwd = effectiveCwd(req);
|
||||
// fleetd #249: refuse rather than silently resume into unverifiable territory. opencode's
|
||||
// `-s <id>` flag itself resumes precisely — the resolved id is what fails, not the resume —
|
||||
// but resolvedSessionId() below can never confirm (or later re-report) this handle's own
|
||||
// identity without a fleetd-provisioned worktree (isProvisionedWorktree(cwd)), because the
|
||||
// directory is shared and sessionIdForDirectory's "most recently updated row" heuristic can
|
||||
// pick a sibling's session. Refusing here, before anything spawns, beats letting the member
|
||||
// start and only then discovering fleetd can never again verify who it actually is.
|
||||
if (req.resumeSessionId() != null && !req.resumeSessionId().isBlank()
|
||||
&& !isProvisionedWorktree(cwd)) {
|
||||
throw new IllegalArgumentException("resumeSessionId requires a fleetd-provisioned "
|
||||
+ "worktree for an opencode profile — without one, this member's cwd is shared "
|
||||
+ "with other sessions, so fleetd can never reliably confirm (now or later) which "
|
||||
+ "conversation it is actually running (fleetd #249). Pass fleet_spawn{worktree:"
|
||||
+ "<ticket-slug>} to resume this member.");
|
||||
}
|
||||
PeerHandle inner = super.spawn(req);
|
||||
// fleetd #175: the same profile config buildLaunch resolved for this spawn (requireProfile
|
||||
// is deterministic on req.profileName(), so re-resolving here costs a map lookup, not a
|
||||
// second decision) — SessionAwareHandle needs cfg.model() to know what THIS session should
|
||||
// be running.
|
||||
FleetConfig.Profile cfg = requireProfile(req.profileName());
|
||||
return new SessionAwareHandle(inner, discovery, effectiveCwd(req), cfg,
|
||||
this::memberHerdrSocketConfigured, discoveryUnavailableWarned, exhaustionSink);
|
||||
return new SessionAwareHandle(inner, discovery, cwd, cfg,
|
||||
this::memberHerdrSocketConfigured, discoveryUnavailableWarned,
|
||||
modelCheckSkippedWarned, exhaustionSink);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -704,6 +732,7 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
private final FleetConfig.Profile cfg;
|
||||
private final BooleanSupplier discoveryUnavailable;
|
||||
private final AtomicBoolean discoveryUnavailableWarned;
|
||||
private final Set<String> modelCheckSkippedWarned;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
/** CAS'd true the first (and only) time a model mismatch is reported for this handle. */
|
||||
private final AtomicBoolean modelMismatchReported = new AtomicBoolean();
|
||||
@@ -718,11 +747,23 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
* the directory right now."
|
||||
*/
|
||||
private final AtomicReference<String> resolvedSessionId = new AtomicReference<>();
|
||||
/**
|
||||
* fleetd #249: whether {@link #cwd} is a fleetd-provisioned git worktree
|
||||
* ({@link HerdrPeerLauncher#isProvisionedWorktree}), computed once at spawn time since
|
||||
* {@code cwd} never changes for this handle. When {@code false} the directory is shared
|
||||
* with other sessions (the default no-worktree spawn inherits the lead's own cwd), so
|
||||
* {@link OpenCodeSessionDiscovery#sessionIdForDirectory}'s "most recently updated row for
|
||||
* this directory" heuristic can and does pick another session's row — see that class's
|
||||
* javadoc. {@link #agentSessionId()} refuses to guess in that case: it reports absent
|
||||
* rather than a possibly-foreign id.
|
||||
*/
|
||||
private final boolean worktreeProvisioned;
|
||||
|
||||
SessionAwareHandle(PeerHandle delegate, OpenCodeSessionDiscovery discovery, String cwd,
|
||||
FleetConfig.Profile cfg,
|
||||
BooleanSupplier discoveryUnavailable,
|
||||
AtomicBoolean discoveryUnavailableWarned,
|
||||
Set<String> modelCheckSkippedWarned,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this.delegate = delegate;
|
||||
this.discovery = discovery;
|
||||
@@ -730,7 +771,9 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
this.cfg = cfg;
|
||||
this.discoveryUnavailable = discoveryUnavailable;
|
||||
this.discoveryUnavailableWarned = discoveryUnavailableWarned;
|
||||
this.modelCheckSkippedWarned = modelCheckSkippedWarned;
|
||||
this.exhaustionSink = exhaustionSink;
|
||||
this.worktreeProvisioned = isProvisionedWorktree(cwd);
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -760,6 +803,9 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
// built from (see OpenCodeLauncher#defaultDiscoveryRoot's javadoc for the full
|
||||
// reasoning). Scanning fleetd's own $HOME under that config would only ever find "no
|
||||
// row" and read as "resume unsupported" — declare it unavailable instead, once, loudly.
|
||||
// Checked before the fleetd #249 worktree gate below: this OS-user mismatch makes
|
||||
// discovery unusable regardless of whether cwd happens to be a provisioned worktree, so
|
||||
// it earns the one-time WARN either way.
|
||||
if (discoveryUnavailable.getAsBoolean()) {
|
||||
if (discoveryUnavailableWarned.compareAndSet(false, true)) {
|
||||
log.warn("opencode session discovery unavailable: memberHerdrSocket is "
|
||||
@@ -771,6 +817,36 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
}
|
||||
return null;
|
||||
}
|
||||
// fleetd #249: cwd is shared with other sessions unless fleetd itself provisioned this
|
||||
// worktree, and sessionIdForDirectory's directory-keyed heuristic cannot tell this
|
||||
// member's row apart from a sibling's in that case (measured: a three-day-old row from
|
||||
// a different profile). Refuse to guess — absent is the honest answer, and it is what
|
||||
// this codebase already returns elsewhere for absent evidence (fleetd #175's UNKNOWN).
|
||||
// This IS the ordinary, expected shape of the large majority of spawns (no worktree
|
||||
// requested), not a configuration gap — but fleetd #267 found that same shape silently
|
||||
// switches off the fleetd #175 model-mismatch check for those spawns too, since
|
||||
// checkModelMatch's only call site is right below this gate. The check cannot be moved
|
||||
// off agentSessionId()'s resolved id: the id is the only safe way to key
|
||||
// actualModelForSessionId to THIS session's own row rather than "whatever is newest in
|
||||
// the shared directory" (fleetd #234) — re-deriving a second, independent answer via
|
||||
// `directory` here would reintroduce exactly the false-positive risk #234 fixed (a
|
||||
// sibling's differently-configured model looking like THIS profile's mismatch). So the
|
||||
// model genuinely is unknowable without a provisioned worktree, and unlike the silence
|
||||
// this branch used to keep, that gap now gets the same one-time, per-profile WARN
|
||||
// treatment discoveryUnavailable already gets above — but keyed by profile, since
|
||||
// several profiles can each hit this independently.
|
||||
if (!worktreeProvisioned) {
|
||||
if (cfg.model() != null && !cfg.model().isBlank()
|
||||
&& modelCheckSkippedWarned.add(cfg.profile())) {
|
||||
log.warn("opencode model-mismatch check (fleetd #175) cannot run for profile "
|
||||
+ "'{}': it was spawned without a fleetd-provisioned worktree (fleetd "
|
||||
+ "#249), so its cwd may be shared with other sessions and the actual "
|
||||
+ "model it is running cannot be safely told apart from a sibling's — "
|
||||
+ "spawn with worktree:true to enable the check for this profile.",
|
||||
cfg.profile());
|
||||
}
|
||||
return null;
|
||||
}
|
||||
// fleetd #234: once resolved, stay resolved. Re-deriving from `directory` on every call
|
||||
// would let this handle's identity drift to a sibling session that later shares the
|
||||
// same cwd and writes a newer row — see resolvedSessionId's javadoc.
|
||||
|
||||
@@ -56,10 +56,13 @@ public final class FleetMetrics {
|
||||
m.describe(REPLIES, "counter",
|
||||
"Worker replies by delivery path (rendezvous=resolved an open send, inbox=stranded and held).");
|
||||
m.describe(PUSH_NUDGES, "counter",
|
||||
"CB-307 push-loop nudges to the primary (delivered|exhausted).");
|
||||
"CB-307 push-loop nudges to the primary (sent|exhausted). fleetd #365: \"sent\" means "
|
||||
+ "the herdr paste-and-submit call succeeded, not that the pane read it — this "
|
||||
+ "layer has no read-receipt concept.");
|
||||
m.describe(HEARTBEAT_NUDGES, "counter",
|
||||
"CB-551 idle-lead heartbeat nudges (delivered|failed|exhausted). Quiet-cap exhaustion "
|
||||
+ "means the lead idled with nothing pending and was told to stand down.");
|
||||
"CB-551 idle-lead heartbeat nudges (sent|failed|exhausted). Quiet-cap exhaustion "
|
||||
+ "means the lead idled with nothing pending and was told to stand down. "
|
||||
+ "fleetd #365: \"sent\" means the herdr call succeeded, not that the lead read it.");
|
||||
m.describe(SPAWNS, "counter",
|
||||
"Worker spawn attempts by peer kind and outcome (ready|timeout|guard_rejected).");
|
||||
m.describe(HERDR_CALLS, "counter",
|
||||
|
||||
@@ -8,6 +8,7 @@ import com.rabbitmq.client.DeliverCallback;
|
||||
import com.rabbitmq.client.Recoverable;
|
||||
import com.rabbitmq.client.RecoveryListener;
|
||||
import com.rabbitmq.client.Return;
|
||||
import com.rabbitmq.client.impl.DefaultExceptionHandler;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -22,6 +23,7 @@ import java.util.concurrent.ConcurrentSkipListMap;
|
||||
import java.util.concurrent.ExecutionException;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.TimeoutException;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
/**
|
||||
* AMQP-backed {@link ReplyInbox} (CB-307 Stage 2): genuine cross-restart durability behind the same
|
||||
@@ -93,8 +95,23 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
private final Channel channel;
|
||||
/** All channel operations (publish/declare/ack/cancel) serialize on this — a Channel is not thread-safe. */
|
||||
private final Object channelLock = new Object();
|
||||
/** target → (msgId → held delivery). Per-target map is guarded by synchronizing on itself. */
|
||||
/**
|
||||
* target → (msgId → held delivery). Per-target map is guarded by synchronizing on itself.
|
||||
*
|
||||
* <p><strong>CB-318 tombstone.</strong> The value {@link #RELEASED} is a reserved sentinel: it
|
||||
* marks a target whose {@link #release} has already run, so {@link #deliverCallback} can tell a
|
||||
* delivery landing after release() apart from a fresh target it has never seen. See both methods'
|
||||
* javadoc for why a plain {@code held.remove(target)} is not enough.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, LinkedHashMap<String, Held>> held = new ConcurrentHashMap<>();
|
||||
|
||||
/**
|
||||
* CB-318 sentinel stored in {@link #held} for a target whose {@link #release} has already run.
|
||||
* Never mutated — every read site compares it by reference ({@code ==}) before touching it as a
|
||||
* map, because it is a single object shared across every released target and calling a mutator on
|
||||
* it would corrupt state for all of them.
|
||||
*/
|
||||
private static final LinkedHashMap<String, Held> RELEASED = new LinkedHashMap<>();
|
||||
/** Targets whose queue is declared and consumer is running, mapped to their broker consumer tag. */
|
||||
private final ConcurrentHashMap<String, String> consumerTags = new ConcurrentHashMap<>();
|
||||
|
||||
@@ -137,17 +154,22 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
/** As {@link #open(String)}, with an explicit consumer prefetch (CB-527: caps the held backlog per target). */
|
||||
public static AmqpReplyInbox open(String uri, int prefetch) {
|
||||
try {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares queues and re-attaches consumers.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
return new AmqpReplyInbox(factory.newConnection("fleetd-reply-inbox"), prefetch);
|
||||
return new AmqpReplyInbox(connectionFactory(uri).newConnection(AmqpConnectionFailureLogger.REPLY_INBOX), prefetch);
|
||||
} catch (Exception e) {
|
||||
throw new IllegalStateException("cannot connect to AMQP broker at " + uri, e);
|
||||
}
|
||||
}
|
||||
|
||||
static ConnectionFactory connectionFactory(String uri) throws Exception {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares queues and re-attaches consumers.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
factory.setExceptionHandler(new AmqpConnectionFailureLogger(AmqpConnectionFailureLogger.REPLY_INBOX, log));
|
||||
return factory;
|
||||
}
|
||||
|
||||
/** Wrap an already-open connection with {@link #DEFAULT_PREFETCH} (injection seam for the contract test). */
|
||||
AmqpReplyInbox(Connection connection) {
|
||||
this(connection, DEFAULT_PREFETCH);
|
||||
@@ -201,6 +223,12 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
channel.queueDeclare(queue, true, false, false, null); // durable, non-exclusive, keep on idle
|
||||
String tag = channel.basicConsume(queue, false, deliverCallback(target), _ -> { });
|
||||
consumerTags.put(target, tag);
|
||||
// CB-318: drop a stale RELEASED tombstone from a prior ownership of this same target
|
||||
// string, so a delivery under this fresh consumer is held normally instead of being
|
||||
// nacked forever by deliverCallback's RELEASED check. Safe to do here, still under
|
||||
// channelLock: no delivery for the consumer tag just registered above can reach
|
||||
// deliverCallback before this basicConsume call returns.
|
||||
held.remove(target, RELEASED);
|
||||
log.debug("AMQP inbox owns queue {} for target {}", queue, target);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot own queue " + queue, e);
|
||||
@@ -208,18 +236,103 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Release ownership of {@code target}: cancel its consumer, then nack-with-requeue every
|
||||
* delivery still held for it instead of just dropping the local record.
|
||||
*
|
||||
* <p><strong>Cancelling a consumer does not requeue its in-flight deliveries.</strong> In AMQP,
|
||||
* a delivery that was pushed to a consumer stays unacked, attached to the still-open
|
||||
* {@link #channel}, until that channel or the connection closes — {@code basicCancel} alone does
|
||||
* neither. So before this method existed with a requeue step, it dropped {@link #held}'s entries
|
||||
* for {@code target} while the broker still considered them outstanding: never acked, never
|
||||
* nacked, never requeued, and no longer reachable by {@link #peek} — permanently invisible. This
|
||||
* is unlike {@link #handleRecovery} and {@link #close()}, whose bare {@code held.clear()} is
|
||||
* correct because each has already made the broker requeue (a real connection drop, or
|
||||
* {@code channel.close()} respectively) before clearing local state.
|
||||
*
|
||||
* <p><strong>Order: cancel first, then nack.</strong> A delivery tag stays valid for
|
||||
* {@code basicNack} on this channel regardless of whether its consumer is still attached — only
|
||||
* a channel/connection close invalidates it — so cancelling {@code target}'s consumer first does
|
||||
* not risk the tags. Doing it the other way round does: nacking a delivery with {@code requeue}
|
||||
* while its consumer is still active hands the message straight back to that <em>same</em>
|
||||
* consumer the instant a prefetch slot frees up (confirmed against a real broker — see
|
||||
* {@code AmqpReplyInboxContractTest.releaseCancelsConsumerAndRequeuesHeldDeliveryForRecovery}),
|
||||
* which races this method's own {@code held.remove(target)}: the redelivery can land after the
|
||||
* clear and leave a stale entry behind, so {@link #peek} is no longer reliably empty right after
|
||||
* {@link #release}. Cancelling first closes that consumer, so the requeued message goes back to
|
||||
* the queue for whichever consumer picks it up next (a later {@link #own}), not this one.
|
||||
*
|
||||
* <p><strong>Failure of the requeue is best-effort, not fatal.</strong> {@link #release} runs
|
||||
* during teardown ({@code Fleetd} calls it right after {@code MessageService.abandon}), and a
|
||||
* throw here would abort cleanups the caller depends on — the same argument fleetd #293 settled
|
||||
* for {@code HerdrPeerLauncher.stop()}'s tab-close step. So a failed {@code basicNack} is logged
|
||||
* at WARN, naming the target and delivery tag that leaked, and release proceeds; the delivery
|
||||
* stays unacked on the broker rather than being silently dropped, so it is still recoverable by a
|
||||
* later connection drop even though this release did not manage to requeue it immediately. A
|
||||
* failed {@code basicCancel} still throws, unchanged from before this fix — that failure means
|
||||
* the consumer may still be attached, so best-effort requeue is not attempted underneath it.
|
||||
*
|
||||
* <p><strong>CB-318: {@code held.remove(target)} alone leaves a second window open.</strong> The
|
||||
* bullet above already explains why cancelling first does not save a tag from going stale — but
|
||||
* that only accounts for a delivery landing before this method starts touching {@link #held}.
|
||||
* {@code basicCancel} stops <em>new</em> dispatches; it does not flush one already handed to the
|
||||
* consumer work pool. So a delivery can still land on that pool's thread and reach
|
||||
* {@link #deliverCallback} at any point during, or after, this method's body — and a plain
|
||||
* {@code held.remove(target)} does nothing to stop it: {@code deliverCallback}'s
|
||||
* {@code computeIfAbsent} finds the key gone and happily creates a brand-new map under it, which
|
||||
* this method — already past its {@code remove} — never looks at again. That entry then sits
|
||||
* delivered-but-unacked on {@link #channel} until the whole inbox closes: never requeued, never
|
||||
* redelivered, and {@link #peek} is never called again for a target nothing owns any more.
|
||||
*
|
||||
* <p>The fix is {@link #held}{@code .compute(target, ...)} instead of {@code remove}: it takes
|
||||
* whatever was held (to nack, same as before) and, in the same atomic step, leaves the
|
||||
* {@link #RELEASED} tombstone behind instead of an absent key. {@code computeIfAbsent} and
|
||||
* {@code compute} calls for the same key are mutually exclusive in {@link ConcurrentHashMap} —
|
||||
* whichever of this call and a concurrent {@code deliverCallback} runs first is fully visible to
|
||||
* the other, with no gap between them. So a delivery that loses the race sees a real map here and
|
||||
* gets nacked by the loop below, same as always; a delivery that wins the race (runs first) is
|
||||
* itself nacked by that same loop, once it settles into {@code held}. A delivery that arrives once
|
||||
* this method has stored {@link #RELEASED} finds it via {@code computeIfAbsent} and refuses itself
|
||||
* — see {@link #deliverCallback}. Either way nothing is silently retained forever, satisfying the
|
||||
* ticket's invariant against dropping a message. This closes the window rather than merely
|
||||
* narrowing it — correctness does not depend on how much time elapses between the swap and this
|
||||
* method returning.
|
||||
*/
|
||||
@Override
|
||||
public void release(String target) {
|
||||
synchronized (channelLock) {
|
||||
String tag = consumerTags.remove(target);
|
||||
held.remove(target); // stale delivery tags must not survive release
|
||||
if (tag == null) {
|
||||
return;
|
||||
if (tag != null) {
|
||||
try {
|
||||
channel.basicCancel(tag);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot cancel consumer for " + target, e);
|
||||
}
|
||||
}
|
||||
try {
|
||||
channel.basicCancel(tag);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot cancel consumer for " + target, e);
|
||||
AtomicReference<LinkedHashMap<String, Held>> previouslyHeld = new AtomicReference<>();
|
||||
held.compute(target, (_, v) -> {
|
||||
previouslyHeld.set(v);
|
||||
return RELEASED;
|
||||
});
|
||||
var perTarget = previouslyHeld.get();
|
||||
if (perTarget != null && perTarget != RELEASED) {
|
||||
synchronized (perTarget) {
|
||||
for (Held h : perTarget.values()) {
|
||||
try {
|
||||
channel.basicNack(h.deliveryTag(), false, true); // requeue, don't drop
|
||||
} catch (IOException | RuntimeException e) {
|
||||
// Caught broadly (not just IOException) for the same reason #293 catches
|
||||
// RuntimeException in HerdrPeerLauncher.stop(): best-effort teardown must
|
||||
// not be guarded only against the expected failure and bare against any
|
||||
// other. The message stays unacked on the broker either way — not lost,
|
||||
// just not proactively requeued — until a connection drop frees it.
|
||||
log.warn("release({}): could not requeue held delivery (msgId={}, tag={})"
|
||||
+ " back to the broker — it stays unacked until a connection"
|
||||
+ " drop frees it: {}",
|
||||
target, h.message().msgId(), h.deliveryTag(), e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -272,7 +385,7 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
if (perTarget == null || perTarget == RELEASED) {
|
||||
return List.of();
|
||||
}
|
||||
synchronized (perTarget) {
|
||||
@@ -283,7 +396,7 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
if (perTarget == null || perTarget == RELEASED) {
|
||||
return;
|
||||
}
|
||||
Held h;
|
||||
@@ -316,6 +429,20 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
String content = new String(delivery.getBody(), StandardCharsets.UTF_8);
|
||||
var perTarget = held.computeIfAbsent(target, _ -> new LinkedHashMap<>());
|
||||
if (perTarget == RELEASED) {
|
||||
// CB-318: release() already ran for this target and left the RELEASED tombstone in
|
||||
// held (see release()'s javadoc) — computeIfAbsent() is guaranteed to see it rather
|
||||
// than recreate a fresh map, because ConcurrentHashMap serializes compute/
|
||||
// computeIfAbsent calls for the same key against each other. Refuse the delivery
|
||||
// instead of holding it somewhere release() will never look at again: requeue it, the
|
||||
// same way release() nacks its own held entries, so a later owner (or a connection
|
||||
// drop) can still recover it. This does not need channelLock across a broker round
|
||||
// trip — basicNack, like the duplicate-ack case just below, does not wait for one.
|
||||
synchronized (channelLock) {
|
||||
channel.basicNack(tag, false, true);
|
||||
}
|
||||
return;
|
||||
}
|
||||
boolean duplicate;
|
||||
synchronized (perTarget) {
|
||||
if (perTarget.containsKey(msgId)) {
|
||||
@@ -450,3 +577,44 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Keeps RabbitMQ's forgiving exception behaviour while adding the connection identity that its
|
||||
* default logger drops. Package-private so both AMQP connections use the same two names.
|
||||
*/
|
||||
final class AmqpConnectionFailureLogger extends DefaultExceptionHandler {
|
||||
|
||||
static final String REPLY_INBOX = "fleetd-reply-inbox";
|
||||
static final String LEAD_MAILBOX = "fleetd-lead-mailbox";
|
||||
|
||||
private final String connectionName;
|
||||
private final Logger logger;
|
||||
|
||||
AmqpConnectionFailureLogger(String connectionName, Logger logger) {
|
||||
this.connectionName = connectionName;
|
||||
this.logger = logger;
|
||||
}
|
||||
|
||||
String connectionName() {
|
||||
return connectionName;
|
||||
}
|
||||
|
||||
@Override
|
||||
protected void log(String message, Throwable cause) {
|
||||
if (isSocketClosedOrConnectionReset(cause)) {
|
||||
logger.warn("AMQP connection {}: {} (Exception message: {})", connectionName, message, cause.getMessage());
|
||||
} else {
|
||||
logger.error("AMQP connection {}: {}", connectionName, message, cause);
|
||||
}
|
||||
}
|
||||
|
||||
private static boolean isSocketClosedOrConnectionReset(Throwable cause) {
|
||||
// Deliberate copy of ForgivingExceptionHandler's private static helper; check it on amqp-client upgrades.
|
||||
if (!(cause instanceof IOException)) {
|
||||
return false;
|
||||
}
|
||||
return "Connection reset".equals(cause.getMessage())
|
||||
|| "Socket closed".equals(cause.getMessage())
|
||||
|| "Connection reset by peer".equals(cause.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,8 +4,8 @@ import java.util.List;
|
||||
|
||||
/**
|
||||
* The lead-to-lead message channel this daemon speaks, as its callers need it — one lead's own
|
||||
* mailbox: publish to a peer's coord-id, look at what has arrived for me, and ack what I have
|
||||
* delivered.
|
||||
* mailbox: publish to a peer's coord-id, look at what has arrived for me, ack what I have
|
||||
* delivered, and (fleetd #361) inspect any coord-id's mailbox from the outside without owning it.
|
||||
*
|
||||
* <p>Extracted from {@link LeadMailbox} purely as a seam. {@code LeadMailbox} is the one production
|
||||
* implementation and owns a live AMQP connection, so a test that wanted to exercise the routing in
|
||||
@@ -32,9 +32,87 @@ public interface LeadChannel {
|
||||
/** Non-destructive FIFO snapshot of the messages held for this daemon's own coord-id. */
|
||||
List<LeadMessage> peek();
|
||||
|
||||
/** Drop {@code msgId} from the held set and ack it on the broker. A no-op if it is not held. */
|
||||
/**
|
||||
* Drop {@code msgId} from the held set and ack it on the broker. A repeated ack that this
|
||||
* connection already completed may be a no-op. Any other unknown msgId must throw rather than
|
||||
* report an ack that did not reach the broker.
|
||||
*/
|
||||
void ack(String msgId);
|
||||
|
||||
/** This daemon's own lead coordination id — the mailbox it owns, and the {@code from} it sends as. */
|
||||
String selfCoordId();
|
||||
|
||||
/**
|
||||
* A non-destructive look at {@code coordId}'s mailbox — does it exist, how many messages are
|
||||
* waiting on it, and how many consumers are attached — without owning, consuming, or otherwise
|
||||
* changing it. {@code consumers == 0} on an existing mailbox is the observable form of "nobody
|
||||
* is reading this right now": a publish to it will sit queued rather than reach a pane.
|
||||
*
|
||||
* <p><strong>Never throws</strong> — this is a best-effort fact-finding call, not an operation a
|
||||
* caller must handle failing. But it must never turn "I could not check" into a false negative:
|
||||
* {@link MailboxState#absent(String)} means the broker positively confirmed there is no such
|
||||
* queue, and {@link MailboxState#unknown(String)} — a distinct value — means the look could not
|
||||
* be completed at all (broker unreachable, timed out, connection closed). A caller that
|
||||
* collapses those two into one, as fleetd #361 initially did, cannot tell "that peer is down"
|
||||
* from "I could not check", and a reader of {@code pending}/{@code consumers} cannot tell a
|
||||
* measured zero from a zero standing in for "not measured".
|
||||
*
|
||||
* <p><strong>Must never share fate with {@link #publish} or {@link #peek}/{@link #ack}.</strong>
|
||||
* fleetd #361: in AMQP 0-9-1 a passive queue declare of a queue that does not exist closes the
|
||||
* channel it was declared on with a 404. An implementation backed by a real broker connection
|
||||
* must inspect on a channel it can afford to lose — never the channel {@link #publish} or the
|
||||
* consume loop depends on — so that looking at a peer that happens to be down can never break
|
||||
* this daemon's own send or receive path.
|
||||
*/
|
||||
MailboxState inspect(String coordId);
|
||||
|
||||
/**
|
||||
* The result of {@link #inspect}. {@code presence} tells apart three states a caller must not
|
||||
* conflate: a confirmed-existing mailbox ({@link Presence#EXISTS}, the only case where
|
||||
* {@code pending}/{@code consumers} are measured facts), a confirmed-absent one
|
||||
* ({@link Presence#ABSENT} — the broker positively said "no such queue"), and one this call
|
||||
* simply could not determine ({@link Presence#UNKNOWN} — broker unreachable, timed out,
|
||||
* connection closed). {@code pending}/{@code consumers} are always {@code 0} and meaningless
|
||||
* outside {@link Presence#EXISTS}; a renderer must gate on {@link #exists()} (or {@code
|
||||
* presence} directly), never present them as measured otherwise.
|
||||
*
|
||||
* @param coordId the coord-id inspected
|
||||
* @param presence whether the mailbox is confirmed to exist, confirmed absent, or unknown
|
||||
* @param pending messages ready for delivery but not yet in a consumer's hands (0 unless EXISTS)
|
||||
* @param consumers how many consumers are attached (0 unless EXISTS)
|
||||
*/
|
||||
record MailboxState(String coordId, Presence presence, int pending, int consumers) {
|
||||
|
||||
/** Whether {@link #inspect} was able to reach a definite answer, of either kind. */
|
||||
public enum Presence { EXISTS, ABSENT, UNKNOWN }
|
||||
|
||||
/** {@code true} only when the broker confirmed this exact queue is currently declared. */
|
||||
public boolean exists() {
|
||||
return presence == Presence.EXISTS;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code true} when {@link #inspect} reached a definite answer (exists or confirmed
|
||||
* absent); {@code false} when it could not determine either way. A caller must never treat
|
||||
* {@code !known()} the same as a confirmed absence — the mailbox may well exist.
|
||||
*/
|
||||
public boolean known() {
|
||||
return presence != Presence.UNKNOWN;
|
||||
}
|
||||
|
||||
/** The broker confirmed this queue exists, with these measured counts. */
|
||||
public static MailboxState exists(String coordId, int pending, int consumers) {
|
||||
return new MailboxState(coordId, Presence.EXISTS, pending, consumers);
|
||||
}
|
||||
|
||||
/** The broker positively confirmed there is no such queue (e.g. a 404 on passive declare). */
|
||||
public static MailboxState absent(String coordId) {
|
||||
return new MailboxState(coordId, Presence.ABSENT, 0, 0);
|
||||
}
|
||||
|
||||
/** The look could not be completed — broker unreachable, timed out, or connection closed. */
|
||||
public static MailboxState unknown(String coordId) {
|
||||
return new MailboxState(coordId, Presence.UNKNOWN, 0, 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,6 +5,7 @@ import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
@@ -43,6 +44,9 @@ public final class LeadCoordLoop {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadCoordLoop.class);
|
||||
|
||||
/** A bounded window is enough: redelivery can only follow a recent failed ack or recovery. */
|
||||
private static final int RECENT_DELIVERY_LIMIT = 1_024;
|
||||
|
||||
/** How an arriving peer message is rendered into the lead's pane — the sender's coord-id, then its text. */
|
||||
static final String DELIVERY_FORMAT = "[lead %s] %s";
|
||||
|
||||
@@ -51,6 +55,8 @@ public final class LeadCoordLoop {
|
||||
private final Supplier<Map<String, String>> leads;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final long intervalMs;
|
||||
/** msgIds already written to the pane, so recovery redelivery is acked without another pane write. */
|
||||
private final LinkedHashMap<String, Boolean> delivered = new LinkedHashMap<>();
|
||||
|
||||
private volatile boolean running;
|
||||
|
||||
@@ -117,6 +123,13 @@ public final class LeadCoordLoop {
|
||||
if (held.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
LeadMessage msg = held.getFirst();
|
||||
if (wasDelivered(msg.msgId())) {
|
||||
// This lives here, rather than in LeadMailbox, because only this loop knows a pane write
|
||||
// happened. The mailbox only knows broker delivery tags and must still redeliver after a crash.
|
||||
ackDelivered(msg);
|
||||
return;
|
||||
}
|
||||
String lead = resolveLocalLead();
|
||||
if (lead == null) {
|
||||
// Left unacked on purpose: the broker keeps holding it until a lead pane exists.
|
||||
@@ -136,7 +149,6 @@ public final class LeadCoordLoop {
|
||||
lead, status, held.size());
|
||||
return;
|
||||
}
|
||||
LeadMessage msg = held.getFirst();
|
||||
try {
|
||||
agents.send(lead, DELIVERY_FORMAT.formatted(msg.from(), msg.content()));
|
||||
} catch (RuntimeException e) {
|
||||
@@ -145,6 +157,13 @@ public final class LeadCoordLoop {
|
||||
msg.msgId(), msg.from(), lead, e.toString());
|
||||
return;
|
||||
}
|
||||
rememberDelivered(msg.msgId());
|
||||
if (ackDelivered(msg)) {
|
||||
log.debug("lead coordination: delivered message {} from {} to lead {}", msg.msgId(), msg.from(), lead);
|
||||
}
|
||||
}
|
||||
|
||||
private boolean ackDelivered(LeadMessage msg) {
|
||||
try {
|
||||
channel.ack(msg.msgId());
|
||||
} catch (RuntimeException e) {
|
||||
@@ -152,9 +171,24 @@ public final class LeadCoordLoop {
|
||||
// deliberate direction of this trade.
|
||||
log.warn("lead coordination: delivered message {} but could not ack it: {}",
|
||||
msg.msgId(), e.toString());
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
private boolean wasDelivered(String msgId) {
|
||||
synchronized (delivered) {
|
||||
return delivered.containsKey(msgId);
|
||||
}
|
||||
}
|
||||
|
||||
private void rememberDelivered(String msgId) {
|
||||
synchronized (delivered) {
|
||||
delivered.put(msgId, Boolean.TRUE);
|
||||
if (delivered.size() > RECENT_DELIVERY_LIMIT) {
|
||||
delivered.remove(delivered.keySet().iterator().next());
|
||||
}
|
||||
}
|
||||
log.debug("lead coordination: delivered message {} from {} to lead {}", msg.msgId(), msg.from(), lead);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -96,7 +96,14 @@ public final class LeadHeartbeatLoop {
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Count one nudge outcome when a registry is wired; a no-op in unit tests. */
|
||||
/**
|
||||
* Count one nudge outcome when a registry is wired; a no-op in unit tests.
|
||||
*
|
||||
* <p>fleetd #365: the {@code "sent"} outcome (renamed from {@code "delivered"}) records only
|
||||
* that {@link #injectNudge} — a one-way herdr {@code agent.prompt} paste-and-submit — returned
|
||||
* without throwing, not that the lead's pane actually read or acted on the text. This layer has
|
||||
* no read-receipt concept, so "sent" is the honest word for what this call can ever establish.
|
||||
*/
|
||||
private void countNudge(String outcome) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(FleetMetrics.HEARTBEAT_NUDGES, "outcome", outcome);
|
||||
@@ -245,7 +252,7 @@ public final class LeadHeartbeatLoop {
|
||||
agents.send(leadTerminal, fleet.nudgeText());
|
||||
log.debug("idle-heartbeat: nudge sent to lead {} (quiet nudges so far in this stretch: {})",
|
||||
leadTerminal, quietCount);
|
||||
countNudge("delivered");
|
||||
countNudge("sent");
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("idle-heartbeat: failed to nudge lead {}: {}", leadTerminal, e.toString());
|
||||
countNudge("failed");
|
||||
|
||||
@@ -10,6 +10,7 @@ import com.rabbitmq.client.DeliverCallback;
|
||||
import com.rabbitmq.client.Recoverable;
|
||||
import com.rabbitmq.client.RecoveryListener;
|
||||
import com.rabbitmq.client.Return;
|
||||
import com.rabbitmq.client.ShutdownSignalException;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -58,7 +59,8 @@ import java.util.concurrent.TimeoutException;
|
||||
* <p><strong>Recovery.</strong> The connection is opened with automatic + topology recovery
|
||||
* enabled, mirroring {@code AmqpReplyInbox}: on reconnect the broker hands out fresh delivery tags,
|
||||
* so the held snapshot is cleared (dedup by {@code msgId} still prevents any double-queue on
|
||||
* redelivery) and any publish still awaiting its confirm is failed rather than left to idle out
|
||||
* redelivery). {@link LeadCoordLoop} separately deduplicates pane writes, since it alone knows
|
||||
* which messages reached a lead. Any publish still awaiting its confirm is failed rather than left to idle out
|
||||
* the confirm timeout against a sequence number that means nothing on the new channel.
|
||||
*/
|
||||
public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
@@ -84,6 +86,10 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
private final Object channelLock = new Object();
|
||||
/** msgId → held delivery, for this mailbox's own queue only (there is exactly one). */
|
||||
private final LinkedHashMap<String, Held> held = new LinkedHashMap<>();
|
||||
/** Successful broker acks on this connection, retained only to make a repeated caller ack quiet. */
|
||||
private final LinkedHashMap<String, Boolean> recentlyAcked = new LinkedHashMap<>();
|
||||
/** Bounds {@link #recentlyAcked}: it is only an idempotency aid, never delivery state. */
|
||||
private static final int RECENT_ACK_LIMIT = 1_024;
|
||||
|
||||
/**
|
||||
* A dedicated channel for {@link #publish}, kept separate from {@link #channel} (consume + ack)
|
||||
@@ -122,17 +128,22 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
/** As {@link #open(String, String)}, with an explicit consumer prefetch. */
|
||||
public static LeadMailbox open(String uri, String selfCoordId, int prefetch) {
|
||||
try {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares the queue and re-attaches the consumer.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
return new LeadMailbox(factory.newConnection("fleetd-lead-mailbox"), selfCoordId, prefetch);
|
||||
return new LeadMailbox(connectionFactory(uri).newConnection(AmqpConnectionFailureLogger.LEAD_MAILBOX), selfCoordId, prefetch);
|
||||
} catch (Exception e) {
|
||||
throw new IllegalStateException("cannot connect to AMQP coordination broker at " + uri, e);
|
||||
}
|
||||
}
|
||||
|
||||
static ConnectionFactory connectionFactory(String uri) throws Exception {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares queues and re-attaches consumers.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
factory.setExceptionHandler(new AmqpConnectionFailureLogger(AmqpConnectionFailureLogger.LEAD_MAILBOX, log));
|
||||
return factory;
|
||||
}
|
||||
|
||||
/** Wrap an already-open connection with {@link #DEFAULT_PREFETCH} (injection seam for tests). */
|
||||
LeadMailbox(Connection connection, String selfCoordId) {
|
||||
this(connection, selfCoordId, DEFAULT_PREFETCH);
|
||||
@@ -156,16 +167,15 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
}
|
||||
// On automatic recovery the broker redelivers unacked messages with FRESH delivery-tags; the
|
||||
// tags we were holding are now stale. Drop the held snapshot so the re-attached consumer
|
||||
// repopulates it with valid tags (dedup by msgId still prevents any double-queue). Any publish
|
||||
// repopulates it with valid tags (dedup by msgId still prevents any double-queue). LeadCoordLoop
|
||||
// remembers successful pane writes separately, so that redelivery cannot write a pane twice. Any publish
|
||||
// confirm still in flight when the connection dropped is equally stale — fail it now rather
|
||||
// than let it silently ride out CONFIRM_TIMEOUT_MS.
|
||||
if (connection instanceof Recoverable recoverable) {
|
||||
recoverable.addRecoveryListener(new RecoveryListener() {
|
||||
@Override
|
||||
public void handleRecovery(Recoverable recoverable) {
|
||||
synchronized (held) {
|
||||
held.clear();
|
||||
}
|
||||
clearHeldForRecovery();
|
||||
failPendingPublishesOnRecovery();
|
||||
log.info("AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery");
|
||||
}
|
||||
@@ -254,6 +264,89 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361: look at {@code coordId}'s mailbox on a fresh, immediately-closed throwaway
|
||||
* channel — never {@link #channel} (consume/ack) or {@link #publishChannel} (publish). A
|
||||
* passive queue declare of a queue that does not exist closes the channel it was declared on
|
||||
* with a 404; using a disposable probe channel means that closure can never touch either
|
||||
* long-lived channel this instance depends on for {@link #publish} or the consume loop.
|
||||
*
|
||||
* <p>Classifies failures rather than collapsing them, both measured against a real broker in
|
||||
* {@code LeadMailboxTest} rather than assumed from the AMQP 0-9-1 spec text:
|
||||
* <ul>
|
||||
* <li>a genuine 404 — an {@link IOException} wrapping a {@link ShutdownSignalException} whose
|
||||
* {@link AMQP.Channel.Close#getReplyCode()} is {@code 404} — reports
|
||||
* {@link MailboxState#absent}; every other declare failure reports
|
||||
* {@link MailboxState#unknown} instead of quietly becoming the same "absent" value;
|
||||
* <li>{@code catch (RuntimeException e)} on both attempts matters as much as the checked
|
||||
* catches: a connection that is already closed makes {@link Connection#createChannel()}
|
||||
* throw {@link com.rabbitmq.client.AlreadyClosedException} (a {@link RuntimeException},
|
||||
* not an {@link IOException}) — an {@code inspect} that only caught {@code IOException}
|
||||
* would let that escape, breaking the "never throws" contract this method promises.
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>Honesty about which catch is measured and which is defensive:</strong> the
|
||||
* {@code createChannel()} catch above is exercised end-to-end against a real broker by
|
||||
* {@code LeadMailboxTest.inspectReportsUnknownRatherThanThrowingWhenTheConnectionIsAlreadyClosed}.
|
||||
* The second {@code catch (RuntimeException e)}, around the passive declare itself — for the
|
||||
* narrower race where the connection drops <em>between</em> {@code createChannel()} succeeding
|
||||
* and the declare landing — has no such test; reaching it needs a connection that dies at that
|
||||
* exact instant, which is not a scenario this suite drives on purpose. It stays purely
|
||||
* defensive: correct by the same reasoning as the first catch, but unproven the way the first
|
||||
* one is proven.
|
||||
*/
|
||||
@Override
|
||||
public MailboxState inspect(String coordId) {
|
||||
String queue = queueName(coordId);
|
||||
Channel probe;
|
||||
try {
|
||||
probe = connection.createChannel();
|
||||
} catch (IOException | RuntimeException e) {
|
||||
log.debug("lead mailbox inspect: cannot open a probe channel for {}: {}", coordId, e.toString());
|
||||
return MailboxState.unknown(coordId);
|
||||
}
|
||||
try {
|
||||
AMQP.Queue.DeclareOk declared = probe.queueDeclarePassive(queue);
|
||||
return MailboxState.exists(coordId, declared.getMessageCount(), declared.getConsumerCount());
|
||||
} catch (IOException e) {
|
||||
// The broker (or the client library) has already closed `probe` for us either way; only
|
||||
// a confirmed 404 means "no such queue" — anything else (a different declare failure) is
|
||||
// "could not determine", never silently reported as the same value as a genuine absence.
|
||||
return isMissingQueue(e) ? MailboxState.absent(coordId) : MailboxState.unknown(coordId);
|
||||
} catch (RuntimeException e) {
|
||||
// E.g. the connection dropped between createChannel() and the declare landing.
|
||||
log.debug("lead mailbox inspect: declare failed unexpectedly for {}: {}", coordId, e.toString());
|
||||
return MailboxState.unknown(coordId);
|
||||
} finally {
|
||||
try {
|
||||
if (probe.isOpen()) {
|
||||
probe.close();
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.debug("lead mailbox inspect: probe channel close for {}: {}", coordId, e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code true} only for the specific shape a missing-queue passive declare actually produces —
|
||||
* measured against a real broker, not assumed from the spec text (see {@code
|
||||
* LeadMailboxTest.passiveDeclareOfAMissingQueueThrowsAnIOExceptionWrappingA404ShutdownSignal}):
|
||||
* an {@link IOException} whose cause is a {@link ShutdownSignalException} carrying an
|
||||
* {@link AMQP.Channel.Close} reason with {@code replyCode == 404}. Any other shape (a different
|
||||
* reply code, a {@code ShutdownSignalException} cause whose reason is not a
|
||||
* {@code Channel.Close}, or no cause at all) is a declare failure of some other kind and must
|
||||
* not be read as "confirmed absent" — pinned hermetically, with no broker needed, by
|
||||
* {@code LeadMailboxIsMissingQueueTest} for exactly those three false shapes. Package-private
|
||||
* (not {@code private}) so that test can call it directly.
|
||||
*/
|
||||
static boolean isMissingQueue(IOException e) {
|
||||
if (!(e.getCause() instanceof ShutdownSignalException sse)) {
|
||||
return false;
|
||||
}
|
||||
return sse.getReason() instanceof AMQP.Channel.Close close && close.getReplyCode() == AMQP.NOT_FOUND;
|
||||
}
|
||||
|
||||
/** Convenience: {@link #peek} the current snapshot, then {@link #ack} every message in it. */
|
||||
public List<LeadMessage> drain() {
|
||||
List<LeadMessage> snapshot = peek();
|
||||
@@ -261,15 +354,26 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
return snapshot;
|
||||
}
|
||||
|
||||
/** Remove the held message {@code msgId} and ack it on the broker. No-op if not held. */
|
||||
/**
|
||||
* Remove the held message {@code msgId} and ack it on the broker.
|
||||
*
|
||||
* <p>A repeated ack that this connection already completed is a no-op, tracked in the bounded
|
||||
* {@link #recentlyAcked} set. Any other missing entry throws: recovery clears {@link #held} while
|
||||
* the broker still owns the unacked delivery, and quiet success there would hide a required retry.
|
||||
* The set is bounded because it only distinguishes a recent duplicate caller ack from an unknown
|
||||
* delivery; it is not a substitute for broker state across a reconnect.
|
||||
*/
|
||||
@Override
|
||||
public void ack(String msgId) {
|
||||
Held h;
|
||||
synchronized (held) {
|
||||
h = held.remove(msgId);
|
||||
if (h == null && recentlyAcked.containsKey(msgId)) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (h == null) {
|
||||
return; // never held (or already acked) — no-op
|
||||
throw new IllegalStateException("cannot ack lead message " + msgId + ": it is not held");
|
||||
}
|
||||
try {
|
||||
synchronized (channelLock) {
|
||||
@@ -283,6 +387,19 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
}
|
||||
throw new IllegalStateException("cannot ack lead message " + msgId, e);
|
||||
}
|
||||
synchronized (held) {
|
||||
recentlyAcked.put(msgId, Boolean.TRUE);
|
||||
if (recentlyAcked.size() > RECENT_ACK_LIMIT) {
|
||||
recentlyAcked.remove(recentlyAcked.keySet().iterator().next());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Clear stale delivery tags after recovery; package-private so the recovery contract test drives this exact path. */
|
||||
void clearHeldForRecovery() {
|
||||
synchronized (held) {
|
||||
held.clear();
|
||||
}
|
||||
}
|
||||
|
||||
private DeliverCallback deliverCallback() {
|
||||
|
||||
@@ -138,6 +138,53 @@ public final class MessageService {
|
||||
public record AskResult(AskOutcome outcome, String answer) {
|
||||
}
|
||||
|
||||
/**
|
||||
* How a worker's {@code fleet_reply} ({@link #reply(String, String)}) actually landed
|
||||
* (fleetd #365) — the two doors that expose it, {@code fleet_reply} and {@code POST
|
||||
* /sessions/{id}/reply}, both used to report the single word "delivered" whichever of these
|
||||
* happened, so a caller could not tell an active handoff from a reply merely held for later
|
||||
* drain. Both are successes; they are not the same fact.
|
||||
*/
|
||||
public enum ReplyOutcome {
|
||||
/** Resolved a {@code fleet_send}/{@code fleet_ask} that was actively waiting on this reply. */
|
||||
RESOLVED_SEND("resolved_send", true,
|
||||
"delivered — resolved the fleet_send that was waiting for it"),
|
||||
/**
|
||||
* No live waiter was open, but the reply completed a parked async ticket directly
|
||||
* ({@link #askAnsweredAsyncTasks}) — a {@code fleet_poll} caller sees it immediately.
|
||||
*/
|
||||
RESOLVED_ASYNC_TICKET("resolved_async_ticket", true,
|
||||
"delivered — resolved a pending async ticket (visible to fleet_poll)"),
|
||||
/** Nothing was waiting; the reply was queued in the inbox for a later drain (CB-307). */
|
||||
QUEUED("queued", false,
|
||||
"queued — no send or ticket was waiting; held in the inbox for a later drain");
|
||||
|
||||
private final String wireName;
|
||||
private final boolean delivered;
|
||||
private final String description;
|
||||
|
||||
ReplyOutcome(String wireName, boolean delivered, String description) {
|
||||
this.wireName = wireName;
|
||||
this.delivered = delivered;
|
||||
this.description = description;
|
||||
}
|
||||
|
||||
/** Stable machine-readable name for a JSON/metrics label (REST's {@code outcome} field). */
|
||||
public String wireName() {
|
||||
return wireName;
|
||||
}
|
||||
|
||||
/** Whether something was actively waiting and received this reply right now. */
|
||||
public boolean delivered() {
|
||||
return delivered;
|
||||
}
|
||||
|
||||
/** Shared human-readable text — the one place both {@code fleet_reply} and REST word this. */
|
||||
public String description() {
|
||||
return description;
|
||||
}
|
||||
}
|
||||
|
||||
/** Lifecycle phase of an async delegation ticket. */
|
||||
public enum Phase {
|
||||
/** Delegated and in flight — queued for the worker or being worked. */
|
||||
@@ -187,6 +234,24 @@ public final class MessageService {
|
||||
private volatile Long completedNanos;
|
||||
private volatile Reply question;
|
||||
private volatile String turnId;
|
||||
/**
|
||||
* Set when this task's {@code fleet_ask} lapsed with no answer (fleetd #307):
|
||||
* {@link #clearAsyncQuestion} then forgets {@link #turnId} (nulls it and drops the task from
|
||||
* {@code asyncTasksByTurn}) so {@link #hasAsyncQuestion} stops reporting the target BUSY — a
|
||||
* later {@code fleet_send} to it must be accepted, not refused. But the worker's turn is
|
||||
* still genuinely live: it resumed on its own and will eventually call its real
|
||||
* {@code fleet_reply}. Losing {@link #turnId} loses {@link #askAnsweredAsyncTasks}' only
|
||||
* signal that such a reply belongs to this task, so that reply used to fall straight to the
|
||||
* inbox and strand — {@code fleet_poll} stayed {@code PENDING} forever, later force-failed by
|
||||
* {@link #abandon} with the misleading "session released before it replied". This flag is a
|
||||
* second, independent signal that survives the forgetting: {@link #askAnsweredAsyncTasks}
|
||||
* accepts it in place of a live {@link #turnId}, without ever re-adding the task to
|
||||
* {@code asyncTasksByTurn} (so the BUSY release is untouched). Cleared implicitly once
|
||||
* {@link #future} resolves — every match in {@link #askAnsweredAsyncTasks} already requires
|
||||
* {@code !future.isDone()}, so a task that recovered (or was later failed by
|
||||
* {@link #abandon}) can never match again regardless of this flag's value.
|
||||
*/
|
||||
private volatile boolean askTimedOut;
|
||||
|
||||
private Task(String ticket, String target, LongSupplier nowNanos) {
|
||||
this.ticket = ticket;
|
||||
@@ -360,6 +425,18 @@ public final class MessageService {
|
||||
* {@link #ask} clears the ticket's question and returns it to {@code PENDING}, but {@link #send}
|
||||
* already closed the forward waiter the instant the question surfaced, so the target has
|
||||
* neither an accepted nor a queued delivery left to show for it.
|
||||
*
|
||||
* <p><strong>Deliberately still {@code question == null} only (fleetd #275).</strong> This
|
||||
* method must not also report a still-{@link Phase#ASKING} task as orphaned: the worker may
|
||||
* genuinely be waiting on a live primary that is about to (or already mid-{@link #answer})
|
||||
* answer it, and {@link dev.ltms.fleet.health.FleetHealthMonitor} would classify that as
|
||||
* {@code DELEGATION_ORPHANED} on nothing more than an active, healthy conversation. {@link
|
||||
* #abandon(String, String, boolean)}'s {@code sweepAsking} path fixes the actual reachable gap
|
||||
* (a target torn down for good while genuinely {@code ASKING}) at the point of teardown itself,
|
||||
* by completing the task's future right there — so by the time this method would ever see it,
|
||||
* {@code task.future.isDone()} is already {@code true} and it is excluded regardless of this
|
||||
* guard. Widening this check instead of that one would trade a real fix for false positives on
|
||||
* every ordinary in-flight question.
|
||||
*/
|
||||
public boolean hasOrphanedDelegation(String target) {
|
||||
if (target == null || hasAcceptedDelivery(target) || hasQueuedDelivery(target)) {
|
||||
@@ -383,45 +460,73 @@ public final class MessageService {
|
||||
* {@link Rendezvous#resolveQuestion} must keep today's {@code NO_WAITER} behaviour — questions
|
||||
* are interactive and must never be queued.
|
||||
*
|
||||
* <p><strong>Ambiguous match also falls to the inbox.</strong> {@link #askAnsweredAsyncTasks}
|
||||
* cannot actually return more than one entry today (see its own javadoc for why — in short,
|
||||
* {@link #hasAsyncQuestion} keeps a target BUSY, so no second task can reach this state, for as
|
||||
* long as an earlier one's {@code turnId} is still stamped). That is an emergent guarantee from
|
||||
* two other facts, not one this method enforces, so this branch stays in as defence in depth
|
||||
* rather than being removed as dead code: if it ever weakens, returning whichever candidate a
|
||||
* {@code ConcurrentHashMap} iteration reaches first would let a genuine reply complete the
|
||||
* <em>wrong</em> ticket — silently handing the lead something that reads like a correct answer to
|
||||
* a delegation the worker never touched, which is worse than a failure because the lead acts on
|
||||
* it. When more than one candidate exists, guessing is not safe: fall back to the inbox exactly
|
||||
* as the zero-candidate case does, and let {@link #abandon} apply the eventual recovery
|
||||
* deterministically instead.
|
||||
* <p><strong>Ambiguous match also falls to the inbox.</strong> {@link #askAnsweredAsyncTasks} can
|
||||
* return more than one entry — a reachable state, not a hypothetical one (see its own javadoc:
|
||||
* an {@code fleet_ask} that lapsed with no answer, fleetd #307, frees the target for a completely fresh
|
||||
* delegation, which can itself go on to ask-and-lapse before the first worker's real reply
|
||||
* arrives). Returning whichever candidate a {@code ConcurrentHashMap} iteration reaches first
|
||||
* would let a genuine reply complete the <em>wrong</em> ticket — silently handing the lead
|
||||
* something that reads like a correct answer to a delegation the worker never touched, which is
|
||||
* worse than a failure because the lead acts on it. When more than one candidate exists, guessing
|
||||
* is not safe: fall back to the inbox exactly as the zero-candidate case does, and let
|
||||
* {@link #abandon} apply the eventual recovery deterministically instead.
|
||||
*
|
||||
* @return always {@code true} — the reply resolved a live send, completed a parked ticket, or
|
||||
* was queued
|
||||
* <p><strong>{@code content} is required (fleetd #302).</strong> Both doors that reach this
|
||||
* method must reject a missing/blank reply the same way, so the check lives here rather than in
|
||||
* either caller: {@code FleetMcp.reply} already refuses a {@code null} content before it ever
|
||||
* calls this method (its own required-arg guard), and no test or production call site anywhere
|
||||
* in the codebase relies on replying with empty content — confirmed by searching every call site
|
||||
* of this method before adding the check, not assumed. Without this guard, a REST {@code
|
||||
* POST /sessions/{id}/reply} whose body omits {@code content} (or a client library that maps a
|
||||
* missing field to {@code ""}) used to reach {@link Rendezvous#resolve} with an empty string,
|
||||
* silently completing the lead's blocking wait with nothing — indistinguishable from a worker
|
||||
* that genuinely replied with nothing, which is worse than a loud failure because it destroys the
|
||||
* information that the reply never arrived.
|
||||
*
|
||||
* @throws IllegalArgumentException if {@code content} is {@code null} or blank — the caller must
|
||||
* report this as a client error (REST: 400 {@code bad_request}) rather than resolve
|
||||
* anything
|
||||
* @return which of the three ways (fleetd #365) the reply actually landed — never {@code null}
|
||||
*/
|
||||
public boolean reply(String session, String content) {
|
||||
public ReplyOutcome reply(String session, String content) {
|
||||
if (content == null || content.isBlank()) {
|
||||
throw new IllegalArgumentException("content is required");
|
||||
}
|
||||
if (rendezvous.resolve(session, content)) {
|
||||
count(FleetMetrics.REPLIES, "path", "rendezvous");
|
||||
return true; // a live send took it — unchanged fast path
|
||||
return ReplyOutcome.RESOLVED_SEND; // a live send took it — unchanged fast path
|
||||
}
|
||||
// #137: no live rendezvous waiter, but this may be the worker's real fleet_reply resuming a
|
||||
// turn that {@link #answer} already gave up waiting on. answer()'s own bounded wait (the
|
||||
// primary's fleet_send{turnId} call, capped well under a minute) can time out and close its
|
||||
// waiter long before the worker — now actually resuming real work — finishes and replies. That
|
||||
// reply used to have nowhere to land but the session inbox, leaving the async ticket's future
|
||||
// unresolved forever: fleet_poll{ticket} stayed PENDING until fleet_stop's abandon() forced it
|
||||
// FAILED with a misleading "session released before it replied" reason, even though the reply
|
||||
// had, in fact, arrived. Completing the matching ticket directly here means fleet_poll{ticket}
|
||||
// sees the real reply instead.
|
||||
// #137/fleetd #307: no live rendezvous waiter, but this may be the worker's real fleet_reply resuming
|
||||
// a turn that either answer() (#137) or ask() (fleetd #307) already gave up waiting on:
|
||||
// - answer()'s own bounded wait (the primary's fleet_send{turnId} call, capped well under a
|
||||
// minute) can time out and close its waiter long before the worker — now actually resuming
|
||||
// real work — finishes and replies.
|
||||
// - ask()'s own wait for the primary can time out first, with the worker resuming on its own
|
||||
// and finishing unanswered.
|
||||
// Either way that reply used to have nowhere to land but the session inbox, leaving the async
|
||||
// ticket's future unresolved forever: fleet_poll{ticket} stayed PENDING until fleet_stop's
|
||||
// abandon() forced it FAILED with a misleading "session released before it replied" reason,
|
||||
// even though the reply had, in fact, arrived. Completing the matching ticket directly here
|
||||
// means fleet_poll{ticket} sees the real reply instead.
|
||||
List<Task> candidates = askAnsweredAsyncTasks(session);
|
||||
if (candidates.size() == 1) {
|
||||
Task orphan = candidates.get(0);
|
||||
if (orphan.future.complete(new Reply(Outcome.REPLIED, content))) {
|
||||
if (orphan.turnId != null) {
|
||||
asyncTasksByTurn.remove(orphan.turnId, orphan);
|
||||
// fleetd #329 (F3): read orphan.turnId once. It used to be read twice, under no
|
||||
// lock — once for this null check, once as the removal key — the same double-read
|
||||
// shape fleetd #324 fixed in finishAsyncTask. clearAsyncQuestion's unlocked
|
||||
// forgetTurn=true path (ask()'s timeout cleanup) can null the field between the two
|
||||
// reads; capturing it once removes the torn read here too.
|
||||
String turnId = orphan.turnId;
|
||||
if (turnId != null) {
|
||||
if (replyOrphanTurnIdRaceHookForTest != null) {
|
||||
// Test-only (fleetd #329, F3): see the field's own javadoc.
|
||||
replyOrphanTurnIdRaceHookForTest.run();
|
||||
}
|
||||
asyncTasksByTurn.remove(turnId, orphan);
|
||||
}
|
||||
count(FleetMetrics.REPLIES, "path", "async-recovered");
|
||||
return true; // the ticket itself took it — no inbox stranding at all
|
||||
return ReplyOutcome.RESOLVED_ASYNC_TICKET; // the ticket itself took it — no inbox stranding
|
||||
}
|
||||
} else if (candidates.size() > 1) {
|
||||
List<String> tickets = candidates.stream().map(t -> t.ticket).toList();
|
||||
@@ -439,38 +544,46 @@ public final class MessageService {
|
||||
if (pushLoop != null) {
|
||||
pushLoop.onReplyQueued(session);
|
||||
}
|
||||
return true; // held, not lost
|
||||
return ReplyOutcome.QUEUED; // held, not lost — but not delivered either
|
||||
}
|
||||
|
||||
/**
|
||||
* Every still-open async task on {@code target} whose {@code fleet_ask} was already answered —
|
||||
* its {@link Task#turnId} is stamped but its {@link Task#question} was cleared by {@link #answer}
|
||||
* — yet whose future is not resolved yet (#137). Empty if no such task exists, including the
|
||||
* common case where {@code target}'s worker never used {@code fleet_ask} at all (a task that was
|
||||
* never asked has {@code turnId == null}, so it can never match here and only ever completes
|
||||
* through the ordinary rendezvous fast path in {@link #reply}).
|
||||
* Every still-open async task on {@code target} whose worker is genuinely expected to send a
|
||||
* real {@code fleet_reply} next with nothing left registered to catch it: either its
|
||||
* {@code fleet_ask} was already answered — {@link Task#turnId} is stamped but {@link
|
||||
* Task#question} was cleared by {@link #answer} — or its {@code fleet_ask} lapsed unanswered and
|
||||
* {@link Task#askTimedOut} marks that (fleetd #307; {@link Task#turnId} is {@code null} by then, forgotten
|
||||
* so the target is not left BUSY — see {@link Task#askTimedOut}'s own javadoc). Either way the
|
||||
* task's future is not resolved yet. Empty if no such task exists, including the common case
|
||||
* where {@code target}'s worker never used {@code fleet_ask} at all (a task that was never asked
|
||||
* has both {@code turnId == null} and {@code askTimedOut == false}, so it can never match here and
|
||||
* only ever completes through the ordinary rendezvous fast path in {@link #reply}).
|
||||
*
|
||||
* <p><strong>Returns at most one entry today — verified, not assumed.</strong> {@link #send}
|
||||
* refuses to open a waiter on {@code target} while {@link #hasAsyncQuestion} is true, and that
|
||||
* check matches ANY task whose {@code turnId} is still stamped in {@code asyncTasksByTurn} —
|
||||
* not only while its question is still open. {@link #answer} deliberately leaves that stamp in
|
||||
* place ({@code clearAsyncQuestion(turnId, false)}) until the resumed turn's own future actually
|
||||
* resolves, at which point {@link #finishAsyncTask} both removes the stamp AND completes that
|
||||
* task's future in the same call. So a second task can never reach "{@code turnId} stamped, future
|
||||
* still open" — the exact pair this method matches on — while a first one already holds it: by
|
||||
* the time the stamp is gone, so is the eligibility. This is an emergent property of those two
|
||||
* facts holding together, not something this method (or its callers) enforces on its own — flip
|
||||
* {@code forgetTurn} to {@code true} in that one {@link #answer} call and it silently stops being
|
||||
* true, with nothing left to fail loudly. The callers below still handle "more than one" as
|
||||
* defence in depth against exactly that, not because they exercise it today: {@link #reply}
|
||||
* treats it as unresolvable and falls back to the inbox; {@link #abandon} would pick the oldest
|
||||
* deterministically (its own {@code matching} list has no such guarantee — see its javadoc).
|
||||
* <p><strong>Can return more than one entry — reachable, not just defence in depth.</strong>
|
||||
* {@link #send} refuses to open a waiter on {@code target} while {@link #hasAsyncQuestion} is
|
||||
* true, and that check matches ANY task whose {@code turnId} is still stamped in
|
||||
* {@code asyncTasksByTurn}. While a task's {@code turnId} stays stamped — {@link #answer} leaves
|
||||
* it in place ({@code clearAsyncQuestion(turnId, false)}) until {@link #finishAsyncTask} removes
|
||||
* the stamp and completes the future in the same call — no second task on the same target can
|
||||
* reach an eligible state, because {@link #send} would refuse it as BUSY first. That single-task
|
||||
* guarantee holds only for the {@code turnId}-stamped half of this method's match: an
|
||||
* {@link Task#askTimedOut} task is, by construction, no longer stamped in {@code asyncTasksByTurn}
|
||||
* (that is the whole point of forgetting {@code turnId} in {@link #clearAsyncQuestion}), so the
|
||||
* target is free the moment one ask lapses. A fresh, independent {@code sendAsync} to the same
|
||||
* target can then be dispatched, itself pause on {@code fleet_ask}, and itself time out — landing
|
||||
* a second {@code askTimedOut} task on the very target the first one is still waiting to answer
|
||||
* for. Two (or more) genuinely open tasks on one target is therefore a real, reachable state
|
||||
* today, not a hypothetical: {@link #reply} treats it as unresolvable and falls back to the
|
||||
* inbox rather than guess which task a reply belongs to (guessing wrong would hand the lead a
|
||||
* plausible-looking answer to a delegation the worker never touched — worse than a failure,
|
||||
* because the lead acts on it); {@link #abandon} instead picks the oldest deterministically (its
|
||||
* own {@code matching} list has a different, wider match — see its javadoc).
|
||||
*/
|
||||
private List<Task> askAnsweredAsyncTasks(String target) {
|
||||
List<Task> candidates = new ArrayList<>();
|
||||
for (Task task : tasks.values()) {
|
||||
if (target.equals(task.target) && task.question == null && task.turnId != null
|
||||
&& !task.future.isDone()) {
|
||||
if (target.equals(task.target) && task.question == null && !task.future.isDone()
|
||||
&& (task.turnId != null || task.askTimedOut)) {
|
||||
candidates.add(task);
|
||||
}
|
||||
}
|
||||
@@ -580,6 +693,39 @@ public final class MessageService {
|
||||
* reply — see the note above)
|
||||
*/
|
||||
public boolean abandon(String target, String reason) {
|
||||
return abandon(target, reason, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #abandon(String, String)}, with control over whether a task still paused in
|
||||
* {@code fleet_ask} ({@link Phase#ASKING}) is swept too (fleetd #275).
|
||||
*
|
||||
* <p>{@code sweepAsking} must be {@code true} only when the caller has independent, certain
|
||||
* knowledge that {@code target} can never resume its turn — today that is only
|
||||
* {@code sessions.onRelease}'s teardown (an explicit {@code fleet_stop}, or the idle reaper):
|
||||
* the worker's pane is being stopped right now, so whatever it was mid-{@code fleet_ask} about
|
||||
* has no turn left to resume into. {@link dev.ltms.fleet.health.FleetHealthMonitor}'s
|
||||
* health-classification call keeps passing {@code false} (via {@link #abandon(String, String)}):
|
||||
* a GONE/NEVER_READY reading is the daemon's best guess from the live agent list, not a teardown
|
||||
* it performed itself, and {@code abandonDoesNotFailAnAsyncTicketWaitingForAnAnswer} documents
|
||||
* why an active ask must survive that guess — the primary may already be mid-{@link #answer} for
|
||||
* the very same turn, and completing it here first would preempt a real answer with a misleading
|
||||
* failure.
|
||||
*
|
||||
* <p><strong>Without {@code sweepAsking} on the release path, a target torn down while
|
||||
* genuinely {@code ASKING} was unrecoverable.</strong> {@link #resolveQuestion} had already
|
||||
* closed the forward waiter the instant the question surfaced (so the {@code waiter} branch
|
||||
* below finds nothing to fail), the {@code question == null} guard excluded the task from
|
||||
* {@code matching} (so the loop below skipped it too), and the worker's own {@code fleet_ask}
|
||||
* clears {@link Task#question} back to {@code null} only once it lapses (the reverse-rendezvous
|
||||
* window — up to {@code FleetMcp.ASK_DEFAULT_TIMEOUT_MS} / {@code FleetApp.MAX_ASK_TIMEOUT_MS},
|
||||
* 55–115s) — by which point the released session no longer appears in {@code sessions.roster()}
|
||||
* for {@link dev.ltms.fleet.health.FleetHealthMonitor} to ever re-observe, so nothing was ever
|
||||
* left to call {@link #abandon} on this target again. The ticket then sat in {@link #tasks}
|
||||
* forever: not terminal, so {@link #pruneTerminalTickets} never dropped it, and
|
||||
* {@code fleet_poll} reported it stuck at {@link Phase#PENDING} for good.
|
||||
*/
|
||||
public boolean abandon(String target, String reason, boolean sweepAsking) {
|
||||
boolean hadStrandedReply = hasStrandedReply(target);
|
||||
// CB-640: the session is gone — nothing will ever accept or deliver into it now.
|
||||
strandedReplies.remove(target);
|
||||
@@ -590,7 +736,8 @@ public final class MessageService {
|
||||
|
||||
List<Task> matching = new ArrayList<>();
|
||||
for (Task task : tasks.values()) {
|
||||
if (target.equals(task.target) && task.question == null && !task.future.isDone()) {
|
||||
if (target.equals(task.target) && (sweepAsking || task.question == null)
|
||||
&& !task.future.isDone()) {
|
||||
matching.add(task);
|
||||
}
|
||||
}
|
||||
@@ -607,19 +754,48 @@ public final class MessageService {
|
||||
for (Task task : matching) {
|
||||
boolean isRecovery = task == recoveryTask && recovered != null;
|
||||
Reply outcome = isRecovery ? recovered : new Reply(Outcome.WORKER_FAILED, reason);
|
||||
if (task.future.complete(outcome)) {
|
||||
if (outcome.outcome() == Outcome.WORKER_FAILED) {
|
||||
asyncFailed = true;
|
||||
} else if (task.turnId != null) {
|
||||
asyncTasksByTurn.remove(task.turnId, task);
|
||||
String turnId = task.turnId;
|
||||
// fleetd #335: completing THIS task's future must not depend on any other task's
|
||||
// cleanup succeeding — every task in `matching` is owed its own outcome regardless of
|
||||
// what happens below, so decide and record that before doing anything that can throw.
|
||||
boolean completedHere = task.future.complete(outcome);
|
||||
if (completedHere && outcome.outcome() == Outcome.WORKER_FAILED) {
|
||||
asyncFailed = true;
|
||||
}
|
||||
try {
|
||||
if (abandonCleanupHookForTest != null) {
|
||||
// Test-only (fleetd #335 site 1): see the field's own javadoc.
|
||||
abandonCleanupHookForTest.run();
|
||||
}
|
||||
} else if (isRecovery) {
|
||||
// The recovered reply was already drained out of the inbox, but this task resolved
|
||||
// through another path (e.g. a concurrent reply() or a second abandon() racing this
|
||||
// one) between us choosing it and completing it here. Put the reply back rather than
|
||||
// lose it silently — it may still belong to some other still-open task, or the next
|
||||
// caller that drains this target's inbox.
|
||||
inbox.publish(target, UUID.randomUUID().toString(), recovered.text());
|
||||
if (completedHere) {
|
||||
if (turnId != null) {
|
||||
// #275: whether this task was swept out of ASKING or was already answered and
|
||||
// only waiting on its resumed turn's real reply (#137), nothing will ever
|
||||
// complete this turnId now — drop it from this class's own bookkeeping AND the
|
||||
// reverse-rendezvous itself, so hasAsyncQuestion(target) stops reporting a turn
|
||||
// that is actually done, and a late answer() sees it as lapsed rather than
|
||||
// resolving a question nothing is listening for any more.
|
||||
asyncTasksByTurn.remove(turnId, task);
|
||||
rendezvous.closeAsk(turnId);
|
||||
}
|
||||
} else if (isRecovery) {
|
||||
// The recovered reply was already drained out of the inbox, but this task resolved
|
||||
// through another path (e.g. a concurrent reply() or a second abandon() racing this
|
||||
// one) between us choosing it and completing it here. Put the reply back rather than
|
||||
// lose it silently — it may still belong to some other still-open task, or the next
|
||||
// caller that drains this target's inbox.
|
||||
inbox.publish(target, UUID.randomUUID().toString(), recovered.text());
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
// fleetd #335: inbox.publish reaches a broker (AmqpReplyInbox throws
|
||||
// IllegalStateException on an unroutable/unconfirmed/interrupted publish) and this
|
||||
// loop has no other teardown path — a caller on the release path, or the health
|
||||
// monitor's GONE/NEVER_READY sweep. Losing this exception uncaught would abort the
|
||||
// loop and leave every task still to come in `matching` PENDING forever (fleetd
|
||||
// #335 site 1). completedHere is already recorded above, so only this task's
|
||||
// best-effort bookkeeping is lost — log it and let the loop reach the rest.
|
||||
log.error("abandon: per-task cleanup failed for ticket {} (target {}, turnId {})",
|
||||
task.ticket, target, turnId, e);
|
||||
}
|
||||
}
|
||||
if (failed) {
|
||||
@@ -753,16 +929,26 @@ public final class MessageService {
|
||||
if (onAccepted != null) {
|
||||
onAccepted.run();
|
||||
}
|
||||
CompletableFuture<Void> delivered = injector.enqueue(target, content, token);
|
||||
Injector.Delivery delivery = injector.enqueue(target, content, token);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return recorded(new Reply(outcomeOf(r.kind()), r.text(), r.turnId()));
|
||||
} catch (TimeoutException e) {
|
||||
boolean wasDelivered = delivered.isDone() && !delivered.isCompletedExceptionally();
|
||||
boolean wasDelivered = delivery.completion().isDone()
|
||||
&& !delivery.completion().isCompletedExceptionally();
|
||||
if (!wasDelivered) {
|
||||
if (timeoutCancellationRaceHookForTest != null) {
|
||||
// Test-only (fleetd #345): see the field's own javadoc.
|
||||
timeoutCancellationRaceHookForTest.run();
|
||||
}
|
||||
// The target monitor makes cancellation atomic with onStatus picking this
|
||||
// Pending up. If pickup won, report TIMED_OUT_WORKING because the text landed.
|
||||
wasDelivered = injector.cancel(delivery) == Injector.Cancellation.DELIVERED;
|
||||
}
|
||||
log.debug("send to {} timed out (delivered={})", target, wasDelivered);
|
||||
if (!wasDelivered) {
|
||||
// CB-640: still sitting in the injector's queue, waiting for the member to
|
||||
// go idle — record the fact for fleet health (see queuedDeliveries).
|
||||
// CB-640: record that delivery did not happen for fleet health (see
|
||||
// queuedDeliveries). The exact Pending was cancelled, so it cannot arrive later.
|
||||
queuedDeliveries.put(target, Boolean.TRUE);
|
||||
}
|
||||
return recorded(new Reply(
|
||||
@@ -825,7 +1011,41 @@ public final class MessageService {
|
||||
return new AskResult(AskOutcome.ANSWERED, answer);
|
||||
} catch (TimeoutException e) {
|
||||
log.debug("fleet_ask from {} went unanswered in {}ms", workerSession, timeoutMillis);
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
// Only the fresh owner tears down the shared turn (mirrors the finally block below).
|
||||
// A duplicate's own timeoutMillis says nothing about whether the SHARED ask is actually
|
||||
// done — it must leave the close/forget bookkeeping to the fresh owner, exactly as it
|
||||
// already leaves closeAsk to it.
|
||||
if (ticket.fresh()) {
|
||||
// fleetd #334: close the ask turn BEFORE forgetting this task's turnId mapping below.
|
||||
// Before this fix the order was reversed — the mapping was forgotten here first, and
|
||||
// rendezvous.closeAsk only ran afterward, in the shared finally. A primary's answer()
|
||||
// call racing this exact timeout could then find rendezvous.askSession(turnId) still
|
||||
// non-null (the ask still "answerable") after the Task mapping was already gone:
|
||||
// answer()'s own asyncTasksByTurn lookup returned null, its task != null guard skipped
|
||||
// the completion, and the async ticket sat at PENDING forever even though answer()
|
||||
// itself reported the worker's real reply. Closing here first removes that window:
|
||||
// any answer() call that still observes askSession(turnId) != null is necessarily
|
||||
// racing a point BEFORE the forgetting below runs (both happen on this one thread, in
|
||||
// this order, with nothing that yields in between), so the Task mapping is still there
|
||||
// for it to find; any call that observes askSession(turnId) == null now correctly
|
||||
// bails out STALE_TURN (see answer()'s own top check) before ever reaching
|
||||
// asyncTasksByTurn. rendezvous.closeAsk is idempotent — a no-op once the turn is
|
||||
// already removed, see its own javadoc — so the shared finally below re-running it
|
||||
// for this same fresh call is harmless.
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
if (askTimeoutRaceHookForTest != null) {
|
||||
// Test-only (fleetd #334): see the field's own javadoc.
|
||||
askTimeoutRaceHookForTest.run();
|
||||
}
|
||||
// fleetd #307: mark the task BEFORE clearAsyncQuestion(forgetTurn=true) below drops it
|
||||
// out of asyncTasksByTurn and nulls its turnId — that forgetting is deliberate and
|
||||
// stays (it is what keeps the target from staying BUSY forever), but it would
|
||||
// otherwise also erase askAnsweredAsyncTasks' only signal that the worker's eventual
|
||||
// real fleet_reply still belongs to this task, stranding it in the inbox with a false
|
||||
// "never replied" verdict.
|
||||
markAskTimedOut(ticket.turnId());
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
}
|
||||
return new AskResult(AskOutcome.TIMED_OUT, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
@@ -872,7 +1092,16 @@ public final class MessageService {
|
||||
}
|
||||
try {
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession);
|
||||
// #282: mirror send()'s registration (:802) so a SECOND fleet_ask inside this same
|
||||
// resumed turn can re-associate the async ticket with its new turnId via
|
||||
// markAsyncQuestion — without this, that second ask has no Task to attach to, and
|
||||
// markAsyncQuestion silently returns null.
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null) {
|
||||
asyncTasksByWaiter.put(reply, task);
|
||||
}
|
||||
if (!rendezvous.answerAsk(turnId, content)) {
|
||||
asyncTasksByWaiter.remove(reply);
|
||||
rendezvous.close(workerSession, reply);
|
||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||
}
|
||||
@@ -880,7 +1109,66 @@ public final class MessageService {
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
Reply result = new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
finishAsyncTask(turnId, result);
|
||||
// #282: this waiter can resolve with a FRESH question rather than a terminal reply —
|
||||
// the worker chained a second fleet_ask before replying. Mirror sendAsync's own guard
|
||||
// (:1000) and leave the ticket open (markAsyncQuestion above already re-armed it under
|
||||
// the new turnId) instead of completing it here with a QUESTION "reply".
|
||||
// Measured when #282 was merged: this guard is DEFENCE IN DEPTH, not the thing
|
||||
// that makes the chained ask work. ask() calls markAsyncQuestion (:860) before
|
||||
// resolveQuestion (:861), so by the time this thread wakes, the task has already
|
||||
// moved to the new turnId — this outcome check, not a Task lookup, is what tells the
|
||||
// two cases apart (see fleetd #329 below). Removing this guard alone leaves the test
|
||||
// green. Keep it anyway: it mirrors sendAsync's sibling guard, and that sibling's own
|
||||
// comment (:1017) warns the two orderings are not something to rely on. Do NOT delete
|
||||
// it as dead code without re-checking that ordering, and do not treat it as the sole
|
||||
// protection either.
|
||||
//
|
||||
// fleetd #329 (F1): complete the SAME Task object this method already looked up at
|
||||
// :991, rather than re-resolving it from turnId a second time. The old
|
||||
// finishAsyncTask(turnId, result) did its own asyncTasksByTurn.get(turnId) here, and
|
||||
// that second lookup races ask()'s own timeout path: ask()'s ticket.answer().get(...)
|
||||
// can time out (or lose that exact race) at essentially the same instant this method's
|
||||
// rendezvous.answerAsk(turnId, ...) above already succeeded, running
|
||||
// markAskTimedOut + clearAsyncQuestion(turnId, true) with no lock at all — which
|
||||
// forgets turnId (removes it from asyncTasksByTurn, nulls Task.turnId) before this
|
||||
// thread ever gets here. The worker's real reply then arrived, this method's own wait
|
||||
// woke up with it, and the by-turnId lookup found nothing: the async ticket's future
|
||||
// was never completed, so fleet_poll{ticket} stayed PENDING forever even though
|
||||
// answer() itself correctly returned REPLIED. Reusing the reference captured at :991
|
||||
// — before that race window opens — sidesteps the second lookup entirely: it is the
|
||||
// identical Task whatever asyncTasksByTurn or Task.turnId say by the time we reach
|
||||
// this point, so finishAsyncTask(Task, Reply) can still complete its future and detach
|
||||
// it using whatever turnId it now reads. The #282 chained-ask case is unaffected
|
||||
// because it is still gated purely by result.outcome() == QUESTION above, which does
|
||||
// not depend on this lookup — widening what "task" means here cannot complete a ticket
|
||||
// the chained ask deliberately left open.
|
||||
//
|
||||
// A null task is NOT only "this was never an async ticket". That reading was in this
|
||||
// comment when #329 merged and it is wrong; it is still not the whole story after
|
||||
// #334. A genuine async ticket can still land here with task == null — a blocking
|
||||
// (wait:true) send's fleet_ask never has a Task at all, so that case is expected and
|
||||
// fine. What #334 fixed was a SECOND, unintended way to get here with task == null:
|
||||
// ask()'s timeout path used to run clearAsyncQuestion(turnId, true) — which drops the
|
||||
// asyncTasksByTurn entry — in its catch block, while rendezvous.closeAsk(turnId) ran
|
||||
// later, in its finally. Between those two the ask was still answerable but the map
|
||||
// entry was already gone, so the lookup at :1053 returned null and this ticket was
|
||||
// never completed. Measured on 2026-09-04: a probe firing only that first half before
|
||||
// answer() ran printed "answer=REPLIED phase=PENDING reply=null" — the same stranded
|
||||
// ticket #329 set out to fix, one step earlier in the same race; the probe used
|
||||
// forgetTurnForTest, which omits ask()'s markAskTimedOut, and that omission does not
|
||||
// change the outcome, because askTimedOut is read only by askAnsweredAsyncTasks, and
|
||||
// reply() never reaches it while this method's own waiter is live. #334's fix
|
||||
// reorders ask()'s timeout catch to run closeAsk before the forgetting (see the
|
||||
// fresh-owner block there), which removes this path entirely rather than narrowing
|
||||
// it further: once closeAsk has run, rendezvous.askSession(turnId) is null and
|
||||
// answer() returns STALE_TURN from its own top check, before it ever reaches this
|
||||
// lookup — see aLateAnswerDuringAskTimeoutTeardownStillCompletesTheAsyncTicket in
|
||||
// MessageServiceTest, which pins the exact window with askTimeoutRaceHookForTest.
|
||||
// So by the time this line runs, task == null means only the ordinary blocking-send
|
||||
// case (or #329's own already-fixed race elsewhere) — not this one.
|
||||
if (result.outcome() != Outcome.QUESTION && task != null) {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
return result;
|
||||
} catch (TimeoutException e) {
|
||||
// The worker resumed but hasn't replied yet — no completion fallback arms an answered
|
||||
@@ -893,6 +1181,7 @@ public final class MessageService {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + workerSession, e);
|
||||
} finally {
|
||||
asyncTasksByWaiter.remove(reply);
|
||||
rendezvous.close(workerSession, reply);
|
||||
}
|
||||
} finally {
|
||||
@@ -929,14 +1218,25 @@ public final class MessageService {
|
||||
// this fires exactly once, from whichever path completes it: finishAsyncTask(task, result)
|
||||
// below on any non-QUESTION outcome of send() — a worker's fleet_reply, the CB-106
|
||||
// completion fallback, a CB-109 wedge, TIMED_OUT, BUSY, or BACKEND_EXHAUSTED — the same
|
||||
// finishAsyncTask reached via answer()'s finishAsyncTask(turnId, result) once a QUESTION
|
||||
// finishAsyncTask reached via answer()'s own finishAsyncTask(task, result) once a QUESTION
|
||||
// is resolved, completeExceptionally(t) just below when send() itself throws, or a CB-516
|
||||
// abandon() on teardown. Without this, MessageService.reply's rendezvous fast path (the
|
||||
// one an async ticket always takes) never told the push loop anything happened — see the
|
||||
// class javadoc on sendAsync/CB-107.
|
||||
task.future.whenComplete((reply, ex) -> {
|
||||
boolean failed = ex != null || reply == null || !reply.completed();
|
||||
pushLoop.onTicketTerminal(ticket, target, failed);
|
||||
try {
|
||||
pushLoop.onTicketTerminal(ticket, target, failed);
|
||||
} catch (Throwable t) {
|
||||
// fleetd #335 (site 2): this stage's own CompletableFuture is discarded, so an
|
||||
// uncaught throw here (e.g. a RejectedExecutionException from
|
||||
// ReplyPushLoop's scheduler, already shut down while this in-flight send's
|
||||
// whenComplete fires during the daemon's own shutdown sequence — messages.close()
|
||||
// only stops accepting NEW async work, it does not cancel a delivery already
|
||||
// running) vanishes with no log line and no metric, and the push loop never learns
|
||||
// the ticket went terminal — the exact thing this hook exists to tell it.
|
||||
log.error("push loop failed to learn ticket {} (target {}) went terminal", ticket, target, t);
|
||||
}
|
||||
});
|
||||
}
|
||||
asyncExecutor.submit(() -> {
|
||||
@@ -949,7 +1249,22 @@ public final class MessageService {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
} catch (Throwable t) {
|
||||
task.future.completeExceptionally(t);
|
||||
// fleetd #329 (F2): finishAsyncTask above already completes task.future — on its very
|
||||
// first line — before doing anything else, so anything that throws afterward (inside
|
||||
// finishAsyncTask's own cleanup, or from a future addition to this try block) lands
|
||||
// here with the future already resolved. completeExceptionally on an already-completed
|
||||
// future is a silent no-op: it returns false and does nothing, so without the check
|
||||
// below the exception simply vanished — no log, no metric, nothing. Measured (see the
|
||||
// ticket): temporarily reintroducing the fleetd #324 NPE reproduced 19 real exceptions
|
||||
// on the ordinary path, with 74/74 tests staying green and not one log line produced.
|
||||
// Do not stop completing the future first — finishAsyncTask completing before it
|
||||
// cleans up is what makes a late failure harmless to the ticket's own result — only
|
||||
// add the missing visibility for the case where that step, or whatever ran after it,
|
||||
// has already lost the race to report through the future.
|
||||
if (!task.future.completeExceptionally(t)) {
|
||||
log.error("async send {} -> {} threw after its ticket was already resolved",
|
||||
ticket, target, t);
|
||||
}
|
||||
}
|
||||
});
|
||||
pruneTerminalTickets();
|
||||
@@ -1051,13 +1366,37 @@ public final class MessageService {
|
||||
private Task markAsyncQuestion(CompletableFuture<Rendezvous.Resolution> waiter, String text, String turnId) {
|
||||
Task task = waiter == null ? null : asyncTasksByWaiter.get(waiter);
|
||||
if (task != null) {
|
||||
String previousTurnId = task.turnId;
|
||||
task.question = new Reply(Outcome.QUESTION, text, turnId);
|
||||
task.turnId = turnId;
|
||||
asyncTasksByTurn.put(turnId, task);
|
||||
// #282: a second fleet_ask in the same resumed turn re-arms an already-answered task
|
||||
// (answer() re-registers it in asyncTasksByWaiter) under a FRESH turnId — drop the old
|
||||
// key so asyncTasksByTurn does not keep growing by one stale entry per chained ask.
|
||||
if (previousTurnId != null && !previousTurnId.equals(turnId)) {
|
||||
asyncTasksByTurn.remove(previousTurnId, task);
|
||||
}
|
||||
}
|
||||
return task;
|
||||
}
|
||||
|
||||
/**
|
||||
* Mark {@code turnId}'s task as having a {@code fleet_ask} that lapsed with no answer (fleetd #307), so
|
||||
* {@link #askAnsweredAsyncTasks} still recognizes the worker's eventual real {@code fleet_reply}
|
||||
* as belonging to it after {@link #clearAsyncQuestion}'s {@code forgetTurn=true} erases
|
||||
* {@link Task#turnId} — see {@link Task#askTimedOut}. Must be called before that forgetting, while
|
||||
* {@code turnId} can still resolve the task in {@code asyncTasksByTurn}; a lookup afterward would
|
||||
* find nothing. Only when it matches the task's current turn — same guard as
|
||||
* {@link #clearAsyncQuestion} — so a chained second {@code fleet_ask} (#282) that already moved
|
||||
* the task to a fresh {@code turnId} cannot mark it for a turn that is no longer its own.
|
||||
*/
|
||||
private void markAskTimedOut(String turnId) {
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null && turnId.equals(task.turnId)) {
|
||||
task.askTimedOut = true;
|
||||
}
|
||||
}
|
||||
|
||||
/** Clear an answered or lapsed question, but only when it matches the ticket's current turn. */
|
||||
private void clearAsyncQuestion(String turnId, boolean forgetTurn) {
|
||||
// CB-582: tell the push loop first — like ticketCollected, a removal for a turnId it never
|
||||
@@ -1076,20 +1415,174 @@ public final class MessageService {
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete and detach an async ticket after its worker's actual terminal reply. */
|
||||
/**
|
||||
* Complete and detach an async ticket after its worker's actual terminal reply.
|
||||
*
|
||||
* <p><strong>fleetd #324.</strong> {@code task.turnId} is read into {@code turnId} exactly once.
|
||||
* It used to be read twice — once for the null check, once as the removal key — and {@code
|
||||
* volatile} makes each of those reads individually fresh but does not make the pair atomic.
|
||||
* {@link #answer} calls this while holding {@code sessionLocks} for the target; {@link #ask}'s
|
||||
* own timeout path calls {@link #clearAsyncQuestion} (which nulls {@link Task#turnId}) under no
|
||||
* lock at all. When that unlocked null-out landed between the two reads here, the second read saw
|
||||
* {@code null} and {@code asyncTasksByTurn.remove(null, task)} threw {@code NullPointerException}
|
||||
* on the lead's own {@code answer()} call — even though {@code task.future.complete(result)} on
|
||||
* the line above had already run, so the answer was in fact delivered. Capturing the field once
|
||||
* removes the torn read; see the ticket for why the wider asymmetry between the locked and
|
||||
* unlocked sides is not fixed by this alone.
|
||||
*/
|
||||
private void finishAsyncTask(Task task, Reply result) {
|
||||
task.future.complete(result);
|
||||
if (task.turnId != null) {
|
||||
asyncTasksByTurn.remove(task.turnId, task);
|
||||
if (afterFinishAsyncTaskCompleteHookForTest != null) {
|
||||
// Test-only (fleetd #329, F2): see the field's own javadoc.
|
||||
afterFinishAsyncTaskCompleteHookForTest.run();
|
||||
}
|
||||
String turnId = task.turnId;
|
||||
if (turnId != null) {
|
||||
if (finishAsyncTaskRaceHook != null) {
|
||||
// Test-only (fleetd #324): see the field's own javadoc.
|
||||
finishAsyncTaskRaceHook.run();
|
||||
}
|
||||
asyncTasksByTurn.remove(turnId, task);
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete the async ticket correlated to a specific answered turn. */
|
||||
private void finishAsyncTask(String turnId, Reply result) {
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null) {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
/**
|
||||
* Null in production; test seam for fleetd #324 — invoked from {@link #finishAsyncTask(Task,
|
||||
* Reply)} right after {@code task.turnId}'s null-check passes and before the (now-local) value is
|
||||
* used for the removal. A test installs this to force, deterministically, the exact interleaving
|
||||
* that a real race between this method and {@link #ask}'s unlocked timeout cleanup can otherwise
|
||||
* only produce by chance: firing it here reproduces "the field went null between the check and the
|
||||
* use" against the pre-fix code, and demonstrates the fix tolerates it (the captured local is used
|
||||
* unconditionally, so a hook that nulls the field afterward cannot affect this call).
|
||||
*/
|
||||
private volatile Runnable finishAsyncTaskRaceHook;
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #324): install {@link #finishAsyncTaskRaceHook}. Package-private so the test,
|
||||
* in the same package, can reach it without widening any production API.
|
||||
*/
|
||||
void setFinishAsyncTaskRaceHookForTest(Runnable hook) {
|
||||
this.finishAsyncTaskRaceHook = hook;
|
||||
}
|
||||
|
||||
/**
|
||||
* Null in production; test seam for fleetd #329 (F2) — invoked from {@link #finishAsyncTask(Task,
|
||||
* Reply)} unconditionally, immediately after {@code task.future.complete(result)} runs (before
|
||||
* {@link Task#turnId} is even read, so it fires regardless of whether this task was ever asked).
|
||||
* A test installs this to force an exception into exactly the shape fleetd #329 identified:
|
||||
* something throws inside {@link #sendAsync}'s executor task after the async ticket's future is
|
||||
* already resolved, so the surrounding {@code catch (Throwable t)} can only report failure through
|
||||
* {@code completeExceptionally} — a silent no-op on an already-completed future. Engineering a
|
||||
* real exception to land in that exact post-completion window is what the ticket itself had to do
|
||||
* by temporarily deleting a production guard (fleetd #324's {@code turnId != null} check); this
|
||||
* hook drives the identical shape deterministically instead.
|
||||
*/
|
||||
private volatile Runnable afterFinishAsyncTaskCompleteHookForTest;
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #329, F2): install {@link #afterFinishAsyncTaskCompleteHookForTest}.
|
||||
* Package-private so the test, in the same package, can reach it without widening any production
|
||||
* API.
|
||||
*/
|
||||
void setAfterFinishAsyncTaskCompleteHookForTest(Runnable hook) {
|
||||
this.afterFinishAsyncTaskCompleteHookForTest = hook;
|
||||
}
|
||||
|
||||
/**
|
||||
* Null in production; test seam for fleetd #345 — invoked in {@link #send}'s timeout path after
|
||||
* {@link Injector.Delivery#completion()} reports incomplete and before {@link Injector#cancel}
|
||||
* takes the target monitor. A test installs this to make {@code onStatus} pick the exact queued
|
||||
* delivery up in that window, so {@code cancel} returns {@link Injector.Cancellation#DELIVERED}.
|
||||
* This deterministically covers the caller's need to use that result rather than relying on a
|
||||
* timing-sensitive real race.
|
||||
*/
|
||||
private volatile Runnable timeoutCancellationRaceHookForTest;
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #345): install {@link #timeoutCancellationRaceHookForTest}. Package-private
|
||||
* so the test, in the same package, can reach it without widening any production API.
|
||||
*/
|
||||
void setTimeoutCancellationRaceHookForTest(Runnable hook) {
|
||||
this.timeoutCancellationRaceHookForTest = hook;
|
||||
}
|
||||
|
||||
/**
|
||||
* Null in production; test seam for fleetd #329 (F3) — invoked from {@link #reply} right after
|
||||
* the single local read of {@code orphan.turnId} passes its null-check and before that (now-local)
|
||||
* value is used as the {@code asyncTasksByTurn} removal key. Mirrors {@link
|
||||
* #finishAsyncTaskRaceHook} exactly, for the structurally identical double-read fleetd #324 fixed
|
||||
* in {@link #finishAsyncTask}: a test installs this to force, deterministically, {@code
|
||||
* clearAsyncQuestion}'s unlocked {@code forgetTurn=true} path nulling {@link Task#turnId} in that
|
||||
* exact window, and to confirm the single-read fix tolerates it (the captured local is used
|
||||
* unconditionally, so a hook that nulls the field afterward cannot affect this call).
|
||||
*/
|
||||
private volatile Runnable replyOrphanTurnIdRaceHookForTest;
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #329, F3): install {@link #replyOrphanTurnIdRaceHookForTest}. Package-private
|
||||
* so the test, in the same package, can reach it without widening any production API.
|
||||
*/
|
||||
void setReplyOrphanTurnIdRaceHookForTest(Runnable hook) {
|
||||
this.replyOrphanTurnIdRaceHookForTest = hook;
|
||||
}
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #324): run the exact production cleanup {@link #ask}'s own timeout path runs
|
||||
* unlocked — {@link #clearAsyncQuestion(String, boolean)} with {@code forgetTurn=true} — so a test
|
||||
* can reproduce that specific mutation instead of hand-rolling an approximation of it.
|
||||
*/
|
||||
void forgetTurnForTest(String turnId) {
|
||||
clearAsyncQuestion(turnId, true);
|
||||
}
|
||||
|
||||
/**
|
||||
* Null in production; test seam for fleetd #335 (site 1) — invoked from {@link #abandon(String,
|
||||
* String, boolean)}'s per-task loop, once per task, right before that task's own cleanup
|
||||
* (turnId bookkeeping, or the stranded-reply put-back) runs. A test installs this to inject a
|
||||
* throw at that exact point deterministically.
|
||||
*
|
||||
* <p>The one production call there that can really throw is {@code inbox.publish} in the
|
||||
* put-back branch — {@link AmqpReplyInbox#publish} reaches a broker and throws {@link
|
||||
* IllegalStateException} on an unroutable, unconfirmed, or interrupted publish — but reaching
|
||||
* that branch requires a second completion of the very task {@code abandon} is about to
|
||||
* complete to win the race first (see the branch's own comment), and the #137 follow-up
|
||||
* investigation above already found the combination this needs (a stranded reply coinciding
|
||||
* with an open matching task) unreachable through the public API, not merely hard to time.
|
||||
* This hook reproduces the resulting shape — a per-task cleanup throw — directly, the same
|
||||
* technique {@link #finishAsyncTaskRaceHook} and {@link
|
||||
* #afterFinishAsyncTaskCompleteHookForTest} already use for their own hard-to-time races.
|
||||
*/
|
||||
private volatile Runnable abandonCleanupHookForTest;
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #335, site 1): install {@link #abandonCleanupHookForTest}. Package-private
|
||||
* so the test, in the same package, can reach it without widening any production API.
|
||||
*/
|
||||
void setAbandonCleanupHookForTest(Runnable hook) {
|
||||
this.abandonCleanupHookForTest = hook;
|
||||
}
|
||||
|
||||
/**
|
||||
* Null in production; test seam for fleetd #334 — invoked from {@link #ask}'s {@code
|
||||
* TimeoutException} catch, only for the fresh owner, right after {@code rendezvous.closeAsk}
|
||||
* has run and before {@link #markAskTimedOut} / {@link #clearAsyncQuestion} forget this task's
|
||||
* turnId mapping. A test installs this to call {@link #answer} for the very same {@code turnId}
|
||||
* synchronously from inside that exact window, deterministically reproducing the race a real
|
||||
* concurrent {@code answer()} call could otherwise only win by timing luck: with the ask already
|
||||
* closed, that call must see {@code rendezvous.askSession(turnId) == null} and return {@link
|
||||
* Outcome#STALE_TURN} immediately, never reaching {@code asyncTasksByTurn} at all — proving the
|
||||
* window fleetd #334 describes (mapping forgotten while the ask was still "answerable") is
|
||||
* closed, rather than merely narrowed the way fleetd #329 narrowed the sibling race in {@link
|
||||
* #finishAsyncTask}.
|
||||
*/
|
||||
private volatile Runnable askTimeoutRaceHookForTest;
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #334): install {@link #askTimeoutRaceHookForTest}. Package-private so the
|
||||
* test, in the same package, can reach it without widening any production API.
|
||||
*/
|
||||
void setAskTimeoutRaceHookForTest(Runnable hook) {
|
||||
this.askTimeoutRaceHookForTest = hook;
|
||||
}
|
||||
|
||||
/** A new send must not open a waiter while an async ticket owns this worker's paused turn. */
|
||||
|
||||
@@ -2,6 +2,7 @@ package dev.ltms.fleet.msg;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
@@ -13,6 +14,7 @@ import java.util.Collection;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
@@ -130,7 +132,15 @@ public final class ReplyPushLoop {
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Count one nudge outcome when a registry is wired; a no-op in unit tests. */
|
||||
/**
|
||||
* Count one nudge outcome when a registry is wired; a no-op in unit tests.
|
||||
*
|
||||
* <p>fleetd #365: the {@code "sent"} outcome (renamed from {@code "delivered"}) records only
|
||||
* that {@code agents.send} — a one-way herdr {@code agent.prompt} paste-and-submit — returned
|
||||
* without throwing. Nothing in this loop, or anywhere downstream of it, confirms the pane
|
||||
* actually read or acted on the text; there is no read-receipt concept at this layer. "Sent"
|
||||
* says exactly that; "delivered" claimed more than this call can ever establish.
|
||||
*/
|
||||
private void countNudge(String outcome) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(FleetMetrics.PUSH_NUDGES, "outcome", outcome);
|
||||
@@ -375,6 +385,80 @@ public final class ReplyPushLoop {
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve who to nudge about {@code target}, the way every public entry point below wants it:
|
||||
* {@link PrimaryRegistry#nudgeTargetFor}, but only after checking the delegating lead it names
|
||||
* is still actually there (fleetd #368).
|
||||
*
|
||||
* <p><strong>The bug this closes.</strong> {@code PrimaryRegistry.forgetDelegation} is wired to
|
||||
* exactly one event — a worker's release — because that is the only teardown the daemon already
|
||||
* observes for a session in this map. Nothing removes a binding when the LEAD half goes away: a
|
||||
* lead that is closed, crashes, or is relaunched leaves {@code leadByTarget} entries pointing at
|
||||
* a terminal that no longer exists. {@code nudgeTargetFor} falls back to the single known
|
||||
* primary only when the map holds nothing for {@code target} — a stale non-null entry beats the
|
||||
* fallback every time, which is exactly backwards: the fallback's own javadoc argues it is safe
|
||||
* precisely in the case a stale entry now hides.
|
||||
*
|
||||
* <p><strong>The fix.</strong> Before trusting a recorded delegation, probe the lead the same
|
||||
* way {@link #decide} already does every tick ({@code agents.status}) — cheap, since it is a
|
||||
* local herdr round-trip, and it is the same signal {@code AgentControl.paneByTerminal} already
|
||||
* trusts to tell a genuinely dead target from a live one. A lead that fails the probe is treated
|
||||
* as if it had never been recorded: the stale entry is forgotten (self-healing, exactly like
|
||||
* {@code AgentControl.paneByTerminal} already does on {@code agent_not_found}) and resolution is
|
||||
* retried, which now reaches the fallback {@code nudgeTargetFor} was built to reach — the same
|
||||
* empty-map state its javadoc already argues is correct.
|
||||
*
|
||||
* <p><strong>fleetd #368 review — only a positive "gone" reading forgets the binding.</strong>
|
||||
* The first version of this method treated <em>any</em> {@code RuntimeException} from the probe
|
||||
* as death, which is the #359 mistake repeated: a transient socket blip or a codec error on a
|
||||
* perfectly live lead would silently and permanently unbind it, with no re-record ever coming.
|
||||
* That is destructive on one bad reading, exactly what #359 shipped a two-reading guard to avoid
|
||||
* for the analogous lead-tab-liveness question. {@link #isLive} now matches
|
||||
* {@code AgentControl.agentCall}'s own narrower rule (see its {@code agent_not_found} check): only
|
||||
* that specific, affirmative "herdr has no such agent" signal counts as gone. Every other failure
|
||||
* — timeout, transport error, a decode error — is treated as still live and the binding is left
|
||||
* alone, because guessing wrong here is unrecoverable while guessing "live" merely costs one more
|
||||
* retry on the next tick, which {@link #decide} already tolerates.
|
||||
*/
|
||||
private Optional<String> resolveLiveLead(String target) {
|
||||
Optional<String> lead = primaryRegistry.nudgeTargetFor(target);
|
||||
if (lead.isEmpty() || isLive(lead.get())) {
|
||||
return lead;
|
||||
}
|
||||
log.debug("push: lead {} delegated to for {} is no longer live, forgetting the stale binding "
|
||||
+ "and falling back", lead.get(), target);
|
||||
primaryRegistry.forgetDelegation(target);
|
||||
return primaryRegistry.nudgeTargetFor(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code lead} should still be trusted: {@code false} only when herdr affirmatively
|
||||
* reports the terminal gone ({@code agent_not_found}), never on a merely inconclusive failure.
|
||||
*
|
||||
* <p>fleetd #368 review: an earlier version returned {@code false} for any {@code RuntimeException},
|
||||
* which made a transient herdr hiccup on a live lead indistinguishable from the lead actually
|
||||
* being dead — and the caller's response to {@code false} ({@code forgetDelegation}) is
|
||||
* destructive and permanent. Narrowed to the one code {@code AgentControl.agentCall} itself
|
||||
* already treats as a genuine, resolvable absence (see its {@code agent_not_found} handling) —
|
||||
* every other {@code RuntimeException} is treated as "still live" and the binding survives to be
|
||||
* probed again next time, which costs nothing worse than one more retry.
|
||||
*/
|
||||
private boolean isLive(String lead) {
|
||||
try {
|
||||
agents.status(lead);
|
||||
return true;
|
||||
} catch (RuntimeException e) {
|
||||
boolean gone = e instanceof HerdrException he && "agent_not_found".equals(he.code());
|
||||
if (gone) {
|
||||
log.debug("push: lead {} no longer exists ({})", lead, e.toString());
|
||||
} else {
|
||||
log.debug("push: liveness check for lead {} was inconclusive ({}); treating as live "
|
||||
+ "rather than risk destroying a live binding", lead, e.toString());
|
||||
}
|
||||
return !gone;
|
||||
}
|
||||
}
|
||||
|
||||
// --- public entrypoints ----------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@@ -385,7 +469,7 @@ public final class ReplyPushLoop {
|
||||
* backstop until a lead is recorded.
|
||||
*/
|
||||
public void onReplyQueued(String target) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on {}, skipping reminder", target);
|
||||
return;
|
||||
@@ -413,7 +497,7 @@ public final class ReplyPushLoop {
|
||||
* @param failed whether the ticket ended in a failure phase rather than {@code DONE}
|
||||
*/
|
||||
public void onTicketTerminal(String ticket, String target, boolean failed) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on ticket {} (target {}), skipping nudge",
|
||||
ticket, target);
|
||||
@@ -448,7 +532,7 @@ public final class ReplyPushLoop {
|
||||
* @param question the question text
|
||||
*/
|
||||
public void onQuestionOpened(String ticket, String target, String turnId, String question) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on {}'s question (turnId {}), skipping nudge",
|
||||
target, turnId);
|
||||
@@ -476,7 +560,7 @@ public final class ReplyPushLoop {
|
||||
Collection<String> profiles, int remainingCoolOffSeconds) {
|
||||
Map<String, List<String>> targetsByLead = new ConcurrentHashMap<>();
|
||||
for (String target : targets) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.warn("push: backend incident {} has no known lead for target {}", incidentId, target);
|
||||
continue;
|
||||
@@ -498,7 +582,7 @@ public final class ReplyPushLoop {
|
||||
* Without an owning lead, emit a warning because no control can act on the target.
|
||||
*/
|
||||
public void onBackendTargetUnmapped(String target, String reason) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.warn("push: backend target {} could not map to a credential: {}", target, reason);
|
||||
return;
|
||||
@@ -654,7 +738,7 @@ public final class ReplyPushLoop {
|
||||
lead, replyReminderCount + 1, maxReminders, ticketReminderCount + 1, maxReminders,
|
||||
questionReminderCount + 1, maxReminders,
|
||||
replyTargets.size(), tickets.size(), questions.size());
|
||||
countNudge("delivered");
|
||||
countNudge("sent");
|
||||
for (PendingIncident incident : incidents) {
|
||||
if (pendingIncidents.remove(incident.key(), incident)) {
|
||||
deliveredIncidents.add(incident.key());
|
||||
|
||||
@@ -29,4 +29,9 @@ public record SpawnRequest(String profileName, String requestedCwd, String calle
|
||||
String sessionName, String resumeSessionId) {
|
||||
this(profileName, requestedCwd, callerCwd, sessionName, resumeSessionId, null);
|
||||
}
|
||||
|
||||
/** Return a copy of this request with {@code profileName} replaced by {@code profile}. */
|
||||
public SpawnRequest withProfile(String profile) {
|
||||
return new SpawnRequest(profile, requestedCwd, callerCwd, sessionName, resumeSessionId, role);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,10 +5,14 @@ import java.util.List;
|
||||
|
||||
/**
|
||||
* Backward-compatible placement: an unqualified spawn always resolves to the configured default
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. This ignores caps and
|
||||
* reachability so that a pre-existing config behaves identically after upgrade.
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. This ignores caps
|
||||
* ({@code maxLoad}) so that a pre-existing config behaves identically after upgrade — capacity
|
||||
* gating for automatic placement is deliberately out of scope for {@code fixed}, exactly as it
|
||||
* always has been. Reachability is a narrower exception (fleetd #315, below): a profile is never
|
||||
* checked for reachability up front, only skipped once it has already failed in <em>this same</em>
|
||||
* spawn call's retry loop — see the unreachable case below.
|
||||
*
|
||||
* <p>Three exceptions walk past the default instead of returning it unconditionally:
|
||||
* <p>Four exceptions walk past the default instead of returning it unconditionally:
|
||||
* <ul>
|
||||
* <li>Quarantine (CB-578 stage B): a quarantined default is a credential that just refused on
|
||||
* a usage limit, not a transient capacity or reachability concern.
|
||||
@@ -16,13 +20,21 @@ import java.util.List;
|
||||
* ({@code BackendOutagePolicy}) — a separate, shorter-lived source from quarantine. When a
|
||||
* profile is both quarantined and cooling off, only the quarantine reason is reported
|
||||
* (exhaustion takes priority), matching {@code CompositePeerLauncher}'s explicit-spawn order.
|
||||
* <li>Unreachable (fleetd #315): {@code CompositePeerLauncher.spawn} retries a failed candidate
|
||||
* on the next one and rebuilds the {@link PlacementContext} so {@code ctx.unreachable()}
|
||||
* names every profile that already failed with {@code PeerUnreachableException} in this same
|
||||
* call. Without this check {@code select} kept handing back the same dead default forever —
|
||||
* the retry loop's own comment says "so the policy excludes this profile", and this is what
|
||||
* makes that true for {@code fixed} too, matching {@code weighted}/{@code round-robin}
|
||||
* (both filter on {@code ctx.unreachable()} via {@link PlacementPolicyUtil#available}).
|
||||
* <li>Weight 0 (CB-554): {@code fixed} is still automatic selection, so a profile the operator
|
||||
* marked "never auto-select me" ({@code weight <= 0}) must be skipped here exactly as
|
||||
* {@code weighted}/{@code round-robin} skip it — an explicit {@code fleet_spawn} naming
|
||||
* the profile is unaffected, only this automatic fallback walk.
|
||||
* </ul>
|
||||
* A fleet where nothing is ever quarantined, cooling off, or weight-0 never exercises any of these
|
||||
* paths, so today's behaviour is unchanged.
|
||||
* A fleet where nothing is ever quarantined, cooling off, unreachable, or weight-0 never exercises
|
||||
* any of these paths, so today's behaviour is unchanged — in particular, the very first selection
|
||||
* of a spawn call always sees an empty {@code unreachable} set, so the first choice is untouched.
|
||||
*/
|
||||
final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
|
||||
@@ -30,12 +42,12 @@ final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
public PlacementCandidate select(PlacementContext ctx) {
|
||||
String d = ctx.defaultProfile();
|
||||
if (d != null && !d.isBlank() && !ctx.quarantined().contains(d) && !ctx.coolingOff().contains(d)
|
||||
&& !weightExcluded(ctx, d)) {
|
||||
&& !ctx.unreachable().contains(d) && !weightExcluded(ctx, d)) {
|
||||
return new PlacementCandidate(d, null, 1.0f, null);
|
||||
}
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (!ctx.quarantined().contains(c.profile()) && !ctx.coolingOff().contains(c.profile())
|
||||
&& !c.excluded()) {
|
||||
&& !ctx.unreachable().contains(c.profile()) && !c.excluded()) {
|
||||
return new PlacementCandidate(c.profile(), null, c.weight(), c.maxLoad());
|
||||
}
|
||||
}
|
||||
@@ -44,8 +56,9 @@ final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
// Exhaustion quarantine takes priority: reported only when quarantine is absent, so the
|
||||
// message never claims "cooling off" for a profile that is really backend-exhausted.
|
||||
boolean dCoolingOff = !dQuarantined && ctx.coolingOff().contains(d);
|
||||
boolean dUnreachable = ctx.unreachable().contains(d);
|
||||
boolean dWeightExcluded = weightExcluded(ctx, d);
|
||||
if (dQuarantined || dCoolingOff || dWeightExcluded) {
|
||||
if (dQuarantined || dCoolingOff || dUnreachable || dWeightExcluded) {
|
||||
List<String> reasons = new ArrayList<>();
|
||||
if (dQuarantined) {
|
||||
reasons.add("is quarantined (backend exhausted)");
|
||||
@@ -53,6 +66,9 @@ final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
if (dCoolingOff) {
|
||||
reasons.add("is cooling off after repeated backend errors");
|
||||
}
|
||||
if (dUnreachable) {
|
||||
reasons.add("is unreachable");
|
||||
}
|
||||
if (dWeightExcluded) {
|
||||
reasons.add("has weight 0 (excluded from automatic selection)");
|
||||
}
|
||||
@@ -62,7 +78,7 @@ final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
}
|
||||
if (!ctx.candidates().isEmpty()) {
|
||||
throw new PlacementException("all worker profiles are excluded from automatic "
|
||||
+ "selection (quarantined, cooling off, or weight-0)");
|
||||
+ "selection (quarantined, cooling off, unreachable, or weight-0)");
|
||||
}
|
||||
throw new PlacementException("no worker profiles configured");
|
||||
}
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.Locale;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
|
||||
/**
|
||||
* Holds macOS idle sleep off by keeping a {@code caffeinate -i} child process alive for the life
|
||||
* of the returned {@link SleepAssertion}.
|
||||
*
|
||||
* <p>{@code -i} asserts only against <em>idle</em> sleep — it does not stop the lid closing or an
|
||||
* operator-requested sleep from taking effect. That is deliberate: this class exists to stop an
|
||||
* unattended host from sleeping out from under a member's long turn, never to override the
|
||||
* operator. {@code -s}/{@code -d} (which also block system/display sleep on demand) are
|
||||
* intentionally not used here.
|
||||
*
|
||||
* <p>{@link #acquire()} never throws. It returns {@code null} — a no-op — off macOS, and again if
|
||||
* starting the {@code caffeinate} child fails for any reason (binary missing, process table full,
|
||||
* …); either case is logged once at INFO, not on every occurrence, so a daemon that runs for
|
||||
* weeks with the tool unavailable does not fill its log.
|
||||
*/
|
||||
public final class CaffeinateSleepAssertionMechanism implements SleepAssertionMechanism {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(CaffeinateSleepAssertionMechanism.class);
|
||||
|
||||
private final AtomicBoolean loggedOnce = new AtomicBoolean(false);
|
||||
|
||||
/** {@code true} when running on macOS, the only platform {@code caffeinate} ships on. */
|
||||
public static boolean isSupportedPlatform() {
|
||||
return isSupportedPlatform(System.getProperty("os.name"));
|
||||
}
|
||||
|
||||
/** Package-visible so a test can drive the platform check without touching a real property. */
|
||||
static boolean isSupportedPlatform(String osName) {
|
||||
return osName != null && osName.toLowerCase(Locale.ROOT).contains("mac");
|
||||
}
|
||||
|
||||
@Override
|
||||
public SleepAssertion acquire() {
|
||||
if (!isSupportedPlatform()) {
|
||||
logOnce("not running on macOS (os.name={}); the idle-sleep guard is a no-op on this platform",
|
||||
System.getProperty("os.name"));
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
Process process = new ProcessBuilder("caffeinate", "-i")
|
||||
.redirectOutput(ProcessBuilder.Redirect.DISCARD)
|
||||
.redirectError(ProcessBuilder.Redirect.DISCARD)
|
||||
.start();
|
||||
return new CaffeinateAssertion(process);
|
||||
} catch (IOException | RuntimeException e) {
|
||||
logOnce("could not start 'caffeinate -i' ({}); the host may idle-sleep while members are live",
|
||||
e.toString());
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
private void logOnce(String format, Object arg) {
|
||||
if (loggedOnce.compareAndSet(false, true)) {
|
||||
log.info("idle-sleep guard: " + format, arg);
|
||||
}
|
||||
}
|
||||
|
||||
/** Wraps the live {@code caffeinate} child; {@link #close} force-destroys it, idempotently. */
|
||||
private static final class CaffeinateAssertion implements SleepAssertion {
|
||||
|
||||
private final Process process;
|
||||
|
||||
CaffeinateAssertion(Process process) {
|
||||
this.process = process;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
if (!process.isAlive()) {
|
||||
return;
|
||||
}
|
||||
process.destroy();
|
||||
try {
|
||||
if (!process.waitFor(2, TimeUnit.SECONDS)) {
|
||||
process.destroyForcibly();
|
||||
}
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
process.destroyForcibly();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,105 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.function.IntSupplier;
|
||||
|
||||
/**
|
||||
* Holds an OS-level assertion against idle sleep for exactly as long as at least one fleet
|
||||
* member is live.
|
||||
*
|
||||
* <p><strong>Why this exists:</strong> a fleetd host was measured idle-sleeping after as little
|
||||
* as one minute of inactivity (its {@code pmset -g custom} reports {@code sleep 1} on battery).
|
||||
* Overnight the daemon's AMQP link to the broker dropped 13 times, and cross-checking every drop
|
||||
* minute against {@code pmset -g log} found a sleep or wake event in the same minute or the one
|
||||
* before, every time. The AMQP churn is only the visible symptom — the real problem is that a
|
||||
* member mid-turn freezes with the host, and a long turn with nobody typing is exactly the case
|
||||
* that goes idle.
|
||||
*
|
||||
* <p><strong>How it tracks "live":</strong> this is driven by {@code SessionManager}'s existing
|
||||
* {@code onAcquire}/{@code onRelease} lifecycle hooks (added for CB-520/CB-516, previously wired
|
||||
* to nothing but the reply inbox) rather than a second member count kept in parallel. Wire it as:
|
||||
* <pre>{@code
|
||||
* IdleSleepGuard guard = new IdleSleepGuard(mechanism, sessions::size);
|
||||
* sessions.onAcquire(_ -> guard.recheck());
|
||||
* sessions.onRelease(_ -> guard.recheck());
|
||||
* }</pre>
|
||||
* Every acquire/release event re-reads {@code SessionManager#size()} — the same registry {@code
|
||||
* fleet_list}'s live/capacity numbers are themselves computed from — and only an actual 0→1 or
|
||||
* 1→0 crossing touches the OS. A listener exception is already caught and logged by {@code
|
||||
* SessionManager} itself (it must never let a listener failure block the acquire/release it is
|
||||
* reacting to), so {@link #recheck()} does not need its own top-level try/catch to honor that.
|
||||
*
|
||||
* <p><strong>Failure posture:</strong> every method here is safe to call whether or not {@link
|
||||
* SleepAssertionMechanism#acquire()} actually works. A mechanism that returns {@code null} (wrong
|
||||
* platform, missing tool, spawn failure) simply means this guard never holds anything — it never
|
||||
* throws and never blocks a spawn, a release, or shutdown.
|
||||
*/
|
||||
public final class IdleSleepGuard implements AutoCloseable {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(IdleSleepGuard.class);
|
||||
|
||||
private final SleepAssertionMechanism mechanism;
|
||||
private final IntSupplier liveCount;
|
||||
private final Object lock = new Object();
|
||||
private SleepAssertion held;
|
||||
|
||||
public IdleSleepGuard(SleepAssertionMechanism mechanism, IntSupplier liveCount) {
|
||||
this.mechanism = mechanism;
|
||||
this.liveCount = liveCount;
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-read the live count and acquire or release the held assertion to match: nothing held and
|
||||
* at least one member live ⇒ acquire; something held and no member live ⇒ release. A steady
|
||||
* count (still zero, still positive) is a no-op either way, so a single spawn or release only
|
||||
* ever touches the OS on the crossing, not on every call.
|
||||
*/
|
||||
public void recheck() {
|
||||
synchronized (lock) {
|
||||
int live = liveCount.getAsInt();
|
||||
if (live > 0 && held == null) {
|
||||
held = mechanism.acquire();
|
||||
if (held != null) {
|
||||
log.debug("idle-sleep guard armed: {} live member(s)", live);
|
||||
}
|
||||
} else if (live == 0 && held != null) {
|
||||
releaseHeldLocked();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** {@code true} while an assertion is actually held. Exposed for tests. */
|
||||
boolean isHeld() {
|
||||
synchronized (lock) {
|
||||
return held != null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Release whatever is held, if anything. Idempotent and safe to call at any time, including
|
||||
* repeatedly — a daemon shutdown hook calls this unconditionally so no assertion (and no
|
||||
* {@code caffeinate} child) survives the process, even if the drain that would otherwise have
|
||||
* driven the live count to zero was itself interrupted or threw.
|
||||
*/
|
||||
@Override
|
||||
public void close() {
|
||||
synchronized (lock) {
|
||||
if (held != null) {
|
||||
releaseHeldLocked();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Caller must hold {@link #lock}. */
|
||||
private void releaseHeldLocked() {
|
||||
try {
|
||||
held.close();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("idle-sleep guard: failed to release its assertion cleanly: {}", e.toString());
|
||||
} finally {
|
||||
held = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
/**
|
||||
* A held OS-level assertion against idle sleep. {@link #close} must be idempotent — safe to call
|
||||
* more than once — and must never throw, matching {@link IdleSleepGuard}'s "never break the
|
||||
* fleet" contract.
|
||||
*/
|
||||
public interface SleepAssertion extends AutoCloseable {
|
||||
@Override
|
||||
void close();
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
/**
|
||||
* The OS mechanism {@link IdleSleepGuard} uses to hold and release an idle-sleep assertion. This
|
||||
* is the seam a test exercises instead of the real effect (a live {@code caffeinate} child) — see
|
||||
* {@code IdleSleepGuardTest}.
|
||||
*
|
||||
* <p>Implementations must never throw. Every failure — wrong platform, missing tool, a spawn
|
||||
* error — must show up as {@link #acquire()} returning {@code null}, so a caller can treat "no
|
||||
* assertion held" and "the mechanism could not be used" identically and the fleet keeps running
|
||||
* either way.
|
||||
*/
|
||||
public interface SleepAssertionMechanism {
|
||||
|
||||
/**
|
||||
* Acquire a fresh assertion against idle sleep, or {@code null} when this mechanism is not
|
||||
* usable right now (wrong platform, the tool is missing, the child process could not start).
|
||||
* Never throws.
|
||||
*/
|
||||
SleepAssertion acquire();
|
||||
}
|
||||
@@ -9,14 +9,17 @@ import dev.ltms.fleet.auth.Principal;
|
||||
import dev.ltms.fleet.guard.GuardException;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.inject.MemberPresence;
|
||||
import dev.ltms.fleet.member.MemberCredentialPolicyView;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.session.ShuttingDownException;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.WorktreeRequest;
|
||||
@@ -32,6 +35,7 @@ import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
@@ -45,6 +49,22 @@ import java.util.stream.Collectors;
|
||||
*/
|
||||
public final class FleetApp {
|
||||
|
||||
/** The authorization action the matching route handler hands to {@link #allow}. */
|
||||
static Authz.Action routeAction(String route) {
|
||||
return switch (route) {
|
||||
case "GET /metrics" -> Authz.Action.METRICS;
|
||||
case "POST /members" -> Authz.Action.SPAWN;
|
||||
case "DELETE /members/{paneId}" -> Authz.Action.STOP;
|
||||
case "POST /sessions/{id}/message" -> Authz.Action.SEND;
|
||||
case "POST /sessions/{id}/reply" -> Authz.Action.REPLY;
|
||||
case "GET /sessions/{id}/replies" -> Authz.Action.DRAIN;
|
||||
case "POST /sessions/{id}/ask" -> Authz.Action.ASK;
|
||||
case "GET /sessions", "GET /agents", "GET /members", "GET /profiles",
|
||||
"GET /member-credentials", "GET /sessions/{id}/status", "GET /tasks/{ticket}" -> Authz.Action.READ;
|
||||
default -> throw new IllegalArgumentException("route has no authorization gate: " + route);
|
||||
};
|
||||
}
|
||||
|
||||
/** Default blocking window for a message; kept under typical HTTP idle timeouts. */
|
||||
private static final long DEFAULT_MESSAGE_TIMEOUT_MS = 25_000;
|
||||
private static final long MAX_MESSAGE_TIMEOUT_MS = 120_000;
|
||||
@@ -64,6 +84,16 @@ public final class FleetApp {
|
||||
private final HttpServlet mcpServlet; // MCP Streamable-HTTP endpoint, mounted at /mcp (nullable)
|
||||
private final CallerResolver auth; // CB-501: null → authz not enforced (legacy behaviour)
|
||||
private final Metrics metrics; // CB-502: null → /metrics not exposed
|
||||
// fleetd #111: re-read per request, same hot-reload shape as every other live config read —
|
||||
// absent() (the honest "no policy configured" view) for every constructor that does not wire
|
||||
// a real one, so existing legacy call sites keep building without knowing this field exists.
|
||||
private final Supplier<MemberCredentialPolicyView> memberCredentials;
|
||||
// fleetd #297: the SAME shared sources FleetMcp.profiles/fleet_profiles reads (BackendQuarantine
|
||||
// and BackendOutagePolicy are each one instance for the whole daemon — see Fleetd wiring) so
|
||||
// GET /profiles cannot drift from fleet_profiles about which profile is quarantined/cooling off.
|
||||
// .none() (the honest "feature not wired" view) for every constructor that does not pass one.
|
||||
private final FleetMcp.QuarantineSource quarantine;
|
||||
private final FleetMcp.OutageSource outage;
|
||||
private final ObjectMapper mapper = new ObjectMapper();
|
||||
|
||||
/**
|
||||
@@ -111,6 +141,36 @@ public final class FleetApp {
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable) {
|
||||
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics,
|
||||
deliverable, MemberCredentialPolicyView::absent);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param memberCredentials live {@code memberCredentials:} policy view (fleetd #111), re-read
|
||||
* per request for {@code GET /member-credentials}; production wiring
|
||||
* passes the same hot-reload shape as every other live config read
|
||||
*/
|
||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials) {
|
||||
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics,
|
||||
deliverable, memberCredentials, FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* @param quarantine the SAME {@link FleetMcp.QuarantineSource} instance passed to {@code
|
||||
* FleetMcp} (fleetd #297), so {@code GET /profiles} reports the identical
|
||||
* exhaustion-quarantine facts as {@code fleet_profiles} rather than a second,
|
||||
* independently-computed copy
|
||||
* @param outage the SAME {@link FleetMcp.OutageSource} instance passed to {@code FleetMcp} —
|
||||
* see {@code quarantine}; a SEPARATE check from it, never merged in
|
||||
*/
|
||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials,
|
||||
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage) {
|
||||
this.herdr = herdr;
|
||||
this.memberHerdr = memberHerdr != null ? memberHerdr : herdr;
|
||||
this.workers = workers;
|
||||
@@ -120,6 +180,9 @@ public final class FleetApp {
|
||||
this.mcpServlet = mcpServlet;
|
||||
this.auth = auth;
|
||||
this.metrics = metrics;
|
||||
this.memberCredentials = memberCredentials != null ? memberCredentials : MemberCredentialPolicyView::absent;
|
||||
this.quarantine = quarantine != null ? quarantine : FleetMcp.QuarantineSource.none();
|
||||
this.outage = outage != null ? outage : FleetMcp.OutageSource.none();
|
||||
}
|
||||
|
||||
/** Wire routes onto a fresh, unstarted Javalin instance. Caller starts it. */
|
||||
@@ -148,6 +211,7 @@ public final class FleetApp {
|
||||
app.get("/agents", this::agents);
|
||||
app.get("/members", this::listMembers); // CB-304: registry roster + live herdr status
|
||||
app.get("/profiles", this::profiles); // configured backend profiles
|
||||
app.get("/member-credentials", this::memberCredentials); // fleetd #111: policy names + counts, never a value
|
||||
app.post("/members", this::spawnMember); // optional ?role=&profile= or {"role":…,"profile":…}
|
||||
app.delete("/members/{paneId}", this::stopMember);
|
||||
app.post("/sessions/{id}/message", this::sendMessage); // fleet_send (primary; blocking, wait:false, or answer via turnId)
|
||||
@@ -201,7 +265,7 @@ public final class FleetApp {
|
||||
|
||||
/** Prometheus scrape endpoint (CB-502). */
|
||||
private void metrics(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.METRICS, null)) {
|
||||
if (!allow(ctx, routeAction("GET /metrics"), null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).contentType("text/plain; version=0.0.4; charset=utf-8").result(metrics.render());
|
||||
@@ -273,7 +337,7 @@ public final class FleetApp {
|
||||
* member workspace (they live on the member daemon only).
|
||||
*/
|
||||
private void sessions(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /sessions"), null)) {
|
||||
return;
|
||||
}
|
||||
List<Map<String, Object>> out = new ArrayList<>();
|
||||
@@ -298,52 +362,99 @@ public final class FleetApp {
|
||||
|
||||
/** Discovery: every agent herdr tracks, keyed by its Claude session UUID. */
|
||||
private void agents(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /agents"), null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).json(Map.of("agents",
|
||||
workers.list().stream().map(Agent.class::cast).map(FleetApp::view).toList()));
|
||||
try {
|
||||
ctx.status(200).json(Map.of("agents",
|
||||
workers.list().stream().map(Agent.class::cast).map(FleetApp::view).toList()));
|
||||
} catch (HerdrException e) {
|
||||
// fleetd #297: workers.list() reaches herdr — a transport failure must land in the same
|
||||
// {error, detail} envelope every other failure path here uses, not escape as a bare
|
||||
// exception and leave Javalin's default handling to respond outside the JSON contract.
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** CB-304: bridge-owned roster merged with live herdr status by paneId. */
|
||||
private void listMembers(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /members"), null)) {
|
||||
return;
|
||||
}
|
||||
// CB-519: the registry key is a host-unique id, not the pane coordinate — join on terminal.
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.map(Agent.class::cast)
|
||||
.filter(a -> a.terminalId() != null)
|
||||
.collect(Collectors.toMap(Agent::terminalId, Function.identity(), (_, b) -> b));
|
||||
// fleetd #209: this REST roster reports agentSessionId via SessionManager.rosterView, so it
|
||||
// uses the resolving roster read (caller-driven, not a timer) rather than the plain one.
|
||||
List<Map<String, Object>> out = sessions.rosterResolved().stream()
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.terminalId())))
|
||||
.toList();
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
// fleetd #199: the endpoint became /members in the CB-634 rename but the body key stayed
|
||||
// "workers", so a caller that read "members" saw an empty fleet and reported no members at
|
||||
// all. "members" is the canonical key; "workers" stays as a deprecated alias so an existing
|
||||
// REST consumer keeps working — the out-of-band path a lead falls back to when its MCP mount
|
||||
// drops reads this endpoint. Drop the alias once nothing reads it.
|
||||
body.put("members", out);
|
||||
body.put("workers", out);
|
||||
// CB-586: operator visibility for the refs/wip snapshot store without shelling into the
|
||||
// repo — how many snapshot refs exist and roughly what they cost. Present only once a
|
||||
// worktree session has established the repo, so a never-snapshotted fleet reports nothing.
|
||||
sessions.wipRefs().ifPresent(st -> body.put("wipRefs",
|
||||
Map.of("count", st.count(), "costBytes", st.costBytes())));
|
||||
ctx.status(200).json(body);
|
||||
try {
|
||||
// CB-519: the registry key is a host-unique id, not the pane coordinate — join on terminal.
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.map(Agent.class::cast)
|
||||
.filter(a -> a.terminalId() != null)
|
||||
.collect(Collectors.toMap(Agent::terminalId, Function.identity(), (_, b) -> b));
|
||||
// fleetd #209: this REST roster reports agentSessionId via SessionManager.rosterView, so it
|
||||
// uses the resolving roster read (caller-driven, not a timer) rather than the plain one.
|
||||
List<Map<String, Object>> out = sessions.rosterResolved().stream()
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.terminalId())))
|
||||
.toList();
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
// fleetd #199: the endpoint became /members in the CB-634 rename but the body key stayed
|
||||
// "workers", so a caller that read "members" saw an empty fleet and reported no members at
|
||||
// all. "members" is the canonical key; "workers" stays as a deprecated alias so an existing
|
||||
// REST consumer keeps working — the out-of-band path a lead falls back to when its MCP mount
|
||||
// drops reads this endpoint. Drop the alias once nothing reads it.
|
||||
body.put("members", out);
|
||||
body.put("workers", out);
|
||||
// CB-586: operator visibility for the refs/wip snapshot store without shelling into the
|
||||
// repo — how many snapshot refs exist and roughly what they cost. Present only once a
|
||||
// worktree session has established the repo, so a never-snapshotted fleet reports nothing.
|
||||
sessions.wipRefs().ifPresent(st -> body.put("wipRefs",
|
||||
Map.of("count", st.count(), "costBytes", st.costBytes())));
|
||||
ctx.status(200).json(body);
|
||||
} catch (HerdrException e) {
|
||||
// fleetd #297: same reasoning as agents() above — this is the out-of-band roster a lead
|
||||
// falls back to when its MCP mount drops, so it must stay inside the JSON error contract
|
||||
// exactly when herdr is briefly unreachable, not escape as a bare exception.
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** The configured worker profiles and which one a no-argument spawn uses. */
|
||||
/**
|
||||
* The configured worker profiles, which one a no-argument spawn uses, and (fleetd #297) the two
|
||||
* outage states {@code fleet_profiles} already reports: {@code quarantined} (CB-578 stage B —
|
||||
* the backend reported it out of capacity) and {@code coolingOff} (fleetd #201 Unit 5 — the
|
||||
* credential threw repeated non-exhaustion backend errors). Both are read from the SAME shared
|
||||
* {@link FleetMcp.QuarantineSource}/{@link FleetMcp.OutageSource} instances {@code FleetMcp}
|
||||
* reads, never recomputed, so the two doors cannot disagree about which profile is down and why.
|
||||
* Independent checks, so a profile can appear in both maps at once; each map is present only
|
||||
* when at least one profile is in that state.
|
||||
*/
|
||||
private void profiles(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /profiles"), null)) {
|
||||
return;
|
||||
}
|
||||
// fleetd #297: ONE body builder, shared with fleet_profiles. Handing both doors the same
|
||||
// QuarantineSource/OutageSource instances stops them reading different facts; rendering
|
||||
// through the same method stops them reporting those facts differently. Both are needed.
|
||||
ctx.status(200).json(FleetMcp.profilesView(workers, quarantine, outage));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #111 (CB-608): the live {@code memberCredentials:} policy as names and counts —
|
||||
* NEVER a value. The daemon does not hold a credential's value in the first place (only the
|
||||
* name it is configured under), so there is nothing to redact here beyond what {@link
|
||||
* MemberCredentialPolicyView} already omits by construction. This is the source
|
||||
* {@code scripts/probe-member-credentials.sh} reads instead of carrying its own hardcoded
|
||||
* name list, which is exactly what let the list drift silently behind the real policy.
|
||||
*/
|
||||
private void memberCredentials(Context ctx) {
|
||||
if (!allow(ctx, routeAction("GET /member-credentials"), null)) {
|
||||
return;
|
||||
}
|
||||
MemberCredentialPolicyView view = memberCredentials.get();
|
||||
ctx.status(200).json(Map.of(
|
||||
"profiles", workers.profiles(),
|
||||
"default", workers.defaultProfile() == null ? "" : workers.defaultProfile()));
|
||||
"present", view.present(),
|
||||
"policy", view.policy() == null ? "" : view.policy(),
|
||||
"known", view.known(),
|
||||
"allowed", view.allowed(),
|
||||
"knownCount", view.knownCount(),
|
||||
"allowedCount", view.allowedCount(),
|
||||
"blockedCount", view.blockedCount()));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -352,7 +463,7 @@ public final class FleetApp {
|
||||
* the subscription boundary, 400 for an unknown profile.
|
||||
*/
|
||||
private void spawnMember(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.SPAWN, null)) {
|
||||
if (!allow(ctx, routeAction("POST /members"), null)) {
|
||||
return;
|
||||
}
|
||||
String role = ctx.queryParam("role");
|
||||
@@ -391,6 +502,11 @@ public final class FleetApp {
|
||||
ctx.status(201).json(view(member));
|
||||
} catch (GuardException e) {
|
||||
ctx.status(403).json(Map.of("error", "subscription_boundary", "detail", e.getMessage()));
|
||||
} catch (ShuttingDownException e) {
|
||||
// fleetd #308: the daemon's shutdown drain has already started — 503, not a bare 500,
|
||||
// so this reads the same as PlacementException below: valid request, refused because
|
||||
// of a transient daemon state rather than a bad argument.
|
||||
ctx.status(503).json(Map.of("error", "shutting_down", "detail", e.getMessage()));
|
||||
} catch (PlacementException e) {
|
||||
// CB-599: no candidate had capacity (maxLoad, quarantine, or all-exhausted) — a benign,
|
||||
// likely-transient refusal, distinct from "profile does not exist" below. 503: the
|
||||
@@ -400,6 +516,12 @@ public final class FleetApp {
|
||||
ctx.status(400).json(Map.of("error", "unknown_profile", "detail", e.getMessage()));
|
||||
} catch (PeerUnreachableException e) {
|
||||
ctx.status(502).json(Map.of("error", "spawn_timeout", "detail", e.getMessage()));
|
||||
} catch (HerdrException e) {
|
||||
// fleetd #304: not every herdr failure on the spawn path is a readiness timeout, so
|
||||
// PeerUnreachableException above does not cover this. Without this catch the exception
|
||||
// escapes to Javalin's default 500, while fleet_spawn reports the same failure as a
|
||||
// clean named error (FleetMcp.spawn) — the #297 one-door-guarded shape.
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -420,13 +542,29 @@ public final class FleetApp {
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
}
|
||||
|
||||
/** Tear a worker down by pane id. */
|
||||
/**
|
||||
* Tear a worker down by pane id.
|
||||
*
|
||||
* <p>fleetd #304: the {@code HerdrException} catch is not cosmetic. {@code release} deregisters
|
||||
* the session, notifies the release listener and preserves a dirty worktree <em>before</em> it
|
||||
* calls {@code launcher.stop}, so a throw from that stop arrives after the teardown the caller
|
||||
* asked for has already happened. Letting it escape gave Javalin's default 500, which tells the
|
||||
* caller to retry — and the retry finds nothing in the registry, reaches the same stop, and
|
||||
* throws again, so it can never succeed. {@code herdrError} instead answers 404 ("the pane is
|
||||
* gone, stop retrying") or 502 ("herdr is upstream and broken, a retry may help"), matching what
|
||||
* {@code fleet_stop} reports for the same failure.
|
||||
*/
|
||||
private void stopMember(Context ctx) {
|
||||
String paneId = ctx.pathParam("paneId");
|
||||
if (!allow(ctx, Authz.Action.STOP, paneId)) {
|
||||
if (!allow(ctx, routeAction("DELETE /members/{paneId}"), paneId)) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
sessions.release(paneId);
|
||||
} catch (HerdrException e) {
|
||||
herdrError(ctx, e);
|
||||
return;
|
||||
}
|
||||
sessions.release(paneId);
|
||||
ctx.status(204);
|
||||
}
|
||||
|
||||
@@ -438,7 +576,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void sendMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.SEND, id)) {
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/message"), id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
@@ -525,7 +663,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void askMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.ASK, id)) {
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/ask"), id)) {
|
||||
return;
|
||||
}
|
||||
String question;
|
||||
@@ -558,13 +696,17 @@ public final class FleetApp {
|
||||
/**
|
||||
* The worker's structured reply ({@code fleet_reply}) — resolves the blocking send awaiting
|
||||
* on this session, or queues the reply in the inbox when no send is open (CB-307).
|
||||
*
|
||||
* <p>fleetd #365: the response body's {@code delivered} field used to be unconditionally
|
||||
* {@code true} for either case; it now reports whether a send/ticket was actually resolved,
|
||||
* with {@code outcome} naming which (see {@link MessageService.ReplyOutcome}).
|
||||
*/
|
||||
private void replyMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
// The rule that matters: a worker may reply only as itself. Over MCP this was already true
|
||||
// structurally (identity comes from the connection, never an argument); over REST the path
|
||||
// id was simply trusted, so this is where the invariant actually gets enforced.
|
||||
if (!allow(ctx, Authz.Action.REPLY, id)) {
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/reply"), id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
@@ -574,8 +716,26 @@ public final class FleetApp {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "body must be JSON"));
|
||||
return;
|
||||
}
|
||||
messages.reply(id, content);
|
||||
ctx.status(200).json(Map.of("sessionId", id, "delivered", true));
|
||||
// fleetd #302: content is required. `.path("content").asText("")` above turns a missing key
|
||||
// into "" rather than throwing, so without this check an empty/blank reply used to reach
|
||||
// messages.reply(...) and silently resolve the lead's waiter — the same class of bug as the
|
||||
// sibling "content is required" guards on sendMessage/askMessage below, except this one wrote
|
||||
// a WRONG value instead of failing loudly. The check lives in MessageService.reply so both
|
||||
// this door and FleetMcp.reply inherit the same rule; this catch only translates it into the
|
||||
// {error, detail} envelope this file uses everywhere else.
|
||||
MessageService.ReplyOutcome outcome;
|
||||
try {
|
||||
outcome = messages.reply(id, content);
|
||||
} catch (IllegalArgumentException e) {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", e.getMessage()));
|
||||
return;
|
||||
}
|
||||
// fleetd #365: "delivered": true used to be unconditional here, whether the reply resolved
|
||||
// a waiting send or was merely queued in the inbox for a later drain — the same gap
|
||||
// FleetMcp.reply had over MCP. `delivered` now reflects which actually happened, and
|
||||
// `outcome` names the specific case (see MessageService.ReplyOutcome).
|
||||
ctx.status(200).json(Map.of("sessionId", id, "delivered", outcome.delivered(),
|
||||
"outcome", outcome.wireName()));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -585,7 +745,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void drainReplies(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.DRAIN, id)) {
|
||||
if (!allow(ctx, routeAction("GET /sessions/{id}/replies"), id)) {
|
||||
return;
|
||||
}
|
||||
var replies = messages.drainReplies(id);
|
||||
@@ -603,7 +763,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void sessionStatus(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.READ, id)) {
|
||||
if (!allow(ctx, routeAction("GET /sessions/{id}/status"), id)) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
@@ -628,7 +788,7 @@ public final class FleetApp {
|
||||
|
||||
/** Poll an async (wait:false) delegation by ticket. 404 for an unknown/expired ticket. */
|
||||
private void taskStatus(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /tasks/{ticket}"), null)) {
|
||||
return;
|
||||
}
|
||||
MessageService.TaskView v = messages.poll(ctx.pathParam("ticket"));
|
||||
|
||||
@@ -92,7 +92,19 @@ public final class GitWorktrees implements Worktrees {
|
||||
private final String configuredRoot;
|
||||
/** OS group name for {@link #shareWithGroup} (fleetd #185 stage 3); {@code null} ⇒ feature off. */
|
||||
private final String group;
|
||||
/** Source directory of skill folders for {@link #seedSkills} (fleetd #362, {@code memberSkills:}
|
||||
* in config); {@code null} ⇒ feature off, no worktree is touched beyond today's behaviour. */
|
||||
private final String memberSkillsSource;
|
||||
/** Extra environment merged into every {@code git} subprocess this instance runs. Always {@code
|
||||
* Map.of()} from every production constructor. Test seam only (fleetd #362 review fix): lets
|
||||
* {@code GitWorktreesTest} point {@code GIT_CONFIG_GLOBAL} at an isolated temp file so it can
|
||||
* drive the real {@link #add} path against a controlled "operator's global git config" and
|
||||
* prove the excludesFile composition below without ever touching the real machine's config. */
|
||||
private final Map<String, String> gitEnv;
|
||||
private final Consumer<String> afterWorktreeAdded;
|
||||
/** How the initial {@code git worktree add} command runs. Package-private test seam for an
|
||||
* interrupted command after Git has made worktree state. */
|
||||
private final Function<String[], String> worktreeAddRunner;
|
||||
/** How {@link #shareWithGroup}'s processes (git config / chgrp / chmod / find) actually run.
|
||||
* Defaults to the real {@link #exec(String...)}. Package-private test seam so a unit test can
|
||||
* prove "no group configured ⇒ zero processes spawned" and inspect exactly what a configured
|
||||
@@ -122,7 +134,20 @@ public final class GitWorktrees implements Worktrees {
|
||||
* config); null/blank ⇒ {@link #shareWithGroup} is a no-op.
|
||||
*/
|
||||
public GitWorktrees(String configuredRoot, String group) {
|
||||
this(configuredRoot, group, _ -> {});
|
||||
this(configuredRoot, group, (String) null);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param configuredRoot nullable absolute or relative path; null/blank derives a sibling of
|
||||
* the repo root.
|
||||
* @param group optional OS group name (fleetd #185 stage 3, {@code worktreeGroup:}
|
||||
* in config); null/blank ⇒ {@link #shareWithGroup} is a no-op.
|
||||
* @param memberSkillsSource fleetd #362: optional directory of skill folders ({@code
|
||||
* memberSkills:} in config) copied into every provisioned worktree's
|
||||
* {@code .claude/skills/}; null/blank ⇒ {@link #seedSkills} is a no-op.
|
||||
*/
|
||||
public GitWorktrees(String configuredRoot, String group, String memberSkillsSource) {
|
||||
this(configuredRoot, group, _ -> {}, null, null, memberSkillsSource);
|
||||
}
|
||||
|
||||
/** Test seam for changing a real worktree between its creation and its security check. */
|
||||
@@ -132,7 +157,20 @@ public final class GitWorktrees implements Worktrees {
|
||||
|
||||
/** Test seam combining a configurable {@code group} with {@link #afterWorktreeAdded}. */
|
||||
GitWorktrees(String configuredRoot, String group, Consumer<String> afterWorktreeAdded) {
|
||||
this(configuredRoot, group, afterWorktreeAdded, null);
|
||||
this(configuredRoot, group, afterWorktreeAdded, null, null, null);
|
||||
}
|
||||
|
||||
/** Test seam for changing how {@link #shareWithGroup}'s processes run. */
|
||||
GitWorktrees(String configuredRoot, String group, Consumer<String> afterWorktreeAdded,
|
||||
Function<String[], String> shareGroupRunner) {
|
||||
this(configuredRoot, group, afterWorktreeAdded, shareGroupRunner, null, null);
|
||||
}
|
||||
|
||||
/** Test seam for changing how the initial {@code git worktree add} command runs, with no
|
||||
* {@code memberSkillsSource} configured. */
|
||||
GitWorktrees(String configuredRoot, String group, Consumer<String> afterWorktreeAdded,
|
||||
Function<String[], String> shareGroupRunner, Function<String[], String> worktreeAddRunner) {
|
||||
this(configuredRoot, group, afterWorktreeAdded, shareGroupRunner, worktreeAddRunner, null);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -141,13 +179,34 @@ public final class GitWorktrees implements Worktrees {
|
||||
* exactly what commands a configured group runs, without a real second OS user/group.
|
||||
*
|
||||
* @param shareGroupRunner {@code null} ⇒ the real {@link #exec(String...)}.
|
||||
* @param worktreeAddRunner {@code null} ⇒ the real {@link #exec(String...)}.
|
||||
* @param memberSkillsSource {@code null}/blank ⇒ {@link #seedSkills} is a no-op.
|
||||
*/
|
||||
GitWorktrees(String configuredRoot, String group, Consumer<String> afterWorktreeAdded,
|
||||
Function<String[], String> shareGroupRunner) {
|
||||
Function<String[], String> shareGroupRunner, Function<String[], String> worktreeAddRunner,
|
||||
String memberSkillsSource) {
|
||||
this(configuredRoot, group, afterWorktreeAdded, shareGroupRunner, worktreeAddRunner,
|
||||
memberSkillsSource, Map.of());
|
||||
}
|
||||
|
||||
/**
|
||||
* Full test seam, plus {@code gitEnv} (fleetd #362 review fix, verification only): extra
|
||||
* environment merged into every {@code git} subprocess this instance runs, so a test can isolate
|
||||
* something like {@code GIT_CONFIG_GLOBAL} from the real machine while still driving the real
|
||||
* {@link #add} path end to end. Every production constructor above delegates here with {@code
|
||||
* Map.of()}.
|
||||
*/
|
||||
GitWorktrees(String configuredRoot, String group, Consumer<String> afterWorktreeAdded,
|
||||
Function<String[], String> shareGroupRunner, Function<String[], String> worktreeAddRunner,
|
||||
String memberSkillsSource, Map<String, String> gitEnv) {
|
||||
this.configuredRoot = configuredRoot;
|
||||
this.group = (group == null || group.isBlank()) ? null : group;
|
||||
this.memberSkillsSource = (memberSkillsSource == null || memberSkillsSource.isBlank())
|
||||
? null : memberSkillsSource;
|
||||
this.gitEnv = gitEnv == null ? Map.of() : gitEnv;
|
||||
this.afterWorktreeAdded = afterWorktreeAdded == null ? _ -> {} : afterWorktreeAdded;
|
||||
this.shareGroupRunner = shareGroupRunner != null ? shareGroupRunner : this::exec;
|
||||
this.worktreeAddRunner = worktreeAddRunner != null ? worktreeAddRunner : this::exec;
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -166,15 +225,71 @@ public final class GitWorktrees implements Worktrees {
|
||||
String wt = path.toAbsolutePath().toString();
|
||||
log.info("adding worktree branch={} path={} base={}", branch, wt, base);
|
||||
removeUserInfoFromHttpsOrigin(repoRoot);
|
||||
exec("git", "-C", repoRoot, "worktree", "add", wt, "-b", branch, base);
|
||||
afterWorktreeAdded.accept(wt);
|
||||
requireCredentialFreeHttpsOrigin(wt);
|
||||
configureEnvironmentCredentialHelper(repoRoot, wt);
|
||||
configureHttpsUrlRewriteForSshOrigin(repoRoot, wt);
|
||||
isolateToolSurface(wt);
|
||||
try {
|
||||
worktreeAddRunner.apply(new String[] {"git", "-C", repoRoot, "worktree", "add", wt, "-b", branch, base});
|
||||
afterWorktreeAdded.accept(wt);
|
||||
requireCredentialFreeHttpsOrigin(wt);
|
||||
configureEnvironmentCredentialHelper(repoRoot, wt);
|
||||
configureHttpsUrlRewriteForSshOrigin(repoRoot, wt);
|
||||
isolateToolSurface(wt);
|
||||
seedSkills(wt);
|
||||
} catch (RuntimeException e) {
|
||||
cleanupAfterAddFailure(repoRoot, wt, branch, e);
|
||||
throw e;
|
||||
}
|
||||
return wt;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code git worktree add} may have created the worktree and its branch by the time it, or any
|
||||
* later step through {@link #isolateToolSurface}, throws. This includes
|
||||
* {@link #requireCredentialFreeHttpsOrigin}, an intended security refusal, not only an IO
|
||||
* accident. A Git-reported {@code worktree add} failure usually creates nothing, but an
|
||||
* interrupted command can leave partial state. Without cleanup, {@code add()} never returns, so its caller
|
||||
* ({@code SessionManager#acquireWithWorktree}) never receives a path to register or clean up:
|
||||
* its local {@code path} stays null, the {@code if (path != null)} guard in its own catch block
|
||||
* never runs, and the worktree directory and branch leak on disk forever with nothing tracking
|
||||
* them (fleetd #274).
|
||||
*
|
||||
* <p>Reuses {@link #remove} — the same {@code git worktree remove --force} path every other
|
||||
* cleanup exit in this class already goes through — rather than a bespoke removal. It
|
||||
* additionally deletes {@code branch}: {@link #remove} alone deliberately leaves a released
|
||||
* session's branch behind (a worker's branch is expected to outlive its worktree, for PRs and
|
||||
* recovery), but a branch that never finished provisioning has no session, no PR, and nothing
|
||||
* else pointing at it, so leaving it behind would just trade one leak for a smaller one. Forced
|
||||
* (`-D`) because the branch is new and unmerged by construction. The worktree is removed first:
|
||||
* a branch checked out by a worktree cannot be deleted until the worktree that holds it is gone.
|
||||
*
|
||||
* <p>Cleanup failure must never mask {@code original} — that is the exception that explains
|
||||
* what actually went wrong — so a failure here is only logged, matching the pattern already
|
||||
* used in {@code SessionManager#acquireWithWorktree}'s own catch block.
|
||||
*/
|
||||
private void cleanupAfterAddFailure(String repoRoot, String worktreePath, String branch, RuntimeException original) {
|
||||
if (!Files.exists(Path.of(worktreePath))) {
|
||||
log.debug("provisioning failed before worktree {} existed; nothing to clean up", worktreePath);
|
||||
return;
|
||||
}
|
||||
log.warn("provisioning failed for branch={} path={}: {} — cleaning up before rethrowing",
|
||||
branch, worktreePath, original.getMessage());
|
||||
try {
|
||||
remove(repoRoot, worktreePath);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to remove leaked worktree {} after provisioning error: {}",
|
||||
worktreePath, cleanup.getMessage());
|
||||
}
|
||||
try {
|
||||
deleteBranch(repoRoot, branch);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to remove leaked branch {} after provisioning error: {}",
|
||||
branch, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void deleteBranch(String repoRoot, String branch) {
|
||||
exec("git", "-C", repoRoot, "branch", "-D", branch);
|
||||
}
|
||||
|
||||
/**
|
||||
* A linked worktree shares its primary checkout's git config. Remove HTTPS user info before
|
||||
* adding one, so a credential accidentally embedded in that config cannot reach the member.
|
||||
@@ -508,6 +623,310 @@ public final class GitWorktrees implements Worktrees {
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #362: copy each skill folder from the configured {@link #memberSkillsSource} directory
|
||||
* into {@code <worktreePath>/.claude/skills/}, so a member spawned against ANY repo — not only
|
||||
* one that already ships its own {@code .claude/skills/} — can load a bridge skill such as
|
||||
* {@code implementer}. Every brief this fleet sends starts with {@code "Load the <name>
|
||||
* skill."}; outside a repo carrying its own copy that line was previously a no-op.
|
||||
*
|
||||
* <p><b>No-op — nothing read, nothing written, nothing logged</b> — when {@link
|
||||
* #memberSkillsSource} is null/blank (today's default), the same off-switch shape as
|
||||
* {@link #shareWithGroup}.
|
||||
*
|
||||
* <p><b>Invariant 1 — a repo's own skill wins.</b> A skill folder already present at
|
||||
* {@code <worktreePath>/.claude/skills/<name>} — because the just-checked-out branch commits its
|
||||
* own copy — is left completely untouched: never overwritten, and never even opened.
|
||||
*
|
||||
* <p><b>Invariant 2 — a seeded skill can never end up in a worker's commit.</b> Every path this
|
||||
* writes is untracked in the target repo (that is the whole reason it is being seeded), so
|
||||
* {@code git status} would otherwise show each one as a new, addable, committable path. The
|
||||
* repo-wide {@code .git/info/exclude} is NOT used for this: measured against a real linked
|
||||
* worktree, that file resolves to the repository's COMMON git dir even from a worktree (the
|
||||
* same file {@link dev.ltms.fleet.member.ClaudeCodeLauncher#writeIdeOverlay writeIdeOverlay}
|
||||
* appends {@code CLAUDE.local.md} to), so an entry written there would hide the seeded skill
|
||||
* from {@code git status} in the PRIMARY's own checkout and every sibling worktree too — not
|
||||
* only this one. Instead, {@link #excludeSeededSkillsFromGitStatus} points {@code
|
||||
* core.excludesFile} at a file scoped {@code --worktree} (the same {@code
|
||||
* extensions.worktreeConfig} mechanism {@link #configureEnvironmentCredentialHelper} already
|
||||
* relies on) that itself lives under this worktree's own private git dir ({@code
|
||||
* .git/worktrees/<nonce>/}, OUTSIDE the working tree) — invisible to this worktree's {@code git
|
||||
* status} and structurally impossible for this worktree to commit, with no effect on any other
|
||||
* worktree or the primary checkout. Proven with a real {@code git status --porcelain} in
|
||||
* {@code GitWorktreesTest}, not by reasoning.
|
||||
*
|
||||
* <p><b>Compose, don't replace.</b> {@code core.excludesFile} is single-valued: the first cut of
|
||||
* this method pointed it at fleetd's own file with {@code --replace-all}, which SHADOWS whatever
|
||||
* the operator's own (global, or repo-local) {@code core.excludesFile} was already resolving to
|
||||
* inside this worktree, rather than adding to it. Measured concretely: this repo's own {@code
|
||||
* .gitignore} does not ignore {@code target/} — only an operator's global excludesFile does — so
|
||||
* every worker's {@code mvn clean install} would otherwise make {@code target/} appear as
|
||||
* untracked, and {@link #hasUncommitted}'s deliberately-untracked-inclusive {@code git status
|
||||
* --porcelain} (CB-576) would then read every such worktree as dirty forever, so it is never
|
||||
* cleaned up. {@link #excludeSeededSkillsFromGitStatus} now reads whatever {@code
|
||||
* core.excludesFile} resolves to BEFORE writing anything (falling back to git's own documented
|
||||
* default, {@code $XDG_CONFIG_HOME/git/ignore} or {@code $HOME/.config/git/ignore}, when the key
|
||||
* is unset entirely — see {@code gitignore(5)}), and writes that content into its OWN exclude
|
||||
* file ahead of the seeded skill patterns, so every operator-configured pattern keeps applying
|
||||
* inside the seeded worktree exactly as it did before seeding ran.
|
||||
*
|
||||
* <p>Instead of using worktree-scoped-config as an add-then-append (a second key does not exist
|
||||
* for {@code core.excludesFile} — it takes exactly one value), an actual second exclude source
|
||||
* was ruled out because git resolves only ONE {@code core.excludesFile}; concatenating the prior
|
||||
* content into fleetd's own file is what "compose" reduces to for a single-valued key.
|
||||
*
|
||||
* <p>Proven the same way as invariant 2's own leak check: {@code
|
||||
* GitWorktreesTest#seedSkillsComposesWithAnAlreadyEffectiveGlobalExcludesFile} isolates a
|
||||
* synthetic "operator's global config" via {@code GIT_CONFIG_GLOBAL} (never the real machine's),
|
||||
* seeds a skill, and asserts {@code git status --porcelain} is still empty for a file matching
|
||||
* that global config's own ignore pattern.
|
||||
*
|
||||
* <p><b>Invariant 3 — best-effort.</b> A missing/unreadable {@link #memberSkillsSource}, or a
|
||||
* copy/exclude failure, is logged and skipped — it must never fail the spawn, the same contract
|
||||
* {@link #overlayParity} and {@link #isolateToolSurface} already hold.
|
||||
*
|
||||
* <p>Recorded for the worker itself the same way fleetd #134 records {@code
|
||||
* fleet.neutralizedConfig}: {@code fleet.seededSkills} (one value per seeded skill folder) and
|
||||
* {@code fleet.seededSkillsNote}, readable with {@code git config --worktree --get-all
|
||||
* fleet.seededSkills}.
|
||||
*
|
||||
* <p><b>Claude Code specific by construction, not by a backend check here.</b> Only {@code
|
||||
* .claude/skills/<name>/SKILL.md} is a path any launcher reads today (opencode's equivalent is a
|
||||
* different shape under {@code .opencode/agent}, out of scope — see issue #362). This method
|
||||
* only copies files; like {@link #isolateToolSurface} — which neutralizes BOTH {@code .mcp.json}
|
||||
* and {@code opencode.json} unconditionally — it runs the same for every worktree regardless of
|
||||
* which backend ultimately spawns into it, because the backend is not yet chosen at {@link #add}
|
||||
* time. A seeded {@code .claude/skills/} directory in an opencode member's worktree is simply
|
||||
* never read by that launcher.
|
||||
*/
|
||||
private void seedSkills(String worktreePath) {
|
||||
if (memberSkillsSource == null) {
|
||||
return;
|
||||
}
|
||||
Path source = Path.of(memberSkillsSource).toAbsolutePath().normalize();
|
||||
if (!Files.isDirectory(source)) {
|
||||
log.warn("memberSkills source '{}' is not a directory — skipping skill seeding for worktree {}",
|
||||
source, worktreePath);
|
||||
return;
|
||||
}
|
||||
Path skillsRoot = Path.of(worktreePath).resolve(".claude").resolve("skills");
|
||||
List<String> seeded = new ArrayList<>();
|
||||
List<String> kept = new ArrayList<>();
|
||||
try (var candidates = Files.list(source)) {
|
||||
for (Path candidate : candidates
|
||||
.filter(Files::isDirectory)
|
||||
.filter(p -> !p.getFileName().toString().startsWith("."))
|
||||
.sorted()
|
||||
.toList()) {
|
||||
String name = candidate.getFileName().toString();
|
||||
Path dst = skillsRoot.resolve(name);
|
||||
if (Files.exists(dst)) {
|
||||
kept.add(name);
|
||||
continue;
|
||||
}
|
||||
copySkillDirectory(candidate, dst);
|
||||
seeded.add(name);
|
||||
}
|
||||
} catch (IOException | RuntimeException e) {
|
||||
log.warn("failed to seed skills into worktree {} from memberSkills source '{}': {}",
|
||||
worktreePath, source, e.getMessage());
|
||||
return;
|
||||
}
|
||||
String detail = seeded.isEmpty() ? "" : "seeded: " + String.join(", ", seeded);
|
||||
if (!kept.isEmpty()) {
|
||||
detail += (detail.isEmpty() ? "" : "; ") + "kept the repo's own copy of: " + String.join(", ", kept);
|
||||
}
|
||||
if (detail.isEmpty()) {
|
||||
detail = "no skill folders found under " + source;
|
||||
}
|
||||
log.info("skill seeding: {} of {} candidate(s) from {} into {}/.claude/skills — {}",
|
||||
seeded.size(), seeded.size() + kept.size(), source, worktreePath, detail);
|
||||
if (seeded.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
excludeSeededSkillsFromGitStatus(worktreePath, seeded);
|
||||
recordSeededSkillsForWorker(worktreePath, seeded);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("seeded skill(s) {} into {} but could not hide them from git status: {} — "
|
||||
+ "they may show as untracked; never commit them", seeded, worktreePath, e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/** Recursively copy a skill folder ({@code src}) into a fresh destination ({@code dst}) that
|
||||
* {@link #seedSkills} has already confirmed does not exist, preserving the directory structure
|
||||
* (e.g. {@code implementer/SKILL.md}, {@code implementer/references/...}). */
|
||||
private static void copySkillDirectory(Path src, Path dst) {
|
||||
try (var walk = Files.walk(src)) {
|
||||
for (Path path : walk.sorted().toList()) {
|
||||
Path target = dst.resolve(src.relativize(path).toString());
|
||||
if (Files.isDirectory(path)) {
|
||||
Files.createDirectories(target);
|
||||
} else {
|
||||
Files.createDirectories(target.getParent());
|
||||
Files.copy(path, target, StandardCopyOption.COPY_ATTRIBUTES);
|
||||
}
|
||||
}
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot copy skill directory " + src + " -> " + dst + ": "
|
||||
+ e.getMessage(), e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Make every path in {@code seededSkillNames} (each a name under {@code .claude/skills/})
|
||||
* invisible to {@code git status} in THIS worktree only — see the invariant-2 discussion on
|
||||
* {@link #seedSkills}. Sets {@code core.excludesFile} scoped {@code --worktree} to a file
|
||||
* written under this worktree's own private git dir ({@code git rev-parse
|
||||
* --absolute-git-dir}), which lives outside the working tree, so the exclude file itself can
|
||||
* never be committed either.
|
||||
*
|
||||
* <p><b>Compose, don't replace.</b> {@code core.excludesFile} is single-valued, so pointing it at
|
||||
* fleetd's own file would otherwise SHADOW whatever excludesFile this worktree was already
|
||||
* resolving (an operator's global config, most commonly) rather than add to it — see the
|
||||
* "Compose, don't replace" discussion on {@link #seedSkills}. {@link
|
||||
* #previouslyEffectiveExcludesFileContent} is read BEFORE this method's own {@code --worktree}
|
||||
* write below, so it still sees whatever was effective beforehand; that content is written into
|
||||
* fleetd's own exclude file ahead of the seeded skill patterns, and the worktree-scoped override
|
||||
* then points at that combined file — so every pattern the operator's own configuration already
|
||||
* applied keeps applying, plus the seeded skill paths.
|
||||
*
|
||||
* <p><b>Assumes a fresh worktree — not idempotent.</b> {@link #seedSkills} only ever calls this
|
||||
* from {@link #add}, which always creates a brand-new worktree, so {@code core.excludesFile} is
|
||||
* never already worktree-scoped-set to fleetd's own file when this runs. A hypothetical second
|
||||
* call on the SAME worktree would read fleetd's own already-composed file back as "previously
|
||||
* effective" (worktree scope now wins) and append the seeded patterns a second time — harmless
|
||||
* to {@code git status} (duplicate exclude lines are a no-op), but not something to rely on. No
|
||||
* guard is added for this because the path does not exist today; if a future caller ever seeds
|
||||
* the same worktree twice, it will need one.
|
||||
*/
|
||||
private void excludeSeededSkillsFromGitStatus(String worktreePath, List<String> seededSkillNames) {
|
||||
exec("git", "-C", worktreePath, "config", "extensions.worktreeConfig", "true");
|
||||
String previouslyEffective = previouslyEffectiveExcludesFileContent(worktreePath);
|
||||
String gitDir = exec("git", "-C", worktreePath, "rev-parse", "--absolute-git-dir").trim();
|
||||
Path excludeFile = Path.of(gitDir, "fleet-seeded-skills-exclude");
|
||||
StringBuilder patterns = new StringBuilder();
|
||||
if (!previouslyEffective.isEmpty()) {
|
||||
patterns.append(previouslyEffective);
|
||||
}
|
||||
for (String name : seededSkillNames) {
|
||||
patterns.append("/.claude/skills/").append(name).append('/').append(System.lineSeparator());
|
||||
}
|
||||
try {
|
||||
Files.writeString(excludeFile, patterns.toString());
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot write skills exclude file " + excludeFile + ": "
|
||||
+ e.getMessage(), e);
|
||||
}
|
||||
exec("git", "-C", worktreePath, "config", "--worktree", "--replace-all", "core.excludesFile",
|
||||
excludeFile.toString());
|
||||
}
|
||||
|
||||
/**
|
||||
* The content of whatever {@code core.excludesFile} resolves to for {@code worktreePath} right
|
||||
* now — BEFORE {@link #excludeSeededSkillsFromGitStatus} points that key at fleetd's own file —
|
||||
* so it can be carried forward instead of shadowed. {@code --type=path} makes git itself perform
|
||||
* {@code ~}/{@code ~user} expansion the same way it would when actually reading the key to build
|
||||
* exclude rules, rather than handing back a raw, unexpanded config string.
|
||||
*
|
||||
* <p>When the key is unset entirely (exit code non-zero), falls back to git's own documented
|
||||
* default excludes file — {@code $XDG_CONFIG_HOME/git/ignore}, or {@code
|
||||
* $HOME/.config/git/ignore} when that variable is unset — per {@code gitignore(5)}: git applies
|
||||
* that file even with no {@code core.excludesFile} configured at all, so skipping it here would
|
||||
* silently drop patterns an operator never had to configure to get.
|
||||
*
|
||||
* <p>Never throws: a missing, unreadable, or unresolvable file is treated as "nothing to carry
|
||||
* forward" (empty string) — this is a best-effort read in service of {@link #seedSkills}'s own
|
||||
* invariant 3, not a new way for skill seeding to fail a spawn.
|
||||
*
|
||||
* <p><b>Review fix, finding 2.</b> The XDG-fallback branch below does not go through {@code git}
|
||||
* at all, so a first cut of it read {@code XDG_CONFIG_HOME}/{@code HOME} straight from the JVM's
|
||||
* own environment ({@link System#getenv} / {@code user.home}) — unlike every other value this
|
||||
* class resolves, which goes through a {@code git} subprocess and therefore already honours
|
||||
* {@link #gitEnv}. That meant no test could make this branch hermetic, and on any machine
|
||||
* carrying a real {@code ~/.config/git/ignore} (this repo's own dev machine does), every
|
||||
* skill-seeding test silently composed with that real file — correct in production, but
|
||||
* machine-dependent in the test suite, and a future broader pattern in that real file could
|
||||
* silently change what a seeded worktree's {@code git status} reports depending on whose home
|
||||
* directory ran the test. {@link #resolveEnv} now checks {@link #gitEnv} first for both
|
||||
* variables, falling back to the JVM's real environment only when the seam does not supply
|
||||
* them — production behaviour (empty {@link #gitEnv}) is unchanged, and a test can now isolate
|
||||
* this branch exactly as it already isolates every {@code git} subprocess call.
|
||||
*
|
||||
* <p><b>Snapshot, not a reference.</b> The content below is read once, at seeding time, and
|
||||
* copied into fleetd's own exclude file. If the operator edits their global excludesFile
|
||||
* afterward, an already-seeded worktree keeps the old copy — acceptable for a worktree's
|
||||
* expected lifetime, but worth knowing before reading a stale pattern as a bug.
|
||||
*
|
||||
* @return the file's content, trailing-newline-normalized, or {@code ""} when there is nothing
|
||||
* to compose with.
|
||||
*/
|
||||
private String previouslyEffectiveExcludesFileContent(String worktreePath) {
|
||||
String resolvedPath;
|
||||
if (exitCode("git", "-C", worktreePath, "config", "--get", "--type=path", "core.excludesFile") == 0) {
|
||||
resolvedPath = exec("git", "-C", worktreePath, "config", "--get", "--type=path",
|
||||
"core.excludesFile").trim();
|
||||
} else {
|
||||
String xdgConfigHome = resolveEnv("XDG_CONFIG_HOME");
|
||||
Path fallback = (xdgConfigHome != null && !xdgConfigHome.isBlank())
|
||||
? Path.of(xdgConfigHome, "git", "ignore")
|
||||
: Path.of(resolveHome(), ".config", "git", "ignore");
|
||||
resolvedPath = fallback.toString();
|
||||
}
|
||||
if (resolvedPath.isBlank()) {
|
||||
return "";
|
||||
}
|
||||
Path file = Path.of(resolvedPath);
|
||||
if (!Files.isRegularFile(file) || !Files.isReadable(file)) {
|
||||
return "";
|
||||
}
|
||||
try {
|
||||
String content = Files.readString(file);
|
||||
return content.isBlank() ? "" : content.stripTrailing() + System.lineSeparator();
|
||||
} catch (IOException e) {
|
||||
log.warn("could not read previously-effective excludesFile {} while seeding skills into "
|
||||
+ "{}: {} — its patterns will not carry forward into the seeded worktree",
|
||||
file, worktreePath, e.getMessage());
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve environment variable {@code name} for {@link #previouslyEffectiveExcludesFileContent}'s
|
||||
* XDG fallback, checking {@link #gitEnv} FIRST so a test can isolate this the same way it
|
||||
* already isolates every {@code git} subprocess this class runs, and falling back to the JVM's
|
||||
* real environment only when the seam does not supply it (always the case in production, where
|
||||
* {@link #gitEnv} is {@code Map.of()}).
|
||||
*/
|
||||
private String resolveEnv(String name) {
|
||||
String fromSeam = gitEnv.get(name);
|
||||
return fromSeam != null ? fromSeam : System.getenv(name);
|
||||
}
|
||||
|
||||
/** Same as {@link #resolveEnv(String)}, for {@code HOME} — falls back to {@code user.home}
|
||||
* (rather than {@code System.getenv("HOME")}) when the seam does not supply it, matching this
|
||||
* class's pre-existing behaviour for every other home-directory resolution. */
|
||||
private String resolveHome() {
|
||||
String fromSeam = gitEnv.get("HOME");
|
||||
return fromSeam != null ? fromSeam : System.getProperty("user.home");
|
||||
}
|
||||
|
||||
/**
|
||||
* The worker-readable half of fleetd #362, mirroring {@link #recordNeutralizedConfigForWorker}:
|
||||
* record which skill folders were seeded where the worker itself can read it, without a
|
||||
* working-tree file that would show up in {@code git status}.
|
||||
*/
|
||||
private void recordSeededSkillsForWorker(String worktreePath, List<String> seeded) {
|
||||
exec("git", "-C", worktreePath, "config", "extensions.worktreeConfig", "true");
|
||||
for (String name : seeded) {
|
||||
exec("git", "-C", worktreePath, "config", "--worktree", "--add", "fleet.seededSkills", name);
|
||||
}
|
||||
exec("git", "-C", worktreePath, "config", "--worktree", "fleet.seededSkillsNote",
|
||||
"each fleet.seededSkills value names a skill folder fleetd copied into "
|
||||
+ ".claude/skills/ because this repo did not already ship it; it is excluded "
|
||||
+ "from git status (core.excludesFile, worktree-scoped) and must never be committed");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void remove(String repoRoot, String worktreePath) {
|
||||
Path p = Path.of(worktreePath);
|
||||
@@ -1018,6 +1437,9 @@ public final class GitWorktrees implements Worktrees {
|
||||
Process p;
|
||||
try {
|
||||
ProcessBuilder pb = new ProcessBuilder(command).redirectErrorStream(true);
|
||||
if (!gitEnv.isEmpty()) {
|
||||
pb.environment().putAll(gitEnv);
|
||||
}
|
||||
if (extraEnv != null && !extraEnv.isEmpty()) {
|
||||
pb.environment().putAll(extraEnv);
|
||||
}
|
||||
@@ -1053,7 +1475,11 @@ public final class GitWorktrees implements Worktrees {
|
||||
private int exitCode(String... command) {
|
||||
Process p;
|
||||
try {
|
||||
p = new ProcessBuilder(command).redirectErrorStream(true).start();
|
||||
ProcessBuilder pb = new ProcessBuilder(command).redirectErrorStream(true);
|
||||
if (!gitEnv.isEmpty()) {
|
||||
pb.environment().putAll(gitEnv);
|
||||
}
|
||||
p = pb.start();
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("failed to start " + command[0] + ": " + e.getMessage(), e);
|
||||
}
|
||||
|
||||
@@ -21,6 +21,7 @@ import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.LongSupplier;
|
||||
@@ -61,6 +62,8 @@ public final class SessionManager implements TurnListener {
|
||||
private final LongSupplier nowNanos;
|
||||
private final int contextCap;
|
||||
private final boolean clearAfterTurn;
|
||||
/** Null in production; test seam for the interval before an idle session's conditional release. */
|
||||
private final Consumer<MemberSession> beforeIdleRelease;
|
||||
private volatile MemberLifecycle memberLifecycle = MemberLifecycle.NONE;
|
||||
/**
|
||||
* CB-586: the repo root the fleet actually works in, remembered the first time a worktree
|
||||
@@ -71,6 +74,16 @@ public final class SessionManager implements TurnListener {
|
||||
*/
|
||||
private volatile String fleetRepoRoot;
|
||||
|
||||
/**
|
||||
* fleetd #308: flips true the instant {@link #drainAll} starts, before its registry snapshot
|
||||
* is even taken — so a spawn already in flight sees the refusal as early as a plain flag can
|
||||
* make it. This alone cannot close the race completely: a caller that read {@code false} just
|
||||
* before the flip can still land in the registry after the snapshot. {@link #drainAll}'s
|
||||
* post-loop sweep is what catches that straggler; the two mechanisms are deliberately paired,
|
||||
* see {@link #drainAll}'s javadoc.
|
||||
*/
|
||||
private final AtomicBoolean draining = new AtomicBoolean(false);
|
||||
|
||||
/** CB-520: notified with a terminalId on every acquire; no-op until wired. */
|
||||
private final List<Consumer<String>> acquireListeners = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
/** CB-516: notified with a {@link ReleaseDetail} on every release; no-op until wired. */
|
||||
@@ -102,13 +115,23 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap, boolean clearAfterTurn) {
|
||||
int contextCap, boolean clearAfterTurn) {
|
||||
this(launcher, worktrees, nowNanos, contextCap, clearAfterTurn, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Package-private constructor for a deterministic reap/delivery race test. Production callers
|
||||
* use the constructor above, whose null hook adds no callback or lock to an ordinary reap.
|
||||
*/
|
||||
SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap, boolean clearAfterTurn, Consumer<MemberSession> beforeIdleRelease) {
|
||||
this.launcher = launcher;
|
||||
this.worktrees = worktrees;
|
||||
this.presence = new PresenceFleet(this);
|
||||
this.nowNanos = nowNanos;
|
||||
this.contextCap = contextCap;
|
||||
this.clearAfterTurn = clearAfterTurn;
|
||||
this.beforeIdleRelease = beforeIdleRelease;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -181,6 +204,14 @@ public final class SessionManager implements TurnListener {
|
||||
public MemberSession acquire(String profile, MemberRole role, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt,
|
||||
String sessionName, String resumeSessionId) {
|
||||
// fleetd #308: refuse before anything else runs — no slot reservation, no launcher spawn —
|
||||
// so a caller learns the daemon is going down instead of getting a session drainAll will
|
||||
// never see again. Checked here because every other acquire(...) overload delegates to
|
||||
// this one, so this is the single point every spawn path passes through.
|
||||
if (draining.get()) {
|
||||
throw new ShuttingDownException("fleetd is shutting down; refusing to spawn a session "
|
||||
+ "the shutdown drain would never see");
|
||||
}
|
||||
MemberRole memberRole = (role == null) ? MemberRole.DEV : role;
|
||||
requireResumeCapability(profile, resumeSessionId);
|
||||
// CB-619 / fleetd #123: an explicit profile bypasses placement (CompositePeerLauncher only
|
||||
@@ -272,10 +303,32 @@ public final class SessionManager implements TurnListener {
|
||||
*/
|
||||
private void release(String paneId, ReleaseCause cause) {
|
||||
MemberSession removed = registry.remove(paneId);
|
||||
releaseRemoved(paneId, removed, handles.remove(paneId), cause);
|
||||
}
|
||||
|
||||
/**
|
||||
* Tear a session down only while {@code expected} is still its registry value. A lifecycle
|
||||
* transition replaces the immutable record, so this prevents a reap based on an old READY or
|
||||
* DONE record from stopping a worker that delivery has made BUSY.
|
||||
*/
|
||||
private boolean releaseIfCurrent(MemberSession expected, ReleaseCause cause) {
|
||||
if (!registry.remove(expected.paneId(), expected)) {
|
||||
// A lifecycle transition replaced the record between the caller's check and this remove.
|
||||
// Log it: this race is by definition unobservable otherwise, and a reaper that silently
|
||||
// declines to reap is the hardest kind of behaviour to diagnose after the fact.
|
||||
log.debug("skipping reap of pane={}: its registry record changed after the idle check "
|
||||
+ "(most likely a delivery made it BUSY)", expected.paneId());
|
||||
return false;
|
||||
}
|
||||
releaseRemoved(expected.paneId(), expected, handles.remove(expected.paneId()), cause);
|
||||
return true;
|
||||
}
|
||||
|
||||
private void releaseRemoved(String paneId, MemberSession removed, PeerHandle removedHandle,
|
||||
ReleaseCause cause) {
|
||||
// fleetd #209: remove right alongside the registry entry so a released session's handle is
|
||||
// never leaked — but keep the local reference below, so the id can still be resolved for
|
||||
// the ReleaseDetail this teardown notifies with.
|
||||
PeerHandle removedHandle = handles.remove(paneId);
|
||||
boolean preserveWorktree = cause == ReleaseCause.SHUTDOWN;
|
||||
String snapshotRef = null;
|
||||
if (removed != null) {
|
||||
@@ -334,7 +387,63 @@ public final class SessionManager implements TurnListener {
|
||||
// slot that no longer appears in the roster and can never be reclaimed.
|
||||
launcher.stop(paneId);
|
||||
if (removed != null && !preserveWorktree && removed.worktree() != null) {
|
||||
worktrees.remove(worktrees.repoRoot(removed.cwd()), removed.worktree());
|
||||
// fleetd #316: the `dirty` read above ran while the worker could still write to this
|
||||
// worktree, so a stale `false` must not be trusted to authorise the --force removal
|
||||
// below. Re-read the worktree's state one more time, right here — immediately before
|
||||
// the one step that would destroy it, and only on the path that is actually about to
|
||||
// do that (invariant 4: no second unconditional `git status` on a release that already
|
||||
// decided to preserve). By now `launcher.stop` has returned, so this read reflects
|
||||
// whatever the worker managed to write up to and including its teardown, not whatever
|
||||
// it had written at release-start time.
|
||||
if (dirtyImmediatelyBeforeRemoval(removed)) {
|
||||
preserveWorktree = true;
|
||||
// The pre-stop snapshot above never ran for this session (the pre-stop read said
|
||||
// clean), so this is the only chance to get the newly-discovered work into
|
||||
// refs/wip/* rather than leaving the on-disk preserve as the sole copy. Best-effort,
|
||||
// like every other snapshot attempt — trySnapshot logs and swallows its own failure.
|
||||
String lateSnapshotRef = trySnapshot(removed, cause);
|
||||
log.warn("release {} preserves worktree {} for pane={} terminal={}: it reported "
|
||||
+ "clean before the pane stopped but dirty immediately before removal — the "
|
||||
+ "worker wrote to it during teardown, and --force removing it now would "
|
||||
+ "have destroyed that work{}",
|
||||
cause, removed.worktree(), paneId, removed.terminalId(),
|
||||
lateSnapshotRef == null ? "" : " (snapshotted to refs/wip/" + removed.branch()
|
||||
+ " commit=" + lateSnapshotRef + ")");
|
||||
}
|
||||
}
|
||||
if (removed != null && !preserveWorktree && removed.worktree() != null) {
|
||||
// fleetd #283: this is the one cleanup step in this method that used to be bare. By the
|
||||
// time it runs, the registry entry, the retained handle, and the pane are all already
|
||||
// gone — so a throw here (a stale index lock, a slow filesystem, `remove`'s own 30s exec
|
||||
// timeout) must not escape release(): there is no retry path (a second stop on this
|
||||
// paneId is a no-op), and the caller would otherwise see a "failed stop" for a session
|
||||
// that is in fact fully torn down. Log and swallow, matching every sibling step above.
|
||||
try {
|
||||
worktrees.remove(worktrees.repoRoot(removed.cwd()), removed.worktree());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("failed to remove worktree {} for pane={} terminal={} after release: the "
|
||||
+ "pane is already stopped and the session already deregistered, so this is "
|
||||
+ "not retryable — the directory must be reclaimed manually: {}",
|
||||
removed.worktree(), paneId, removed.terminalId(), e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #316: the read that actually authorises {@code worktrees.remove}, taken with the
|
||||
* worker's pane already stopped. Fails toward preserving (returns {@code true}) on any
|
||||
* exception — the same rule the pre-stop check applies (CB-581): once we can no longer tell
|
||||
* whether the worktree is dirty, preserving costs disk while deleting on a guess can destroy
|
||||
* work that has no other copy.
|
||||
*/
|
||||
private boolean dirtyImmediatelyBeforeRemoval(MemberSession removed) {
|
||||
try {
|
||||
return worktrees.hasUncommitted(removed.worktree());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("release could not re-check worktree {} for pane={} terminal={} immediately "
|
||||
+ "before removal; preserving it rather than risk destroying unsaved work: {}",
|
||||
removed.worktree(), removed.paneId(), removed.terminalId(), e.toString());
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -504,11 +613,26 @@ public final class SessionManager implements TurnListener {
|
||||
log.warn("spawn failed for profile={} role={} branch={} path={}: {}",
|
||||
preResolvedProfile, memberRole, branch, path, e.getMessage());
|
||||
if (path != null) {
|
||||
// fleetd #283: this catch covers every failure AFTER worktrees.add() returned —
|
||||
// overlayParity, shareWithGroup, launcher.spawn itself — so by this point `branch`
|
||||
// was actually created in git. #274 fixed the sibling failure INSIDE add() by having
|
||||
// GitWorktrees.cleanupAfterAddFailure delete both the worktree and the branch it
|
||||
// provisioned; this path removed only the worktree and left the branch orphaned. A
|
||||
// spawn failure here is routine (a quarantined credential, a backend refusal), so
|
||||
// every occurrence leaked a `worker/<slug>-<nonce>` branch nothing ever pointed at
|
||||
// again. Reuse the same Worktrees.deleteBranch GitWorktrees already has, rather than
|
||||
// a second copy of the git command. Best-effort and log-only, like the worktree
|
||||
// removal right above it — neither cleanup step may mask the original exception.
|
||||
try {
|
||||
worktrees.remove(repoRoot, path);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to clean up worktree {} after spawn error: {}", path, cleanup.getMessage());
|
||||
}
|
||||
try {
|
||||
worktrees.deleteBranch(repoRoot, branch);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to clean up branch {} after spawn error: {}", branch, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
@@ -818,14 +942,18 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
long idleNanos = now - s.lastActivityAtNanos();
|
||||
if (idleNanos > idleTtlNanos) {
|
||||
log.debug("reaping idle session terminal={} pane={}: idle {}s exceeds the {}s ttl",
|
||||
s.terminalId(), s.paneId(), TimeUnit.NANOSECONDS.toSeconds(idleNanos),
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos));
|
||||
// CB-581: one session that fails to release must not abort the whole reaping pass —
|
||||
// match drainAll's per-session try/catch so the rest of the roster still gets reaped.
|
||||
try {
|
||||
release(s.paneId());
|
||||
reaped++;
|
||||
if (beforeIdleRelease != null) {
|
||||
beforeIdleRelease.accept(s);
|
||||
}
|
||||
if (releaseIfCurrent(s, ReleaseCause.COMPLETED)) {
|
||||
log.debug("reaping idle session terminal={} pane={}: idle {}s exceeds the {}s ttl",
|
||||
s.terminalId(), s.paneId(), TimeUnit.NANOSECONDS.toSeconds(idleNanos),
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos));
|
||||
reaped++;
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("reap failed for pane={} terminal={} worktree={}; continuing with "
|
||||
+ "remaining sessions", s.paneId(), s.terminalId(), s.worktree(), e);
|
||||
@@ -836,20 +964,60 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
|
||||
/**
|
||||
* Gracefully drain all registered sessions on daemon shutdown. For each session that is
|
||||
* {@code BUSY}, poll up to {@code timeoutNanos} for it to leave {@code BUSY}, then release it
|
||||
* regardless. Non-busy sessions are released immediately. A failure releasing one session is
|
||||
* logged and does not abort the rest.
|
||||
* Gracefully drain all registered sessions on daemon shutdown. Non-busy sessions are released
|
||||
* immediately; a {@code BUSY} one is polled until it leaves {@code BUSY}, then released
|
||||
* regardless. A failure releasing one session is logged and does not abort the rest.
|
||||
*
|
||||
* <p>{@code timeoutNanos} is a budget for the WHOLE drain, not a grace period per session: the
|
||||
* deadline is taken once, before the loop. So the first BUSY session can spend all of it, and a
|
||||
* later BUSY one is then released with no wait at all. That is deliberate. This drain is only
|
||||
* one phase of shutdown — {@code Fleetd} closes the message service, the push loop, the
|
||||
* heartbeat, MCP and the router after it — and the whole sequence has to finish inside
|
||||
* launchd's exit window. A per-session grace would let N busy members drain for N * the
|
||||
* timeout, overrun that window, and get the daemon SIGKILLed part-way through; the members not
|
||||
* yet reached would then get no clean release, no preserved-worktree log, and no snapshot.
|
||||
* Cutting one turn short is the cheaper failure, and it is not silent: an abandoned BUSY
|
||||
* session is preserved, snapshotted, and logged at WARN by {@code logPreservedForShutdown}.
|
||||
*
|
||||
* <p>CB-544: this is a {@link ReleaseCause#SHUTDOWN} release — the worker's pane is stopped
|
||||
* (the process must end) but its worktree is preserved and its path logged. Shutdown is never
|
||||
* a reason to delete a worker's only copy of its uncommitted work. A session still {@code BUSY}
|
||||
* when the timeout expired is abandoned mid-turn and logged loudly so an operator can find its
|
||||
* kept worktree.
|
||||
*
|
||||
* <p>fleetd #308: {@code roster()} is a one-shot snapshot (see its javadoc), and nothing used
|
||||
* to stop a new session from registering after it was taken — {@link #acquire} stayed open for
|
||||
* as long as this drain waited on a {@code BUSY} session, up to the whole {@code timeoutNanos}
|
||||
* budget. Two things close that window, deliberately paired because neither alone is complete:
|
||||
* {@link #draining} is flipped true before the snapshot is even taken, so {@link #acquire}
|
||||
* refuses (invariant 3: loudly, via {@link ShuttingDownException}) as much of the window as a
|
||||
* plain flag can close; and the sweep below re-reads the registry once the initial snapshot has
|
||||
* fully drained and drains whatever a straggler — a caller that read the flag as {@code false}
|
||||
* a moment before it flipped — still managed to register. The sweep shares the same
|
||||
* {@code deadline} rather than getting its own: {@code timeoutNanos} is a budget for the WHOLE
|
||||
* drain (see above), and a straggler must not buy the drain more time than the flag it lost the
|
||||
* race against would have. In the ordinary case the sweep finds nothing and costs one empty
|
||||
* {@link #roster()} call.
|
||||
*/
|
||||
void drainAll(long timeoutNanos) {
|
||||
long deadline = System.nanoTime() + timeoutNanos;
|
||||
for (MemberSession s : roster()) {
|
||||
draining.set(true);
|
||||
drainSnapshot(roster(), deadline);
|
||||
List<MemberSession> stragglers = roster();
|
||||
if (!stragglers.isEmpty()) {
|
||||
log.warn("drain sweep found {} session(s) registered after the drain snapshot was "
|
||||
+ "taken (raced past the shutdown guard); draining them too", stragglers.size());
|
||||
drainSnapshot(stragglers, deadline);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Drain exactly the sessions in {@code snapshot}, waiting out a {@code BUSY} one against the
|
||||
* shared whole-drain {@code deadline} before releasing it. Shared by {@link #drainAll}'s main
|
||||
* pass and its post-loop straggler sweep (fleetd #308) so both honor the same one budget.
|
||||
*/
|
||||
private void drainSnapshot(List<MemberSession> snapshot, long deadline) {
|
||||
for (MemberSession s : snapshot) {
|
||||
try {
|
||||
if (s.state() == MemberSession.State.BUSY) {
|
||||
while (System.nanoTime() < deadline) {
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
package dev.ltms.fleet.session;
|
||||
|
||||
/**
|
||||
* Thrown by {@link SessionManager#acquire} when a spawn is requested after the daemon's shutdown
|
||||
* drain has already begun (fleetd #308).
|
||||
*
|
||||
* <p>{@link SessionManager#drainAll} snapshots the registry once and tears down exactly what is
|
||||
* in that snapshot. A session registered after the snapshot is invisible to the drain loop: its
|
||||
* pane is left running and its worktree is never preserved, and nothing else ever reclaims
|
||||
* either — the daemon's in-memory registry dies with the process. Refusing the spawn here,
|
||||
* loudly, is what stops that session from ever being created in the first place, rather than
|
||||
* silently handing the caller a session the daemon can no longer manage.
|
||||
*/
|
||||
public final class ShuttingDownException extends RuntimeException {
|
||||
public ShuttingDownException(String message) {
|
||||
super(message);
|
||||
}
|
||||
}
|
||||
@@ -11,6 +11,15 @@ public interface Worktrees {
|
||||
/** git -C <repoRoot> worktree remove --force <path>. Idempotent (already-gone tolerated). */
|
||||
void remove(String repoRoot, String worktreePath);
|
||||
|
||||
/**
|
||||
* git -C {@code repoRoot} branch -D {@code branch}. Force-deletes a branch that has no other
|
||||
* owner — used only on the failed-provisioning path (fleetd #274, #283), never on a normal
|
||||
* release: {@link SessionManager#release} deliberately leaves a released session's branch
|
||||
* behind so a lead can still recover the work, and this method must never be called from
|
||||
* that path.
|
||||
*/
|
||||
void deleteBranch(String repoRoot, String branch);
|
||||
|
||||
/**
|
||||
* True when the worktree holds uncommitted changes the bridge cannot see: tracked
|
||||
* modifications, staged files, or untracked files. {@code git status --porcelain} is the
|
||||
|
||||
@@ -226,4 +226,44 @@ class FleetdBackendErrorSinkTest {
|
||||
assertTrue(remaining.isPresent(), "two distinct targets must start a cool-off");
|
||||
assertEquals(1, leadClient.sendCount());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a backend-error session no longer blocks the real maxLoad spawn gate")
|
||||
void backendErrorSessionDoesNotBlockFreshSpawnAtMaxLoad() {
|
||||
SessionManager sessions = capacityLimitedSessions();
|
||||
MemberSession failed = sessions.acquire("terra", null, null, null);
|
||||
|
||||
assertTrue(sessions.onBackendError(failed.terminalId(), "backend exited"));
|
||||
|
||||
MemberSession fresh = sessions.acquire("terra", null, null, null);
|
||||
assertEquals("terra", fresh.profile(), "the real maxLoad gate grants a fresh spawn after a backend error");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a failed session no longer blocks the real maxLoad spawn gate")
|
||||
void failedSessionDoesNotBlockFreshSpawnAtMaxLoad() {
|
||||
SessionManager sessions = capacityLimitedSessions();
|
||||
MemberSession failed = sessions.acquire("terra", null, null, null);
|
||||
|
||||
sessions.onTurnFailed(failed.terminalId());
|
||||
|
||||
MemberSession fresh = sessions.acquire("terra", null, null, null);
|
||||
assertEquals("terra", fresh.profile(), "the real maxLoad gate grants a fresh spawn after a failed turn");
|
||||
}
|
||||
|
||||
private static SessionManager capacityLimitedSessions() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile profile = new FleetConfig.Profile("terra", "http://gx00.gw:8000", "coder",
|
||||
null, "FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, 1.0f, 1);
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("terra", profile);
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), profiles, "terra", _ -> "tok");
|
||||
AtomicReference<SessionManager> sessionsRef = new AtomicReference<>();
|
||||
CompositePeerLauncher workers = new CompositePeerLauncher(List.of(adapter), "terra", profiles,
|
||||
PlacementPolicies.fixed(), name -> Fleetd.liveSessionCount(sessionsRef.get().roster(), name));
|
||||
SessionManager sessions = new SessionManager(workers);
|
||||
sessionsRef.set(sessions);
|
||||
return sessions;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -80,7 +80,7 @@ class FleetdLeadMailboxSelectionTest {
|
||||
@Test
|
||||
void opensTheMailboxWhenAUriAndSelfIdAreConfigured() {
|
||||
var opener = new RecordingOpener();
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null);
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null, null);
|
||||
|
||||
Fleetd.openLeadMailbox(coordinator, Map.of(), opener);
|
||||
|
||||
@@ -94,7 +94,7 @@ class FleetdLeadMailboxSelectionTest {
|
||||
void honoursUriEnvOverALiteralUri() {
|
||||
var opener = new RecordingOpener();
|
||||
var coordinator = new FleetConfig.Coordinator("amqp://stale:stale@old:5672/x", "COORD_URI",
|
||||
"mac-opus", 8);
|
||||
"mac-opus", 8, null);
|
||||
|
||||
Fleetd.openLeadMailbox(coordinator, Map.of("COORD_URI", RESOLVED_URI), opener);
|
||||
|
||||
@@ -105,7 +105,7 @@ class FleetdLeadMailboxSelectionTest {
|
||||
@Test
|
||||
void turnsOffWhenUriEnvDoesNotResolve() {
|
||||
var opener = new RecordingOpener();
|
||||
var coordinator = new FleetConfig.Coordinator(null, "COORD_URI", "mac-opus", null);
|
||||
var coordinator = new FleetConfig.Coordinator(null, "COORD_URI", "mac-opus", null, null);
|
||||
|
||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener));
|
||||
|
||||
@@ -116,7 +116,7 @@ class FleetdLeadMailboxSelectionTest {
|
||||
void warnsAndStaysOffWhenSelfIdIsMissing() {
|
||||
var appender = captureFleetdLogs();
|
||||
var opener = new RecordingOpener();
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, null, null);
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, null, null, null);
|
||||
|
||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener));
|
||||
|
||||
@@ -131,7 +131,7 @@ class FleetdLeadMailboxSelectionTest {
|
||||
var appender = captureFleetdLogs();
|
||||
var opener = new RecordingOpener();
|
||||
opener.unreachable = true;
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null);
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null, null);
|
||||
|
||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener),
|
||||
"a down coordination broker turns the feature off; it must never take the daemon down");
|
||||
|
||||
@@ -0,0 +1,241 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #377: the startup git host line must report the <em>shape</em> of the value a
|
||||
* member receives as GITEA_HOST — set or unset, length, scheme, trailing slash — and the
|
||||
* value itself must never reach the log.
|
||||
*/
|
||||
class GitHostShapeReportTest {
|
||||
|
||||
/** A profile that opts in to the git-forge token, so GITEA_HOST is the var that matters. */
|
||||
private static final String GIT_TOKEN_CONFIG = """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
gitTokenEnv: WORKER_GITEA_TOKEN
|
||||
""";
|
||||
|
||||
private static FleetConfig load(Path dir, String yaml) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml);
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
/**
|
||||
* The level this logger had before {@link #attach()} raised it, so {@link #detach} can put it
|
||||
* back. {@code null} is a real value here — it means "inherit from the parent" — and that is
|
||||
* exactly the state this logger starts in, so it must be restored as {@code null} rather than
|
||||
* as some concrete level.
|
||||
*/
|
||||
private static Level originalLevel;
|
||||
|
||||
/**
|
||||
* fleetd #377: the shape lines are logged at INFO, and {@code logback-test.xml} sets
|
||||
* {@code dev.ltms.fleet} to WARN — so INFO events are dropped by the level check BEFORE any
|
||||
* appender sees them. Attaching an appender is therefore not enough: without raising the level
|
||||
* the list stays empty and every assertion below fails against correct production code. The
|
||||
* sibling report tests do the same thing at each call site (see
|
||||
* {@code MemberTrustModelReportTest}); doing it here keeps it in one place.
|
||||
*/
|
||||
private static ListAppender<ILoggingEvent> attach() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
originalLevel = logger.getLevel();
|
||||
logger.setLevel(Level.INFO);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(originalLevel);
|
||||
}
|
||||
|
||||
private static List<String> messages(ListAppender<ILoggingEvent> appender) {
|
||||
return appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList();
|
||||
}
|
||||
|
||||
@Test
|
||||
void setLineReportsLengthSchemeAndTrailingSlash(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, GIT_TOKEN_CONFIG);
|
||||
String value = "https://git.example.test/";
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportGitHostShape(cfg, Map.of("GITEA_HOST", value));
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
String expected = "startup git host GITEA_HOST: set (profile 'local' gitHostEnv) — "
|
||||
+ "length=" + value.length() + ", startsWithScheme=true, trailingSlash=true";
|
||||
assertTrue(messages(appender).contains(expected),
|
||||
"a value with a scheme and a trailing slash must be reported by shape only: "
|
||||
+ "its length, startsWithScheme=true, trailingSlash=true");
|
||||
}
|
||||
|
||||
@Test
|
||||
void bareHostLineReportsNoSchemeNoTrailingSlash(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, GIT_TOKEN_CONFIG);
|
||||
String value = "git.example.test";
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportGitHostShape(cfg, Map.of("GITEA_HOST", value));
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
String expected = "startup git host GITEA_HOST: set (profile 'local' gitHostEnv) — "
|
||||
+ "length=" + value.length() + ", startsWithScheme=false, trailingSlash=false";
|
||||
assertTrue(messages(appender).contains(expected),
|
||||
"a bare host name must report startsWithScheme=false, trailingSlash=false");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unsetVariableIsLoggedAtInfoAndDoesNotThrow(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, GIT_TOKEN_CONFIG);
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
// GITEA_HOST absent from the map entirely — the unset case must not throw.
|
||||
Fleetd.reportGitHostShape(cfg, Map.of());
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
assertTrue(appender.list.stream().anyMatch(e ->
|
||||
e.getLevel() == Level.INFO
|
||||
&& e.getFormattedMessage()
|
||||
.startsWith("startup git host GITEA_HOST: unset (profile 'local' gitHostEnv)")),
|
||||
"an unset git host is useful information, not an error — say it at INFO level");
|
||||
}
|
||||
|
||||
/**
|
||||
* The important test: the value must NEVER appear in the log output. This test fails if the
|
||||
* line is ever changed to include the value, because it drives the real logging path with a
|
||||
* value that carries a marker no shape field could contain.
|
||||
*/
|
||||
@Test
|
||||
void theValueNeverAppearsInLogOutput(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, GIT_TOKEN_CONFIG);
|
||||
String marker = "never-logged-host-shape-377";
|
||||
String value = "https://" + marker + "/";
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportGitHostShape(cfg, Map.of("GITEA_HOST", value));
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
List<String> msgs = messages(appender);
|
||||
assertFalse(msgs.stream().anyMatch(m -> m.contains(value)),
|
||||
"the full GITEA_HOST value must never reach the log");
|
||||
assertFalse(msgs.stream().anyMatch(m -> m.contains(marker)),
|
||||
"no fragment of the value may reach the log — a line that embeds the value"
|
||||
+ " would leak at least this marker");
|
||||
assertTrue(msgs.stream().anyMatch(m -> m.contains("startsWithScheme=true")
|
||||
&& m.contains("trailingSlash=true")),
|
||||
"the shape must still be reported although the value is not");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unsetReportAlsoCarriesNoValue(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, GIT_TOKEN_CONFIG);
|
||||
// A blank value is not injected either (putIfPresent skips it), so it must report unset
|
||||
// without echoing anything of it.
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportGitHostShape(cfg, Map.of("GITEA_HOST", " "));
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
assertTrue(messages(appender).stream().anyMatch(m ->
|
||||
m.startsWith("startup git host GITEA_HOST: unset")),
|
||||
"a blank value is skipped by the launcher, so the line reports unset");
|
||||
}
|
||||
|
||||
@Test
|
||||
void defaultGitHostEnvIsGITEA_HOST(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, GIT_TOKEN_CONFIG);
|
||||
|
||||
assertEquals(Map.of("GITEA_HOST", List.of("profile 'local' gitHostEnv")),
|
||||
Fleetd.gitHostEnvVars(cfg));
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitGitHostEnvReportsUnderItsOwnName(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
gitTokenEnv: WORKER_GITEA_TOKEN
|
||||
gitHostEnv: MY_FORGE_HOST
|
||||
""");
|
||||
String value = "https://forge.example.test/";
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportGitHostShape(cfg, Map.of("MY_FORGE_HOST", value));
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
assertTrue(messages(appender).contains(
|
||||
"startup git host MY_FORGE_HOST: set (profile 'local' gitHostEnv) — "
|
||||
+ "length=" + value.length()
|
||||
+ ", startsWithScheme=true, trailingSlash=true"),
|
||||
"an explicitly named gitHostEnv is reported under that name");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noGitTokenMeansNoHostLineIsNeeded(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
""");
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportGitHostShape(cfg, Map.of());
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
assertEquals(List.of("startup git host: no profile sets a gitTokenEnv — nothing to check"),
|
||||
messages(appender));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aHostWithAPortIsNotTreatedAsAScheme() {
|
||||
assertFalse(Fleetd.startsWithScheme("git.example.test"));
|
||||
assertFalse(Fleetd.startsWithScheme("git.example.test:3000"));
|
||||
assertFalse(Fleetd.startsWithScheme("git.example.test:3000/"));
|
||||
assertTrue(Fleetd.startsWithScheme("https://git.example.test"));
|
||||
assertTrue(Fleetd.startsWithScheme("https://git.example.test:3000/"));
|
||||
assertTrue(Fleetd.startsWithScheme("ssh://git.example.test"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* fleetd #184: startup must state whether members share fleetd's OS user or use a separate herdr.
|
||||
*/
|
||||
class MemberTrustModelReportTest {
|
||||
|
||||
private static FleetConfig load(Path dir, String yaml) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml);
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
private static ListAppender<ILoggingEvent> attach() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
||||
((Logger) LoggerFactory.getLogger(Fleetd.class)).detachAppender(appender);
|
||||
}
|
||||
|
||||
private static String report(FleetConfig cfg) {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
ch.qos.logback.classic.Level original = logger.getLevel();
|
||||
logger.setLevel(ch.qos.logback.classic.Level.INFO);
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportMemberTrustModel(cfg);
|
||||
} finally {
|
||||
detach(appender);
|
||||
logger.setLevel(original);
|
||||
}
|
||||
return appender.list.getFirst().getFormattedMessage();
|
||||
}
|
||||
|
||||
@Test
|
||||
void unsetMemberHerdrSocketStatesThatMembersAreNotSandboxed(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, "bind:\n host: 127.0.0.1\n port: 8765\n");
|
||||
|
||||
assertEquals("member trust model: members run as the same OS user as fleetd, not in a sandbox. "
|
||||
+ "A member can read any file this user can read, including SSH keys and credential "
|
||||
+ "stores, whatever memberCredentials says. To add a real boundary, route members to "
|
||||
+ "a second herdr under a different OS user with memberHerdrSocket.",
|
||||
report(cfg));
|
||||
}
|
||||
|
||||
@Test
|
||||
void configuredMemberHerdrSocketStatesThatFleetdCannotConfirmTheBoundary(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, "memberHerdrSocket: /tmp/member-herdr.sock\n");
|
||||
|
||||
assertEquals("member trust model: members are routed to a separate herdr through "
|
||||
+ "memberHerdrSocket. fleetd cannot see that herdr's uid, so confirm it runs "
|
||||
+ "as a different OS user before treating it as a boundary.",
|
||||
report(cfg));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import com.tngtech.archunit.base.DescribedPredicate;
|
||||
import com.tngtech.archunit.core.domain.JavaClass;
|
||||
import com.tngtech.archunit.core.domain.JavaClass.Predicates;
|
||||
import com.tngtech.archunit.core.importer.ClassFileImporter;
|
||||
import com.tngtech.archunit.core.importer.ImportOption;
|
||||
import com.tngtech.archunit.library.dependencies.SliceRule;
|
||||
import com.tngtech.archunit.library.dependencies.SlicesRuleDefinition;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
/**
|
||||
* fleetd #131 (CB-627): enforce package boundaries with an ArchUnit test instead of a
|
||||
* Maven module split.
|
||||
*
|
||||
* <p>This test fails the build the moment a NEW cycle appears between the top-level
|
||||
* {@code dev.ltms.fleet.*} packages. Today's cycles are recorded below as explicit,
|
||||
* narrow exceptions: each one ignores dependencies between exactly the two named
|
||||
* packages, in both directions, and nothing else. A cycle through any other pair of
|
||||
* packages -- or a brand new pair -- still fails this test.
|
||||
*
|
||||
* <p><b>Main code only.</b> The import excludes test classes
|
||||
* ({@link ImportOption.Predefined#DO_NOT_INCLUDE_TESTS}). Test code legitimately wires
|
||||
* across many packages for setup and mocking; that is not part of the shipped
|
||||
* architecture this rule protects. Verified: importing test classes too pulls in a much
|
||||
* larger, noisier cycle set -- {@code herdr}, {@code member}, {@code peer}, {@code
|
||||
* config}, {@code guard} and {@code placement} all show up in cycles that disappear the
|
||||
* moment test classes are excluded. Scanning off the classpath via {@code
|
||||
* importPackages(...)} (not a hardcoded {@code target/classes} path) also keeps this
|
||||
* test correct regardless of the working directory the build is invoked from.
|
||||
*
|
||||
* <p><b>No package moves here</b> -- ticket #131 is explicit that removing a cycle is
|
||||
* its own, later PR. See the comment on each exception below for which ticket step
|
||||
* removes it.
|
||||
*/
|
||||
class PackageCyclesTest {
|
||||
|
||||
@Test
|
||||
void packagesAreFreeOfCycles() {
|
||||
var classes = new ClassFileImporter()
|
||||
.withImportOption(ImportOption.Predefined.DO_NOT_INCLUDE_TESTS)
|
||||
.importPackages("dev.ltms.fleet");
|
||||
|
||||
SliceRule rule = SlicesRuleDefinition.slices()
|
||||
.matching("dev.ltms.fleet.(*)..")
|
||||
.should().beFreeOfCycles();
|
||||
|
||||
// fleetd #131 step 1: move ConnectionIdentity so authz stops depending on the
|
||||
// MCP layer. Evidence: auth/CallerResolver.java:3 imports mcp.ConnectionIdentity;
|
||||
// mcp/FleetMcp.java:3-7 imports auth.AuditLog, Authz, CallerResolver, Principal,
|
||||
// Role.
|
||||
rule = ignoreCycle(rule, "auth", "mcp");
|
||||
|
||||
// fleetd #131 step 2: PrimaryRegistry is used by loops in msg; move it, or put
|
||||
// an interface between msg and mcp. Evidence: msg/ReplyPushLoop.java:5 and
|
||||
// msg/LeadHeartbeatLoop.java:5 import mcp.PrimaryRegistry; mcp/FleetMcp.java:15-18
|
||||
// imports msg.LeadChannel, LeadMessage, MessageService, Rendezvous.
|
||||
rule = ignoreCycle(rule, "mcp", "msg");
|
||||
|
||||
// fleetd #131 -- found while implementing this test, NOT one of the ticket's
|
||||
// original three; it names its own follow-up step before removal. Evidence:
|
||||
// inject/CompletionResolver.java:4-5, inject/Injector.java:6 and
|
||||
// inject/TurnListener.java:3 import msg.Rendezvous / msg.TurnToken;
|
||||
// msg/MessageService.java:6 imports inject.Injector.
|
||||
rule = ignoreCycle(rule, "inject", "msg");
|
||||
|
||||
// fleetd #131 -- same as above, its own follow-up. Evidence:
|
||||
// metrics/FleetMetrics.java:3 imports msg.ReplyInbox; msg/MessageService.java:7-8,
|
||||
// msg/LeadHeartbeatLoop.java:6-7 and msg/ReplyPushLoop.java:6-7 import
|
||||
// metrics.FleetMetrics / metrics.Metrics.
|
||||
rule = ignoreCycle(rule, "metrics", "msg");
|
||||
|
||||
// fleetd #131 -- same as above, its own follow-up. Evidence:
|
||||
// session/SessionManager.java:7 imports msg.TurnToken;
|
||||
// msg/LeadHeartbeatLoop.java:8 imports session.MemberSession.
|
||||
rule = ignoreCycle(rule, "msg", "session");
|
||||
|
||||
rule.check(classes);
|
||||
}
|
||||
|
||||
/**
|
||||
* Accepts today's known cycle between two top-level packages, and nothing else.
|
||||
* Ignoring both directions removes exactly this pair from cycle detection; every
|
||||
* other dependency -- including any new one added later, between these same two
|
||||
* packages or any other pair -- is still checked.
|
||||
*/
|
||||
private static SliceRule ignoreCycle(SliceRule rule, String packageA, String packageB) {
|
||||
return rule
|
||||
.ignoreDependency(residesIn(packageA), residesIn(packageB))
|
||||
.ignoreDependency(residesIn(packageB), residesIn(packageA));
|
||||
}
|
||||
|
||||
private static DescribedPredicate<JavaClass> residesIn(String topLevelPackage) {
|
||||
return Predicates.resideInAPackage("dev.ltms.fleet." + topLevelPackage + "..");
|
||||
}
|
||||
}
|
||||
@@ -103,6 +103,54 @@ class CallerResolverTest {
|
||||
assertEquals(Role.PRIMARY, p.role(), "the historical behaviour, now an explicit choice");
|
||||
}
|
||||
|
||||
// ── fleetd #317: an unresolvable caller must never be promoted to the primary ──────────────────
|
||||
// #305 closed the trigger where a resolved pid matched no pane *and* had no ancestry walk to
|
||||
// save it. This is the other trigger PaneLocator's javadoc names: the pid never resolves at
|
||||
// all — LsofPeerPidLookup returns -1 on any failure, including (silently) "lsof found no
|
||||
// match" — so there is no candidate pid for an ancestry walk to even attempt.
|
||||
|
||||
/**
|
||||
* The failing-without-the-fix case. Before #317's fix, {@code c.terminal() == null} was the
|
||||
* only test in the loopback-trust fallback, and an unresolved pid produces exactly that same
|
||||
* {@code null} terminal as a genuine primary — so this caller was handed
|
||||
* {@code Principal.primary(...)}, a real worker's failed lookup becoming indistinguishable from
|
||||
* the lead.
|
||||
*/
|
||||
@Test
|
||||
void aFailedPeerPidLookupIsRefusedNotPromotedToPrimary() {
|
||||
ConnectionIdentity unresolved = new ConnectionIdentity(new PaneLocator(herdr), _ -> -1);
|
||||
Principal p = new CallerResolver(unresolved).resolve("127.0.0.1", 55555, null);
|
||||
|
||||
assertEquals(Role.ANONYMOUS, p.role(),
|
||||
"an unresolvable caller must never be silently promoted to the primary");
|
||||
}
|
||||
|
||||
/**
|
||||
* The companion invariant #317 must not break: a caller whose lookup genuinely succeeded, and
|
||||
* who simply owns no herdr pane — the real primary's own connection — is still the primary.
|
||||
* This is {@link #loopbackTrustTreatsANonWorkerLoopbackCallerAsThePrimary} pinned again here,
|
||||
* named for #317 and placed next to the test it must be distinguished from: same {@code null}
|
||||
* terminal, opposite verdict, because {@code Caller.resolved()} tells them apart.
|
||||
*/
|
||||
@Test
|
||||
void aRealPidThatOwnsNoPaneIsStillThePrimaryNotRefused() {
|
||||
Principal p = new CallerResolver(nonWorkerIdentity()).resolve("127.0.0.1", 55555, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
}
|
||||
|
||||
/** #317 point 4: token mode never consults {@code c.pid()}, so a failed lookup must not change it. */
|
||||
@Test
|
||||
void tokenModeIsUndisturbedByAnUnresolvedLookup() {
|
||||
ConnectionIdentity unresolved = new ConnectionIdentity(new PaneLocator(herdr), _ -> -1);
|
||||
CallerResolver r = new CallerResolver(unresolved, true, "s3cret");
|
||||
|
||||
assertEquals(Role.ANONYMOUS, r.resolve("127.0.0.1", 55555, null).role(),
|
||||
"no credential is still just ANONYMOUS, as before #317 — unchanged by the lookup failing");
|
||||
assertEquals(Role.PRIMARY, r.resolve("127.0.0.1", 55555, "Bearer s3cret").role(),
|
||||
"a valid token still authenticates the primary even though the peer-pid lookup failed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void tokenModeRefusesANonWorkerCallerThatPresentsNoToken() {
|
||||
Principal p = new CallerResolver(nonWorkerIdentity(), true, "s3cret")
|
||||
@@ -413,4 +461,29 @@ class CallerResolverTest {
|
||||
assertThrows(IllegalArgumentException.class, () -> new CallerResolver(id, true, null));
|
||||
assertThrows(IllegalArgumentException.class, () -> new CallerResolver(id, true, " "));
|
||||
}
|
||||
@Test
|
||||
void aWorkerOnAnyLoopbackSourceAddressIsStillAWorkerNotThePrimary() {
|
||||
// fleetd #305: the escalation. ConnectionIdentity used to accept only 127.0.0.1, so a
|
||||
// worker connecting from 127.0.0.2 resolved to no terminal, and this resolver's own
|
||||
// (wider) loopback check then made it the PRIMARY — granting spawn, stop, send and drain.
|
||||
// Measured on the Linux fleet host: binding a source of 127.0.0.2 succeeds there, so the
|
||||
// path is real and not theoretical.
|
||||
CallerResolver r = new CallerResolver(workerIdentity(), false, null);
|
||||
for (String src : new String[]{"127.0.0.1", "127.0.0.2", "127.1.2.3", "::ffff:127.0.0.2"}) {
|
||||
Principal p = r.resolve(src, 55555, null);
|
||||
assertEquals(Role.WORKER, p.role(), "a worker must stay a worker from source " + src);
|
||||
assertEquals("term_a", p.terminal(), "worker terminal from source " + src);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonWorkerOnAnyLoopbackSourceAddressIsStillThePrimary() {
|
||||
// The other direction of the same fix: widening the identity check must not demote a
|
||||
// legitimate same-host primary that happens to connect from another 127.* address.
|
||||
CallerResolver r = new CallerResolver(nonWorkerIdentity(), false, null);
|
||||
for (String src : new String[]{"127.0.0.1", "127.0.0.2", "::ffff:127.0.0.1"}) {
|
||||
assertEquals(Role.PRIMARY, r.resolve(src, 55555, null).role(), "source " + src);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -0,0 +1,202 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.lang.reflect.Constructor;
|
||||
import java.lang.reflect.RecordComponent;
|
||||
import java.util.Arrays;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #323: {@code ConfigRef.sameLaunchSettings} javadoc used to claim it "compares every
|
||||
* component the launcher reads at spawn". It did not — {@code ideProjectDir}, {@code
|
||||
* ideOpenCommand} and {@code autoCompactWindow} were all baked in at daemon startup (see
|
||||
* {@code ClaudeCodeLauncher}/{@code OpenCodeLauncher}) and missing from the comparison, so a reload
|
||||
* that changed only one of them reported "config reloaded" and the running daemon kept the old
|
||||
* value.
|
||||
*
|
||||
* <p>This class is the mechanism the issue asked for: it enumerates every record component of
|
||||
* {@link FleetConfig.Profile} by reflection and proves — by actually mutating a base profile one
|
||||
* field at a time and calling the real method — that each component is either compared by
|
||||
* {@link ConfigRef#sameLaunchSettings} or named in {@link ConfigRef#LAUNCH_SETTINGS_EXCLUDED} with
|
||||
* a reason. A new profile field that is neither fails this test by name, not a hand-maintained list
|
||||
* going stale.
|
||||
*/
|
||||
class ConfigRefProfileCoverageTest {
|
||||
|
||||
private static final RecordComponent[] COMPONENTS = FleetConfig.Profile.class.getRecordComponents();
|
||||
|
||||
/**
|
||||
* One valid, non-blank value per record component — "the a value". None of these trip any
|
||||
* defaulting/normalization in {@code Profile}'s compact constructor (see {@code
|
||||
* FleetConfig.java}), so what goes in is what {@code sameLaunchSettings} sees back out.
|
||||
*/
|
||||
private static final Map<String, Object> BASE = baseValues();
|
||||
|
||||
/** The same shape, each value distinct from {@link #BASE} — "the b value". */
|
||||
private static final Map<String, Object> ALT = altValues();
|
||||
|
||||
private static Map<String, Object> baseValues() {
|
||||
Map<String, Object> v = new LinkedHashMap<>();
|
||||
v.put("profile", "sonnet");
|
||||
v.put("baseUrl", "http://gx00.gw:8000");
|
||||
v.put("model", "sonnet");
|
||||
v.put("configDir", "/config/a");
|
||||
v.put("tokenEnv", "TOKEN_A");
|
||||
v.put("argv", List.of("claude", "--flag-a"));
|
||||
v.put("placement", "tab");
|
||||
v.put("workspace", "workspace-a");
|
||||
v.put("tabLabel", "label-a");
|
||||
v.put("mcpUrl", "http://mcp-a");
|
||||
v.put("cwd", "/cwd/a");
|
||||
v.put("parityOverlay", List.of(".env", ".env.a"));
|
||||
v.put("gitTokenEnv", "GIT_TOKEN_A");
|
||||
v.put("gitHostEnv", "GITEA_HOST_A");
|
||||
v.put("kind", "claude-code");
|
||||
v.put("env", Map.of("K", "A"));
|
||||
v.put("weight", 1.0f);
|
||||
v.put("maxLoad", 5);
|
||||
v.put("subscription", Boolean.TRUE);
|
||||
v.put("exhaustedPattern", "usage limit a");
|
||||
v.put("credentialId", "cred-a");
|
||||
v.put("ideMcpUrl", "http://ide-mcp-a");
|
||||
v.put("ideProjectDir", "modules/a");
|
||||
v.put("ideOpenCommand", "open-cmd-a {dir}");
|
||||
v.put("autoCompactWindow", 150000);
|
||||
v.put("errorPattern", "error a");
|
||||
assertNamesMatchComponents(v);
|
||||
return v;
|
||||
}
|
||||
|
||||
private static Map<String, Object> altValues() {
|
||||
Map<String, Object> v = new LinkedHashMap<>();
|
||||
v.put("profile", "sonnet-b");
|
||||
v.put("baseUrl", "http://gx01.gw:8000");
|
||||
v.put("model", "haiku");
|
||||
v.put("configDir", "/config/b");
|
||||
v.put("tokenEnv", "TOKEN_B");
|
||||
v.put("argv", List.of("claude", "--flag-b"));
|
||||
v.put("placement", "weighted");
|
||||
v.put("workspace", "workspace-b");
|
||||
v.put("tabLabel", "label-b");
|
||||
v.put("mcpUrl", "http://mcp-b");
|
||||
v.put("cwd", "/cwd/b");
|
||||
v.put("parityOverlay", List.of(".env", ".env.b"));
|
||||
v.put("gitTokenEnv", "GIT_TOKEN_B");
|
||||
v.put("gitHostEnv", "GITEA_HOST_B");
|
||||
v.put("kind", "opencode");
|
||||
v.put("env", Map.of("K", "B"));
|
||||
v.put("weight", 2.0f);
|
||||
v.put("maxLoad", 9);
|
||||
v.put("subscription", Boolean.FALSE);
|
||||
v.put("exhaustedPattern", "usage limit b");
|
||||
v.put("credentialId", "cred-b");
|
||||
v.put("ideMcpUrl", "http://ide-mcp-b");
|
||||
v.put("ideProjectDir", "modules/b");
|
||||
v.put("ideOpenCommand", "open-cmd-b {dir}");
|
||||
v.put("autoCompactWindow", 250000);
|
||||
v.put("errorPattern", "error b");
|
||||
assertNamesMatchComponents(v);
|
||||
return v;
|
||||
}
|
||||
|
||||
private static void assertNamesMatchComponents(Map<String, Object> values) {
|
||||
Set<String> componentNames = new TreeSet<>();
|
||||
for (RecordComponent rc : COMPONENTS) {
|
||||
componentNames.add(rc.getName());
|
||||
}
|
||||
assertEquals(componentNames, new TreeSet<>(values.keySet()),
|
||||
"this test's value map has drifted from FleetConfig.Profile's actual components — "
|
||||
+ "update BASE/ALT alongside the record");
|
||||
}
|
||||
|
||||
private static FleetConfig.Profile profileOf(Map<String, Object> values) throws ReflectiveOperationException {
|
||||
Class<?>[] types = Arrays.stream(COMPONENTS).map(RecordComponent::getType).toArray(Class<?>[]::new);
|
||||
Object[] args = Arrays.stream(COMPONENTS)
|
||||
.map(rc -> values.get(rc.getName()))
|
||||
.toArray();
|
||||
Constructor<FleetConfig.Profile> ctor = FleetConfig.Profile.class.getDeclaredConstructor(types);
|
||||
return ctor.newInstance(args);
|
||||
}
|
||||
|
||||
/** {@code BASE} with exactly one named component swapped for its {@code ALT} value. */
|
||||
private static FleetConfig.Profile mutate(String componentName) throws ReflectiveOperationException {
|
||||
Map<String, Object> values = new LinkedHashMap<>(BASE);
|
||||
values.put(componentName, ALT.get(componentName));
|
||||
return profileOf(values);
|
||||
}
|
||||
|
||||
/**
|
||||
* The mechanism fleetd #323 asked for: enumerate {@link FleetConfig.Profile}'s record
|
||||
* components, mutate each non-excluded one, and prove {@code sameLaunchSettings} actually
|
||||
* notices — not just that some hand-maintained list claims it does. Prints the denominator
|
||||
* (total / compared / excluded) the issue required: a checker that cannot state its own
|
||||
* denominator is the failure this repo keeps hitting.
|
||||
*/
|
||||
@Test
|
||||
void sameLaunchSettingsComparesEveryProfileComponentOrExcludesIt() throws ReflectiveOperationException {
|
||||
int total = COMPONENTS.length;
|
||||
Set<String> excluded = ConfigRef.LAUNCH_SETTINGS_EXCLUDED;
|
||||
|
||||
Set<String> allNames = new TreeSet<>();
|
||||
for (RecordComponent rc : COMPONENTS) {
|
||||
allNames.add(rc.getName());
|
||||
}
|
||||
assertTrue(allNames.containsAll(excluded),
|
||||
"ConfigRef.LAUNCH_SETTINGS_EXCLUDED names a component that does not exist on "
|
||||
+ "FleetConfig.Profile — check for a typo: " + excluded);
|
||||
|
||||
// The exclusion set is this mechanism's own escape hatch, so it has to be pinned too.
|
||||
// Found by mutation while verifying fleetd #323: moving autoCompactWindow and
|
||||
// ideOpenCommand OUT of the comparison and INTO the exclusion set left the whole suite
|
||||
// green — the loop below simply skips them, and the denominator assertion still balances.
|
||||
// That is exactly the lazy move a failing coverage test invites, and it silently restores
|
||||
// the #323 bug. Only ideProjectDir and worktreeGroup were saved by a behavioural test in
|
||||
// ConfigRefTest; the other two had none. So: growing this set now requires editing this
|
||||
// line as well, which is a visible, deliberate diff rather than a quiet one.
|
||||
assertEquals(Set.of("weight", "maxLoad", "credentialId"), excluded,
|
||||
"ConfigRef.LAUNCH_SETTINGS_EXCLUDED changed. A component belongs in it ONLY if it "
|
||||
+ "is read live off the config supplier, not baked into a launcher at "
|
||||
+ "startup. If you are adding one to silence this test, that is fleetd #323 "
|
||||
+ "happening again: compare it in sameLaunchSettings instead. If it really "
|
||||
+ "is read live, name where it is read and update this assertion.");
|
||||
|
||||
FleetConfig.Profile base = profileOf(BASE);
|
||||
List<String> uncovered = new java.util.ArrayList<>();
|
||||
int compared = 0;
|
||||
for (RecordComponent rc : COMPONENTS) {
|
||||
String name = rc.getName();
|
||||
if (excluded.contains(name)) {
|
||||
continue;
|
||||
}
|
||||
FleetConfig.Profile mutated = mutate(name);
|
||||
if (ConfigRef.sameLaunchSettings(base, mutated)) {
|
||||
uncovered.add(name);
|
||||
} else {
|
||||
compared++;
|
||||
}
|
||||
}
|
||||
|
||||
System.out.printf(
|
||||
"ConfigRef.sameLaunchSettings coverage — %d Profile components total, %d compared, "
|
||||
+ "%d excluded (%s)%n",
|
||||
total, compared, excluded.size(), excluded);
|
||||
|
||||
assertEquals(List.of(), uncovered,
|
||||
"these FleetConfig.Profile components changed but ConfigRef.sameLaunchSettings "
|
||||
+ "reported no difference — add each one to the comparison (it is read at "
|
||||
+ "spawn and baked in until a restart) or to ConfigRef.LAUNCH_SETTINGS_EXCLUDED "
|
||||
+ "with a reason it is genuinely read live: " + uncovered);
|
||||
assertEquals(total, compared + excluded.size(),
|
||||
"every FleetConfig.Profile record component must be either compared or excluded — "
|
||||
+ total + " components, " + compared + " compared, " + excluded.size()
|
||||
+ " excluded");
|
||||
}
|
||||
}
|
||||
@@ -434,6 +434,364 @@ class ConfigRefTest {
|
||||
assertEquals("provider 5xx", ref.get().profiles().get("sonnet").errorPattern());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #323 instance 1: {@code ideProjectDir} is read at spawn off the frozen profile map
|
||||
* (see {@code ClaudeCodeLauncher}/{@code OpenCodeLauncher}) exactly like {@code model}, but was
|
||||
* missing from {@code sameLaunchSettings} — a reload changing only this field used to report a
|
||||
* bare "config reloaded" and the running daemon kept launching with the old value.
|
||||
*/
|
||||
@Test
|
||||
void changingAProfilesIdeProjectDirIsReportedAsDeferred(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
ideProjectDir: fleetd
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
ideProjectDir: fleetd-renamed
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(1, out.deferred().size(), out.deferred().toString());
|
||||
assertTrue(out.deferred().getFirst().contains("sonnet"), out.deferred().toString());
|
||||
assertTrue(out.deferred().getFirst().contains("launch settings"), out.deferred().toString());
|
||||
// The snapshot still carries the new value — a restart is what makes it take effect.
|
||||
assertEquals("fleetd-renamed", ref.get().profiles().get("sonnet").ideProjectDir());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #323 instance 2: {@code worktreeGroup} is baked into the same {@code GitWorktrees}
|
||||
* as {@code worktreeRoot} (Fleetd.java:251) and never rebuilt, but only {@code worktreeRoot}
|
||||
* was on {@code changedDeferredKeys} — a reload changing only the group reported a bare
|
||||
* "config reloaded" and newly provisioned worktrees kept the old sharing behaviour.
|
||||
*/
|
||||
@Test
|
||||
void changingWorktreeGroupIsReportedAsDeferred(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml("worktreeGroup: devgroup\n"));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, yaml("worktreeGroup: devgroup2\n"));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(java.util.List.of("worktreeGroup"), out.deferred());
|
||||
assertTrue(out.summary().contains("needs a restart") || out.summary().contains("need a restart"),
|
||||
out.summary());
|
||||
// The snapshot still carries the new value — a restart is what makes it take effect.
|
||||
assertEquals("devgroup2", ref.get().worktreeGroup());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #326: {@code primary} is read only off the startup snapshot — {@code Fleetd.java:506,
|
||||
* 519, 520} feed {@code PrimaryRegistry} and {@code ReplyPushLoop} at construction and neither is
|
||||
* rebuilt on reload — but it was missing from {@link ConfigRef#changedDeferredKeys}, so a reload
|
||||
* that only changed the pinned primary terminal reported a bare "config reloaded" while a lead
|
||||
* whose tab no longer matched stayed demoted to worker.
|
||||
*/
|
||||
@Test
|
||||
void changingPrimaryIsReportedAsDeferred(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
primary:
|
||||
terminal: term-a
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
primary:
|
||||
terminal: term-b
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(java.util.List.of("primary"), out.deferred());
|
||||
assertTrue(out.summary().contains("need") && out.summary().contains("restart"), out.summary());
|
||||
// The snapshot still carries the new value — a restart is what makes it take effect.
|
||||
assertEquals("term-b", ref.get().primary().terminal());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #326: {@code configReload} itself is read only at startup ({@code Fleetd.java:679-680})
|
||||
* to decide whether to build a {@code ConfigWatcher} at all, and with what interval — the watcher
|
||||
* that would apply a later change is itself built once, so it is deferred rather than cold (see
|
||||
* {@link ConfigRef}'s class doc: cold means an already-open resource would go inconsistent with
|
||||
* the new value, and there is no such resource here — a running watcher just keeps polling on its
|
||||
* original enabled/interval until a restart, exactly like {@code lifecycle} or {@code guard}).
|
||||
* Before this fix, turning reload off (or changing its interval) through a reload reported a bare
|
||||
* "config reloaded" — the obvious joke the issue names.
|
||||
*/
|
||||
@Test
|
||||
void changingConfigReloadIsReportedAsDeferred(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
configReload:
|
||||
enabled: true
|
||||
intervalSeconds: 10
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
configReload:
|
||||
enabled: false
|
||||
intervalSeconds: 30
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(java.util.List.of("configReload"), out.deferred());
|
||||
assertTrue(out.summary().contains("need") && out.summary().contains("restart"), out.summary());
|
||||
// The snapshot still carries the new value — a restart is what makes it take effect.
|
||||
assertFalse(ref.get().configReload().isEnabled());
|
||||
assertEquals(30, ref.get().configReload().intervalSeconds());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #330: {@code health:} is read both ways — {@code Fleetd.java:556-563} builds the
|
||||
* monitor off the startup snapshot and never rebuilds it, but {@code Fleetd.java:648-650} reads
|
||||
* {@code config.get().health()} live on every {@code fleet_profiles} call. A changed value is
|
||||
* neither purely hot nor purely deferred, so it gets its own {@code split} report naming both
|
||||
* halves rather than a bare "config reloaded" (which would hide the frozen half) or a plain
|
||||
* {@code deferred} entry (which would hide that the coverage string already applied).
|
||||
*/
|
||||
@Test
|
||||
void changingHealthIsReportedAsSplit(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
health:
|
||||
enabled: true
|
||||
intervalSeconds: 30
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
health:
|
||||
enabled: true
|
||||
intervalSeconds: 90
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertTrue(out.deferred().isEmpty(), out.deferred().toString());
|
||||
assertEquals(1, out.split().size(), out.split().toString());
|
||||
assertTrue(out.split().getFirst().startsWith("health:"), out.split().toString());
|
||||
assertTrue(out.split().getFirst().contains("restart"), out.split().toString());
|
||||
assertTrue(out.split().getFirst().contains("live"), out.split().toString());
|
||||
assertTrue(out.summary().contains("partially live"), out.summary());
|
||||
// The snapshot still carries the new value — the monitor itself is what waits for a restart.
|
||||
assertEquals(90, ref.get().health().intervalSeconds());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #330: {@code coordinator:} is the other split key — {@code Fleetd.java:502} opens the
|
||||
* {@code LeadMailbox} off the startup snapshot and never reopens it, but
|
||||
* {@code HerdrPeerLauncher.java:1530} reads {@code config.get().coordinator()} live on every
|
||||
* spawn to keep the broker URI env-var name out of a member's environment.
|
||||
*/
|
||||
@Test
|
||||
void changingCoordinatorIsReportedAsSplit(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
coordinator:
|
||||
selfId: mac-a
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
coordinator:
|
||||
selfId: mac-b
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertTrue(out.deferred().isEmpty(), out.deferred().toString());
|
||||
assertEquals(1, out.split().size(), out.split().toString());
|
||||
assertTrue(out.split().getFirst().startsWith("coordinator:"), out.split().toString());
|
||||
assertTrue(out.split().getFirst().contains("restart"), out.split().toString());
|
||||
assertTrue(out.split().getFirst().contains("live"), out.split().toString());
|
||||
// The snapshot still carries the new value — the LeadMailbox connection is what waits for a
|
||||
// restart; selfId names this daemon's own inbox queue and a peer cannot discover a rename.
|
||||
assertEquals("mac-b", ref.get().coordinator().selfId());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #333: {@code fleet:} is split too — {@code fleet.leaders} is read only at
|
||||
* {@code Fleetd.java:281} to build the {@code LeadTabScanner}'s identity map (fed into
|
||||
* {@code CallerResolver}) and to auto-launch leads via {@code LeadLauncher.ensureLeads()}, and
|
||||
* neither is rebuilt on reload, while the rest of {@code fleet:} (role pools, charters,
|
||||
* tabLabel) is read live through the {@code CompositePeerLauncher} supplier. Before this fix
|
||||
* {@code fleet} sat in the coverage checker's hot-excluded escape hatch, so a reload that
|
||||
* changed only {@code fleet.leaders} reported a bare "config reloaded" — exactly the
|
||||
* under-claim the split class exists to prevent for {@code health}/{@code coordinator}.
|
||||
*/
|
||||
@Test
|
||||
void changingFleetLeadersIsReportedAsSplit(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
leaders:
|
||||
opus:
|
||||
tab: "lead: opus-a"
|
||||
profile: sonnet
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
leaders:
|
||||
opus:
|
||||
tab: "lead: opus-b"
|
||||
profile: sonnet
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertTrue(out.deferred().isEmpty(), out.deferred().toString());
|
||||
assertEquals(1, out.split().size(), out.split().toString());
|
||||
assertTrue(out.split().getFirst().startsWith("fleet:"), out.split().toString());
|
||||
assertTrue(out.split().getFirst().contains("restart"), out.split().toString());
|
||||
assertTrue(out.split().getFirst().contains("live"), out.split().toString());
|
||||
assertTrue(out.summary().contains("partially live"), out.summary());
|
||||
// The snapshot still carries the new value — the LeadTabScanner's identity map and
|
||||
// LeadLauncher's auto-launch are what wait for a restart; a reload rebuilds neither.
|
||||
assertEquals("lead: opus-b", ref.get().fleet().leaders().get("opus").tab());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #333: {@code fleet.leaders} is the ONLY frozen part of {@code fleet:}. A reload that
|
||||
* changes {@code tabLabel} (or charters, or a role pool) without touching {@code fleet.leaders}
|
||||
* must stay fully hot with nothing reported — proving {@link ConfigRef#changedSplitKeys}
|
||||
* compares {@code fleet.leaders} specifically rather than the whole {@code Fleet} record, which
|
||||
* would over-claim "needs a restart" for a change that is genuinely all live (the mirror
|
||||
* mistake of the under-claim this class exists to prevent).
|
||||
*/
|
||||
@Test
|
||||
void changingFleetTabLabelWithoutLeadersStaysFullyHot(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
tabLabel: "{role}: {profile} #{n}"
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
tabLabel: "[{profile}] {role}"
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertTrue(out.deferred().isEmpty(), out.deferred().toString());
|
||||
assertTrue(out.split().isEmpty(), out.split().toString());
|
||||
assertEquals("config reloaded", out.summary());
|
||||
assertEquals("[{profile}] {role}", ref.get().fleet().tabLabel());
|
||||
}
|
||||
|
||||
/**
|
||||
* A split change must not refuse the reload (invariant 2 of fleetd #330) and
|
||||
* {@code Outcome.applied()} must stay {@code true} (invariant 3) — unlike a cold change, the
|
||||
* live half of a split key genuinely took effect, so refusing would throw that away.
|
||||
*/
|
||||
@Test
|
||||
void aSplitChangeDoesNotRefuseTheReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml("coordinator:\n selfId: mac-a\n"));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, yaml("coordinator:\n selfId: mac-b\n"));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertTrue(out.error() == null);
|
||||
assertTrue(out.coldKeys().isEmpty());
|
||||
}
|
||||
|
||||
/**
|
||||
* Both split keys can change in one reload — the report names both, and a caller reading
|
||||
* {@code split} does not have to guess which half of which key already applied.
|
||||
*/
|
||||
@Test
|
||||
void changingBothSplitKeysReportsBoth(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
health:
|
||||
enabled: true
|
||||
coordinator:
|
||||
selfId: mac-a
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
health:
|
||||
enabled: false
|
||||
coordinator:
|
||||
selfId: mac-b
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(2, out.split().size(), out.split().toString());
|
||||
assertTrue(out.split().stream().anyMatch(s -> s.startsWith("health:")), out.split().toString());
|
||||
assertTrue(out.split().stream().anyMatch(s -> s.startsWith("coordinator:")), out.split().toString());
|
||||
}
|
||||
|
||||
/**
|
||||
* A split key and a deferred key changing in the same reload must both show up, each in its own
|
||||
* list — proving the two fields do not step on each other and {@link ConfigRef.Outcome#summary()}
|
||||
* reports both halves of the message.
|
||||
*/
|
||||
@Test
|
||||
void aSplitChangeAndADeferredChangeCoexist(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
coordinator:
|
||||
selfId: mac-a
|
||||
lifecycle:
|
||||
drainTimeoutSeconds: 30
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
coordinator:
|
||||
selfId: mac-b
|
||||
lifecycle:
|
||||
drainTimeoutSeconds: 60
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(java.util.List.of("lifecycle"), out.deferred());
|
||||
assertEquals(1, out.split().size(), out.split().toString());
|
||||
assertTrue(out.split().getFirst().startsWith("coordinator:"), out.split().toString());
|
||||
assertTrue(out.summary().contains("need a restart") || out.summary().contains("needs a restart"),
|
||||
out.summary());
|
||||
assertTrue(out.summary().contains("partially live"), out.summary());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFixedRefHasNoFileAndRefusesToReload() {
|
||||
FleetConfig cfg = new FleetConfig(null, null, null, null, null, null,
|
||||
|
||||
@@ -0,0 +1,171 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.lang.reflect.RecordComponent;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #330: a new top-level {@link FleetConfig} record component must be triaged into a reload
|
||||
* class before it ships, or it repeats fleetd #323 ({@code worktreeGroup} missing from {@code
|
||||
* ConfigRef.changedDeferredKeys}) and fleetd #326 ({@code primary}/{@code configReload} missing the
|
||||
* same way) — a key silently absent from {@link ConfigRef}'s reload machinery, so a reload changing
|
||||
* only that key reports a bare "config reloaded" for a change the running daemon never picked up.
|
||||
*
|
||||
* <p>This is the top-level counterpart of {@link ConfigRefProfileCoverageTest}: instead of
|
||||
* enumerating {@link FleetConfig.Profile}'s record components, it enumerates {@link FleetConfig}'s
|
||||
* own — {@code bind}, {@code health}, {@code memberCredentials}, and so on — and requires each to
|
||||
* fall into exactly one of four homes: {@link ConfigRef#COLD_KEYS}, {@link #DEFERRED_TOP_LEVEL_KEYS}
|
||||
* (compared in {@code ConfigRef.changedDeferredKeys}), {@link ConfigRef#SPLIT_KEYS}, or
|
||||
* {@link #HOT_EXCLUDED_TOP_LEVEL_KEYS} (the escape hatch: read live off the config supplier, so no
|
||||
* reload bookkeeping is needed for it at all).
|
||||
*
|
||||
* <h2>What this checker can and cannot prove</h2>
|
||||
* It proves the record's <em>shape</em> is fully triaged: every one of {@code FleetConfig}'s
|
||||
* components sits in exactly one of the four sets, none sits in two, and the escape hatch
|
||||
* ({@link #HOT_EXCLUDED_TOP_LEVEL_KEYS}) cannot silently grow without a visible diff to this file.
|
||||
* That is what "a new component cannot be added without someone triaging it" means in practice.
|
||||
*
|
||||
* <p>It CANNOT prove that any of the citations are <em>true</em>. "Compared in {@code
|
||||
* changedDeferredKeys}" and "read live off {@code config.get()}" are facts about {@code
|
||||
* ConfigRef.java}, {@code Fleetd.java} and {@code HerdrPeerLauncher.java} that a reflection-only
|
||||
* test over {@code FleetConfig}'s shape has no way to inspect — this test would pass identically
|
||||
* whether or not the cited line still does what the comment next to it says. Trust the citation
|
||||
* because a person read the source (the exact call sites are named next to each set below), not
|
||||
* because this test is green. {@link ConfigRefTest} is what behaviourally proves the deferred and
|
||||
* split keys it covers actually get reported; {@link ConfigRefProfileCoverageTest} does the same,
|
||||
* behaviourally, for {@code FleetConfig.Profile}'s own fields.
|
||||
*/
|
||||
class ConfigRefTopLevelCoverageTest {
|
||||
|
||||
private static final RecordComponent[] COMPONENTS = FleetConfig.class.getRecordComponents();
|
||||
|
||||
/**
|
||||
* Top-level components whose change {@code ConfigRef.changedDeferredKeys} reads and reports on.
|
||||
* Fleetd #337 promoted this out of a hand-maintained copy here into {@link ConfigRef#DEFERRED_KEYS}
|
||||
* itself, the same reason {@link ConfigRef#COLD_KEYS} and {@link ConfigRef#SPLIT_KEYS} are
|
||||
* production constants rather than test-side copies: two lists that are supposed to describe the
|
||||
* same method are exactly the shape that silently drifts apart. {@code spawnReadyTimeoutMs}/
|
||||
* {@code spawnReadyPollMs} are compared together and reported under one combined label
|
||||
* ({@code "spawnReady*"}); {@code profiles} is compared twice over — once for added/removed
|
||||
* profile names, once for an existing profile's launch settings — and that second comparison
|
||||
* excludes {@code weight}/{@code maxLoad}/{@code credentialId} as hot sub-fields, which is what
|
||||
* {@link ConfigRefProfileCoverageTest} exists to keep honest at the sub-field level. {@code
|
||||
* profiles} itself still belongs here, not in the hot-exclusion set below: most of a profile's
|
||||
* fields are NOT read live, so citing "read live off the config supplier" for the whole
|
||||
* top-level key would be false.
|
||||
*/
|
||||
private static final Set<String> DEFERRED_TOP_LEVEL_KEYS = ConfigRef.DEFERRED_KEYS;
|
||||
|
||||
/**
|
||||
* The escape hatch: top-level components with no reload bookkeeping at all, because every read
|
||||
* of them goes live through {@link ConfigRef#get()} rather than off a startup snapshot. A
|
||||
* component belongs here ONLY if that is true — never because adding it here makes this test
|
||||
* pass. fleetd #323 is the cautionary tale for exactly this pattern: the identical hatch on
|
||||
* {@code ConfigRef.LAUNCH_SETTINGS_EXCLUDED} let two profile fields be silently re-broken with
|
||||
* the whole suite green, and it was only caught by mutating the checker itself (see this class's
|
||||
* own mutation test below, and {@code ConfigRefProfileCoverageTest}'s equivalent).
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@code placement} — read live by the placement policy on every spawn (class doc, Hot
|
||||
* bullet; {@code ConfigRefTest.aConsumerHoldingTheRefSeesTheNewValue} proves it
|
||||
* behaviourally).</li>
|
||||
* <li>{@code memberCredentials} — read live at {@code Fleetd.java:198, 205, 729}.</li>
|
||||
* <li>{@code memberLoginShell} — read live at
|
||||
* {@code HerdrPeerLauncher#configuredMemberLoginShell}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>{@code fleet} used to sit here too, on the strength of most of it (role pools, charters,
|
||||
* tabLabel) being read the same live way — but {@code fleet.leaders} inside the same key is
|
||||
* read only at startup and genuinely needs a restart, which is a real gap this escape hatch
|
||||
* cannot represent: it excuses a whole top-level key from reload bookkeeping, and {@code fleet}
|
||||
* needed exactly half of it excused. fleetd #333 moved it into {@link ConfigRef#SPLIT_KEYS}
|
||||
* instead, where {@code changedSplitKeys} reports the frozen half by name. That is the
|
||||
* cautionary tale this test's own javadoc already told: this checker proves the record's shape
|
||||
* is triaged, never that a bucket a key sits in is the right one — a person has to read the
|
||||
* source, which is exactly how fleetd #333 found {@code fleet} sitting in the wrong bucket
|
||||
* while this test stayed green throughout.</p>
|
||||
*/
|
||||
private static final Set<String> HOT_EXCLUDED_TOP_LEVEL_KEYS =
|
||||
Set.of("placement", "memberCredentials", "memberLoginShell");
|
||||
|
||||
@Test
|
||||
void everyTopLevelComponentIsAccountedForInExactlyOneClass() {
|
||||
Set<String> allNames = new TreeSet<>();
|
||||
for (RecordComponent rc : COMPONENTS) {
|
||||
allNames.add(rc.getName());
|
||||
}
|
||||
|
||||
Set<String> cold = ConfigRef.COLD_KEYS;
|
||||
Set<String> split = ConfigRef.SPLIT_KEYS;
|
||||
Set<String> deferred = DEFERRED_TOP_LEVEL_KEYS;
|
||||
Set<String> hot = HOT_EXCLUDED_TOP_LEVEL_KEYS;
|
||||
|
||||
// Typo guard on each set — the same check ConfigRefProfileCoverageTest runs on
|
||||
// LAUNCH_SETTINGS_EXCLUDED. A name that does not exist on FleetConfig is a silent no-op.
|
||||
assertTrue(allNames.containsAll(cold),
|
||||
"ConfigRef.COLD_KEYS names a component that does not exist on FleetConfig: " + cold);
|
||||
assertTrue(allNames.containsAll(split),
|
||||
"ConfigRef.SPLIT_KEYS names a component that does not exist on FleetConfig: " + split);
|
||||
assertTrue(allNames.containsAll(deferred),
|
||||
"DEFERRED_TOP_LEVEL_KEYS names a component that does not exist on FleetConfig: " + deferred);
|
||||
assertTrue(allNames.containsAll(hot),
|
||||
"HOT_EXCLUDED_TOP_LEVEL_KEYS names a component that does not exist on FleetConfig: " + hot);
|
||||
|
||||
// The escape hatch is pinned. Growing it requires editing this line — a visible, deliberate
|
||||
// diff, not a quiet one. See the field javadoc above for what "belongs here" actually means.
|
||||
assertEquals(Set.of("placement", "memberCredentials", "memberLoginShell"), hot,
|
||||
"HOT_EXCLUDED_TOP_LEVEL_KEYS changed. A component belongs here ONLY if it is read "
|
||||
+ "live off the config supplier, never because adding it makes this test "
|
||||
+ "pass. If you are adding one to silence this test, that is fleetd #323 "
|
||||
+ "happening again: account for it in ConfigRef.changedDeferredKeys (or "
|
||||
+ "COLD_KEYS/SPLIT_KEYS) instead. If it really is read live, name where and "
|
||||
+ "update this assertion and the field javadoc together.");
|
||||
|
||||
// No component may sit in two buckets at once — the denominator check below could not catch
|
||||
// that on its own (two buckets double-booking one key still sums to the right total if
|
||||
// another key is simultaneously missing), so check every pair directly and name the culprit.
|
||||
record Bucket(String name, Set<String> keys) {}
|
||||
List<Bucket> buckets = List.of(
|
||||
new Bucket("COLD_KEYS", cold), new Bucket("SPLIT_KEYS", split),
|
||||
new Bucket("DEFERRED_TOP_LEVEL_KEYS", deferred), new Bucket("HOT_EXCLUDED_TOP_LEVEL_KEYS", hot));
|
||||
for (int i = 0; i < buckets.size(); i++) {
|
||||
for (int j = i + 1; j < buckets.size(); j++) {
|
||||
Set<String> overlap = new LinkedHashSet<>(buckets.get(i).keys());
|
||||
overlap.retainAll(buckets.get(j).keys());
|
||||
assertEquals(Set.of(), overlap, "a component is in both " + buckets.get(i).name()
|
||||
+ " and " + buckets.get(j).name() + ": " + overlap);
|
||||
}
|
||||
}
|
||||
|
||||
Set<String> union = new TreeSet<>();
|
||||
union.addAll(cold);
|
||||
union.addAll(split);
|
||||
union.addAll(deferred);
|
||||
union.addAll(hot);
|
||||
|
||||
System.out.printf(
|
||||
"FleetConfig top-level coverage — %d components total: %d cold %s, %d deferred %s, "
|
||||
+ "%d split %s, %d hot-excluded %s%n",
|
||||
allNames.size(), cold.size(), cold, deferred.size(), deferred, split.size(), split,
|
||||
hot.size(), hot);
|
||||
|
||||
Set<String> missing = new TreeSet<>(allNames);
|
||||
missing.removeAll(union);
|
||||
assertEquals(Set.of(), missing,
|
||||
"these FleetConfig components are in none of COLD_KEYS, DEFERRED_TOP_LEVEL_KEYS, "
|
||||
+ "SPLIT_KEYS or HOT_EXCLUDED_TOP_LEVEL_KEYS — triage each one into whichever "
|
||||
+ "actually describes it: " + missing);
|
||||
assertEquals(allNames.size(), cold.size() + deferred.size() + split.size() + hot.size(),
|
||||
"counts don't sum to the component total even though every component was found in "
|
||||
+ "the union — " + allNames.size() + " components, " + cold.size()
|
||||
+ " cold + " + deferred.size() + " deferred + " + split.size() + " split + "
|
||||
+ hot.size() + " hot-excluded");
|
||||
}
|
||||
}
|
||||
+288
@@ -0,0 +1,288 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.lang.reflect.Constructor;
|
||||
import java.lang.reflect.RecordComponent;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Arrays;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* fleetd #333, finding F2: {@link ConfigRefTopLevelCoverageTest} proves every {@link FleetConfig}
|
||||
* top-level component sits in exactly one of {@link ConfigRef#COLD_KEYS}, {@link
|
||||
* ConfigRef#DEFERRED_KEYS}, {@link ConfigRef#SPLIT_KEYS} or the hot-excluded set. It does
|
||||
* <strong>not</strong> prove that a key's membership in one of the first three sets corresponds to
|
||||
* any actual comparison in {@link ConfigRef}: a key can sit in a set with no branch in {@code
|
||||
* changedColdKeys}/{@code changedSplitKeys}/{@code changedDeferredKeys} checking it, and both
|
||||
* {@link ConfigRefTopLevelCoverageTest} and the "kept in step" {@code assert} inside the first two
|
||||
* of those methods stay green, because neither one reads the method body — the coverage test only
|
||||
* reads set membership, and the assert only checks that reported entries are a SUBSET of the set,
|
||||
* never that every set member produced a reported entry. {@code changedDeferredKeys} does not even
|
||||
* have a "kept in step" assert of its own.
|
||||
*
|
||||
* <p>Measured directly, live, while fixing fleetd #333: dropping the {@code coordinator} branch out
|
||||
* of {@code ConfigRef.changedSplitKeys} while leaving {@code "coordinator"} in
|
||||
* {@link ConfigRef#SPLIT_KEYS} left {@link ConfigRefTopLevelCoverageTest} and the in-method assert
|
||||
* both green — only a hand-written behavioural case in {@link ConfigRefTest} caught it, because it
|
||||
* happened to name that exact key. This is the {@link ConfigRefProfileCoverageTest} mechanism one
|
||||
* level up, generalised over every {@code COLD_KEYS}/{@code SPLIT_KEYS}/{@code DEFERRED_KEYS}
|
||||
* member rather than one hand-picked field: enumerate {@link FleetConfig}'s own record components by
|
||||
* reflection, build "a config where only {@code <key>} differs" for each key, and call the real
|
||||
* {@link ConfigRef#changedColdKeys}/{@link ConfigRef#changedSplitKeys}/
|
||||
* {@link ConfigRef#changedDeferredKeys} methods (all package-private for exactly this, the same
|
||||
* reason {@link ConfigRef#sameLaunchSettings} already is) to prove each one is actually reported —
|
||||
* not assumed from a set literal.
|
||||
*
|
||||
* <h2>fleetd #337 — DEFERRED_KEYS was the gap left open here</h2>
|
||||
* This class originally covered {@code COLD_KEYS} and {@code SPLIT_KEYS} only — {@code
|
||||
* DEFERRED_TOP_LEVEL_KEYS} (now {@link ConfigRef#DEFERRED_KEYS}) carried the identical one-way risk
|
||||
* in principle, unexercised. fleetd #337 measured the real consequence rather than assuming it from
|
||||
* the shape of the gap: dropping {@code guard}'s comparison out of {@code changedDeferredKeys} while
|
||||
* {@code "guard"} stayed in the set left all 1355 tests green — the same failure mode {@code
|
||||
* coordinator} demonstrated for {@code SPLIT_KEYS} in fleetd #333, now confirmed for {@code
|
||||
* DEFERRED_KEYS} too. Re-deriving the full list by mutation (drop each key's branch in turn, run the
|
||||
* suite, restore) found six of the eleven {@code DEFERRED_KEYS} members with no behavioural test in
|
||||
* {@link ConfigRefTest} naming them: {@code guard}, {@code leadHeartbeat}, {@code worktreeRoot},
|
||||
* {@code spawnReadyTimeoutMs}, {@code spawnReadyPollMs} and {@code quarantineCooldownSeconds}. That
|
||||
* list corrects fleetd #333's own guess at it in two ways the mutation proved and a reading did not:
|
||||
* {@code lifecycle} is NOT on it — {@code ConfigRefTest.aDeferredChangeIsAppliedAndReported} already
|
||||
* names it, and dropping its branch fails that test — and {@code worktreeRoot} IS on it, which #333
|
||||
* never named at all. The other five {@code DEFERRED_KEYS} members ({@code lifecycle}, {@code
|
||||
* worktreeGroup}, {@code primary}, {@code configReload}, {@code profiles}) already had a hand-written
|
||||
* case each. {@link #everyDeferredKeyIsActuallyReportedByChangedDeferredKeys} below now covers all
|
||||
* eleven the reflective way, so the six with no hand-written test are no longer silently unpinned —
|
||||
* every {@code DEFERRED_KEYS} component turned out to be a scalar or a simple record, so, unlike
|
||||
* {@code fleet.leaders} in fleetd #333, none needed an exclusion set: {@link #BASE}/{@link #ALT} give
|
||||
* every top-level component (not only {@code COLD_KEYS}/{@code SPLIT_KEYS}) a real, distinct value.
|
||||
*/
|
||||
class ConfigRefTopLevelReportingCoverageTest {
|
||||
|
||||
private static final RecordComponent[] COMPONENTS = FleetConfig.class.getRecordComponents();
|
||||
|
||||
/**
|
||||
* One valid value per top-level {@link FleetConfig} component — "the a value". fleetd #337 gave
|
||||
* every {@code DEFERRED_KEYS} component a real value here too (previously left {@code null} on
|
||||
* both sides, which meant {@code mutate(key)} produced no actual difference for any of them);
|
||||
* only {@code placement}, {@code memberCredentials} and {@code memberLoginShell} — the
|
||||
* hot-excluded set, never compared by any {@code changed*Keys} method — stay {@code null}.
|
||||
* {@link FleetConfig}'s compact constructor only normalizes {@code profiles}, so every other
|
||||
* field accepts whatever is put here unmutated.
|
||||
*/
|
||||
private static final Map<String, Object> BASE = baseValues();
|
||||
|
||||
/** The same shape, each value distinct from {@link #BASE} — "the b value". */
|
||||
private static final Map<String, Object> ALT = altValues();
|
||||
|
||||
private static Map<String, Object> baseValues() {
|
||||
Map<String, Object> v = new LinkedHashMap<>();
|
||||
v.put("bind", new FleetConfig.Bind("127.0.0.1", 8765));
|
||||
v.put("herdrSocket", "~/.config/herdr/a.sock");
|
||||
v.put("memberHerdrSocket", "~/.config/herdr/member-a.sock");
|
||||
v.put("profiles", Map.of());
|
||||
v.put("guard", new FleetConfig.Guard(List.of("host-a")));
|
||||
v.put("worktreeRoot", "/wt/a");
|
||||
v.put("lifecycle", new FleetConfig.Lifecycle(300, 5, 30, false));
|
||||
v.put("spawnReadyTimeoutMs", 5000);
|
||||
v.put("spawnReadyPollMs", 100);
|
||||
v.put("broker", new FleetConfig.Broker("amqp://a", null, 1));
|
||||
v.put("primary", new FleetConfig.Primary("term-a", 1, 1000));
|
||||
v.put("fleet", new FleetConfig.Fleet(
|
||||
Map.of("opus", new FleetConfig.Leader("sonnet", "lead: opus-a", 1, null, 10,
|
||||
"claude", null, null, null)),
|
||||
Map.of(), Map.of(), Map.of(), Map.of(), "{role}: {profile} #{n}"));
|
||||
v.put("leadHeartbeat", new FleetConfig.LeadHeartbeat(300, 60_000L, 3));
|
||||
v.put("health", new FleetConfig.Health(true, 30, 600, null, null));
|
||||
v.put("placement", null);
|
||||
v.put("auth", new FleetConfig.Auth("loopback-trust", null));
|
||||
v.put("configReload", new FleetConfig.ConfigReload(true, 10));
|
||||
v.put("quarantineCooldownSeconds", 1800);
|
||||
v.put("memberCredentials", null);
|
||||
v.put("coordinator", new FleetConfig.Coordinator("amqp://coord-a", null, "self-a", 1, null));
|
||||
v.put("worktreeGroup", "group-a");
|
||||
v.put("memberLoginShell", null);
|
||||
v.put("memberSkills", "/skills/a");
|
||||
v.put("idleSleepGuard", new FleetConfig.IdleSleepGuard(true));
|
||||
assertNamesMatchComponents(v);
|
||||
return v;
|
||||
}
|
||||
|
||||
private static Map<String, Object> altValues() {
|
||||
Map<String, Object> v = new LinkedHashMap<>();
|
||||
v.put("bind", new FleetConfig.Bind("127.0.0.2", 8766));
|
||||
v.put("herdrSocket", "~/.config/herdr/b.sock");
|
||||
v.put("memberHerdrSocket", "~/.config/herdr/member-b.sock");
|
||||
// A single added profile — enough to trip the "added/removed" comparison in
|
||||
// ConfigRef.changedDeferredKeys, which is all this mechanism needs to prove "profiles" has
|
||||
// a branch behind it; the launch-settings comparison already has its own hand-written cases
|
||||
// in ConfigRefTest (changingAProfilesLaunchSettingsIsReportedAsDeferred and siblings).
|
||||
v.put("profiles", Map.of("sonnet", minimalProfile("sonnet")));
|
||||
v.put("guard", new FleetConfig.Guard(List.of("host-b")));
|
||||
v.put("worktreeRoot", "/wt/b");
|
||||
v.put("lifecycle", new FleetConfig.Lifecycle(600, 10, 60, true));
|
||||
v.put("spawnReadyTimeoutMs", 10_000);
|
||||
v.put("spawnReadyPollMs", 200);
|
||||
v.put("broker", new FleetConfig.Broker("amqp://b", null, 2));
|
||||
v.put("primary", new FleetConfig.Primary("term-b", 2, 2000));
|
||||
// Differs from BASE.fleet only in fleet.leaders (a different tab for "opus") — the frozen
|
||||
// sub-field ConfigRef.changedSplitKeys actually compares. A Fleet that instead differed only
|
||||
// in tabLabel would correctly NOT be reported (see
|
||||
// ConfigRefTest.changingFleetTabLabelWithoutLeadersStaysFullyHot) and would wrongly fail this
|
||||
// test — that is by design, not a gap: this map exists to prove fleet.leaders is covered.
|
||||
v.put("fleet", new FleetConfig.Fleet(
|
||||
Map.of("opus", new FleetConfig.Leader("sonnet", "lead: opus-b", 1, null, 10,
|
||||
"claude", null, null, null)),
|
||||
Map.of(), Map.of(), Map.of(), Map.of(), "{role}: {profile} #{n}"));
|
||||
v.put("leadHeartbeat", new FleetConfig.LeadHeartbeat(600, 120_000L, 5));
|
||||
v.put("health", new FleetConfig.Health(false, 90, 900, null, null));
|
||||
v.put("placement", null);
|
||||
v.put("auth", new FleetConfig.Auth("token", "TOKEN_ENV"));
|
||||
v.put("configReload", new FleetConfig.ConfigReload(false, 20));
|
||||
v.put("quarantineCooldownSeconds", 3600);
|
||||
v.put("memberCredentials", null);
|
||||
v.put("coordinator", new FleetConfig.Coordinator("amqp://coord-b", null, "self-b", 2, null));
|
||||
v.put("worktreeGroup", "group-b");
|
||||
v.put("memberLoginShell", null);
|
||||
v.put("memberSkills", "/skills/b");
|
||||
v.put("idleSleepGuard", new FleetConfig.IdleSleepGuard(false));
|
||||
assertNamesMatchComponents(v);
|
||||
return v;
|
||||
}
|
||||
|
||||
/** A minimal, otherwise-null {@link FleetConfig.Profile} — just enough to name one in a map. */
|
||||
private static FleetConfig.Profile minimalProfile(String name) {
|
||||
return new FleetConfig.Profile(name, null, null, null, null, null, null, null, null, null,
|
||||
null, null, null, null, null, null, null, null, null, null, null, null, null, null,
|
||||
null, null);
|
||||
}
|
||||
|
||||
private static void assertNamesMatchComponents(Map<String, Object> values) {
|
||||
Set<String> names = new TreeSet<>();
|
||||
for (RecordComponent rc : COMPONENTS) {
|
||||
names.add(rc.getName());
|
||||
}
|
||||
assertEquals(names, new TreeSet<>(values.keySet()),
|
||||
"this test's value map has drifted from FleetConfig's actual top-level components — "
|
||||
+ "update BASE/ALT alongside the record");
|
||||
}
|
||||
|
||||
private static FleetConfig configOf(Map<String, Object> values) throws ReflectiveOperationException {
|
||||
Class<?>[] types = Arrays.stream(COMPONENTS).map(RecordComponent::getType).toArray(Class<?>[]::new);
|
||||
Object[] args = Arrays.stream(COMPONENTS).map(rc -> values.get(rc.getName())).toArray();
|
||||
Constructor<FleetConfig> ctor = FleetConfig.class.getDeclaredConstructor(types);
|
||||
return ctor.newInstance(args);
|
||||
}
|
||||
|
||||
/** {@code BASE} with exactly one named top-level component swapped for its {@code ALT} value. */
|
||||
private static FleetConfig mutate(String componentName) throws ReflectiveOperationException {
|
||||
Map<String, Object> values = new LinkedHashMap<>(BASE);
|
||||
values.put(componentName, ALT.get(componentName));
|
||||
return configOf(values);
|
||||
}
|
||||
|
||||
/**
|
||||
* The mechanism fleetd #333 F2 asked for, applied to {@link ConfigRef#COLD_KEYS}: mutate each
|
||||
* cold key in isolation and prove {@code changedColdKeys} actually names it, not just that
|
||||
* {@code COLD_KEYS} claims it does.
|
||||
*/
|
||||
@Test
|
||||
void everyColdKeyIsActuallyReportedByChangedColdKeys() throws ReflectiveOperationException {
|
||||
FleetConfig base = configOf(BASE);
|
||||
List<String> uncovered = new ArrayList<>();
|
||||
for (String key : new TreeSet<>(ConfigRef.COLD_KEYS)) {
|
||||
FleetConfig mutated = mutate(key);
|
||||
if (!ConfigRef.changedColdKeys(base, mutated).contains(key)) {
|
||||
uncovered.add(key);
|
||||
}
|
||||
}
|
||||
System.out.printf(
|
||||
"ConfigRef.changedColdKeys reporting coverage — %d COLD_KEYS, %d verified%n",
|
||||
ConfigRef.COLD_KEYS.size(), ConfigRef.COLD_KEYS.size() - uncovered.size());
|
||||
assertEquals(List.of(), uncovered,
|
||||
"these keys are in ConfigRef.COLD_KEYS but mutating them alone produces no matching "
|
||||
+ "entry from changedColdKeys — a set entry with no comparison behind it: "
|
||||
+ uncovered);
|
||||
}
|
||||
|
||||
/**
|
||||
* The same mechanism applied to {@link ConfigRef#SPLIT_KEYS}: mutate each split key in
|
||||
* isolation and prove {@code changedSplitKeys} actually names it (message starting with
|
||||
* {@code "<key>:"}), not just that {@code SPLIT_KEYS} claims it does. This is the exact check
|
||||
* that would have failed fleetd #333's own reproduction — dropping the {@code coordinator}
|
||||
* branch from {@code changedSplitKeys} while {@code "coordinator"} stayed in {@code SPLIT_KEYS}
|
||||
* — see the mutation proof in the fleetd #333 PR description; the earlier checkers here (the
|
||||
* shape test and the in-method assert) do not.
|
||||
*/
|
||||
@Test
|
||||
void everySplitKeyIsActuallyReportedByChangedSplitKeys() throws ReflectiveOperationException {
|
||||
FleetConfig base = configOf(BASE);
|
||||
List<String> uncovered = new ArrayList<>();
|
||||
for (String key : new TreeSet<>(ConfigRef.SPLIT_KEYS)) {
|
||||
FleetConfig mutated = mutate(key);
|
||||
List<String> split = ConfigRef.changedSplitKeys(base, mutated);
|
||||
if (split.stream().noneMatch(s -> s.startsWith(key + ":"))) {
|
||||
uncovered.add(key);
|
||||
}
|
||||
}
|
||||
System.out.printf(
|
||||
"ConfigRef.changedSplitKeys reporting coverage — %d SPLIT_KEYS, %d verified%n",
|
||||
ConfigRef.SPLIT_KEYS.size(), ConfigRef.SPLIT_KEYS.size() - uncovered.size());
|
||||
assertEquals(List.of(), uncovered,
|
||||
"these keys are in ConfigRef.SPLIT_KEYS but mutating them alone produces no matching "
|
||||
+ "entry from changedSplitKeys — a set entry with no comparison behind it, "
|
||||
+ "exactly the fleetd #333 F2 shape: " + uncovered);
|
||||
}
|
||||
|
||||
/**
|
||||
* For most {@link ConfigRef#DEFERRED_KEYS} members, {@code changedDeferredKeys} reports the key
|
||||
* name verbatim — the default this map assumes. Two entries don't: {@code spawnReadyTimeoutMs}
|
||||
* and {@code spawnReadyPollMs} are compared together in one branch and reported under the
|
||||
* combined label {@code "spawnReady*"} (see {@link ConfigRef#changedDeferredKeys}). {@code
|
||||
* profiles} keeps the default: mutating it here only exercises the added/removed comparison
|
||||
* (see {@link #altValues}), which reports {@code "profiles (added/removed: …)"} — starts with
|
||||
* {@code "profiles"}, same as the default would expect.
|
||||
*/
|
||||
private static final Map<String, String> DEFERRED_REPORT_PREFIX = Map.of(
|
||||
"spawnReadyTimeoutMs", "spawnReady*",
|
||||
"spawnReadyPollMs", "spawnReady*");
|
||||
|
||||
/**
|
||||
* fleetd #337: the same mechanism applied to {@link ConfigRef#DEFERRED_KEYS}, closing the gap
|
||||
* this class's own javadoc left open since fleetd #333. Mutate each deferred key in isolation
|
||||
* and prove {@code changedDeferredKeys} actually names it (message starting with the key's
|
||||
* expected report prefix — see {@link #DEFERRED_REPORT_PREFIX}), not just that {@code
|
||||
* DEFERRED_KEYS} claims it does. This is the exact check that fails for {@code guard} the way
|
||||
* {@code coordinator} failed {@link #everySplitKeyIsActuallyReportedByChangedSplitKeys} in
|
||||
* fleetd #333 — verified live: dropping {@code guard}'s branch from {@code changedDeferredKeys}
|
||||
* while {@code "guard"} stayed in {@code DEFERRED_KEYS} left the whole 1355-test suite green,
|
||||
* and this test is what now catches it (it fails naming {@code guard} with that mutation in
|
||||
* place).
|
||||
*/
|
||||
@Test
|
||||
void everyDeferredKeyIsActuallyReportedByChangedDeferredKeys() throws ReflectiveOperationException {
|
||||
FleetConfig base = configOf(BASE);
|
||||
List<String> uncovered = new ArrayList<>();
|
||||
for (String key : new TreeSet<>(ConfigRef.DEFERRED_KEYS)) {
|
||||
FleetConfig mutated = mutate(key);
|
||||
List<String> deferred = ConfigRef.changedDeferredKeys(base, mutated);
|
||||
String prefix = DEFERRED_REPORT_PREFIX.getOrDefault(key, key);
|
||||
if (deferred.stream().noneMatch(s -> s.startsWith(prefix))) {
|
||||
uncovered.add(key);
|
||||
}
|
||||
}
|
||||
System.out.printf(
|
||||
"ConfigRef.changedDeferredKeys reporting coverage — %d DEFERRED_KEYS, %d verified%n",
|
||||
ConfigRef.DEFERRED_KEYS.size(), ConfigRef.DEFERRED_KEYS.size() - uncovered.size());
|
||||
assertEquals(List.of(), uncovered,
|
||||
"these keys are in ConfigRef.DEFERRED_KEYS but mutating them alone produces no "
|
||||
+ "matching entry from changedDeferredKeys — a set entry with no comparison "
|
||||
+ "behind it, exactly the fleetd #333 F2 shape, confirmed here for "
|
||||
+ "DEFERRED_KEYS by fleetd #337: " + uncovered);
|
||||
}
|
||||
}
|
||||
+172
@@ -0,0 +1,172 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.lang.reflect.Constructor;
|
||||
import java.lang.reflect.RecordComponent;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Arrays;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* Fleetd #358, the same "defect factory" #357 guarded on {@code FleetConfig.withDefaults()}
|
||||
* (see {@code FleetConfigWithDefaultsPreservesEveryComponentTest}), reproduced here on
|
||||
* {@link FleetConfig.Profile}. {@code Profile} carries a long back-compat constructor ladder — 8
|
||||
* constructors, re-counted directly against the source rather than trusted from the ticket, at
|
||||
* arities 25, 24, 22, 20, 18, 15, 14 and 12, against a canonical arity of 26 — and exactly ONE
|
||||
* rebuild site, {@link FleetConfig.Profile#withProfile(String)}, whose own
|
||||
* {@code return new Profile(...)} call is written at a literal 26-arg count. Add a 27th component
|
||||
* and its established back-compat constructor at the old (26-arg) arity, and {@code withProfile}'s
|
||||
* own call becomes a legal match for that new overload — silently dropping the new component every
|
||||
* time a profile's name is defaulted from its {@code workers:} key.
|
||||
*
|
||||
* <p>Builds one {@link FleetConfig.Profile} through the TRUE canonical constructor — resolved by
|
||||
* the record's own component types via {@code getDeclaredConstructor}, never by argument count —
|
||||
* with a real, distinctive, non-null value in every component, calls {@link
|
||||
* FleetConfig.Profile#withProfile(String)}, and asserts every component except {@code profile}
|
||||
* itself survives unchanged, while {@code profile} comes back as the new name it was given.
|
||||
*
|
||||
* <p>Every value here is chosen so {@code Profile}'s own compact constructor (which normalizes
|
||||
* several components — defaults {@code argv}/{@code kind}/{@code placement}/{@code workspace}/
|
||||
* {@code gitHostEnv}, nulls a handful of blank-checked strings, clamps {@code weight}, coerces
|
||||
* {@code subscription}) leaves it unchanged: every String is non-blank and already in the shape the
|
||||
* compact constructor would otherwise coerce it to (e.g. {@code placement} is already lowercase),
|
||||
* and every collection is non-empty. That is what makes "must survive unchanged" a valid assertion
|
||||
* for every component below, the same reasoning {@code FleetConfigWithDefaultsPreservesEveryComponentTest}
|
||||
* documents for {@code withDefaults()}.
|
||||
*
|
||||
* <p>{@link #EXCLUDED_FROM_SURVIVAL_CHECK} is kept deliberately empty and size-pinned by
|
||||
* {@link #exclusionListSizeIsPinned()} — a checker whose escape hatch can grow to silence a failure
|
||||
* is not a checker. Every one of {@code Profile}'s 26 current components has a real, non-null,
|
||||
* non-blank value here and none is excluded.
|
||||
*/
|
||||
class FleetConfigProfileWithProfilePreservesEveryComponentTest {
|
||||
|
||||
private static final RecordComponent[] COMPONENTS = FleetConfig.Profile.class.getRecordComponents();
|
||||
|
||||
/** Deliberately empty today; grow it only with a matching justification, and re-pin the size. */
|
||||
private static final Set<String> EXCLUDED_FROM_SURVIVAL_CHECK = Set.of();
|
||||
|
||||
/** One real, distinctive, non-null value per component, chosen to survive the compact ctor. */
|
||||
private static Map<String, Object> baseValues() {
|
||||
Map<String, Object> v = new LinkedHashMap<>();
|
||||
v.put("profile", "profile-guard");
|
||||
v.put("baseUrl", "https://guard.example/base");
|
||||
v.put("model", "model-guard");
|
||||
v.put("configDir", "/config/guard");
|
||||
v.put("tokenEnv", "GUARD_TOKEN");
|
||||
v.put("argv", List.of("guard-cmd"));
|
||||
v.put("placement", "guard-placement");
|
||||
v.put("workspace", "workspace-guard");
|
||||
v.put("tabLabel", "tab-guard");
|
||||
v.put("mcpUrl", "https://mcp.guard/");
|
||||
v.put("cwd", "/cwd/guard");
|
||||
v.put("parityOverlay", List.of(".guardrc"));
|
||||
v.put("gitTokenEnv", "GUARD_GIT_TOKEN");
|
||||
v.put("gitHostEnv", "GUARD_GIT_HOST");
|
||||
v.put("kind", "claude-code");
|
||||
v.put("env", Map.of("GUARD_ENV", "1"));
|
||||
v.put("weight", 2.5f);
|
||||
v.put("maxLoad", 4);
|
||||
v.put("subscription", Boolean.TRUE);
|
||||
v.put("exhaustedPattern", "pattern-guard");
|
||||
v.put("credentialId", "cred-guard");
|
||||
v.put("ideMcpUrl", "https://ide.guard/");
|
||||
v.put("ideProjectDir", "ide-project-guard");
|
||||
v.put("ideOpenCommand", "open-guard {dir}");
|
||||
v.put("autoCompactWindow", 150_000);
|
||||
v.put("errorPattern", "error-pattern-guard");
|
||||
assertNamesMatchComponents(v);
|
||||
return v;
|
||||
}
|
||||
|
||||
/**
|
||||
* Guards {@link #baseValues()} itself against drifting from the record's real shape — forgetting
|
||||
* to add a new component here fails this assertion by name, rather than silently checking one
|
||||
* component fewer than the record has.
|
||||
*/
|
||||
private static void assertNamesMatchComponents(Map<String, Object> values) {
|
||||
Set<String> names = new TreeSet<>();
|
||||
for (RecordComponent rc : COMPONENTS) {
|
||||
names.add(rc.getName());
|
||||
}
|
||||
assertEquals(names, new TreeSet<>(values.keySet()),
|
||||
"this test's value map has drifted from FleetConfig.Profile's actual components — "
|
||||
+ "update baseValues() alongside the record");
|
||||
}
|
||||
|
||||
/**
|
||||
* Builds a {@link FleetConfig.Profile} through the TRUE canonical constructor — resolved by the
|
||||
* record's own component types, not by argument count — so this never accidentally exercises a
|
||||
* back-compat overload the way a literal {@code new Profile(...)} call risks doing.
|
||||
*/
|
||||
private static FleetConfig.Profile profileOf(Map<String, Object> values) throws ReflectiveOperationException {
|
||||
Class<?>[] types = Arrays.stream(COMPONENTS).map(RecordComponent::getType).toArray(Class<?>[]::new);
|
||||
Object[] args = Arrays.stream(COMPONENTS).map(rc -> values.get(rc.getName())).toArray();
|
||||
Constructor<FleetConfig.Profile> ctor = FleetConfig.Profile.class.getDeclaredConstructor(types);
|
||||
return ctor.newInstance(args);
|
||||
}
|
||||
|
||||
@Test
|
||||
void exclusionListSizeIsPinned() {
|
||||
assertEquals(0, EXCLUDED_FROM_SURVIVAL_CHECK.size(),
|
||||
"EXCLUDED_FROM_SURVIVAL_CHECK grew from 0 — every entry needs a justification in "
|
||||
+ "this test class's javadoc AND this assertion re-pinned to the new size; a "
|
||||
+ "growing exclusion list that silences failures on its own is not a guard");
|
||||
}
|
||||
|
||||
/**
|
||||
* The mutation this is built to catch: make {@code withProfile(String)}'s final constructor call
|
||||
* literal at some arg count, add one more component to the record with a new back-compat
|
||||
* constructor at the old arity, and the stale call silently rebinds. Every component here is real
|
||||
* and non-null/non-blank, so none of it should be replaced by {@code withProfile}, except
|
||||
* {@code profile} itself, which the method is documented to replace.
|
||||
*/
|
||||
@Test
|
||||
void withProfilePreservesEveryOtherComponent() throws ReflectiveOperationException {
|
||||
Map<String, Object> base = baseValues();
|
||||
FleetConfig.Profile profile = profileOf(base);
|
||||
FleetConfig.Profile renamed = profile.withProfile("renamed-profile-guard");
|
||||
|
||||
List<String> dropped = new ArrayList<>();
|
||||
int checked = 0;
|
||||
for (RecordComponent rc : COMPONENTS) {
|
||||
String name = rc.getName();
|
||||
if (EXCLUDED_FROM_SURVIVAL_CHECK.contains(name)) {
|
||||
continue;
|
||||
}
|
||||
checked++;
|
||||
Object expected = "profile".equals(name) ? "renamed-profile-guard" : base.get(name);
|
||||
Object actual;
|
||||
try {
|
||||
actual = rc.getAccessor().invoke(renamed);
|
||||
} catch (ReflectiveOperationException e) {
|
||||
throw new RuntimeException("failed to read FleetConfig.Profile." + name + "()", e);
|
||||
}
|
||||
if (!Objects.equals(expected, actual)) {
|
||||
dropped.add(String.format(Locale.ROOT,
|
||||
"%s: withProfile() was expected to carry (%s) for '%s' but returned %s — a "
|
||||
+ "component silently dropped by withProfile(), the shape of the "
|
||||
+ "defect this test exists to catch (its final \"return new "
|
||||
+ "Profile(...)\" call binding to a back-compat constructor instead "
|
||||
+ "of the true canonical one)",
|
||||
name, expected, name, actual));
|
||||
}
|
||||
}
|
||||
|
||||
System.out.printf(Locale.ROOT,
|
||||
"FleetConfig.Profile.withProfile() component-survival coverage — %d components, %d "
|
||||
+ "checked, %d excluded, %d survived%n",
|
||||
COMPONENTS.length, checked, EXCLUDED_FROM_SURVIVAL_CHECK.size(), checked - dropped.size());
|
||||
assertEquals(List.of(), dropped,
|
||||
"withProfile() silently dropped these components: " + dropped);
|
||||
}
|
||||
}
|
||||
@@ -6,11 +6,18 @@ import dev.ltms.fleet.peer.MemberRole;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.ParameterizedType;
|
||||
import java.lang.reflect.RecordComponent;
|
||||
import java.lang.reflect.Type;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.TreeMap;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
@@ -148,6 +155,90 @@ class FleetConfigTest {
|
||||
assertTrue(e.getMessage().contains("errorPattern"), "the offending key is named: " + e.getMessage());
|
||||
}
|
||||
|
||||
// ── fleetd #273: exhaustedPattern gets the same load-time validation as its sibling errorPattern ──
|
||||
|
||||
@Test
|
||||
void aProfileWithAMalformedExhaustedPatternIsRejectedAtLoadNamingTheProfileAndKey(@TempDir Path dir)
|
||||
throws Exception {
|
||||
Path f = dir.resolve("malformed-exhausted-pattern.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
ltms-local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
exhaustedPattern: "["
|
||||
""");
|
||||
|
||||
IllegalStateException e = assertThrows(IllegalStateException.class, () -> FleetConfig.load(f));
|
||||
assertTrue(e.getMessage().contains("ltms-local"), "the offending profile is named: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("exhaustedPattern"), "the offending key is named: " + e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMalformedErrorPatternAndAMalformedExhaustedPatternAreBothReportedFromOneLoad(@TempDir Path dir)
|
||||
throws Exception {
|
||||
Path f = dir.resolve("both-malformed.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
errorPattern: "(unterminated["
|
||||
terra:
|
||||
baseUrl: http://gx01.gw:8000
|
||||
exhaustedPattern: "["
|
||||
""");
|
||||
|
||||
IllegalStateException e = assertThrows(IllegalStateException.class, () -> FleetConfig.load(f));
|
||||
assertTrue(e.getMessage().contains("sonnet"), "the errorPattern profile is named: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("errorPattern"), e.getMessage());
|
||||
assertTrue(e.getMessage().contains("terra"), "the exhaustedPattern profile is named: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("exhaustedPattern"), e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void validErrorPatternAndExhaustedPatternBothLoadFine(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("both-valid.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
ltms-local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
errorPattern: "credential outage"
|
||||
exhaustedPattern: "usage limit has been reached"
|
||||
""");
|
||||
|
||||
FleetConfig.Profile w = FleetConfig.load(f).profiles().get("ltms-local");
|
||||
assertEquals("credential outage", w.errorPattern());
|
||||
assertEquals("usage limit has been reached", w.exhaustedPattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBlankExhaustedPatternNormalizesToNullJustLikeUnset(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("blank-exhausted-pattern.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
ltms-local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
exhaustedPattern: " "
|
||||
""");
|
||||
|
||||
FleetConfig.Profile w = FleetConfig.load(f).profiles().get("ltms-local");
|
||||
assertNull(w.exhaustedPattern());
|
||||
assertFalse(w.hasExhaustedPattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aProfileWithNoExhaustedPatternLoadsFine(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("no-exhausted-pattern.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
ltms-local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
""");
|
||||
|
||||
FleetConfig.Profile w = FleetConfig.load(f).profiles().get("ltms-local");
|
||||
assertNull(w.exhaustedPattern());
|
||||
assertFalse(w.hasExhaustedPattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
void withProfileCarriesErrorPatternThrough(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("with-profile-error-pattern.yaml");
|
||||
@@ -1098,7 +1189,7 @@ class FleetConfigTest {
|
||||
@Test
|
||||
void coordinatorEffectiveUriHonorsUriEnv() {
|
||||
FleetConfig.Coordinator withEnv = new FleetConfig.Coordinator(
|
||||
"amqp://stale-clear-text@127.0.0.1:5672/coord", "LEAD_COORD_URI", "fleet01-lead", null);
|
||||
"amqp://stale-clear-text@127.0.0.1:5672/coord", "LEAD_COORD_URI", "fleet01-lead", null, null);
|
||||
|
||||
assertEquals("amqp://from-env@127.0.0.1:5672/coord",
|
||||
withEnv.effectiveUri(Map.of("LEAD_COORD_URI", "amqp://from-env@127.0.0.1:5672/coord")),
|
||||
@@ -1109,12 +1200,61 @@ class FleetConfigTest {
|
||||
"a blank uriEnv variable must not fall back to the literal uri");
|
||||
|
||||
FleetConfig.Coordinator noEnv = new FleetConfig.Coordinator(
|
||||
"amqp://guest:guest@127.0.0.1:5672/coord", null, null, null);
|
||||
"amqp://guest:guest@127.0.0.1:5672/coord", null, null, null, null);
|
||||
assertEquals("amqp://guest:guest@127.0.0.1:5672/coord", noEnv.effectiveUri(Map.of()),
|
||||
"the literal uri is used when no uriEnv is configured");
|
||||
assertEquals(LeadMailbox.DEFAULT_PREFETCH, noEnv.prefetchOrDefault());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361: the live block ships as just {@code uriEnv} + {@code selfId} (see
|
||||
* {@code Fleetd.example.yaml} / the operator's real {@code fleetd.yaml}, gitignored). That exact
|
||||
* shape, with no {@code peers:} key at all, must keep parsing unchanged after this field is added.
|
||||
*/
|
||||
@Test
|
||||
void coordinatorBlockWithNoPeersKeyStillParses(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("coordinator-no-peers.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
port: 8080
|
||||
coordinator:
|
||||
uriEnv: COORD_AMQP_URI
|
||||
selfId: mac
|
||||
""");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertNotNull(cfg.coordinator());
|
||||
assertEquals("mac", cfg.coordinator().selfId());
|
||||
assertEquals(List.of(), cfg.coordinator().peers(), "no peers: key means no configured peers, never null");
|
||||
}
|
||||
|
||||
@Test
|
||||
void coordinatorPeersParsesAndDropsBlankEntries(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("coordinator-peers.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
port: 8080
|
||||
coordinator:
|
||||
selfId: mac
|
||||
uri: amqp://guest:guest@127.0.0.1:5672/coord
|
||||
peers:
|
||||
- fleet01
|
||||
- ""
|
||||
- fleet02
|
||||
""");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertEquals(List.of("fleet01", "fleet02"), cfg.coordinator().peers(),
|
||||
"a blank peer entry must be dropped, never kept as an empty coord-id");
|
||||
}
|
||||
|
||||
@Test
|
||||
void coordinatorPeersDefaultsToEmptyWhenConstructedWithNull() {
|
||||
FleetConfig.Coordinator c = new FleetConfig.Coordinator(
|
||||
"amqp://guest:guest@127.0.0.1:5672/coord", null, "mac", null, null);
|
||||
assertEquals(List.of(), c.peers(), "a null peers list must default to empty, never NPE downstream");
|
||||
}
|
||||
|
||||
@Test
|
||||
void absentWorktreeGroupLeavesItNull(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("no-worktree-group.yaml");
|
||||
@@ -1366,9 +1506,18 @@ class FleetConfigTest {
|
||||
}
|
||||
|
||||
/**
|
||||
* Every optional knob the example documents must bind under the exact spelling used there.
|
||||
* Keep this list in step with {@code fleetd.example.yaml}: a rename that updates the record
|
||||
* but not the example (or vice versa) fails here instead of silently no-op'ing in production.
|
||||
* A spot check that the knobs listed below bind under the exact spelling the example uses —
|
||||
* it asserts real VALUES arrive in the record, which no name-matching guard can do.
|
||||
*
|
||||
* <p><b>This is NOT a coverage guard, and must not be read as one</b> (fleetd #113). The list
|
||||
* inside it is hand-written, so it only ever covers what someone remembered to add. Coverage
|
||||
* — "is every key the code reads documented, and does every documented key bind?" — comes
|
||||
* from {@link #everyNestedConfigKeyIsDocumentedInTheExample} and
|
||||
* {@link #everyLiveKeyInTheExampleBindsToARecordComponent}, both of which derive their key
|
||||
* set from the record tree and therefore cannot drift.
|
||||
*
|
||||
* <p>Adding a knob here is optional. Leaving one out is not a coverage gap, because the two
|
||||
* derived guards above already fail on an undocumented or unbindable key.
|
||||
*/
|
||||
@Test
|
||||
void everyOptionalKnobDocumentedInTheExampleBinds(@TempDir Path dir) throws Exception {
|
||||
@@ -1525,6 +1674,230 @@ class FleetConfigTest {
|
||||
return p.matcher(yaml).find();
|
||||
}
|
||||
|
||||
/**
|
||||
* The nested half of {@link #everyKnownTopLevelKeyIsDocumentedInTheExample} (fleetd #113).
|
||||
*
|
||||
* <p>That guard walks {@link FleetConfig#KNOWN_TOP_LEVEL_KEYS} and anchors its regex at
|
||||
* column 0, so it sees ONLY top-level keys. Every nested key — {@code profiles.<name>.model},
|
||||
* {@code health.paneProbeIntervalSeconds} and a hundred others — is outside its scope, and
|
||||
* neither its name nor its output says so. A green run then reads as "the example documents
|
||||
* the schema" when most of the schema was never looked at.
|
||||
*
|
||||
* <p>This walks the record tree rather than a name list, so a key added to any nested record
|
||||
* is covered the moment it compiles, with no edit here. That is the point: a hand-maintained
|
||||
* second copy of a list always drifts from the thing it mirrors.
|
||||
*
|
||||
* <p><b>Scope, stated on purpose</b> (fleetd #113 criterion 3 — every check reports what it
|
||||
* did and did not look at):
|
||||
* <ul>
|
||||
* <li>It checks each key NAME appears somewhere in the example as a YAML key, live or
|
||||
* commented out. It does NOT check the key sits at the right path.</li>
|
||||
* <li>It does NOT check a documented key is read by anything. {@code paneProbeIntervalSeconds}
|
||||
* is parsed into {@link FleetConfig.Health} and used nowhere, and this guard passes it.
|
||||
* Proving a key is live code needs a call graph, which this is not.</li>
|
||||
* </ul>
|
||||
*/
|
||||
@Test
|
||||
void everyNestedConfigKeyIsDocumentedInTheExample() throws Exception {
|
||||
Path example = Path.of("fleetd.example.yaml");
|
||||
assertTrue(Files.exists(example), "fleetd.example.yaml must ship next to the pom");
|
||||
String text = Files.readString(example);
|
||||
|
||||
Map<String, String> pathByName = configKeyPaths();
|
||||
|
||||
// The denominator. An under-counting walk passes every subset check vacuously, which is
|
||||
// the exact shape fleetd #113 collects — so the walk has to prove it descended at all.
|
||||
// The floor is DERIVED, not a literal: the nested walk must find substantially more keys
|
||||
// than the top-level set the old guard used, or it has not gone below the first level.
|
||||
int topLevel = FleetConfig.KNOWN_TOP_LEVEL_KEYS.size();
|
||||
assertTrue(pathByName.size() > topLevel * 2,
|
||||
"the record walk found " + pathByName.size() + " config key(s) against "
|
||||
+ topLevel + " top-level key(s) — it has stopped descending into the "
|
||||
+ "nested records, so this guard would pass vacuously. Fix the walk "
|
||||
+ "before trusting a green run.");
|
||||
|
||||
List<String> undocumented = pathByName.entrySet().stream()
|
||||
.filter(e -> !keyDocumentedAnywhere(text, e.getKey()))
|
||||
.map(Map.Entry::getValue)
|
||||
.sorted()
|
||||
.toList();
|
||||
|
||||
assertTrue(undocumented.isEmpty(), () -> "checked " + pathByName.size()
|
||||
+ " config key(s) that FleetConfig can bind; " + undocumented.size()
|
||||
+ " appear nowhere in fleetd.example.yaml: " + undocumented
|
||||
+ " — document each one there, commented out if optional. fleetd.yaml is "
|
||||
+ "gitignored, so the example is the only committed description of the schema.");
|
||||
}
|
||||
|
||||
/**
|
||||
* The other direction: a LIVE key in the example that {@link FleetConfig} cannot bind. That is
|
||||
* a key an operator would copy into {@code fleetd.yaml} expecting it to do something, where it
|
||||
* would be silently ignored.
|
||||
*
|
||||
* <p><b>Scope, stated on purpose:</b> only live (uncommented) keys are checked. Most of the
|
||||
* example is commented-out prose, and that prose contains lines like {@code # mode: token}
|
||||
* that are indistinguishable from keys by text alone. Parsing them would produce false
|
||||
* failures, so they are deliberately out of scope — and saying so here is the point, rather
|
||||
* than letting a reader assume the whole file was validated.
|
||||
*/
|
||||
@Test
|
||||
void everyLiveKeyInTheExampleBindsToARecordComponent() throws Exception {
|
||||
Path example = Path.of("fleetd.example.yaml");
|
||||
String text = Files.readString(example);
|
||||
|
||||
List<List<String>> paths = liveKeyPaths(text);
|
||||
assertTrue(paths.size() >= 20,
|
||||
"only " + paths.size() + " live key path(s) were parsed out of the example — the "
|
||||
+ "parser is not seeing the file, so this guard would pass vacuously.");
|
||||
|
||||
List<String> unbindable = paths.stream()
|
||||
.filter(path -> !pathBinds(path))
|
||||
.map(path -> String.join(".", path))
|
||||
.distinct()
|
||||
.sorted()
|
||||
.toList();
|
||||
|
||||
assertTrue(unbindable.isEmpty(), () -> "checked " + paths.size()
|
||||
+ " live key path(s) in fleetd.example.yaml; " + unbindable.size()
|
||||
+ " bind to nothing in FleetConfig: " + unbindable
|
||||
+ " — an operator copying one of these into fleetd.yaml gets silence, not an error.");
|
||||
}
|
||||
|
||||
/**
|
||||
* Every configuration key {@link FleetConfig} can bind, at every depth, as
|
||||
* {@code name -> a dotted path to one place it appears}. Derived from the record components,
|
||||
* so it cannot drift from the code.
|
||||
*/
|
||||
private static Map<String, String> configKeyPaths() {
|
||||
Map<String, String> out = new TreeMap<>();
|
||||
collectConfigKeys(FleetConfig.class, "", new HashSet<>(), out);
|
||||
return out;
|
||||
}
|
||||
|
||||
private static void collectConfigKeys(Class<?> type, String prefix, Set<String> seen,
|
||||
Map<String, String> out) {
|
||||
if (!type.isRecord() || !seen.add(type.getName())) {
|
||||
return;
|
||||
}
|
||||
for (RecordComponent rc : type.getRecordComponents()) {
|
||||
String path = prefix.isEmpty() ? rc.getName() : prefix + "." + rc.getName();
|
||||
out.putIfAbsent(rc.getName(), path);
|
||||
Class<?> nested = rc.getType();
|
||||
if (nested.isRecord()) {
|
||||
collectConfigKeys(nested, path, seen, out);
|
||||
} else if (Map.class.isAssignableFrom(nested) || List.class.isAssignableFrom(nested)) {
|
||||
Class<?> element = elementRecord(rc);
|
||||
if (element != null) {
|
||||
String childPrefix = Map.class.isAssignableFrom(nested)
|
||||
? path + ".<name>" : path + "[]";
|
||||
collectConfigKeys(element, childPrefix, seen, out);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** The record type inside a {@code Map<String, X>} or {@code List<X>} component, else null. */
|
||||
private static Class<?> elementRecord(RecordComponent rc) {
|
||||
if (rc.getGenericType() instanceof ParameterizedType pt) {
|
||||
Type[] args = pt.getActualTypeArguments();
|
||||
if (args.length > 0 && args[args.length - 1] instanceof Class<?> c && c.isRecord()) {
|
||||
return c;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when {@code key} is documented in the example, in either of the two conventions that
|
||||
* file actually uses:
|
||||
* <ol>
|
||||
* <li>as a YAML key at any indentation, live or commented out ({@code key:}); or</li>
|
||||
* <li>in a prose block that describes a section's sub-keys, one per line, as
|
||||
* {@code # key → what it does}.</li>
|
||||
* </ol>
|
||||
*
|
||||
* <p>The second form is not decoration. {@code broker.uri} is documented ONLY that way, on
|
||||
* purpose: writing it out as a copy-pasteable {@code uri: amqp://user:pass@host} invites an
|
||||
* operator to paste a password into a file, which is the very thing {@code uriEnv} exists to
|
||||
* avoid. A guard that demanded the key form would push the file toward doing that. So this
|
||||
* encodes the convention the example really uses rather than imposing a new one.
|
||||
*/
|
||||
private static boolean keyDocumentedAnywhere(String yaml, String key) {
|
||||
String quoted = Pattern.quote(key);
|
||||
Pattern asYamlKey = Pattern.compile("(?m)^\\s*(?:#\\s*)?" + quoted + ":");
|
||||
Pattern asProseEntry = Pattern.compile("(?m)^\\s*#\\s*" + quoted + "\\s+\u2192");
|
||||
return asYamlKey.matcher(yaml).find() || asProseEntry.matcher(yaml).find();
|
||||
}
|
||||
|
||||
/** Every live (uncommented) key in {@code yaml}, as a path from the document root. */
|
||||
private static List<List<String>> liveKeyPaths(String yaml) {
|
||||
Pattern keyLine = Pattern.compile("^(\\s*)([A-Za-z][A-Za-z0-9_]*):(\\s.*)?$");
|
||||
List<String> stack = new ArrayList<>();
|
||||
List<Integer> indents = new ArrayList<>();
|
||||
List<List<String>> paths = new ArrayList<>();
|
||||
for (String line : yaml.split("\n", -1)) {
|
||||
if (line.isBlank() || line.stripLeading().startsWith("#")) {
|
||||
continue;
|
||||
}
|
||||
Matcher m = keyLine.matcher(line);
|
||||
if (!m.matches()) {
|
||||
continue;
|
||||
}
|
||||
int indent = m.group(1).length();
|
||||
while (!indents.isEmpty() && indents.get(indents.size() - 1) >= indent) {
|
||||
indents.remove(indents.size() - 1);
|
||||
stack.remove(stack.size() - 1);
|
||||
}
|
||||
indents.add(indent);
|
||||
stack.add(m.group(2));
|
||||
paths.add(List.copyOf(stack));
|
||||
}
|
||||
return paths;
|
||||
}
|
||||
|
||||
/** True when a dotted YAML path resolves to something {@link FleetConfig} can bind. */
|
||||
private static boolean pathBinds(List<String> path) {
|
||||
Class<?> type = FleetConfig.class;
|
||||
boolean nextSegmentIsAFreeFormName = false;
|
||||
for (int i = 0; i < path.size(); i++) {
|
||||
if (nextSegmentIsAFreeFormName) {
|
||||
nextSegmentIsAFreeFormName = false;
|
||||
continue;
|
||||
}
|
||||
RecordComponent rc = componentNamed(type, path.get(i));
|
||||
if (rc == null) {
|
||||
return false;
|
||||
}
|
||||
Class<?> t = rc.getType();
|
||||
if (t.isRecord()) {
|
||||
type = t;
|
||||
} else if (Map.class.isAssignableFrom(t)) {
|
||||
Class<?> element = elementRecord(rc);
|
||||
if (element == null) {
|
||||
return true; // Map<String,String>: its entries are data, not schema
|
||||
}
|
||||
type = element;
|
||||
nextSegmentIsAFreeFormName = true;
|
||||
} else {
|
||||
// A scalar or a list of scalars: nothing may legitimately nest under it.
|
||||
return i == path.size() - 1;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
private static RecordComponent componentNamed(Class<?> type, String name) {
|
||||
if (type == null || !type.isRecord()) {
|
||||
return null;
|
||||
}
|
||||
for (RecordComponent rc : type.getRecordComponents()) {
|
||||
if (rc.getName().equals(name)) {
|
||||
return rc;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
@Test
|
||||
void placementDefaultsToFixedForExistingConfigs(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("no-placement.yaml");
|
||||
@@ -1770,22 +2143,93 @@ class FleetConfigTest {
|
||||
"deny-list normalizes onto the canonical deny-by-default value");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-633: {@code SSH_AUTH_SOCK} is a decision, never a default — absent, blank, or misspelled,
|
||||
* it stays BLOCKED; only the literal "allow" (any case) passes it through. A typo like "alow"
|
||||
* failing safe here is the whole point of making it a knob.
|
||||
*/
|
||||
@Test
|
||||
void sshAuthSockDefaultsToBlockedAndOnlyExplicitAllowUnblocksIt() {
|
||||
assertTrue(new FleetConfig.MemberCredentials("allow-list", List.of(), List.of()).sshAuthSock().equals("block"),
|
||||
"absent knob blocks SSH_AUTH_SOCK");
|
||||
assertFalse(new FleetConfig.MemberCredentials("allow-list", List.of(), List.of()).sshAuthSockAllowed());
|
||||
assertFalse(new FleetConfig.MemberCredentials(null, null, null, "").sshAuthSockAllowed(),
|
||||
"blank knob blocks SSH_AUTH_SOCK");
|
||||
assertFalse(new FleetConfig.MemberCredentials(null, null, null, "alow").sshAuthSockAllowed(),
|
||||
"a misspelled value fails SAFE, not open");
|
||||
assertTrue(new FleetConfig.MemberCredentials(null, null, null, "ALLOW").sshAuthSockAllowed(),
|
||||
"the literal allow (case-insensitive) unblocks SSH_AUTH_SOCK");
|
||||
void sshAgentEnvAcceptsEveryCompatibleKeyAndValuePair(@TempDir Path dir) throws Exception {
|
||||
String[][] spellings = {
|
||||
{"sshAuthSock", "block", "omit"},
|
||||
{"sshAuthSock", "allow", "inherit"},
|
||||
{"sshAuthSock", "omit", "omit"},
|
||||
{"sshAuthSock", "inherit", "inherit"},
|
||||
{"sshAgentEnv", "block", "omit"},
|
||||
{"sshAgentEnv", "allow", "inherit"},
|
||||
{"sshAgentEnv", "omit", "omit"},
|
||||
{"sshAgentEnv", "inherit", "inherit"}
|
||||
};
|
||||
|
||||
for (int i = 0; i < spellings.length; i++) {
|
||||
Path file = dir.resolve("member-credentials-ssh-agent-" + i + ".yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
port: 8080
|
||||
memberCredentials:
|
||||
policy: allow-list
|
||||
""" + " " + spellings[i][0] + ": " + spellings[i][1] + "\n");
|
||||
|
||||
FleetConfig.MemberCredentials credentials = FleetConfig.load(file).memberCredentials();
|
||||
assertEquals(spellings[i][2], credentials.sshAgentEnv(),
|
||||
spellings[i][0] + ": " + spellings[i][1] + " must normalize correctly");
|
||||
assertEquals("inherit".equals(spellings[i][2]), credentials.sshAgentEnvInherited());
|
||||
}
|
||||
}
|
||||
|
||||
/** Protects the live {@code sshAuthSock: block} allow-list configuration during the rename. */
|
||||
@Test
|
||||
void legacySshAuthSockBlockKeepsLiveAllowListConfigOmitted(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("live-member-credentials.yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
port: 8080
|
||||
memberCredentials:
|
||||
policy: allow-list
|
||||
sshAuthSock: block
|
||||
""");
|
||||
|
||||
FleetConfig.MemberCredentials credentials = FleetConfig.load(file).memberCredentials();
|
||||
|
||||
assertEquals("omit", credentials.sshAgentEnv());
|
||||
assertFalse(credentials.sshAgentEnvInherited(), "the live config must omit SSH_AUTH_SOCK");
|
||||
}
|
||||
|
||||
@Test
|
||||
void sshAgentEnvWinsWhenBothCompatibleKeysArePresent(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("both-ssh-agent-keys.yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
port: 8080
|
||||
memberCredentials:
|
||||
policy: allow-list
|
||||
sshAuthSock: allow
|
||||
sshAgentEnv: omit
|
||||
""");
|
||||
|
||||
FleetConfig.MemberCredentials credentials = FleetConfig.load(file).memberCredentials();
|
||||
|
||||
assertEquals("omit", credentials.sshAgentEnv());
|
||||
assertFalse(credentials.sshAgentEnvInherited());
|
||||
}
|
||||
|
||||
@Test
|
||||
void sshAgentEnvDefaultsToOmitAndUnknownValuesFailClosed(@TempDir Path dir) throws Exception {
|
||||
Path absent = dir.resolve("member-credentials-ssh-agent-absent.yaml");
|
||||
Files.writeString(absent, """
|
||||
bind:
|
||||
port: 8080
|
||||
memberCredentials:
|
||||
policy: allow-list
|
||||
""");
|
||||
assertEquals("omit", FleetConfig.load(absent).memberCredentials().sshAgentEnv());
|
||||
|
||||
Path unknown = dir.resolve("member-credentials-ssh-agent-unknown.yaml");
|
||||
Files.writeString(unknown, """
|
||||
bind:
|
||||
port: 8080
|
||||
memberCredentials:
|
||||
policy: allow-list
|
||||
sshAgentEnv: inhert
|
||||
""");
|
||||
FleetConfig.MemberCredentials credentials = FleetConfig.load(unknown).memberCredentials();
|
||||
assertEquals("omit", credentials.sshAgentEnv());
|
||||
assertFalse(credentials.sshAgentEnvInherited(), "an unknown value must fail closed");
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -2185,4 +2629,58 @@ class FleetConfigTest {
|
||||
"with no pool to choose from, every configured profile is a candidate and the "
|
||||
+ "first one wins");
|
||||
}
|
||||
|
||||
// ── idle-sleep guard: default-on config block ───────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void idleSleepGuardIsOnByDefaultWhenTheBlockIsEntirelyAbsent(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8080
|
||||
""");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertNull(cfg.idleSleepGuard(), "an absent block parses to null, unlike most other blocks here");
|
||||
// The block itself is absent, but the FEATURE stays on: Fleetd treats a null block the
|
||||
// same as enabled: true (see FleetConfig.idleSleepGuard's javadoc) — this test only pins
|
||||
// the parse result, the on-by-default behaviour is Fleetd's own null check.
|
||||
}
|
||||
|
||||
@Test
|
||||
void idleSleepGuardExplicitlyEnabledIsOn(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
idleSleepGuard:
|
||||
enabled: true
|
||||
""");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertTrue(cfg.idleSleepGuard().isEnabled());
|
||||
}
|
||||
|
||||
@Test
|
||||
void idleSleepGuardExplicitlyDisabledIsOff(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
""");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertFalse(cfg.idleSleepGuard().isEnabled());
|
||||
}
|
||||
|
||||
@Test
|
||||
void idleSleepGuardBlockPresentButEmptyDefaultsToEnabled(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
idleSleepGuard: {}
|
||||
""");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertTrue(cfg.idleSleepGuard().isEnabled(),
|
||||
"unlike ConfigReload/Health, this block defaults to ON even when present but empty");
|
||||
}
|
||||
}
|
||||
|
||||
+194
@@ -0,0 +1,194 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.lang.reflect.Constructor;
|
||||
import java.lang.reflect.RecordComponent;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Arrays;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* Guards against a "defect factory" built into this file's own established pattern, found live
|
||||
* while building the (parked) idle-sleep-guard PR: every time a component is added to
|
||||
* {@link FleetConfig}, the record grows by one arg AND a new back-compat constructor is added at
|
||||
* the OLD arity, so existing callers keep compiling. That is correct and required — see the
|
||||
* constructor ladder just below the record header. But {@link #withDefaults()}'s own {@code return
|
||||
* new FleetConfig(...)} call sits in this same file, written at a literal argument count. The very
|
||||
* next time a component is added, the freshly-added back-compat constructor at the OLD arity
|
||||
* silently captures that stale call, because it is now a legal overload at that arg count too. It
|
||||
* compiles. Every other test passes, because nothing else exercises the new field. The new
|
||||
* component is defaulted away — {@code null}, or whatever that back-compat overload defaults it to
|
||||
* — on every {@link FleetConfig#load}. Measured, not theoretical: this exact sequence happened
|
||||
* live when the {@code idleSleepGuard} component was added on a sibling branch; it was caught only
|
||||
* because that branch's own new tests happened to assert on the new field's value.
|
||||
*
|
||||
* <p>This test proves the opposite property, and does it in a way that survives the next field
|
||||
* being added without being rewritten: reflectively enumerate {@link FleetConfig}'s own record
|
||||
* components (never a hardcoded count — the arity is exactly what changes over time), build one
|
||||
* config through the true canonical constructor with a real, distinctive, non-null value in EVERY
|
||||
* component (reusing the exact reflective-construction pattern
|
||||
* {@link ConfigRefTopLevelReportingCoverageTest} already established for this file:
|
||||
* {@code getDeclaredConstructor(exact record-component types)}, which resolves the canonical
|
||||
* constructor by its true shape, not by binding to whichever overload happens to match arg count —
|
||||
* the same way Jackson resolves it), call the real {@link FleetConfig#withDefaults()}, and assert
|
||||
* every one of those values survives unchanged.
|
||||
*
|
||||
* <p>Why this is a valid check for every component, not just some: {@link #withDefaults()}'s own
|
||||
* comments document that it only ever REPLACES a component when the incoming value is {@code null}
|
||||
* (or blank, for {@code placement}) — {@code broker}/{@code primary}/{@code leadHeartbeat}/
|
||||
* {@code configReload}/{@code coordinator}/{@code worktreeGroup}/{@code memberLoginShell}/
|
||||
* {@code memberSkills}/{@code idleSleepGuard} are left as-is unconditionally, and {@code bind}/{@code guard}/{@code lifecycle}/{@code auth}/
|
||||
* {@code fleet}/{@code quarantineCooldownSeconds}/{@code memberCredentials}/{@code placement} are
|
||||
* replaced only on null/blank input. A value that is never null or blank going in must therefore
|
||||
* never change coming out, for every current component. No exclusion is needed today.
|
||||
*
|
||||
* <p>{@link #EXCLUDED_FROM_SURVIVAL_CHECK} exists anyway, kept deliberately empty and size-pinned
|
||||
* by {@link #exclusionListSizeIsPinned()}: a future component that {@code withDefaults()} is
|
||||
* <em>documented</em> to transform unconditionally (unlike every field today) would legitimately
|
||||
* need one. Pinning the size at 0 means growing that set to make a failure go away is itself a
|
||||
* visible diff to this test, not a silent one — a checker that can be silenced by adding to its
|
||||
* own escape hatch is not a checker.
|
||||
*/
|
||||
class FleetConfigWithDefaultsPreservesEveryComponentTest {
|
||||
|
||||
private static final RecordComponent[] COMPONENTS = FleetConfig.class.getRecordComponents();
|
||||
|
||||
/** See the class javadoc — deliberately empty today; grow it only with a matching justification. */
|
||||
private static final Set<String> EXCLUDED_FROM_SURVIVAL_CHECK = Set.of();
|
||||
|
||||
/** One real, distinctive, non-null (non-blank where blankness would mean "unset") value per component. */
|
||||
private static Map<String, Object> baseValues() {
|
||||
Map<String, Object> v = new LinkedHashMap<>();
|
||||
v.put("bind", new FleetConfig.Bind("127.0.0.1", 8765));
|
||||
v.put("herdrSocket", "~/.config/herdr/guard.sock");
|
||||
v.put("memberHerdrSocket", "~/.config/herdr/member-guard.sock");
|
||||
v.put("profiles", Map.of("sonnet", minimalProfile("sonnet")));
|
||||
v.put("guard", new FleetConfig.Guard(List.of("host-guard")));
|
||||
v.put("worktreeRoot", "/wt/guard");
|
||||
v.put("lifecycle", new FleetConfig.Lifecycle(300, 5, 30, true));
|
||||
v.put("spawnReadyTimeoutMs", 12_345);
|
||||
v.put("spawnReadyPollMs", 234);
|
||||
v.put("broker", new FleetConfig.Broker("amqp://guard", null, 7));
|
||||
v.put("primary", new FleetConfig.Primary("term-guard", 4, 4000));
|
||||
v.put("fleet", new FleetConfig.Fleet(
|
||||
Map.of("opus", new FleetConfig.Leader("sonnet", "lead: opus-guard", 1, null, 10,
|
||||
"claude", null, null, null)),
|
||||
Map.of(), Map.of(), Map.of(), Map.of(), "{role}: {profile} #{n}"));
|
||||
v.put("leadHeartbeat", new FleetConfig.LeadHeartbeat(301, 61_000L, 4));
|
||||
v.put("health", new FleetConfig.Health(true, 31, 601, 61, null));
|
||||
v.put("placement", "round-robin");
|
||||
v.put("auth", new FleetConfig.Auth("loopback-trust", null));
|
||||
v.put("configReload", new FleetConfig.ConfigReload(true, 11));
|
||||
v.put("quarantineCooldownSeconds", 1801);
|
||||
v.put("memberCredentials", new FleetConfig.MemberCredentials(
|
||||
FleetConfig.MemberCredentials.POLICY_DENY_BY_DEFAULT,
|
||||
List.of("git"), List.of("git", "ssh"), null));
|
||||
v.put("coordinator", new FleetConfig.Coordinator("amqp://coord-guard", null, "self-guard", 3, null));
|
||||
v.put("worktreeGroup", "group-guard");
|
||||
v.put("memberLoginShell", "/bin/zsh");
|
||||
v.put("memberSkills", "/skills/guard");
|
||||
v.put("idleSleepGuard", new FleetConfig.IdleSleepGuard(true));
|
||||
assertNamesMatchComponents(v);
|
||||
return v;
|
||||
}
|
||||
|
||||
/** A minimal, otherwise-null {@link FleetConfig.Profile} — just enough to name one in a map. */
|
||||
private static FleetConfig.Profile minimalProfile(String name) {
|
||||
return new FleetConfig.Profile(name, null, null, null, null, null, null, null, null, null,
|
||||
null, null, null, null, null, null, null, null, null, null, null, null, null, null,
|
||||
null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Guards {@link #baseValues()} itself against drifting from the record's real shape — the same
|
||||
* assurance {@link ConfigRefTopLevelReportingCoverageTest} already relies on. This is what makes
|
||||
* "no hardcoded arity" true in practice: forgetting to add a new component here fails this
|
||||
* assertion by name, rather than silently checking one component fewer than the record has.
|
||||
*/
|
||||
private static void assertNamesMatchComponents(Map<String, Object> values) {
|
||||
Set<String> names = new TreeSet<>();
|
||||
for (RecordComponent rc : COMPONENTS) {
|
||||
names.add(rc.getName());
|
||||
}
|
||||
assertEquals(names, new TreeSet<>(values.keySet()),
|
||||
"this test's value map has drifted from FleetConfig's actual top-level components — "
|
||||
+ "update baseValues() alongside the record");
|
||||
}
|
||||
|
||||
/**
|
||||
* Builds a {@link FleetConfig} through the TRUE canonical constructor — resolved by the record's
|
||||
* own component types, not by argument count — so this never accidentally exercises a
|
||||
* back-compat overload the way a literal {@code new FleetConfig(...)} call risks doing.
|
||||
*/
|
||||
private static FleetConfig configOf(Map<String, Object> values) throws ReflectiveOperationException {
|
||||
Class<?>[] types = Arrays.stream(COMPONENTS).map(RecordComponent::getType).toArray(Class<?>[]::new);
|
||||
Object[] args = Arrays.stream(COMPONENTS).map(rc -> values.get(rc.getName())).toArray();
|
||||
Constructor<FleetConfig> ctor = FleetConfig.class.getDeclaredConstructor(types);
|
||||
return ctor.newInstance(args);
|
||||
}
|
||||
|
||||
@Test
|
||||
void exclusionListSizeIsPinned() {
|
||||
assertEquals(0, EXCLUDED_FROM_SURVIVAL_CHECK.size(),
|
||||
"EXCLUDED_FROM_SURVIVAL_CHECK grew from 0 — every entry needs a justification in "
|
||||
+ "this test class's javadoc AND this assertion re-pinned to the new size; a "
|
||||
+ "growing exclusion list that silences failures on its own is not a guard");
|
||||
}
|
||||
|
||||
/**
|
||||
* The mutation this is built to catch: make {@code withDefaults()}'s final constructor call
|
||||
* literal at some arg count, add one more component to the record with a new back-compat
|
||||
* constructor at the old arity, and the stale call silently rebinds. Every component here is
|
||||
* real and non-null (non-blank for the one String — {@code placement} — where blank has
|
||||
* meaning), so none of it should be replaced by {@code withDefaults()}; any component that
|
||||
* comes back different was silently dropped.
|
||||
*/
|
||||
@Test
|
||||
void everyComponentGivenARealValueSurvivesWithDefaults() throws ReflectiveOperationException {
|
||||
Map<String, Object> base = baseValues();
|
||||
FleetConfig config = configOf(base);
|
||||
FleetConfig defaulted = config.withDefaults();
|
||||
|
||||
List<String> dropped = new ArrayList<>();
|
||||
int checked = 0;
|
||||
for (RecordComponent rc : COMPONENTS) {
|
||||
String name = rc.getName();
|
||||
if (EXCLUDED_FROM_SURVIVAL_CHECK.contains(name)) {
|
||||
continue;
|
||||
}
|
||||
checked++;
|
||||
Object expected = base.get(name);
|
||||
Object actual;
|
||||
try {
|
||||
actual = rc.getAccessor().invoke(defaulted);
|
||||
} catch (ReflectiveOperationException e) {
|
||||
throw new RuntimeException("failed to read FleetConfig." + name + "()", e);
|
||||
}
|
||||
if (!Objects.equals(expected, actual)) {
|
||||
dropped.add(String.format(Locale.ROOT,
|
||||
"%s: withDefaults() was given a real, non-null value (%s) for '%s' but "
|
||||
+ "returned %s — a component silently dropped by withDefaults(), the "
|
||||
+ "shape of the defect this test exists to catch (its final "
|
||||
+ "\"return new FleetConfig(...)\" call binding to a back-compat "
|
||||
+ "constructor instead of the true canonical one)",
|
||||
name, expected, name, actual));
|
||||
}
|
||||
}
|
||||
|
||||
System.out.printf(Locale.ROOT,
|
||||
"FleetConfig.withDefaults() component-survival coverage — %d components, %d checked, "
|
||||
+ "%d excluded, %d survived%n",
|
||||
COMPONENTS.length, checked, EXCLUDED_FROM_SURVIVAL_CHECK.size(), checked - dropped.size());
|
||||
assertEquals(List.of(), dropped,
|
||||
"withDefaults() silently dropped these real, given components: " + dropped);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,232 @@
|
||||
package dev.ltms.fleet.deploy;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.params.ParameterizedTest;
|
||||
import org.junit.jupiter.params.provider.MethodSource;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #360: {@code deploy/fleetd.service} started clean on fleet01 and broke the daemon in
|
||||
* three ways that nothing logs at startup.
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@code ProtectSystem=}, {@code ProtectHome=}, {@code ProtectKernelTunables=} and
|
||||
* {@code ProtectControlGroups=} each give the unit its own mount namespace. fleetd resolves
|
||||
* a caller's role by running {@code lsof} to find the loopback peer PID (see
|
||||
* {@code mcp/LsofPeerPidLookup}). Inside such a namespace {@code lsof} returns nothing,
|
||||
* every caller falls back to ANONYMOUS, and the primary is refused every orchestration call
|
||||
* with "unauthenticated: anonymous may not SPAWN" -- while healthz still reports ok.
|
||||
* Measured by bisection on fleet01 2026-09-05 (lsof line count): no sandbox 3,
|
||||
* {@code ProtectSystem=strict} 0, {@code ProtectHome=read-only} 0,
|
||||
* {@code ProtectKernelTunables} 0, {@code ProtectControlGroups} 0,
|
||||
* {@code RestrictSUIDSGID} 3, {@code NoNewPrivileges} 3 -- so only the first four are
|
||||
* forbidden.
|
||||
* <li>{@code PrivateTmp=true} gives the unit its own {@code /tmp}. fleetd writes the member
|
||||
* ZDOTDIR credential-scrub directory and the opencode config directory under
|
||||
* {@code java.io.tmpdir}, and the member pane -- a child of a *different* unit
|
||||
* (herdr.service) -- has to read them back. A private {@code /tmp} on either unit turns the
|
||||
* credential scrub into a silent no-op.
|
||||
* <li>Running {@code java} directly from {@code ExecStart} skips the login shell. Every secret
|
||||
* fleetd needs (AI_GATEWAY_TOKEN, WORKER_GITEA_TOKEN, LAVINMQ_URI, COORD_AMQP_URI) lives in
|
||||
* a file only the login shell sources; systemd runs no login shell on its own. Started that
|
||||
* way the daemon boots fine with empty credentials, and the failure appears hours later as a
|
||||
* member that cannot open a pull request.
|
||||
* </ul>
|
||||
*
|
||||
* <p>Review round 2 on fleetd #360 found the same shape one file further down the chain:
|
||||
* {@code herdr.service}'s {@code ExecStart} only names {@code deploy/herdr-inner.sh}, so that
|
||||
* script is the ONLY place two more of these properties live, and nothing was reading it.
|
||||
*
|
||||
* <ul>
|
||||
* <li>If the script does not exec a login shell, herdr -- and every member pane it spawns as a
|
||||
* child -- starts with none of the secrets only a login shell sources, the same silent
|
||||
* empty-credentials failure as fleetd.service's {@code ExecStart}, one process further away.
|
||||
* <li>If the script does not set a real, non-zero pty size before starting herdr (with
|
||||
* {@code stty}), herdr reports a 0x0 window and every pane spawn fails with
|
||||
* "ghostty error -2" -- which surfaces as a fleetd spawn failure, not a herdr one.
|
||||
* </ul>
|
||||
*
|
||||
* <p>None of these five failures makes the unit (or the script) fail to start, or makes
|
||||
* {@code /healthz} report unhealthy, so nothing short of reading the files catches a regression.
|
||||
* This test is that read.
|
||||
*
|
||||
* <p><b>It checks source text, not behaviour</b> -- it cannot start systemd or fork an mount
|
||||
* namespace in a build sandbox. It parses the same two files an operator would install and fails
|
||||
* if a forbidden directive is active, exactly the way {@code McpContractDocTest} guards
|
||||
* {@code docs/MCP-Contract.md} against naming a tool that does not exist.
|
||||
*/
|
||||
class SystemdUnitSafetyTest {
|
||||
|
||||
/** Tests run with the module directory (fleetd/) as cwd; deploy/ is the repo-root sibling. */
|
||||
private static final Path FLEETD_SERVICE = Path.of("../deploy/fleetd.service");
|
||||
private static final Path HERDR_SERVICE = Path.of("../deploy/herdr.service");
|
||||
private static final Path HERDR_INNER_SCRIPT = Path.of("../deploy/herdr-inner.sh");
|
||||
|
||||
private static final List<String> NAMESPACING_DIRECTIVES = List.of(
|
||||
"ProtectSystem", "ProtectHome", "ProtectKernelTunables", "ProtectControlGroups");
|
||||
|
||||
/**
|
||||
* A shell invoked with a login flag, e.g. {@code zsh -lc '...'} or {@code /bin/sh -l}. The
|
||||
* property under test is the {@code -l}, not the specific shell or the rest of its flags, so
|
||||
* this matches a token ending in "sh" followed by a flag cluster containing "l" -- tighter
|
||||
* than a bare {@code contains("-lc")}, which a "-lc" anywhere in the line would also satisfy.
|
||||
*/
|
||||
private static final Pattern LOGIN_SHELL_INVOCATION =
|
||||
Pattern.compile("(?:^|\\s)\\S*sh\\s+-[A-Za-z]*l[A-Za-z]*\\b");
|
||||
|
||||
private static final Pattern POSITIVE_ROWS = Pattern.compile("\\brows\\s+([1-9]\\d*)\\b");
|
||||
private static final Pattern POSITIVE_COLS = Pattern.compile("\\bcols\\s+([1-9]\\d*)\\b");
|
||||
|
||||
static Stream<Path> bothUnits() {
|
||||
return Stream.of(FLEETD_SERVICE, HERDR_SERVICE);
|
||||
}
|
||||
|
||||
private static boolean invokesALoginShell(List<String> activeLines) {
|
||||
return activeLines.stream().anyMatch(l -> LOGIN_SHELL_INVOCATION.matcher(l).find());
|
||||
}
|
||||
|
||||
/** An active {@code stty} line naming both a positive row count and a positive column count. */
|
||||
private static boolean setsANonZeroPtySize(List<String> activeLines) {
|
||||
return activeLines.stream()
|
||||
.filter(l -> l.startsWith("stty"))
|
||||
.anyMatch(l -> POSITIVE_ROWS.matcher(l).find() && POSITIVE_COLS.matcher(l).find());
|
||||
}
|
||||
|
||||
/** Lines that are actually in force: comments and blank lines don't count. */
|
||||
private static List<String> activeLines(Path unit) throws Exception {
|
||||
List<String> active = new ArrayList<>();
|
||||
for (String line : Files.readAllLines(unit)) {
|
||||
String stripped = line.strip();
|
||||
if (!stripped.isEmpty() && !stripped.startsWith("#")) {
|
||||
active.add(stripped);
|
||||
}
|
||||
}
|
||||
return active;
|
||||
}
|
||||
|
||||
@ParameterizedTest
|
||||
@MethodSource("bothUnits")
|
||||
@DisplayName("[SOURCE TEXT] no active mount-namespacing directive -- it blinds lsof and turns every caller ANONYMOUS")
|
||||
void doesNotActivateAMountNamespace(Path unit) throws Exception {
|
||||
List<String> active = activeLines(unit);
|
||||
|
||||
for (String directive : NAMESPACING_DIRECTIVES) {
|
||||
Pattern activeDirective = Pattern.compile("^" + Pattern.quote(directive) + "\\s*=");
|
||||
List<String> hits = active.stream().filter(l -> activeDirective.matcher(l).find()).toList();
|
||||
assertTrue(hits.isEmpty(),
|
||||
unit + " sets " + directive + " (" + hits + "). That directive gives the unit its "
|
||||
+ "own mount namespace; inside it, fleetd's lsof-based caller lookup "
|
||||
+ "(mcp/LsofPeerPidLookup) returns nothing, so every MCP caller falls back to "
|
||||
+ "ANONYMOUS and the primary is refused every orchestration call with "
|
||||
+ "\"unauthenticated: anonymous may not SPAWN\" -- while the daemon still "
|
||||
+ "starts and /healthz still reports ok. Measured on fleet01 2026-09-05 "
|
||||
+ "(fleetd #360). A commented-out mention in the file's own DO-NOT-add block "
|
||||
+ "is fine; an active directive is not.");
|
||||
}
|
||||
}
|
||||
|
||||
@ParameterizedTest
|
||||
@MethodSource("bothUnits")
|
||||
@DisplayName("[SOURCE TEXT] PrivateTmp is not true -- it silently no-ops the credential scrub")
|
||||
void privateTmpIsNotTrue(Path unit) throws Exception {
|
||||
List<String> active = activeLines(unit);
|
||||
Pattern privateTmpTrue = Pattern.compile("^PrivateTmp\\s*=\\s*true\\b");
|
||||
|
||||
boolean hasPrivateTmpTrue = active.stream().anyMatch(l -> privateTmpTrue.matcher(l).find());
|
||||
assertFalse(hasPrivateTmpTrue,
|
||||
unit + " sets PrivateTmp=true. fleetd writes the member ZDOTDIR credential-scrub "
|
||||
+ "directory and the opencode config directory under java.io.tmpdir, and the "
|
||||
+ "member pane -- a child of a DIFFERENT unit -- has to read them back. A "
|
||||
+ "private /tmp on either unit turns the credential scrub into a silent no-op: "
|
||||
+ "no error, no log line, the scrub just never happens (fleetd #360).");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] fleetd.service's ExecStart goes through a login shell -- otherwise every secret is empty")
|
||||
void execStartUsesALoginShell() throws Exception {
|
||||
List<String> active = activeLines(FLEETD_SERVICE);
|
||||
List<String> execStartLines = active.stream().filter(l -> l.startsWith("ExecStart=")).toList();
|
||||
|
||||
assertTrue(execStartLines.size() == 1,
|
||||
FLEETD_SERVICE + " must have exactly one active ExecStart= line; found "
|
||||
+ execStartLines.size() + ": " + execStartLines);
|
||||
|
||||
String execStart = execStartLines.get(0);
|
||||
assertTrue(LOGIN_SHELL_INVOCATION.matcher(execStart).find(),
|
||||
FLEETD_SERVICE + "'s ExecStart (" + execStart + ") does not run a login shell (a shell "
|
||||
+ "invoked with a \"-l\" flag, e.g. \"zsh -lc\"). Every secret fleetd needs "
|
||||
+ "(AI_GATEWAY_TOKEN, WORKER_GITEA_TOKEN, LAVINMQ_URI, COORD_AMQP_URI) lives in a "
|
||||
+ "file only the login shell sources; systemd runs no login shell on its own. "
|
||||
+ "Running java directly boots fine with every credential empty, and the failure "
|
||||
+ "surfaces hours later as a member that cannot open a pull request (fleetd #360).");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] herdr-inner.sh execs a login shell -- otherwise herdr and every member it spawns start with empty credentials")
|
||||
void herdrInnerScriptUsesALoginShell() throws Exception {
|
||||
List<String> active = activeLines(HERDR_INNER_SCRIPT);
|
||||
|
||||
assertTrue(invokesALoginShell(active),
|
||||
HERDR_INNER_SCRIPT + " does not exec a login shell (a shell invoked with a \"-l\" flag, "
|
||||
+ "e.g. \"zsh -lc\"). herdr.service's ExecStart only names this script, so this "
|
||||
+ "is the ONLY place herdr's login-shell property lives. Without it, herdr -- and "
|
||||
+ "every member pane it spawns as a child of herdr -- starts with none of the "
|
||||
+ "secrets that only a login shell sources (this host: ~/.fleet/secrets.sh via "
|
||||
+ "~/.zprofile), and the failure surfaces hours later as a member with no "
|
||||
+ "credentials at all, not just fleetd (fleetd #360).");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] herdr-inner.sh sets a non-zero pty size before starting herdr -- otherwise every pane spawn fails with \"ghostty error -2\"")
|
||||
void herdrInnerScriptSetsANonZeroPtySize() throws Exception {
|
||||
List<String> active = activeLines(HERDR_INNER_SCRIPT);
|
||||
|
||||
assertTrue(setsANonZeroPtySize(active),
|
||||
HERDR_INNER_SCRIPT + " does not run \"stty rows N cols M\" with both N and M positive "
|
||||
+ "before starting herdr. Without a real pty size, herdr reports a 0x0 window and "
|
||||
+ "every pane spawn fails with \"ghostty error -2\" -- which surfaces as a fleetd "
|
||||
+ "spawn failure, not a herdr one, and gives no hint that the actual cause is this "
|
||||
+ "script (fleetd #360).");
|
||||
}
|
||||
|
||||
/**
|
||||
* The denominator guard, same shape as {@code McpContractDocTest}'s: a check that scans for a
|
||||
* forbidden pattern passes trivially if it is handed nothing to scan. Pin that all three files
|
||||
* exist, are non-trivial, and that the DO-NOT block's commented mentions are still there -- so
|
||||
* the "comment survives, directive doesn't" distinction above is actually being exercised.
|
||||
*/
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] the namespacing check is not vacuous -- all three files exist and the DO-NOT block still mentions every forbidden directive in a comment")
|
||||
void theCheckActuallyHasSomethingToCheck() throws Exception {
|
||||
assertTrue(Files.size(FLEETD_SERVICE) > 200,
|
||||
FLEETD_SERVICE + " is missing or unexpectedly small -- the checks above would pass "
|
||||
+ "vacuously against an empty or absent file");
|
||||
assertTrue(Files.size(HERDR_SERVICE) > 200,
|
||||
HERDR_SERVICE + " is missing or unexpectedly small -- the checks above would pass "
|
||||
+ "vacuously against an empty or absent file");
|
||||
assertTrue(Files.size(HERDR_INNER_SCRIPT) > 50,
|
||||
HERDR_INNER_SCRIPT + " is missing or unexpectedly small -- the login-shell and "
|
||||
+ "pty-size checks above would either pass vacuously or fail with an opaque "
|
||||
+ "\"file not found\" instead of naming the actual consequence (empty "
|
||||
+ "credentials / \"ghostty error -2\") against an empty or absent file");
|
||||
|
||||
String fleetdService = Files.readString(FLEETD_SERVICE);
|
||||
for (String directive : NAMESPACING_DIRECTIVES) {
|
||||
assertTrue(fleetdService.contains(directive),
|
||||
FLEETD_SERVICE + " no longer mentions " + directive + " anywhere, not even in the "
|
||||
+ "DO-NOT-add comment block that explains why it must stay out. That comment "
|
||||
+ "is the whole point of fleetd #360 -- it is what stops the next edit from "
|
||||
+ "re-adding the directive without knowing why.");
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -24,10 +24,13 @@ import org.slf4j.LoggerFactory;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledFuture;
|
||||
import java.util.concurrent.ScheduledThreadPoolExecutor;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
@@ -204,6 +207,139 @@ class FleetHealthMonitorTest {
|
||||
.workingSuspectAfterOrDefault());
|
||||
}
|
||||
|
||||
// --- fleetd #386: a stall detector whose only clock freezes with a sleeping host is worse
|
||||
// than a silent one — it reports "quiet" for a member that was genuinely busy for hours.
|
||||
|
||||
@Test void monotonicClockFrozenPastThresholdOnRealClockStillReportsStallSuspected() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
AtomicLong mono = new AtomicLong(0);
|
||||
AtomicLong real = new AtomicLong(0);
|
||||
FakeHerdr herdr = new FakeHerdr().withAgent("busy", "term_busy", "pane_busy", "tab_busy");
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
FleetHealthMonitor monitor = monitorWithClocks(herdr,
|
||||
List.of(member("term_busy", MemberSession.State.BUSY, 0, 0)), scheduler,
|
||||
mono::get, real::get, 60, 600, (_, _) -> { });
|
||||
|
||||
monitor.tick(); // establishes the clock baseline; nothing has diverged yet
|
||||
assertEquals(0, appender.list.stream().filter(event -> event.getFormattedMessage()
|
||||
.contains("state=STALL_SUSPECTED")).count());
|
||||
|
||||
// The host "sleeps": the monotonic clock stands completely still while the real clock
|
||||
// keeps moving, past the 600s stall threshold.
|
||||
real.set(TimeUnit.SECONDS.toNanos(700));
|
||||
monitor.tick();
|
||||
monitor.stop();
|
||||
|
||||
assertTrue(appender.list.stream().anyMatch(event -> event.getFormattedMessage()
|
||||
.contains("member=term_busy state=STALL_SUSPECTED")),
|
||||
"the real clock crossed the stall threshold even though the monotonic clock never moved");
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
@Test void clockDivergenceIsLoggedOnceNotOncePerTick() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
AtomicLong mono = new AtomicLong(0);
|
||||
AtomicLong real = new AtomicLong(0);
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
FleetHealthMonitor monitor = monitorWithClocks(new FakeHerdr(), List.of(), scheduler,
|
||||
mono::get, real::get, 60, 600, (_, _) -> { });
|
||||
|
||||
monitor.tick(); // baseline: no divergence possible yet
|
||||
|
||||
// One sleep gap: the monotonic clock is frozen while the real clock jumps far past one
|
||||
// tick interval (60s).
|
||||
real.set(TimeUnit.SECONDS.toNanos(700));
|
||||
monitor.tick();
|
||||
|
||||
// The host is awake again: both clocks advance together from here, so no more divergence.
|
||||
mono.set(TimeUnit.SECONDS.toNanos(10));
|
||||
real.set(TimeUnit.SECONDS.toNanos(710));
|
||||
monitor.tick();
|
||||
mono.set(TimeUnit.SECONDS.toNanos(20));
|
||||
real.set(TimeUnit.SECONDS.toNanos(720));
|
||||
monitor.tick();
|
||||
monitor.stop();
|
||||
|
||||
assertEquals(1, appender.list.stream().filter(event -> event.getFormattedMessage()
|
||||
.contains("the monotonic clock did not advance"))
|
||||
.count(), "one sleep gap must produce exactly one divergence line, not one per tick");
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #386 follow-up, added on merge. The fix carries a PER-MEMBER drift baseline, so drift
|
||||
* from a sleep that happened BEFORE a member went busy is never charged to that member. The
|
||||
* two tests shipped with the fix both start with the member already BUSY, so a single global
|
||||
* baseline passes them — this one fails without the per-member map.
|
||||
*
|
||||
* <p>Order matters: the host sleeps while nothing is busy, and only then does a member take a
|
||||
* turn. Its stall clock must start at zero.
|
||||
*/
|
||||
@Test void driftFromASleepBeforeAMemberWentBusyIsNotChargedToThatMember() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
AtomicLong mono = new AtomicLong(0);
|
||||
AtomicLong real = new AtomicLong(0);
|
||||
AtomicReference<List<MemberSession>> roster = new AtomicReference<>(List.of());
|
||||
FakeHerdr herdr = new FakeHerdr().withAgent("busy", "term_busy", "pane_busy", "tab_busy");
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents, roster::get,
|
||||
new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox()),
|
||||
scheduler, mono::get, real::get, 60, 600, (_, _) -> { });
|
||||
|
||||
monitor.tick(); // baseline, no members yet
|
||||
|
||||
// The host sleeps for 700s with nobody busy: the monotonic clock stands still.
|
||||
real.set(TimeUnit.SECONDS.toNanos(700));
|
||||
monitor.tick();
|
||||
|
||||
// Awake again. Only NOW does a member start a turn, with a fresh activity stamp taken
|
||||
// from the monotonic clock. Both clocks advance together from here.
|
||||
mono.set(TimeUnit.SECONDS.toNanos(10));
|
||||
real.set(TimeUnit.SECONDS.toNanos(710));
|
||||
roster.set(List.of(member("term_busy", MemberSession.State.BUSY,
|
||||
0, TimeUnit.SECONDS.toNanos(10))));
|
||||
monitor.tick();
|
||||
|
||||
mono.set(TimeUnit.SECONDS.toNanos(20));
|
||||
real.set(TimeUnit.SECONDS.toNanos(720));
|
||||
monitor.tick();
|
||||
monitor.stop();
|
||||
|
||||
assertEquals(0, appender.list.stream().filter(event -> event.getFormattedMessage()
|
||||
.contains("member=term_busy state=STALL_SUSPECTED")).count(),
|
||||
"the member has been busy for 10s, not 710s — the earlier sleep is not its stall");
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetHealthMonitor monitorWithClocks(FakeHerdr herdr, List<MemberSession> roster,
|
||||
java.util.concurrent.ScheduledExecutorService scheduler, LongSupplier clock,
|
||||
LongSupplier realtimeClock, long intervalSeconds, long workingSuspectAfterSeconds,
|
||||
BiConsumer<String, String> failTarget) {
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
return new FleetHealthMonitor(agents, () -> roster,
|
||||
new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox()),
|
||||
scheduler, clock, realtimeClock, intervalSeconds, workingSuspectAfterSeconds, failTarget);
|
||||
}
|
||||
|
||||
@Test void goneMemberRecoveryLogsOnceWithoutRefiringTargetFailure() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
Level previousLevel = logger.getLevel();
|
||||
@@ -434,4 +570,120 @@ class FleetHealthMonitorTest {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
// --- fleetd #280: a GONE/NEVER_READY guess whose fleet_ask lapses AFTER the first sweep must
|
||||
// still be swept, without ever reaching into a target that has since recovered or left the
|
||||
// roster. See FleetHealthMonitor.recheckTerminalTarget's javadoc for the full reachability chain.
|
||||
|
||||
@Test void terminalTransitionSchedulesExactlyOneDelayedRecheck() {
|
||||
ScheduledThreadPoolExecutor scheduler = new ScheduledThreadPoolExecutor(1);
|
||||
FleetHealthMonitor monitor = monitor(new FakeHerdr(),
|
||||
List.of(member("term_a", MemberSession.State.BUSY, 0, 0)), scheduler, () -> 1, 600,
|
||||
(_, _) -> { });
|
||||
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
|
||||
// Driven via reportTransition directly (not tick()), so the queue holds only the recheck.
|
||||
assertEquals(1, scheduler.getQueue().size());
|
||||
ScheduledFuture<?> scheduled = (ScheduledFuture<?>) scheduler.getQueue().peek();
|
||||
assertTrue(scheduled.getDelay(TimeUnit.SECONDS) > 100,
|
||||
"the delay must clear the worst-case fleet_ask lapse window (up to 115s)");
|
||||
|
||||
// An unchanged tick must not queue a second one (CB-580's fire-once rule extends to this).
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
assertEquals(1, scheduler.getQueue().size());
|
||||
monitor.stop();
|
||||
}
|
||||
|
||||
@Test void recheckIsANoOpOnceTheTargetHasRecovered() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
assertEquals(1, failTarget.calls.size());
|
||||
|
||||
monitor.reportTransition("term_a", HealthState.IDLE); // recovered before the recheck fired
|
||||
monitor.recheckTerminalTarget("term_a", HealthState.GONE);
|
||||
|
||||
assertEquals(1, failTarget.calls.size(), "a recovered target must not be reached into again");
|
||||
monitor.stop();
|
||||
}
|
||||
|
||||
@Test void recheckIsANoOpForATargetItNeverObserved() {
|
||||
// Mirrors "left the roster": tick() prunes states.keySet() to the current roster on release
|
||||
// (see FleetHealthMonitor.tick), so a target this monitor never recorded is the same case.
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
|
||||
monitor.recheckTerminalTarget("term_never_seen", HealthState.GONE);
|
||||
|
||||
assertEquals(0, failTarget.calls.size(), "an untracked/released target must not be reached into");
|
||||
monitor.stop();
|
||||
}
|
||||
|
||||
/**
|
||||
* The scenario from the ticket, end to end, driven through the real {@link MessageService}: a
|
||||
* target's ask is still genuinely open when health first observes GONE (sweep must skip it,
|
||||
* exactly as {@code abandonDoesNotFailAnAsyncTicketWaitingForAnAnswer} pins), the ask then lapses
|
||||
* on its own, an unchanged tick still must not refire, and only the delayed recheck sweeps the
|
||||
* now-lapsed ticket to FAILED.
|
||||
*/
|
||||
@Test void delayedRecheckSweepsATicketWhoseAskLapsedAfterGoneWasFirstObserved() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr().withAgent("worker", "term_a", "pane-term_a", "tab_a")
|
||||
.readText("$ prompt");
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
Injector injector = new Injector(agents);
|
||||
InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||
inbox.own("term_a");
|
||||
MessageService messages = new MessageService(agents, injector, rendezvous, inbox);
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents,
|
||||
() -> List.of(member("term_a", MemberSession.State.READY, 0, 0)), messages,
|
||||
new ScheduledThreadPoolExecutor(1), () -> 1, 60, 600, messages::abandon);
|
||||
|
||||
String ticket = messages.sendAsync("term_a", "task that asks");
|
||||
long deadline = System.currentTimeMillis() + 2000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the async send should have opened its waiter");
|
||||
injector.onStatus("term_a", AgentStatus.IDLE); // deliver the task
|
||||
injector.onStatus("term_a", AgentStatus.WORKING); // the worker picks it up
|
||||
|
||||
// The worker asks, with a short timeout so its own fleet_ask lapses quickly in test time.
|
||||
CompletableFuture<MessageService.AskResult> ask = CompletableFuture.supplyAsync(
|
||||
() -> messages.ask("term_a", "which config?", 200));
|
||||
awaitPhase(messages, ticket, MessageService.Phase.ASKING);
|
||||
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
assertEquals(MessageService.Phase.ASKING, messages.poll(ticket).phase(),
|
||||
"the first sweep must not fail a ticket that is still genuinely being asked");
|
||||
|
||||
// The worker's own fleet_ask now lapses on its own — task.question clears to null.
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT, ask.get(5, TimeUnit.SECONDS).outcome());
|
||||
|
||||
// An unchanged tick still must not refire (CB-580).
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket).phase());
|
||||
|
||||
// The delayed recheck scheduled for the original transition finally sweeps it.
|
||||
monitor.recheckTerminalTarget("term_a", HealthState.GONE);
|
||||
assertEquals(MessageService.Phase.FAILED, messages.poll(ticket).phase());
|
||||
|
||||
monitor.stop();
|
||||
}
|
||||
|
||||
private static MessageService.TaskView awaitPhase(MessageService messages, String ticket,
|
||||
MessageService.Phase phase) throws InterruptedException {
|
||||
long deadline = System.currentTimeMillis() + 2000;
|
||||
MessageService.TaskView view;
|
||||
do {
|
||||
view = messages.poll(ticket);
|
||||
if (view.phase() == phase) {
|
||||
return view;
|
||||
}
|
||||
Thread.sleep(5);
|
||||
} while (System.currentTimeMillis() < deadline);
|
||||
assertEquals(phase, view.phase());
|
||||
return view;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -7,6 +7,7 @@ import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
|
||||
/**
|
||||
@@ -39,8 +40,12 @@ public final class FakeHerdr implements HerdrClient {
|
||||
private final Map<String, List<String>> extraTabs = new LinkedHashMap<>();
|
||||
private int agentNameTakenFor = 0;
|
||||
private int agentPaneBusyFor = 0;
|
||||
private String agentStartErrorCode = null;
|
||||
private int workerTabPaneCount = 1;
|
||||
private String paneCloseErrorCode = null;
|
||||
private final Map<String, String> paneCloseErrorCodeFor = new ConcurrentHashMap<>();
|
||||
private String tabCloseErrorCode = null;
|
||||
private final Map<String, String> tabCloseErrorCodeFor = new ConcurrentHashMap<>();
|
||||
private String agentSendErrorCode = null;
|
||||
private boolean noPanes = false;
|
||||
private volatile String agentStatus = "idle"; // steady-state agent.get status
|
||||
@@ -80,18 +85,56 @@ public final class FakeHerdr implements HerdrClient {
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make every {@code agent.start} call fail with this herdr error code. */
|
||||
public FakeHerdr agentStartFailsWith(String code) {
|
||||
this.agentStartErrorCode = code;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make the worker tab (w9:t2) report this many panes in {@code tab.list} (default 1). */
|
||||
public FakeHerdr withWorkerTabPaneCount(int n) {
|
||||
this.workerTabPaneCount = n;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make {@code pane.close} fail with this herdr error code. */
|
||||
/** Make {@code pane.close} fail with this herdr error code, for every pane. */
|
||||
public FakeHerdr paneCloseFailsWith(String code) {
|
||||
this.paneCloseErrorCode = code;
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Make {@code pane.close} fail with this herdr error code, but only for the given {@code
|
||||
* pane_id} — every other pane's {@code pane.close} still succeeds. Unlike {@link
|
||||
* #paneCloseFailsWith}, which fails every call regardless of which pane it targets, this lets a
|
||||
* test reap/release several sessions at once and make exactly one of them fail to stop, so the
|
||||
* others' teardown can be asserted to proceed normally (fleetd #290).
|
||||
*/
|
||||
public FakeHerdr paneCloseFailsForPane(String paneId, String code) {
|
||||
this.paneCloseErrorCodeFor.put(paneId, code);
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make {@code tab.close} fail with this herdr error code, for every tab. */
|
||||
public FakeHerdr tabCloseFailsWith(String code) {
|
||||
this.tabCloseErrorCode = code;
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Make {@code tab.close} fail with this herdr error code, but only for the given {@code
|
||||
* tab_id} — every other tab's {@code tab.close} still succeeds. The {@code tab.close}
|
||||
* counterpart to {@link #paneCloseFailsForPane} (fleetd #290): lets a test make exactly one
|
||||
* session's tab teardown fail while proving the rest of {@code stop()} — {@code
|
||||
* releaseZdotdir}, and the caller's worktree removal — still runs (fleetd #293). Named "ForTab"
|
||||
* rather than "ForPane" (unlike its sibling) because {@code tab.close} keys on {@code tab_id},
|
||||
* not a pane id.
|
||||
*/
|
||||
public FakeHerdr tabCloseFailsForTab(String tabId, String code) {
|
||||
this.tabCloseErrorCodeFor.put(tabId, code);
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Make {@code pane.list} report no panes at all — models a second herdr daemon (CB-185) that
|
||||
* simply does not host the pane a {@link PaneLocator} is searching for.
|
||||
@@ -275,6 +318,10 @@ public final class FakeHerdr implements HerdrClient {
|
||||
+ required + "`", "invalid_request", null);
|
||||
}
|
||||
}
|
||||
if (agentStartErrorCode != null) {
|
||||
throw new HerdrException("herdr error [" + agentStartErrorCode + "]: agent.start failed",
|
||||
agentStartErrorCode, null);
|
||||
}
|
||||
long starts = calls.stream().filter(c -> c.method().equals("agent.start")).count();
|
||||
if (starts <= agentPaneBusyFor) {
|
||||
throw new HerdrException(
|
||||
@@ -338,7 +385,17 @@ public final class FakeHerdr implements HerdrClient {
|
||||
.formatted(workerTabPaneCount,
|
||||
seeded.isEmpty() ? "" : "," + String.join(",", seeded)));
|
||||
}
|
||||
case "tab.close" -> mapper.readTree("{\"type\":\"ok\"}");
|
||||
case "tab.close" -> {
|
||||
Object tabIdParam = params instanceof Map<?, ?> m ? m.get("tab_id") : null;
|
||||
String perTabCode = tabIdParam == null ? null
|
||||
: tabCloseErrorCodeFor.get(String.valueOf(tabIdParam));
|
||||
String code = perTabCode != null ? perTabCode : tabCloseErrorCode;
|
||||
if (code != null) {
|
||||
throw new HerdrException("herdr error [" + code + "]: tab.close failed",
|
||||
code, null);
|
||||
}
|
||||
yield mapper.readTree("{\"type\":\"ok\"}");
|
||||
}
|
||||
case "pane.get" -> mapper.readTree("""
|
||||
{"type":"pane_info","pane":{"pane_id":"w9:pW","workspace_id":"w9",
|
||||
"tab_id":"w9:t2","agent_status":"idle"}}""");
|
||||
@@ -360,9 +417,13 @@ public final class FakeHerdr implements HerdrClient {
|
||||
"foreground_processes":[]}}""");
|
||||
}
|
||||
case "pane.close" -> {
|
||||
if (paneCloseErrorCode != null) {
|
||||
throw new HerdrException("herdr error [" + paneCloseErrorCode + "]: pane.close failed",
|
||||
paneCloseErrorCode, null);
|
||||
Object paneIdParam = params instanceof Map<?, ?> m ? m.get("pane_id") : null;
|
||||
String perPaneCode = paneIdParam == null ? null
|
||||
: paneCloseErrorCodeFor.get(String.valueOf(paneIdParam));
|
||||
String code = perPaneCode != null ? perPaneCode : paneCloseErrorCode;
|
||||
if (code != null) {
|
||||
throw new HerdrException("herdr error [" + code + "]: pane.close failed",
|
||||
code, null);
|
||||
}
|
||||
yield mapper.readTree("{\"type\":\"ok\"}");
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
@@ -35,6 +36,12 @@ class LeadTabScannerTest {
|
||||
final Map<String, String[]> tabs = new LinkedHashMap<>();
|
||||
/** pane_id → [tab_id, terminal_id]. */
|
||||
final Map<String, String[]> panes = new LinkedHashMap<>();
|
||||
/**
|
||||
* tab_id → whether herdr reports a running agent there. Defaults to {@code true} for every
|
||||
* tab that has a pane, so every existing fixture keeps meaning "a live lead" unless a test
|
||||
* says otherwise via {@link #deadAgent}.
|
||||
*/
|
||||
final Set<String> deadTabs = new LinkedHashSet<>();
|
||||
int calls;
|
||||
boolean failing;
|
||||
|
||||
@@ -53,6 +60,18 @@ class LeadTabScannerTest {
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Mark {@code tabId} as labelled but agent-less — a dead lead's leftover tab (fleetd #359). */
|
||||
TopologyHerdr deadAgent(String tabId) {
|
||||
deadTabs.add(tabId);
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Undo {@link #deadAgent} — models {@code agent.list} reporting the tab live again. */
|
||||
TopologyHerdr reviveAgent(String tabId) {
|
||||
deadTabs.remove(tabId);
|
||||
return this;
|
||||
}
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
calls++;
|
||||
@@ -83,6 +102,19 @@ class LeadTabScannerTest {
|
||||
.formatted(id, p[0], p[1])));
|
||||
return read("{\"panes\":[%s]}".formatted(String.join(",", items)));
|
||||
}
|
||||
case "agent.list" -> {
|
||||
// One agent per distinct tab that has a pane and isn't marked dead — mirrors
|
||||
// AgentControl.list()'s "agents" shape closely enough for the scanner's join,
|
||||
// which only reads tab_id off each entry.
|
||||
Set<String> seen = new LinkedHashSet<>();
|
||||
panes.forEach((paneId, p) -> {
|
||||
String tabId = p[0];
|
||||
if (!deadTabs.contains(tabId) && seen.add(tabId)) {
|
||||
items.add("{\"tab_id\":\"%s\"}".formatted(tabId));
|
||||
}
|
||||
});
|
||||
return read("{\"agents\":[%s]}".formatted(String.join(",", items)));
|
||||
}
|
||||
default -> throw new AssertionError("unexpected herdr call: " + method);
|
||||
}
|
||||
}
|
||||
@@ -208,6 +240,114 @@ class LeadTabScannerTest {
|
||||
scanner(herdr, twoLeadsConfigured(), new AtomicLong()).get().get("term_opus_split"));
|
||||
}
|
||||
|
||||
// ── fleetd #359: a labelled tab is only a lead when something is running in it ─────────────────
|
||||
|
||||
/**
|
||||
* The core invariant this ticket restores: {@code get()} must never report a terminal for a
|
||||
* lead whose pane no longer runs an agent. Before this fix the scanner joined labelled tabs to
|
||||
* panes with no liveness check at all, so a tab left behind by a crashed/relaunched lead (see
|
||||
* {@code LeadLauncher}'s own staleness handling) was reported as live forever — which is exactly
|
||||
* what let duplicate lead tabs make {@code LeadCoordLoop.resolveLocalLead()} permanently unable
|
||||
* to pick one. Mutate this away (drop the {@code agent.list} cross-check in {@link
|
||||
* LeadTabScanner#scan()}) and this test must fail.
|
||||
*/
|
||||
@Test
|
||||
void aLabelledTabWithNoRunningAgentIsNotReported() {
|
||||
TopologyHerdr herdr = twoLeads().deadAgent("w1:t1"); // opus-5.0's tab is labelled but dead
|
||||
|
||||
Map<String, String> leads = scanner(herdr, twoLeadsConfigured(), new AtomicLong()).get();
|
||||
|
||||
assertFalse(leads.containsKey("term_opus"),
|
||||
"a labelled tab with no running agent must never be reported as a live lead");
|
||||
assertEquals("gpt-sol-5.6", leads.get("term_gpt"),
|
||||
"the other, genuinely live lead must be unaffected");
|
||||
}
|
||||
|
||||
/**
|
||||
* The other direction, pinned separately so a fix cannot satisfy the test above by simply
|
||||
* returning nothing: a labelled tab that DOES have a running agent must still be reported. A
|
||||
* scanner that always comes back empty is worse than the bug it fixes.
|
||||
*/
|
||||
@Test
|
||||
void aLabelledTabWithARunningAgentIsStillReported() {
|
||||
Map<String, String> leads = scanner(twoLeads(), twoLeadsConfigured(), new AtomicLong()).get();
|
||||
|
||||
assertEquals("opus-5.0", leads.get("term_opus"));
|
||||
assertEquals("gpt-sol-5.6", leads.get("term_gpt"));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #359 review, finding 2 — the exact scenario the ticket's own evidence showed:
|
||||
* {@code agent.list} can come back successfully but short, without the herdr call ever throwing.
|
||||
* A lead this class already reported as live must not be dropped on the strength of one such
|
||||
* read: {@link LeadTabScanner#get()}'s "keep the cache on failure" contract only fires on an
|
||||
* exception, so without a fix a single short {@code agent.list} silently empties the cached map —
|
||||
* which would resolve that lead's pane as {@code Role.WORKER} downstream, refusing every
|
||||
* orchestration call. Mutate this away (drop the one-scan grace in {@link
|
||||
* LeadTabScanner#scan()}) and this test must fail.
|
||||
*/
|
||||
@Test
|
||||
void aTransientAgentListMissDoesNotDemoteALeadAlreadyKnownLive() {
|
||||
TopologyHerdr herdr = twoLeads();
|
||||
AtomicLong clock = new AtomicLong();
|
||||
LeadTabScanner s = scanner(herdr, twoLeadsConfigured(), clock);
|
||||
assertTrue(s.get().containsKey("term_opus"), "opus must be known live before the miss");
|
||||
|
||||
// One scan where agent.list comes back without opus's tab, even though the tab and pane are
|
||||
// completely unchanged — the tab/pane are still there, only the liveness read is short.
|
||||
herdr.deadAgent("w1:t1");
|
||||
clock.addAndGet(TTL);
|
||||
|
||||
assertTrue(s.get().containsKey("term_opus"),
|
||||
"a single missed detection must not empty the cached map for a lead already known live");
|
||||
assertEquals("gpt-sol-5.6", s.get().get("term_gpt"), "the unaffected lead is unchanged");
|
||||
}
|
||||
|
||||
/**
|
||||
* The other half of finding 2, so the grace above cannot be mistaken for permanent amnesty: the
|
||||
* original #359 invariant (a genuinely dead tab is not reported forever) must still hold once a
|
||||
* SECOND, independent scan agrees the agent is gone.
|
||||
*/
|
||||
@Test
|
||||
void aLeadMissingFromAgentListOnTwoConsecutiveScansIsFinallyDropped() {
|
||||
TopologyHerdr herdr = twoLeads();
|
||||
AtomicLong clock = new AtomicLong();
|
||||
LeadTabScanner s = scanner(herdr, twoLeadsConfigured(), clock);
|
||||
assertTrue(s.get().containsKey("term_opus"));
|
||||
|
||||
herdr.deadAgent("w1:t1");
|
||||
clock.addAndGet(TTL);
|
||||
assertTrue(s.get().containsKey("term_opus"), "first miss is a grace period, not a verdict");
|
||||
|
||||
clock.addAndGet(TTL); // a second, independent scan — still no agent
|
||||
assertFalse(s.get().containsKey("term_opus"),
|
||||
"a second consecutive miss for the same terminal must finally drop it");
|
||||
}
|
||||
|
||||
/** A lead that recovers between the two misses keeps its grace spent, not renewed for free. */
|
||||
@Test
|
||||
void aLeadThatRecoversBetweenMissesIsReportedNormallyAndResetsItsGrace() {
|
||||
TopologyHerdr herdr = twoLeads();
|
||||
AtomicLong clock = new AtomicLong();
|
||||
LeadTabScanner s = scanner(herdr, twoLeadsConfigured(), clock);
|
||||
assertTrue(s.get().containsKey("term_opus"));
|
||||
|
||||
herdr.deadAgent("w1:t1");
|
||||
clock.addAndGet(TTL);
|
||||
assertTrue(s.get().containsKey("term_opus"), "graced on the first miss");
|
||||
|
||||
herdr.reviveAgent("w1:t1"); // the miss really was transient
|
||||
clock.addAndGet(TTL);
|
||||
assertTrue(s.get().containsKey("term_opus"), "found live again — reported normally");
|
||||
|
||||
// A later, unrelated miss must get its own fresh grace scan rather than being dropped
|
||||
// immediately because the earlier miss had already "used up" a slot for this terminal.
|
||||
herdr.deadAgent("w1:t1");
|
||||
clock.addAndGet(TTL);
|
||||
assertTrue(s.get().containsKey("term_opus"),
|
||||
"a miss after a genuine recovery is a new event and deserves its own grace scan");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-579 acceptance (6): this is the bug the ticket closes. A stale pin used to be merged back
|
||||
* over every scan and never expire; now a scan is the whole answer, so a lead whose tab is gone
|
||||
|
||||
@@ -2,6 +2,7 @@ package dev.ltms.fleet.inject;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.fleet.Fleetd;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
@@ -146,40 +147,16 @@ class BackendOutageFlowTest {
|
||||
pushLoop = new ReplyPushLoop(registry, new AgentControl(leadClient), inbox, scheduler, 3, 50);
|
||||
AtomicReference<ReplyPushLoop> pushLoopRef = new AtomicReference<>(pushLoop);
|
||||
|
||||
// --- mirrors Fleetd.main's backendErrorSink lambda EXACTLY: (1) mark BACKEND_ERROR,
|
||||
// (2) resolve profile/credential via the roster, fail-loud + notify unmapped-target,
|
||||
// (3) record in BackendOutagePolicy, (4) on a NEW incident, notify the lead. ------------
|
||||
BackendErrorSink backendErrorSink = (target, matchedLine, reason) -> {
|
||||
sessions.onBackendError(target, reason);
|
||||
// fleetd #248 follow-up: this used to be a 30-line hand-copy of Fleetd.main's
|
||||
// backendErrorSink lambda, with a comment promising it mirrored production "EXACTLY".
|
||||
// That promise is exactly the problem: a copy proves the copy. Editing or deleting the
|
||||
// real sink left this whole flow test green, because it never touched the real sink.
|
||||
// #248 made Fleetd.backendErrorSink public precisely so a cross-package test could
|
||||
// drive the real object, so this now calls it. Every assertion below is about
|
||||
// production code again.
|
||||
BackendErrorSink backendErrorSink =
|
||||
Fleetd.backendErrorSink(sessions, () -> profiles, outagePolicy, pushLoopRef::get);
|
||||
|
||||
String profileName = sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(MemberSession::profile)
|
||||
.orElse(null);
|
||||
FleetConfig.Profile profile = profileName == null ? null : profiles.get(profileName);
|
||||
if (profile == null) {
|
||||
ReplyPushLoop loop = pushLoopRef.get();
|
||||
if (loop != null) {
|
||||
loop.onBackendTargetUnmapped(target, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
String credentialId = profile.effectiveCredentialId();
|
||||
Optional<BackendOutagePolicy.Incident> incident = outagePolicy.record(credentialId, target, reason);
|
||||
incident.ifPresent(inc -> {
|
||||
List<String> affectedProfiles = profiles.values().stream()
|
||||
.filter(p -> credentialId.equals(p.effectiveCredentialId()))
|
||||
.map(FleetConfig.Profile::profile)
|
||||
.sorted()
|
||||
.toList();
|
||||
ReplyPushLoop loop = pushLoopRef.get();
|
||||
if (loop != null) {
|
||||
loop.onBackendIncident(inc.id(), inc.targets(), credentialId, affectedProfiles,
|
||||
(int) inc.remainingCoolOffSeconds());
|
||||
}
|
||||
});
|
||||
};
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(),
|
||||
ExhaustionSink.none(), patterns, backendErrorSink);
|
||||
|
||||
@@ -357,6 +357,54 @@ class CompletionResolverTest {
|
||||
"the failure carries whatever was on screen: " + waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPlausibleLookingReplyInsideTheFloorStillFails() {
|
||||
// fleetd#376 guard. A fix was attempted that inspected the pane inside the floor and resolved
|
||||
// a COMPLETION when the text "looked like a real reply". Every cheap test for that is unsafe:
|
||||
// lastAssistantBlock falls back to the WHOLE pane when there is no ⏺ marker, and a crash pane
|
||||
// almost always contains sentence punctuation — in a file path, a version, or a hostname.
|
||||
// This pane is the trap: it reads like a finished answer and it is a backend failure.
|
||||
FakeHerdr herdr = new FakeHerdr().readText(
|
||||
"Error: connection reset while loading src/main/java/Foo.java v1.2.3\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
long[] clock = {10_000_000_000L};
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), () -> clock[0]);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null, clock[0]);
|
||||
clock[0] += CompletionResolver.MIN_TURN_NANOS - 1; // inside the floor
|
||||
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"inside the floor the verdict is always FAILED — never guess a completion from pane text");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theTooFastFailureDoesNotAssertACauseItCannotKnow() {
|
||||
// fleetd#376: the message used to say "most likely a backend error before any work started".
|
||||
// When no error pattern matches, that cause is a guess, and a reader who believes it stops
|
||||
// looking at the pane. The verdict stays FAILED; only the claim about WHY is withdrawn.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("I am running on opencode/mimo-v2.5-free.\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
long[] clock = {10_000_000_000L};
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), () -> clock[0]);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null, clock[0]);
|
||||
clock[0] += CompletionResolver.MIN_TURN_NANOS - 1; // inside the floor
|
||||
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
String text = waiter.getNow(null).text();
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(), "still fails, still loud");
|
||||
assertFalse(text.contains("most likely a backend error"),
|
||||
"an unmatched fast turn must not assert a backend error: " + text);
|
||||
assertTrue(text.contains("mimo-v2.5-free"), "the pane is still carried: " + text);
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBusyToDoneTransitionJustOutsideTheFloorResolvesNormally() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ a real, if quick, answer\n❯ ");
|
||||
@@ -484,6 +532,27 @@ class CompletionResolverTest {
|
||||
|
||||
// --- CB-578 stage A: backend-exhausted classification ---------------------------------
|
||||
|
||||
@Test
|
||||
void aNormalMemberReportMentioningTheExhaustionPatternDoesNotNotifyTheSink() {
|
||||
String block = "⏺ I reviewed capacity handling. The usage limit has been reached means no more work can start.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, waiter.getNow(null).kind(),
|
||||
"a matching report still fails the send as exhausted");
|
||||
assertTrue(waiter.getNow(null).text().contains("I reviewed capacity handling."),
|
||||
"the exhausted result keeps the whole matched pane line");
|
||||
assertTrue(notified.isEmpty(),
|
||||
"a normal report mentioning an exhaustion pattern must not quarantine a credential");
|
||||
}
|
||||
|
||||
@Test
|
||||
void classifiesAMatchingScrapeAsBackendExhaustedInsteadOfACompletedReply() {
|
||||
String block = "⏺ Working on it...\nThe usage limit has been reached. Try again later.\n❯ ";
|
||||
@@ -534,6 +603,53 @@ class CompletionResolverTest {
|
||||
"the sink is told the matched reason: " + notified.get(0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aRealExhaustionBehindTerminalChromeStillNotifiesTheSink() {
|
||||
String block = "⏺ │ The usage limit has been reached. Try again later.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, waiter.getNow(null).kind(),
|
||||
"a real exhaustion must still fail the send as exhausted");
|
||||
assertEquals(1, notified.size(),
|
||||
"a real exhaustion behind terminal chrome must reach the sink");
|
||||
}
|
||||
|
||||
/**
|
||||
* The live fleet configures {@code exhaustedPattern: "The usage limit has been reached"} — with
|
||||
* the leading {@code "The"}. Every other test here uses a pattern without it, which is the shape
|
||||
* that made fleetd #348 need a looser rule than a start-of-line check. This pins the deployed
|
||||
* shape as well, so a later tightening of {@link CompletionResolver} cannot silently stop
|
||||
* recording the exhaustion this fleet actually reports.
|
||||
*
|
||||
* <p>What it does not prove: that this is the only pattern shape an operator will write.
|
||||
*/
|
||||
@Test
|
||||
void anExhaustionPatternCarryingItsLeadingWordsStillNotifiesTheSink() {
|
||||
String block = "⏺ │ The usage limit has been reached. Try again later.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("The usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, waiter.getNow(null).kind(),
|
||||
"the live pattern shape must still fail the send as exhausted");
|
||||
assertEquals(1, notified.size(),
|
||||
"the live pattern shape must still reach the sink");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLosingBackendExhaustedClassificationNeverNotifiesTheExhaustionSink() {
|
||||
// The waiter was already resolved (e.g. by the worker's own reply) before this scrape landed —
|
||||
@@ -804,7 +920,28 @@ class CompletionResolverTest {
|
||||
// --- fleetd#201 Unit 1: target-keyed backend-error pattern + typed sink ----------------------
|
||||
|
||||
@Test
|
||||
void aConfiguredBackendErrorPatternClassifiesAMatchAsAFailureAndNotifiesTheSinkOnce() {
|
||||
void aNormalMemberReportMentioningTheFallbackErrorPatternFailsButDoesNotNotifyTheSink() {
|
||||
String block = "⏺ I checked the retry path. An API Error: makes it back off.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), BackendErrorPatternLookup.legacy(), sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"a scrape mentioning the pattern must still fail the send");
|
||||
assertTrue(waiter.getNow(null).text().contains("I checked the retry path. An API Error: makes it back off."),
|
||||
"the failure must keep the whole pane tail");
|
||||
assertTrue(notified.isEmpty(),
|
||||
"a normal report mentioning the fallback pattern must not record a credential failure");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aConfiguredBackendErrorPatternAtTheStartOfALineClassifiesAMatchAndNotifiesTheSinkOnce() {
|
||||
String block = "⏺ 503 Service Unavailable: upstream credential rejected\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
@@ -924,7 +1061,7 @@ class CompletionResolverTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void aConfiguredPatternAlsoClassifiesTheRawScrapeFallbackAndNotifiesTheSink() {
|
||||
void aConfiguredPatternAtTheStartOfALineAlsoClassifiesTheRawScrapeFallbackAndNotifiesTheSink() {
|
||||
// No ⏺ marker and leading TUI chrome ⇒ lastAssistantBlock() yields "", so classification must
|
||||
// fall back to the raw scrape (fleetd#211) — and it must use the configured pattern too.
|
||||
String block = """
|
||||
@@ -949,6 +1086,40 @@ class CompletionResolverTest {
|
||||
assertTrue(notified.get(0).contains("503 Service Unavailable"), notified.get(0));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #339 follow-up: a genuine backend error rendered behind terminal chrome must still
|
||||
* record the credential outage. #339 added a start-of-line check to stop a member's own prose
|
||||
* being counted as an outage, and a bare {@code lookingAt} also rejected this — the send failed
|
||||
* but the sink never fired. #339's invariant 3 named that direction as the worse one: a real
|
||||
* outage going unrecorded leaves the fleet spawning into a dead credential.
|
||||
*
|
||||
* <p>The raw-scrape path is where this matters, because its own comment says to expect leading
|
||||
* TUI chrome there.
|
||||
*/
|
||||
@Test
|
||||
void aRealErrorBehindTerminalChromeStillNotifiesTheSink() {
|
||||
String block = """
|
||||
╭──────────────────────────────────────╮
|
||||
│ 503 Service Unavailable: upstream credential rejected
|
||||
""";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"a real backend error must still fail the send");
|
||||
assertEquals(1, notified.size(),
|
||||
"a real error line behind box chrome is still a real outage — it must reach the sink, "
|
||||
+ "or the fleet keeps spawning into a dead credential");
|
||||
}
|
||||
|
||||
// --- fleetd#201 Unit 1: classification inside the fleetd#164 MIN_TURN_NANOS floor -------------
|
||||
|
||||
@Test
|
||||
@@ -994,8 +1165,16 @@ class CompletionResolverTest {
|
||||
assertTrue(waiter.isDone());
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"still a failure — the floor itself, not the pattern, is why");
|
||||
assertTrue(waiter.getNow(null).text().contains("too fast to be real work"),
|
||||
"a non-match inside the floor stays the generic too-fast reason: " + waiter.getNow(null).text());
|
||||
// fleetd#376: this used to assert the phrase "too fast to be real work", which carried the
|
||||
// claim "most likely a backend error before any work started". With no pattern matched that
|
||||
// cause is a guess, so the wording was withdrawn. What this test really guards is unchanged:
|
||||
// the floor alone still fails the turn, it stays generic, and it never notifies the sink.
|
||||
String reason = waiter.getNow(null).text();
|
||||
assertTrue(reason.contains("inside the floor"),
|
||||
"a non-match inside the floor stays the generic floor reason: " + reason);
|
||||
assertFalse(reason.contains("most likely a backend error"),
|
||||
"a non-match must not assert a cause it did not establish: " + reason);
|
||||
assertTrue(reason.contains("still starting up"), "the pane is still carried: " + reason);
|
||||
assertTrue(notified.isEmpty(), "a non-match must never notify the typed sink");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,7 +48,7 @@ class InjectorTest {
|
||||
|
||||
@Test
|
||||
void deliversWhenIdle() {
|
||||
CompletableFuture<Void> f = injector.enqueue(T, "hello", TestTurnTokens.inert(T));
|
||||
CompletableFuture<Void> f = injector.enqueue(T, "hello", TestTurnTokens.inert(T)).completion();
|
||||
assertFalse(f.isDone(), "not delivered until an injectable status arrives");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
assertTrue(f.isDone());
|
||||
@@ -170,6 +170,32 @@ class InjectorTest {
|
||||
assertEquals(List.of("a", "b", "c"), sent());
|
||||
}
|
||||
|
||||
@Test
|
||||
void cancellingTheMiddleDeliveryKeepsTheFollowingDeliveryReachable() {
|
||||
Injector.Delivery first = injector.enqueue(T, "same text", TestTurnTokens.inert(T));
|
||||
Injector.Delivery cancelled = injector.enqueue(T, "same text", TestTurnTokens.inert(T));
|
||||
injector.enqueue(T, "after cancelled", TestTurnTokens.inert(T));
|
||||
|
||||
assertEquals(Injector.Cancellation.CANCELLED, injector.cancel(cancelled));
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
|
||||
assertEquals(List.of("same text", "after cancelled"), sent(),
|
||||
"cancellation must match the exact Delivery and preserve the remaining FIFO queue");
|
||||
assertTrue(first.completion().isDone());
|
||||
}
|
||||
|
||||
@Test
|
||||
void cancellationReportsDeliveredWhenPickupWonTheRace() {
|
||||
Injector.Delivery delivery = injector.enqueue(T, "already sent", TestTurnTokens.inert(T));
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
|
||||
assertEquals(Injector.Cancellation.DELIVERED, injector.cancel(delivery),
|
||||
"a cancellation after pickup must not claim that the text stayed queued");
|
||||
assertEquals(List.of("already sent"), sent());
|
||||
}
|
||||
|
||||
@Test
|
||||
void activeWhileQueuedOrInFlightThenQuietAfterTurnCompletes() {
|
||||
assertTrue(injector.activeTargets().isEmpty());
|
||||
@@ -297,6 +323,68 @@ class InjectorTest {
|
||||
assertTrue(inj.activeTargets().isEmpty(), "the wedged target is reclaimed, not polled forever");
|
||||
}
|
||||
|
||||
/** A listener whose post-turn housekeeping always starts, as SessionManager's does with clearAfterTurn on. */
|
||||
private static final class PostTurnListener implements TurnListener {
|
||||
@Override public void onTurnComplete(String target) { }
|
||||
@Override public boolean hasPostTurnAction(String target) { return true; }
|
||||
@Override public boolean onTurnCompleteWithPostAction(String target) { return true; }
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerThatWedgesInUnknownAwaitingPostTurnPickupIsReleased() {
|
||||
// fleetd #306: the post-turn phase had no way out of a sustained unknown streak, so the
|
||||
// pickup latch stayed set, the target was polled forever, and every later message to it was
|
||||
// blocked by the delivery gate — while the session still looked healthy.
|
||||
Captor cap = new Captor();
|
||||
Injector inj = new Injector(new AgentControl(herdr), new PostTurnListener());
|
||||
inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||
|
||||
inj.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
inj.onStatus(T, AgentStatus.WORKING); // turn starts
|
||||
inj.onStatus(T, AgentStatus.IDLE); // turn completes; housekeeping dispatched
|
||||
for (int i = 0; i < STALL_SAMPLES; i++) inj.onStatus(T, AgentStatus.UNKNOWN); // then wedges
|
||||
|
||||
assertTrue(inj.activeTargets().isEmpty(),
|
||||
"a target wedged awaiting post-turn pickup must be reclaimed, not polled forever");
|
||||
assertEquals(List.of(), cap.failed,
|
||||
"the delegated turn already completed — a stuck /clear must not be reported as a failed turn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerThatWedgesInUnknownAfterPickingUpTheResetIsReleased() {
|
||||
// The sibling latch. postTurnObserved is set when the reset is seen picked up (WORKING) and
|
||||
// is cleared only on a later injectable sample, so a wedge right after pickup sticks too.
|
||||
Captor cap = new Captor();
|
||||
Injector inj = new Injector(new AgentControl(herdr), new PostTurnListener());
|
||||
inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||
|
||||
inj.onStatus(T, AgentStatus.IDLE);
|
||||
inj.onStatus(T, AgentStatus.WORKING);
|
||||
inj.onStatus(T, AgentStatus.IDLE); // turn complete; reset dispatched
|
||||
inj.onStatus(T, AgentStatus.WORKING); // reset picked up -> postTurnObserved
|
||||
for (int i = 0; i < STALL_SAMPLES; i++) inj.onStatus(T, AgentStatus.UNKNOWN);
|
||||
|
||||
assertTrue(inj.activeTargets().isEmpty(), "a wedge after reset pickup must also be reclaimed");
|
||||
assertEquals(List.of(), cap.failed, "still not a turn failure");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBriefUnknownDuringPostTurnHousekeepingDoesNotDropTheLatch() {
|
||||
// The other direction: the escape must not fire on a glitch, or the queued next delegation
|
||||
// would overtake housekeeping that is still running.
|
||||
Injector inj = new Injector(new AgentControl(herdr), new PostTurnListener());
|
||||
inj.enqueue(T, "first", TestTurnTokens.inert(T));
|
||||
inj.enqueue(T, "second", TestTurnTokens.inert(T));
|
||||
|
||||
inj.onStatus(T, AgentStatus.IDLE);
|
||||
inj.onStatus(T, AgentStatus.WORKING);
|
||||
inj.onStatus(T, AgentStatus.IDLE); // first completes; reset dispatched
|
||||
for (int i = 0; i < 10; i++) inj.onStatus(T, AgentStatus.UNKNOWN); // well under the grace
|
||||
|
||||
assertFalse(inj.activeTargets().isEmpty(), "a brief glitch must not release the post-turn latch");
|
||||
assertEquals(List.of("first"), sent(), "the queued delegation must not overtake housekeeping");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aTransientUnknownGlitchNeitherFailsNorBlocksCompletion() {
|
||||
Captor cap = new Captor();
|
||||
@@ -316,7 +404,7 @@ class InjectorTest {
|
||||
void sendFailureDropsMessageAndFailsItsFuture() {
|
||||
FakeHerdr failing = new FakeHerdr().agentSendFailsWith("send_failed");
|
||||
Injector inj = new Injector(new AgentControl(failing));
|
||||
CompletableFuture<Void> f = inj.enqueue(T, "boom", TestTurnTokens.inert(T));
|
||||
CompletableFuture<Void> f = inj.enqueue(T, "boom", TestTurnTokens.inert(T)).completion();
|
||||
|
||||
inj.onStatus(T, AgentStatus.IDLE);
|
||||
assertTrue(f.isCompletedExceptionally());
|
||||
@@ -325,7 +413,7 @@ class InjectorTest {
|
||||
|
||||
@Test
|
||||
void dropFailsPendingWaiters() {
|
||||
CompletableFuture<Void> f = injector.enqueue(T, "orphan", TestTurnTokens.inert(T));
|
||||
CompletableFuture<Void> f = injector.enqueue(T, "orphan", TestTurnTokens.inert(T)).completion();
|
||||
injector.drop(T, new HerdrException("worker gone", "pane_not_found", null));
|
||||
assertTrue(f.isCompletedExceptionally(), "queued waiters unblock when the worker vanishes");
|
||||
}
|
||||
@@ -334,8 +422,8 @@ class InjectorTest {
|
||||
void dropPassesTheRealCauseForQueuedAndDeliveredWork() {
|
||||
Captor cap = new Captor();
|
||||
Injector inj = new Injector(new AgentControl(herdr), cap);
|
||||
CompletableFuture<Void> delivered = inj.enqueue(T, "delivered", TestTurnTokens.inert(T));
|
||||
CompletableFuture<Void> queued = inj.enqueue(T, "queued", TestTurnTokens.inert(T));
|
||||
CompletableFuture<Void> delivered = inj.enqueue(T, "delivered", TestTurnTokens.inert(T)).completion();
|
||||
CompletableFuture<Void> queued = inj.enqueue(T, "queued", TestTurnTokens.inert(T)).completion();
|
||||
|
||||
inj.onStatus(T, AgentStatus.IDLE); // deliver the first message
|
||||
inj.onStatus(T, AgentStatus.WORKING); // its turn is now in flight; one remains queued
|
||||
@@ -387,7 +475,7 @@ class InjectorTest {
|
||||
Captor cap = new Captor();
|
||||
List<String> forgotten = new ArrayList<>();
|
||||
Injector inj = new Injector(new AgentControl(herdr), cap, _ -> false, forgotten::add);
|
||||
CompletableFuture<Void> f = inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||
CompletableFuture<Void> f = inj.enqueue(T, "task", TestTurnTokens.inert(T)).completion();
|
||||
|
||||
for (int i = 0; i < READINESS_SAMPLES; i++) inj.onStatus(T, AgentStatus.IDLE);
|
||||
|
||||
@@ -501,7 +589,7 @@ class InjectorTest {
|
||||
StatusPoller poller = new StatusPoller(new AgentControl(idle), inj, 10);
|
||||
poller.start();
|
||||
try {
|
||||
CompletableFuture<Void> delivered = inj.enqueue(T, "via-poller", TestTurnTokens.inert(T));
|
||||
CompletableFuture<Void> delivered = inj.enqueue(T, "via-poller", TestTurnTokens.inert(T)).completion();
|
||||
delivered.get(2, TimeUnit.SECONDS); // completes when the poller drives the send
|
||||
} finally {
|
||||
poller.stop();
|
||||
@@ -520,7 +608,7 @@ class InjectorTest {
|
||||
void deliveredFutureCarriesSendFailure() {
|
||||
FakeHerdr failing = new FakeHerdr().agentSendFailsWith("send_failed");
|
||||
Injector inj = new Injector(new AgentControl(failing));
|
||||
CompletableFuture<Void> f = inj.enqueue(T, "boom", TestTurnTokens.inert(T));
|
||||
CompletableFuture<Void> f = inj.enqueue(T, "boom", TestTurnTokens.inert(T)).completion();
|
||||
inj.onStatus(T, AgentStatus.IDLE);
|
||||
ExecutionException ex = assertThrows(ExecutionException.class, f::get);
|
||||
assertInstanceOf(HerdrException.class, ex.getCause());
|
||||
|
||||
@@ -41,7 +41,7 @@ class StatusPollerRoutingTest {
|
||||
poller.start();
|
||||
try {
|
||||
CompletableFuture<Void> delivered =
|
||||
injector.enqueue(LEAD_TARGET, "via-poller", TestTurnTokens.inert(LEAD_TARGET));
|
||||
injector.enqueue(LEAD_TARGET, "via-poller", TestTurnTokens.inert(LEAD_TARGET)).completion();
|
||||
// Must resolve quickly: refining against the WRONG daemon (member) never classifies
|
||||
// out of UNKNOWN, so this would time out under the bug.
|
||||
delivered.get(2, TimeUnit.SECONDS);
|
||||
@@ -65,7 +65,7 @@ class StatusPollerRoutingTest {
|
||||
poller.start();
|
||||
try {
|
||||
CompletableFuture<Void> delivered =
|
||||
injector.enqueue(LEAD_TARGET, "via-poller", TestTurnTokens.inert(LEAD_TARGET));
|
||||
injector.enqueue(LEAD_TARGET, "via-poller", TestTurnTokens.inert(LEAD_TARGET)).completion();
|
||||
assertThrows(TimeoutException.class, () -> delivered.get(500, TimeUnit.MILLISECONDS),
|
||||
"a lead target must never be refined from the member daemon's pane content");
|
||||
} finally {
|
||||
|
||||
@@ -9,6 +9,7 @@ import org.junit.jupiter.api.Test;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
@@ -106,6 +107,7 @@ class LeadLauncherTest {
|
||||
|
||||
assertEquals(0, launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads());
|
||||
assertFalse(herdr.called("agent.start"), "the live lead must not be duplicated");
|
||||
assertFalse(herdr.called("tab.close"), "a labelled tab WITH a live agent must never be closed");
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -122,6 +124,121 @@ class LeadLauncherTest {
|
||||
"a stale label is not a lead; the lead must be relaunched");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #359 review finding 1 — the exact scenario the ticket's own live evidence produced: on a
|
||||
* real host, {@code agent.list} reported "0 live" for a tab a plain {@code ps} confirmed was
|
||||
* running a real session. A single such reading must never close the tab outright — that would
|
||||
* destroy the operator's actual lead, a worse failure than the stale-tab bug this ticket exists
|
||||
* to fix. The first dead reading only flags the tab; mutate this away (make the first reading
|
||||
* close instead of flag) and this test must fail.
|
||||
*/
|
||||
@Test
|
||||
void aStaleLabelledTabIsFlaggedRatherThanClosedOnTheFirstReconcile() {
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.withWorkspace("wL", "fleet")
|
||||
.withTab("wL", "wL:t1", "lead: opus"); // label only, first look — could be a live session agent.list missed
|
||||
|
||||
assertEquals(1, launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads());
|
||||
|
||||
assertFalse(herdr.called("tab.close"),
|
||||
"one missed reading must never close a tab that might be hosting a live session");
|
||||
assertTrue(flaggedTabIds(herdr).contains("wL:t1"),
|
||||
"the stale tab must be flagged pending-close so a later reconcile can confirm it");
|
||||
}
|
||||
|
||||
/**
|
||||
* The operator's own trace on fleet01 (#359): a daemon that restarted several times, each
|
||||
* occasion finding "0 live" for whatever reason, had left several identically-labelled dead
|
||||
* tabs sitting side by side. Every one of them is flagged on its first dead reading, not closed —
|
||||
* none is more or less trustworthy than another.
|
||||
*/
|
||||
@Test
|
||||
void allStaleLabelledTabsAreFlaggedRatherThanClosedOnTheFirstReconcile() {
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.withWorkspace("wL", "fleet")
|
||||
.withTab("wL", "wL:t1", "lead: opus")
|
||||
.withTab("wL", "wL:t2", "lead: opus")
|
||||
.withTab("wL", "wL:t3", "lead: opus"); // three restarts' worth of debris, none confirmed twice yet
|
||||
|
||||
assertEquals(1, launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads());
|
||||
|
||||
assertFalse(herdr.called("tab.close"), "no tab may be closed on its first dead reading");
|
||||
assertEquals(Set.of("wL:t1", "wL:t2", "wL:t3"), flaggedTabIds(herdr),
|
||||
"every dead labelled tab must be flagged, not just the first one found");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #359 review finding 1, the other half — a tab already flagged pending-close by an
|
||||
* earlier reconcile, and STILL dead on this one, has now been read dead on two independent,
|
||||
* separately-connected reconciles. That is strong enough evidence to actually close it. Mutate
|
||||
* this away (never close a flagged tab) and this test must fail — the original #359 growth bug
|
||||
* would come back for good.
|
||||
*/
|
||||
@Test
|
||||
void aTabAlreadyFlaggedPendingCloseIsClosedWhenStillDeadOnALaterReconcile() {
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.withWorkspace("wL", "fleet")
|
||||
.withTab("wL", "wL:t1", "lead: opus [fleetd:pending-close]"); // flagged last reconcile, still dead
|
||||
|
||||
assertEquals(1, launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads());
|
||||
|
||||
assertTrue(herdr.called("tab.close"),
|
||||
"a tab dead on two independent reconciles must finally be closed");
|
||||
assertEquals("wL:t1", ((Map<?, ?>) herdr.lastCall("tab.close").params()).get("tab_id"));
|
||||
}
|
||||
|
||||
/** All of several already-flagged, still-dead tabs are closed — not just the first found. */
|
||||
@Test
|
||||
void allTabsAlreadyFlaggedPendingCloseAreClosedWhenStillDead() {
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.withWorkspace("wL", "fleet")
|
||||
.withTab("wL", "wL:t1", "lead: opus [fleetd:pending-close]")
|
||||
.withTab("wL", "wL:t2", "lead: opus [fleetd:pending-close]")
|
||||
.withTab("wL", "wL:t3", "lead: opus [fleetd:pending-close]");
|
||||
|
||||
assertEquals(1, launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads());
|
||||
|
||||
List<Object> closedTabIds = herdr.calls.stream()
|
||||
.filter(c -> c.method().equals("tab.close"))
|
||||
.<Object>map(c -> ((Map<?, ?>) c.params()).get("tab_id"))
|
||||
.toList();
|
||||
assertEquals(3, closedTabIds.size(),
|
||||
"every confirmed-dead labelled tab must be closed, not just the first one found");
|
||||
assertEquals(Set.of("wL:t1", "wL:t2", "wL:t3"), Set.copyOf(closedTabIds));
|
||||
}
|
||||
|
||||
/**
|
||||
* A tab flagged pending-close on a previous reconcile that is running an agent again — the miss
|
||||
* that flagged it was transient. It must never be closed, and its flag must be cleared so a
|
||||
* future, unrelated miss starts its own two-reading count from zero.
|
||||
*/
|
||||
@Test
|
||||
void aFlaggedTabRunningAnAgentAgainHasItsFlagClearedInsteadOfBeingClosed() {
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.withWorkspace("wL", "fleet")
|
||||
.withTab("wL", "wL:t1", "lead: opus [fleetd:pending-close]")
|
||||
.withAgent("lead-opus", "term_lead", "wL:p1", "wL:t1"); // it recovered — really alive now
|
||||
|
||||
assertEquals(0, launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads(),
|
||||
"the lead is live again — nothing to relaunch");
|
||||
|
||||
assertFalse(herdr.called("tab.close"), "a tab running an agent again must never be closed");
|
||||
assertFalse(herdr.called("agent.start"), "the lead is live again — nothing to relaunch");
|
||||
assertEquals("wL:t1", ((Map<?, ?>) herdr.lastCall("tab.rename").params()).get("tab_id"));
|
||||
assertEquals("lead: opus", ((Map<?, ?>) herdr.lastCall("tab.rename").params()).get("label"),
|
||||
"the pending-close flag must be cleared once the tab is confirmed live again");
|
||||
}
|
||||
|
||||
/** Every {@code tab.rename} call whose label carries the pending-close marker, by tab id. */
|
||||
private static Set<String> flaggedTabIds(FakeHerdr herdr) {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> c.method().equals("tab.rename"))
|
||||
.filter(c -> String.valueOf(((Map<?, ?>) c.params()).get("label"))
|
||||
.endsWith("[fleetd:pending-close]"))
|
||||
.map(c -> String.valueOf(((Map<?, ?>) c.params()).get("tab_id")))
|
||||
.collect(java.util.stream.Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* A lead the operator opened by hand is live once its tab carries the configured `tab:` label —
|
||||
* CB-579 retired the `terminal:` pin, so a hand-opened lead is found the same way an
|
||||
|
||||
@@ -20,6 +20,17 @@ class ConnectionIdentityTest {
|
||||
assertEquals("term_a", with(_ -> FakeHerdr.WORKER_PID).callerTerminal("127.0.0.1", 55555));
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesWorkerFromAnyLoopbackSourceAddressNotJust127001() {
|
||||
// fleetd #305. On Linux the whole 127.0.0.0/8 is bound to lo, so a worker can connect with
|
||||
// a source address of 127.0.0.2. If identity resolution skips that address the caller has
|
||||
// no terminal, and a caller with no terminal is the primary under loopback-trust — so this
|
||||
// must resolve the worker, not null.
|
||||
assertEquals("term_a", with(_ -> FakeHerdr.WORKER_PID).callerTerminal("127.0.0.2", 55555));
|
||||
assertEquals("term_a", with(_ -> FakeHerdr.WORKER_PID).callerTerminal("127.1.2.3", 55555));
|
||||
assertEquals("term_a", with(_ -> FakeHerdr.WORKER_PID).callerTerminal("::ffff:127.0.0.2", 55555));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForOffHostCaller() {
|
||||
// A non-loopback peer can't be an on-host worker → treat as primary/unknown.
|
||||
@@ -32,6 +43,22 @@ class ConnectionIdentityTest {
|
||||
assertNull(with(_ -> 999_999).callerTerminal("127.0.0.1", 55555));
|
||||
}
|
||||
|
||||
@Test
|
||||
void callerIsUnresolvedWhenThePeerPidLookupFails() {
|
||||
// fleetd #317: LsofPeerPidLookup returns -1 on any failure — a fork error, or (silently)
|
||||
// simply no matching lsof line. Caller.resolved() is the one place that sentinel is tested.
|
||||
ConnectionIdentity.Caller c = with(_ -> -1).resolve("127.0.0.1", 55555);
|
||||
assertFalse(c.resolved(), "a -1 pid means the lookup failed, not that this pid owns no pane");
|
||||
}
|
||||
|
||||
@Test
|
||||
void callerIsResolvedWhenThePidIsRealEvenThoughItOwnsNoPane() {
|
||||
// The primary's own connection: a real, lsof-found pid that just isn't a worker pane. This
|
||||
// must read as "resolved" — the distinction #317 turns on.
|
||||
ConnectionIdentity.Caller c = with(_ -> 999_999).resolve("127.0.0.1", 55555);
|
||||
assertTrue(c.resolved());
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesTheCallersPidAndCwd() {
|
||||
// CB-112: the primary maps to no pane, but its PID and cwd are still readable.
|
||||
|
||||
@@ -24,8 +24,13 @@ import io.modelcontextprotocol.spec.McpSchema;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
@@ -44,6 +49,9 @@ import static org.junit.jupiter.api.Assertions.*;
|
||||
*/
|
||||
class FleetMcpAuthzTest {
|
||||
|
||||
private static final Path MCP_SOURCE = Path.of("src/main/java/dev/ltms/fleet/mcp/FleetMcp.java");
|
||||
private static final Pattern TOOL_REGISTRATION = Pattern.compile("tool\\(\\\"(fleet_[a-z_]+)\\\"");
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
private final AgentControl agents = new AgentControl(herdr);
|
||||
private Metrics metrics;
|
||||
@@ -179,6 +187,89 @@ class FleetMcpAuthzTest {
|
||||
"no CallerResolver supplied ⇒ authorization not enforced (legacy behaviour)");
|
||||
}
|
||||
|
||||
// --- which action each tool hands the gate (fleetd #272) ------------------------------------
|
||||
|
||||
/**
|
||||
* fleetd #272: {@code fleet_poll{target}} drains a session's reply inbox, so it needs
|
||||
* {@link Authz.Action#DRAIN} -- not the {@link Authz.Action#READ} the handler passed for both
|
||||
* of its branches until this ticket.
|
||||
*
|
||||
* <p>This asserts against {@link FleetMcp#pollAction}, the method the handler itself calls, so
|
||||
* the handler holds no separate copy of the rule that this test could miss. Every other test in
|
||||
* this class checks the policy table (is a worker allowed to DRAIN?) and all of them passed for
|
||||
* the whole time the defect was live -- the table was right, the action fed to it was wrong.
|
||||
*/
|
||||
@Test
|
||||
void pollingByTargetIsADrainAndPollingByTicketIsARead() {
|
||||
assertEquals(Authz.Action.DRAIN, FleetMcp.pollAction("term_b"),
|
||||
"poll by target removes the replies — that is a drain, not an observation");
|
||||
assertEquals(Authz.Action.READ, FleetMcp.pollAction(null),
|
||||
"poll by ticket changes nothing");
|
||||
assertEquals(Authz.Action.READ, FleetMcp.pollAction(" "),
|
||||
"a blank target is an absent target");
|
||||
}
|
||||
|
||||
@Test
|
||||
void everyRegisteredToolHasItsHandlerActionPinned() {
|
||||
Set<String> registered = toolsTheServerRegisters();
|
||||
assertTrue(registered.size() >= 10,
|
||||
"scraped only " + registered.size() + " tool registrations from FleetMcp (" + registered
|
||||
+ "); the server registers eleven, so the tool(\"…\") scrape has stopped matching");
|
||||
registered.forEach(tool -> assertDoesNotThrow(() -> FleetMcp.toolAction(tool, Map.of()),
|
||||
() -> tool + " is registered but has no pinned authorization action"));
|
||||
|
||||
assertEquals(Authz.Action.SEND, FleetMcp.toolAction("fleet_send", Map.of()));
|
||||
assertEquals(Authz.Action.REPLY, FleetMcp.toolAction("fleet_reply", Map.of()));
|
||||
assertEquals(Authz.Action.ASK, FleetMcp.toolAction("fleet_ask", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_status", Map.of()));
|
||||
assertEquals(Authz.Action.DRAIN, FleetMcp.toolAction("fleet_ack", Map.of()));
|
||||
assertEquals(Authz.Action.SPAWN, FleetMcp.toolAction("fleet_spawn", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_list", Map.of()));
|
||||
assertEquals(Authz.Action.STOP, FleetMcp.toolAction("fleet_stop", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_profiles", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_whoami", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_poll", Map.of("ticket", "task")));
|
||||
assertEquals(Authz.Action.DRAIN, FleetMcp.toolAction("fleet_poll", Map.of("target", "term_b")));
|
||||
}
|
||||
|
||||
private static Set<String> toolsTheServerRegisters() {
|
||||
try {
|
||||
Matcher matcher = TOOL_REGISTRATION.matcher(Files.readString(MCP_SOURCE));
|
||||
Set<String> tools = new LinkedHashSet<>();
|
||||
while (matcher.find()) {
|
||||
tools.add(matcher.group(1));
|
||||
}
|
||||
return tools;
|
||||
} catch (Exception e) {
|
||||
throw new AssertionError("could not scrape FleetMcp tool registrations", e);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerMayNotDrainAnotherSessionsInboxByPolling() {
|
||||
FleetMcp m = mcp(true);
|
||||
|
||||
assertNotNull(m.denyFor(WORKER_A, FleetMcp.pollAction("term_b"), "term_b"),
|
||||
"a worker draining a peer's inbox would destroy replies queued for the primary");
|
||||
assertNotNull(m.denyFor(ARCH_DESIGN, FleetMcp.pollAction("term_b"), "term_b"),
|
||||
"an architect has no lifecycle rights either — same gate as fleet_ack");
|
||||
assertNull(m.denyFor(PRIMARY, FleetMcp.pollAction("term_b"), "term_b"),
|
||||
"collecting a held reply is the primary's job");
|
||||
}
|
||||
|
||||
/**
|
||||
* The tightening must not close the branch that legitimately serves non-primary callers: an
|
||||
* architect may {@code fleet_send}, so it owns tickets and must be able to poll them.
|
||||
*/
|
||||
@Test
|
||||
void pollingAnOwnTicketStaysOpenToWorkersAndArchitects() {
|
||||
FleetMcp m = mcp(true);
|
||||
|
||||
assertNull(m.denyFor(WORKER_A, FleetMcp.pollAction(null), null));
|
||||
assertNull(m.denyFor(ARCH_DESIGN, FleetMcp.pollAction(null), null),
|
||||
"an architect delegates with wait:false, so it must be able to poll its ticket");
|
||||
}
|
||||
|
||||
// --- identity reconstruction from the transport context ------------------------------------
|
||||
|
||||
@Test
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import dev.ltms.fleet.msg.FakeLeadChannel;
|
||||
import dev.ltms.fleet.msg.LeadChannel;
|
||||
import dev.ltms.fleet.msg.LeadMessage;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import org.junit.jupiter.api.Test;
|
||||
@@ -93,4 +94,56 @@ class FleetMcpLeadCoordTest {
|
||||
assertTrue(FleetMcp.sendToLead(channel, PEER, " ", null, null).isError());
|
||||
assertEquals(0, channel.published().size());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361: "delivered" overstated what publish actually proves — the broker's confirm means
|
||||
* durably queued, not read. A zero-consumer target is the observable form of "this will not
|
||||
* reach a pane right now", so the (still-successful) result must say so.
|
||||
*/
|
||||
@Test
|
||||
void warnsWhenThePeerMailboxHasNoConsumersButStillReportsSuccess() {
|
||||
var channel = new FakeLeadChannel(SELF)
|
||||
.withMailbox(PEER, LeadChannel.MailboxState.exists(PEER, 0, 0));
|
||||
|
||||
McpSchema.CallToolResult res = FleetMcp.sendToLead(channel, PEER, "hi", null, null);
|
||||
|
||||
assertFalse(res.isError(), "a zero-consumer mailbox is still a successful, durably-queued publish");
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("durably confirmed"), out);
|
||||
assertTrue(out.toLowerCase().contains("no consumers"), () -> "must warn nobody is reading it: " + out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void staysQuietAboutConsumersWhenThePeerMailboxHasOne() {
|
||||
var channel = new FakeLeadChannel(SELF)
|
||||
.withMailbox(PEER, LeadChannel.MailboxState.exists(PEER, 0, 1));
|
||||
|
||||
McpSchema.CallToolResult res = FleetMcp.sendToLead(channel, PEER, "hi", null, null);
|
||||
|
||||
assertFalse(res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("durably confirmed"), out);
|
||||
assertFalse(out.toLowerCase().contains("no consumers"), () -> "a consumer IS attached: " + out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361 review finding 1: an unmeasured fact must never render as a definite one. When
|
||||
* the post-publish probe could not determine the mailbox's consumer count at all (broker slow,
|
||||
* unreachable, or the probe timed out — {@link LeadChannel.MailboxState#unknown}), the result
|
||||
* must stay just as quiet as the has-a-consumer case — never assert "no consumers" for a mailbox
|
||||
* this call never actually measured.
|
||||
*/
|
||||
@Test
|
||||
void staysQuietAboutConsumersWhenThePeerMailboxStateIsUnknown() {
|
||||
var channel = new FakeLeadChannel(SELF)
|
||||
.withMailbox(PEER, LeadChannel.MailboxState.unknown(PEER));
|
||||
|
||||
McpSchema.CallToolResult res = FleetMcp.sendToLead(channel, PEER, "hi", null, null);
|
||||
|
||||
assertFalse(res.isError(), "an unresolved post-publish probe must never turn a durably-confirmed publish into an error");
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("durably confirmed"), out);
|
||||
assertFalse(out.toLowerCase().contains("no consumers"),
|
||||
() -> "an unmeasured fact must never be reported as a definite zero-consumer mailbox: " + out);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,6 +10,9 @@ import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.msg.FakeLeadChannel;
|
||||
import dev.ltms.fleet.msg.LeadChannel;
|
||||
import dev.ltms.fleet.msg.LeadMessage;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.session.FakeWorktrees;
|
||||
@@ -31,9 +34,13 @@ import org.junit.jupiter.api.Test;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.EnumSet;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.function.Function;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
@@ -101,8 +108,10 @@ class FleetMcpTest {
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "send should have opened its waiter");
|
||||
|
||||
// fleetd #365: a resolved live send must read distinctly from a merely-queued reply —
|
||||
// see replyWithNoPendingSendIsQueuedNotError below for the other case.
|
||||
McpSchema.CallToolResult reply = FleetMcp.reply(messages, "term_a", "LGTM");
|
||||
assertEquals("delivered", textOf(reply));
|
||||
assertEquals(MessageService.ReplyOutcome.RESOLVED_SEND.description(), textOf(reply));
|
||||
|
||||
McpSchema.CallToolResult res = send.get(6, TimeUnit.SECONDS);
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
@@ -128,7 +137,7 @@ class FleetMcpTest {
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "send should have opened its waiter");
|
||||
|
||||
McpSchema.CallToolResult reply = FleetMcp.reply(messages, "term_a", "async LGTM");
|
||||
assertEquals("delivered", textOf(reply));
|
||||
assertEquals(MessageService.ReplyOutcome.RESOLVED_SEND.description(), textOf(reply));
|
||||
|
||||
// Poll until the async send completes and reports the reply.
|
||||
McpSchema.CallToolResult polled = FleetMcp.poll(messages, ticket, null);
|
||||
@@ -187,7 +196,7 @@ class FleetMcpTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void unansweredAsyncAskReturnsTheTicketToPending() throws Exception {
|
||||
void unansweredAsyncAskReturnsTheTicketToPendingThenAWorkersLateReplyStillCompletesIt() throws Exception {
|
||||
McpSchema.CallToolResult accepted = FleetMcp.sendAsync(messages, "term_a", "do it", null, Set.of());
|
||||
String ticket = textOf(accepted).substring(textOf(accepted).indexOf("ticket=") + "ticket=".length()).trim();
|
||||
|
||||
@@ -201,8 +210,15 @@ class FleetMcpTest {
|
||||
assertTrue(textOf(ask).contains("no answer"), textOf(ask));
|
||||
assertTrue(textOf(FleetMcp.poll(messages, ticket, null)).startsWith("[pending"));
|
||||
|
||||
// fleetd #307: the worker resumed on its own after the primary never answered, and its real
|
||||
// fleet_reply must complete its OWN async ticket — not strand in the inbox with
|
||||
// fleet_poll{ticket} stuck PENDING forever and later force-failed with a false "session
|
||||
// released before it replied" reason. This used to land in the inbox instead (see the old
|
||||
// assertion this replaced: messages.drainReplies("term_a").getFirst()...) — that was the bug.
|
||||
FleetMcp.reply(messages, "term_a", "finished after timeout");
|
||||
assertEquals("finished after timeout", messages.drainReplies("term_a").getFirst().content());
|
||||
assertEquals("finished after timeout", textOf(FleetMcp.poll(messages, ticket, null)));
|
||||
assertTrue(messages.drainReplies("term_a").isEmpty(),
|
||||
"the reply completed its own ticket directly and never touched the inbox");
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -314,9 +330,10 @@ class FleetMcpTest {
|
||||
@Test
|
||||
void replyWithNoPendingSendIsQueuedNotError() {
|
||||
// CB-307: a reply with no open send is now queued in the inbox, not an error.
|
||||
// fleetd #365: it must also no longer claim "delivered" — nothing was waiting for it.
|
||||
McpSchema.CallToolResult res = FleetMcp.reply(messages, "term_a", "orphan");
|
||||
assertNotEquals(Boolean.TRUE, res.isError(), "a queued reply is not an error");
|
||||
assertEquals("delivered", textOf(res));
|
||||
assertEquals(MessageService.ReplyOutcome.QUEUED.description(), textOf(res));
|
||||
|
||||
// The reply is drainable by target.
|
||||
var drained = messages.drainReplies("term_a");
|
||||
@@ -324,6 +341,26 @@ class FleetMcpTest {
|
||||
assertEquals("orphan", drained.getFirst().content());
|
||||
}
|
||||
|
||||
@Test
|
||||
void replyWithBlankContentIsACleanToolErrorNotAnUncaughtException() {
|
||||
// fleetd #302: MessageService.reply now REJECTS blank content by throwing. fleet_reply's
|
||||
// handler is a bare BiFunction with no try/catch around it, so if this guard only checked
|
||||
// `== null` (as it did), a whitespace-only reply would leave the handler as an uncaught
|
||||
// IllegalArgumentException instead of a tool error the caller can read. Null and whitespace
|
||||
// are the same caller mistake and must get the same answer — the sibling fleet_send guard
|
||||
// has always used isBlank for exactly this reason.
|
||||
for (String blank : new String[] {null, "", " ", "\n\t"}) {
|
||||
McpSchema.CallToolResult res = assertDoesNotThrow(
|
||||
() -> FleetMcp.reply(messages, "term_a", blank),
|
||||
"blank content must be refused as a tool error, never thrown out of the handler");
|
||||
assertEquals(Boolean.TRUE, res.isError(), "blank content is an error result");
|
||||
assertTrue(textOf(res).contains("content is required"),
|
||||
"the error names the missing argument: " + textOf(res));
|
||||
}
|
||||
assertEquals(0, messages.drainReplies("term_a").size(),
|
||||
"a refused reply must not reach the inbox");
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgePollWithTargetDrainsReplies() {
|
||||
// A reply with no open send queues it in the inbox.
|
||||
@@ -379,7 +416,7 @@ class FleetMcpTest {
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the answer should have reopened a waiter");
|
||||
McpSchema.CallToolResult reply = FleetMcp.reply(messages, "term_a", "done");
|
||||
assertEquals("delivered", textOf(reply));
|
||||
assertEquals(MessageService.ReplyOutcome.RESOLVED_SEND.description(), textOf(reply));
|
||||
assertEquals("done", textOf(answer.get(6, TimeUnit.SECONDS)));
|
||||
}
|
||||
|
||||
@@ -547,17 +584,48 @@ class FleetMcpTest {
|
||||
void listReportsThisDaemonsOwnCoordIdWhenLeadCoordinationIsOn() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
FakeLeadChannel channel = new FakeLeadChannel("mac-opus")
|
||||
.withMailbox("mac-opus", LeadChannel.MailboxState.exists("mac-opus", 0, 1));
|
||||
|
||||
McpSchema.CallToolResult res = FleetMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, null,
|
||||
FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), Map.of(), "", "mac-opus");
|
||||
FleetMcp.QuarantineSource.none(), Map.of(), "",
|
||||
new FleetMcp.CoordinationSource(channel, List.of()));
|
||||
|
||||
String out = textOf(res);
|
||||
// There is no peer-discovery surface yet, so this row answers the one question an operator
|
||||
// cannot answer any other way: which coord-id a peer must use to reach ME.
|
||||
// fleetd #361: reports both which coord-id a peer must use to reach ME, and this daemon's
|
||||
// own mailbox state (a self-diagnosis: is my own consumer actually attached?).
|
||||
assertTrue(out.contains("\"coordinator\""), out);
|
||||
assertTrue(out.contains("\"selfId\":\"mac-opus\""), out);
|
||||
assertTrue(out.contains("\"mailbox\":{\"status\":\"exists\",\"pending\":0,\"consumers\":1}"), out);
|
||||
assertTrue(out.contains("\"held\":[]"), out);
|
||||
assertTrue(out.contains("\"peers\":[]"), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361 review finding 1: a self-probe that could not complete (broker unreachable, timed
|
||||
* out) must never render the same as a measured "0 pending, 0 consumers" — that was exactly the
|
||||
* bug: a reader could not tell "my mailbox is empty and idle" from "I could not check", and the
|
||||
* second one is the far more alarming state.
|
||||
*/
|
||||
@Test
|
||||
void listReportsAnUnresolvedSelfProbeAsUnknownNeverAsAMeasuredZero() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
FakeLeadChannel channel = new FakeLeadChannel("mac-opus")
|
||||
.withMailbox("mac-opus", LeadChannel.MailboxState.unknown("mac-opus"));
|
||||
|
||||
McpSchema.CallToolResult res = FleetMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, null,
|
||||
FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), Map.of(), "",
|
||||
new FleetMcp.CoordinationSource(channel, List.of()));
|
||||
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"mailbox\":{\"status\":\"unknown\"}"), out);
|
||||
assertFalse(out.contains("\"pending\""), "an unresolved probe must never carry a pending count at all: " + out);
|
||||
assertFalse(out.contains("\"consumers\""), "an unresolved probe must never carry a consumers count at all: " + out);
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -572,6 +640,110 @@ class FleetMcpTest {
|
||||
"an ordinary fleet's output must be unchanged by this feature");
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsHeldMessagesWithATruncatedPreviewNeverTheFullBody() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
String longContent = "x".repeat(200);
|
||||
FakeLeadChannel channel = new FakeLeadChannel("mac-opus")
|
||||
.hold(new LeadMessage("m1", "fleet01-lead", "mac-opus", longContent));
|
||||
|
||||
McpSchema.CallToolResult res = FleetMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, null,
|
||||
FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), Map.of(), "",
|
||||
new FleetMcp.CoordinationSource(channel, List.of()));
|
||||
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"msgId\":\"m1\""), out);
|
||||
assertTrue(out.contains("\"from\":\"fleet01-lead\""), out);
|
||||
assertFalse(out.contains(longContent), "fleet_list must never dump a held message's full body: " + out);
|
||||
assertTrue(out.contains("x".repeat(80) + "…"), "expected an 80-char preview with an ellipsis: " + out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsEachDeclaredPeersLiveReachability() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
FakeLeadChannel channel = new FakeLeadChannel("mac-opus")
|
||||
.withMailbox("fleet01-lead", LeadChannel.MailboxState.exists("fleet01-lead", 2, 1))
|
||||
.withMailbox("fleet03-lead", LeadChannel.MailboxState.unknown("fleet03-lead"));
|
||||
// "fleet02-lead" is declared as a peer but never configured on the fake — inspect() falls
|
||||
// back to MailboxState.absent, exactly as a real down (never-run) peer would report.
|
||||
// "fleet03-lead" IS configured, as unknown — a broker that could not be reached in time,
|
||||
// which review finding 1 says must render distinctly from "fleet02-lead"'s confirmed absence.
|
||||
|
||||
McpSchema.CallToolResult res = FleetMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, null,
|
||||
FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), Map.of(), "",
|
||||
new FleetMcp.CoordinationSource(channel, List.of("fleet01-lead", "fleet02-lead", "fleet03-lead")));
|
||||
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"coordId\":\"fleet01-lead\",\"status\":\"exists\",\"pending\":2,\"consumers\":1"), out);
|
||||
assertTrue(out.contains("\"coordId\":\"fleet02-lead\",\"status\":\"absent\""), out);
|
||||
assertTrue(out.contains("\"coordId\":\"fleet03-lead\",\"status\":\"unknown\""), out);
|
||||
assertFalse(out.contains("\"coordId\":\"fleet02-lead\",\"status\":\"absent\",\"pending\""),
|
||||
"pending/consumers must be omitted, not faked as zero, for a confirmed-absent peer: " + out);
|
||||
assertFalse(out.contains("\"coordId\":\"fleet03-lead\",\"status\":\"unknown\",\"pending\""),
|
||||
"pending/consumers must be omitted, not faked as zero, for an unresolved peer probe: " + out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361 review finding 2: {@code get(timeout)} alone times out the CALLER but leaves the
|
||||
* submitted {@link LeadChannel#inspect} task running forever on its own virtual thread — against
|
||||
* a hung (not down) broker every probe would orphan one more thread holding an AMQP channel
|
||||
* until the connection's channel-max is exhausted, which would break {@code publish} too. This
|
||||
* proves {@link FleetMcp#probe(LeadChannel, String, long)} does not merely give up on a slow
|
||||
* task: it interrupts it, so the task does not go on running unbounded after the caller has
|
||||
* already moved on. No hung broker needed — a {@link LeadChannel} fake that blocks until
|
||||
* interrupted is enough to observe the same mechanism.
|
||||
*/
|
||||
@Test
|
||||
void aTimedOutProbeInterruptsTheOrphanedTaskRatherThanAbandoningIt() throws Exception {
|
||||
CountDownLatch started = new CountDownLatch(1);
|
||||
AtomicBoolean wasInterrupted = new AtomicBoolean(false);
|
||||
LeadChannel hangs = new LeadChannel() {
|
||||
@Override
|
||||
public void publish(String toCoordId, LeadMessage m) { }
|
||||
|
||||
@Override
|
||||
public List<LeadMessage> peek() { return List.of(); }
|
||||
|
||||
@Override
|
||||
public void ack(String msgId) { }
|
||||
|
||||
@Override
|
||||
public String selfCoordId() { return "mac-opus"; }
|
||||
|
||||
@Override
|
||||
public MailboxState inspect(String coordId) {
|
||||
started.countDown();
|
||||
try {
|
||||
Thread.sleep(60_000);
|
||||
} catch (InterruptedException e) {
|
||||
wasInterrupted.set(true);
|
||||
Thread.currentThread().interrupt();
|
||||
}
|
||||
return MailboxState.unknown(coordId);
|
||||
}
|
||||
};
|
||||
|
||||
LeadChannel.MailboxState result = FleetMcp.probe(hangs, "fleet01-lead", 100L);
|
||||
|
||||
assertFalse(result.exists(), "a timed-out probe must never claim the mailbox exists");
|
||||
assertFalse(result.known(), "a timed-out probe proves nothing either way — it must report unknown");
|
||||
assertTrue(started.await(2, TimeUnit.SECONDS), "the probe task must actually have started");
|
||||
// The interrupt is delivered asynchronously to the orphaned task's own thread — poll briefly
|
||||
// rather than assume it has already landed the instant probe() returns.
|
||||
long deadline = System.nanoTime() + TimeUnit.SECONDS.toNanos(2);
|
||||
while (!wasInterrupted.get() && System.nanoTime() < deadline) {
|
||||
Thread.sleep(20);
|
||||
}
|
||||
assertTrue(wasInterrupted.get(),
|
||||
"probe() must cancel the orphaned task (interrupt it) instead of leaving it to run forever");
|
||||
}
|
||||
|
||||
@Test
|
||||
void capacityUsesThePlacementLiveCount() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
@@ -603,6 +775,60 @@ class FleetMcpTest {
|
||||
assertTrue(out.contains("\"reclaimable\":0"), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #284: {@code reclaimable} has exactly one definition, and this pins it over EVERY
|
||||
* {@link MemberSession.State} — so a state added later cannot slip through unconsidered. Both
|
||||
* views in a {@code fleet_list} response call {@link FleetMcp#reclaimable}, so they cannot
|
||||
* drift apart.
|
||||
*
|
||||
* <p>{@code BACKEND_ERROR} and {@code FAILED} are NOT reclaimable on purpose. The ticket asked
|
||||
* for them to be; that half of the ticket was wrong. Their seat is already out of
|
||||
* {@code live}, so it is already in {@code free} — counting it here too would report the same
|
||||
* seat twice.
|
||||
*/
|
||||
@Test
|
||||
void onlyReadyAndDoneSessionsAreReclaimable() {
|
||||
Set<MemberSession.State> expected = EnumSet.of(MemberSession.State.READY, MemberSession.State.DONE);
|
||||
for (MemberSession.State state : MemberSession.State.values()) {
|
||||
MemberSession session = new MemberSession("p1", "term1", "ltms-local", null,
|
||||
"/tmp", null, 0L, 0L, 0, state, null, null);
|
||||
assertEquals(expected.contains(state), FleetMcp.reclaimable(session, null),
|
||||
"state " + state + " must " + (expected.contains(state) ? "" : "not ")
|
||||
+ "count as reclaimable");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #284, the operator-visible half. {@code liveCount} here is the value the real counter
|
||||
* ({@code Fleetd.liveSessionCount}, proven against the actual spawn gate in
|
||||
* {@code FleetdBackendErrorSinkTest}) produces for this roster: 0, because both sessions are
|
||||
* terminal. What this test pins is what {@code fleet_list} says around it — the two seats show
|
||||
* up once, in {@code free}, and are NOT counted a second time as {@code reclaimable}; neither
|
||||
* is any member row; and both dead sessions are still listed so the lead can see why.
|
||||
*/
|
||||
@Test
|
||||
void terminalFailureSessionsFreeTheirSeatWithoutBeingCountedReclaimable() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
MemberSession backendError = sessions.acquire("ltms-local", null, null, null);
|
||||
MemberSession failed = sessions.acquire("ltms-local", null, null, null);
|
||||
assertTrue(sessions.onBackendError(backendError.terminalId(), "backend exited"));
|
||||
sessions.onTurnFailed(failed.terminalId());
|
||||
|
||||
String out = textOf(FleetMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
sessions, null, new FleetMcp.CapacitySource(profile -> 0, profile -> 2,
|
||||
() -> Set.of("ltms-local"), () -> 0), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), Map.of(), ""));
|
||||
|
||||
assertTrue(out.contains("\"free\":2"), "both seats are back: " + out);
|
||||
assertTrue(out.contains("\"reclaimable\":0"),
|
||||
"the freed seats must not be counted a second time as reclaimable: " + out);
|
||||
assertFalse(out.contains("\"reclaimable\":true"),
|
||||
"no member row may claim a seat the profile count says is not held: " + out);
|
||||
assertTrue(out.contains("\"state\":\"backend_error\""), "the dead session stays visible: " + out);
|
||||
assertTrue(out.contains("\"state\":\"failed\""), "the failed session stays visible: " + out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void inertCapacitySourceOmitsCapacityBlock() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
@@ -810,13 +1036,16 @@ class FleetMcpTest {
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176: this is the exact shape measured on the Mac fleet — {@code maxLoad:3, live:2},
|
||||
* where one of the "free" three is really the lead's own seat on the same subscription. The old
|
||||
* formula ({@code max(0, cap - live)}) reported {@code free:1}; the real ceiling is {@code 0}
|
||||
* (two members plus the lead's own seat already fill all three).
|
||||
* fleetd #257: this is the exact shape measured on the Mac fleet — {@code maxLoad:3, live:2},
|
||||
* one lead sharing the subscription. The OLD formula ({@code max(0, cap - live - leadSeats)})
|
||||
* reported {@code free:0} here, disagreeing with the real spawn gate (which never read
|
||||
* {@code leadSeats} and would still grant one more spawn — see
|
||||
* {@code freeMatchesWhatTheRealPlacementGateActuallyGrants} below for that proof against the
|
||||
* actual gate). {@code free} must report {@code 1}: {@code leadSeats} is carried as a fact, not
|
||||
* subtracted.
|
||||
*/
|
||||
@Test
|
||||
void leadSeatSubtractsFromFreeTheSameWayLiveDoes() {
|
||||
void leadSeatIsReportedButNeverSubtractedFromFree() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
FleetMcp.LeadSeatSource leadSeats = new FleetMcp.LeadSeatSource(
|
||||
@@ -825,21 +1054,22 @@ class FleetMcpTest {
|
||||
String out = textOf(FleetMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
sessions, null, new FleetMcp.CapacitySource(profile -> 2, profile -> 3,
|
||||
() -> Set.of("sonnet"), () -> 0), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none(), leadSeats, Map.of(), "", null));
|
||||
FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none(), leadSeats, Map.of(), "",
|
||||
FleetMcp.CoordinationSource.none()));
|
||||
|
||||
assertTrue(out.contains("\"maxLoad\":3"), "maxLoad itself must be left untouched: " + out);
|
||||
assertTrue(out.contains("\"live\":2"), out);
|
||||
assertTrue(out.contains("\"free\":0"), "2 live + 1 lead seat fills all 3: " + out);
|
||||
assertTrue(out.contains("\"leadSeats\":1"), out);
|
||||
assertTrue(out.contains("\"free\":1"), "free is maxLoad - live only, never minus leadSeats: " + out);
|
||||
assertTrue(out.contains("\"leadSeats\":1"), "leadSeats is still reported, just not subtracted: " + out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176: the OTHER measurement in the issue — a completely idle fleet still overstates
|
||||
* {@code free} by the lead's own seat. {@code maxLoad:3, live:0} must report {@code free:2}, the
|
||||
* real fan-out ceiling, not {@code 3}.
|
||||
* fleetd #257: the OTHER measurement in the issue — a completely idle fleet with a lead sharing
|
||||
* the subscription. {@code maxLoad:3, live:0} must report {@code free:3}, matching what the
|
||||
* spawn gate (which has no notion of a lead's seat) would actually grant.
|
||||
*/
|
||||
@Test
|
||||
void leadSeatLowersFreeOnAnOtherwiseIdleSubscriptionProfile() {
|
||||
void leadSeatDoesNotLowerFreeOnAnOtherwiseIdleSubscriptionProfile() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
FleetMcp.LeadSeatSource leadSeats = new FleetMcp.LeadSeatSource(
|
||||
@@ -848,13 +1078,80 @@ class FleetMcpTest {
|
||||
String out = textOf(FleetMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
sessions, null, new FleetMcp.CapacitySource(profile -> 0, profile -> 3,
|
||||
() -> Set.of("sonnet"), () -> 0), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none(), leadSeats, Map.of(), "", null));
|
||||
FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none(), leadSeats, Map.of(), "",
|
||||
FleetMcp.CoordinationSource.none()));
|
||||
|
||||
assertTrue(out.contains("\"live\":0"), out);
|
||||
assertTrue(out.contains("\"free\":2"), "an idle fleet's real ceiling is 3 minus the lead's own seat: " + out);
|
||||
assertTrue(out.contains("\"free\":3"), "the real gate never subtracts the lead's seat: " + out);
|
||||
assertTrue(out.contains("\"leadSeats\":1"), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #257 — the defect, driven against the REAL placement gate, not a copy of its
|
||||
* arithmetic. {@code free} must equal exactly how many more spawns
|
||||
* {@link CompositePeerLauncher#spawn} (routed through the same {@link SessionManager} fleet_spawn
|
||||
* itself uses) will grant on this profile right now: this test reads whatever number
|
||||
* {@code fleet_list}'s real {@code listFleet} call reports, then drives that many spawns through
|
||||
* the REAL composite launcher and asserts every one succeeds, and the next one — one past what
|
||||
* fleet_list promised — is refused. A test that instead hand-computed {@code cap - live} and
|
||||
* compared it to {@code free} would pass even if both sides shared the same wrong formula (this
|
||||
* repo has been bitten by exactly that before); this one only passes when fleet_list's number and
|
||||
* the gate's real behaviour actually agree.
|
||||
*
|
||||
* <p>{@code liveCount} here is wired the same way {@code Fleetd.main} wires it in production: one
|
||||
* function, read by both the {@link CompositePeerLauncher}'s {@code maxLoad} gate and
|
||||
* {@code fleet_list}'s {@code CapacitySource}, off the SAME {@link SessionManager#roster()} — so
|
||||
* the two paths cannot silently drift on what "live" means.
|
||||
*/
|
||||
@Test
|
||||
void freeMatchesWhatTheRealPlacementGateActuallyGrants() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
FleetConfig.Profile wcfg = new FleetConfig.Profile(
|
||||
"sonnet", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN", null,
|
||||
"tab", "fleetd-workers", "worker: {profile} #{n}", null,
|
||||
null, null, null, null, null, null, null, 3);
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(wcfg.profile(), wcfg);
|
||||
ClaudeCodeLauncher delegate = new ClaudeCodeLauncher(
|
||||
new AgentControl(h), new WorkspaceControl(h), new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
profiles, wcfg.profile(), k -> "FLEETD_WORKER_TOKEN".equals(k) ? "tok" : null);
|
||||
|
||||
java.util.concurrent.atomic.AtomicReference<SessionManager> smRef =
|
||||
new java.util.concurrent.atomic.AtomicReference<>();
|
||||
Function<String, Integer> liveCount = profile -> (int) smRef.get().roster().stream()
|
||||
.filter(s -> profile.equals(s.profile())).count();
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(delegate), wcfg.profile(), profiles, PlacementPolicies.fixed(), liveCount);
|
||||
SessionManager sm = new SessionManager(composite);
|
||||
smRef.set(sm);
|
||||
|
||||
// Two members already live — the exact shape measured in fleetd #257 (maxLoad:3, live:2).
|
||||
assertFalse(FleetMcp.spawn(sm, "sonnet").isError(), "setup: first live member must spawn cleanly");
|
||||
assertFalse(FleetMcp.spawn(sm, "sonnet").isError(), "setup: second live member must spawn cleanly");
|
||||
|
||||
// A lead session shares this profile's subscription — leadSeats:1, same as the ticket.
|
||||
FleetMcp.LeadSeatSource leadSeats = new FleetMcp.LeadSeatSource(p -> "sonnet".equals(p) ? 1 : 0);
|
||||
String out = textOf(FleetMcp.listFleet(composite, sm, null,
|
||||
new FleetMcp.CapacitySource(liveCount, p -> profiles.get(p).maxLoad(), profiles::keySet, () -> 0),
|
||||
new FleetMcp.HealthCoverageSource(() -> "off"), FleetMcp.QuarantineSource.none(),
|
||||
FleetMcp.OutageSource.none(), leadSeats, Map.of(), "", FleetMcp.CoordinationSource.none()));
|
||||
int reportedFree = extractInt(out, "free");
|
||||
|
||||
for (int i = 0; i < reportedFree; i++) {
|
||||
McpSchema.CallToolResult res = FleetMcp.spawn(sm, "sonnet");
|
||||
assertFalse(res.isError(), "fleet_list promised free:" + reportedFree + "; spawn #" + (i + 1)
|
||||
+ " of that many was refused by the real gate: " + textOf(res));
|
||||
}
|
||||
McpSchema.CallToolResult overflow = FleetMcp.spawn(sm, "sonnet");
|
||||
assertTrue(overflow.isError(), "fleet_list reported free:" + reportedFree
|
||||
+ " but the real placement gate granted at least one more spawn than that: " + textOf(overflow));
|
||||
}
|
||||
|
||||
private static int extractInt(String json, String key) {
|
||||
java.util.regex.Matcher m = java.util.regex.Pattern.compile("\"" + key + "\":(-?\\d+)").matcher(json);
|
||||
assertTrue(m.find(), "no \"" + key + "\" field in: " + json);
|
||||
return Integer.parseInt(m.group(1));
|
||||
}
|
||||
|
||||
/** A profile with no lead seats reported must be byte-identical to before this ticket. */
|
||||
@Test
|
||||
void zeroLeadSeatsOmitsTheKeyAndLeavesFreeUnchanged() {
|
||||
@@ -865,7 +1162,7 @@ class FleetMcpTest {
|
||||
sessions, null, new FleetMcp.CapacitySource(profile -> 0, profile -> 2,
|
||||
() -> Set.of("terra"), () -> 0), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none(), FleetMcp.LeadSeatSource.none(),
|
||||
Map.of(), "", null));
|
||||
Map.of(), "", FleetMcp.CoordinationSource.none()));
|
||||
|
||||
assertTrue(out.contains("\"free\":2"), out);
|
||||
assertFalse(out.contains("leadSeats"), "no lead shares this profile's credential: " + out);
|
||||
|
||||
@@ -0,0 +1,121 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #114 (CB-609): the guard that lets {@code docs/MCP-Contract.md} name a tool at all.
|
||||
*
|
||||
* <p>That page was written in July 2026, before any MCP code existed, and then did not follow the
|
||||
* code. By August it named two tools that had never been built, omitted five that shipped, had the
|
||||
* wrong name for nearly every parameter, and still described a caller-identity rule that was a
|
||||
* privilege bug by then. Nothing failed, because nothing checked it — and {@code CLAUDE.md} sends
|
||||
* every session in the fleet to that page.
|
||||
*
|
||||
* <p>The fix was to delete the tool catalogue rather than correct it: a hand-maintained second copy
|
||||
* of the tool surface is the defect, not the particular errors it had accumulated. What survives is
|
||||
* the flows, which are shapes rather than names. But the flows still have to say {@code fleet_send}
|
||||
* somewhere to be readable, and that is exactly the sentence that rots. This test is what makes it
|
||||
* safe to write.
|
||||
*
|
||||
* <p><b>It checks source text, not behaviour.</b> It reads the Markdown and reads {@link FleetMcp}'s
|
||||
* source, and it only catches a name in the doc that the server does not register. It cannot catch a
|
||||
* flow that describes the wrong order, or a parameter name in prose — those are not name-shaped. The
|
||||
* doc's own header carries that caveat for its readers.
|
||||
*/
|
||||
class McpContractDocTest {
|
||||
|
||||
/** Tests run with the module directory as cwd, so the repo-root doc is one level up. */
|
||||
private static final Path DOC = Path.of("../docs/MCP-Contract.md");
|
||||
private static final Path MCP_SOURCE = Path.of("src/main/java/dev/ltms/fleet/mcp/FleetMcp.java");
|
||||
|
||||
private static Set<String> matches(Path file, String regex) throws Exception {
|
||||
Matcher m = Pattern.compile(regex).matcher(Files.readString(file));
|
||||
Set<String> found = new LinkedHashSet<>();
|
||||
while (m.find()) {
|
||||
found.add(m.group(1));
|
||||
}
|
||||
return found;
|
||||
}
|
||||
|
||||
/** Every {@code fleet_*} the doc mentions, in prose or in a diagram. */
|
||||
private static Set<String> toolsNamedInTheDoc() throws Exception {
|
||||
return matches(DOC, "(fleet_[a-z_]+)");
|
||||
}
|
||||
|
||||
/** Every tool {@link FleetMcp} actually registers, read from its {@code tool("…")} calls. */
|
||||
private static Set<String> toolsTheServerRegisters() throws Exception {
|
||||
return matches(MCP_SOURCE, "tool\\(\"(fleet_[a-z_]+)\"");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] every fleet_* tool named in MCP-Contract.md is one the server registers")
|
||||
void theDocNamesNoToolThatDoesNotExist() throws Exception {
|
||||
Set<String> registered = toolsTheServerRegisters();
|
||||
Set<String> named = toolsNamedInTheDoc();
|
||||
|
||||
Set<String> unknown = new LinkedHashSet<>(named);
|
||||
unknown.removeAll(registered);
|
||||
|
||||
assertTrue(unknown.isEmpty(),
|
||||
"docs/MCP-Contract.md names " + unknown + ", which FleetMcp does not register. "
|
||||
+ "Checked " + named.size() + " name(s) in the doc against " + registered.size()
|
||||
+ " registered tool(s): " + registered + ". This is the fleetd #114 defect "
|
||||
+ "recurring — the doc named fleet_read and fleet_cancel for weeks after the "
|
||||
+ "code shipped without them. Either fix the name or drop it from the page; do "
|
||||
+ "NOT weaken this test.");
|
||||
}
|
||||
|
||||
/**
|
||||
* The denominator guard. The check above passes trivially if the doc stops naming any tool at
|
||||
* all — an empty set is a subset of everything. A checker that can silently check nothing is the
|
||||
* fleetd #113 shape, so this pins that the doc really is still describing the flows, and that
|
||||
* the registration scrape really did find the server's tools.
|
||||
*/
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] the doc/server name check is not vacuous — both sides found names")
|
||||
void theCheckActuallyHasSomethingToCheck() throws Exception {
|
||||
Set<String> registered = toolsTheServerRegisters();
|
||||
Set<String> named = toolsNamedInTheDoc();
|
||||
|
||||
assertTrue(registered.size() >= 10,
|
||||
"scraped only " + registered.size() + " tool registrations from FleetMcp (" + registered
|
||||
+ "); the server registers eleven, so the tool(\"…\") scrape has stopped matching "
|
||||
+ "and the check above is now vacuous");
|
||||
assertTrue(named.size() >= 4,
|
||||
"docs/MCP-Contract.md names only " + named.size() + " fleet_* tool(s) (" + named + "). "
|
||||
+ "The flows describe delegation, clarification, detached delivery and the "
|
||||
+ "turn-done fallback, so it should name several. Too few means the page has been "
|
||||
+ "gutted and this test is guarding nothing.");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #114's actual lesson. The catalogue was deleted on purpose; a well-meaning "let me just
|
||||
* document the tools here" restores the exact second copy that drifted for a month.
|
||||
*/
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] MCP-Contract.md still says it is not the tool reference")
|
||||
void theDocStillDisclaimsBeingTheToolReference() throws Exception {
|
||||
String doc = Files.readString(DOC);
|
||||
assertTrue(doc.contains("**What this page is NOT: a tool reference.**"),
|
||||
"docs/MCP-Contract.md must keep saying it is not the tool reference. That sentence is "
|
||||
+ "the fix for fleetd #114: the page carried a hand-maintained tool catalogue that "
|
||||
+ "drifted from the code for a month while CLAUDE.md pointed every session at it.");
|
||||
assertEquals(0, countTables(doc.substring(0, doc.indexOf("## 1. Rendezvous flows"))),
|
||||
"the header of docs/MCP-Contract.md must not grow a tool/parameter table — that is the "
|
||||
+ "second copy fleetd #114 deleted");
|
||||
}
|
||||
|
||||
private static int countTables(String markdown) {
|
||||
return (int) markdown.lines().filter(l -> l.strip().startsWith("|")).count();
|
||||
}
|
||||
}
|
||||
@@ -19,11 +19,14 @@ import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import org.junit.jupiter.api.AfterAll;
|
||||
import org.junit.jupiter.api.BeforeAll;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.attribute.PosixFileAttributeView;
|
||||
@@ -32,10 +35,12 @@ import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
@@ -147,18 +152,25 @@ class ClaudeCodeLauncherTest {
|
||||
// every ide_* call to the member's own worktree via the charter.
|
||||
|
||||
/** A profile carrying an ideMcpUrl (plus optional bridge mcpUrl and cwd). ideMcpUrl is the last record component. */
|
||||
private FleetConfig.Profile ideProfile(String mcpUrl, String ideMcpUrl, String cwd) {
|
||||
/**
|
||||
* {@code configDir} is first and mandatory on purpose (fleetd #258). A profile that sets no
|
||||
* {@code configDir} sends {@code seedTrustDialog}'s write to the operator's real
|
||||
* {@code ~/.claude.json}, and the fleetd #149 gate does not stop that when the fixture's
|
||||
* {@code cwd} is worktree-shaped — which every IDE-overlay fixture's is. Pass a {@code @TempDir}
|
||||
* whenever {@code cwd} has a {@code .git} FILE; {@code null} is only safe when it does not.
|
||||
*/
|
||||
private FleetConfig.Profile ideProfile(String configDir, String mcpUrl, String ideMcpUrl, String cwd) {
|
||||
return new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", configDir, "FLEETD_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "fleetd-workers", "w #{n}", mcpUrl, cwd, null,
|
||||
null, null, null, null, null, null, null, null, null, ideMcpUrl);
|
||||
}
|
||||
|
||||
/** As {@link #ideProfile} but carrying the CB-634 auto-open fields (module subdir + open command). */
|
||||
private FleetConfig.Profile ideProfileModule(String ideMcpUrl, String cwd, String ideProjectDir,
|
||||
String ideOpenCommand) {
|
||||
private FleetConfig.Profile ideProfileModule(String configDir, String ideMcpUrl, String cwd,
|
||||
String ideProjectDir, String ideOpenCommand) {
|
||||
return new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", configDir, "FLEETD_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "fleetd-workers", "w #{n}", null, cwd, null,
|
||||
null, null, null, null, null, null, null, null, null, ideMcpUrl, ideProjectDir, ideOpenCommand);
|
||||
}
|
||||
@@ -171,7 +183,7 @@ class ClaudeCodeLauncherTest {
|
||||
@Test
|
||||
void mountsIdeMcpAsSecondServerWhenIdeMcpUrlSet() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
launcher(herdr, ideProfile("http://127.0.0.1:8765/mcp",
|
||||
launcher(herdr, ideProfile(null, "http://127.0.0.1:8765/mcp",
|
||||
"http://127.0.0.1:29170/index-mcp/streamable-http", null)).spawn();
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
@@ -186,7 +198,8 @@ class ClaudeCodeLauncherTest {
|
||||
@Test
|
||||
void ideMcpUrlAloneStillEmitsTheMount() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
launcher(herdr, ideProfile(null, "http://127.0.0.1:29170/index-mcp/streamable-http", null)).spawn();
|
||||
launcher(herdr, ideProfile(null, null, "http://127.0.0.1:29170/index-mcp/streamable-http", null))
|
||||
.spawn();
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
assertTrue(args.contains("--mcp-config"),
|
||||
@@ -201,7 +214,7 @@ class ClaudeCodeLauncherTest {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
String roleCharter = "You review changes.";
|
||||
String worktree = "/tmp/.fleet-worktrees/rev-1";
|
||||
FleetConfig.Profile cfg = ideProfile("http://127.0.0.1:8765/mcp",
|
||||
FleetConfig.Profile cfg = ideProfile(null, "http://127.0.0.1:8765/mcp",
|
||||
"http://127.0.0.1:29170/index-mcp/streamable-http", worktree);
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null,
|
||||
@@ -226,6 +239,62 @@ class ClaudeCodeLauncherTest {
|
||||
}, "the --append-system-prompt-file path must be a readable file");
|
||||
}
|
||||
|
||||
// ---- fleetd #258: no fixture in this class may seed the operator's real ~/.claude.json ----
|
||||
//
|
||||
// seedTrustDialog writes <configDir>/.claude.json, or ~/.claude.json when the profile sets no
|
||||
// configDir. The fleetd #149 gate (isProvisionedWorktree) closes the case where a fixture leaves
|
||||
// cwd unset and it falls back to user.dir. It does NOT close the case where a fixture builds a
|
||||
// worktree-shaped @TempDir on purpose — the gate opens, and a null configDir still points the
|
||||
// write at the real home. Two IDE-overlay fixtures did exactly that on EVERY run; by 2026-09-03
|
||||
// the operator's ~/.claude.json carried 116 dead JUnit temp paths, none of which still existed.
|
||||
//
|
||||
// DIFFERENTIAL, not absolute: it snapshots the temp-dir project keys already present and fails
|
||||
// only on keys this class ADDS. An absolute check would fail on any host still carrying the
|
||||
// historical entries, and a check that fails for a reason nobody can fix gets deleted, not fixed.
|
||||
|
||||
private static Set<String> tempProjectKeysBefore;
|
||||
|
||||
@BeforeAll
|
||||
static void snapshotTempProjectKeysInTheDefaultClaudeJson() {
|
||||
tempProjectKeysBefore = tempProjectKeysInDefaultClaudeJson();
|
||||
}
|
||||
|
||||
@AfterAll
|
||||
static void noFixtureSeededTheDefaultClaudeJson() {
|
||||
Set<String> added = new TreeSet<>(tempProjectKeysInDefaultClaudeJson());
|
||||
added.removeAll(tempProjectKeysBefore);
|
||||
assertTrue(added.isEmpty(),
|
||||
"a fixture in this class seeded the DEFAULT .claude.json (the operator's real file "
|
||||
+ "when user.home is not redirected) with " + added.size() + " temp-dir "
|
||||
+ "project entry/entries: " + added + ". Give that fixture's profile a "
|
||||
+ "@TempDir configDir — see ideProfile's javadoc.");
|
||||
}
|
||||
|
||||
/**
|
||||
* Project keys under the JVM temp dir in {@code <user.home>/.claude.json}, or an empty set when
|
||||
* the file is absent or unreadable. Only key NAMES are read; nothing in the operator's file is
|
||||
* copied, asserted on, or written back.
|
||||
*/
|
||||
private static Set<String> tempProjectKeysInDefaultClaudeJson() {
|
||||
Set<String> keys = new TreeSet<>();
|
||||
Path target = Path.of(System.getProperty("user.home"), ".claude.json");
|
||||
if (!Files.isRegularFile(target)) {
|
||||
return keys;
|
||||
}
|
||||
String tmp = System.getProperty("java.io.tmpdir");
|
||||
try {
|
||||
JsonNode projects = new ObjectMapper().readTree(target.toFile()).path("projects");
|
||||
projects.fieldNames().forEachRemaining(name -> {
|
||||
if (name.startsWith(tmp) || name.contains("/junit-")) {
|
||||
keys.add(name);
|
||||
}
|
||||
});
|
||||
} catch (IOException e) {
|
||||
return keys; // unreadable file proves nothing either way
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
|
||||
// CB-634: the IDE guidance is delivered as a CLAUDE.local.md overlay (written only into a
|
||||
// provisioned worktree — cwd with a `.git` FILE) and registered in the repository's COMMON
|
||||
// info/exclude. git reads a worktree's excludes from the common dir, not the per-worktree
|
||||
@@ -241,8 +310,8 @@ class ClaudeCodeLauncherTest {
|
||||
Files.writeString(worktree.resolve(".git"), "gitdir: " + gitDir);
|
||||
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
launcher(herdr, ideProfile(null, "http://127.0.0.1:29170/index-mcp/streamable-http",
|
||||
worktree.toString())).spawn();
|
||||
launcher(herdr, ideProfile(root.toString(), null,
|
||||
"http://127.0.0.1:29170/index-mcp/streamable-http", worktree.toString())).spawn();
|
||||
|
||||
Path overlay = worktree.resolve("CLAUDE.local.md");
|
||||
assertTrue(Files.exists(overlay), "the overlay is written beside the project's CLAUDE.md");
|
||||
@@ -268,8 +337,9 @@ class ClaudeCodeLauncherTest {
|
||||
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// ideProjectDir "fleetd" ⇒ the pin is <worktree>/fleetd, not <worktree>.
|
||||
launcher(herdr, ideProfileModule("http://127.0.0.1:29170/index-mcp/streamable-http",
|
||||
worktree.toString(), "fleetd", null)).spawn();
|
||||
launcher(herdr, ideProfileModule(root.toString(),
|
||||
"http://127.0.0.1:29170/index-mcp/streamable-http", worktree.toString(), "fleetd", null))
|
||||
.spawn();
|
||||
|
||||
Path overlay = worktree.resolve("CLAUDE.local.md");
|
||||
assertTrue(Files.exists(overlay), "the overlay file still lives at the worktree root");
|
||||
@@ -304,8 +374,8 @@ class ClaudeCodeLauncherTest {
|
||||
Files.createDirectories(worktree.resolve(".git"));
|
||||
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
launcher(herdr, ideProfile(null, "http://127.0.0.1:29170/index-mcp/streamable-http",
|
||||
worktree.toString())).spawn();
|
||||
launcher(herdr, ideProfile(root.toString(), null,
|
||||
"http://127.0.0.1:29170/index-mcp/streamable-http", worktree.toString())).spawn();
|
||||
|
||||
assertFalse(Files.exists(worktree.resolve("CLAUDE.local.md")),
|
||||
"the safety gate refuses to write into a non-worktree cwd (.git directory)");
|
||||
@@ -855,6 +925,71 @@ class ClaudeCodeLauncherTest {
|
||||
assertTrue(herdr.called("pane.close"), "stop via handle.id() must close the pane");
|
||||
}
|
||||
|
||||
/** A tab-placement launcher with {@code memberCredentials policy=allow-list} under a zsh shell — the
|
||||
* combination that makes {@link HerdrPeerLauncher#spawn} generate a real ZDOTDIR, so {@code
|
||||
* releaseZdotdir}'s effect (the directory's deletion) is observable from a test. */
|
||||
private ClaudeCodeLauncher serviceWithAllowList(FakeHerdr herdr) {
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
Supplier<FleetConfig.MemberCredentials> creds = () -> new FleetConfig.MemberCredentials(
|
||||
FleetConfig.MemberCredentials.POLICY_ALLOW_LIST, List.of(), List.of(), null);
|
||||
Function<String, String> env = name -> "SHELL".equals(name) ? "/bin/zsh" : null;
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
env, 0, System::currentTimeMillis, () -> { }, null, creds);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #293: {@code stop()} used to run {@code spaces.closeTab} bare — any non-{@code
|
||||
* *_not_found} herdr error propagated straight out of {@code stop()}, skipping {@code
|
||||
* releaseZdotdir} entirely (the pane was already closed by that point, so the tab-close failure
|
||||
* is cosmetic, not a real teardown failure). Proves both halves of the fix: {@code stop()} no
|
||||
* longer throws for this failure, and {@code releaseZdotdir} still runs — observed here by the
|
||||
* generated ZDOTDIR actually being deleted, since {@code releaseZdotdir}'s last line is {@code
|
||||
* EnvAllowListScrub.deleteRecursively(dir)}.
|
||||
*/
|
||||
@Test
|
||||
@SuppressWarnings("unchecked")
|
||||
void stopStillReleasesZdotdirWhenCloseTabFailsWithANonNotFoundCode() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = serviceWithAllowList(herdr);
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
Map<String, Object> tabCreateParams = (Map<String, Object>) herdr.lastCall("tab.create").params();
|
||||
Map<String, String> tabEnv = (Map<String, String>) tabCreateParams.get("env");
|
||||
String zdotdir = tabEnv.get("ZDOTDIR");
|
||||
assertNotNull(zdotdir, "policy=allow-list under a zsh shell must have generated a ZDOTDIR: " + tabEnv);
|
||||
Path dir = Path.of(zdotdir);
|
||||
assertTrue(Files.isDirectory(dir), "the generated ZDOTDIR must exist before stop(): " + dir);
|
||||
herdr.tabCloseFailsForTab("w9:t2", "internal_error");
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
assertDoesNotThrow(() -> svc.stop(handle.id()),
|
||||
"fleetd #293: a failing tab.close is cosmetic — it must not propagate out of stop()");
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
|
||||
assertTrue(herdr.called("tab.close"), "tab.close was still attempted");
|
||||
assertFalse(Files.exists(dir),
|
||||
"releaseZdotdir must still run and delete the generated ZDOTDIR despite the tab.close "
|
||||
+ "failure: " + dir);
|
||||
String warn = appender.list.stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.filter(m -> m.contains("tab.close") && m.contains("w9:t2"))
|
||||
.findFirst()
|
||||
.orElse(null);
|
||||
assertNotNull(warn, "the failing tab.close must be logged at WARN naming the tab id — a "
|
||||
+ "silently swallowed failure with no message is not an improvement. Log lines: "
|
||||
+ appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList());
|
||||
}
|
||||
|
||||
// --- CB-519: host-unique id, decoupled from the pane coordinate ------------------------------
|
||||
|
||||
@Test
|
||||
@@ -1030,7 +1165,8 @@ class ClaudeCodeLauncherTest {
|
||||
@Test
|
||||
void spawnLetsAnUnrelatedHerdrErrorPropagateUnchanged() {
|
||||
// Fix 1 must only special-case a "*_not_found" answer. Any other herdr failure keeps
|
||||
// propagating as-is — this gate does not know how to recover from it.
|
||||
// propagating as-is — this gate does not know how to recover from it. The pane still needs
|
||||
// closing because spawn throws before it can return the pane id to a caller that could stop it.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown");
|
||||
herdr.agentGetFailsWithAfter(0, "internal_error");
|
||||
@@ -1047,8 +1183,27 @@ class ClaudeCodeLauncherTest {
|
||||
() -> svc.spawn(new SpawnRequest(null, null, null)));
|
||||
|
||||
assertEquals("internal_error", ex.code());
|
||||
assertEquals(0, paneCloseCount(herdr, "w9:pRoot_1"),
|
||||
"an error this gate does not recognize is not this gate's teardown to run");
|
||||
assertEquals(1, paneCloseCount(herdr, "w9:pRoot_1"),
|
||||
"the unchanged error leaves spawn without a pane id, so this gate closes its orphaned pane");
|
||||
}
|
||||
|
||||
@Test
|
||||
void panePlacementClosesTheSplitPaneWhenAgentStartFails() {
|
||||
FakeHerdr herdr = new FakeHerdr().agentStartFailsWith("internal_error");
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||
List.of("claude"), "pane", "fleetd-workers", "w #{n}", null, null, null);
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
|
||||
dev.ltms.fleet.herdr.HerdrException ex = assertThrows(
|
||||
dev.ltms.fleet.herdr.HerdrException.class,
|
||||
() -> svc.spawn(new SpawnRequest(null, null, null)));
|
||||
|
||||
assertEquals("internal_error", ex.code(), "agent.start failure propagates unchanged");
|
||||
assertEquals(1, paneCloseCount(herdr, "w1:pSplit"),
|
||||
"the pane split for a peer that never starts is closed instead of left orphaned");
|
||||
}
|
||||
|
||||
// --- fleetd #176 fix 2: corroborated UNKNOWN refinement --------------------------------------
|
||||
@@ -1409,14 +1564,16 @@ class ClaudeCodeLauncherTest {
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-633 follow-up (#192): the mirror of the zsh test above. On a non-zsh login shell {@code
|
||||
* ZDOTDIR} is ignored, so no scrub ever runs — the report must keep the WARN wording (a name here
|
||||
* really is inherited unblocked) rather than claiming a scrub protects it. This is the trap PR
|
||||
* #174 fell into the other direction: keying the wording on the shell, not on {@code
|
||||
* creds.isAllowList()}, is what keeps this branch correct.
|
||||
* fleetd #155 (was CB-633 follow-up (#192) "allowListPolicyOnNonZshKeepsTheWarnWording"): on a
|
||||
* non-zsh login shell {@code ZDOTDIR} is ignored, so no scrub could ever run there. Before #155
|
||||
* the launcher degraded to the weaker overlay and spawned anyway; that is exactly the "control
|
||||
* silently does nothing" failure #155 exists to close, since {@code policy: allow-list} is the
|
||||
* operator explicitly asking for a control a sourced file cannot undo. Now the spawn is REFUSED
|
||||
* instead — this test asserts the refusal, naming the shell, on the real spawn path ({@link
|
||||
* ClaudeCodeLauncher#spawn}), not merely on the launcher method that computes it.
|
||||
*/
|
||||
@Test
|
||||
void allowListPolicyOnNonZshKeepsTheWarnWording() {
|
||||
void allowListPolicyOnNonZshRefusesTheSpawn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||
@@ -1430,27 +1587,12 @@ class ClaudeCodeLauncherTest {
|
||||
0, System::currentTimeMillis, () -> {}, null, () -> creds,
|
||||
() -> Set.of("AI_GATEWAY_TOKEN", "A_BRAND_NEW_SECRET_TOKEN"));
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
svc.spawn();
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class, svc::spawn,
|
||||
"policy=allow-list on a non-zsh shell must refuse the spawn, not silently degrade "
|
||||
+ "to the weaker overlay");
|
||||
|
||||
assertTrue(appender.list.stream().anyMatch(e ->
|
||||
e.getFormattedMessage().contains("memberCredentials gap")
|
||||
&& e.getFormattedMessage().contains("UNBLOCKED")
|
||||
&& e.getFormattedMessage().contains("A_BRAND_NEW_SECRET_TOKEN")),
|
||||
"no scrub runs on a non-zsh shell, so the WARN wording must be kept — got: "
|
||||
+ appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList());
|
||||
assertFalse(appender.list.stream().anyMatch(e ->
|
||||
e.getFormattedMessage().contains("memberCredentials gap")
|
||||
&& e.getFormattedMessage().toLowerCase(java.util.Locale.ROOT).contains("scrub")),
|
||||
"nothing is scrubbed on this path, so the report must not claim otherwise — got: "
|
||||
+ appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList());
|
||||
assertTrue(e.getMessage().contains("allow-list") && e.getMessage().contains("/bin/bash"),
|
||||
"the refusal must name the policy and the actual shell, got: " + e.getMessage());
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -2159,8 +2301,7 @@ class ClaudeCodeLauncherTest {
|
||||
}
|
||||
JsonNode root = new ObjectMapper().readTree(claudeJson.toFile());
|
||||
JsonNode project = root.path("projects").path(worktree.toString());
|
||||
seededBeforeStart.set(project.path("hasTrustDialogAccepted").asBoolean(false)
|
||||
&& project.path("hasCompletedProjectOnboarding").asBoolean(false));
|
||||
seededBeforeStart.set(project.path("hasTrustDialogAccepted").asBoolean(false));
|
||||
} catch (IOException e) {
|
||||
seededBeforeStart.set(false);
|
||||
}
|
||||
@@ -2178,7 +2319,7 @@ class ClaudeCodeLauncherTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void seedTrustDialogWritesBothTrustFlagsForTheResolvedCwd(
|
||||
void seedTrustDialogWritesOnlyTheTrustFlagAndNotTheOnboardingKey(
|
||||
@TempDir Path configDir, @TempDir Path worktree) throws Exception {
|
||||
markAsProvisionedWorktree(worktree);
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
@@ -2192,7 +2333,13 @@ class ClaudeCodeLauncherTest {
|
||||
JsonNode project = new ObjectMapper().readTree(claudeJson.toFile())
|
||||
.path("projects").path(worktree.toString());
|
||||
assertTrue(project.path("hasTrustDialogAccepted").asBoolean(false));
|
||||
assertTrue(project.path("hasCompletedProjectOnboarding").asBoolean(false));
|
||||
// fleetd #247: the onboarding key must NOT be written. Claude Code strips it on every
|
||||
// save (measured: 0 of 28 live entries had it, including its own), so writing it only
|
||||
// adds a contested key to a file two processes share. This assertion is the guard that
|
||||
// stops it coming back as a plausible-looking "completeness" fix.
|
||||
assertFalse(project.has("hasCompletedProjectOnboarding"),
|
||||
"hasCompletedProjectOnboarding must not be written — Claude Code drops it on "
|
||||
+ "every save, and the member reaches idle on hasTrustDialogAccepted alone");
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -2238,7 +2385,6 @@ class ClaudeCodeLauncherTest {
|
||||
|
||||
JsonNode mine = root.path("projects").path(worktree.toString());
|
||||
assertTrue(mine.path("hasTrustDialogAccepted").asBoolean(false));
|
||||
assertTrue(mine.path("hasCompletedProjectOnboarding").asBoolean(false));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -2535,4 +2681,272 @@ class ClaudeCodeLauncherTest {
|
||||
+ tornRead.get());
|
||||
assertEquals(newContent, Files.readString(target), "the final content must be the new content");
|
||||
}
|
||||
|
||||
// --- fleetd #247: CAS against a writer TRUST_JSON_LOCK cannot reach --------------------------
|
||||
//
|
||||
// fleetd #149's lock only serialises seedTrustDialog calls THIS launcher makes inside this one
|
||||
// JVM. It does nothing about the one writer that actually shares this file on a real host: the
|
||||
// operator's own live Claude Code, whose CLAUDE_CONFIG_DIR is routinely the very configDir this
|
||||
// profile is given. A plain read-modify-write there is a routine lost update — fleetd reads v1,
|
||||
// the operator's session writes v2, fleetd's ATOMIC_MOVE lands v3 built from v1, and v2 is gone,
|
||||
// atomically. The fix re-reads the file's exact bytes immediately before the move and compares
|
||||
// them with what the update was built from, retrying from fresh bytes on a mismatch.
|
||||
//
|
||||
// Both tests below drive the race through the real seedTrustDialog/spawn() path (not a
|
||||
// hand-rolled call to some extracted primitive) using trustJsonCasTestHook — a seam fired once
|
||||
// per CAS attempt, at the exact point between the read and the final compare, so the race is
|
||||
// deterministic instead of depending on real thread timing.
|
||||
|
||||
@Test
|
||||
void seedTrustDialogRetriesAndPreservesAConcurrentExternalWritersChange(
|
||||
@TempDir Path configDir, @TempDir Path worktree) throws Exception {
|
||||
markAsProvisionedWorktree(worktree);
|
||||
Path claudeJson = configDir.resolve(".claude.json");
|
||||
Files.writeString(claudeJson, "{\"projects\":{}}");
|
||||
|
||||
// Fires exactly once, on the first CAS attempt — simulating the operator's own live Claude
|
||||
// Code landing its own write to this SAME file in the gap between fleetd's read and write.
|
||||
AtomicBoolean fired = new AtomicBoolean(false);
|
||||
ClaudeCodeLauncher.trustJsonCasTestHook = () -> {
|
||||
if (fired.compareAndSet(false, true)) {
|
||||
try {
|
||||
Files.writeString(claudeJson,
|
||||
"{\"projects\":{\"/operator/own/project\":"
|
||||
+ "{\"hasTrustDialogAccepted\":true}}}");
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException(e);
|
||||
}
|
||||
}
|
||||
};
|
||||
try {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile cfg = trustProfile(configDir.toString(), worktree.toString());
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null)
|
||||
.spawn();
|
||||
} finally {
|
||||
ClaudeCodeLauncher.trustJsonCasTestHook = () -> {};
|
||||
}
|
||||
|
||||
assertTrue(fired.get(), "the race hook must actually have fired during the spawn");
|
||||
JsonNode root = new ObjectMapper().readTree(claudeJson.toFile());
|
||||
assertTrue(root.path("projects").path("/operator/own/project")
|
||||
.path("hasTrustDialogAccepted").asBoolean(false),
|
||||
"the external writer's change, landed between fleetd's read and write, must SURVIVE "
|
||||
+ "— this is the whole point of the CAS: without it, fleetd's stale-built "
|
||||
+ "ATOMIC_MOVE would have silently discarded it. Final file: "
|
||||
+ Files.readString(claudeJson));
|
||||
assertTrue(root.path("projects").path(worktree.toString())
|
||||
.path("hasTrustDialogAccepted").asBoolean(false),
|
||||
"fleetd's own retry must still land its own trust entry, built from the fresh bytes");
|
||||
}
|
||||
|
||||
/**
|
||||
* Retry-exhaustion path: an external writer that changes the file on EVERY attempt (not just
|
||||
* once) exhausts all {@code MAX_TRUST_JSON_CAS_ATTEMPTS} retries. fleetd must then write nothing
|
||||
* at all — not even a partial/best-effort write — and log a WARN naming the file it gave up on.
|
||||
*/
|
||||
@Test
|
||||
void seedTrustDialogWritesNothingAndWarnsWhenCasRetriesAreExhausted(
|
||||
@TempDir Path configDir, @TempDir Path worktree) throws Exception {
|
||||
markAsProvisionedWorktree(worktree);
|
||||
Path claudeJson = configDir.resolve(".claude.json");
|
||||
Files.writeString(claudeJson, "{\"marker\":\"start\"}");
|
||||
|
||||
AtomicInteger hookCalls = new AtomicInteger();
|
||||
ClaudeCodeLauncher.trustJsonCasTestHook = () -> {
|
||||
try {
|
||||
Files.writeString(claudeJson,
|
||||
"{\"marker\":\"race-" + hookCalls.incrementAndGet() + "\"}");
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException(e);
|
||||
}
|
||||
};
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(ClaudeCodeLauncher.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile cfg = trustProfile(configDir.toString(), worktree.toString());
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null)
|
||||
.spawn();
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
ClaudeCodeLauncher.trustJsonCasTestHook = () -> {};
|
||||
}
|
||||
|
||||
assertEquals(5, hookCalls.get(), "the race hook must fire exactly once per CAS attempt");
|
||||
String finalContent = Files.readString(claudeJson);
|
||||
assertEquals("{\"marker\":\"race-" + hookCalls.get() + "\"}", finalContent,
|
||||
"the file must be left exactly as the external writer last left it — fleetd must not "
|
||||
+ "have written at all once retries are exhausted");
|
||||
assertFalse(finalContent.contains("hasTrustDialogAccepted"),
|
||||
"the trust entry must never appear — its presence would mean the CAS gave up and "
|
||||
+ "wrote anyway instead of skipping the seed");
|
||||
|
||||
assertTrue(appender.list.stream().anyMatch(e -> e.getLevel() == Level.WARN
|
||||
&& e.getFormattedMessage().contains("gave up seeding workspace-trust")
|
||||
&& e.getFormattedMessage().contains(worktree.toString())
|
||||
&& e.getFormattedMessage().contains(claudeJson.toString())),
|
||||
"a WARN naming both the cwd and the file it gave up on must be logged: "
|
||||
+ appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList());
|
||||
}
|
||||
|
||||
/**
|
||||
* The loud-default WARN (fleetd #247): a profile with no {@code configDir} targets the
|
||||
* operator's real {@code ~/.claude.json} (redirected here to a {@code @TempDir}), and every time
|
||||
* that happens must be logged, not just detected.
|
||||
*/
|
||||
@Test
|
||||
void seedTrustDialogWarnsEveryTimeItTargetsTheDefaultClaudeJson(
|
||||
@TempDir Path fakeHome, @TempDir Path worktree) throws Exception {
|
||||
markAsProvisionedWorktree(worktree);
|
||||
String originalHome = System.getProperty("user.home");
|
||||
System.setProperty("user.home", fakeHome.toString());
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(ClaudeCodeLauncher.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile cfg = trustProfile(null, worktree.toString());
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null)
|
||||
.spawn();
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
System.setProperty("user.home", originalHome);
|
||||
}
|
||||
|
||||
assertTrue(appender.list.stream().anyMatch(e -> e.getLevel() == Level.WARN
|
||||
&& e.getFormattedMessage().contains("no configDir set")
|
||||
&& e.getFormattedMessage().contains(worktree.toString())),
|
||||
"a WARN naming the cwd must fire when the profile sets no configDir: "
|
||||
+ appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList());
|
||||
}
|
||||
|
||||
// --- fleetd #285: seedTrustDialog under memberHerdrSocket --------------------------------------
|
||||
//
|
||||
// seedTrustDialog gated only on isProvisionedWorktree(cwd) and, being static, could not see
|
||||
// memberHerdrSocketConfigured() at all — unlike its sibling writeCharterFile one method below,
|
||||
// which already refuses the spawn when it cannot place the charter where a different-uid member
|
||||
// can read it. Under memberHerdrSocket + configDir unset, seedTrustDialog wrote fleetd's OWN
|
||||
// ~/.claude.json (the operator's real file) while believing it was seeding the member's. The fix
|
||||
// makes the method an instance method so it can see memberHerdrSocketConfigured(), and applies
|
||||
// the same "refuse, don't silently write somewhere wrong" rule writeCharterFile already uses.
|
||||
|
||||
@Test
|
||||
void seedTrustDialogUnderMemberHerdrSocketSharesTheFileWithTheConfiguredGroup(
|
||||
@TempDir Path configDir, @TempDir Path worktree, @TempDir Path worktreeRoot) throws Exception {
|
||||
markAsProvisionedWorktree(worktree);
|
||||
String group = currentUserGroup();
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// trustProfile always sets a bridge mcpUrl, so a reply charter is generated too, which
|
||||
// means writeCharterFile ALSO runs under memberHerdrSocket and needs its own worktreeRoot —
|
||||
// pass one so this test isolates the trust-seed behaviour instead of tripping that refusal.
|
||||
FleetConfig.Profile cfg = trustProfile(configDir.toString(), worktree.toString());
|
||||
serviceWithConfig(herdr, cfg, () -> configWithMemberHerdrSocket(worktreeRoot.toString(), group)).spawn();
|
||||
|
||||
Path claudeJson = configDir.resolve(".claude.json");
|
||||
assertTrue(Files.exists(claudeJson), "still seeded into <configDir>/.claude.json under memberHerdrSocket");
|
||||
JsonNode project = new ObjectMapper().readTree(claudeJson.toFile())
|
||||
.path("projects").path(worktree.toString());
|
||||
assertTrue(project.path("hasTrustDialogAccepted").asBoolean(false));
|
||||
|
||||
assertEquals("rw-r-----", PosixFilePermissions.toString(Files.getPosixFilePermissions(claudeJson)),
|
||||
"under memberHerdrSocket the file must be shared group-readable, mirroring the "
|
||||
+ "charter file's own per-file mode (fleetd #222) — a 0600 file (Claude "
|
||||
+ "Code's own default) is unreadable by the member's different OS user");
|
||||
String actualGroup = Files.getFileAttributeView(claudeJson, PosixFileAttributeView.class)
|
||||
.readAttributes().group().getName();
|
||||
assertEquals(group, actualGroup, "the file must be chgrp'd to the configured worktreeGroup");
|
||||
}
|
||||
|
||||
/**
|
||||
* The bug itself: {@code memberHerdrSocket} configured, {@code configDir} unset. Before the fix
|
||||
* this wrote fleetd's own default {@code ~/.claude.json} (here redirected to {@code fakeHome} so
|
||||
* a reintroduced bug still cannot touch the real operator file); after the fix it must refuse the
|
||||
* spawn instead, naming {@code configDir} as the missing key, before the member is ever started.
|
||||
*/
|
||||
@Test
|
||||
void seedTrustDialogUnderMemberHerdrSocketRefusesWhenConfigDirUnset(
|
||||
@TempDir Path fakeHome, @TempDir Path worktree) throws Exception {
|
||||
markAsProvisionedWorktree(worktree);
|
||||
String originalHome = System.getProperty("user.home");
|
||||
System.setProperty("user.home", fakeHome.toString());
|
||||
try {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile cfg = trustProfile(null, worktree.toString());
|
||||
ClaudeCodeLauncher launcher = serviceWithConfig(herdr, cfg,
|
||||
() -> configWithMemberHerdrSocket(null, "some-group"));
|
||||
|
||||
IllegalStateException ex = assertThrows(IllegalStateException.class, launcher::spawn,
|
||||
"memberHerdrSocket + no configDir must refuse the spawn, not write fleetd's own "
|
||||
+ "default ~/.claude.json");
|
||||
assertTrue(ex.getMessage().contains("configDir"),
|
||||
"the refusal must name the missing config key — got: " + ex.getMessage());
|
||||
assertFalse(Files.exists(fakeHome.resolve(".claude.json")),
|
||||
"nothing may be written to fleetd's own default home — this is the exact fleetd "
|
||||
+ "#285 defect: writing the operator's own home instead of the member's");
|
||||
assertFalse(herdr.called("agent.start"),
|
||||
"the spawn must be refused BEFORE the member is ever started — got calls: " + herdr.calls);
|
||||
} finally {
|
||||
System.setProperty("user.home", originalHome);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code configDir} alone is not enough — without {@code worktreeGroup} the file fleetd writes
|
||||
* stays {@code 0600} and the member's different OS user still cannot read it, so this must also
|
||||
* refuse, naming {@code worktreeGroup} this time.
|
||||
*/
|
||||
@Test
|
||||
void seedTrustDialogUnderMemberHerdrSocketRefusesWhenWorktreeGroupUnset(
|
||||
@TempDir Path configDir, @TempDir Path worktree) throws Exception {
|
||||
markAsProvisionedWorktree(worktree);
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile cfg = trustProfile(configDir.toString(), worktree.toString());
|
||||
ClaudeCodeLauncher launcher = serviceWithConfig(herdr, cfg,
|
||||
() -> configWithMemberHerdrSocket(null, null));
|
||||
|
||||
IllegalStateException ex = assertThrows(IllegalStateException.class, launcher::spawn,
|
||||
"memberHerdrSocket + no worktreeGroup must refuse the spawn, not write an unreadable file");
|
||||
assertTrue(ex.getMessage().contains("worktreeGroup"),
|
||||
"configDir alone is not enough — got: " + ex.getMessage());
|
||||
assertFalse(Files.exists(configDir.resolve(".claude.json")),
|
||||
"nothing may be written when the file cannot be shared with the member's group");
|
||||
assertFalse(herdr.called("agent.start"),
|
||||
"the spawn must be refused BEFORE the member is ever started");
|
||||
}
|
||||
|
||||
/**
|
||||
* Regression proof: with {@code memberHerdrSocket} ABSENT — even given a LIVE, non-null {@code
|
||||
* config} supplier (not merely {@code config == null}, which every other seedTrustDialog test in
|
||||
* this file already exercises) — the write must stay byte-identical to before this fix: existing
|
||||
* {@code 0600} permissions preserved, no chgrp/chmod attempted. This is the proof the ticket asks
|
||||
* for: the path the whole live fleet uses today is unchanged by this fix.
|
||||
*/
|
||||
@Test
|
||||
void seedTrustDialogPreservesExisting0600PermissionsWhenMemberHerdrSocketAbsentEvenWithALiveConfigSupplier(
|
||||
@TempDir Path configDir, @TempDir Path worktree) throws Exception {
|
||||
markAsProvisionedWorktree(worktree);
|
||||
Path claudeJson = configDir.resolve(".claude.json");
|
||||
Files.writeString(claudeJson, "{}");
|
||||
assumeTrue(Files.getFileAttributeView(claudeJson, PosixFileAttributeView.class) != null,
|
||||
"no POSIX permissions on this filesystem — skipping rather than failing");
|
||||
Files.setPosixFilePermissions(claudeJson, PosixFilePermissions.fromString("rw-------"));
|
||||
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile cfg = trustProfile(configDir.toString(), worktree.toString());
|
||||
FleetConfig config = new FleetConfig(null, null, null, Map.of(), null, null, null, null, null,
|
||||
null, null, null, null, null, null, null, null, null, null, null, null, null).withDefaults();
|
||||
serviceWithConfig(herdr, cfg, () -> config).spawn();
|
||||
|
||||
assertEquals("rw-------", PosixFilePermissions.toString(Files.getPosixFilePermissions(claudeJson)),
|
||||
"with memberHerdrSocket absent — even given a live config supplier — the seed must "
|
||||
+ "stay byte-identical to before this fix: no chgrp/chmod attempted");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -299,6 +299,41 @@ class CompositePeerLauncherTest {
|
||||
"two adapter kinds sharing one daemon keep the fallback route");
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopThroughTheSingleDaemonShortcutStillClosesTheTabWhenTheFallbackDelegateUsesPanePlacement() {
|
||||
// fleetd #342: the single-daemon shortcut (spawnedBy empty, herdrDaemonCount()==1) always
|
||||
// routes stop() through delegates.getFirst() — here the claude adapter, configured for
|
||||
// PANE placement (its own profiles never create a dedicated tab). The pane being torn
|
||||
// down here actually belongs to the opencode adapter's TAB placement, sharing the same
|
||||
// herdr daemon — mixing placements is the point: every existing stop-fallback test in this
|
||||
// class configures BOTH adapters as "tab", so usesTabPlacement() was true either way and
|
||||
// the mis-routing never showed.
|
||||
//
|
||||
// Before the fix, HerdrPeerLauncher#stop gated tab resolution on usesTabPlacement() of the
|
||||
// delegate it happened to be called through, so the wrongly-routed (pane-placement) claude
|
||||
// adapter never even looked for a tab to close, and the now-empty tab leaked with nothing
|
||||
// to reap it. The fix (fleetd #342) resolves the pane's real tab unconditionally, so the
|
||||
// decision follows the pane's actual placement rather than the fallback delegate's static
|
||||
// config.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile claudePane = new FleetConfig.Profile("claude", "http://gx00.gw:8000",
|
||||
"coder", null, "FLEETD_WORKER_TOKEN", List.of("claude"), "pane", "fleetd-workers",
|
||||
"w #{n}", null, null, null);
|
||||
ClaudeCodeLauncher claude = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("claude", claudePane), "claude", _ -> null);
|
||||
PeerLauncher composite = new CompositePeerLauncher(List.of(claude, opencodeAdapter(herdr)), "claude");
|
||||
|
||||
// "w9:pW" was never spawned through this composite instance, so spawnedBy has no entry for
|
||||
// it (the same in-memory-cache-miss shape a daemon restart leaves behind) — stop() falls
|
||||
// through to the single-daemon shortcut and hands it to delegates.getFirst() (claude).
|
||||
composite.stop("w9:pW");
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "the pane itself is still closed on every routed path");
|
||||
assertTrue(herdr.called("tab.close"),
|
||||
"the pane's real (sole-occupant) tab must be closed even though the fallback routed "
|
||||
+ "through a delegate configured for pane placement");
|
||||
}
|
||||
|
||||
@Test
|
||||
void listKeepsBothPanesWhenTwoDaemonsShareAPaneId() {
|
||||
// herdr pane ids are per-daemon counters, so two daemons really can both hold w1:p1 on
|
||||
@@ -615,6 +650,67 @@ class CompositePeerLauncherTest {
|
||||
assertEquals(1, adapter.spawnCount("b"));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #315: {@code CompositePeerLauncher.spawn} rebuilds the {@link PlacementContext} after
|
||||
* every failed attempt "so the policy excludes this profile" (see the comment at the retry call
|
||||
* site) — but {@code FixedPlacementPolicy} never read {@code ctx.unreachable()}, so under the
|
||||
* default {@code fixed} placement every retry re-picked the same dead default and a second,
|
||||
* healthy, configured profile was never tried. This is the same scenario as
|
||||
* {@link #failoverRetriesNextCandidateWhenProfileIsUnreachable}, but pinned to {@code fixed()}
|
||||
* instead of {@code weighted()} — the three existing failover tests all use {@code weighted()},
|
||||
* which is exactly why nobody caught this: the retry loop's contract has no coverage under its
|
||||
* own default policy.
|
||||
*/
|
||||
@Test
|
||||
void failoverRetriesNextCandidateUnderFixedPlacementWhenProfileIsUnreachable() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of("a"));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), _ -> 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(),
|
||||
"fixed placement must fail over from the unreachable default a to the healthy b");
|
||||
assertEquals(1, adapter.spawnCount("a"), "a was tried once and failed");
|
||||
assertEquals(1, adapter.spawnCount("b"), "b was tried once and succeeded");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #315: the same fix — {@code FixedPlacementPolicy} consulting {@code ctx.unreachable()}
|
||||
* — also covers the wiring-bug branch in {@code CompositePeerLauncher.spawn}: a profile that
|
||||
* placement is allowed to choose (it is in the configured candidate list) but that no delegate
|
||||
* declares ({@code byProfile.get(chosen.profile()) == null}). That branch adds the profile to
|
||||
* {@code unreachable} and {@code continue}s without ever calling a launcher, so before this fix
|
||||
* {@code fixed} handed back the same adapterless profile on every remaining attempt too.
|
||||
*/
|
||||
@Test
|
||||
void failoverSkipsAConfiguredProfileNoAdapterDeclaresUnderFixedPlacement() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// Placement's candidate list has three profiles, in this order (LinkedHashMap preserves it,
|
||||
// and the fixed default resolves to the first — see the `ordered` helper's own javadoc).
|
||||
Map<String, FleetConfig.Profile> profiles = new LinkedHashMap<>();
|
||||
profiles.put("c", stubWorker("c"));
|
||||
profiles.put("a", stubWorker("a"));
|
||||
profiles.put("b", stubWorker("b"));
|
||||
// The adapter only declares a and b — c is a configured profile with no owning adapter,
|
||||
// the "wiring bug" the comment in CompositePeerLauncher.spawn calls out.
|
||||
Map<String, FleetConfig.Profile> adapterProfiles = new LinkedHashMap<>();
|
||||
adapterProfiles.put("a", stubWorker("a"));
|
||||
adapterProfiles.put("b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, adapterProfiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), _ -> 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("a", h.profile(),
|
||||
"c has no adapter, so fixed placement must skip it and land on the next candidate, a");
|
||||
assertEquals(0, adapter.spawnCount("c"), "c is never spawned — no adapter owns it");
|
||||
assertEquals(1, adapter.spawnCount("a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnAtMaxLoadThrowsPlacementExceptionNamingProfileLiveAndCap() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
|
||||
@@ -8,6 +8,7 @@ import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
@@ -17,6 +18,7 @@ import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
@@ -45,6 +47,15 @@ class EnvAllowListScrubTest {
|
||||
/** Env var names appearing in command output; anything else (prompts, wrapped lines) is noise. */
|
||||
private static final Pattern ENV_NAME = Pattern.compile("^([A-Za-z_][A-Za-z0-9_]*)$");
|
||||
|
||||
/**
|
||||
* fleetd #388: the sentinel name the generated {@code .zshenv} guard uses, kept here as a
|
||||
* literal rather than referencing {@link EnvAllowListScrub#SCRUB_SENTINEL} — the two tests that
|
||||
* use it must still compile and run against the pre-fix production class (which has no such
|
||||
* constant), so the revert-and-prove-it-fails step exercises a real assertion instead of a
|
||||
* compilation error.
|
||||
*/
|
||||
private static final String SCRUB_SENTINEL_NAME = "_CB633_SCRUBBED";
|
||||
|
||||
/**
|
||||
* The equality test. Expected survivors = baseline exports ∩ allowed — i.e. every survivor is
|
||||
* allowed AND every allowed name that existed survives. The operator's own secret-store exports
|
||||
@@ -108,6 +119,68 @@ class EnvAllowListScrubTest {
|
||||
"allowed N of M with N <= M — the denominator is always reported");
|
||||
}
|
||||
|
||||
/**
|
||||
* A pane inherits {@code UID}; a cleared test parent does not. The scrub must survive it.
|
||||
*
|
||||
* <p>Every other test here starts zsh from a CLEARED environment, so {@code UID} is never an
|
||||
* exported name and never reaches the blanking loop. In a real member pane it is exported and
|
||||
* it IS reached — and {@code export UID=} is a fatal zsh parameter error that aborts the whole
|
||||
* sourced file, leaving every later name unscrubbed and writing no report at all. The abort is
|
||||
* silent: the loop is wrapped in {@code 2>/dev/null}.
|
||||
*
|
||||
* <p>The assertion is deliberately "a report exists" rather than "the canary is blanked". The
|
||||
* report is written by the last statement in the file, so its presence proves the script ran
|
||||
* to completion; the canary alone would depend on where it happens to sit in {@code env} order.
|
||||
* Both are checked, but only the first one fails deterministically without the fix.
|
||||
*/
|
||||
@Test
|
||||
void scrubSurvivesAnInheritedUidTheWayARealPaneHasIt(@TempDir Path tmp) throws Exception {
|
||||
assumeTrue(Files.isExecutable(ZSH), "/bin/zsh not present — nothing to prove here");
|
||||
Set<String> allowed = MemberEnvAllowList.derive(List.of());
|
||||
Path zdotdir = EnvAllowListScrub.generate(tmp, allowed);
|
||||
|
||||
// The production shape: UID present and exported, as every pane shell inherits it.
|
||||
Map<String, String> paneLikeParent = Map.of(
|
||||
"HOME", System.getProperty("user.home"),
|
||||
"PATH", "/usr/bin:/bin",
|
||||
"SHELL", "/bin/zsh",
|
||||
"UID", "1000",
|
||||
"CB633_CANARY", "must-not-survive-the-scrub");
|
||||
Set<String> survivors = exportedNamesFromCleanParent(paneLikeParent, zdotdir);
|
||||
|
||||
EnvAllowListScrub.ScrubReport report = EnvAllowListScrub.readReport(zdotdir);
|
||||
assertNotNull(report,
|
||||
"an inherited UID must not abort the scrub — no report means the file died mid-loop "
|
||||
+ "and every name after UID in `env` order was left unscrubbed");
|
||||
assertFalse(survivors.contains("CB633_CANARY"),
|
||||
"a non-allow-listed name must still be blanked when UID is in the environment");
|
||||
}
|
||||
|
||||
/** A group-shared ZDOTDIR still lets the member truncate and write its pre-created receipt. */
|
||||
@Test
|
||||
void groupSharedScrubWritesAndReadsItsReport(@TempDir Path tmp) throws Exception {
|
||||
assumeTrue(Files.isExecutable(ZSH), "/bin/zsh not present — nothing to prove here");
|
||||
Set<String> allowed = MemberEnvAllowList.derive(List.of());
|
||||
Path zdotdir = EnvAllowListScrub.generate(tmp, allowed, currentUserGroup());
|
||||
|
||||
Map<String, String> cleanParent = Map.of(
|
||||
"HOME", System.getProperty("user.home"),
|
||||
"PATH", "/usr/bin:/bin",
|
||||
"SHELL", "/bin/zsh");
|
||||
exportedNamesFromCleanParent(cleanParent, zdotdir);
|
||||
|
||||
EnvAllowListScrub.ScrubReport report = EnvAllowListScrub.readReport(zdotdir);
|
||||
assertNotNull(report, "a group-shared completed login shell must leave a report behind");
|
||||
assertTrue(report.allowed() >= 0 && report.total() >= report.allowed(),
|
||||
"allowed N of M with N <= M — the denominator is always reported");
|
||||
assertEquals("rw-rw----", java.nio.file.attribute.PosixFilePermissions.toString(
|
||||
Files.getPosixFilePermissions(zdotdir.resolve(EnvAllowListScrub.REPORT_FILE))),
|
||||
"the pre-created receipt must be group-writable");
|
||||
assertEquals("rwxr-x---", java.nio.file.attribute.PosixFilePermissions.toString(
|
||||
Files.getPosixFilePermissions(zdotdir)),
|
||||
"group sharing must not make the ZDOTDIR directory group-writable");
|
||||
}
|
||||
|
||||
/** Report parsing is lenient: absent file → null (no measurement), not an exception. */
|
||||
@Test
|
||||
void readReportReturnsNullForADirectoryWithoutOne(@TempDir Path dir) {
|
||||
@@ -224,6 +297,140 @@ class EnvAllowListScrubTest {
|
||||
+ "shell. A difference here means the scrub is dead on Linux.");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #388: the actual gap. zsh reads {@code .zshenv} always, {@code .zprofile}/
|
||||
* {@code .zlogin} only for a LOGIN shell, and {@code .zshrc} only for an INTERACTIVE one — so a
|
||||
* shell that is NEITHER (a bare {@code /bin/zsh} reading a script off a non-tty stdin, no
|
||||
* {@code -l}, no {@code -i}) reads only {@code .zshenv} and stops. Before this fix, that shell
|
||||
* never reached {@code scrub.zsh} at all: the decoy secret below would survive untouched. This
|
||||
* test injects that decoy directly into the process environment (not via a sourced dotfile,
|
||||
* since the whole point of the gap is that {@code .zshenv} is normally close to empty) so the
|
||||
* test does not depend on any real {@code ~/.zshrc} content existing on the host.
|
||||
*/
|
||||
@Test
|
||||
void scrubRunsInAShellThatIsNeitherLoginNorInteractive(@TempDir Path tmp) throws Exception {
|
||||
assumeTrue(Files.isExecutable(ZSH), "/bin/zsh not present — nothing to prove here");
|
||||
Set<String> allowed = MemberEnvAllowList.derive(List.of());
|
||||
Path zdotdir = EnvAllowListScrub.generate(tmp, allowed);
|
||||
|
||||
Map<String, String> cleanParent = new HashMap<>(Map.of(
|
||||
"HOME", System.getProperty("user.home"),
|
||||
"PATH", "/usr/bin:/bin",
|
||||
"SHELL", "/bin/zsh",
|
||||
"USER", System.getProperty("user.name", "nobody"),
|
||||
"TMPDIR", tmp.toString()));
|
||||
cleanParent.put("FLEETD_TEST_DECOY_SECRET", "x"); // not on any allow-list; must be blanked
|
||||
|
||||
List<String> neither = List.of(); // no -l, no -i; stdin is a pipe (never a tty) either way
|
||||
Set<String> baseline = exportedNamesFromCleanParent(cleanParent, null, neither);
|
||||
Set<String> scrubbed = exportedNamesFromCleanParent(cleanParent, zdotdir, neither);
|
||||
|
||||
Set<String> expected = new TreeSet<>();
|
||||
for (String name : baseline) {
|
||||
if (MemberEnvAllowList.keeps(allowed, name)) {
|
||||
expected.add(name);
|
||||
}
|
||||
}
|
||||
assertTrue(baseline.contains("FLEETD_TEST_DECOY_SECRET"),
|
||||
"sanity: the decoy must actually reach the un-scrubbed baseline, or this test proves "
|
||||
+ "nothing");
|
||||
expected.add("ZDOTDIR"); // the harness set it and it is infrastructure, so it must survive
|
||||
expected.add(SCRUB_SENTINEL_NAME); // set by the new .zshenv guard once scrubbed
|
||||
assertEquals(expected, scrubbed,
|
||||
"a pane shell that is NEITHER login nor interactive must still be scrubbed — its "
|
||||
+ "surviving exported names must EQUAL baseline ∩ allow-list, plus the "
|
||||
+ "sentinel the guard sets once it has run. FLEETD_TEST_DECOY_SECRET surviving "
|
||||
+ "here means the gap is still open.");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #388 invariants 3 and 4, which a name-set equality cannot show: a member's own tooling
|
||||
* forks plain, non-login, non-interactive zsh processes for a single command (the same shape as
|
||||
* the pane shell itself), and such a child must (a) keep whatever its parent deliberately set
|
||||
* for it, never (b) re-run the scrub and blank it, and never (c) overwrite the pane's own
|
||||
* {@code scrub-report.txt} with a description of itself instead of the pane. All three can only
|
||||
* be shown by actually running a child process from within the scrubbed pane shell.
|
||||
*
|
||||
* <p>The pane and the child both report presence via {@code ${NAME:+present}} — empty when a
|
||||
* name is unset OR blanked (exported empty), {@code present} when it is set and non-empty. No
|
||||
* value is ever printed, only these two shapes and the literal word {@code set}/{@code unset}
|
||||
* for the sentinel.
|
||||
*/
|
||||
@Test
|
||||
void neitherShellChildKeepsParentVariablesAndReceiptStillDescribesThePane(@TempDir Path tmp) throws Exception {
|
||||
assumeTrue(Files.isExecutable(ZSH), "/bin/zsh not present — nothing to prove here");
|
||||
Set<String> allowed = MemberEnvAllowList.derive(List.of());
|
||||
Path zdotdir = EnvAllowListScrub.generate(tmp, allowed);
|
||||
|
||||
Map<String, String> paneEnv = new HashMap<>(Map.of(
|
||||
"HOME", System.getProperty("user.home"),
|
||||
"PATH", "/usr/bin:/bin",
|
||||
"SHELL", "/bin/zsh",
|
||||
"USER", System.getProperty("user.name", "nobody"),
|
||||
"TMPDIR", tmp.toString()));
|
||||
paneEnv.put("FLEETD_TEST_DECOY_SECRET", "x"); // not allow-listed; the pane must blank it
|
||||
paneEnv.put("ZDOTDIR", zdotdir.toAbsolutePath().toString());
|
||||
|
||||
// The pane's own script reports what IT sees, then forks a plain non-login, non-interactive
|
||||
// child — the shape a member's own tooling uses — carrying a variable the "parent" (this
|
||||
// pane) deliberately set for it, the way git sets GIT_DIR for a hook.
|
||||
String outerScript = """
|
||||
print -r -- "PANE_SENTINEL=${%1$s:+set}"
|
||||
print -r -- "PANE_DECOY=${FLEETD_TEST_DECOY_SECRET:+present}"
|
||||
FLEETD_TEST_TOOL_VAR=keep /bin/zsh <<'CHILD'
|
||||
print -r -- "CHILD_LOGIN=$([[ -o login ]] && echo yes || echo no)"
|
||||
print -r -- "CHILD_INTERACTIVE=$([[ -o interactive ]] && echo yes || echo no)"
|
||||
print -r -- "CHILD_TOOL_VAR=${FLEETD_TEST_TOOL_VAR:+present}"
|
||||
print -r -- "CHILD_DECOY=${FLEETD_TEST_DECOY_SECRET:+present}"
|
||||
print -r -- "CHILD_SENTINEL=${%1$s:+set}"
|
||||
CHILD
|
||||
exit
|
||||
""".formatted(SCRUB_SENTINEL_NAME);
|
||||
|
||||
ProcessBuilder pb = new ProcessBuilder("/bin/zsh"); // no -l, no -i: the pane's own shape
|
||||
pb.environment().clear();
|
||||
pb.environment().putAll(paneEnv);
|
||||
pb.redirectError(ProcessBuilder.Redirect.DISCARD);
|
||||
Process zsh = pb.start();
|
||||
zsh.getOutputStream().write(outerScript.getBytes(StandardCharsets.UTF_8));
|
||||
zsh.getOutputStream().flush();
|
||||
zsh.getOutputStream().close();
|
||||
String stdout = new String(zsh.getInputStream().readAllBytes(), StandardCharsets.UTF_8);
|
||||
assertTrue(zsh.waitFor(60, java.util.concurrent.TimeUnit.SECONDS),
|
||||
"the pane+child probe did not exit within 60s");
|
||||
assertTrue(zsh.exitValue() == 0, "probe zsh exited non-zero: " + stdout);
|
||||
|
||||
Map<String, String> reported = new HashMap<>();
|
||||
for (String line : stdout.split("\n")) {
|
||||
int eq = line.indexOf('=');
|
||||
if (eq > 0) {
|
||||
reported.put(line.substring(0, eq).trim(), line.substring(eq + 1).trim());
|
||||
}
|
||||
}
|
||||
|
||||
assertEquals("set", reported.get("PANE_SENTINEL"),
|
||||
"the pane itself is neither login nor interactive, so the .zshenv guard must have "
|
||||
+ "run the scrub and exported the sentinel");
|
||||
assertEquals("", reported.get("PANE_DECOY"),
|
||||
"the pane must blank a non-allow-listed name — invariant 1");
|
||||
assertEquals("no", reported.get("CHILD_LOGIN"), "sanity: the child must also be non-login");
|
||||
assertEquals("no", reported.get("CHILD_INTERACTIVE"), "sanity: the child must also be non-interactive");
|
||||
assertEquals("present", reported.get("CHILD_TOOL_VAR"),
|
||||
"invariant 3: a variable the pane deliberately set for its child must survive — a "
|
||||
+ "child that re-ran the scrub would have blanked it");
|
||||
assertEquals("", reported.get("CHILD_DECOY"),
|
||||
"a name already blanked by the pane must stay blanked in the child, never resurrected");
|
||||
assertEquals("set", reported.get("CHILD_SENTINEL"),
|
||||
"the child must inherit the sentinel from the pane's environment, or it would re-scrub");
|
||||
|
||||
EnvAllowListScrub.ScrubReport report = EnvAllowListScrub.readReport(zdotdir);
|
||||
assertNotNull(report, "the pane's own scrub pass must leave a report behind");
|
||||
assertTrue(report.blanked().stream().noneMatch(n -> n.startsWith("FLEETD_TEST_TOOL_VAR")),
|
||||
"invariant 4: the receipt must still describe the PANE, not the child — a child that "
|
||||
+ "re-ran the scrub would have rewritten this file to list its own "
|
||||
+ "FLEETD_TEST_TOOL_VAR as blanked");
|
||||
}
|
||||
|
||||
/**
|
||||
* Run {@code /bin/zsh -l -i} from a clean parent and return the NAMES it has exported by prompt
|
||||
* time. With {@code zdotdir} non-null, {@code ZDOTDIR} points at a generated scrub directory, so
|
||||
|
||||
+303
-41
@@ -23,12 +23,14 @@ import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
import static org.junit.jupiter.api.Assumptions.assumeTrue;
|
||||
|
||||
@@ -99,18 +101,26 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
}
|
||||
|
||||
/**
|
||||
* A non-zsh shell cannot read {@code ZDOTDIR} at all. The launcher must fall back rather than
|
||||
* generate a directory nothing will ever read — a directory that would look like protection.
|
||||
* fleetd #155: a non-zsh shell cannot read {@code ZDOTDIR} at all, so under {@code
|
||||
* policy: allow-list} — the policy the operator picked specifically for a control a sourced file
|
||||
* cannot undo — the launcher must REFUSE the spawn rather than silently generate a directory
|
||||
* nothing will ever read (protection theatre) or fall back to the weaker overlay (exactly the
|
||||
* "control silently does nothing" defect this ticket exists to close). Real path: through {@link
|
||||
* HerdrPeerLauncher#spawn}, the method the daemon actually calls.
|
||||
*/
|
||||
@Test
|
||||
void aNonZshShellGeneratesNothingAndFallsBack() {
|
||||
void aNonZshShellUnderAllowListPolicyRefusesTheSpawn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/bash");
|
||||
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class,
|
||||
() -> launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV)),
|
||||
"policy=allow-list on a non-zsh shell must refuse the spawn, not silently degrade");
|
||||
|
||||
assertTrue(e.getMessage().contains("allow-list") && e.getMessage().contains("/bin/bash"),
|
||||
"the refusal must name the policy and the actual shell, got: " + e.getMessage());
|
||||
assertFalse(launcher.env.containsKey("ZDOTDIR"),
|
||||
"bash ignores ZDOTDIR; setting it would be protection theatre");
|
||||
"a refused spawn must not have generated (or wired in) a scrub directory: " + launcher.env);
|
||||
}
|
||||
|
||||
private static Supplier<FleetConfig.MemberCredentials> allowList() {
|
||||
@@ -161,10 +171,10 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
/**
|
||||
* {@code SSH_AUTH_SOCK} is a live ssh-agent handle, not a value — it must stay blocked under
|
||||
* {@code allow-list} even when the operator lists it under {@code allow:}, because {@code
|
||||
* sshAuthSock} defaults to blocked. Governed ONLY by {@code memberCredentials.sshAuthSock}.
|
||||
* sshAgentEnv} defaults to omit. Governed ONLY by {@code memberCredentials.sshAgentEnv}.
|
||||
*/
|
||||
@Test
|
||||
void sshAuthSockStaysBlockedEvenWhenListedInMemberCredentialsAllow() {
|
||||
void sshAgentEnvStaysOmittedEvenWhenListedInMemberCredentialsAllow() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
WiringLauncher launcher = new WiringLauncher(herdr,
|
||||
allowListWithAllow(List.of("SSH_AUTH_SOCK")));
|
||||
@@ -175,7 +185,7 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
String scrub = readAll(dir.resolve(EnvAllowListScrub.SCRUB_FILE));
|
||||
assertFalse(scrub.contains("'SSH_AUTH_SOCK'"),
|
||||
"SSH_AUTH_SOCK must not be on the derived allow-list just because the operator put "
|
||||
+ "it under allow: — sshAuthSock is unset here, so it defaults to block");
|
||||
+ "it under allow: — sshAgentEnv is unset here, so it defaults to omit");
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -241,16 +251,70 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
+ appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #269 follow-up: the same overclaim the WARN in {@code logCredentialGap} was fixed for,
|
||||
* in the INFO line beside it. With {@code memberHerdrSocket} configured, member panes are routed
|
||||
* to a second herdr whose environment fleetd has no channel to inspect, so the counts come from
|
||||
* fleetd's OWN environment. The bare line "member credentials: allowed 1 of 3" reads as a fact
|
||||
* about the member's pane, and there it is not one.
|
||||
*
|
||||
* <p>#269 reworded four sites and stopped at the sibling below; this pins the pair together so
|
||||
* a future edit cannot fix one and leave the other. Real path: asserted after a real {@link
|
||||
* HerdrPeerLauncher#spawn}, reading the log production actually emits.
|
||||
*/
|
||||
@Test
|
||||
void theAllowedCountLineSaysWhoseEnvironmentItCountedWhenMemberHerdrSocketIsSet(@TempDir Path worktreeRoot)
|
||||
throws IOException {
|
||||
String group = currentUserGroup();
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Set<String> hostEnvNames = Set.of(INJECTED, "SOME_UNRELATED_NAME", "ANOTHER_UNRELATED_NAME");
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/bash-should-be-ignored",
|
||||
() -> hostEnvNames,
|
||||
() -> configWithMemberHerdrSocketRootAndGroup("/tmp/other-user-herdr.sock", "/bin/zsh",
|
||||
worktreeRoot.toString(), group));
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
Level original = logger.getLevel();
|
||||
logger.setLevel(Level.INFO);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(original);
|
||||
}
|
||||
|
||||
List<String> lines = appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList();
|
||||
String coverage = lines.stream()
|
||||
.filter(l -> l.startsWith("member credentials: allowed "))
|
||||
.findFirst()
|
||||
.orElse(null);
|
||||
assertNotNull(coverage, "the coverage line must still be logged — narrowing the claim must "
|
||||
+ "not silently delete the line: " + lines);
|
||||
assertTrue(coverage.contains("fleetd's OWN environment"),
|
||||
"the line must say whose environment it counted: " + coverage);
|
||||
assertTrue(coverage.contains("NOT the member pane's"),
|
||||
"and must say plainly that it is not the member's: " + coverage);
|
||||
// The counts themselves stay real — narrowing the claim must not turn them into constants.
|
||||
assertTrue(coverage.startsWith("member credentials: allowed 1 of 3"),
|
||||
"the real counts must survive the rewording: " + coverage);
|
||||
}
|
||||
|
||||
/**
|
||||
* Lead-review fix: on a NON-zsh shell no scrub ever runs (bash ignores {@code ZDOTDIR}), so the
|
||||
* "allowed N of M" line — which describes what the scrub does — must not be printed there either.
|
||||
* Before this fix the line was logged BEFORE the zsh gate, so a non-zsh host printed e.g.
|
||||
* "allowed 1 of 3" while blocking nothing at all, telling an operator a control ran when it did
|
||||
* not. Real path: goes through {@link HerdrPeerLauncher#spawn}, same as the sibling test above,
|
||||
* with the shell fixed to bash so the fallback branch is the one exercised.
|
||||
* not. fleetd #155: the spawn itself is now refused on this path (see {@code
|
||||
* aNonZshShellUnderAllowListPolicyRefusesTheSpawn}) rather than falling back — this test's own
|
||||
* concern still holds under the refusal: the "allowed N of M" line describes a scrub that never
|
||||
* ran here, so it must still never appear. Real path: goes through {@link
|
||||
* HerdrPeerLauncher#spawn}, same as the sibling test above, with the shell fixed to bash.
|
||||
*/
|
||||
@Test
|
||||
void noAllowedCountLineIsEmittedOnTheNonZshFallbackPath() {
|
||||
void noAllowedCountLineIsEmittedOnTheNonZshRefusalPath() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Set<String> hostEnvNames = Set.of(INJECTED, "SOME_UNRELATED_NAME", "ANOTHER_UNRELATED_NAME");
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/bash", () -> hostEnvNames);
|
||||
@@ -262,7 +326,9 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV)),
|
||||
"policy=allow-list on a non-zsh shell must refuse the spawn");
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(original);
|
||||
@@ -305,18 +371,180 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
"expected the pre-existing 'scrub blanks them' INFO unchanged, got: " + messages);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #341: {@code unprotectedGapLogged} guarded TWO WARN branches that name DIFFERENT env
|
||||
* var names — the allow-list branch ({@code keptByDerivedList}, below) and the deny-by-default
|
||||
* / non-zsh-fallback branch ({@link HerdrPeerLauncher#warnGapUnprotected}). {@code
|
||||
* memberCredentials} is a live, re-read-per-spawn supplier, so the policy can change between
|
||||
* two spawns on the same launcher instance — a config reload needs no restart. Spawn 1 runs
|
||||
* under {@code deny-by-default} with a gap of {@code SPAWN_ONE_UNCOVERED_TOKEN}, which trips
|
||||
* the (before this fix) SHARED one-shot flag. The policy is then reloaded to {@code
|
||||
* allow-list}; spawn 2's gap is {@code FLEETD_WORKER_TOKEN} instead — the test profile's own
|
||||
* {@code tokenEnv}, which the derived allow-list keeps even though it is on neither {@code
|
||||
* known:} nor {@code allow:}, so it is genuinely unprotected and deserves its own WARN. Before
|
||||
* this fix that WARN never fires, because the shared flag was already {@code true} — the
|
||||
* operator is never told {@code FLEETD_WORKER_TOKEN} reaches every member pane unblocked. Real
|
||||
* path: two real {@link HerdrPeerLauncher#spawn} calls on ONE launcher instance, with mutable
|
||||
* {@code memberCredentials}/host-env suppliers standing in for a live config reload between
|
||||
* spawns.
|
||||
*/
|
||||
@Test
|
||||
void aDifferentUnprotectedGapOnALaterSpawnIsNotSuppressedByAnEarlierSpawnsWarn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AtomicReference<FleetConfig.MemberCredentials> credsState = new AtomicReference<>(
|
||||
new FleetConfig.MemberCredentials(null, List.of(), List.of(), null)); // deny-by-default
|
||||
AtomicReference<Set<String>> hostEnvState =
|
||||
new AtomicReference<>(Set.of("SPAWN_ONE_UNCOVERED_TOKEN"));
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, credsState::get, "/bin/zsh", hostEnvState::get);
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
Level original = logger.getLevel();
|
||||
logger.setLevel(Level.WARN);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
// Spawn 1: deny-by-default, gap = {SPAWN_ONE_UNCOVERED_TOKEN} — the effectiveAllowed ==
|
||||
// null branch, via warnGapUnprotected.
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
// Live policy reload to allow-list, with a DIFFERENT gap name.
|
||||
credsState.set(new FleetConfig.MemberCredentials(
|
||||
FleetConfig.MemberCredentials.POLICY_ALLOW_LIST, List.of(), List.of(), null));
|
||||
hostEnvState.set(Set.of("FLEETD_WORKER_TOKEN"));
|
||||
// Spawn 2: allow-list, gap = {FLEETD_WORKER_TOKEN} — kept by the derived allow-list
|
||||
// (the profile's own tokenEnv), so it is the effectiveAllowed != null / keptByDerivedList
|
||||
// branch, at the SAME log line HerdrPeerLauncher:1819 guards with the shared flag.
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(original);
|
||||
}
|
||||
|
||||
List<String> messages = appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList();
|
||||
assertTrue(messages.stream().anyMatch(
|
||||
m -> m.contains("UNBLOCKED") && m.contains("SPAWN_ONE_UNCOVERED_TOKEN")),
|
||||
"spawn 1's deny-by-default gap must still warn — got: " + messages);
|
||||
assertTrue(messages.stream().anyMatch(
|
||||
m -> m.contains("UNBLOCKED") && m.contains("FLEETD_WORKER_TOKEN")),
|
||||
"spawn 2's gap names a DIFFERENT env var than spawn 1 (FLEETD_WORKER_TOKEN, not "
|
||||
+ "SPAWN_ONE_UNCOVERED_TOKEN) — it must still be warned about even though a "
|
||||
+ "flag already fired once for spawn 1's unrelated name. Before fleetd #341's "
|
||||
+ "fix this WARN never fires because unprotectedGapLogged was already true. "
|
||||
+ "Got: " + messages);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #341 follow-up: the OTHER half of the guard's contract. The set exists to report every
|
||||
* distinct name, but it must still report each one only ONCE — the noise control is the reason
|
||||
* a guard is here at all, and the ticket named it as invariant 1. Two spawns, same policy, same
|
||||
* gap name: exactly one WARN mentioning it.
|
||||
*
|
||||
* <p>Measured before this test existed: replacing {@code .filter(unprotectedGapNamesWarned::add)}
|
||||
* with a filter that adds and always returns {@code true} — so every name is logged on every
|
||||
* spawn — left all 1358 tests green. The fix was correct and nothing held it there. That is the
|
||||
* "a test on the seam does not prove the caller" shape: the {@code Set} behaves, and nothing
|
||||
* proved this class used it as a guard rather than as a record.
|
||||
*/
|
||||
@Test
|
||||
void theSameUnprotectedNameIsWarnedAboutOnlyOnceAcrossSpawns() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AtomicReference<FleetConfig.MemberCredentials> credsState = new AtomicReference<>(
|
||||
new FleetConfig.MemberCredentials(null, List.of(), List.of(), null)); // deny-by-default
|
||||
AtomicReference<Set<String>> hostEnvState =
|
||||
new AtomicReference<>(Set.of("REPEATED_UNCOVERED_TOKEN"));
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, credsState::get, "/bin/zsh", hostEnvState::get);
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
Level original = logger.getLevel();
|
||||
logger.setLevel(Level.WARN);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
// Same policy, same gap, second spawn. Nothing new to tell the operator.
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(original);
|
||||
}
|
||||
|
||||
List<String> messages = appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList();
|
||||
long mentioning = messages.stream()
|
||||
.filter(m -> m.contains("REPEATED_UNCOVERED_TOKEN"))
|
||||
.count();
|
||||
assertEquals(1, mentioning,
|
||||
"one unchanged unprotected name across two spawns must produce exactly one WARN — "
|
||||
+ "the set is a guard, not just a record. Got: " + messages);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #341 follow-up: the reverse policy order. The original defect was found going
|
||||
* deny-by-default then allow-list, and a guard that is fixed in one direction is not
|
||||
* necessarily fixed in the other — "ask which states still OPEN the gate". Here spawn 1 runs
|
||||
* under {@code allow-list} (the {@code keptByDerivedList} branch) and spawn 2 under
|
||||
* {@code deny-by-default} ({@link HerdrPeerLauncher#warnGapUnprotected}), with a different name
|
||||
* each time. Both must be reported.
|
||||
*/
|
||||
@Test
|
||||
void anAllowListWarnDoesNotSuppressALaterDenyByDefaultWarnForADifferentName() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AtomicReference<FleetConfig.MemberCredentials> credsState = new AtomicReference<>(
|
||||
new FleetConfig.MemberCredentials(
|
||||
FleetConfig.MemberCredentials.POLICY_ALLOW_LIST, List.of(), List.of(), null));
|
||||
AtomicReference<Set<String>> hostEnvState =
|
||||
new AtomicReference<>(Set.of("FLEETD_WORKER_TOKEN"));
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, credsState::get, "/bin/zsh", hostEnvState::get);
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
Level original = logger.getLevel();
|
||||
logger.setLevel(Level.WARN);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
// Spawn 1: allow-list, gap kept by the derived list — the keptByDerivedList WARN.
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
// Live reload the OTHER way: back to deny-by-default, with a different name.
|
||||
credsState.set(new FleetConfig.MemberCredentials(null, List.of(), List.of(), null));
|
||||
hostEnvState.set(Set.of("LATER_UNCOVERED_TOKEN"));
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(original);
|
||||
}
|
||||
|
||||
List<String> messages = appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList();
|
||||
assertTrue(messages.stream().anyMatch(
|
||||
m -> m.contains("UNBLOCKED") && m.contains("FLEETD_WORKER_TOKEN")),
|
||||
"spawn 1's allow-list gap must warn — got: " + messages);
|
||||
assertTrue(messages.stream().anyMatch(
|
||||
m -> m.contains("UNBLOCKED") && m.contains("LATER_UNCOVERED_TOKEN")),
|
||||
"spawn 2's deny-by-default gap names a different variable and must still be warned "
|
||||
+ "about, even though an allow-list WARN already fired. Got: " + messages);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #185 stage 2: with {@code memberHerdrSocket:} configured, member panes run under a
|
||||
* different OS user — {@link HerdrPeerLauncher#hostEnvNames} describes fleetd's own process, not
|
||||
* that user's. Neither "inherits them UNBLOCKED" nor "scrub blanks them" is evidence-backed
|
||||
* there, so neither may print; the single unknown-environment WARN must, naming the config key.
|
||||
*
|
||||
* <p>fleetd #155: {@code memberLoginShell} is now given explicitly as zsh, so the spawn reaches
|
||||
* this WARN through the (still-degrading, not refusing) missing-{@code worktreeRoot}/{@code
|
||||
* worktreeGroup} fallback rather than through the zsh gate, which now refuses instead of falling
|
||||
* back — see {@code aNonZshShellUnderAllowListPolicyRefusesTheSpawn}. This test's own concern
|
||||
* (the unknown-environment WARN) is orthogonal to which fallback reached {@code
|
||||
* logCredentialGap}, so it still holds.
|
||||
*/
|
||||
@Test
|
||||
void gapDetectorReportsUnknownInsteadOfAConclusionWhenMemberHerdrSocketIsConfigured() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Set<String> hostEnvNames = Set.of("FLEETD_WORKER_TOKEN", "SOME_UNKNOWN_SECRET_TOKEN");
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/zsh", () -> hostEnvNames,
|
||||
() -> configWithMemberHerdrSocket("/tmp/other-user-herdr.sock"));
|
||||
() -> configWithMemberHerdrSocket("/tmp/other-user-herdr.sock", "/bin/zsh"));
|
||||
|
||||
List<String> messages = spawnAndCaptureLogs(launcher);
|
||||
|
||||
@@ -340,8 +568,9 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
void theUnknownEnvironmentWarnFiresOnceNotOncePerSpawn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Set<String> hostEnvNames = Set.of("FLEETD_WORKER_TOKEN", "SOME_UNKNOWN_SECRET_TOKEN");
|
||||
// fleetd #155: memberLoginShell given explicitly as zsh — see the sibling test above for why.
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/zsh", () -> hostEnvNames,
|
||||
() -> configWithMemberHerdrSocket("/tmp/other-user-herdr.sock"));
|
||||
() -> configWithMemberHerdrSocket("/tmp/other-user-herdr.sock", "/bin/zsh"));
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
Level original = logger.getLevel();
|
||||
@@ -365,6 +594,33 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
+ appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #184 item 5: the unknown-environment WARN must state the honest reason for the
|
||||
* UNKNOWN conclusion — fleetd has no channel to confirm what OS user the second herdr runs
|
||||
* as — and must NOT assert as fact that member panes run under a different OS user just
|
||||
* because {@code memberHerdrSocket} is configured. An operator may point it at a second herdr
|
||||
* running as the SAME user, for pane isolation alone; in that case members DO inherit fleetd's
|
||||
* environment, and asserting otherwise would tell the operator to disregard a real, known gap.
|
||||
*/
|
||||
@Test
|
||||
void unknownEnvironmentWarnStatesUncertaintyNotAnAssertedDifferentUser() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Set<String> hostEnvNames = Set.of("FLEETD_WORKER_TOKEN", "SOME_UNKNOWN_SECRET_TOKEN");
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/zsh", () -> hostEnvNames,
|
||||
() -> configWithMemberHerdrSocket("/tmp/other-user-herdr.sock", "/bin/zsh"));
|
||||
|
||||
List<String> messages = spawnAndCaptureLogs(launcher);
|
||||
|
||||
assertTrue(messages.stream().anyMatch(m -> m.contains("memberHerdrSocket")
|
||||
&& m.contains("no channel to confirm what OS user that herdr runs as")),
|
||||
"expected the WARN to name the actual uncertainty (no channel to confirm the "
|
||||
+ "herdr's uid), got: " + messages);
|
||||
assertFalse(messages.stream().anyMatch(m -> m.contains("member panes run under a different OS "
|
||||
+ "user than fleetd's own process")),
|
||||
"the WARN must not assert as fact that members run under a different OS user just "
|
||||
+ "because memberHerdrSocket is configured — got: " + messages);
|
||||
}
|
||||
|
||||
/**
|
||||
* Hard constraint: the gap detector must never log an env var VALUE, only its NAME. {@code
|
||||
* SOME_UNKNOWN_SECRET_TOKEN} resolves to a distinctive canary value through the same {@code env}
|
||||
@@ -389,42 +645,44 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213 defect 1, acceptance criterion 1: {@code memberHerdrSocket} configured and {@code
|
||||
* memberLoginShell} configured as non-zsh must fall back to the sentinel overlay exactly like a
|
||||
* non-zsh {@code $SHELL} does today — and the "generated ZDOTDIR" INFO must not appear, since no
|
||||
* scrub actually runs. The WiringLauncher's own {@code env("SHELL")} is deliberately set to
|
||||
* {@code /bin/zsh} — the OPPOSITE of what {@code memberLoginShell} says — so a launcher that
|
||||
* (incorrectly) fell back to fleetd's own {@code $SHELL} here would wrongly pass the gate and
|
||||
* fail this test.
|
||||
* fleetd #213 defect 1, acceptance criterion 1 — updated by fleetd #155: {@code
|
||||
* memberHerdrSocket} configured and {@code memberLoginShell} configured as non-zsh must now
|
||||
* REFUSE the spawn (not fall back to the sentinel overlay — see {@code
|
||||
* aNonZshShellUnderAllowListPolicyRefusesTheSpawn}'s javadoc for why). The WiringLauncher's own
|
||||
* {@code env("SHELL")} is deliberately set to {@code /bin/zsh} — the OPPOSITE of what {@code
|
||||
* memberLoginShell} says — so a launcher that (incorrectly) fell back to fleetd's own {@code
|
||||
* $SHELL} here would wrongly pass the gate and fail this test.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketWithNonZshMemberLoginShellFallsBackToTheOverlay() {
|
||||
void memberHerdrSocketWithNonZshMemberLoginShellRefusesTheSpawn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowListWithKnown(List.of("SOME_TOKEN")),
|
||||
"/bin/zsh", null,
|
||||
() -> configWithMemberHerdrSocket("/tmp/other-user-herdr.sock", "/bin/bash"));
|
||||
|
||||
List<String> messages = spawnAndCaptureLogs(launcher);
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class,
|
||||
() -> launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV)),
|
||||
"a non-zsh configured memberLoginShell under policy=allow-list must refuse the spawn");
|
||||
|
||||
assertEquals("blocked-by-fleetd-cb596-see-gitea-issue-82", launcher.env.get("SOME_TOKEN"),
|
||||
"a non-zsh memberLoginShell must fall back to the CB-596 sentinel overlay, exactly "
|
||||
+ "like a non-zsh $SHELL does when memberHerdrSocket is absent");
|
||||
assertTrue(e.getMessage().contains("/bin/bash"),
|
||||
"the refusal must name the configured memberLoginShell, got: " + e.getMessage());
|
||||
assertFalse(launcher.env.containsKey("SOME_TOKEN"),
|
||||
"a refused spawn must not have touched the pane env at all: " + launcher.env);
|
||||
assertFalse(launcher.env.containsKey("ZDOTDIR"),
|
||||
"no scrub directory may be generated when the configured member login shell is not zsh");
|
||||
assertFalse(messages.stream().anyMatch(m -> m.contains("generated ZDOTDIR")),
|
||||
"the 'generated ZDOTDIR' INFO must not appear when the scrub never runs — got: " + messages);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213 defect 1, acceptance criterion 2: {@code memberHerdrSocket} configured and NO
|
||||
* {@code memberLoginShell} configured must fall back exactly like criterion 1 above — AND
|
||||
* fleetd's own {@code $SHELL} must never even be consulted (not merely "not decisive"). The
|
||||
* fixture's {@code env} function reports {@code /bin/zsh} for {@code SHELL} — a value that would
|
||||
* WRONGLY pass the zsh gate if the fix regressed to reading it — while flagging whether it was
|
||||
* ever asked for at all, so this test fails loudly on either kind of regression.
|
||||
* fleetd #213 defect 1, acceptance criterion 2 — updated by fleetd #155: {@code
|
||||
* memberHerdrSocket} configured and NO {@code memberLoginShell} configured must now REFUSE the
|
||||
* spawn exactly like criterion 1 above — AND fleetd's own {@code $SHELL} must never even be
|
||||
* consulted (not merely "not decisive"). The fixture's {@code env} function reports {@code
|
||||
* /bin/zsh} for {@code SHELL} — a value that would WRONGLY pass the zsh gate if the fix
|
||||
* regressed to reading it — while flagging whether it was ever asked for at all, so this test
|
||||
* fails loudly on either kind of regression.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketWithNoMemberLoginShellFallsBackAndNeverConsultsFleetdsOwnShell() {
|
||||
void memberHerdrSocketWithNoMemberLoginShellRefusesAndNeverConsultsFleetdsOwnShell() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AtomicBoolean shellQueried = new AtomicBoolean(false);
|
||||
Function<String, String> env = name -> {
|
||||
@@ -437,15 +695,18 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowListWithKnown(List.of("SOME_TOKEN")), env,
|
||||
() -> configWithMemberHerdrSocket("/tmp/other-user-herdr.sock", null));
|
||||
|
||||
List<String> messages = spawnAndCaptureLogs(launcher);
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV)),
|
||||
"no configured memberLoginShell under policy=allow-list must refuse the spawn, same "
|
||||
+ "as an explicit non-zsh one");
|
||||
|
||||
assertFalse(shellQueried.get(), "fleetd's own $SHELL must never be consulted once "
|
||||
+ "memberHerdrSocket is configured — only memberLoginShell: may decide the gate");
|
||||
assertEquals("blocked-by-fleetd-cb596-see-gitea-issue-82", launcher.env.get("SOME_TOKEN"),
|
||||
"no memberLoginShell configured must fall back to the sentinel overlay, same as a "
|
||||
assertFalse(launcher.env.containsKey("SOME_TOKEN"),
|
||||
"a refused spawn must not have touched the pane env at all, same as a "
|
||||
+ "configured non-zsh shell");
|
||||
assertFalse(messages.stream().anyMatch(m -> m.contains("generated ZDOTDIR")),
|
||||
"no scrub may run without a configured memberLoginShell — got: " + messages);
|
||||
assertFalse(launcher.env.containsKey("ZDOTDIR"),
|
||||
"no scrub may run without a configured memberLoginShell");
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -700,7 +961,8 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
/**
|
||||
* fleetd #213: as above, plus {@code worktreeRoot:}/{@code worktreeGroup:} — both required for
|
||||
* the ZDOTDIR scrub to run at all once {@code memberHerdrSocket} is configured; either missing
|
||||
* falls back to the sentinel overlay, same as a non-zsh {@code memberLoginShell}.
|
||||
* falls back to the sentinel overlay (fleetd #155 left this branch alone — it is not a shell
|
||||
* problem, so it is not this ticket's refusal).
|
||||
*/
|
||||
private static FleetConfig configWithMemberHerdrSocketRootAndGroup(String memberHerdrSocket,
|
||||
String memberLoginShell, String worktreeRoot, String worktreeGroup) {
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #111 (CB-608): {@link MemberCredentialPolicyView} is the one place that turns a {@code
|
||||
* memberCredentials:} policy into names-and-counts, so the startup log line and {@code
|
||||
* GET /member-credentials} cannot drift apart. These tests pin: the counts always match the
|
||||
* policy that produced them, an absent/empty policy is represented honestly (never as "nothing
|
||||
* blocked"), and the view carries names only — no value ever flows through it, because it is
|
||||
* built only from {@link FleetConfig.MemberCredentials}, which itself never holds a value.
|
||||
*/
|
||||
class MemberCredentialPolicyViewTest {
|
||||
|
||||
@Test
|
||||
void nullPolicyIsAbsentNotClean() {
|
||||
MemberCredentialPolicyView view = MemberCredentialPolicyView.of(null);
|
||||
|
||||
assertFalse(view.present(), "a null policy must be reported as absent");
|
||||
assertEquals(0, view.knownCount());
|
||||
assertEquals(0, view.allowedCount());
|
||||
assertEquals(0, view.blockedCount());
|
||||
assertTrue(view.known().isEmpty());
|
||||
assertTrue(view.allowed().isEmpty());
|
||||
assertTrue(view.blocked().isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void emptyKnownListIsAbsentEvenWithAPolicyModeSet() {
|
||||
// A memberCredentials: block can be present in YAML with policy: set but known: empty —
|
||||
// that must still read as "no policy configured", the same as a fully absent block,
|
||||
// because zero known names means the daemon blocks nothing either way.
|
||||
FleetConfig.MemberCredentials creds =
|
||||
new FleetConfig.MemberCredentials("deny-by-default", List.of(), List.of());
|
||||
|
||||
MemberCredentialPolicyView view = MemberCredentialPolicyView.of(creds);
|
||||
|
||||
assertFalse(view.present());
|
||||
assertEquals(0, view.knownCount());
|
||||
}
|
||||
|
||||
@Test
|
||||
void countsMatchARealPolicyExactly() {
|
||||
FleetConfig.MemberCredentials creds = new FleetConfig.MemberCredentials(
|
||||
"deny-by-default",
|
||||
List.of("AI_GATEWAY_TOKEN"),
|
||||
List.of("AI_GATEWAY_TOKEN", "GITEA_ACCESS_TOKEN", "WORKER_GITEA_TOKEN"));
|
||||
|
||||
MemberCredentialPolicyView view = MemberCredentialPolicyView.of(creds);
|
||||
|
||||
assertTrue(view.present());
|
||||
assertEquals("deny-by-default", view.policy());
|
||||
assertEquals(List.of("AI_GATEWAY_TOKEN", "GITEA_ACCESS_TOKEN", "WORKER_GITEA_TOKEN"), view.known());
|
||||
assertEquals(List.of("AI_GATEWAY_TOKEN"), view.allowed());
|
||||
assertEquals(3, view.knownCount());
|
||||
assertEquals(1, view.allowedCount());
|
||||
// known minus allowed — the two names actually shadowed on a spawn.
|
||||
assertEquals(2, view.blockedCount());
|
||||
assertTrue(view.blocked().containsAll(List.of("GITEA_ACCESS_TOKEN", "WORKER_GITEA_TOKEN")));
|
||||
}
|
||||
|
||||
@Test
|
||||
void presentPolicyThatBlocksNothingIsStillDistinctFromAbsent() {
|
||||
// known == allow => blockedCount is 0, exactly like an absent policy's blockedCount — the
|
||||
// two must still be told apart by `present`, or a reader cannot tell "policy configured,
|
||||
// nothing currently blocked" from "no policy at all".
|
||||
FleetConfig.MemberCredentials creds = new FleetConfig.MemberCredentials(
|
||||
"deny-by-default", List.of("X"), List.of("X"));
|
||||
|
||||
MemberCredentialPolicyView view = MemberCredentialPolicyView.of(creds);
|
||||
|
||||
assertTrue(view.present());
|
||||
assertEquals(1, view.knownCount());
|
||||
assertEquals(0, view.blockedCount());
|
||||
assertFalse(MemberCredentialPolicyView.absent().present());
|
||||
}
|
||||
|
||||
@Test
|
||||
void namesPassThroughUnchangedNeverAValue() {
|
||||
// The view is built only from FleetConfig.MemberCredentials, which itself carries names,
|
||||
// never values (see its javadoc) — so there is no code path here that could substitute a
|
||||
// secret's value for its name. This pins the identity: what goes into `known`/`allow` is
|
||||
// exactly what comes out, character for character.
|
||||
List<String> known = List.of("SOME_TOKEN_NAME", "ANOTHER_NAME");
|
||||
FleetConfig.MemberCredentials creds =
|
||||
new FleetConfig.MemberCredentials("deny-by-default", List.of(), known);
|
||||
|
||||
MemberCredentialPolicyView view = MemberCredentialPolicyView.of(creds);
|
||||
|
||||
assertEquals(known, view.known());
|
||||
}
|
||||
}
|
||||
@@ -109,11 +109,11 @@ class MemberEnvAllowListTest {
|
||||
/**
|
||||
* {@code SSH_AUTH_SOCK} is a live handle to the operator's ssh-agent, never a value — so it must
|
||||
* stay excluded from the derived set even when the operator lists it under {@code allow:} for an
|
||||
* unrelated reason. It is governed ONLY by {@code memberCredentials.sshAuthSock}, applied
|
||||
* unrelated reason. It is governed ONLY by {@code memberCredentials.sshAgentEnv}, applied
|
||||
* separately by the caller ({@code HerdrPeerLauncher}).
|
||||
*/
|
||||
@Test
|
||||
void sshAuthSockInMemberCredentialsAllowIsStillExcluded() {
|
||||
void sshAgentEnvInMemberCredentialsAllowIsStillExcluded() {
|
||||
Set<String> derived = MemberEnvAllowList.derive(List.of(), Set.of("SSH_AUTH_SOCK", "OTHER_NAME"));
|
||||
|
||||
assertFalse(derived.contains("SSH_AUTH_SOCK"),
|
||||
@@ -146,7 +146,7 @@ class MemberEnvAllowListTest {
|
||||
return new FleetConfig(null, null, null, Map.of(), null, null, null, null, null,
|
||||
new FleetConfig.Broker(null, brokerUriEnv, null), null, null, null, null, null,
|
||||
null, null, null, null,
|
||||
new FleetConfig.Coordinator(null, coordinatorUriEnv, null, null)).withDefaults();
|
||||
new FleetConfig.Coordinator(null, coordinatorUriEnv, null, null, null)).withDefaults();
|
||||
}
|
||||
|
||||
/** {@code LC_*} categories are infrastructure by prefix; everything else needs an exact match. */
|
||||
|
||||
@@ -323,11 +323,39 @@ class OpenCodeLauncherTest {
|
||||
|
||||
// --- CB-547: resume + post-hoc session discovery --------------------------------------------
|
||||
|
||||
/**
|
||||
* Give {@code dir} the exact signature {@link HerdrPeerLauncher#isProvisionedWorktree} checks
|
||||
* for: a {@code .git} REGULAR FILE, never a directory. Content is never parsed by that gate, so
|
||||
* any {@code gitdir:} pointer is fine. Mirrors {@code ClaudeCodeLauncherTest}'s helper of the
|
||||
* same shape (fleetd #249).
|
||||
*/
|
||||
private static void markAsProvisionedWorktree(Path dir) throws IOException {
|
||||
Files.writeString(dir.resolve(".git"), "gitdir: /tmp/not-a-real-gitdir");
|
||||
}
|
||||
|
||||
/**
|
||||
* A fresh subdirectory of {@code configRoot}, marked as a provisioned worktree (fleetd #249),
|
||||
* for tests that predate this gate and stood in a bare {@code "/work/dir"} string as their
|
||||
* member's cwd — a directory that never existed on disk and, post-#249, would never pass
|
||||
* {@link HerdrPeerLauncher#isProvisionedWorktree} either. Those tests are about the model
|
||||
* mismatch / late-resolve machinery (fleetd #175/#234/#209), not about the worktree gate
|
||||
* itself, so they need a cwd the gate accepts without changing what each test demonstrates.
|
||||
*/
|
||||
private static String provisionedWorkDir(Path configRoot) throws IOException {
|
||||
Path dir = Files.createDirectories(configRoot.resolve("work-dir"));
|
||||
markAsProvisionedWorktree(dir);
|
||||
return dir.toString();
|
||||
}
|
||||
|
||||
@Test
|
||||
void aResumeSpawnPassesTheSessionIdAsDashS(@TempDir Path root) {
|
||||
void aResumeSpawnIntoAProvisionedWorktreePassesTheSessionIdAsDashS(@TempDir Path root,
|
||||
@TempDir Path worktree)
|
||||
throws Exception {
|
||||
markAsProvisionedWorktree(worktree);
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", null, null))
|
||||
.spawn(new SpawnRequest(null, null, null, null, "ses_41b79fc90ffeI9E8uZv6VprUn2"));
|
||||
.spawn(new SpawnRequest(null, worktree.toString(), null, null,
|
||||
"ses_41b79fc90ffeI9E8uZv6VprUn2"));
|
||||
|
||||
List<String> args = startArgs(herdr);
|
||||
int s = args.indexOf("-s");
|
||||
@@ -336,6 +364,27 @@ class OpenCodeLauncherTest {
|
||||
"the resume target id follows -s");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #249 acceptance criterion 3: without a fleetd-provisioned worktree, the member's cwd
|
||||
* is shared with other sessions, so fleetd can never reliably confirm (now or later via {@link
|
||||
* OpenCodeSessionDiscovery}) which conversation it is actually running. Refuse the spawn itself
|
||||
* rather than silently launching opencode's {@code -s <id>} into unverifiable territory.
|
||||
*/
|
||||
@Test
|
||||
void aResumeSpawnWithoutAProvisionedWorktreeIsRefused(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher = service(herdr, root,
|
||||
opencodeCfg("google/gemini-2.5-pro", null, null));
|
||||
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class, () ->
|
||||
launcher.spawn(new SpawnRequest(null, null, null, null,
|
||||
"ses_41b79fc90ffeI9E8uZv6VprUn2")));
|
||||
|
||||
assertTrue(e.getMessage().contains("worktree"), e.getMessage());
|
||||
assertFalse(herdr.called("agent.start"),
|
||||
"the refusal must happen before anything spawns — no pane, no process");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFreshSpawnCarriesNoSessionFlag(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
@@ -348,25 +397,59 @@ class OpenCodeLauncherTest {
|
||||
|
||||
@Test
|
||||
void theHandleDiscoversTheSessionIdForTheWorkersCwdOnlyAfterItAppears(@TempDir Path root,
|
||||
@TempDir Path discRoot)
|
||||
@TempDir Path discRoot,
|
||||
@TempDir Path worktree)
|
||||
throws Exception {
|
||||
markAsProvisionedWorktree(worktree);
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher = new OpenCodeLauncher(new AgentControl(herdr),
|
||||
new WorkspaceControl(herdr), Map.of("gemini", opencodeCfg(null, null, null)),
|
||||
"gemini", _ -> null, 0, System::currentTimeMillis, () -> { }, root, discRoot);
|
||||
|
||||
PeerHandle handle = launcher.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
PeerHandle handle = launcher.spawn(new SpawnRequest(null, worktree.toString(), null));
|
||||
|
||||
// opencode writes the record only when the session is first persisted — the instant the
|
||||
// pane is ready it does not exist, so agentSessionId() is null (never a spawn failure).
|
||||
assertNull(handle.agentSessionId(), "no record yet → null, not a spawn-time block");
|
||||
// Once the record appears (here: same cwd), lazy discovery resolves it — the handle's
|
||||
// session id matches its own worktree, not another's.
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_resolved", "/work/dir", 1000L);
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_resolved", worktree.toString(), 1000L);
|
||||
assertEquals("ses_resolved", handle.agentSessionId(),
|
||||
"agentSessionId() re-scans and picks up a record that has since been written");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #249 acceptance criterion 1, exercised through the real caller path (the handle
|
||||
* {@code fleet_list} actually reads), not {@link OpenCodeSessionDiscovery} directly. Without a
|
||||
* fleetd-provisioned worktree the member's cwd is shared — the default no-worktree spawn
|
||||
* inherits the lead's own long-lived cwd — so even once a matching row appears (here:
|
||||
* simulating another profile's session that happens to share the directory) the handle must
|
||||
* report absence rather than guess. Measured real-world case (2026-09-03): the row it would
|
||||
* otherwise pick was three days old and belonged to a different profile.
|
||||
*/
|
||||
@Test
|
||||
void theHandleNeverReportsAnIdForANonProvisionedCwdEvenAfterARowAppears(@TempDir Path root,
|
||||
@TempDir Path discRoot)
|
||||
throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher = new OpenCodeLauncher(new AgentControl(herdr),
|
||||
new WorkspaceControl(herdr), Map.of("gemini", opencodeCfg(null, null, null)),
|
||||
"gemini", _ -> null, 0, System::currentTimeMillis, () -> { }, root, discRoot);
|
||||
// No markAsProvisionedWorktree — this cwd has no .git file, the shared-cwd shape a
|
||||
// no-worktree spawn (or a real checkout) actually has.
|
||||
String sharedCwd = root.resolve("shared-cwd").toString();
|
||||
|
||||
PeerHandle handle = launcher.spawn(new SpawnRequest(null, sharedCwd, null));
|
||||
|
||||
assertNull(handle.agentSessionId(), "no record yet → null, same as the provisioned case");
|
||||
// A row for this exact directory now appears — e.g. a sibling member, or a stale session
|
||||
// from days earlier, sharing the same unprovisioned cwd.
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_someone_elses", sharedCwd, 1000L);
|
||||
assertNull(handle.agentSessionId(),
|
||||
"a non-provisioned cwd must NEVER report an id, even once a row for it exists — "
|
||||
+ "the row could belong to any other session sharing this directory");
|
||||
}
|
||||
|
||||
@Test
|
||||
void foreignWorkerMatchesOpencodePrefixButNotClaude() {
|
||||
String nonce = "abc123";
|
||||
@@ -920,6 +1003,7 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void theRealSessionManagerLateResolvePathCatchesAModelMismatch(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// xf's real shape (fleetd #175): weight:80, model "opencode/nemotron-3-ultra-free", no
|
||||
// credentialId — the profile that actually escaped the fleet's accounting.
|
||||
@@ -929,7 +1013,7 @@ class OpenCodeLauncherTest {
|
||||
OpenCodeLauncher launcher = serviceWithSink(herdr, configRoot, discRoot, cfg, sink);
|
||||
|
||||
SessionManager sessions = new SessionManager(launcher);
|
||||
MemberSession acquired = sessions.acquire(cfg.profile(), "/work/dir", null, null);
|
||||
MemberSession acquired = sessions.acquire(cfg.profile(), workDir, null, null);
|
||||
|
||||
// Real late-resolve path, driven BEFORE opencode has written its session row — same shape
|
||||
// as production the instant a pane goes ready.
|
||||
@@ -940,7 +1024,7 @@ class OpenCodeLauncherTest {
|
||||
|
||||
// opencode writes its row late, running gpt-5.6-sol (a PAID credential) instead of the
|
||||
// withdrawn free model the profile actually asked for — the exact fleetd #175 scenario.
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", workDir, 1000L,
|
||||
"{\"id\":\"gpt-5.6-sol\",\"providerID\":\"openai\"}");
|
||||
|
||||
// Drive the SAME real late-resolve path again: sessions.get() -> resolveAgentSessionId ->
|
||||
@@ -959,12 +1043,13 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void aProviderPrefixedModelMatchingBothIdAndProviderIsNotAMismatch(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("openai/gpt-5.6-terra", null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
.spawn(new SpawnRequest(null, workDir, null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", workDir, 1000L,
|
||||
"{\"id\":\"gpt-5.6-terra\",\"providerID\":\"openai\"}");
|
||||
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
@@ -975,12 +1060,13 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void aGxProviderPrefixedModelMatchingBothIdAndProviderIsNotAMismatch(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("gx/deepseek-v4-flash", null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
.spawn(new SpawnRequest(null, workDir, null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", workDir, 1000L,
|
||||
"{\"id\":\"deepseek-v4-flash\",\"providerID\":\"gx\"}");
|
||||
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
@@ -997,12 +1083,13 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void aMissingProviderIdInTheEvidenceIsUnknownNotAMismatchWhenTheIdMatches(
|
||||
@TempDir Path configRoot, @TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("openai/gpt-5.6-terra", null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
.spawn(new SpawnRequest(null, workDir, null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", workDir, 1000L,
|
||||
"{\"id\":\"gpt-5.6-terra\"}");
|
||||
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
@@ -1018,12 +1105,13 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void aMissingProviderIdInTheEvidenceStillCatchesARealIdMismatch(
|
||||
@TempDir Path configRoot, @TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("openai/gpt-5.6-terra", null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
.spawn(new SpawnRequest(null, workDir, null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", workDir, 1000L,
|
||||
"{\"id\":\"gpt-5.6-sol\"}");
|
||||
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
@@ -1042,12 +1130,13 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void aBareModelWithNoProviderPrefixMatchesOnIdAloneAndIsNotAMismatch(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("deepseek-v4-flash", null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
.spawn(new SpawnRequest(null, workDir, null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", workDir, 1000L,
|
||||
"{\"id\":\"deepseek-v4-flash\",\"providerID\":\"gx\"}");
|
||||
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
@@ -1064,6 +1153,7 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void aRealIdMismatchLogsAnErrorNamingBothModelsAndQuarantinesThroughTheSink(
|
||||
@TempDir Path configRoot, @TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> exhausted.add(target + "|" + reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("opencode/nemotron-3-ultra-free", null, null);
|
||||
@@ -1075,8 +1165,8 @@ class OpenCodeLauncherTest {
|
||||
PeerHandle handle;
|
||||
try {
|
||||
handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
.spawn(new SpawnRequest(null, workDir, null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", workDir, 1000L,
|
||||
"{\"id\":\"gpt-5.6-sol\",\"providerID\":\"openai\"}");
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
} finally {
|
||||
@@ -1111,11 +1201,12 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void unknownOrUnparseableModelEvidenceNeverQuarantines(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("openai/gpt-5.6-terra", null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
.spawn(new SpawnRequest(null, workDir, null));
|
||||
|
||||
// No row yet at all.
|
||||
assertNull(handle.agentSessionId());
|
||||
@@ -1137,12 +1228,13 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void aProfileWithNoConfiguredModelIsNeverCheckedForAMismatch(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg(null, null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
.spawn(new SpawnRequest(null, workDir, null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", workDir, 1000L,
|
||||
"{\"id\":\"anything-at-all\",\"providerID\":\"anyone\"}");
|
||||
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
@@ -1166,21 +1258,22 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void modelCheckReadsTheResolvedSessionsOwnRowNotWhateverIsNewestInTheSharedDirectory(
|
||||
@TempDir Path configRoot, @TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("openai/gpt-5.6-terra", null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
.spawn(new SpawnRequest(null, workDir, null));
|
||||
|
||||
// Our own session's row, correctly matching the profile's requested model.
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_ours", "/work/dir", 1000L,
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_ours", workDir, 1000L,
|
||||
"{\"id\":\"gpt-5.6-terra\",\"providerID\":\"openai\"}");
|
||||
assertEquals("ses_ours", handle.agentSessionId(), "resolves to our own session");
|
||||
assertTrue(exhausted.isEmpty(), "matching model → no mismatch on first resolve: " + exhausted);
|
||||
|
||||
// A sibling member, spawned later into the SAME shared directory (no worktree, fleetd
|
||||
// #234's default), writes a newer row running a totally different model.
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_sibling", "/work/dir", 9000L,
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_sibling", workDir, 9000L,
|
||||
"{\"id\":\"deepseek-v4-flash\",\"providerID\":\"gx\"}");
|
||||
|
||||
assertEquals("ses_ours", handle.agentSessionId(),
|
||||
@@ -1212,6 +1305,7 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void aSpawnTimeModelMismatchActuallyQuarantinesTheCredentialThroughTheRealAcquirePath(
|
||||
@TempDir Path configRoot, @TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
FleetConfig.Profile cfg = opencodeCfgWithCredential(
|
||||
"terra", "opencode/nemotron-3-ultra-free", "openai-shared");
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(cfg.profile(), cfg);
|
||||
@@ -1232,14 +1326,14 @@ class OpenCodeLauncherTest {
|
||||
|
||||
// The mismatching row exists BEFORE the spawn — reproducing fleetd #234's exact timing:
|
||||
// opencode's session table already carries evidence by the moment acquire() first asks.
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", workDir, 1000L,
|
||||
"{\"id\":\"gpt-5.6-sol\",\"providerID\":\"openai\"}");
|
||||
|
||||
assertFalse(quarantine.isQuarantined("openai-shared"), "nothing quarantined before the spawn");
|
||||
|
||||
// The real production entrypoint: acquire() builds the MemberSession by calling
|
||||
// handle.agentSessionId() BEFORE registry.put() runs.
|
||||
MemberSession acquired = sessions.acquire(cfg.profile(), "/work/dir", null, null);
|
||||
MemberSession acquired = sessions.acquire(cfg.profile(), workDir, null, null);
|
||||
|
||||
assertEquals("ses_x", acquired.agentSessionId(), "the id itself still resolves correctly");
|
||||
assertTrue(quarantine.isQuarantined("openai-shared"),
|
||||
@@ -1258,6 +1352,7 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void aRosterOnlySinkSilentlyDropsTheSpawnTimeQuarantine(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
FleetConfig.Profile cfg = opencodeCfgWithCredential(
|
||||
"terra", "opencode/nemotron-3-ultra-free", "openai-shared");
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.SECONDS.toNanos(1800));
|
||||
@@ -1270,10 +1365,10 @@ class OpenCodeLauncherTest {
|
||||
OpenCodeLauncher launcher = serviceWithSink(herdr, configRoot, discRoot, cfg, rosterOnlySink);
|
||||
SessionManager sessions = new SessionManager(launcher);
|
||||
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", workDir, 1000L,
|
||||
"{\"id\":\"gpt-5.6-sol\",\"providerID\":\"openai\"}");
|
||||
|
||||
MemberSession acquired = sessions.acquire(cfg.profile(), "/work/dir", null, null);
|
||||
MemberSession acquired = sessions.acquire(cfg.profile(), workDir, null, null);
|
||||
|
||||
assertEquals("ses_x", acquired.agentSessionId(), "the id itself still resolves correctly");
|
||||
assertFalse(quarantine.isQuarantined("openai-shared"),
|
||||
@@ -1301,6 +1396,7 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void theSpawnTimeQuarantineSurvivesTheFleetdStyleForwardingHop(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
FleetConfig.Profile cfg = opencodeCfgWithCredential(
|
||||
"terra", "opencode/nemotron-3-ultra-free", "openai-shared");
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(cfg.profile(), cfg);
|
||||
@@ -1332,16 +1428,143 @@ class OpenCodeLauncherTest {
|
||||
};
|
||||
exhaustionSinkRef.set(realSink);
|
||||
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", workDir, 1000L,
|
||||
"{\"id\":\"gpt-5.6-sol\",\"providerID\":\"openai\"}");
|
||||
|
||||
assertFalse(quarantine.isQuarantined("openai-shared"), "nothing quarantined before the spawn");
|
||||
|
||||
MemberSession acquired = sessions.acquire(cfg.profile(), "/work/dir", null, null);
|
||||
MemberSession acquired = sessions.acquire(cfg.profile(), workDir, null, null);
|
||||
|
||||
assertEquals("ses_x", acquired.agentSessionId(), "the id itself still resolves correctly");
|
||||
assertTrue(quarantine.isQuarantined("openai-shared"),
|
||||
"the profile hint must survive the Fleetd-style forwarding hop and reach the real "
|
||||
+ "sink — a lambda forwarder drops it and this must go red");
|
||||
}
|
||||
|
||||
// --- fleetd #267: the #175 check never ran for the ordinary (no-worktree) spawn shape --------
|
||||
|
||||
/**
|
||||
* fleetd #267 acceptance criterion 2, half 1 — a regression guard for the NEW code path only:
|
||||
* a spawn WITH a fleetd-provisioned worktree must keep running the fleetd #175 model check
|
||||
* exactly as before (already proven thoroughly above), and must now ALSO never emit the new
|
||||
* fleetd #267 "cannot run" WARN, since the check is not skipped in this shape. Driven through
|
||||
* the real {@code SessionManager.acquire()}/{@code get()} late-resolve path (fleetd #209),
|
||||
* the same path the existing #175 tests already exercise.
|
||||
*/
|
||||
@Test
|
||||
void aProvisionedWorktreeSpawnRunsTheModelCheckThroughSessionManagerAndNeverLogsTheSkipWarn(
|
||||
@TempDir Path configRoot, @TempDir Path discRoot) throws Exception {
|
||||
String workDir = provisionedWorkDir(configRoot);
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile cfg = opencodeCfg("opencode/nemotron-3-ultra-free", null, null);
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> exhausted.add(reason);
|
||||
OpenCodeLauncher launcher = serviceWithSink(herdr, configRoot, discRoot, cfg, sink);
|
||||
SessionManager sessions = new SessionManager(launcher);
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(OpenCodeLauncher.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
MemberSession acquired = sessions.acquire(cfg.profile(), workDir, null, null);
|
||||
assertNull(acquired.agentSessionId(), "no opencode row yet");
|
||||
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", workDir, 1000L,
|
||||
"{\"id\":\"gpt-5.6-sol\",\"providerID\":\"openai\"}");
|
||||
|
||||
Optional<MemberSession> after = sessions.get(acquired.paneId());
|
||||
assertEquals("ses_x", after.get().agentSessionId());
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
|
||||
assertEquals(1, exhausted.size(),
|
||||
"the mismatch check still runs on the real path with a provisioned worktree: " + exhausted);
|
||||
boolean cannotRunWarn = appender.list.stream()
|
||||
.filter(e -> e.getLevel() == Level.WARN)
|
||||
.anyMatch(e -> e.getFormattedMessage().contains("cannot run"));
|
||||
assertFalse(cannotRunWarn, "a provisioned-worktree spawn must never log the fleetd #267 "
|
||||
+ "'cannot run' WARN — the check ran, it was not skipped: " + appender.list);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #267's central defect, reproduced and fixed: {@code OpenCodeLauncher.SessionAwareHandle
|
||||
* .agentSessionId()} is the ONLY caller of {@code checkModelMatch}, and it sits behind the
|
||||
* fleetd #249 worktree gate — so a plain {@code fleet_spawn} with no {@code worktree:true}
|
||||
* (the ticket's "ordinary, expected shape of the large majority of spawns") never reached
|
||||
* {@code checkModelMatch} at all. A test that called {@code checkModelMatch} directly, or built
|
||||
* a {@link OpenCodeLauncher.SessionAwareHandle}/{@link PeerHandle} in isolation, would have
|
||||
* passed on every single day this gap existed — it never drives {@code agentSessionId()}
|
||||
* through the worktree gate the way production does. This test instead drives the REAL
|
||||
* late-resolve path: {@code SessionManager.acquire()} (which calls {@code handle
|
||||
* .agentSessionId()} to build the very first {@code MemberSession}) and a re-poll via {@code
|
||||
* SessionManager.get()} (fleetd #209's retained-handle mechanism) — the exact sequence a live
|
||||
* pane goes through.
|
||||
*
|
||||
* <p>The fix chosen (see {@code OpenCodeLauncher}'s javadoc on the {@code !worktreeProvisioned}
|
||||
* branch) is the WARN path, not a decoupled check: {@code actualModelForSessionId} can only be
|
||||
* keyed safely by a RESOLVED session id (fleetd #234's fix for exactly this false-positive
|
||||
* risk), and without a provisioned worktree no id can ever be safely resolved (fleetd #249) —
|
||||
* re-deriving "whatever is newest in this shared directory" here would silently reintroduce the
|
||||
* false-positive risk #234 fixed. This test proves both halves: a plausible-looking mismatch
|
||||
* row for the shared, non-provisioned cwd never quarantines anything, AND the new one-time,
|
||||
* per-profile WARN replaces the old total silence.
|
||||
*/
|
||||
@Test
|
||||
void aSpawnWithoutAProvisionedWorktreeNeverRunsTheModelCheckButWarnsOncePerProfile(
|
||||
@TempDir Path configRoot, @TempDir Path discRoot) throws Exception {
|
||||
// Deliberately NOT provisionedWorkDir(...) / markAsProvisionedWorktree(...): a plain
|
||||
// directory with no .git marker — the exact "fleet_spawn with no worktree:" shape fleetd
|
||||
// #267 is about, and the ordinary shape the ticket says most spawns actually take.
|
||||
Path workDir = Files.createDirectories(configRoot.resolve("shared-cwd"));
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile cfg = opencodeCfg("opencode/nemotron-3-ultra-free", null, null);
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> exhausted.add(reason);
|
||||
OpenCodeLauncher launcher = serviceWithSink(herdr, configRoot, discRoot, cfg, sink);
|
||||
SessionManager sessions = new SessionManager(launcher);
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(OpenCodeLauncher.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
// The real production entrypoint: acquire() calls handle.agentSessionId() itself to
|
||||
// build the very first MemberSession, BEFORE any row exists.
|
||||
MemberSession acquired = sessions.acquire(cfg.profile(), workDir.toString(), null, null);
|
||||
assertNull(acquired.agentSessionId(),
|
||||
"still refuses to guess an identity for a shared, non-provisioned cwd (fleetd #249)");
|
||||
|
||||
// A row for this exact (shared) directory appears, running a model that WOULD look like
|
||||
// a mismatch against cfg.model() if fleetd trusted the shared-directory heuristic —
|
||||
// exactly the false-positive shape fleetd #234 fixed for the id-resolved case.
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_sibling", workDir.toString(), 1000L,
|
||||
"{\"id\":\"gpt-5.6-sol\",\"providerID\":\"openai\"}");
|
||||
|
||||
// Re-drive the SAME real late-resolve path (fleetd #209) — repeatedly, to also prove
|
||||
// the new WARN fires at most once per profile, not once per poll.
|
||||
Optional<MemberSession> resolved = sessions.get(acquired.paneId());
|
||||
assertNull(resolved.get().agentSessionId(), "still no identity — the gate never opens");
|
||||
sessions.get(acquired.paneId());
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
|
||||
assertTrue(exhausted.isEmpty(),
|
||||
"must never quarantine off a shared-directory row it cannot trust as this session's "
|
||||
+ "own — fleetd #234's exact concern, now also for the model check: " + exhausted);
|
||||
|
||||
List<String> skipWarnings = appender.list.stream()
|
||||
.filter(e -> e.getLevel() == Level.WARN)
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.filter(m -> m.contains("cannot run"))
|
||||
.toList();
|
||||
assertEquals(1, skipWarnings.size(),
|
||||
"exactly one 'cannot run' WARN across acquire() + two get() re-polls — the old code "
|
||||
+ "logged NOTHING here, which is the bug this ticket fixes; got: " + skipWarnings);
|
||||
assertTrue(skipWarnings.get(0).contains(cfg.profile()),
|
||||
"the WARN must name the profile, same treatment discoveryUnavailable already gets: "
|
||||
+ skipWarnings.get(0));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,134 @@
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import com.rabbitmq.client.Channel;
|
||||
import com.rabbitmq.client.ConnectionFactory;
|
||||
import com.rabbitmq.client.impl.DefaultExceptionHandler;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.lang.reflect.Proxy;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertInstanceOf;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
class AmqpConnectionFailureLoggerTest {
|
||||
|
||||
@Test
|
||||
void installedHandlersLogTheirOwnConnectionNamesAtErrorWithTheCause() throws Exception {
|
||||
ConnectionFactory inboxFactory = AmqpReplyInbox.connectionFactory("amqp://127.0.0.1");
|
||||
ConnectionFactory mailboxFactory = LeadMailbox.connectionFactory("amqp://127.0.0.1");
|
||||
|
||||
AmqpConnectionFailureLogger inboxHandler = installedStrictHandler(inboxFactory, "reply inbox");
|
||||
AmqpConnectionFailureLogger mailboxHandler = installedStrictHandler(mailboxFactory, "lead mailbox");
|
||||
assertEquals(AmqpConnectionFailureLogger.REPLY_INBOX, inboxHandler.connectionName());
|
||||
assertEquals(AmqpConnectionFailureLogger.LEAD_MAILBOX, mailboxHandler.connectionName());
|
||||
|
||||
ListAppender<ILoggingEvent> inboxEvents = attach(AmqpReplyInbox.class);
|
||||
ListAppender<ILoggingEvent> mailboxEvents = attach(LeadMailbox.class);
|
||||
IllegalStateException inboxFailure = new IllegalStateException("inbox failure");
|
||||
IllegalStateException mailboxFailure = new IllegalStateException("mailbox failure");
|
||||
try {
|
||||
inboxHandler.handleUnexpectedConnectionDriverException(null, inboxFailure);
|
||||
mailboxHandler.handleConnectionRecoveryException(null, mailboxFailure);
|
||||
|
||||
assertError(inboxEvents, "AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred",
|
||||
inboxFailure, "inbox failure line");
|
||||
assertError(mailboxEvents, "AMQP connection fleetd-lead-mailbox: Caught an exception during connection recovery!",
|
||||
mailboxFailure, "mailbox recovery line");
|
||||
} finally {
|
||||
detach(AmqpReplyInbox.class, inboxEvents);
|
||||
detach(LeadMailbox.class, mailboxEvents);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void connectionResetKeepsForgivingHandlerWarningSemantics() {
|
||||
AmqpConnectionFailureLogger handler = new AmqpConnectionFailureLogger(
|
||||
AmqpConnectionFailureLogger.REPLY_INBOX, LoggerFactory.getLogger(AmqpReplyInbox.class));
|
||||
ListAppender<ILoggingEvent> events = attach(AmqpReplyInbox.class);
|
||||
try {
|
||||
handler.handleUnexpectedConnectionDriverException(null, new IOException("Connection reset"));
|
||||
assertEquals(1, events.list.size(), "the handler must still log a reset");
|
||||
ILoggingEvent event = events.list.getFirst();
|
||||
assertEquals(Level.WARN, event.getLevel(), "ForgivingExceptionHandler logs connection resets at WARN");
|
||||
assertEquals("AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred "
|
||||
+ "(Exception message: Connection reset)", event.getFormattedMessage());
|
||||
assertTrue(event.getThrowableProxy() == null, "ForgivingExceptionHandler does not attach a reset stack trace");
|
||||
} finally {
|
||||
detach(AmqpReplyInbox.class, events);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void connectionNamesStayDistinct() {
|
||||
assertNotEquals(AmqpConnectionFailureLogger.REPLY_INBOX, AmqpConnectionFailureLogger.LEAD_MAILBOX,
|
||||
"reply-inbox and lead-mailbox failures must be distinguishable");
|
||||
}
|
||||
|
||||
@Test
|
||||
void strictConsumerExceptionStillClosesItsChannel() {
|
||||
AtomicInteger closes = new AtomicInteger();
|
||||
Channel channel = (Channel) Proxy.newProxyInstance(getClass().getClassLoader(), new Class<?>[] {Channel.class},
|
||||
(_, method, _) -> switch (method.getName()) {
|
||||
case "close" -> {
|
||||
closes.incrementAndGet();
|
||||
yield null;
|
||||
}
|
||||
case "toString" -> "test-channel";
|
||||
default -> throw new UnsupportedOperationException(method.getName());
|
||||
});
|
||||
AmqpConnectionFailureLogger handler = new AmqpConnectionFailureLogger(
|
||||
AmqpConnectionFailureLogger.REPLY_INBOX, LoggerFactory.getLogger(AmqpReplyInbox.class));
|
||||
|
||||
handler.handleConsumerException(channel, new IllegalStateException("consumer failed"), null, "tag", "handleDelivery");
|
||||
|
||||
assertEquals(1, closes.get(), "DefaultExceptionHandler must close a channel after a consumer exception");
|
||||
}
|
||||
|
||||
@Test
|
||||
void handlerOnlyChangesDefaultHandlerLogging() {
|
||||
assertEquals(DefaultExceptionHandler.class,
|
||||
AmqpConnectionFailureLogger.class.getSuperclass());
|
||||
assertFalse(java.util.Arrays.stream(AmqpConnectionFailureLogger.class.getDeclaredMethods())
|
||||
.anyMatch(method -> method.getName().startsWith("handle")),
|
||||
"all exception-handling methods must remain inherited from DefaultExceptionHandler");
|
||||
}
|
||||
|
||||
private static ListAppender<ILoggingEvent> attach(Class<?> owner) {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(owner);
|
||||
logger.setLevel(Level.DEBUG);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(Class<?> owner, ListAppender<ILoggingEvent> appender) {
|
||||
((Logger) LoggerFactory.getLogger(owner)).detachAppender(appender);
|
||||
}
|
||||
|
||||
private static AmqpConnectionFailureLogger installedStrictHandler(ConnectionFactory factory, String connection) {
|
||||
assertInstanceOf(DefaultExceptionHandler.class, factory.getExceptionHandler(),
|
||||
connection + " must keep DefaultExceptionHandler: replacing the strict handler with a forgiving one "
|
||||
+ "changes when a channel is closed");
|
||||
return assertInstanceOf(AmqpConnectionFailureLogger.class, factory.getExceptionHandler());
|
||||
}
|
||||
|
||||
private static void assertError(ListAppender<ILoggingEvent> events, String message, Throwable cause, String name) {
|
||||
assertEquals(1, events.list.size(), name);
|
||||
ILoggingEvent event = events.list.getFirst();
|
||||
assertEquals(Level.ERROR, event.getLevel(), name);
|
||||
assertEquals(message, event.getFormattedMessage(), name);
|
||||
assertEquals(cause.toString(), event.getThrowableProxy().getClassName() + ": "
|
||||
+ event.getThrowableProxy().getMessage(), name);
|
||||
}
|
||||
}
|
||||
@@ -154,7 +154,7 @@ class AmqpReplyInboxContractTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void releaseCancelsConsumerAndClearsHeld() throws Exception {
|
||||
void releaseCancelsConsumerAndRequeuesHeldDeliveryForRecovery() throws Exception {
|
||||
String target = "worker-release-" + System.nanoTime();
|
||||
try (AmqpReplyInbox inbox = AmqpReplyInbox.open(uri())) {
|
||||
inbox.own(target);
|
||||
@@ -164,6 +164,22 @@ class AmqpReplyInboxContractTest {
|
||||
inbox.release(target);
|
||||
assertTrue(inbox.peek(target).isEmpty(),
|
||||
"release clears the local held snapshot");
|
||||
|
||||
// fleetd #298: release() must not just drop the local record — the broker delivery was
|
||||
// never acked, so cancelling the consumer alone leaves it unacked-but-orphaned on the
|
||||
// still-open channel unless release() nacks it back with requeue=true. Prove the message
|
||||
// is genuinely recoverable, not merely absent from peek: re-own the same target and
|
||||
// confirm the broker redelivers it to the fresh consumer.
|
||||
inbox.own(target);
|
||||
List<ReplyInbox.InboxMessage> recovered = awaitPeek(inbox, target);
|
||||
assertEquals(1, recovered.size(),
|
||||
"a reply held (but undrained) at release() time must still be recoverable — "
|
||||
+ "release() must requeue it, not silently drop it while the broker still "
|
||||
+ "considers it outstanding");
|
||||
assertEquals("m1", recovered.getFirst().msgId());
|
||||
assertEquals("release me", recovered.getFirst().content());
|
||||
|
||||
inbox.ack(target, "m1");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,264 @@
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import com.rabbitmq.client.AMQP;
|
||||
import com.rabbitmq.client.Channel;
|
||||
import com.rabbitmq.client.Connection;
|
||||
import com.rabbitmq.client.DeliverCallback;
|
||||
import com.rabbitmq.client.Delivery;
|
||||
import com.rabbitmq.client.Envelope;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.Timeout;
|
||||
|
||||
import java.lang.reflect.InvocationHandler;
|
||||
import java.lang.reflect.Proxy;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.time.Duration;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-318: a delivery landing on the consumer work-pool thread <em>after</em>
|
||||
* {@link AmqpReplyInbox#release} has already swapped the target's {@code held} entry for its
|
||||
* tombstone, but <em>before</em> {@code release()} itself returns, must be nacked-with-requeue —
|
||||
* never silently retained in a fresh map {@code release()} has already stopped looking at.
|
||||
*
|
||||
* <p><strong>This forces the actual interleaving, not a sequence of calls.</strong> {@code release()}
|
||||
* runs on its own thread and is made to block <em>inside</em> its nack loop, via a fake
|
||||
* {@link Channel} whose {@code basicNack} blocks on its first invocation. That block is only
|
||||
* reachable after {@code release()}'s {@code held.compute(...)} has already swapped in the
|
||||
* {@code RELEASED} tombstone (the compute call happens strictly before the loop that calls
|
||||
* {@code basicNack}), so observing it is direct, ordering-guaranteed proof that the tombstone is in
|
||||
* place and {@code release()} has not yet returned — still holding {@code channelLock} — when a
|
||||
* second thread fires {@code own()}'s captured {@link DeliverCallback} for a brand-new message on the
|
||||
* same target. No mocking library is on the classpath, so the fake broker is a {@link Proxy}, the
|
||||
* same pattern {@code AmqpReplyInboxRecoveryRaceTest} already uses.
|
||||
*
|
||||
* <p><strong>What this does and does not prove.</strong> It proves that a delivery whose
|
||||
* {@code computeIfAbsent} call is ordered strictly after {@code release()}'s tombstone swap — while
|
||||
* {@code release()} is still running — is nacked-with-requeue rather than silently parked forever.
|
||||
* It does not drive a real broker: {@code basicNack} here is a recorded call on a fake channel, not a
|
||||
* verified requeue-and-redeliver. That half of the contract (a nacked-with-requeue delivery really
|
||||
* does come back to a later owner) is already covered against a real broker by
|
||||
* {@code AmqpReplyInboxContractTest.releaseCancelsConsumerAndRequeuesHeldDeliveryForRecovery}, which
|
||||
* this test does not duplicate.
|
||||
*/
|
||||
class AmqpReplyInboxReleaseRaceTest {
|
||||
|
||||
@Test
|
||||
@Timeout(15)
|
||||
void deliveryArrivingWhileReleaseIsStillRunningIsNackedNotStranded() throws Exception {
|
||||
String target = "worker-release-race";
|
||||
|
||||
List<long[]> nacks = new CopyOnWriteArrayList<>(); // {deliveryTag, requeue(1/0)}
|
||||
List<Long> acks = new CopyOnWriteArrayList<>();
|
||||
AtomicReference<DeliverCallback> deliverCallback = new AtomicReference<>();
|
||||
CountDownLatch nackStarted = new CountDownLatch(1);
|
||||
CountDownLatch releaseMayFinishNack = new CountDownLatch(1);
|
||||
AtomicInteger nackCallCount = new AtomicInteger();
|
||||
|
||||
Channel consumeChannel = fakeConsumeChannel(deliverCallback, nacks, acks, nackCallCount,
|
||||
nackStarted, releaseMayFinishNack);
|
||||
Channel publishChannel = fakeInertChannel();
|
||||
Connection connection = fakeConnection(consumeChannel, publishChannel);
|
||||
|
||||
AmqpReplyInbox inbox = new AmqpReplyInbox(connection, AmqpReplyInbox.DEFAULT_PREFETCH);
|
||||
inbox.own(target);
|
||||
assertTrue(deliverCallback.get() != null, "own() must have registered a DeliverCallback");
|
||||
|
||||
// Seed one already-held delivery (m0) so release()'s nack loop has something to iterate, and
|
||||
// therefore somewhere to block, before it can return.
|
||||
deliverCallback.get().handle("ctag", delivery(1L, "m0", "first"));
|
||||
|
||||
AtomicReference<Throwable> releaseError = new AtomicReference<>();
|
||||
Thread releaseThread = new Thread(() -> {
|
||||
try {
|
||||
inbox.release(target);
|
||||
} catch (Throwable t) {
|
||||
releaseError.set(t);
|
||||
}
|
||||
}, "release-under-test");
|
||||
releaseThread.start();
|
||||
|
||||
// This latch only fires from inside the fake channel's basicNack — i.e. from inside
|
||||
// release()'s nack loop, which release()'s code only reaches AFTER held.compute(...) has
|
||||
// already swapped in RELEASED. Waiting for it is direct proof the swap has happened and
|
||||
// release() has not yet returned (it is stuck mid-loop, still holding channelLock).
|
||||
assertTrue(nackStarted.await(10, TimeUnit.SECONDS),
|
||||
"release() never reached its nack loop — it may not have started");
|
||||
|
||||
// The exact interleaving CB-318 describes: a delivery for a NEW message on the same target
|
||||
// lands on the "consumer work-pool thread" (this second thread) while release() is still
|
||||
// running. With the pre-fix code (a bare held.remove(target)) this created a brand-new map
|
||||
// under computeIfAbsent that release() — already past its remove — never looks at again.
|
||||
AtomicReference<Throwable> deliveryError = new AtomicReference<>();
|
||||
Thread deliveryThread = new Thread(() -> {
|
||||
try {
|
||||
deliverCallback.get().handle("ctag", delivery(2L, "m1", "second"));
|
||||
} catch (Throwable t) {
|
||||
deliveryError.set(t);
|
||||
}
|
||||
}, "concurrent-delivery");
|
||||
deliveryThread.start();
|
||||
|
||||
// Head start for the delivery thread to reach (and, on the fixed code, block on)
|
||||
// channelLock — release() still holds it at this point, so a correct fix cannot have
|
||||
// resolved m1's nack yet. Purely in-memory work (computeIfAbsent, a reference compare)
|
||||
// separates deliveryThread.start() from that block point, so 300ms is a large margin, not a
|
||||
// tight timing assumption.
|
||||
Thread.sleep(300);
|
||||
assertEquals(1, nackCallCount.get(),
|
||||
"the concurrent delivery must not resolve its nack before release() gives up "
|
||||
+ "channelLock — if this is 2 already, the interleaving below is not being "
|
||||
+ "tested, only a sequential call");
|
||||
|
||||
releaseMayFinishNack.countDown(); // let release() finish nacking m0 and return
|
||||
assertTrue(releaseThread.join(Duration.ofSeconds(10)), "release() did not finish");
|
||||
assertTrue(deliveryThread.join(Duration.ofSeconds(10)), "the concurrent delivery did not finish");
|
||||
|
||||
assertNull(releaseError.get(), "release() threw: " + releaseError.get());
|
||||
assertNull(deliveryError.get(), "the concurrent delivery threw: " + deliveryError.get());
|
||||
|
||||
assertEquals(2, nacks.size(),
|
||||
"both the pre-held m0 and the concurrently-arriving m1 must be nacked, got: "
|
||||
+ nacks.stream().map(n -> "[tag=" + n[0] + " requeue=" + n[1] + "]").toList());
|
||||
assertTrue(nacks.stream().allMatch(n -> n[1] == 1L),
|
||||
"invariant 1 (never drop): every nack must set requeue=true");
|
||||
assertTrue(nacks.stream().anyMatch(n -> n[0] == 1L), "m0's delivery tag must be nacked");
|
||||
assertTrue(nacks.stream().anyMatch(n -> n[0] == 2L),
|
||||
"m1 — delivered while release() was still running, after the tombstone swap — must be "
|
||||
+ "nacked, not silently retained in a map release() will never look at again");
|
||||
assertTrue(acks.isEmpty(), "invariant 1 (never drop): a held reply must never be basicAck'd");
|
||||
}
|
||||
|
||||
private static Delivery delivery(long tag, String msgId, String body) {
|
||||
Envelope envelope = new Envelope(tag, false, "", "irrelevant");
|
||||
AMQP.BasicProperties props = new AMQP.BasicProperties.Builder().messageId(msgId).build();
|
||||
return new Delivery(envelope, props, body.getBytes(StandardCharsets.UTF_8));
|
||||
}
|
||||
|
||||
/** A {@link Proxy}-backed consume {@link Channel}: blocks the FIRST {@code basicNack} call on
|
||||
* {@code releaseMayFinishNack}, after signalling {@code nackStarted} — everything else records
|
||||
* the call and returns a harmless default, matching the style already used by
|
||||
* {@code AmqpReplyInboxRecoveryRaceTest}. */
|
||||
private static Channel fakeConsumeChannel(AtomicReference<DeliverCallback> deliverCallback,
|
||||
List<long[]> nacks, List<Long> acks,
|
||||
AtomicInteger nackCallCount,
|
||||
CountDownLatch nackStarted,
|
||||
CountDownLatch releaseMayFinishNack) {
|
||||
InvocationHandler handler = (proxy, method, args) -> {
|
||||
String name = method.getName();
|
||||
if (name.equals("basicConsume")) {
|
||||
deliverCallback.set((DeliverCallback) args[2]);
|
||||
return "ctag";
|
||||
}
|
||||
if (name.equals("basicNack")) {
|
||||
long tag = (long) args[0];
|
||||
boolean requeue = (boolean) args[2];
|
||||
if (nackCallCount.incrementAndGet() == 1) {
|
||||
nackStarted.countDown();
|
||||
if (!releaseMayFinishNack.await(10, TimeUnit.SECONDS)) {
|
||||
throw new IllegalStateException("test never released the nack latch");
|
||||
}
|
||||
}
|
||||
nacks.add(new long[] {tag, requeue ? 1L : 0L});
|
||||
return null;
|
||||
}
|
||||
if (name.equals("basicAck")) {
|
||||
acks.add((long) args[0]);
|
||||
return null;
|
||||
}
|
||||
if (name.equals("equals")) {
|
||||
return proxy == args[0];
|
||||
}
|
||||
if (name.equals("hashCode")) {
|
||||
return System.identityHashCode(proxy);
|
||||
}
|
||||
if (name.equals("toString")) {
|
||||
return "FakeConsumeChannel";
|
||||
}
|
||||
return defaultValue(method.getReturnType());
|
||||
};
|
||||
return (Channel) Proxy.newProxyInstance(AmqpReplyInboxReleaseRaceTest.class.getClassLoader(),
|
||||
new Class<?>[] {Channel.class}, handler);
|
||||
}
|
||||
|
||||
/** A {@link Proxy}-backed {@link Channel} that answers every call with a harmless default — used
|
||||
* as the publish channel, which this test never actually publishes on. */
|
||||
private static Channel fakeInertChannel() {
|
||||
InvocationHandler handler = (proxy, method, args) -> {
|
||||
String name = method.getName();
|
||||
if (name.equals("equals")) {
|
||||
return proxy == args[0];
|
||||
}
|
||||
if (name.equals("hashCode")) {
|
||||
return System.identityHashCode(proxy);
|
||||
}
|
||||
if (name.equals("toString")) {
|
||||
return "FakeInertChannel";
|
||||
}
|
||||
return defaultValue(method.getReturnType());
|
||||
};
|
||||
return (Channel) Proxy.newProxyInstance(AmqpReplyInboxReleaseRaceTest.class.getClassLoader(),
|
||||
new Class<?>[] {Channel.class}, handler);
|
||||
}
|
||||
|
||||
/** A {@link Proxy}-backed {@link Connection} handing out {@code first} then {@code second} from
|
||||
* successive {@code createChannel()} calls, matching {@link AmqpReplyInbox}'s constructor. */
|
||||
private static Connection fakeConnection(Channel first, Channel second) {
|
||||
AtomicInteger calls = new AtomicInteger();
|
||||
InvocationHandler handler = (proxy, method, args) -> {
|
||||
String name = method.getName();
|
||||
if (name.equals("createChannel") && (args == null || args.length == 0)) {
|
||||
return calls.getAndIncrement() == 0 ? first : second;
|
||||
}
|
||||
if (name.equals("equals")) {
|
||||
return proxy == args[0];
|
||||
}
|
||||
if (name.equals("hashCode")) {
|
||||
return System.identityHashCode(proxy);
|
||||
}
|
||||
if (name.equals("toString")) {
|
||||
return "FakeConnection";
|
||||
}
|
||||
return defaultValue(method.getReturnType());
|
||||
};
|
||||
return (Connection) Proxy.newProxyInstance(AmqpReplyInboxReleaseRaceTest.class.getClassLoader(),
|
||||
new Class<?>[] {Connection.class}, handler);
|
||||
}
|
||||
|
||||
private static Object defaultValue(Class<?> type) {
|
||||
if (!type.isPrimitive() || type == void.class) {
|
||||
return null;
|
||||
}
|
||||
if (type == boolean.class) {
|
||||
return Boolean.FALSE;
|
||||
}
|
||||
if (type == long.class) {
|
||||
return 0L;
|
||||
}
|
||||
if (type == short.class) {
|
||||
return (short) 0;
|
||||
}
|
||||
if (type == byte.class) {
|
||||
return (byte) 0;
|
||||
}
|
||||
if (type == char.class) {
|
||||
return (char) 0;
|
||||
}
|
||||
if (type == double.class) {
|
||||
return 0.0d;
|
||||
}
|
||||
if (type == float.class) {
|
||||
return 0.0f;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
@@ -3,6 +3,8 @@ package dev.ltms.fleet.msg;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* Hermetic stand-in for {@link LeadChannel}: an in-memory mailbox that records what was published
|
||||
@@ -24,11 +26,19 @@ public final class FakeLeadChannel implements LeadChannel {
|
||||
private final List<String> acked = Collections.synchronizedList(new ArrayList<>());
|
||||
/** When set, every {@link #publish} throws it — the unroutable/nacked/timed-out peer. */
|
||||
private volatile IllegalStateException publishFailure;
|
||||
/** Canned {@link #inspect} results by coord-id — absent for any coord-id not configured here. */
|
||||
private final Map<String, MailboxState> mailboxes = new ConcurrentHashMap<>();
|
||||
|
||||
public FakeLeadChannel(String selfCoordId) {
|
||||
this.selfCoordId = selfCoordId;
|
||||
}
|
||||
|
||||
/** Make {@link #inspect(String)} return {@code state} for {@code coordId} instead of "absent". */
|
||||
public FakeLeadChannel withMailbox(String coordId, MailboxState state) {
|
||||
mailboxes.put(coordId, state);
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make every publish fail as an unreachable peer would. */
|
||||
public FakeLeadChannel failPublishWith(String message) {
|
||||
this.publishFailure = new IllegalStateException(message);
|
||||
@@ -65,6 +75,11 @@ public final class FakeLeadChannel implements LeadChannel {
|
||||
return selfCoordId;
|
||||
}
|
||||
|
||||
@Override
|
||||
public MailboxState inspect(String coordId) {
|
||||
return mailboxes.getOrDefault(coordId, MailboxState.absent(coordId));
|
||||
}
|
||||
|
||||
public List<LeadMessage> published() {
|
||||
return List.copyOf(published);
|
||||
}
|
||||
|
||||
@@ -54,6 +54,40 @@ class LeadCoordLoopTest {
|
||||
assertTrue(channel.peek().isEmpty(), "and is no longer held");
|
||||
}
|
||||
|
||||
@Test
|
||||
void redeliveryOfAMessageAlreadyWrittenToThePaneIsAckedWithoutAnotherPaneWrite() {
|
||||
var channel = new FakeLeadChannel(SELF).hold(new LeadMessage("m1", PEER, SELF, "recover me"));
|
||||
var herdr = new FakeHerdr().agentStatus("idle");
|
||||
var loop = loop(channel, herdr, Map.of(LEAD_TERM, SELF));
|
||||
|
||||
loop.tick();
|
||||
channel.hold(new LeadMessage("m1", PEER, SELF, "recover me"));
|
||||
loop.tick();
|
||||
|
||||
assertEquals(1, prompts(herdr).size(), "a redelivery must not consume the lead pane twice");
|
||||
assertEquals(List.of("m1", "m1"), channel.acked(), "the redelivery still needs a fresh broker ack");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aRedeliveryIsAckedEvenWhileTheLeadIsMidTurn() {
|
||||
var channel = new FakeLeadChannel(SELF).hold(new LeadMessage("m1", PEER, SELF, "recover me"));
|
||||
var herdr = new FakeHerdr().agentStatus("idle");
|
||||
var loop = loop(channel, herdr, Map.of(LEAD_TERM, SELF));
|
||||
|
||||
loop.tick();
|
||||
channel.hold(new LeadMessage("m1", PEER, SELF, "recover me"));
|
||||
herdr.agentStatus("working");
|
||||
loop.tick();
|
||||
|
||||
assertEquals(1, prompts(herdr).size(), "the pane is still written exactly once");
|
||||
assertEquals(List.of("m1", "m1"), channel.acked(),
|
||||
"a message already written to the pane must be acked even mid-turn: the mid-turn "
|
||||
+ "gate exists to protect the pane, and this message needs no pane. Gating the ack "
|
||||
+ "on it leaves the message held on a lead that is busy most of the time, and every "
|
||||
+ "recovery redelivers it again — which is the loop this fix exists to stop");
|
||||
assertTrue(channel.peek().isEmpty(), "so it is no longer held");
|
||||
}
|
||||
|
||||
@Test
|
||||
void leavesTheMessageUnackedWhenTheLeadIsMidTurn() {
|
||||
var channel = new FakeLeadChannel(SELF).hold(new LeadMessage("m1", PEER, SELF, "hello"));
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import com.rabbitmq.client.ShutdownSignalException;
|
||||
import com.rabbitmq.client.impl.AMQImpl;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.io.IOException;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #361 review round 2: a mutation that made {@link LeadMailbox#isMissingQueue} return
|
||||
* {@code true} unconditionally still left {@code mvn clean install} green — 1389 tests, 0
|
||||
* failures — because nothing exercised its false branch. That branch is the whole discriminator
|
||||
* between {@link LeadChannel.MailboxState#absent} and {@link LeadChannel.MailboxState#unknown};
|
||||
* without a test pinning it, a future refactor that widens it back to "always true" (restoring the
|
||||
* exact overstatement fleetd #361 exists to fix) would pass this suite.
|
||||
*
|
||||
* <p>Hermetic — no broker needed, per the review's own suggestion. {@code isMissingQueue} takes a
|
||||
* plain {@link IOException}, so every input here is constructed directly rather than provoked from
|
||||
* a live connection. The real 404 shape itself is still pinned against a real broker, in
|
||||
* {@code LeadMailboxTest.passiveDeclareOfAMissingQueueThrowsAnIOExceptionWrappingA404ShutdownSignal}
|
||||
* — this class covers the three false shapes {@link LeadMailbox#isMissingQueue}'s own javadoc
|
||||
* lists, so both directions of the discriminator are proven somewhere.
|
||||
*/
|
||||
class LeadMailboxIsMissingQueueTest {
|
||||
|
||||
@Test
|
||||
void aConfirmedMissingQueueIsRecognized() {
|
||||
ShutdownSignalException sse = new ShutdownSignalException(true, false,
|
||||
new AMQImpl.Channel.Close(404, "NOT_FOUND - no queue 'lead.x.inbox' in vhost '/'", 50, 10), null);
|
||||
IOException e = new IOException("channel error", sse);
|
||||
|
||||
assertTrue(LeadMailbox.isMissingQueue(e), "a genuine 404 Channel.Close must be recognized as a missing queue");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aDifferentReplyCodeIsNotAMissingQueue() {
|
||||
// E.g. 403 ACCESS_REFUSED — the queue may well exist; this call was simply refused.
|
||||
ShutdownSignalException sse = new ShutdownSignalException(true, false,
|
||||
new AMQImpl.Channel.Close(403, "ACCESS_REFUSED", 50, 10), null);
|
||||
IOException e = new IOException("channel error", sse);
|
||||
|
||||
assertFalse(LeadMailbox.isMissingQueue(e),
|
||||
"a non-404 reply code must never be read as a confirmed absence — the mailbox's real state is unknown");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aShutdownSignalWhoseReasonIsNotAChannelCloseIsNotAMissingQueue() {
|
||||
// A Connection.Close (a whole different broker-level shutdown) is still a ShutdownSignalException,
|
||||
// but its reason is not a Channel.Close at all — must not be misread as "no such queue".
|
||||
ShutdownSignalException sse = new ShutdownSignalException(true, false,
|
||||
new AMQImpl.Connection.Close(404, "coincidentally 404, but this is a CONNECTION close", 10, 50), null);
|
||||
IOException e = new IOException("connection error", sse);
|
||||
|
||||
assertFalse(LeadMailbox.isMissingQueue(e),
|
||||
"a ShutdownSignalException whose reason is not a Channel.Close must never be read as a missing queue,"
|
||||
+ " even if its reply code happens to be 404");
|
||||
}
|
||||
|
||||
@Test
|
||||
void anIOExceptionWithNoCauseAtAllIsNotAMissingQueue() {
|
||||
IOException e = new IOException("some other declare failure, no cause attached");
|
||||
|
||||
assertFalse(LeadMailbox.isMissingQueue(e),
|
||||
"an IOException with no ShutdownSignalException cause must never be read as a confirmed absence");
|
||||
}
|
||||
|
||||
@Test
|
||||
void anIOExceptionWithAnUnrelatedCauseIsNotAMissingQueue() {
|
||||
IOException e = new IOException("wrapped something else entirely", new RuntimeException("boom"));
|
||||
|
||||
assertFalse(LeadMailbox.isMissingQueue(e),
|
||||
"a cause that isn't even a ShutdownSignalException must never be read as a confirmed absence");
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,9 @@
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import com.rabbitmq.client.AMQP;
|
||||
import com.rabbitmq.client.Channel;
|
||||
import com.rabbitmq.client.Connection;
|
||||
import com.rabbitmq.client.ShutdownSignalException;
|
||||
import org.junit.jupiter.api.BeforeAll;
|
||||
import org.junit.jupiter.api.Tag;
|
||||
import org.junit.jupiter.api.Test;
|
||||
@@ -7,11 +11,14 @@ import org.testcontainers.containers.RabbitMQContainer;
|
||||
import org.testcontainers.junit.jupiter.Testcontainers;
|
||||
import org.testcontainers.utility.DockerImageName;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertInstanceOf;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
@@ -85,6 +92,33 @@ class LeadMailboxTest {
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void ackThrowsWhenRecoveryClearedTheHeldMessage() throws Exception {
|
||||
String to = coordId("lead-recovery-ack");
|
||||
try (LeadMailbox inbox = LeadMailbox.open(uri(), to)) {
|
||||
inbox.publish(to, new LeadMessage("recovery-ack", "lead-from", to, "in flight"));
|
||||
assertEquals(1, awaitPeek(inbox).size(), "the broker delivery must be held before recovery clears it");
|
||||
|
||||
inbox.clearHeldForRecovery();
|
||||
|
||||
assertThrows(IllegalStateException.class, () -> inbox.ack("recovery-ack"),
|
||||
"a cleared delivery has no valid tag, so ack must report that it did not reach the broker");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void ackOfAMessageAlreadyAckedOnThisConnectionStaysQuiet() throws Exception {
|
||||
String to = coordId("lead-repeat-ack");
|
||||
try (LeadMailbox inbox = LeadMailbox.open(uri(), to)) {
|
||||
inbox.publish(to, new LeadMessage("repeat-ack", "lead-from", to, "once"));
|
||||
awaitPeek(inbox);
|
||||
|
||||
inbox.ack("repeat-ack");
|
||||
|
||||
inbox.ack("repeat-ack");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void duplicateMsgIdIsNotDoubleQueued() throws Exception {
|
||||
String to = coordId("lead-dedup");
|
||||
@@ -159,6 +193,145 @@ class LeadMailboxTest {
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void inspectReportsAnOwnedMailboxAsExistingWithItsOwnConsumer() throws Exception {
|
||||
// A LeadMailbox declares AND consumes its own queue the moment open() returns (see own()),
|
||||
// so inspecting a coord-id this same process owns must always find exactly one consumer.
|
||||
String self = coordId("lead-inspect-self");
|
||||
try (LeadMailbox mailbox = LeadMailbox.open(uri(), self)) {
|
||||
LeadChannel.MailboxState state = mailbox.inspect(self);
|
||||
assertEquals(self, state.coordId());
|
||||
assertTrue(state.exists(), "this daemon owns and has declared this exact queue");
|
||||
assertEquals(0, state.pending(), "nothing has been published to it yet");
|
||||
assertEquals(1, state.consumers(), "the mailbox's own constructor already attached a consumer");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void inspectReportsAMissingMailboxAsAbsentRatherThanThrowing() throws Exception {
|
||||
String nobody = coordId("lead-inspect-nobody");
|
||||
try (LeadMailbox mailbox = LeadMailbox.open(uri(), coordId("lead-inspect-caller"))) {
|
||||
LeadChannel.MailboxState state = mailbox.inspect(nobody);
|
||||
assertEquals(LeadChannel.MailboxState.absent(nobody), state,
|
||||
"a queue nobody has ever declared must report absent, never throw");
|
||||
assertTrue(state.known(), "a confirmed 404 IS a definite answer — this is not the unknown case");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361 review finding 3: {@code inspect} is specified to never throw, but the original
|
||||
* implementation caught only {@link IOException} — and {@link Connection#createChannel()} on an
|
||||
* already-closed connection throws {@link com.rabbitmq.client.AlreadyClosedException}, an
|
||||
* unchecked {@link RuntimeException} (pinned by {@code
|
||||
* createChannelOnAnAlreadyClosedConnectionThrowsAnUncheckedException} above). This drives that
|
||||
* exact scenario through the real {@link LeadMailbox#inspect} — not the raw client call — and
|
||||
* checks both halves of finding 1 and finding 3 at once: no exception escapes, and the result is
|
||||
* {@code UNKNOWN} rather than the wrong-but-plausible-looking {@code ABSENT}.
|
||||
*/
|
||||
@Test
|
||||
void inspectReportsUnknownRatherThanThrowingWhenTheConnectionIsAlreadyClosed() throws Exception {
|
||||
LeadMailbox mailbox = LeadMailbox.open(uri(), coordId("lead-inspect-dead-connection"));
|
||||
mailbox.close(); // tears down the connection `inspect` will try to open a probe channel on
|
||||
|
||||
LeadChannel.MailboxState state = mailbox.inspect(coordId("lead-inspect-irrelevant-target"));
|
||||
|
||||
assertFalse(state.exists());
|
||||
assertFalse(state.known(), "a dead connection proves nothing about the target mailbox — it must be unknown, not absent");
|
||||
assertEquals(LeadChannel.MailboxState.Presence.UNKNOWN, state.presence());
|
||||
}
|
||||
|
||||
@Test
|
||||
void inspectReportsPendingMessagesAndZeroConsumersWhenNobodyIsReadingAnymore() throws Exception {
|
||||
// Publish into a mailbox this test owns, then never consume from it, to prove `pending` and
|
||||
// `consumers` really come off the broker rather than off this process's own in-memory state.
|
||||
String to = coordId("lead-inspect-pending");
|
||||
String observerId = coordId("lead-inspect-observer");
|
||||
try (LeadMailbox owner = LeadMailbox.open(uri(), to);
|
||||
LeadMailbox observer = LeadMailbox.open(uri(), observerId)) {
|
||||
owner.publish(to, new LeadMessage("m1", "lead-from", to, "sitting in the queue"));
|
||||
awaitPeek(owner); // make sure the broker has actually enqueued it before inspecting
|
||||
} // `owner` closes here: its consumer disconnects, but the durable, unacked message stays queued.
|
||||
|
||||
try (LeadMailbox observer = LeadMailbox.open(uri(), coordId("lead-inspect-observer-2"))) {
|
||||
// The broker requeues `owner`'s unacked delivery asynchronously once its connection drops,
|
||||
// so poll rather than assume the very first passive declare already sees the settled state.
|
||||
LeadChannel.MailboxState state = awaitInspect(observer, to, s -> s.consumers() == 0);
|
||||
assertTrue(state.exists());
|
||||
assertEquals(0, state.consumers(), "the only owner just closed — nobody is reading this anymore");
|
||||
assertEquals(1, state.pending(), "the unacked message must be requeued, never dropped");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361's central invariant, proved rather than assumed: a passive queue declare of a
|
||||
* missing queue closes ITS channel with a 404 in AMQP 0-9-1. {@link LeadMailbox#inspect} is
|
||||
* specified to run on its own disposable channel for exactly this reason — this test is the one
|
||||
* that actually exercises the failure mode and shows {@link LeadMailbox#publish} on the SAME
|
||||
* instance is unaffected by it.
|
||||
*/
|
||||
@Test
|
||||
void inspectingAMissingMailboxNeverBreaksPublishOnTheSameInstance() throws Exception {
|
||||
String self = coordId("lead-invariant-self");
|
||||
try (LeadMailbox mailbox = LeadMailbox.open(uri(), self)) {
|
||||
// Miss on a queue that has never existed — this is exactly the 404-closes-the-channel case.
|
||||
LeadChannel.MailboxState missed = mailbox.inspect(coordId("lead-invariant-nobody-home"));
|
||||
assertFalse(missed.exists());
|
||||
assertTrue(missed.known(), "a genuine 404 on a queue that never existed is a confirmed fact, not an unknown");
|
||||
|
||||
// publish() must still work on THIS SAME instance: if inspect() had reused `publishChannel`
|
||||
// (or `channel`), the broker's 404 would have closed it out from underneath publish().
|
||||
LeadMessage sent = new LeadMessage("after-miss", "lead-from", self, "still alive");
|
||||
mailbox.publish(self, sent);
|
||||
List<LeadMessage> got = awaitPeek(mailbox);
|
||||
assertEquals(1, got.size(), "publish must still reach this mailbox's own queue after a missed inspect");
|
||||
assertEquals("after-miss", got.getFirst().msgId());
|
||||
|
||||
// And a second inspect() — of a mailbox that DOES exist this time — must also still work,
|
||||
// proving the miss did not wedge inspect() itself either.
|
||||
LeadChannel.MailboxState self2 = mailbox.inspect(self);
|
||||
assertTrue(self2.exists());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Pins the exact exception shape {@link LeadMailbox#inspect} relies on to tell a genuine 404
|
||||
* (mailbox confirmed absent) apart from everything else (mailbox state unknown) — measured
|
||||
* against a real broker rather than assumed from the AMQP 0-9-1 spec text. If this ever fails,
|
||||
* the classification in {@code inspect} is reading the wrong shape and must be revisited.
|
||||
*/
|
||||
@Test
|
||||
void passiveDeclareOfAMissingQueueThrowsAnIOExceptionWrappingA404ShutdownSignal() throws Exception {
|
||||
try (Connection conn = LeadMailbox.connectionFactory(uri()).newConnection()) {
|
||||
Channel probe = conn.createChannel();
|
||||
String missing = LeadMailbox.queueName(coordId("lead-404-shape"));
|
||||
IOException thrown = assertThrows(IOException.class, () -> probe.queueDeclarePassive(missing));
|
||||
assertInstanceOf(ShutdownSignalException.class, thrown.getCause(),
|
||||
() -> "expected the IOException to wrap a ShutdownSignalException, got: " + thrown);
|
||||
ShutdownSignalException sse = (ShutdownSignalException) thrown.getCause();
|
||||
assertInstanceOf(AMQP.Channel.Close.class, sse.getReason(),
|
||||
() -> "expected a Channel.Close reason: " + sse);
|
||||
AMQP.Channel.Close close = (AMQP.Channel.Close) sse.getReason();
|
||||
assertEquals(404, close.getReplyCode(), () -> "expected AMQP NOT_FOUND (404): " + close);
|
||||
assertFalse(probe.isOpen(), "the 404 must have closed the channel the declare ran on");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The other half of the same measurement: calling {@code createChannel()} on an
|
||||
* already-closed connection — the shape {@link LeadMailbox#inspect} hits when the broker
|
||||
* connection itself is gone — throws {@link com.rabbitmq.client.AlreadyClosedException}, an
|
||||
* unchecked {@link RuntimeException}, not an {@link IOException}. An {@code inspect} that only
|
||||
* caught {@code IOException} here would let this escape instead of reporting "unknown".
|
||||
*/
|
||||
@Test
|
||||
void createChannelOnAnAlreadyClosedConnectionThrowsAnUncheckedException() throws Exception {
|
||||
Connection conn = LeadMailbox.connectionFactory(uri()).newConnection();
|
||||
conn.close();
|
||||
RuntimeException thrown = assertThrows(RuntimeException.class, conn::createChannel);
|
||||
assertInstanceOf(com.rabbitmq.client.AlreadyClosedException.class, thrown,
|
||||
() -> "expected AlreadyClosedException, got: " + thrown);
|
||||
}
|
||||
|
||||
/** Poll peek until at least one message is held, or ~10s elapse (broker delivery is async). */
|
||||
@SuppressWarnings("BusyWait")
|
||||
private static List<LeadMessage> awaitPeek(LeadMailbox inbox) throws InterruptedException {
|
||||
@@ -170,4 +343,18 @@ class LeadMailboxTest {
|
||||
}
|
||||
return msgs;
|
||||
}
|
||||
|
||||
/** Poll inspect(coordId) until it satisfies {@code done}, or ~10s elapse (broker state settles async). */
|
||||
@SuppressWarnings("BusyWait")
|
||||
private static LeadChannel.MailboxState awaitInspect(
|
||||
LeadMailbox observer, String coordId, java.util.function.Predicate<LeadChannel.MailboxState> done)
|
||||
throws InterruptedException {
|
||||
long deadline = System.nanoTime() + TimeUnit.SECONDS.toNanos(10);
|
||||
LeadChannel.MailboxState state = observer.inspect(coordId);
|
||||
while (!done.test(state) && System.nanoTime() < deadline) {
|
||||
Thread.sleep(50);
|
||||
state = observer.inspect(coordId);
|
||||
}
|
||||
return state;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
@@ -11,6 +15,7 @@ import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
@@ -253,7 +258,7 @@ class MessageServiceTest {
|
||||
injector.onStatus(T, AgentStatus.IDLE); // first delivery
|
||||
injector.onStatus(T, AgentStatus.WORKING); // first turn in flight
|
||||
|
||||
CompletableFuture<Void> queued = injector.enqueue(T, "second task", TestTurnTokens.inert(T));
|
||||
CompletableFuture<Void> queued = injector.enqueue(T, "second task", TestTurnTokens.inert(T)).completion();
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(T);
|
||||
injector.drop(T, new HerdrException("agent target sol not found", "agent_not_found", null));
|
||||
|
||||
@@ -372,6 +377,127 @@ class MessageServiceTest {
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #334. {@code ask()}'s {@code TimeoutException} catch used to forget this task's
|
||||
* {@code turnId} mapping ({@code clearAsyncQuestion(turnId, true)}) BEFORE closing the ask
|
||||
* ({@code rendezvous.closeAsk}, in the shared {@code finally}). A primary's {@code answer()}
|
||||
* call racing that exact window found the ask still "answerable" ({@code
|
||||
* rendezvous.askSession(turnId)} still non-null) while the {@code Task} was already forgotten,
|
||||
* so its {@code task != null} guard skipped the completion and the async ticket sat at
|
||||
* {@code PENDING} forever even though {@code answer()} itself reported a result. The fix
|
||||
* (closing the ask first) makes this window impossible: a racing {@code answer()} call either
|
||||
* still finds the ask open (and the {@code Task} mapping guaranteed intact) or finds it already
|
||||
* closed (and bails {@code STALE_TURN} before ever touching the {@code Task}). This test pins
|
||||
* the exact window with {@code askTimeoutRaceHookForTest} and proves both invariants the ticket
|
||||
* named: (1) a late/racing {@code answer()} sees the ask as already lapsed ({@code STALE_TURN}),
|
||||
* never made answerable again, and (2) the async ticket still resolves {@code DONE} once the
|
||||
* worker's real {@code fleet_reply} lands — it is never stranded {@code PENDING}.
|
||||
*/
|
||||
/**
|
||||
* fleetd #334 gated the ask-timeout teardown on {@code ticket.fresh()}, matching the {@code
|
||||
* finally} block that already did. This pins that gate. A coalesced duplicate passes its own
|
||||
* {@code timeoutMillis}, which says nothing about whether the shared ask is done — so a
|
||||
* duplicate timing out first must leave the fresh owner's still-open ask answerable.
|
||||
*
|
||||
* <p>Measured on merge: without this test, removing the {@code ticket.fresh()} gate left all
|
||||
* 1371 tests green. The gate shipped with the reorder and nothing held it there.
|
||||
*
|
||||
* <p>What this does not prove: anything about the ordering inside the gate — that is
|
||||
* {@code aLateAnswerDuringAskTimeoutTeardownStillCompletesTheAsyncTicket}'s job.
|
||||
*/
|
||||
@Test
|
||||
void aCoalescedDuplicateAskTimingOutLeavesTheFreshOwnersAskOpen() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "long task");
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up, then pauses to ask
|
||||
|
||||
CompletableFuture<MessageService.AskResult> fresh =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||
|
||||
MessageService.TaskView asking = null;
|
||||
long deadline = System.currentTimeMillis() + 2000;
|
||||
while ((asking == null || asking.phase() != MessageService.Phase.ASKING)
|
||||
&& System.currentTimeMillis() < deadline) {
|
||||
asking = messages.poll(ticket);
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertNotNull(asking, "the fresh owner's question must surface before the duplicate asks");
|
||||
String turnId = asking.turnId();
|
||||
assertNotNull(turnId, "an ASKING view carries the turnId to answer on");
|
||||
|
||||
// A coalesced duplicate on the same session, with its own much shorter timeout.
|
||||
MessageService.AskResult duplicate = messages.ask(T, "which config file?", 100);
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT, duplicate.outcome(),
|
||||
"the duplicate's own timeout elapses first");
|
||||
|
||||
CompletableFuture<MessageService.Reply> answered =
|
||||
CompletableFuture.supplyAsync(() -> messages.answer(turnId, "fleetd.yaml", 500));
|
||||
|
||||
MessageService.AskResult a = fresh.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.AskOutcome.ANSWERED, a.outcome(),
|
||||
"a duplicate's timeout must not lapse the ask the fresh owner still holds");
|
||||
assertEquals("fleetd.yaml", a.answer());
|
||||
assertFalse(answered.get(5, TimeUnit.SECONDS).outcome() == MessageService.Outcome.STALE_TURN,
|
||||
"the answer must not be rejected as stale");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLateAnswerDuringAskTimeoutTeardownStillCompletesTheAsyncTicket() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "long task");
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up, then pauses to ask
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 200));
|
||||
|
||||
// Wait for the question to actually surface (poll sees ASKING) before racing the timeout.
|
||||
MessageService.TaskView asking = null;
|
||||
long deadline = System.currentTimeMillis() + 2000;
|
||||
while ((asking == null || asking.phase() != MessageService.Phase.ASKING)
|
||||
&& System.currentTimeMillis() < deadline) {
|
||||
asking = messages.poll(ticket);
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertNotNull(asking, "the question must surface before the ask times out");
|
||||
String turnId = asking.turnId();
|
||||
assertNotNull(turnId, "an ASKING view carries the turnId to answer on");
|
||||
|
||||
CompletableFuture<MessageService.Reply> lateAnswer = new CompletableFuture<>();
|
||||
messages.setAskTimeoutRaceHookForTest(() ->
|
||||
lateAnswer.complete(messages.answer(turnId, "too late", 500)));
|
||||
try {
|
||||
MessageService.AskResult a = ask.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT, a.outcome());
|
||||
|
||||
MessageService.Reply late = lateAnswer.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.STALE_TURN, late.outcome(),
|
||||
"a late answer racing the timeout teardown must see the ask as already lapsed");
|
||||
|
||||
// The worker resumes on its own (per the ask() contract) and eventually sends its real
|
||||
// fleet_reply; the async ticket must still resolve with it, not strand at PENDING.
|
||||
assertEquals(MessageService.ReplyOutcome.RESOLVED_ASYNC_TICKET, messages.reply(T, "real result"),
|
||||
"the worker's real reply must still be accepted, resolving the parked async ticket");
|
||||
} finally {
|
||||
messages.setAskTimeoutRaceHookForTest(null);
|
||||
}
|
||||
|
||||
MessageService.TaskView done = null;
|
||||
deadline = System.currentTimeMillis() + 2000;
|
||||
while ((done == null || done.phase() == MessageService.Phase.PENDING)
|
||||
&& System.currentTimeMillis() < deadline) {
|
||||
done = messages.poll(ticket);
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertNotNull(done);
|
||||
assertEquals(MessageService.Phase.DONE, done.phase(), "the async ticket must not be stranded PENDING");
|
||||
assertEquals("real result", done.reply());
|
||||
}
|
||||
|
||||
@Test
|
||||
void answeringAnUnknownTurnIsStale() {
|
||||
MessageService.Reply r = messages.answer(T + "#999", "too late", 500);
|
||||
@@ -405,6 +531,29 @@ class MessageServiceTest {
|
||||
"a delivered send whose worker never replies times out as still working");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #345. This forces the injector to pick up the exact pending delivery after {@code send}
|
||||
* first observes its completion as incomplete, but before {@code cancel} takes the target monitor.
|
||||
* The timeout must use {@link Injector.Cancellation#DELIVERED} from {@code cancel} and report
|
||||
* {@link MessageService.Outcome#TIMED_OUT_WORKING}, because the text landed.
|
||||
*
|
||||
* <p>What this does not prove: that this precise interleaving happens by itself under production
|
||||
* timing. The test forces it through a test-only hook; it proves the timeout caller handles the
|
||||
* injector result when the interleaving occurs.
|
||||
*/
|
||||
@Test
|
||||
void sendTimeoutUsesCancellationDeliveredWhenPickupWinsTheRace() {
|
||||
messages.setTimeoutCancellationRaceHookForTest(() -> injector.onStatus(T, AgentStatus.IDLE));
|
||||
try {
|
||||
MessageService.Reply reply = messages.send(T, "race delivery", 50);
|
||||
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_WORKING, reply.outcome(),
|
||||
"cancel reporting DELIVERED means the worker received the timed-out message");
|
||||
} finally {
|
||||
messages.setTimeoutCancellationRaceHookForTest(null);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void answerTimesOutWhenTheResumedWorkerNeverReplies() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
@@ -619,7 +768,9 @@ class MessageServiceTest {
|
||||
@Test
|
||||
void replyQueuesInInboxWhenNoSendIsOpen() {
|
||||
// No send is open for this session — reply should queue in the inbox.
|
||||
assertTrue(messages.reply(T, "queued-text"), "reply should succeed (queued)");
|
||||
// fleetd #365: this is the case that must read as QUEUED, not "delivered".
|
||||
assertEquals(MessageService.ReplyOutcome.QUEUED, messages.reply(T, "queued-text"),
|
||||
"reply should succeed but only as queued — nothing was waiting for it");
|
||||
|
||||
var drained = messages.drainReplies(T);
|
||||
assertEquals(1, drained.size());
|
||||
@@ -632,7 +783,9 @@ class MessageServiceTest {
|
||||
awaitUninterruptibly(T);
|
||||
|
||||
// An explicit reply resolves the open send.
|
||||
assertTrue(messages.reply(T, "send-resolved"), "reply should succeed (resolved live send)");
|
||||
// fleetd #365: this is the other case — RESOLVED_SEND, distinct from QUEUED above.
|
||||
assertEquals(MessageService.ReplyOutcome.RESOLVED_SEND, messages.reply(T, "send-resolved"),
|
||||
"reply should succeed by resolving the live waiting send");
|
||||
|
||||
// The inbox should be empty — the reply went to the send, not the inbox.
|
||||
assertTrue(messages.drainReplies(T).isEmpty(), "no reply in the inbox");
|
||||
@@ -780,6 +933,49 @@ class MessageServiceTest {
|
||||
assertFailedTicket(third, "agent target term_a not found");
|
||||
}
|
||||
|
||||
// --- fleetd #335 (site 1): a per-task cleanup failure inside the abandon() loop must not -----
|
||||
// strand the tasks that come after it. abandon()'s own comment on the loop documents the one
|
||||
// real production call that can throw there (inbox.publish, in the stranded-reply put-back
|
||||
// branch, reached when a concurrent reply() or a second abandon() races this one) — but
|
||||
// reaching that branch requires the exact combination the #137 follow-up above already found
|
||||
// unreachable through the public API. abandonCleanupHookForTest reproduces the resulting SHAPE
|
||||
// (one task's cleanup throws) directly instead, the same technique this file already uses for
|
||||
// fleetd #324/#329's own hard-to-time races.
|
||||
@Test
|
||||
void aPerTaskCleanupFailureDoesNotStrandTheRemainingMatchingTasks() throws Exception {
|
||||
ListAppender<ILoggingEvent> appender = attachMessageServiceLog();
|
||||
try {
|
||||
String first = messages.sendAsync(T, "first task");
|
||||
awaitWaiting(); // first task owns the target lock and rendezvous waiter
|
||||
String second = messages.sendAsync(T, "second task"); // parked on the same lock
|
||||
String third = messages.sendAsync(T, "third task"); // parked too — the whole sweep must survive
|
||||
|
||||
java.util.concurrent.atomic.AtomicInteger calls = new java.util.concurrent.atomic.AtomicInteger();
|
||||
messages.setAbandonCleanupHookForTest(() -> {
|
||||
if (calls.getAndIncrement() == 0) {
|
||||
throw new RuntimeException("PROBE-335-SITE1");
|
||||
}
|
||||
});
|
||||
|
||||
assertTrue(messages.abandon(T, "agent target term_a not found"));
|
||||
|
||||
// Every task in the loop still gets its own outcome — the one whose cleanup threw
|
||||
// included — even though the loop had no way to know in advance which one that would be.
|
||||
assertFailedTicket(first, "agent target term_a not found");
|
||||
assertFailedTicket(second, "agent target term_a not found");
|
||||
assertFailedTicket(third, "agent target term_a not found");
|
||||
|
||||
assertTrue(appender.list.stream().anyMatch(e ->
|
||||
e.getLevel() == Level.ERROR
|
||||
&& e.getThrowableProxy() != null
|
||||
&& "PROBE-335-SITE1".equals(e.getThrowableProxy().getMessage())),
|
||||
"a per-task cleanup failure must still reach the log, not vanish silently");
|
||||
} finally {
|
||||
messages.setAbandonCleanupHookForTest(null);
|
||||
detachMessageServiceLog(appender);
|
||||
}
|
||||
}
|
||||
|
||||
// --- #137 follow-up: abandon() must not guess when more than one task is open ---------------
|
||||
//
|
||||
// A test combining a genuine stranded reply (hasStrandedReply(T)==true) with two simultaneously
|
||||
@@ -828,6 +1024,56 @@ class MessageServiceTest {
|
||||
assertEquals(MessageService.Outcome.REPLIED, answer.get(5, TimeUnit.SECONDS).outcome());
|
||||
}
|
||||
|
||||
// --- fleetd #275: a target torn down FOR GOOD while genuinely ASKING must not orphan --------
|
||||
//
|
||||
// sessions.onRelease (fleet_stop, or the idle reaper) is the one abandon() caller that knows
|
||||
// for certain the target can never resume: its pane is being stopped right now. Unlike the
|
||||
// health-classification caller above (a GONE/NEVER_READY guess, not a teardown it performed),
|
||||
// it must sweep an ASKING ticket right here — see MessageService.abandon(String, String,
|
||||
// boolean)'s javadoc for the full reachability chain this closes: without this, the forward
|
||||
// waiter is already closed by the time the question surfaces, the ASKING guard skips the task,
|
||||
// and by the time the worker's own fleet_ask lapses (~55-115s later) the released session no
|
||||
// longer appears in FleetHealthMonitor's roster for anything to ever sweep it again — leaving
|
||||
// fleet_poll{ticket} stuck PENDING forever.
|
||||
|
||||
@Test
|
||||
void abandonWithSweepAskingFailsATornDownTargetsAskingTicket() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config?", 300));
|
||||
MessageService.TaskView asking = awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
|
||||
assertTrue(messages.abandon(T, "the worker session was released before it replied", true),
|
||||
"a released target's open ask can never resume, so it must fail right here");
|
||||
|
||||
MessageService.TaskView failed = awaitTicketPhase(ticket, MessageService.Phase.FAILED);
|
||||
assertEquals("the worker session was released before it replied", failed.detail());
|
||||
|
||||
// The reverse-rendezvous ask is torn down too: the worker's still-blocked fleet_ask rides
|
||||
// out its own timeout (nothing completed its answer future), and a late answer() for the
|
||||
// same turnId must see it as lapsed rather than resolving a question nobody is waiting on.
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT, ask.get(5, TimeUnit.SECONDS).outcome());
|
||||
assertEquals(MessageService.Outcome.STALE_TURN,
|
||||
messages.answer(asking.turnId(), "config.yaml", 200).outcome());
|
||||
}
|
||||
|
||||
@Test
|
||||
void abandonWithoutSweepAskingBehavesLikeTheTwoArgOverload() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config?", 5000));
|
||||
awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
|
||||
assertFalse(messages.abandon(T, "agent target term_a not found", false),
|
||||
"sweepAsking=false must match the plain abandon(target, reason) overload");
|
||||
assertEquals(MessageService.Phase.ASKING, messages.poll(ticket).phase());
|
||||
}
|
||||
|
||||
// --- #137: a fleet_ask round-trip must not orphan the ticket's own reply -------------------
|
||||
//
|
||||
// The primary's fleet_send{turnId} answer call is itself bounded (a real MCP call, capped well
|
||||
@@ -853,8 +1099,11 @@ class MessageServiceTest {
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_WORKING, answerReply.outcome(),
|
||||
"the primary's own bounded wait gives up before the worker finishes resuming");
|
||||
|
||||
// The worker keeps working past that window and only now calls fleet_reply.
|
||||
assertTrue(messages.reply(T, "PR opened: https://example/pulls/42"));
|
||||
// The worker keeps working past that window and only now calls fleet_reply. The forward
|
||||
// waiter answer() opened already timed out, so this resolves via the parked async ticket,
|
||||
// not a live send (fleetd #365).
|
||||
assertEquals(MessageService.ReplyOutcome.RESOLVED_ASYNC_TICKET,
|
||||
messages.reply(T, "PR opened: https://example/pulls/42"));
|
||||
|
||||
MessageService.TaskView done = awaitTicketPhase(ticket, MessageService.Phase.DONE);
|
||||
assertEquals("PR opened: https://example/pulls/42", done.reply(),
|
||||
@@ -878,7 +1127,10 @@ class MessageServiceTest {
|
||||
assertEquals("config.yaml", ask.get(5, TimeUnit.SECONDS).answer());
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_WORKING, answerReply.outcome());
|
||||
|
||||
assertTrue(messages.reply(T, "PR opened: https://example/pulls/42"));
|
||||
// No live waiter (answer()'s own forward wait already timed out) — resolves the parked
|
||||
// async ticket instead (fleetd #365).
|
||||
assertEquals(MessageService.ReplyOutcome.RESOLVED_ASYNC_TICKET,
|
||||
messages.reply(T, "PR opened: https://example/pulls/42"));
|
||||
|
||||
// fleet_stop tears the worker's session down right after the reply landed — this must never
|
||||
// report the misleading "the worker session was released before it replied": a reply is
|
||||
@@ -890,6 +1142,56 @@ class MessageServiceTest {
|
||||
assertEquals("PR opened: https://example/pulls/42", view.reply());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #324: {@code answer()} holds {@code sessionLocks} for the target and, once the worker's
|
||||
* real terminal reply arrives, calls {@code finishAsyncTask}, which used to read the volatile
|
||||
* {@code task.turnId} twice — once to check it is non-null, once as the key for
|
||||
* {@code asyncTasksByTurn.remove}. {@code ask()}'s own timeout path mutates the same field with no
|
||||
* lock at all. This test does not wait for a real race to land in that narrow window between the
|
||||
* two reads — instead it drives the exact sequence the ticket describes (worker asks, primary
|
||||
* answers, worker's real reply arrives) and, via a package-private test hook wired to fire at
|
||||
* precisely that point, runs the identical production cleanup {@code ask()}'s timeout catch block
|
||||
* runs ({@code clearAsyncQuestion(turnId, true)}) so the field goes {@code null} between the two
|
||||
* reads deterministically rather than by chance.
|
||||
*
|
||||
* <p>What this proves: given that exact interleaving, {@code answer()} must not throw and the
|
||||
* ticket must still resolve to the worker's real reply. What it does not prove: that the
|
||||
* interleaving itself is reachable in production — that is established by reading the code (see
|
||||
* the ticket), not by this test, since forcing it via a hook is not the same as two independent
|
||||
* threads racing on their own schedules.
|
||||
*/
|
||||
@Test
|
||||
void finishAsyncTaskSurvivesTurnIdGoingNullBetweenItsTwoReads() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config?", 5000));
|
||||
MessageService.TaskView asking = awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
String turnId = asking.turnId();
|
||||
|
||||
// Fire ask()'s own unlocked timeout cleanup at the moment finishAsyncTask has already checked
|
||||
// task.turnId is non-null but has not yet used it — the exact torn-read window fleetd #324
|
||||
// describes.
|
||||
messages.setFinishAsyncTaskRaceHookForTest(() -> messages.forgetTurnForTest(turnId));
|
||||
|
||||
CompletableFuture<MessageService.Reply> answer =
|
||||
CompletableFuture.supplyAsync(() -> messages.answer(turnId, "config.yaml", 5000));
|
||||
assertEquals("config.yaml", ask.get(5, TimeUnit.SECONDS).answer());
|
||||
awaitWaiting(); // answer() opened its own forward waiter for the resumed worker turn
|
||||
|
||||
// A live waiter is open (the forward wait above) — this resolves it directly (fleetd #365).
|
||||
assertEquals(MessageService.ReplyOutcome.RESOLVED_SEND,
|
||||
messages.reply(T, "PR opened: https://example/pulls/42"));
|
||||
|
||||
assertEquals(MessageService.Outcome.REPLIED, answer.get(5, TimeUnit.SECONDS).outcome(),
|
||||
"the lead's own answer() call must not throw because ask()'s timeout cleanup raced it");
|
||||
MessageService.TaskView done = awaitTicketPhase(ticket, MessageService.Phase.DONE);
|
||||
assertEquals("PR opened: https://example/pulls/42", done.reply(),
|
||||
"the ticket must still resolve to the worker's real reply despite the forced race");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unansweredAsyncQuestionReturnsTheTicketToPendingAndReleasesItsTarget() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks");
|
||||
@@ -907,6 +1209,182 @@ class MessageServiceTest {
|
||||
assertEquals(MessageService.Phase.DONE, awaitTicketPhase(next, MessageService.Phase.DONE).phase());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #307: a worker's {@code fleet_ask} can time out because the primary never answers —
|
||||
* distinct from {@link #aReplyAfterAnswerTimesOutStillCompletesTheAsyncTicket}, where the
|
||||
* primary DID answer and only its own bounded wait for the resumed turn expired.
|
||||
* {@code ask()}'s timeout path deliberately forgets the task's {@code turnId} (so
|
||||
* {@code hasAsyncQuestion} stops reporting the target BUSY — see
|
||||
* {@code unansweredAsyncQuestionReturnsTheTicketToPendingAndReleasesItsTarget} above), which used
|
||||
* to also erase the one signal {@code askAnsweredAsyncTasks} needed to recognize the worker's
|
||||
* eventual real {@code fleet_reply}. That reply then had nowhere to land but the inbox, and
|
||||
* {@code fleet_poll{ticket}} stayed PENDING forever — later force-failed with the false reason
|
||||
* "session released before it replied", even though the worker had, in fact, replied.
|
||||
*/
|
||||
@Test
|
||||
void aReplyAfterAnAskTimeoutStillCompletesTheAsyncTicket() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks then finishes alone");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT,
|
||||
messages.ask(T, "which config?", 200).outcome());
|
||||
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket).phase(),
|
||||
"only the question wait ended; the delegated turn may still finish");
|
||||
|
||||
// The worker keeps working past the timeout and only now calls fleet_reply — with no live
|
||||
// rendezvous waiter open (ask()'s timeout already closed it) and no new send() having
|
||||
// reopened one for this target. So it resolves the parked async ticket (fleetd #365).
|
||||
assertEquals(MessageService.ReplyOutcome.RESOLVED_ASYNC_TICKET,
|
||||
messages.reply(T, "PR opened: https://example/pulls/42"));
|
||||
|
||||
MessageService.TaskView done = awaitTicketPhase(ticket, MessageService.Phase.DONE);
|
||||
assertEquals("PR opened: https://example/pulls/42", done.reply(),
|
||||
"fleet_poll{ticket} must return the worker's real reply, not stay pending forever");
|
||||
assertEquals("reply", done.replySource());
|
||||
assertFalse(messages.hasStrandedReply(T),
|
||||
"the reply completed its own ticket directly and never touched the inbox");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #329 (F1). {@code answer()} completes the async ticket by looking {@code turnId} up in
|
||||
* {@code asyncTasksByTurn} a SECOND time (the first is at :991, purely to re-register the
|
||||
* {@code asyncTasksByWaiter} entry for #282's chained-ask case). That second lookup races
|
||||
* {@code ask()}'s own unlocked timeout cleanup ({@code clearAsyncQuestion(turnId, true)}, run from
|
||||
* {@code markAskTimedOut} + forgetting): {@code ask()}'s {@code ticket.answer().get(timeoutMillis)}
|
||||
* can time out at essentially the same instant {@code answer()}'s {@code rendezvous.answerAsk}
|
||||
* call above already succeeded and unblocked the worker. fleetd #324's single-read fix does not
|
||||
* help here — that fixed a torn read of one already-held {@link MessageService} internal
|
||||
* {@code Task}; this is a second, independent map lookup by a caller that no longer holds the
|
||||
* {@code Task} it already found once.
|
||||
*
|
||||
* <p>This test does not wait for that real race to land on its own schedule — it drives the exact
|
||||
* sequence the ticket describes (worker asks, primary answers, worker's real reply arrives) and
|
||||
* fires the identical production cleanup {@code ask()}'s timeout path runs
|
||||
* ({@code clearAsyncQuestion(turnId, true)}, via the {@code forgetTurnForTest} seam fleetd #324
|
||||
* already merged) at the point between "the worker's own {@code ask()} call has unblocked" and
|
||||
* "the worker's real {@code fleet_reply} arrives" — the exact window fleetd #329 names.
|
||||
*
|
||||
* <p>What this proves: given that exact interleaving, the async ticket must still resolve to the
|
||||
* worker's real reply, not stay stuck {@link MessageService.Phase#PENDING} forever. What it does
|
||||
* not prove: that the interleaving itself is reachable in production on its own timing — that is
|
||||
* established by reading the code (see the ticket's "path in"), not by this test, for the same
|
||||
* reason fleetd #324's own race test says so.
|
||||
*/
|
||||
@Test
|
||||
void aReplyRacingAsksTimeoutCleanupStillCompletesTheAsyncTicket() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config?", 5000));
|
||||
MessageService.TaskView asking = awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
String turnId = asking.turnId();
|
||||
|
||||
CompletableFuture<MessageService.Reply> answer =
|
||||
CompletableFuture.supplyAsync(() -> messages.answer(turnId, "config.yaml", 5000));
|
||||
assertEquals("config.yaml", ask.get(5, TimeUnit.SECONDS).answer(),
|
||||
"the worker's own ask() call must have already unblocked with the primary's answer "
|
||||
+ "before we force the race below");
|
||||
|
||||
// ask()'s own unlocked timeout cleanup can forget this exact turnId at essentially the same
|
||||
// instant answer() has already unblocked the worker and is now waiting on the resumed turn's
|
||||
// real reply — reproduce that interleaving directly instead of trying to win a real race.
|
||||
messages.forgetTurnForTest(turnId);
|
||||
|
||||
// answer() is still waiting on its own forward waiter for the resumed turn — a live send —
|
||||
// so this resolves it directly, not the async ticket (fleetd #365).
|
||||
assertEquals(MessageService.ReplyOutcome.RESOLVED_SEND,
|
||||
messages.reply(T, "PR opened: https://example/pulls/42"));
|
||||
|
||||
assertEquals(MessageService.Outcome.REPLIED, answer.get(5, TimeUnit.SECONDS).outcome(),
|
||||
"the primary's own answer() call must still see the worker's real reply");
|
||||
MessageService.TaskView done = awaitTicketPhase(ticket, MessageService.Phase.DONE);
|
||||
assertEquals("PR opened: https://example/pulls/42", done.reply(),
|
||||
"fleet_poll{ticket} must return the worker's real reply, not stay PENDING forever "
|
||||
+ "just because ask()'s timeout cleanup forgot this turnId first");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #329 (F3). {@link MessageService#reply} reads the same orphan {@code Task}'s {@code
|
||||
* turnId} twice on this path (once to check it is non-null, once as the {@code
|
||||
* asyncTasksByTurn.remove} key) — the exact double-read shape fleetd #324 fixed in {@code
|
||||
* finishAsyncTask}. This test drives the same real sequence as {@code
|
||||
* aReplyAfterAnswerTimesOutStillCompletesTheAsyncTicket} (worker asks, primary answers, the
|
||||
* primary's own bounded wait for the resumed turn expires) to reach a task with {@code turnId}
|
||||
* genuinely stamped and no live rendezvous waiter open — the state {@code askAnsweredAsyncTasks}
|
||||
* matches here — then fires the identical production cleanup {@code ask()}'s own timeout path
|
||||
* runs ({@code clearAsyncQuestion(turnId, true)}, via the {@code forgetTurnForTest} seam fleetd
|
||||
* #324 merged) at the point between the check and the removal use, via a dedicated test hook
|
||||
* mirroring {@code finishAsyncTaskRaceHook}.
|
||||
*/
|
||||
@Test
|
||||
void replyToAnOrphanedTaskSurvivesTurnIdGoingNullBetweenItsTwoReads() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config?", 5000));
|
||||
MessageService.TaskView asking = awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
String turnId = asking.turnId();
|
||||
|
||||
MessageService.Reply answerReply = messages.answer(turnId, "config.yaml", 150);
|
||||
assertEquals("config.yaml", ask.get(5, TimeUnit.SECONDS).answer());
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_WORKING, answerReply.outcome(),
|
||||
"the primary's own bounded wait must give up first, leaving turnId stamped with no "
|
||||
+ "live waiter — the state reply()'s F3 code path matches");
|
||||
|
||||
messages.setReplyOrphanTurnIdRaceHookForTest(() -> messages.forgetTurnForTest(turnId));
|
||||
try {
|
||||
// No live waiter — resolves the parked async ticket (fleetd #365).
|
||||
assertEquals(MessageService.ReplyOutcome.RESOLVED_ASYNC_TICKET,
|
||||
messages.reply(T, "PR opened: https://example/pulls/42"));
|
||||
|
||||
MessageService.TaskView done = awaitTicketPhase(ticket, MessageService.Phase.DONE);
|
||||
assertEquals("PR opened: https://example/pulls/42", done.reply(),
|
||||
"the ticket must still resolve to the worker's real reply despite the forced race");
|
||||
} finally {
|
||||
messages.setReplyOrphanTurnIdRaceHookForTest(null);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #307's ambiguity guard: an ask timeout frees its target ({@code hasAsyncQuestion}
|
||||
* becomes false the instant it lapses — proven above), so a second, independent delegation can
|
||||
* be dispatched to the same target and itself go on to ask-and-lapse before the first worker's
|
||||
* real reply ever arrives. Two open tasks are then both eligible candidates on one target with
|
||||
* no live waiter to disambiguate them. A reply arriving now must not guess which one it answers
|
||||
* — guessing wrong would hand the lead a plausible-looking answer to a delegation the worker
|
||||
* never touched, worse than a failure because the lead acts on it — so it must fall back to the
|
||||
* inbox exactly as the zero-candidate case does.
|
||||
*/
|
||||
@Test
|
||||
void twoAskTimedOutTicketsOnOneTargetFallBackToTheInboxRatherThanGuess() throws Exception {
|
||||
String ticket1 = messages.sendAsync(T, "first task that asks");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT, messages.ask(T, "Q1?", 200).outcome());
|
||||
|
||||
String ticket2 = messages.sendAsync(T, "second task that asks");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT, messages.ask(T, "Q2?", 200).outcome());
|
||||
|
||||
// Ambiguous — two candidates, so it must fall back to the inbox rather than guess (fleetd #365).
|
||||
assertEquals(MessageService.ReplyOutcome.QUEUED, messages.reply(T, "which task does this answer?"));
|
||||
|
||||
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket1).phase(),
|
||||
"an ambiguous reply must not guess ticket1");
|
||||
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket2).phase(),
|
||||
"an ambiguous reply must not guess ticket2");
|
||||
assertTrue(messages.hasStrandedReply(T));
|
||||
var drained = messages.drainReplies(T);
|
||||
assertEquals(1, drained.size());
|
||||
assertEquals("which task does this answer?", drained.get(0).content());
|
||||
}
|
||||
|
||||
@Test
|
||||
void asyncQuestionBelongsToTheTaskThatOwnsItsForwardWaiter() throws Exception {
|
||||
String first = messages.sendAsync(T, "first task");
|
||||
@@ -925,6 +1403,121 @@ class MessageServiceTest {
|
||||
assertEquals(MessageService.Outcome.REPLIED, answer.get(5, TimeUnit.SECONDS).outcome());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #282: a worker that chains a SECOND {@code fleet_ask} inside the same resumed turn —
|
||||
* before it ever calls {@code fleet_reply} — used to kill its own async ticket. {@code answer()}
|
||||
* opens a fresh forward waiter but (unlike {@code send()}) never registered it in
|
||||
* {@code asyncTasksByWaiter}, so the second ask's {@code markAsyncQuestion} found no {@code Task}
|
||||
* to re-associate. That waiter still resolved with the second {@code QUESTION} once the worker
|
||||
* asked again, and {@code answer()} completed the ticket's future with that QUESTION "reply"
|
||||
* unconditionally — so {@code fleet_poll} reported FAILED while the worker was still alive and
|
||||
* the primary was mid-conversation with it.
|
||||
*
|
||||
* <p>Driven entirely through {@code MessageService}'s public API (sendAsync/ask/answer/poll) —
|
||||
* never by reaching into {@link Rendezvous} or the task maps directly, so this test cannot pass
|
||||
* for a reason unrelated to the real bug.
|
||||
*/
|
||||
@Test
|
||||
void secondFleetAskInTheSameResumedTurnDoesNotKillTheAsyncTicket() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks twice");
|
||||
awaitWaiting();
|
||||
|
||||
// The worker's first fleet_ask.
|
||||
CompletableFuture<MessageService.AskResult> ask1 =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "Q1", 5000));
|
||||
MessageService.TaskView asking1 = awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
assertEquals("Q1", asking1.reply());
|
||||
|
||||
// The primary answers it — answer() resumes the turn and blocks for what comes next.
|
||||
CompletableFuture<MessageService.Reply> answer1 = CompletableFuture.supplyAsync(
|
||||
() -> messages.answer(asking1.turnId(), "a1", 5000));
|
||||
assertEquals("a1", ask1.get(5, TimeUnit.SECONDS).answer());
|
||||
|
||||
// Still in the SAME resumed turn — before replying — the worker asks again.
|
||||
CompletableFuture<MessageService.AskResult> ask2 =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "Q2", 5000));
|
||||
|
||||
// answer1's own call unblocks with the second QUESTION (documented QUESTION-chaining
|
||||
// behaviour — see FleetMcp.answer's javadoc: "Answer it by calling fleet_send again with
|
||||
// turnId=..."). The bug: this used to also kill the async ticket in the process.
|
||||
MessageService.Reply firstAnswerResult = answer1.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, firstAnswerResult.outcome());
|
||||
String turnId2 = firstAnswerResult.turnId();
|
||||
|
||||
MessageService.TaskView asking2 = awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
assertEquals("Q2", asking2.reply(),
|
||||
"the ticket must surface the SECOND question, not be dead/FAILED");
|
||||
assertEquals(turnId2, asking2.turnId());
|
||||
|
||||
// The primary answers the second question; the worker finally sends its real fleet_reply.
|
||||
CompletableFuture<MessageService.Reply> answer2 = CompletableFuture.supplyAsync(
|
||||
() -> messages.answer(turnId2, "a2", 5000));
|
||||
assertEquals("a2", ask2.get(5, TimeUnit.SECONDS).answer());
|
||||
awaitWaiting();
|
||||
assertTrue(rendezvous.resolve(T, "done"));
|
||||
assertEquals(MessageService.Outcome.REPLIED, answer2.get(5, TimeUnit.SECONDS).outcome());
|
||||
|
||||
MessageService.TaskView done = awaitTicketPhase(ticket, MessageService.Phase.DONE);
|
||||
assertEquals("done", done.reply());
|
||||
}
|
||||
|
||||
// --- fleetd #329 (F2): an exception after the ticket's future completes must reach a log -----
|
||||
//
|
||||
// sendAsync's own executor task ends with catch (Throwable t) { task.future.completeExceptionally(t); }
|
||||
// — but finishAsyncTask completes that same future on its first line, so anything that throws
|
||||
// afterward hits an already-completed future: completeExceptionally returns false and does
|
||||
// nothing, and (before this fix) nothing logged it either. Measured on the ticket: temporarily
|
||||
// reintroducing the fleetd #324 NPE reproduced 19 real exceptions on the ordinary path with
|
||||
// 74/74 tests staying green and zero log lines. There is no reachable production call site where
|
||||
// finishAsyncTask throws after completing the future (fleetd #324 already closed the one that
|
||||
// used to), so this test drives the shape directly via a dedicated test-only hook
|
||||
// (afterFinishAsyncTaskCompleteHookForTest) rather than trying to engineer a real exception into
|
||||
// that narrow window — the same technique fleetd #324's own race test and this ticket's F1/F3
|
||||
// tests use for their own races.
|
||||
|
||||
private static ListAppender<ILoggingEvent> attachMessageServiceLog() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(MessageService.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detachMessageServiceLog(ListAppender<ILoggingEvent> appender) {
|
||||
((Logger) LoggerFactory.getLogger(MessageService.class)).detachAppender(appender);
|
||||
}
|
||||
|
||||
@Test
|
||||
void anExceptionAfterTheTicketFutureCompletesStillReachesTheLog() throws Exception {
|
||||
ListAppender<ILoggingEvent> appender = attachMessageServiceLog();
|
||||
try {
|
||||
messages.setAfterFinishAsyncTaskCompleteHookForTest(() -> {
|
||||
throw new RuntimeException("PROBE-329-F2");
|
||||
});
|
||||
|
||||
String ticket = messages.sendAsync(T, "do the task");
|
||||
awaitWaiting();
|
||||
assertTrue(rendezvous.resolve(T, "done"));
|
||||
|
||||
// finishAsyncTask's own first line already completed the future with the real result
|
||||
// BEFORE the hook threw — F2 is a visibility gap, not a correctness gap for the ticket
|
||||
// itself, so the ticket's own outcome must be unaffected by the swallowed exception.
|
||||
MessageService.TaskView done = awaitTicketPhase(ticket, MessageService.Phase.DONE);
|
||||
assertEquals("done", done.reply());
|
||||
|
||||
assertTrue(appender.list.stream().anyMatch(e ->
|
||||
e.getLevel() == Level.ERROR
|
||||
&& e.getFormattedMessage().contains(ticket)
|
||||
&& e.getThrowableProxy() != null
|
||||
&& "PROBE-329-F2".equals(e.getThrowableProxy().getMessage())),
|
||||
"an exception thrown after the ticket's future already completed must still "
|
||||
+ "reach the log, not vanish silently");
|
||||
} finally {
|
||||
messages.setAfterFinishAsyncTaskCompleteHookForTest(null);
|
||||
detachMessageServiceLog(appender);
|
||||
}
|
||||
}
|
||||
|
||||
// --- CB-582: fleet_status pendingAsk() ------------------------------------------------------
|
||||
|
||||
@Test
|
||||
@@ -1085,6 +1678,42 @@ class MessageServiceTest {
|
||||
}
|
||||
}
|
||||
|
||||
// --- fleetd #335 (site 2): task.future.whenComplete's own returned stage is discarded, so an --
|
||||
// uncaught throw from ReplyPushLoop.onTicketTerminal used to vanish with no log line and no
|
||||
// metric. The real production trigger is the daemon's own shutdown sequence (Fleetd's shutdown
|
||||
// hook): messages.close() only stops the async executor from taking NEW work — it does not
|
||||
// cancel a send already in flight — while pushLoop.close() shuts its scheduler down immediately
|
||||
// right after, so a ticket that completes in that narrow window has onTicketTerminal's own
|
||||
// scheduler.schedule(...) throw a real RejectedExecutionException. Reproduced here by shutting
|
||||
// the very same scheduler down before the ticket resolves — no test-only hook needed, this
|
||||
// reachable path throws for real.
|
||||
@Test
|
||||
void aTicketTerminalPushFailureDoesNotVanishSilently() throws Exception {
|
||||
ListAppender<ILoggingEvent> appender = attachMessageServiceLog();
|
||||
try (var wiring = wireWithPushLoop(1, 50)) {
|
||||
String ticket = wiring.service().sendAsync(T, "long task");
|
||||
awaitWaiting();
|
||||
wiring.scheduler().shutdownNow(); // simulate pushLoop.close() racing an in-flight send
|
||||
injectDelivery();
|
||||
assertTrue(rendezvous.resolve(T, "async result"));
|
||||
|
||||
// The ticket's own outcome must be unaffected by the swallowed exception — finishAsyncTask
|
||||
// completes task.future before whenComplete's action (and thus onTicketTerminal) ever runs.
|
||||
MessageService.TaskView done = awaitTicketPhaseOn(wiring.service(), ticket, MessageService.Phase.DONE);
|
||||
assertEquals("async result", done.reply());
|
||||
|
||||
assertTrue(appender.list.stream().anyMatch(e ->
|
||||
e.getLevel() == Level.ERROR
|
||||
&& e.getFormattedMessage().contains(ticket)
|
||||
&& e.getThrowableProxy() != null
|
||||
&& "java.util.concurrent.RejectedExecutionException"
|
||||
.equals(e.getThrowableProxy().getClassName())),
|
||||
"onTicketTerminal throwing must still reach the log, not vanish silently");
|
||||
} finally {
|
||||
detachMessageServiceLog(appender);
|
||||
}
|
||||
}
|
||||
|
||||
// --- CB-582: fleet_ask question-open nudges --------------------------------------------------
|
||||
|
||||
@Test
|
||||
@@ -1381,7 +2010,10 @@ class MessageServiceTest {
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_QUEUED, r.outcome());
|
||||
|
||||
assertTrue(messages.hasQueuedDelivery(T),
|
||||
"a TIMED_OUT_QUEUED send leaves the message still queued in the injector");
|
||||
"a TIMED_OUT_QUEUED send still records the undelivered delivery for fleet health");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
assertTrue(herdr.calls.stream().noneMatch(c -> c.method().equals("agent.prompt")),
|
||||
"a TIMED_OUT_QUEUED send must be cancelled, not delivered when the worker later goes idle");
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -1420,7 +2052,7 @@ class MessageServiceTest {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
|
||||
assertTrue(messages.reply(T, "resolved-live"));
|
||||
assertEquals(MessageService.ReplyOutcome.RESOLVED_SEND, messages.reply(T, "resolved-live"));
|
||||
assertFalse(messages.hasStrandedReply(T), "a reply that resolved an open send is not stranded");
|
||||
|
||||
MessageService.Reply r = send.get(5, TimeUnit.SECONDS);
|
||||
@@ -1430,14 +2062,14 @@ class MessageServiceTest {
|
||||
@Test
|
||||
void hasStrandedReplyIsTrueWhenNoSendWasWaiting() {
|
||||
// No send is open for T — the reply queues into the inbox and is recorded as stranded.
|
||||
assertTrue(messages.reply(T, "nobody was waiting"));
|
||||
assertEquals(MessageService.ReplyOutcome.QUEUED, messages.reply(T, "nobody was waiting"));
|
||||
assertTrue(messages.hasStrandedReply(T),
|
||||
"a reply with no open send strands, even though it is safely queued in the inbox");
|
||||
}
|
||||
|
||||
@Test
|
||||
void hasStrandedReplyClearsOnceTheTargetsNextDeliveryIsAccepted() throws Exception {
|
||||
assertTrue(messages.reply(T, "stray"));
|
||||
assertEquals(MessageService.ReplyOutcome.QUEUED, messages.reply(T, "stray"));
|
||||
assertTrue(messages.hasStrandedReply(T));
|
||||
|
||||
// The next accepted delivery for T clears the stale stranding fact — the one case the
|
||||
@@ -1455,7 +2087,7 @@ class MessageServiceTest {
|
||||
|
||||
@Test
|
||||
void hasStrandedReplyClearsOnAbandon() {
|
||||
assertTrue(messages.reply(T, "stray"));
|
||||
assertEquals(MessageService.ReplyOutcome.QUEUED, messages.reply(T, "stray"));
|
||||
assertTrue(messages.hasStrandedReply(T));
|
||||
|
||||
messages.abandon(T, "session released");
|
||||
|
||||
@@ -4,6 +4,7 @@ import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
@@ -44,6 +45,7 @@ class ReplyPushLoopTest {
|
||||
private static final String WORKER = "term_worker";
|
||||
private static final String WORKER2 = "term_worker2";
|
||||
private static final String OTHER_PRIMARY = "term_other_primary";
|
||||
private static final String DEAD_LEAD = "term_dead_lead";
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
|
||||
private PrimaryRegistry registry;
|
||||
@@ -207,6 +209,101 @@ class ReplyPushLoopTest {
|
||||
"exactly " + cap + " agent.prompt calls (cap=" + cap + ")");
|
||||
}
|
||||
|
||||
// --- fleetd #368: a lead's delegation binding must not outlive the lead ---------------------
|
||||
|
||||
/**
|
||||
* The bug: {@code PrimaryRegistry.forgetDelegation} is wired to a worker's release, never to
|
||||
* the delegating lead's own disappearance, so a lead that closed, crashed, or was relaunched
|
||||
* leaves {@code leadByTarget} pointing at a terminal herdr no longer knows. Before the fix,
|
||||
* {@code onReplyQueued} took that stale, non-null entry at face value — {@code nudgeTargetFor}
|
||||
* only ever falls back to the pinned primary when the map holds nothing for the target — so
|
||||
* the nudge's only schedule ran against the dead terminal forever and the live primary never
|
||||
* heard about the reply through this path.
|
||||
*
|
||||
* <p>This drives {@link ReplyPushLoop#onReplyQueued(String)} itself (not {@code PrimaryRegistry}
|
||||
* directly), because the registry lookup was never the defect — the caller trusting it without
|
||||
* checking liveness was. A test that only asserted on {@code PrimaryRegistry.nudgeTargetFor}
|
||||
* would pass whether or not {@code ReplyPushLoop} ever adopted the fix.
|
||||
*/
|
||||
@Test
|
||||
void aStaleLeadBindingFallsBackToTheLiveLeadInsteadOfNudgingADeadTerminal() throws Exception {
|
||||
// PRIMARY is the single known (pinned) lead — set up in @BeforeEach via `registry`.
|
||||
// DEAD_LEAD is a second lead that once delegated to WORKER and is now gone: herdr reports
|
||||
// agent_not_found for it, exactly as it would for a closed/crashed/relaunched terminal.
|
||||
registry.recordDelegation(WORKER, DEAD_LEAD);
|
||||
|
||||
var rec = new DeadLeadHerdrClient(DEAD_LEAD);
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
|
||||
loop(1, 50).onReplyQueued(WORKER);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS),
|
||||
"the nudge should still reach the live primary, not silently vanish with the dead lead");
|
||||
assertEquals(List.of(PRIMARY), rec.promptTargets(),
|
||||
"the nudge must be sent to the live primary, never to the dead lead's terminal");
|
||||
assertEquals(PRIMARY, registry.nudgeTargetFor(WORKER).orElseThrow(),
|
||||
"the stale binding must be forgotten (self-healed) once found dead, exactly like "
|
||||
+ "AgentControl.paneByTerminal already does on agent_not_found");
|
||||
}
|
||||
|
||||
/**
|
||||
* Same dead binding, but with no pinned primary to fall back to (the multi-lead, no-fallback
|
||||
* case {@code PrimaryRegistry.nudgeTargetFor}'s own javadoc already covers): the loop must
|
||||
* never nudge the dead terminal, and must not spin — no schedule starts at all once the stale
|
||||
* binding resolves to empty, same as if the map had never held an entry for this target.
|
||||
*/
|
||||
@Test
|
||||
void aStaleLeadBindingWithNoFallbackNeverNudgesTheDeadTerminal() throws Exception {
|
||||
var unpinned = new PrimaryRegistry(null);
|
||||
unpinned.recordDelegation(WORKER, DEAD_LEAD);
|
||||
|
||||
var rec = new DeadLeadHerdrClient(DEAD_LEAD);
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
|
||||
var loop = new ReplyPushLoop(unpinned, agents, inbox, scheduler, 1, 50);
|
||||
loop.onReplyQueued(WORKER);
|
||||
|
||||
Thread.sleep(200);
|
||||
assertEquals(0, rec.sendCount(), "no lead is live to nudge, so nothing should ever be sent");
|
||||
assertTrue(unpinned.nudgeTargetFor(WORKER).isEmpty(),
|
||||
"the stale binding must be forgotten even when there is no fallback to hand back");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #368 review, must-fix: the first version of {@code isLive} treated <em>any</em>
|
||||
* {@code RuntimeException} from the liveness probe as "the lead is gone" — indistinguishable
|
||||
* from a transient herdr hiccup (a socket blip, a decode error) on a lead that is actually
|
||||
* still live. The consequence of that misdiagnosis is destructive and permanent
|
||||
* ({@code forgetDelegation}), which is the exact #359 mistake repeated two days later: a single
|
||||
* bad reading must never destroy a live binding. This pins the narrower rule — only an
|
||||
* affirmative {@code agent_not_found} may forget a binding; a merely inconclusive failure must
|
||||
* leave the binding alone, and the lead must still be nudged once the probe recovers.
|
||||
*/
|
||||
@Test
|
||||
void aTransientLivenessFailureMustNotForgetABindingToAStillLiveLead() throws Exception {
|
||||
// OTHER_PRIMARY is delegated to and genuinely live — its FIRST agent.get call fails with a
|
||||
// transient, non-agent_not_found HerdrException (a transport-level failure, code null,
|
||||
// exactly what a socket blip looks like), then succeeds on every call after.
|
||||
registry.recordDelegation(WORKER, OTHER_PRIMARY);
|
||||
|
||||
var rec = new FlakyThenLiveHerdrClient(OTHER_PRIMARY);
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
|
||||
loop(2, 50).onReplyQueued(WORKER);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS),
|
||||
"the nudge must still reach the live lead once the transient failure clears");
|
||||
assertEquals(List.of(OTHER_PRIMARY), rec.promptTargets(),
|
||||
"the nudge must go to the lead that was only transiently unreachable, not the "
|
||||
+ "unrelated pinned primary");
|
||||
assertEquals(OTHER_PRIMARY, registry.nudgeTargetFor(WORKER).orElseThrow(),
|
||||
"a merely transient failure must not forget the binding to a lead that is actually "
|
||||
+ "still live");
|
||||
}
|
||||
|
||||
// --- nudge format --------------------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
@@ -887,7 +984,7 @@ class ReplyPushLoopTest {
|
||||
// --- metrics (CB-512) ----------------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void successfulNudgeIncrementsDelivered() throws Exception {
|
||||
void successfulNudgeIncrementsSent() throws Exception {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
@@ -897,11 +994,13 @@ class ReplyPushLoopTest {
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS),
|
||||
"one nudge (1 agent.prompt call) should have been sent");
|
||||
// The delivered count is bumped on the scheduler thread right after the send that releases
|
||||
// The sent count is bumped on the scheduler thread right after the send that releases
|
||||
// the latch — settle briefly so the counter is published before we read it.
|
||||
Thread.sleep(200);
|
||||
assertEquals(1, metrics.count(FleetMetrics.PUSH_NUDGES, "outcome", "delivered"),
|
||||
"a successfully sent nudge must count as delivered");
|
||||
// fleetd #365: "sent", not "delivered" — this only proves the herdr call succeeded, not
|
||||
// that the primary's pane read it.
|
||||
assertEquals(1, metrics.count(FleetMetrics.PUSH_NUDGES, "outcome", "sent"),
|
||||
"a successfully sent nudge must count as sent");
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -916,11 +1015,11 @@ class ReplyPushLoopTest {
|
||||
|
||||
assertEquals(1, metrics.count(FleetMetrics.PUSH_NUDGES, "outcome", "exhausted"),
|
||||
"hitting the reminder cap must count as exhausted");
|
||||
assertEquals(0, metrics.count(FleetMetrics.PUSH_NUDGES, "outcome", "delivered"));
|
||||
assertEquals(0, metrics.count(FleetMetrics.PUSH_NUDGES, "outcome", "sent"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void successfulTicketNudgeIncrementsDelivered() throws Exception {
|
||||
void successfulTicketNudgeIncrementsSent() throws Exception {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
Metrics metrics = new Metrics();
|
||||
@@ -929,8 +1028,8 @@ class ReplyPushLoopTest {
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS), "one ticket nudge should have been sent");
|
||||
Thread.sleep(200);
|
||||
assertEquals(1, metrics.count(FleetMetrics.PUSH_NUDGES, "outcome", "delivered"),
|
||||
"a successfully sent ticket nudge must count as delivered, same metric as CB-307");
|
||||
assertEquals(1, metrics.count(FleetMetrics.PUSH_NUDGES, "outcome", "sent"),
|
||||
"a successfully sent ticket nudge must count as sent, same metric as CB-307");
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -1098,4 +1197,101 @@ class ReplyPushLoopTest {
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fake herdr client for fleetd #368: {@code deadTarget} is a terminal herdr genuinely no
|
||||
* longer knows about — {@code agent.get} fails with {@code agent_not_found} exactly as
|
||||
* {@code AgentControl.agentCall} expects for a real dead/closed pane (see its javadoc). Every
|
||||
* other target reports {@code idle} (injectable). Records the {@code target} named by every
|
||||
* {@code agent.prompt} call, so a test can prove which terminal actually got nudged.
|
||||
*/
|
||||
private static final class DeadLeadHerdrClient implements HerdrClient {
|
||||
private final String deadTarget;
|
||||
private final List<String> promptTargets = Collections.synchronizedList(new ArrayList<>());
|
||||
volatile CountDownLatch sendLatch = new CountDownLatch(1);
|
||||
|
||||
DeadLeadHerdrClient(String deadTarget) {
|
||||
this.deadTarget = deadTarget;
|
||||
}
|
||||
|
||||
@Override
|
||||
@SuppressWarnings("unchecked")
|
||||
public JsonNode call(String method, Object params) {
|
||||
Map<String, Object> p = params instanceof Map ? (Map<String, Object>) params : Map.of();
|
||||
if ("agent.get".equals(method)) {
|
||||
String target = String.valueOf(p.get("target"));
|
||||
if (deadTarget.equals(target)) {
|
||||
throw new HerdrException("no such agent: " + target, "agent_not_found", null);
|
||||
}
|
||||
return MAPPER.createObjectNode()
|
||||
.set("agent", MAPPER.createObjectNode()
|
||||
.put("terminal_id", target)
|
||||
.put("agent_status", "idle"));
|
||||
}
|
||||
if ("agent.prompt".equals(method)) {
|
||||
promptTargets.add(String.valueOf(p.get("target")));
|
||||
sendLatch.countDown();
|
||||
}
|
||||
return MAPPER.createObjectNode();
|
||||
}
|
||||
|
||||
List<String> promptTargets() {
|
||||
return List.copyOf(promptTargets);
|
||||
}
|
||||
|
||||
long sendCount() {
|
||||
return promptTargets.size();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fake herdr client for fleetd #368 review: {@code flakyTarget}'s FIRST {@code agent.get} call
|
||||
* fails with a transient, non-{@code agent_not_found} {@code HerdrException} — a transport-level
|
||||
* failure (code {@code null}), exactly what a socket blip or a decode error on a perfectly live
|
||||
* lead looks like — then succeeds ({@code idle}) on every call after. Used to prove a merely
|
||||
* inconclusive failure must not be treated as the lead being gone.
|
||||
*/
|
||||
private static final class FlakyThenLiveHerdrClient implements HerdrClient {
|
||||
private final String flakyTarget;
|
||||
private final AtomicInteger getCalls = new AtomicInteger();
|
||||
private final List<String> promptTargets = Collections.synchronizedList(new ArrayList<>());
|
||||
volatile CountDownLatch sendLatch = new CountDownLatch(1);
|
||||
|
||||
FlakyThenLiveHerdrClient(String flakyTarget) {
|
||||
this.flakyTarget = flakyTarget;
|
||||
}
|
||||
|
||||
@Override
|
||||
@SuppressWarnings("unchecked")
|
||||
public JsonNode call(String method, Object params) {
|
||||
Map<String, Object> p = params instanceof Map ? (Map<String, Object>) params : Map.of();
|
||||
if ("agent.get".equals(method)) {
|
||||
String target = String.valueOf(p.get("target"));
|
||||
if (flakyTarget.equals(target) && getCalls.getAndIncrement() == 0) {
|
||||
throw new HerdrException("herdr socket read timed out"); // transport failure, code == null
|
||||
}
|
||||
return MAPPER.createObjectNode()
|
||||
.set("agent", MAPPER.createObjectNode()
|
||||
.put("terminal_id", target)
|
||||
.put("agent_status", "idle"));
|
||||
}
|
||||
if ("agent.prompt".equals(method)) {
|
||||
promptTargets.add(String.valueOf(p.get("target")));
|
||||
sendLatch.countDown();
|
||||
}
|
||||
return MAPPER.createObjectNode();
|
||||
}
|
||||
|
||||
List<String> promptTargets() {
|
||||
return List.copyOf(promptTargets);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+144
@@ -0,0 +1,144 @@
|
||||
package dev.ltms.fleet.peer;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.lang.reflect.Constructor;
|
||||
import java.lang.reflect.RecordComponent;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Arrays;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* Fleetd #382, the same "defect factory" #357 and #358 guarded on {@code FleetConfig.withDefaults()}
|
||||
* and {@link dev.ltms.fleet.session.MemberSession}'s rebuild sites — reproduced here on
|
||||
* {@link SpawnRequest#withProfile(String)}.
|
||||
*
|
||||
* <p>{@code SpawnRequest} has back-compat constructors at arity 3 and 5 alongside its canonical
|
||||
* arity-6 constructor. Before this ticket, {@code CompositePeerLauncher} routed a profile by
|
||||
* building a fresh {@code SpawnRequest} from a literal {@code new SpawnRequest(...)} call listing
|
||||
* six of the original request's own accessors. That call is only correct because it happens to
|
||||
* name exactly six arguments today — add a 7th component and the established back-compat pattern
|
||||
* (a new constructor at the old, now-shorter arity) and a call one argument short of the new
|
||||
* canonical arity would silently rebind to that back-compat constructor, dropping the new
|
||||
* component on every profile-routed spawn without any compile error. {@link #withProfile} replaces
|
||||
* that literal call, so this test guards the ONE rebuild site instead of a call site scattered
|
||||
* through a launcher.
|
||||
*
|
||||
* <p>The check below builds a {@link SpawnRequest} through the TRUE canonical constructor —
|
||||
* resolved by the record's own component types via {@code getDeclaredConstructor}, never by
|
||||
* argument count, so it can never itself land on a back-compat overload — with a real, distinctive,
|
||||
* non-null value in every component, calls {@link SpawnRequest#withProfile(String)}, and asserts
|
||||
* every component the method is not documented to change survives unchanged, while {@code
|
||||
* profileName} comes back as the new value it was given. A component that comes back anything else
|
||||
* was silently dropped — the shape of the defect this test exists to catch.
|
||||
*
|
||||
* <p>{@link #EXCLUDED_FROM_SURVIVAL_CHECK} is kept deliberately empty and size-pinned by
|
||||
* {@link #exclusionListSizeIsPinned()}, for the same reason the other two guards pin theirs at
|
||||
* zero: a checker whose escape hatch can grow to silence a failure is not a checker. Every one of
|
||||
* {@link SpawnRequest}'s 6 current components has a real, non-null value here and none is excluded.
|
||||
*/
|
||||
class SpawnRequestWithProfilePreservesEveryComponentTest {
|
||||
|
||||
private static final RecordComponent[] COMPONENTS = SpawnRequest.class.getRecordComponents();
|
||||
|
||||
/** Deliberately empty today; grow it only with a matching justification, and re-pin the size. */
|
||||
private static final Set<String> EXCLUDED_FROM_SURVIVAL_CHECK = Set.of();
|
||||
|
||||
/** One real, distinctive, non-null value per component — none of the 6 is excluded. */
|
||||
private static Map<String, Object> baseValues() {
|
||||
Map<String, Object> v = new LinkedHashMap<>();
|
||||
v.put("profileName", "profile-guard");
|
||||
v.put("requestedCwd", "/wt/requested-guard");
|
||||
v.put("callerCwd", "/wt/caller-guard");
|
||||
v.put("sessionName", "session-guard");
|
||||
v.put("resumeSessionId", "resume-guard");
|
||||
v.put("role", MemberRole.REVIEWER);
|
||||
assertNamesMatchComponents(v);
|
||||
return v;
|
||||
}
|
||||
|
||||
/**
|
||||
* Guards {@link #baseValues()} itself against drifting from the record's real shape — forgetting
|
||||
* to add a new component here fails this assertion by name, rather than silently checking one
|
||||
* component fewer than the record has.
|
||||
*/
|
||||
private static void assertNamesMatchComponents(Map<String, Object> values) {
|
||||
Set<String> names = new TreeSet<>();
|
||||
for (RecordComponent rc : COMPONENTS) {
|
||||
names.add(rc.getName());
|
||||
}
|
||||
assertEquals(names, new TreeSet<>(values.keySet()),
|
||||
"this test's value map has drifted from SpawnRequest's actual components — "
|
||||
+ "update baseValues() alongside the record");
|
||||
}
|
||||
|
||||
/**
|
||||
* Builds a {@link SpawnRequest} through the TRUE canonical constructor — resolved by the
|
||||
* record's own component types, not by argument count — so this never accidentally exercises a
|
||||
* back-compat overload the way a literal {@code new SpawnRequest(...)} call risks doing.
|
||||
*/
|
||||
private static SpawnRequest requestOf(Map<String, Object> values) throws ReflectiveOperationException {
|
||||
Class<?>[] types = Arrays.stream(COMPONENTS).map(RecordComponent::getType).toArray(Class<?>[]::new);
|
||||
Object[] args = Arrays.stream(COMPONENTS).map(rc -> values.get(rc.getName())).toArray();
|
||||
Constructor<SpawnRequest> ctor = SpawnRequest.class.getDeclaredConstructor(types);
|
||||
return ctor.newInstance(args);
|
||||
}
|
||||
|
||||
@Test
|
||||
void exclusionListSizeIsPinned() {
|
||||
assertEquals(0, EXCLUDED_FROM_SURVIVAL_CHECK.size(),
|
||||
"EXCLUDED_FROM_SURVIVAL_CHECK grew from 0 — every entry needs a justification in "
|
||||
+ "this test class's javadoc AND this assertion re-pinned to the new size; a "
|
||||
+ "growing exclusion list that silences failures on its own is not a guard");
|
||||
}
|
||||
|
||||
@Test
|
||||
void withProfilePreservesEveryOtherComponent() throws ReflectiveOperationException {
|
||||
Map<String, Object> base = baseValues();
|
||||
SpawnRequest request = requestOf(base);
|
||||
SpawnRequest result = request.withProfile("profile-updated");
|
||||
|
||||
Map<String, Object> expectedOverrides = Map.of("profileName", "profile-updated");
|
||||
|
||||
List<String> dropped = new ArrayList<>();
|
||||
int checked = 0;
|
||||
for (RecordComponent rc : COMPONENTS) {
|
||||
String name = rc.getName();
|
||||
if (EXCLUDED_FROM_SURVIVAL_CHECK.contains(name)) {
|
||||
continue;
|
||||
}
|
||||
checked++;
|
||||
Object expected = expectedOverrides.containsKey(name) ? expectedOverrides.get(name) : base.get(name);
|
||||
Object actual;
|
||||
try {
|
||||
actual = rc.getAccessor().invoke(result);
|
||||
} catch (ReflectiveOperationException e) {
|
||||
throw new RuntimeException("failed to read SpawnRequest." + name + "()", e);
|
||||
}
|
||||
if (!Objects.equals(expected, actual)) {
|
||||
dropped.add(String.format(Locale.ROOT,
|
||||
"%s: withProfile() was expected to carry (%s) for '%s' but returned %s — a "
|
||||
+ "component silently dropped by withProfile(), the shape of the defect "
|
||||
+ "this test exists to catch (its final \"return new SpawnRequest(...)\" "
|
||||
+ "call binding to a back-compat constructor instead of the true "
|
||||
+ "canonical one)",
|
||||
name, expected, name, actual));
|
||||
}
|
||||
}
|
||||
|
||||
System.out.printf(Locale.ROOT,
|
||||
"SpawnRequest.withProfile() component-survival coverage — %d components, %d checked, "
|
||||
+ "%d excluded, %d survived%n",
|
||||
COMPONENTS.length, checked, EXCLUDED_FROM_SURVIVAL_CHECK.size(), checked - dropped.size());
|
||||
assertEquals(List.of(), dropped,
|
||||
"withProfile() silently dropped these components: " + dropped);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* Platform-detection unit tests for {@link CaffeinateSleepAssertionMechanism}.
|
||||
*
|
||||
* <p>This deliberately never calls {@link CaffeinateSleepAssertionMechanism#acquire()} itself —
|
||||
* doing so on a real macOS machine would actually start a live {@code caffeinate} child and hold
|
||||
* a real idle-sleep assertion, which the ticket this class exists for explicitly forbids testing
|
||||
* with. Instead this exercises the pure {@code isSupportedPlatform(String)} predicate that
|
||||
* {@code acquire()} consults before ever touching {@link ProcessBuilder} — so it proves the
|
||||
* platform check itself is correct on any CI OS, but it does <strong>not</strong> prove that a
|
||||
* real {@code caffeinate -i} spawn succeeds or that its child is torn down correctly; that half is
|
||||
* exercised indirectly by {@link IdleSleepGuardTest} against a {@link FakeSleepAssertionMechanism}
|
||||
* instead, which is the seam invariant 2/3 in the ticket call for.
|
||||
*/
|
||||
class CaffeinateSleepAssertionMechanismTest {
|
||||
|
||||
@Test
|
||||
void macOsNamesAreSupported() {
|
||||
assertTrue(CaffeinateSleepAssertionMechanism.isSupportedPlatform("Mac OS X"));
|
||||
assertTrue(CaffeinateSleepAssertionMechanism.isSupportedPlatform("macOS"));
|
||||
assertTrue(CaffeinateSleepAssertionMechanism.isSupportedPlatform("MAC OS X"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nonMacNamesAreNotSupported() {
|
||||
assertFalse(CaffeinateSleepAssertionMechanism.isSupportedPlatform("Linux"));
|
||||
assertFalse(CaffeinateSleepAssertionMechanism.isSupportedPlatform("Windows 11"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullOsNameIsNotSupported() {
|
||||
assertFalse(CaffeinateSleepAssertionMechanism.isSupportedPlatform(null));
|
||||
}
|
||||
|
||||
/**
|
||||
* The overload {@code isSupportedPlatform()} (no args) reads the JVM's real {@code os.name} —
|
||||
* proves the wiring is live, without asserting a specific answer (this suite itself must pass
|
||||
* on both macOS and Linux CI).
|
||||
*/
|
||||
@Test
|
||||
void noArgOverloadReadsRealSystemProperty() {
|
||||
boolean expected = CaffeinateSleepAssertionMechanism
|
||||
.isSupportedPlatform(System.getProperty("os.name"));
|
||||
boolean actual = CaffeinateSleepAssertionMechanism.isSupportedPlatform();
|
||||
assertEquals(expected, actual);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,56 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
|
||||
/**
|
||||
* Recording fake {@link SleepAssertionMechanism} — the seam behind the real OS effect (a live
|
||||
* {@code caffeinate} child process). No test in this package ever spawns that real process; every
|
||||
* assertion here is against this fake's own call log instead.
|
||||
*
|
||||
* <p>Each acquired {@link FakeAssertion} records its own {@code close()} calls, and every
|
||||
* acquired instance is kept in {@link #acquired} so a test can inspect all of them, including
|
||||
* ones {@link IdleSleepGuard} has already released.
|
||||
*/
|
||||
final class FakeSleepAssertionMechanism implements SleepAssertionMechanism {
|
||||
|
||||
/** Every {@link FakeAssertion} this mechanism has ever handed out, in order. */
|
||||
final CopyOnWriteArrayList<FakeAssertion> acquired = new CopyOnWriteArrayList<>();
|
||||
|
||||
private final AtomicInteger acquireCalls = new AtomicInteger();
|
||||
private volatile boolean unavailable = false;
|
||||
|
||||
/** Make the next (and every subsequent) {@link #acquire()} return {@code null}, like a missing tool. */
|
||||
void makeUnavailable() {
|
||||
unavailable = true;
|
||||
}
|
||||
|
||||
int acquireCallCount() {
|
||||
return acquireCalls.get();
|
||||
}
|
||||
|
||||
@Override
|
||||
public SleepAssertion acquire() {
|
||||
acquireCalls.incrementAndGet();
|
||||
if (unavailable) {
|
||||
return null;
|
||||
}
|
||||
FakeAssertion a = new FakeAssertion();
|
||||
acquired.add(a);
|
||||
return a;
|
||||
}
|
||||
|
||||
/** A held fake assertion; records how many times {@code close()} was actually called. */
|
||||
static final class FakeAssertion implements SleepAssertion {
|
||||
private final AtomicInteger closeCalls = new AtomicInteger();
|
||||
|
||||
int closeCallCount() {
|
||||
return closeCalls.get();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
closeCalls.incrementAndGet();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,118 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* {@link IdleSleepGuard} against a {@link FakeSleepAssertionMechanism} — the seam that stands in
|
||||
* for a real {@code caffeinate} child process. No test in this class ever spawns a real OS
|
||||
* process or asserts against real idle sleep; every assertion is against the fake's call log
|
||||
* (how many times {@code acquire()}/{@code close()} were actually called). That proves the
|
||||
* <em>orchestration</em> — when the guard decides to hold or release an assertion, and that it
|
||||
* never throws — but it does <strong>not</strong> prove that {@code caffeinate -i} itself
|
||||
* actually stops macOS from idle-sleeping; that half is outside what a unit test can safely
|
||||
* exercise (see {@link CaffeinateSleepAssertionMechanismTest}'s class doc).
|
||||
*/
|
||||
class IdleSleepGuardTest {
|
||||
|
||||
@Test
|
||||
void acquiresOnZeroToOneAndReleasesOnOneToZero() {
|
||||
FakeSleepAssertionMechanism mechanism = new FakeSleepAssertionMechanism();
|
||||
AtomicInteger liveCount = new AtomicInteger(0);
|
||||
IdleSleepGuard guard = new IdleSleepGuard(mechanism, liveCount::get);
|
||||
|
||||
assertFalse(guard.isHeld(), "nothing held before any member is live");
|
||||
|
||||
liveCount.set(1);
|
||||
guard.recheck();
|
||||
assertTrue(guard.isHeld(), "an assertion must be held once a member is live");
|
||||
assertEquals(1, mechanism.acquired.size());
|
||||
assertEquals(0, mechanism.acquired.get(0).closeCallCount());
|
||||
|
||||
liveCount.set(0);
|
||||
guard.recheck();
|
||||
assertFalse(guard.isHeld(), "the assertion must be released once the last member goes");
|
||||
assertEquals(1, mechanism.acquired.get(0).closeCallCount(), "the SAME held assertion must be closed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void steadyLiveCountDoesNotReacquireOrRerelease() {
|
||||
FakeSleepAssertionMechanism mechanism = new FakeSleepAssertionMechanism();
|
||||
AtomicInteger liveCount = new AtomicInteger(2);
|
||||
IdleSleepGuard guard = new IdleSleepGuard(mechanism, liveCount::get);
|
||||
|
||||
guard.recheck(); // 0 -> 2 crossing: acquires
|
||||
guard.recheck(); // still 2: must be a no-op
|
||||
guard.recheck(); // still 2: must be a no-op
|
||||
assertEquals(1, mechanism.acquireCallCount(), "only the crossing touches the mechanism");
|
||||
|
||||
liveCount.set(1); // 2 -> 1: still > 0, still a no-op
|
||||
guard.recheck();
|
||||
assertTrue(guard.isHeld());
|
||||
assertEquals(0, mechanism.acquired.get(0).closeCallCount());
|
||||
assertEquals(1, mechanism.acquireCallCount());
|
||||
}
|
||||
|
||||
/**
|
||||
* Invariant 2: a missing/unavailable mechanism must never throw, and the guard must simply
|
||||
* hold nothing. {@link FakeSleepAssertionMechanism#makeUnavailable()} makes {@code acquire()}
|
||||
* return {@code null}, exactly like {@link CaffeinateSleepAssertionMechanism} does off macOS
|
||||
* or when the {@code caffeinate} binary is missing.
|
||||
*/
|
||||
@Test
|
||||
void unavailableMechanismNeverThrowsAndHoldsNothing() {
|
||||
FakeSleepAssertionMechanism mechanism = new FakeSleepAssertionMechanism();
|
||||
mechanism.makeUnavailable();
|
||||
AtomicInteger liveCount = new AtomicInteger(1);
|
||||
IdleSleepGuard guard = new IdleSleepGuard(mechanism, liveCount::get);
|
||||
|
||||
guard.recheck(); // must not throw
|
||||
assertFalse(guard.isHeld(), "acquire() returned null, so nothing is held");
|
||||
assertEquals(1, mechanism.acquireCallCount());
|
||||
|
||||
// still must not throw or leak on release, even though nothing was ever actually held
|
||||
liveCount.set(0);
|
||||
guard.recheck();
|
||||
assertFalse(guard.isHeld());
|
||||
|
||||
guard.close(); // teardown with nothing held must also be a safe no-op
|
||||
}
|
||||
|
||||
/**
|
||||
* Invariant 3 (teardown). This is the test the mutation testing step removes the production
|
||||
* release call to fail: with {@code releaseHeldLocked()} not invoked from {@link
|
||||
* IdleSleepGuard#close()}, the held fake assertion's {@code close()} would never be called and
|
||||
* this assertion would fail.
|
||||
*/
|
||||
@Test
|
||||
void closeReleasesAHeldAssertionEvenWithoutAZeroCrossing() {
|
||||
FakeSleepAssertionMechanism mechanism = new FakeSleepAssertionMechanism();
|
||||
AtomicInteger liveCount = new AtomicInteger(1);
|
||||
IdleSleepGuard guard = new IdleSleepGuard(mechanism, liveCount::get);
|
||||
|
||||
guard.recheck();
|
||||
assertTrue(guard.isHeld());
|
||||
|
||||
guard.close();
|
||||
|
||||
assertFalse(guard.isHeld(), "close() must release whatever is held, independent of live count");
|
||||
assertEquals(1, mechanism.acquired.get(0).closeCallCount());
|
||||
}
|
||||
|
||||
@Test
|
||||
void closeIsIdempotent() {
|
||||
FakeSleepAssertionMechanism mechanism = new FakeSleepAssertionMechanism();
|
||||
AtomicInteger liveCount = new AtomicInteger(1);
|
||||
IdleSleepGuard guard = new IdleSleepGuard(mechanism, liveCount::get);
|
||||
|
||||
guard.recheck();
|
||||
guard.close();
|
||||
guard.close(); // must not throw, must not double-release
|
||||
assertEquals(1, mechanism.acquired.get(0).closeCallCount());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,72 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* Proves the wiring {@code Fleetd.main} actually performs — {@code
|
||||
* sessions.onAcquire(_ -> guard.recheck())} / {@code sessions.onRelease(_ -> guard.recheck())} —
|
||||
* not just {@link IdleSleepGuard}'s own orchestration logic in isolation
|
||||
* ({@link IdleSleepGuardTest} already covers that in isolation, which on its own would not catch
|
||||
* a wiring gap — e.g. an {@code onAcquire} call typo'd to a no-op lambda, or the listener wired to
|
||||
* the wrong SessionManager instance — see fleetd's own "a test on the seam does not prove the
|
||||
* caller" lesson). This test builds a real {@link SessionManager} exactly as
|
||||
* {@code SessionManagerTest} does (a {@link FakeHerdr}-backed {@link ClaudeCodeLauncher}, no live
|
||||
* herdr process), wires it to an {@link IdleSleepGuard} the same two lines {@code Fleetd.main}
|
||||
* uses, and drives real {@link SessionManager#acquire} / {@link SessionManager#release} calls.
|
||||
*/
|
||||
class IdleSleepGuardWiringTest {
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr) {
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
return new SessionManager(workers);
|
||||
}
|
||||
|
||||
@Test
|
||||
void acquiringAndReleasingRealSessionsDrivesTheGuardThroughTheSameWiringFleetdUses() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
FakeSleepAssertionMechanism mechanism = new FakeSleepAssertionMechanism();
|
||||
IdleSleepGuard guard = new IdleSleepGuard(mechanism, sessions::size);
|
||||
|
||||
// The exact two lines Fleetd.main wires up.
|
||||
sessions.onAcquire(_ -> guard.recheck());
|
||||
sessions.onRelease(_ -> guard.recheck());
|
||||
|
||||
assertFalse(guard.isHeld(), "no member yet: nothing held");
|
||||
|
||||
MemberSession a = sessions.acquire("ltms-local", "/a", "/caller", "ownerA");
|
||||
assertTrue(guard.isHeld(), "0 -> 1: the first live member must arm the guard");
|
||||
|
||||
MemberSession b = sessions.acquire("ltms-local", "/b", "/caller", "ownerB");
|
||||
assertEquals(1, mechanism.acquireCallCount(), "2nd member: still just 1 live-to-2 step, no new acquire");
|
||||
|
||||
sessions.release(a.paneId());
|
||||
assertTrue(guard.isHeld(), "one member still live: the guard must stay armed");
|
||||
assertEquals(0, mechanism.acquired.get(0).closeCallCount());
|
||||
|
||||
sessions.release(b.paneId());
|
||||
assertFalse(guard.isHeld(), "1 -> 0: the last member releasing must disarm the guard");
|
||||
assertEquals(1, mechanism.acquired.get(0).closeCallCount());
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user