Compare commits
239 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| d9fedfbc5a | |||
| ac535503ff | |||
| 654b3b5e14 | |||
| 1fae8b81a0 | |||
| 4ef815682b | |||
| 6596458ce6 | |||
| 1cc0322782 | |||
| c043d149cf | |||
| fc786d0f67 | |||
| b314cb4d51 | |||
| a3d296f639 | |||
| 79423787d0 | |||
| 364b229db9 | |||
| ac436efefb | |||
| 7dad054045 | |||
| 4e414c475f | |||
| 9084667493 | |||
| 2289e94223 | |||
| 92adfcfae5 | |||
| 5f7f388e69 | |||
| 39accf73e6 | |||
| 5c2f296bc3 | |||
| 0f2ec7a6b5 | |||
| 02eff4c532 | |||
| 92b6d9406b | |||
| c4498607e1 | |||
| e42eab5b4c | |||
| b1d2cb48ac | |||
| 001367d82c | |||
| 9fdcaaa8fd | |||
| 459a523e2c | |||
| 291dc02c77 | |||
| be835aa259 | |||
| 80506c79d0 | |||
| 6677ec8c63 | |||
| ca0c965932 | |||
| e8ab933cbc | |||
| 2adb950a12 | |||
| 7f0c4a8464 | |||
| 682991a846 | |||
| abe617c48c | |||
| 886ce1521d | |||
| d41aff4012 | |||
| 8a1d73b39e | |||
| 70a735b638 | |||
| 803c91ea6c | |||
| d438a74575 | |||
| 7467ffa252 | |||
| f4176ae455 | |||
| 6ab3a81af7 | |||
| 8fa8d18c97 | |||
| 9d653e86df | |||
| f6d1131d7a | |||
| a0505dc614 | |||
| d2f30f1654 | |||
| cd1f04cbb4 | |||
| 73137f198f | |||
| 4ffe49f3bb | |||
| efd9cdb983 | |||
| aabecce901 | |||
| 6754b4edbc | |||
| 787ae0ed7a | |||
| e3050efe8b | |||
| 11998cd626 | |||
| 8e5394f63f | |||
| 428a12af62 | |||
| a332dfdb2c | |||
| d0f4ae057b | |||
| 7b3beaa209 | |||
| b3b2bf3da6 | |||
| 8bb2aa0be4 | |||
| 4ce3149bfd | |||
| 1fc9e85bf1 | |||
| 38544d467c | |||
| 809b7d9b20 | |||
| 8cf7215d56 | |||
| 1ef93e57cc | |||
| cf0c9b9316 | |||
| 337dbd491e | |||
| 7cf6075b79 | |||
| fbdcd709c9 | |||
| 8d3f10d291 | |||
| 38f4fd64ee | |||
| 3fc39b981d | |||
| 70328ca0f8 | |||
| efab9b8c49 | |||
| 2374de28e4 | |||
| dd18bd1f38 | |||
| efeffb4ab7 | |||
| ed4f4b08ad | |||
| 28a1f3d6f5 | |||
| b12d70716b | |||
| 0f5985b419 | |||
| 6e06058b07 | |||
| e5f4fb81ab | |||
| d28ab0968b | |||
| 9a64d42599 | |||
| f4e0ca41e6 | |||
| 29a2f97c25 | |||
| d2db8c7dc9 | |||
| 95311c6e8e | |||
| 10ab58e4fc | |||
| 0ba597e394 | |||
| c468953963 | |||
| 25d53e6ef7 | |||
| a9a37af957 | |||
| 7d497aa423 | |||
| b205bcc2aa | |||
| 11eccc3a1b | |||
| 4939d40362 | |||
| e7a7711a4e | |||
| 1fd2cfa716 | |||
| 3e8e314656 | |||
| 20fc42b572 | |||
| f82073717a | |||
| 16b52fac6c | |||
| 13b6ae2628 | |||
| 7e838ba8b9 | |||
| c3e3554bde | |||
| b5bc5d4ab5 | |||
| d83972ede2 | |||
| 02c6909546 | |||
| b92a669ddc | |||
| bb29b001e4 | |||
| f0ff25221e | |||
| 780cb342ad | |||
| 5c563f02c8 | |||
| ad593c9bb9 | |||
| 736fd9cf4b | |||
| 05244a82b3 | |||
| ef4996a01e | |||
| edbd8d816a | |||
| 9425a9b696 | |||
| d0688c8a60 | |||
| 2c467c2553 | |||
| c6430d8edd | |||
| bfee23acc3 | |||
| 7dec74f1b4 | |||
| 482598e2a6 | |||
| 5f5d16fbd4 | |||
| 7df7985a16 | |||
| 724b35b46e | |||
| 2eb2d6112e | |||
| 7c458e8bf2 | |||
| 133f03e428 | |||
| 804279175d | |||
| 9dea289975 | |||
| cb4a6869b9 | |||
| 28b45d97e5 | |||
| 1a397e962e | |||
| 209e1231ea | |||
| 5d8b9d365c | |||
| a6aeda39e7 | |||
| 31b3c24caa | |||
| d105da978d | |||
| 5051a06443 | |||
| a134eccc57 | |||
| 6f275227d2 | |||
| 283ccf8423 | |||
| b96fba4a03 | |||
| 6d97d210b4 | |||
| 37b23cd704 | |||
| 367facf6a6 | |||
| 7f9a9c09f9 | |||
| e854957247 | |||
| b4b7cf5155 | |||
| cbb35ad947 | |||
| 656588f597 | |||
| e33377b2ca | |||
| 4b4a8688c2 | |||
| 7e48d4b86c | |||
| 41cc785534 | |||
| 60fa86a107 | |||
| f288cee2bb | |||
| a5d6ce1a37 | |||
| a52ca35d34 | |||
| 9417de1123 | |||
| 03d92be751 | |||
| 136bec8e28 | |||
| ae94d511d7 | |||
| 436b026696 | |||
| 905fa3a454 | |||
| 011ee80067 | |||
| 3fab743152 | |||
| 52eb9c2277 | |||
| 6794fd8200 | |||
| 9b5c1cdcff | |||
| 3b69e0103b | |||
| c97b1bba5a | |||
| a42253f597 | |||
| e69eafcc9f | |||
| a42b12440c | |||
| a06426c33c | |||
| 28ea0de575 | |||
| 91792e11fc | |||
| 6539efe9fa | |||
| 68397f78d5 | |||
| af901ff1d2 | |||
| dac5f88812 | |||
| 2e663e5968 | |||
| 8c14ed2846 | |||
| 141ae3b04d | |||
| faefea14c4 | |||
| ea6896f2ef | |||
| d7f94cafa2 | |||
| 096f08c866 | |||
| 4eb720029c | |||
| 0db6d31dc2 | |||
| 158a2a84b5 | |||
| 41ebc9cf69 | |||
| 619769bb52 | |||
| 180c953c42 | |||
| 4af919af0c | |||
| 09c37061c1 | |||
| 2be287ea03 | |||
| c3b0406826 | |||
| e76fa1660b | |||
| 9dca376604 | |||
| 91be5079d6 | |||
| 26f198675c | |||
| 72f46d7c0c | |||
| 640f4d5f23 | |||
| 6edeb70bc4 | |||
| bc49d87cb8 | |||
| 14b169c410 | |||
| 2b52324d9a | |||
| b6147a39f6 | |||
| dbfc34cb6d | |||
| a20cb96730 | |||
| cda1a6a917 | |||
| 608e4496be | |||
| fa61dc587c | |||
| 526e3b7459 | |||
| b6b006c651 | |||
| 63eec8a0da | |||
| cbe872b538 | |||
| 7d9a807243 | |||
| 8915e40c7d | |||
| 3f7bc3815e |
@@ -59,7 +59,7 @@ as `matches HEAD`, `drift`, or `unknown`; do not turn an unclear timestamp into
|
||||
Report the process identifier (PID) and uptime too:
|
||||
|
||||
```bash
|
||||
PIDS="$(pgrep -f 'target/fleetd.jar' || true)"
|
||||
PIDS="$(pgrep -f 'fleetd.jar' || true)"
|
||||
if [ -z "$PIDS" ]; then
|
||||
printf '%s\n' 'fleetd: not running'
|
||||
else
|
||||
|
||||
@@ -13,7 +13,7 @@ writes a handover file, and the new session reads that file and carries on.
|
||||
- **By hand.** You write the file, then tell the operator where it is. The operator starts the new
|
||||
session and points it at the file. This always works.
|
||||
- **With `fleet_handover`** (fleetd #480, merged 2026-09-11). You ask fleetd to do the swap: it
|
||||
checks the file, clears your pane, and tells the fresh session to read it. This needs
|
||||
checks the file, restarts your pane, and tells the fresh session to read it. This needs
|
||||
`leadRollover:` in `fleetd.yaml`; without it every action answers a clean refusal naming
|
||||
`NOT_CONFIGURED`, and you fall back to the manual path. Section 11 below is the procedure.
|
||||
|
||||
@@ -133,8 +133,17 @@ A handover file is a record of state and decisions. It is not a diary.
|
||||
**Run the three steps in this order. The order is not a style choice — the wrong order is
|
||||
refused.**
|
||||
|
||||
1. **`fleet_handover{action: "open", reason: "<why now>"}`.** It returns a `token` and the
|
||||
`handoverPath` you must write to. Nothing has happened to your pane yet.
|
||||
1. **`fleet_handover{action: "open", reason: "<why now>"}`.** It returns a `token`, the
|
||||
`handoverPath` you must write to, and `outstandingTickets` plus `openAsks`. Nothing has happened
|
||||
to your pane yet.
|
||||
|
||||
**Copy `outstandingTickets` and `openAsks` into the handover file.** Your successor keeps the
|
||||
authority to poll those tickets and answer those asks, because both are gated on the lead's
|
||||
name, which does not change when your pane does. What it does not keep is the ids — they exist
|
||||
only in your context and in this response. A ticket already in a terminal phase is the urgent
|
||||
one: its reply lives only in memory and is deleted once the ticket TTL passes, so an uncollected
|
||||
report is lost for good. Poll those before you confirm, or name them in the file so your
|
||||
successor polls them first.
|
||||
|
||||
**Write to exactly that path, and do not resolve it yourself.** It is always absolute, even when
|
||||
the operator configured a relative `handoverPath`: fleetd resolves a relative one against your
|
||||
@@ -159,27 +168,104 @@ fails.
|
||||
|
||||
`{action: "cancel", token}` drops a pending request without rolling.
|
||||
|
||||
**Things that will surprise you:**
|
||||
## 12. A roll restarts your process
|
||||
|
||||
- **`accepted` does not mean your pane has been cleared.** It means every gate passed and the roll
|
||||
The daemon ends your pane, launches a fresh one, waits for the new terminal to be recognised as a
|
||||
lead, and only then sends the bootstrap text. Your `claude` process really exits, so a newer CLI on
|
||||
disk is loaded. It does not type `/clear`.
|
||||
|
||||
**The restart path is live in the code but has not run yet.** Measured 2026-10-05: the log holds no
|
||||
`lead-rollover:` line since the current daemon started, and no occurrence of any new outcome name.
|
||||
All 20 rolls recorded further down ran under the older `/clear` behaviour, so read them as history
|
||||
rather than as evidence about your own roll. Re-measure with:
|
||||
|
||||
```bash
|
||||
grep -c "lead-rollover: rolled" fleetd/fleetd.out # successful rolls
|
||||
grep -c "lead-rollover:" fleetd/fleetd.out # positive control: must be larger
|
||||
```
|
||||
|
||||
Run the control line too. A broken pattern returns a clean `0` that reads exactly like good news.
|
||||
If the first number has grown past 20, somebody has rolled under the restart path, and this section
|
||||
should be replaced with what they measured.
|
||||
|
||||
**Three separate timeouts bound a roll.** `leadRollover.relaunchReadySeconds` (default 45,
|
||||
`FleetConfig.java:1483`) bounds **each** of two waits that run after the relaunch, so the worst case
|
||||
there is about twice that number, not 45 seconds in total. A third bound gives your old pane 10
|
||||
seconds to die (`LeadRollover.PANE_DEATH_TIMEOUT_SECONDS`).
|
||||
|
||||
`fleet_handover{action: "status", token}` answers with one of these:
|
||||
|
||||
| Outcome | What it means |
|
||||
|---|---|
|
||||
| `IN_PROGRESS` | still running; it always ends on one of the rows below |
|
||||
| `ROLLED` | the roll succeeded |
|
||||
| `TURN_NEVER_SETTLED` | your turn ran past `leadRollover.turnSettleSeconds`; nothing was touched |
|
||||
| `OLD_PANE_NEVER_DIED` | your pane did not exit within the 10-second bound |
|
||||
| `RELAUNCH_FAILED` | launching the fresh pane failed |
|
||||
| `RELAUNCH_NEVER_READY` | the fresh pane never became ready within `relaunchReadySeconds` |
|
||||
| `RELAUNCH_NOT_RECOGNISED` | the fresh terminal never resolved as a lead |
|
||||
| `FAILED` | the roll threw; `runRollover`'s catch records this rather than leaving it stuck |
|
||||
|
||||
Only `TURN_NEVER_SETTLED` guarantees your context is intact. The other failures can leave you
|
||||
already gone, so you may never read them yourself — they are in the daemon log for whoever looks
|
||||
next.
|
||||
|
||||
## 13. Things that will surprise you
|
||||
|
||||
- **`accepted` does not mean your pane has been restarted.** It means every gate passed and the roll
|
||||
is scheduled to run once your current turn ends. Say your goodbye in the same turn — you will not
|
||||
get another one.
|
||||
- **If you are still running after that turn, the roll did not happen.** A roll that works ends your
|
||||
process, so surviving your own goodbye is itself the signal that it refused. Check with
|
||||
`fleet_handover{action: "status", token}`, using the token you confirmed. `TURN_NEVER_SETTLED`
|
||||
means your turn ran past `leadRollover.turnSettleSeconds` and **your pane was never ended**: your
|
||||
context is intact and nothing was lost. Open a fresh request and retry. Never assume the roll
|
||||
succeeded because `confirm` answered `accepted` — by the time it refuses, there is no caller left
|
||||
to tell, so this check is the only thing that closes that gap.
|
||||
- **There is no terminal or session parameter, on purpose.** The pane is always your own, resolved
|
||||
from your connection, so you can only ever roll yourself.
|
||||
- **`operatorConfirmed` is your report of what a human told you.** Do not pass `true` because you
|
||||
are confident. Ask, wait for the answer, then pass what they said. `requireOperatorConfirm`
|
||||
defaults to `true` and this is the only thing standing between a judgement call and a wiped
|
||||
session.
|
||||
are confident. Ask, wait for the answer, then pass what they said.
|
||||
|
||||
- **Whether you must ask at all depends on `leadRollover.requireOperatorConfirm`. Check it; do not
|
||||
assume.** The default is `true` (`FleetConfig.java:1426`), and then `confirm` refuses unless you
|
||||
also pass `operatorConfirmed: true`. **This host set it to `false` on 2026-09-22**, on the
|
||||
operator's explicit grant, because they do not want to approve routine context rolls. Where it is
|
||||
`false`, the three handover-file checks are the whole gate: the file must exist, be fresher than
|
||||
`maxDocAgeSeconds`, and have been modified after the open request.
|
||||
|
||||
Read the live value rather than trusting this line:
|
||||
|
||||
```bash
|
||||
grep -A1 'requireOperatorConfirm' fleetd/fleetd.yaml
|
||||
```
|
||||
|
||||
No match means the key is unset, so the default `true` applies and you must ask. The key is
|
||||
**deferred, not hot** — it is read once at boot, so an edit does nothing until the daemon is
|
||||
redeployed.
|
||||
|
||||
The context nudge tracks this value, so its text and the config agree (fleetd #621 —
|
||||
`LeadHeartbeatLoop.contextNotice` takes `requireOperatorConfirm`). You still read the config
|
||||
rather than the nudge, because the nudge only reaches you when your context is already high.
|
||||
|
||||
- **The roll can still refuse after `confirm` returns**, and by then there is no caller to tell.
|
||||
Those outcomes are logged only, as `lead-rollover:` lines in the daemon log.
|
||||
- **The bootstrap prompt has never yet landed, and the fix is unproven (fleetd #489).** The first
|
||||
real rollover, on 2026-09-12, joined `/clear` and the bootstrap text into one line and Claude Code
|
||||
refused it as `Unknown command: /clearFresh`. The pane was never cleared and no context was lost,
|
||||
so the failure was safe — the roll simply did nothing. PR #490 fixed the cause and is deployed,
|
||||
but no roll has bootstrapped a fresh session end to end yet. **Assume it may still fail, and tell
|
||||
the operator so before you confirm.** The recovery is the same either way: the file is already
|
||||
written, so the operator starts a session and points it at the file. That is why you write the
|
||||
file before you confirm, and never the other way round.
|
||||
- **The bootstrap prompt works end to end — measured under the older `/clear` path.** On 2026-09-22
|
||||
the daemon log held four `lead-rollover: rolled` lines; on 2026-10-04 it held **20**, against a
|
||||
control of 86 `lead-rollover:` lines. Each roll started a fresh session against the handover file,
|
||||
with the configured `bootstrapText` arriving as its first message, and no context was lost. So the
|
||||
bootstrap half of the roll is proven, and that half did not change. Section 12 says how to check
|
||||
whether anything has rolled under the restart path since.
|
||||
|
||||
19 of the 20 carry an `elapsedMs`: median 16507 ms, maximum 48261 ms, and two above 45000 ms. That
|
||||
figure times the **whole** roll, and the wait for your own turn to end dominates it. Expect a roll
|
||||
to take tens of seconds, and do not treat a slow one as a failed one. The restart path adds a pane
|
||||
death and a relaunch to that work, so expect it to be slower rather than faster — but nobody has
|
||||
measured it, so do not quote a number for it.
|
||||
|
||||
**You still write the file before you confirm, and never the other way round.** That order is not
|
||||
about the bootstrap being unreliable. It is what the daemon checks: the handover file must have
|
||||
been modified *after* the open request, or `confirm` refuses it as stale.
|
||||
|
||||
## Writing style
|
||||
|
||||
|
||||
@@ -22,13 +22,22 @@ scripts/redeploy-fleetd.sh --yes # skip the drain prompt (fleet already chec
|
||||
scripts/redeploy-fleetd.sh --no-build # restart the jar already on disk
|
||||
```
|
||||
|
||||
`--no-build` skips the build and restarts whatever jar is at `fleetd/target/fleetd.jar`. Use it only
|
||||
when you just built and nothing changed since. It gives up the protection in the next paragraph: no
|
||||
build runs, so a stale or missing jar is not caught early. The script still checks the file is there
|
||||
and dies with `no jar at … — run without --no-build` if it is not, but it cannot tell you the jar is
|
||||
old. A `mvn clean` in the tree deletes that jar while the daemon keeps running on it, and nothing
|
||||
degrades until the next restart. Run `--check` first: it prints the jar's hash and its modification
|
||||
time, so you can see for yourself whether the jar is missing or older than the code you mean to ship.
|
||||
`--no-build` skips the build and restarts whatever jar is at `fleetd/run/fleetd.jar` — the runtime
|
||||
path, not Maven's output path. Use it only when you just built and nothing changed since. It gives
|
||||
up the protection in the next paragraph: no build runs, so a stale or missing jar is not caught
|
||||
early. The script still checks the file is there and dies with `no jar at … — run without
|
||||
--no-build` if it is not, but it cannot tell you the jar is old.
|
||||
|
||||
The daemon runs from `fleetd/run/fleetd.jar`, not from `fleetd/target/fleetd.jar` where Maven
|
||||
writes its output (fleetd #664). That split is what makes a bare `mvn install`/`mvn clean` in the
|
||||
main clone harmless now: neither can reach the file the running daemon holds open, because that
|
||||
file no longer lives under `target/` at all. Verify a merge by building in a throwaway git
|
||||
worktree anyway — a build still produces nothing the fleet runs until this script's own `mv` of
|
||||
`target/fleetd.jar` onto `run/fleetd.jar`, performed only after the old daemon is confirmed gone.
|
||||
Let only `scripts/redeploy-fleetd.sh` touch `fleetd/run/fleetd.jar`. Run `--check` first: it prints
|
||||
the BUILT jar (`target/fleetd.jar`) and the RUNNING jar (`run/fleetd.jar`) as two separately
|
||||
labelled hash-and-mtime facts, so a mismatch between them — a build sitting unswapped, or a stale
|
||||
runtime jar — is visible before you decide anything.
|
||||
|
||||
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
||||
old process to exit rather than assuming; it polls `/healthz`; and it anchors its log checks to a
|
||||
@@ -50,9 +59,12 @@ if the script is unavailable or a step fails, this is what it was protecting you
|
||||
3. **A restart is the only way deferred config keys take effect.** That is usually the reason to do
|
||||
it. The startup log names which keys it accepted and which it deferred — read those lines rather
|
||||
than assuming.
|
||||
4. **Re-check identity afterwards.** Call `fleet_whoami` and confirm it still answers `primary`. The
|
||||
lead is found by its tab label (`fleet.leaders.*.tab`), and a lead whose tab no longer matches is
|
||||
demoted to worker, which refuses every orchestration call.
|
||||
4. **Re-check identity afterwards.** Call `fleet_whoami` and confirm it still answers `primary`. A
|
||||
lead is found by two things together: its tab is labelled `lead`, and that tab sits in the space
|
||||
named by `fleet.leaders.<name>.workspace`. Both must match, so a renamed tab *and* a space whose
|
||||
label differs from the config each demote the lead to worker, which refuses every orchestration
|
||||
call. A `tab:` still in config is accepted as a second label for that lead, and the daemon logs
|
||||
one deprecation warning naming it at startup.
|
||||
5. **Prove the new jar is the one running.** Confirm a *fresh* `fleetd listening` line at the end of
|
||||
`fleetd/fleetd.out`, dated after the restart. An old daemon that never died looks identical from
|
||||
the outside.
|
||||
|
||||
@@ -20,6 +20,13 @@ fleetd.out
|
||||
fleetd/fleetd.out
|
||||
logs/
|
||||
|
||||
# fleetd #635 follow-up — scripts/config-edit.sh's backup directory. No leading slash, so this is
|
||||
# ignored at every depth: the real one lives under fleetd/ (also named in fleetd/.gitignore, next
|
||||
# to the config it backs up), and scripts/test-config-edit.sh's own throwaway fixtures build one
|
||||
# under the repo root while the suite runs. --config can point anywhere, so the directory name is
|
||||
# ignored everywhere rather than only where the live daemon happens to use it.
|
||||
.config-backups/
|
||||
|
||||
# fleetd #480: the lead rollover handover file. `leadRollover.handoverPath` points here, and the
|
||||
# outgoing lead rewrites it on every rollover. It is a snapshot of one moment's live state —
|
||||
# unpushed branches, running builds, open questions — so it is stale the moment it is written and
|
||||
|
||||
@@ -27,23 +27,37 @@ through its `fleet_*` tools. No session addresses a peer, a broker, or the netwo
|
||||
**Every role reads this file.** A member runs in a git worktree of this same repo, so it inherits
|
||||
this `CLAUDE.md` verbatim, and every rule below is role-conditional.
|
||||
|
||||
**Call `fleet_whoami`.** It returns `primary`, `worker`, or `architect`, resolved by the daemon from
|
||||
your connection — unforgeable, and the same resolution its authorization gate uses. A worker also
|
||||
carries its `sessionId`, `profile`, `worktree` and `branch`; an architect carries the slot name it
|
||||
was bound to. Don't infer what you can ask.
|
||||
**Call `fleet_whoami`.** It returns `primary`, `worker`, `architect`, `collaborator`, or `observer`,
|
||||
resolved by the daemon from your connection — unforgeable, and the same resolution its authorization
|
||||
gate uses. A worker also carries its `sessionId`, `profile`, `worktree` and `branch`; an architect
|
||||
carries the slot name it was bound to; a collaborator carries its registry name and its own
|
||||
`sessionId`, and **no `leader` key** — a collaborator is a named peer, not a primary. An **observer**
|
||||
carries only its own `sessionId`: a pane the daemon could not place as any of the above, authorized
|
||||
to `READ`/`METRICS`, to `REPLY`/`ASK` on its own pane, and to `SEND` only to a target that resolves
|
||||
as an observer too — never to a lead, a collaborator, or a spawned member, and never a ticket. It
|
||||
finds such a target in `fleet_list`'s `panes` array, which for an observer is filtered to exactly
|
||||
what it may send to and reduced to `sessionId`, `label`, `status`, `role` and `deliverable`.
|
||||
Don't infer what you can ask.
|
||||
|
||||
Only if that call is unavailable, fall back to these — each is one-way, so keep reading until one
|
||||
fires: the reply charter in your system prompt (*"You are a spawned member in the
|
||||
claude-bridge fleet"*) ⇒ **spawned member**; fleet tools prefixed `mcp__fleet__*` ⇒ **spawned
|
||||
member** (the launcher fixes that mount name; a primary's mount is named by whoever wrote its
|
||||
`.mcp.json`, so it varies — and a member spawned before CB-632 still says `mcp__bridge__*`); `ANTHROPIC_BASE_URL` set ⇒ **spawned member** (Claude-model members run
|
||||
on a clean env, so its *absence* proves nothing). None of these separate a worker from an architect —
|
||||
only `fleet_whoami` does. **Still unsure ⇒ act as a worker**, the most restricted member role. The
|
||||
two mistakes are not symmetric: a primary acting as a worker is refused by the authorization gate —
|
||||
loud and self-correcting — while a member acting as the primary ends its turn with no `fleet_reply`,
|
||||
on a clean env, so its *absence* proves nothing). None of these separate a worker from an architect,
|
||||
or a worker from an **observer** — an observer is just as unspawned as a collaborator and carries
|
||||
none of these signals either, so only `fleet_whoami` tells the two apart. **And none of them fires
|
||||
for a collaborator at all**: every signal in the ladder detects a *spawned* member, while a
|
||||
collaborator is a tab a person opened by hand, so it has no charter, no fixed mount name and a
|
||||
normal environment. A collaborator — or an observer — that cannot call `fleet_whoami` therefore falls
|
||||
to the line below and acts as a worker. That is the safe direction — it under-privileges, and the
|
||||
refusals are loud — but it means a collaborator or an observer has no way to learn what it is except
|
||||
by asking. **Still unsure ⇒ act as a worker**, the most restricted member role this ladder can name.
|
||||
The two mistakes are not symmetric: a primary acting as a worker is refused by the authorization gate
|
||||
— loud and self-correcting — while a member acting as the primary ends its turn with no `fleet_reply`,
|
||||
and the sender silently receives nothing. Fail toward the recoverable error.
|
||||
|
||||
### Invariants — both roles, no exceptions
|
||||
### Invariants — every role, no exceptions
|
||||
|
||||
1. **Never set, export, or forward `ANTHROPIC_BASE_URL`** (or `ANTHROPIC_AUTH_TOKEN`). The primary
|
||||
stays on subscription; only the bridge puts a member off it, at spawn. Mounting the bridge must
|
||||
@@ -51,13 +65,22 @@ and the sender silently receives nothing. Fail toward the recoverable error.
|
||||
2. **The bridge is the only channel.** Text you print in your terminal reaches nobody — the other
|
||||
side cannot see your screen. An answer that isn't in a `fleet_*` call is silently discarded.
|
||||
3. **Identity comes from the connection, never an argument.** Workers never pass a target; you
|
||||
cannot act as another session. Spawn/stop/drain are lead-only; **send is lead or architect**;
|
||||
reply/ask are only-as-itself — any peer may answer for its own pane, and for no other. A call
|
||||
outside your role is refused, not queued.
|
||||
cannot act as another session. Spawn/stop/drain are lead-only; **send is lead, architect,
|
||||
collaborator, or observer** — and a collaborator may send only to a lead or another collaborator,
|
||||
never to a spawned member's terminal, while an observer may send only to another observer pane;
|
||||
reply/ask are only-as-itself — any peer may answer for its own pane,
|
||||
and for no other. A call outside your role is refused, not queued.
|
||||
4. **Delivery is status-gated: one message per turn.** Don't busy-poll a peer's terminal and don't
|
||||
re-send because a call looks slow — the bridge delivers when the peer is `idle`, `blocked` or
|
||||
`done`. A spawned member must **also** have mounted the bridge MCP: until it has, it is not
|
||||
deliverable, and a send waits on that gate for ~60s and then fails without ever reaching its pane.
|
||||
**A lead's own pane has a second gate: its input box must be empty.** The multiplexer pastes and
|
||||
submits in one step, so a delivery that lands while the operator is typing submits their
|
||||
half-written line. A heartbeat, a ticket nudge and lead-to-lead mail therefore wait until the box
|
||||
is clear, and a pane the daemon cannot read as a box waits too. Nothing is lost — every one of
|
||||
those paths retries — but a lead that leaves text sitting in its box receives nothing until it
|
||||
clears, and the only sign is one warning in `fleetd.out` after 20 held checks in a row. Delivery
|
||||
to a *member* is not gated this way, because nobody types in a member's pane.
|
||||
5. **Never move a fleet session, pane or peer except through the bridge.** The bridge owns policy;
|
||||
the multiplexer owns PTYs. Any route that changes fleet state without the bridge's checks
|
||||
bypasses every rule above — the `herdr` CLI and its socket are the usual example.
|
||||
@@ -93,7 +116,11 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
5. **Collect** — `fleet_poll{ticket}` → `fleet_ack{target, msgId}`. Answer a worker's `fleet_ask`
|
||||
with `fleet_send{turnId, content}` — **not** `sessionId`. A worker gone quiet is diagnosed with
|
||||
`fleet_status`, never by reading its terminal; it also reports an open question and the `turnId`
|
||||
that answers it. **A worker's ask waits ~55 seconds, and no nudge makes that longer** — so never
|
||||
that answers it — but **only to the caller that created that delegation**, so a question raised
|
||||
under an architect's brief is invisible to you, and seeing none does not mean there is none.
|
||||
**Only that same creator can answer it.** A `turnId` you came by any other way is refused, so an
|
||||
architect's worker waits for that architect and not for you.
|
||||
**A worker's ask waits ~55 seconds, and no nudge makes that longer** — so never
|
||||
brief a worker to "ask me". Decide before you delegate, or give it an explicit default.
|
||||
**A correction cannot reach a busy member.** A `fleet_send` to a working member is *accepted* and
|
||||
returns a ticket, and is then never delivered — measured here three times in one session, and the
|
||||
@@ -158,9 +185,11 @@ you decide.
|
||||
| See the fleet | `fleet_list` → `leads` (your peers) + `members` (each carries `agentSessionId` when its backend knows one) + `loopHealth` (`RUNNING`, `STALLED`, or `STOPPED` for `statusPoller` and `sessionReaper`) · one peer's state: `fleet_status{sessionId}` |
|
||||
| Delegate (blocking) | `fleet_send{sessionId, content}` |
|
||||
| Delegate (long task) | `fleet_send{sessionId, content, wait:false}` → ticket → `fleet_poll{ticket}` |
|
||||
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId` |
|
||||
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId`, and only the caller that created that delegation |
|
||||
| Message a **peer lead** on this host | `fleet_send{sessionId: <their terminal>, content}` — `fleet_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Message a **peer lead** on another daemon or host | `fleet_send{coordId: <their coord-id>, content}` — needs a `coordinator:` block; your own coord-id is in `fleet_list`. Coordination only, **never** a task |
|
||||
| Message a **collaborator** on this host | `fleet_send{sessionId: <their terminal>, content}` — `fleet_list` reports a `collaborators` array, and each row carries that peer's `name` and the `sessionId` you send to. It is visible to you, to an architect and to another collaborator, never to a worker. Coordination only, **never** a task |
|
||||
| Message an **unconfigured pane** — a tab a person opened by hand | `fleet_send{sessionId: <their terminal>, content}` — it needs **no** `fleet.collaborators` entry and no restart, because a pane becomes deliverable the moment its agent connects the bridge MCP. `fleet_list`'s `panes` array reports every such pane with its label and the terminal id to send to — the full row for you, an architect or a collaborator; filtered and reduced for an observer. **`ListAgents` still never lists these**, and joining `herdr tab list` to `GET /agents` on `tab_id` stays the read-only fallback if the array is missing. Such a pane resolves as an `observer`: it can answer you with `fleet_reply`, and it can `fleet_send` to another observer pane, but never to you. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `fleet_send{coordId}` — or `{sessionId}` if they are on this host. **Not** `fleet_reply`: it has no peer route and the publish is refused |
|
||||
| Read your own held lead-to-lead mail (no ack) | `fleet_poll{coordId: <your own coord-id, from fleet_list's coordinator.selfId>}` — primary-only; never acks, so `fleet_list`'s `held[]` still shows it after. `fleet_list`'s `held[]` gives only a truncated preview — this is the only way to read the full body |
|
||||
| Collect a held reply | `fleet_poll{target}` · then `fleet_ack{target, msgId}` |
|
||||
@@ -234,6 +263,27 @@ simply complies has thrown away the reason there are two of you.
|
||||
7. **Never merge.** Stage files explicitly — never `git add -A` — and leave alone anything the
|
||||
project marks as not-yours-to-commit.
|
||||
|
||||
### Collaborator — a named peer, not a member
|
||||
|
||||
`fleet_whoami` answered `collaborator`, so your pane's tab matches a `fleet.collaborators.<name>.tab`
|
||||
entry. You are **not** a member: nothing delegates to you, you have no brief, no worktree and no
|
||||
ticket, and **you owe no `fleet_reply`** — the turn contract above is for a session a lead spawned,
|
||||
and it does not apply to you. Read it only to understand what the members around you are doing.
|
||||
|
||||
What you may do: observe the fleet (`fleet_list`, `fleet_profiles`, `fleet_whoami`), and send to a
|
||||
lead or to another collaborator. What you may not: spawn, stop or drain anything, roll a lead's
|
||||
session, answer a member's `fleet_ask`, poll a ticket, or send to a spawned member's terminal. Each
|
||||
of those is refused at the gate, not queued.
|
||||
|
||||
Two limits worth knowing before you hit them. **You cannot reach a worker** — not even to help one —
|
||||
because a worker belongs to the lead that spawned it, and routing around that would make you a
|
||||
second orchestrator with no plan. Send to the lead instead. And **you cannot read a ticket**, so you
|
||||
cannot collect a delegation's reply: `fleet_poll` refuses you at the role gate, and a ticket also
|
||||
records the terminal that created it, so even a leaked id reads nothing.
|
||||
|
||||
Being named buys you a channel, not authority. Your `fleet_send` to a lead is coordination between
|
||||
peers: the lead owes you no obedience, and you owe it none.
|
||||
|
||||
### Where each rule lives (don't duplicate — extend the right layer)
|
||||
|
||||
| Layer | Scope | Reaches |
|
||||
@@ -301,10 +351,22 @@ must obey belongs in the charter, not here.
|
||||
(#362). **Read `plugin/` before designing anything about onboarding a project.** Two limits are
|
||||
structural, not bugs: a plugin cannot carry the role agent files, because
|
||||
`ClaudeCodeLauncher.java:371` requires `<cwd>/.claude/agents/<role>.md` in the member's own
|
||||
worktree; and a plugin cannot deliver anything to members at all, because
|
||||
`ClaudeCodeLauncher.java:285` exports `CLAUDE_CONFIG_DIR` and every Claude profile here sets it,
|
||||
so a member never reads the operator's plugin store. **The plugin is the lead-side surface;
|
||||
member-facing assets travel in the worktree.**
|
||||
worktree; and a plugin reaches a member only through `CLAUDE_CONFIG_DIR`, which
|
||||
`ClaudeCodeLauncher.java:286` exports with `putIfPresent` — so only for a profile that sets
|
||||
`configDir`. Every `claude-code` profile does set one (the four without are `opencode`, which
|
||||
never reads that variable). **But measured 2026-10-04: two of them point at
|
||||
`~/.ccs/instances/ltms`, which is the operator's own `CLAUDE_CONFIG_DIR` on this host.** So for an
|
||||
`opus` or `sonnet` member, "a member never reads the operator's plugin store" is false — it reads
|
||||
the same store, because that store is the one its `configDir` names. It stays true for `local` and
|
||||
`local-direct`, which point at `~/.ccs/instances/gx10`. `ClaudeCodeLauncher`'s own javadoc names
|
||||
the related hazard: that file is rewritten on every spawn, so for those two profiles fleetd and the
|
||||
operator's live session write the same `.claude.json`, and its compare-and-swap "narrows the
|
||||
lost-update window, it does not close it". Re-measure which profiles share the operator's dir with
|
||||
`awk '/^profiles:/{i=1;next} /^[a-z]/{i=0} i&&/^ [a-z-]+:$/{p=$1} i&&/configDir:/{print p,$2}'
|
||||
fleetd/fleetd.yaml` against `echo $CLAUDE_CONFIG_DIR`; delete this note once no profile names the
|
||||
operator's dir. **Treat the plugin as the lead-side surface and put member-facing assets in the
|
||||
worktree** — that conclusion holds either way, because a worktree asset does not depend on which
|
||||
config dir a member reads.
|
||||
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
||||
(a submodule with its own remote).
|
||||
- **A provisioned worktree neutralizes `.mcp.json`, `opencode.json` and `.autoenv`** — the repo's
|
||||
@@ -323,6 +385,13 @@ must obey belongs in the charter, not here.
|
||||
reference**, with the intent→tool table above as the short form. `McpContractDocTest` fails if
|
||||
that page names a `fleet_*` tool the server does not register. The flows are kept out of this
|
||||
file because this file loads into every session's context.
|
||||
- **The daemon runs from `fleetd/run/fleetd.jar`, not `fleetd/target/fleetd.jar`** (fleetd #664).
|
||||
Maven's own output still lands at `fleetd/target/fleetd.jar` — that part of the build is
|
||||
unchanged — but the running daemon never has that file open, so a bare `mvn install`/`mvn clean`
|
||||
in the main clone no longer corrupts anything a live process is reading. Verify merges in a
|
||||
throwaway git worktree anyway: a build in the main clone still ships nothing until
|
||||
`scripts/redeploy-fleetd.sh` moves it into place with its own atomic `mv`, performed only after
|
||||
the old daemon is confirmed gone. Let only that script touch `fleetd/run/fleetd.jar`.
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
@@ -352,7 +421,7 @@ Before you call any work done, check the row that matches what you touched:
|
||||
| `ConnectionIdentity` / how a caller is resolved | the `fleet_whoami` paragraph and the fallback ladder |
|
||||
| `REPLY_CHARTER`, or a launcher's mount/flags | the fallback ladder (`mcp__fleet__*`), and the layering table's top row |
|
||||
| the injector / status gating | invariant 4 |
|
||||
| worktree provisioning or the parity overlay | the "both roles read this file" premise — it rests on the worker's worktree being a checkout of this repo |
|
||||
| worktree provisioning or the parity overlay | the "every role reads this file" premise — it rests on the worker's worktree being a checkout of this repo |
|
||||
| `.claude/skills/**` | the addendum's skill list, and the "name the playbook" rule |
|
||||
| a new peer kind (non-Claude adapter) | what that peer can read — anything it must obey belongs in its charter, not in the block |
|
||||
| **anything an operator can use, configure, or observe** — an MCP tool, a `fleetd.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
||||
@@ -445,3 +514,33 @@ to replace them.
|
||||
|
||||
Prefer the unnamed lambda parameter `_` for required-but-unused params; a non-public
|
||||
`static void main(String[])` is valid (JEP 512) and boots via `java -jar`.
|
||||
|
||||
## Code quality — five rules, and what each already cost (enforced)
|
||||
|
||||
Measured at `7f0c4a8`: 124 main files, 36,278 lines, of which **17,850 are code** — 44% is comment,
|
||||
and only **two** files exceed 1000 *code* lines. Encapsulation and inheritance are already sound (0
|
||||
public mutable fields; 12 `extends`, 8 of them exceptions; 120 records). So there is **no Clean Code
|
||||
section, no SOLID list and no pattern catalogue** here: two architect reviews rejected those
|
||||
independently as text that would change no behaviour. These five rules are the whole standard.
|
||||
|
||||
1. **A comment states the current contract or a current maintainer constraint — nothing else.** No
|
||||
tickets, history, dates, measurements or review rationale; those go in the commit message or the
|
||||
MR description. Source code only — this rule never applies to Markdown.
|
||||
2. **A javadoc block stops at 30 lines.** Longer means it is a design argument, so it moves to
|
||||
`docs/<subject>.md` and is linked in one line. The longest here is 235 lines
|
||||
(`config/ConfigRef.java`) and the knowledge in it is load-bearing: **move it, never delete it.**
|
||||
This project has **no ADR** — subject pages under `docs/` are the destination.
|
||||
3. **A comment in main source never names a test class.** There are 44 such names in 76 places and
|
||||
**2 are already dead**, because a name inside `{@code}` is invisible to the compiler and rots in
|
||||
silence. Say what the code guarantees; the test is found by looking.
|
||||
4. **Never relieve a testing problem by reshaping production code.** `Fleetd` carries 54 static
|
||||
factories, `MessageService` carries 7 `volatile` race hooks, and 8 tests assert on main source as
|
||||
*text*. Make the part injectable instead. `FleetdAssembly.assembleAndStart` is 452 lines and may
|
||||
not grow; no new source-text test may be added.
|
||||
5. **No new package cycle, and no widening of a recorded one.** Five pairs are frozen as an exact
|
||||
edge baseline in `PackageCyclesTest` — four of them involve `msg`.
|
||||
|
||||
Rules 1, 2, 3 and 5 have build checks, and Gitea CI runs them on every PR, so they bind members too.
|
||||
**Rule 4's judgement half has no mechanism**: a cap stops a count growing, but no test tells a good
|
||||
decomposition from a bad one. That half is a review obligation, and saying so is deliberate — a rule
|
||||
dressed as a gate it does not have is worse than an honest review item.
|
||||
|
||||
@@ -47,7 +47,7 @@
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/scripts/fleetd-launchd-wrapper.sh</string>
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin/java</string>
|
||||
<string>-jar</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/target/fleetd.jar</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/run/fleetd.jar</string>
|
||||
<string>fleetd.yaml</string>
|
||||
</array>
|
||||
|
||||
|
||||
@@ -50,7 +50,7 @@ WorkingDirectory=%h/LTMS/fleetd/fleetd
|
||||
# and looks healthy, and the failure appears hours later as a member that cannot open a pull
|
||||
# request. exec keeps it one process, so systemd tracks the right PID.
|
||||
# This also avoids a SECOND copy of the secrets in a systemd drop-in: one source of truth.
|
||||
ExecStart=/bin/zsh -lc "exec java -jar target/fleetd.jar fleetd.yaml"
|
||||
ExecStart=/bin/zsh -lc "exec java -jar run/fleetd.jar fleetd.yaml"
|
||||
|
||||
# PrivateTmp MUST stay false -- see herdr.service. fleetd creates the member ZDOTDIR scrub dir and
|
||||
# the opencode config dir under java.io.tmpdir, and the member pane (a herdr child, a different
|
||||
|
||||
@@ -182,6 +182,12 @@ Two consequences a lead feels directly:
|
||||
is fine; the message simply waits, and then restarts the member when it next goes idle.
|
||||
- **A spawned member is not deliverable until it has mounted the MCP.** Until then a send waits on
|
||||
that gate for about 60 seconds and then fails without ever reaching the pane.
|
||||
- **The same gate is what makes an unconfigured pane deliverable.** `contextExtractor` runs on
|
||||
every MCP request, `initialize` included, and `markTrackedCallerPresent` enrols a spawned member
|
||||
*or* an observer into `MemberPresence`; `deliverableTo` then tests presence before the lead and
|
||||
collaborator maps. So mounting the server is the enrolment, and a tab a person opened by hand can
|
||||
be sent to with no config and no restart. It answers with `fleet_reply` — it cannot `fleet_send`,
|
||||
because `Authz` keeps `SEND` to a primary, an architect or a collaborator.
|
||||
|
||||
`UNKNOWN` is deliberately neither injectable nor a pickup. A pane whose status cannot be read is
|
||||
not a pane that is safe to write to — see fleetd #176 for what happens when a gate treats an
|
||||
|
||||
@@ -2,11 +2,22 @@
|
||||
target/
|
||||
dependency-reduced-pom.xml
|
||||
|
||||
# The daemon's runtime jar (fleetd #664). scripts/redeploy-fleetd.sh moves the built jar here
|
||||
# with a same-filesystem rename; this is never Maven's output path and never belongs in git.
|
||||
run/
|
||||
|
||||
# Local runtime config (copy from fleetd.example.yaml). Both names are ignored: fleetd.yaml is
|
||||
# the current name, and bridged.yaml is the legacy name Fleetd still falls back to.
|
||||
fleetd.yaml
|
||||
bridged.yaml
|
||||
|
||||
# fleetd #635 follow-up — scripts/config-edit.sh's backups of fleetd.yaml. A backup of a file
|
||||
# that must never be committed inherits that requirement. The directory is the real protection
|
||||
# (it keeps working even if the backup naming changes); the glob is a backstop for a stray
|
||||
# backup written the old way, directly beside fleetd.yaml, or by an older copy of the script.
|
||||
.config-backups/
|
||||
fleetd.yaml.bak.*
|
||||
|
||||
# CB-505 audit trail + daemon stdout/stderr — runtime records, never source
|
||||
logs/
|
||||
|
||||
|
||||
+48
-25
@@ -62,11 +62,12 @@ bind:
|
||||
#
|
||||
# Leads are configured under `fleet.leaders:` — see THE FLEET further down.
|
||||
#
|
||||
# Two things stop the tab-name convention from becoming a way to claim leadership: the configured
|
||||
# member spaces are excluded from the scan, so nothing fleetd places can land in a matching tab;
|
||||
# and startup REFUSES a `tabPrefix` that the fleet tabLabel template, or any per-profile `tabLabel`
|
||||
# override, also matches — so the two namespaces cannot overlap by accident. The label is a NAME,
|
||||
# never a capability: what a pane may do is decided by the role the daemon resolves for it.
|
||||
# Two things stop the tab-name convention from becoming a way to claim leadership: startup REFUSES
|
||||
# a `tabPrefix` that the fleet tabLabel template, or any per-profile `tabLabel` override, also
|
||||
# matches, so the two namespaces cannot overlap by accident; and the CallerResolver asks the live
|
||||
# spawned-member roster BEFORE any tab map, so a live member is never mistaken for a lead no matter
|
||||
# what its tab says. The label is a NAME, never a capability: what a pane may do is decided by the
|
||||
# role the daemon resolves for it.
|
||||
|
||||
# CB-551: IDLE-LEAD HEARTBEAT — nudge the single lead back to work when it has been continuously
|
||||
# idle (no open fleet_send driving it) past the quiet period. The fleet is one lead + architects +
|
||||
@@ -98,17 +99,17 @@ bind:
|
||||
# contextHighNudge: false
|
||||
|
||||
# Lead rollover (fleetd #480): replace a lead session that has decided it is ready to be replaced,
|
||||
# without an operator doing it by hand. A lead writes a handover file, then asks fleetd to clear its
|
||||
# own pane and bootstrap a fresh session against that file.
|
||||
# without an operator doing it by hand. A lead writes a handover file, then asks fleetd to end its
|
||||
# own pane, launch a fresh one, and bootstrap that fresh session against the handover file.
|
||||
#
|
||||
# Opt-in on purpose — it clears the lead's own pane on request, so upgrading the daemon must never
|
||||
# acquire that ability for you. Absent block = feature off, and nothing is constructed at all. Even
|
||||
# once present, nothing but an explicit confirm() call — one that passes every check — can ever
|
||||
# cause a /clear: there is no recurring timer, heartbeat or scheduler anywhere in this feature that
|
||||
# fires one on its own initiative. confirm() itself is called FROM the calling lead's own turn, so
|
||||
# it cannot clear the pane inline (that pane is still WORKING); instead it schedules a one-shot
|
||||
# Opt-in on purpose — it tears down the lead's own pane on request, so upgrading the daemon must
|
||||
# never acquire that ability for you. Absent block = feature off, and nothing is constructed at all.
|
||||
# Even once present, nothing but an explicit confirm() call — one that passes every check — can ever
|
||||
# tear a pane down: there is no recurring timer, heartbeat or scheduler anywhere in this feature that
|
||||
# fires one on its own initiative. confirm() itself is called FROM the calling lead's own turn, so it
|
||||
# cannot act on the pane inline (that pane is still WORKING); instead it schedules a one-shot
|
||||
# continuation that waits for the SAME confirm() call's turn to end, then does the actual work. See
|
||||
# dev.ltms.fleet.lead.LeadRollover's class javadoc for the exact order (fleetd #480 correction).
|
||||
# dev.ltms.fleet.lead.LeadRollover's class javadoc for the exact order.
|
||||
#
|
||||
# handoverPath: REQUIRED when this block is present — where the handover file a fresh lead session
|
||||
# reads must live. No default (an operator-specific path); a present block with no
|
||||
@@ -117,25 +118,30 @@ bind:
|
||||
# working directory when that lead has none configured) — never against whatever
|
||||
# directory the daemon process happens to have been started in. An absolute path is
|
||||
# used unchanged. Prefer an absolute path if the daemon and the lead's pane might not
|
||||
# share a working directory (fleetd #480 follow-up).
|
||||
# share a working directory.
|
||||
# requireOperatorConfirm: true # default true — confirm() refuses unless the caller also passes
|
||||
# # operatorConfirmed: true
|
||||
# maxDocAgeSeconds: 3600 # default 3600 — refuse a handover file older than this
|
||||
# turnSettleSeconds: 20 # default 20 — how long the deferred roll waits for the CALLING
|
||||
# # lead's own turn to end (its pane to report injectable again)
|
||||
# # before sending /clear at all. If this elapses, /clear is NEVER
|
||||
# # sent — a lead that never goes idle is still doing real work.
|
||||
# clearSettleSeconds: 20 # default 20 — how long to wait for the pane to become injectable
|
||||
# # again AFTER /clear before giving up (never sends bootstrapText
|
||||
# # if this elapses). A separate, second wait from turnSettleSeconds.
|
||||
# # before tearing the old pane down at all. If this elapses, nothing
|
||||
# # is torn down — a lead that never goes idle is still doing real
|
||||
# # work.
|
||||
# relaunchReadySeconds: 45 # default 45 — bounds two later waits, after the old pane is gone
|
||||
# # and a fresh one has been launched: first, for the fresh pane to
|
||||
# # reach a real turn boundary (never sends bootstrapText if THIS one
|
||||
# # elapses); second, for the new terminal to be recognised as this
|
||||
# # lead (bootstrapText is sent either way once the first wait
|
||||
# # passes). A separate, later pair of waits from turnSettleSeconds.
|
||||
# bootstrapText: "..." # default names the RESOLVED (absolute) handoverPath — sent to
|
||||
# # the lead once its pane settles after /clear
|
||||
# # the freshly relaunched lead's pane once it reaches a real turn
|
||||
# # boundary
|
||||
# leadRollover:
|
||||
# handoverPath: /path/to/handover.md
|
||||
# requireOperatorConfirm: true
|
||||
# maxDocAgeSeconds: 3600
|
||||
# turnSettleSeconds: 20
|
||||
# clearSettleSeconds: 20
|
||||
# relaunchReadySeconds: 45
|
||||
# bootstrapText: "Fresh lead session: read the handover file and carry on."
|
||||
|
||||
# Fleet health detection is dormant unless enabled (CB-573). It reads one whole-fleet agent list
|
||||
@@ -659,9 +665,10 @@ fleet:
|
||||
# tabPrefix: "lead:" # only used to guard against a worker tabLabel colliding with
|
||||
# # this convention at startup; plays no part in matching a lead
|
||||
# scanIntervalSeconds: 10 # rescan cadence, and the worst case before a new tab is seen
|
||||
# workspace: leads # where a launched lead's tab is created (default "leads").
|
||||
# # MUST NOT be a member workspace — those are excluded from the
|
||||
# # scan, so a lead placed in one is never found again.
|
||||
# workspace: leads # where a launched lead's tab is created (default "fleet",
|
||||
# # the same shared space the members use). Sharing that space
|
||||
# # with members is the normal shipped shape: the scanner tells
|
||||
# # a lead from a member by the exact tab label, not by workspace.
|
||||
# cwd: /path/to/repo # the launched lead's working directory (default: fleetd's own)
|
||||
# kind: claude # descriptive; reported by fleet_whoami
|
||||
# gpt-sol-5.6:
|
||||
@@ -669,6 +676,22 @@ fleet:
|
||||
# kind: opencode
|
||||
# model: openai/gpt-5.6-terra
|
||||
|
||||
# A collaborator tab, keyed by name (fleetd #669). A pane whose tab matches resolves as the
|
||||
# COLLABORATOR role: `fleet_whoami` answers `collaborator`, and the session may observe the fleet
|
||||
# (`fleet_list`, `fleet_profiles`, `fleet_whoami`) and `fleet_send` to a lead or to another
|
||||
# collaborator. It may NOT spawn, stop or drain anything, roll a lead's session, answer a
|
||||
# member's `fleet_ask`, poll a ticket, send across hosts, or send to a spawned member's terminal.
|
||||
# Each of those is refused at the gate rather than queued.
|
||||
# Recognise-only, like a profile-less `leaders:` entry above: there is no `profile:`, no
|
||||
# `instances:` and no `kind:`. `tab:` is REQUIRED and is the only field identity depends on,
|
||||
# matched case-insensitively — the same GET-THE-VALUE-RIGHT warning above the `leaders:` block
|
||||
# applies here too.
|
||||
# A lead cannot discover a collaborator yet (#703): `fleet_list` has no `collaborators` key, so
|
||||
# the collaborator must speak first, or pass on the `sessionId` its own `fleet_whoami` reports.
|
||||
# collaborators:
|
||||
# reviewer-alex:
|
||||
# tab: "collab: alex"
|
||||
|
||||
# architects:
|
||||
# architect-1:
|
||||
# profile: opus # a strong model, on the operator's subscription
|
||||
|
||||
+4
-2
@@ -189,8 +189,10 @@
|
||||
|
||||
<build>
|
||||
<!-- CB-634: the cutover renamed the module dir (bridged/ -> fleetd/), the jar, and the
|
||||
launchd plist together. The installed plist names fleetd/target/fleetd.jar and
|
||||
KeepAlive is armed, so this name, the plist, and the wrapper must move as one. -->
|
||||
launchd plist together. fleetd #664: the installed plist and the systemd unit now name
|
||||
fleetd/run/fleetd.jar, not this plugin's own output path — see
|
||||
scripts/redeploy-fleetd.sh for the mv that gets a build from here to there. KeepAlive is
|
||||
armed, so this name, the plist, and the wrapper must still move as one. -->
|
||||
<finalName>fleetd</finalName>
|
||||
<plugins>
|
||||
<plugin>
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
|
||||
/**
|
||||
* fleetd #612 Unit A: everything {@code Fleetd.main} has ready once config is loaded, reported
|
||||
* and validated — the exact point {@code main} used to keep going straight into socket and broker
|
||||
* work. Building this (and handing it to {@link FleetdAssembly#assembleAndStart}) is the seam a
|
||||
* test now has to drive the real boot composition without being {@code main} itself.
|
||||
*
|
||||
* @param cfg the boot-time {@link FleetConfig} snapshot every one-time wiring decision reads —
|
||||
* see {@code Fleetd.main}'s own comment on why this must never be swapped for a live
|
||||
* reference once loaded
|
||||
* @param config the live {@link ConfigRef} the hot-reloadable paths read per use
|
||||
* @param guard the same {@link SubscriptionGuard} {@code Fleetd.main} already used to assert the
|
||||
* launching environment is clean, reused rather than rebuilt so the assembled
|
||||
* launchers see the identical instance {@code main} already validated with
|
||||
*/
|
||||
record AssemblyInputs(FleetConfig cfg, ConfigRef config, SubscriptionGuard guard) {
|
||||
}
|
||||
@@ -42,6 +42,7 @@ import dev.ltms.fleet.mcp.LsofProcessCwdLookup;
|
||||
import dev.ltms.fleet.msg.AmqpReplyInbox;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.fleet.msg.LeadChannel;
|
||||
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||
import dev.ltms.fleet.msg.LeadCoordLoop;
|
||||
import dev.ltms.fleet.msg.LeadMailbox;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
@@ -130,6 +131,27 @@ public final class Fleetd {
|
||||
}
|
||||
|
||||
static void main(String[] args) {
|
||||
main(args, ResourcePorts.system());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #625: package-private so a test can drive the literal startup sequence — not a copy
|
||||
* of it — against a non-production {@link ResourcePorts} whose {@link
|
||||
* ResourcePorts#environment()} is tainted, without ever touching the real process environment.
|
||||
* The real process environment is exactly what a test cannot taint from inside the JVM, which
|
||||
* is why nothing could pin {@link SubscriptionGuard#assertPrimaryClean} running at this call
|
||||
* site before this ticket.
|
||||
*
|
||||
* <p>{@link #main(String[])} is the one production caller, passing {@link
|
||||
* ResourcePorts#system()}. Every statement below, in the same order, is otherwise unchanged
|
||||
* from before this ticket — in particular the guard still runs here, before {@code
|
||||
* cfg.validateAll()} and before {@link FleetdAssembly#assembleAndStart} ever touches a socket,
|
||||
* a broker, or HTTP, exactly as it always has. {@code ports} is reused for the guard check and
|
||||
* then handed on to the assembly, rather than a second instance being constructed there, so a
|
||||
* test's fake backs the whole boot path with one consistent view — see {@code
|
||||
* FleetdSubscriptionGuardOrderingTest}.
|
||||
*/
|
||||
static void main(String[] args, ResourcePorts ports) {
|
||||
Path configPath = args.length > 0 ? Path.of(args[0]) : chooseDefaultConfigFile(Path.of(""));
|
||||
// The config file was renamed bridged.yaml -> fleetd.yaml. Name the file we actually
|
||||
// loaded, whichever of the two names it carries.
|
||||
@@ -165,7 +187,7 @@ public final class Fleetd {
|
||||
|
||||
// The primary/host env that launched fleetd must not be tainted.
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
guard.assertPrimaryClean(System.getenv());
|
||||
guard.assertPrimaryClean(ports.environment());
|
||||
|
||||
// Every FleetConfig.validateXxx() the operator's config can fail — CB-501's auth-exposure
|
||||
// check, CB-531's lead-tab-prefix check, CB-542's subscription-profile check, the charter
|
||||
@@ -177,636 +199,45 @@ public final class Fleetd {
|
||||
// and the Fleetd-startup tests actually pin — see FleetConfig#validateAll's javadoc for
|
||||
// why a name-by-name list here would have the same defect it replaces.
|
||||
cfg.validateAll();
|
||||
// fleetd #613: validateAll() (validateMembers() inside it) only refuses a slot that names a
|
||||
// bad role or profile — it says nothing about a role that has NO pool or NO charter at all,
|
||||
// because both are legitimate ("unconstrained") states, not errors. Report them here, right
|
||||
// after validation passes, so an operator sees the gap once per restart instead of finding
|
||||
// it later in a roster row (see reportRoleFallbackGaps' javadoc for the measured cause).
|
||||
reportRoleFallbackGaps(cfg);
|
||||
// fleetd #469, follow-up to #464: validateAll() (and validateCharters() inside it) only
|
||||
// checks that a charter's KEY is a role wire name and its text is non-blank — it never
|
||||
// looks at what the text actually names. This is the separate check that does: it asks
|
||||
// dev.ltms.fleet.mcp.FleetTool (the canonical registered-tool set) whether every fleet_*/
|
||||
// bridge_* token a charter names is a tool this server actually registers. It cannot live
|
||||
// inside FleetConfig#validateCharters() — config loads before the MCP server exists, and
|
||||
// must not gain a dependency on the mcp package — so it runs here instead, at the one seam
|
||||
// that already holds both a loaded FleetConfig and the mcp package, before anything below
|
||||
// opens a socket or spawns a member. fleetd #474: the same check is also wired into `config`
|
||||
// above as ConfigRef's extraValidation, so a reload refuses what this line refuses at startup.
|
||||
assertChartersNameOnlyRegisteredTools(cfg);
|
||||
|
||||
Path socket = cfg.herdrSocket() != null && !cfg.herdrSocket().isBlank()
|
||||
? Path.of(cfg.herdrSocket())
|
||||
: UnixSocketHerdrClient.defaultSocketPath();
|
||||
|
||||
UnixSocketHerdrClient herdr = UnixSocketHerdrClient.connect(socket, new com.fasterxml.jackson.databind.ObjectMapper());
|
||||
UnixSocketHerdrClient memberHerdr = cfg.memberHerdrSocket() != null && !cfg.memberHerdrSocket().isBlank()
|
||||
? UnixSocketHerdrClient.connect(Path.of(cfg.memberHerdrSocket()), new com.fasterxml.jackson.databind.ObjectMapper())
|
||||
: herdr;
|
||||
AtomicReference<Supplier<Map<String, String>>> leadsRef = new AtomicReference<>(Map::of);
|
||||
HerdrRouter router = new HerdrRouter(herdr, memberHerdr,
|
||||
target -> leadsRef.get().get().containsKey(target));
|
||||
// CB-402: one adapter per configured peer kind, fronted by a composite router. A profile's
|
||||
// `kind:` selects its adapter — claude-code (the default) and opencode partition the profile
|
||||
// set — and the composite dispatches each SPI call to the adapter that owns the profile/pane.
|
||||
Map<String, FleetConfig.Profile> claudeProfiles = new LinkedHashMap<>();
|
||||
Map<String, FleetConfig.Profile> opencodeProfiles = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, w) -> {
|
||||
if (w.isOpenCode()) {
|
||||
opencodeProfiles.put(name, w);
|
||||
} else {
|
||||
claudeProfiles.put(name, w);
|
||||
}
|
||||
});
|
||||
List<HerdrPeerLauncher> adapters = new ArrayList<>();
|
||||
// fleetd #175: the daemon's real ExhaustionSink can only be built once `sessions` exists
|
||||
// (below), but `sessions` needs `workers`, which needs the adapters built right here — a
|
||||
// genuine cycle. Break it exactly like liveCountRef below: a forwarding sink built now,
|
||||
// pointed at the real one once it exists.
|
||||
AtomicReference<ExhaustionSink> exhaustionSinkRef = new AtomicReference<>(ExhaustionSink.none());
|
||||
// fleetd #234, round 4: routed through the shared ExhaustionSink.forwardingTo factory
|
||||
// rather than written inline here — not because a lambda at this call site is unsafe
|
||||
// anymore (it is not: the 3-arg overload is now the interface's single abstract method, so
|
||||
// there is no 2-arg overload left for any lambda to silently bind to instead), but so a
|
||||
// test can call the exact same object this line builds, instead of asserting a copy of its
|
||||
// shape (round 3's lesson).
|
||||
// fleetd #589 Group 1: extracted to forwardingExhaustionSink(...) below (see that method's
|
||||
// javadoc) so a dedicated test can prove this factory keeps reading the reference live,
|
||||
// rather than a rebuilt copy of its shape.
|
||||
ExhaustionSink forwardingExhaustionSink = forwardingExhaustionSink(exhaustionSinkRef);
|
||||
// The claude-code adapter is the always-present default; keep it even with no profiles (so a
|
||||
// bridge configured with no workers, or opencode-only, still has a well-defined base adapter)
|
||||
// unless opencode is the only kind configured.
|
||||
// fleetd #589 Group 2: extracted to claudeCodeLauncher(...)/openCodeLauncher(...) below (see
|
||||
// those methods' javadoc) so a dedicated test can prove the CB-596 memberCredentials policy
|
||||
// supplier is actually wired to each adapter, not silently replaced with `() -> null`.
|
||||
if (!claudeProfiles.isEmpty() || opencodeProfiles.isEmpty()) {
|
||||
adapters.add(claudeCodeLauncher(router.memberAgents(), router.memberSpaces(), guard,
|
||||
claudeProfiles, cfg, config));
|
||||
}
|
||||
if (!opencodeProfiles.isEmpty()) {
|
||||
adapters.add(openCodeLauncher(router.memberAgents(), router.memberSpaces(),
|
||||
opencodeProfiles, cfg, config, forwardingExhaustionSink));
|
||||
}
|
||||
AtomicReference<Function<String, Integer>> liveCountRef = new AtomicReference<>(_ -> 0);
|
||||
// CB-578 stage B: one quarantine tracker for the whole daemon, shared between the launcher
|
||||
// (checked at spawn) and the exhaustion sink wired in below (written on BACKEND_EXHAUSTED).
|
||||
// The cooldown is deferred (see FleetConfig#quarantineCooldownSeconds): it is read once
|
||||
// here, at startup, and a config reload only changes it for a daemon restart.
|
||||
// fleetd #466: escalating, not flat — a credential that keeps reporting exhaustion (e.g. a
|
||||
// weekly subscription limit, which would otherwise be retried on every ~30-minute cooldown,
|
||||
// about 336 times across the week) backs off further each consecutive time, capped at
|
||||
// BackendQuarantine.DEFAULT_MAX_COOLDOWN_MULTIPLE x the base cooldown. See BackendQuarantine's
|
||||
// class doc for the mechanism, why this never fires on cooling-off (a separate, unescalated
|
||||
// mechanism — BackendOutagePolicy below), and the reset.
|
||||
BackendQuarantine quarantine = BackendQuarantine.withEscalation(System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));
|
||||
// fleetd #201 Unit 5: one outage-cool-off tracker for the whole daemon, shared between the
|
||||
// launcher (checked at spawn, like `quarantine` above) and the backend-error sink wired in
|
||||
// below (written on a classified backend error). A SEPARATE, shorter-lived mechanism from
|
||||
// `quarantine` — see BackendOutagePolicy's class doc — never merged with it.
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(System::nanoTime);
|
||||
PeerLauncher workers = new CompositePeerLauncher(
|
||||
adapters,
|
||||
cfg.effectiveDefaultProfile(),
|
||||
config,
|
||||
profileName -> liveCountRef.get().apply(profileName),
|
||||
quarantine,
|
||||
outagePolicy);
|
||||
// fleetd #422 follow-up: say which of the three model-gate states the daemon booted into —
|
||||
// no models: block at all, a block armed with nothing off, or a block with N off — the same
|
||||
// way exhaustedPatternCoverageLine/errorPatternCoverageLine report CB-578 stage A/fleetd
|
||||
// #201 Unit 5 coverage just below. Read from workers.modelGateState() (never a separate
|
||||
// config.get().models() here) so this line and fleet_profiles' modelGateArmed can never
|
||||
// disagree about what CompositePeerLauncher's spawn gate actually enforces.
|
||||
log.info("model gate (fleetd #422): {}", modelGateCoverageLine(workers.modelGateState()));
|
||||
// CB-504: under supervision (launchd/systemd) fleetd can start before herdr's socket
|
||||
// exists. The client itself is lazy — it connects per call — but the orphan reap below is
|
||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||
// would crash the daemon into a restart loop. Wait, then degrade rather than die: serving
|
||||
// with /healthz reporting "degraded" is strictly more useful than exiting.
|
||||
HerdrAwaitOutcome herdrOutcome = awaitHerdr(herdr, System::nanoTime, Fleetd::sleepHerdrPoll);
|
||||
boolean herdrUp = logHerdrWaitOutcomeAndShouldReap(herdrOutcome);
|
||||
if (herdrUp) {
|
||||
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
||||
// with the previous process — reap those leaked orphans now, before we start serving.
|
||||
workers.reapOrphanWorkers();
|
||||
}
|
||||
|
||||
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
||||
// CB-301-ext: worktree provisioning seam, optionally rooted at a configured directory.
|
||||
// CB-303 part 2: context cap is opt-in and disabled (0) when absent/null.
|
||||
int contextCap = 0;
|
||||
if (cfg.lifecycle() != null && cfg.lifecycle().contextCap() != null
|
||||
&& cfg.lifecycle().contextCap() > 0) {
|
||||
contextCap = cfg.lifecycle().contextCap();
|
||||
}
|
||||
boolean clearAfterTurn = cfg.lifecycle() != null && cfg.lifecycle().clearAfterTurn();
|
||||
SessionManager sessions = new SessionManager(workers,
|
||||
new GitWorktrees(cfg.worktreeRoot(), cfg.worktreeGroup(), cfg.memberSkills()),
|
||||
System::nanoTime, contextCap, clearAfterTurn);
|
||||
liveCountRef.set(profileName -> liveSessionCount(sessions.roster(), profileName));
|
||||
|
||||
// Idle-sleep guard: hold an OS-level assertion against idle sleep while at least one
|
||||
// member is live, so an unattended host does not idle-sleep out from under a member's
|
||||
// long turn (see FleetConfig.IdleSleepGuard / dev.ltms.fleet.power.IdleSleepGuard for the
|
||||
// measurement that motivated this). Opt-out via idleSleepGuard.enabled: false; on by
|
||||
// default. Hangs off SessionManager's own onAcquire/onRelease hooks (CB-520/CB-516,
|
||||
// previously wired only to the reply inbox) and SessionManager#size() — the exact registry
|
||||
// fleet_list's live/capacity numbers are themselves computed from — rather than tracking
|
||||
// members a second way. No-op (never constructed) off macOS or when idleSleepGuard.enabled
|
||||
// is explicitly false; the mechanism itself is additionally a no-op if 'caffeinate' cannot
|
||||
// be started, so this can never fail a spawn, a release, or startup.
|
||||
boolean idleSleepGuardEnabled = cfg.idleSleepGuard() == null || cfg.idleSleepGuard().isEnabled();
|
||||
final IdleSleepGuard idleSleepGuard;
|
||||
if (idleSleepGuardEnabled) {
|
||||
idleSleepGuard = new IdleSleepGuard(new CaffeinateSleepAssertionMechanism(), sessions::size);
|
||||
sessions.onAcquire(_ -> idleSleepGuard.recheck());
|
||||
sessions.onRelease(_ -> idleSleepGuard.recheck());
|
||||
} else {
|
||||
idleSleepGuard = null;
|
||||
}
|
||||
|
||||
// CB-303 part 1: idle-ttl reaper — only when configured, defaults to disabled.
|
||||
final SessionReaper reaper;
|
||||
if (cfg.lifecycle() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() > 0) {
|
||||
reaper = new SessionReaper(sessions, cfg.lifecycle().idleTtlSeconds());
|
||||
reaper.start();
|
||||
} else {
|
||||
reaper = null;
|
||||
}
|
||||
|
||||
// CB-530: every pane the config names as a lead, merged from `leaders:` and the legacy
|
||||
// singular pin. PrimaryRegistry below still tracks ONE terminal — it addresses the push
|
||||
// loop's nudges, which need a single destination — so it keeps the legacy pin.
|
||||
Map<String, String> leadTerminals = cfg.leaderTerminals();
|
||||
if (leadTerminals.size() > 1) {
|
||||
log.info("leads: {} panes recognised {}", leadTerminals.size(), leadTerminals.values());
|
||||
}
|
||||
// CB-531: on top of the legacy primary.terminal pin, discover leads by the tab labels the
|
||||
// operator writes. CB-557 moved the settings onto the lead they describe, so scanning is on
|
||||
// whenever a `fleet.leaders:` entry exists — with no leads configured the supplier is a
|
||||
// constant and never touches herdr, exactly as a missing `leadScan:` block used to behave.
|
||||
// CB-579: each lead now names its own exact `tab:` label, so one scanner discovers every
|
||||
// configured lead regardless of how differently their tabs are labelled — the old
|
||||
// single-shared-tabPrefix limitation (and its warning) is gone.
|
||||
final Supplier<Map<String, String>> leads;
|
||||
var leaders = cfg.fleet().leaders();
|
||||
if (!leaders.isEmpty()) {
|
||||
Map<String, String> tabToName = new LinkedHashMap<>();
|
||||
leaders.forEach((name, leader) -> {
|
||||
if (leader != null && leader.tab() != null && !leader.tab().isBlank()) {
|
||||
tabToName.put(leader.tab(), name);
|
||||
}
|
||||
});
|
||||
// The lead and the members now share ONE workspace (the operator asked for a single
|
||||
// "session" with many tabs), so no workspace can be excluded — the lead lives in the
|
||||
// members' space by design. A lead is told from a member by its exact tab label alone:
|
||||
// a lead carries its configured `tab`, a member its `worker: {profile} #{n}` template,
|
||||
// and the two never collide. (The scanner still supports an exclusion set for a split
|
||||
// layout; the fleet's policy here is simply not to use one.)
|
||||
//
|
||||
// One shared rescan cadence: still taken from the first entry, as before — it is an
|
||||
// operational cadence, not identity, so there is no correctness reason to give every
|
||||
// lead its own scanner.
|
||||
int scanIntervalSeconds = leaders.values().iterator().next().scanIntervalSeconds();
|
||||
// This must use the lead daemon: scanning member tabs would demote the lead to a worker.
|
||||
leads = new LeadTabScanner(herdr, tabToName, Set.of(),
|
||||
TimeUnit.SECONDS.toNanos(scanIntervalSeconds), System::nanoTime);
|
||||
log.info("lead scan: tabs {} host a lead (rescan every {}s, shared fleet space)",
|
||||
tabToName.keySet(), scanIntervalSeconds);
|
||||
} else {
|
||||
leads = () -> leadTerminals;
|
||||
}
|
||||
leadsRef.set(leads);
|
||||
|
||||
// CB-558: start any declared lead that is not already running. After the scanner is built,
|
||||
// because both read the same tab labels and the ordering makes that dependency visible; and
|
||||
// only when herdr answered, because the launcher's whole safety property is that it can
|
||||
// count live leads first — it must never guess and risk a second orchestrator.
|
||||
if (herdrUp && !leaders.isEmpty()) {
|
||||
int launched = new LeadLauncher(router.leadAgents(), router.leadSpaces(), cfg).ensureLeads();
|
||||
if (launched > 0) {
|
||||
log.info("lead auto-launch: {} lead(s) started", launched);
|
||||
}
|
||||
}
|
||||
|
||||
// CB-548: config-declared architect slots. Config supplies only the stable name → profile
|
||||
// map; the terminal → slot binding is owned by the registry and is empty at startup, so no
|
||||
// pane resolves to an architect until the later spawn lifecycle binds one. The registry is
|
||||
// what CallerResolver resolves against and what that lifecycle will read profiles from;
|
||||
// nothing here spawns a slot.
|
||||
// fleetd #424: MemberRegistry.live re-reads fleet.architects through `config` on every
|
||||
// reserve/requireSlotFor call, so a reload that removes or adds an architect slot governs
|
||||
// the next spawn with no restart — the frozen `new MemberRegistry(cfg.fleet())` this used
|
||||
// to be let a "revoked" slot keep granting new architect spawns forever.
|
||||
MemberRegistry members = MemberRegistry.live(() -> config.get().fleet());
|
||||
sessions.setMemberLifecycle(members);
|
||||
if (!members.slots().isEmpty()) {
|
||||
log.info("member slots: {} configured {} — none bound yet (a slot is idle until the "
|
||||
+ "spawn lifecycle binds a live terminal to it)",
|
||||
members.slots().size(), members.slots().keySet());
|
||||
}
|
||||
|
||||
// Status-gated injector (CB-103): the single writer into workers, fed by a poller.
|
||||
// The blocking message endpoint (CB-104) is the producer; the poller is inert until then.
|
||||
// CB-106: a confirmed turn completion resolves a blocked send whose worker never replied.
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
// CB-578 stage A: classify a completion-fallback scrape that matches a profile's configured
|
||||
// usage-limit refusal as BACKEND_EXHAUSTED rather than handing it back as a real answer.
|
||||
// fleetd #446: read live off `config` per lookup, cached by profile name — see
|
||||
// LiveExhaustedPatterns's class doc for why this replaces the old compiled-once-at-startup
|
||||
// map. A profile with no exhaustedPattern simply returns null here, so its workers keep
|
||||
// today's completion-fallback behaviour unchanged.
|
||||
// fleetd #589 Group 1: both extracted to liveExhaustedPatterns(...)/
|
||||
// exhaustedPatternLookup(...) below (see those methods' javadoc) — this is the worst
|
||||
// consequence in the whole #589 sweep: silently losing either wiring means a genuine
|
||||
// usage-limit refusal is handed back as a real completion instead of BACKEND_EXHAUSTED.
|
||||
LiveExhaustedPatterns liveExhaustedPatterns = liveExhaustedPatterns(config);
|
||||
ExhaustedPatternLookup exhaustedPatterns = exhaustedPatternLookup(sessions::roster, liveExhaustedPatterns);
|
||||
// The startup coverage line still reports the boot-time snapshot only — it is printed once,
|
||||
// here, and a reload no longer needs to change what it said; exhaustionDetectionArmed (via
|
||||
// liveExhaustedPatterns.armed, wired into quarantineSource below) is what stays live.
|
||||
Set<String> exhaustedConfiguredAtStartup = cfg.profiles().entrySet().stream()
|
||||
.filter(e -> e.getValue().hasExhaustedPattern())
|
||||
.map(Map.Entry::getKey)
|
||||
.collect(Collectors.toCollection(LinkedHashSet::new));
|
||||
log.info("backend-exhausted classification (CB-578 stage A): {}",
|
||||
exhaustedPatternCoverageLine(cfg.profiles().keySet(), exhaustedConfiguredAtStartup));
|
||||
// fleetd #201 Unit 5: classify a completion-fallback scrape that matches a profile's
|
||||
// configured backend-error refusal (a credential outage, a provider 5xx) as a backend error
|
||||
// rather than handing it back as a real answer. Still compiled once at startup, keyed by
|
||||
// profile name — unlike exhaustedPattern above (fleetd #446), errorPattern was left DEFERRED
|
||||
// on purpose: the ticket that made exhaustedPattern hot scoped errorPattern/cooling-off out
|
||||
// explicitly. A profile with no configured errorPattern is simply absent here, so
|
||||
// CompletionResolver falls back to its built-in narrow {@code (?i)\bAPI Error\s*:}
|
||||
// compatibility pattern for that profile's targets (BackendErrorPatternLookup#legacy's
|
||||
// contract — see backendErrorPatterns below).
|
||||
Map<String, Pattern> errorPatternsByProfile = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (profile.hasErrorPattern()) {
|
||||
errorPatternsByProfile.put(name, Pattern.compile(profile.errorPattern()));
|
||||
}
|
||||
});
|
||||
// fleetd #248: extracted to a static factory (see backendErrorPatternLookup below) so a
|
||||
// test can prove main() actually PASSES this into CompletionResolver, not only that the
|
||||
// lookup itself behaves correctly — the exact gap fleetd #248 exists to close.
|
||||
BackendErrorPatternLookup backendErrorPatterns = backendErrorPatternLookup(sessions::roster,
|
||||
errorPatternsByProfile);
|
||||
log.info("backend-error classification (fleetd #201 Unit 5): {}",
|
||||
errorPatternCoverageLine(cfg.profiles().keySet(), errorPatternsByProfile.keySet()));
|
||||
// CB-578 stage B: on a classification that actually wins, quarantine the exhausted profile's
|
||||
// CREDENTIAL — not the profile name — so a profile sharing that credential (e.g. two models
|
||||
// on one OpenAI account) is refused too, not just the one that happened to report it. Reads
|
||||
// the profile config live off `config`, so a credentialId edit is hot: no restart needed.
|
||||
// fleetd #446 criterion 3: the backend text that triggered the most recent quarantine of
|
||||
// each credential, so fleet_profiles can report WHY a limit was hit, not only that it was.
|
||||
// Keyed by credential id — the same key BackendQuarantine's own remainingSeconds uses —
|
||||
// and written at the one call site that actually quarantines (inside exhaustionSink below),
|
||||
// so a reason can never be reported for a quarantine that never happened. Bounded the same
|
||||
// way BackendQuarantine's own internal map is documented to be: by the number of distinct
|
||||
// credentials ever exhausted, not by how often the config is edited.
|
||||
Map<String, String> quarantineReasonByCredential = new ConcurrentHashMap<>();
|
||||
// fleetd #446 follow-up (round 3): extracted into exhaustionSink(...) below — see that
|
||||
// method's javadoc for the full fleetd #175/#234/#446 history this used to carry inline —
|
||||
// so a dedicated test can drive the exact ExhaustionSink main() builds, not a hand-rebuilt
|
||||
// copy of its shape.
|
||||
// fleetd #175: point the forwarding sink handed to OpenCodeLauncher above at the real one,
|
||||
// now that `sessions` exists to resolve target -> session -> profile.
|
||||
// fleetd #589 Group 1: both statements (build + set) folded into publishExhaustionSink(...)
|
||||
// below (see that method's javadoc), so a test can prove the reference is actually
|
||||
// repointed at the real sink, not silently left at ExhaustionSink.none().
|
||||
ExhaustionSink exhaustionSink = publishExhaustionSink(exhaustionSinkRef, sessions, config,
|
||||
quarantine, quarantineReasonByCredential, cfg);
|
||||
// fleetd #201 Unit 5: the production BackendErrorSink needs `pushLoop` (built further below,
|
||||
// after `sessions`) to tell a lead about an incident or an unmapped target — the same
|
||||
// construction-order cycle `exhaustionSinkRef` breaks above, broken the same way: a mutable
|
||||
// holder set once `pushLoop` exists, read lazily from inside the lambda built here.
|
||||
AtomicReference<ReplyPushLoop> pushLoopRef = new AtomicReference<>();
|
||||
// fleetd #248: extracted to a static factory (see backendErrorSink below), public rather
|
||||
// than package-private like the other two factories here, so
|
||||
// dev.ltms.fleet.inject.BackendOutageFlowTest can exercise the REAL production sink
|
||||
// directly instead of a hand-mirrored copy of this lambda — that copy was precisely the
|
||||
// gap fleetd #248 exists to close (see that test's class doc for the history).
|
||||
BackendErrorSink backendErrorSink = backendErrorSink(sessions, () -> config.get().profiles(),
|
||||
outagePolicy, pushLoopRef::get);
|
||||
AgentControl agents = router.memberAgents();
|
||||
// Both fleetd#201 Unit 5 (backend-error patterns + sink) and fleetd#241 (the worktree/branch
|
||||
// lookup the fallback report names) land on this one call. The full constructor takes both,
|
||||
// so neither feature is dropped; nowNanos must be passed explicitly to reach it.
|
||||
//
|
||||
// fleetd #248: every argument built specifically for this call (backendErrorPatterns and
|
||||
// backendErrorSink above, and the worktree/branch lookup right here) now comes from a
|
||||
// static factory tested on its own; FleetdCompletionResolverWiringTest source-asserts that
|
||||
// THIS call actually passes them, which is the coverage that was missing before.
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous, exhaustedPatterns,
|
||||
exhaustionSink, backendErrorPatterns, backendErrorSink, System::nanoTime,
|
||||
worktreeBranchLookup(sessions::roster));
|
||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||
MemberPresence presence = sessions.asPresence();
|
||||
// fleetd #561: extracted to a static factory (see turnListener below) — the anonymous class
|
||||
// this replaced had two bare, unguarded statements per callback, and nothing enforced that
|
||||
// the completion half went first beyond call order in the source.
|
||||
TurnListener turnListener = turnListener(completion, sessions);
|
||||
Predicate<String> deliverable = deliverableTo(presence, leads);
|
||||
// fleetd #556: registration is wired directly to `completion`, not folded into the
|
||||
// `turnListener` fan-out above — so it survives `sessions.onDelivered` (or any future
|
||||
// listener) throwing, regardless of call order. See TurnRegistrar's javadoc.
|
||||
// fleetd #589 Group 3 (:505): extracted to turnRegistrar(...) — see FleetdTurnRegistrarWiringTest.
|
||||
Injector injector = new Injector(router, turnListener, deliverable,
|
||||
presence::forget, turnRegistrar(completion));
|
||||
StatusPoller poller = new StatusPoller(router, injector, Injector.POLL_INTERVAL_MILLIS);
|
||||
poller.start();
|
||||
|
||||
// CB-307: reply inbox. A broker: block selects the AMQP-backed durable adapter; absent (or
|
||||
// unusable), fleetd stays soft-state on the in-memory inbox. The AMQP inbox owns a broker
|
||||
// connection, so keep the reference to close it in the ordered shutdown hook.
|
||||
// fleetd #589 Group 3 (:512): opener extracted to replyInboxOpener() — see
|
||||
// FleetdReplyInboxOpenerWiringTest.
|
||||
final ReplyInbox replyInbox = selectReplyInbox(cfg.broker(), System.getenv(), replyInboxOpener());
|
||||
// CB-637: this daemon's lead-to-lead mailbox on the SHARED coordination vhost — a separate
|
||||
// broker from the reply inbox by design (see FleetConfig.Coordinator). Absent a coordinator:
|
||||
// block this is null and every lead path below is simply not wired, which is exactly the
|
||||
// behaviour before this ticket. It owns a broker connection, so keep the reference for the
|
||||
// ordered shutdown hook.
|
||||
// fleetd #589 Group 3 (:518): opener extracted to leadMailboxOpener() — see
|
||||
// FleetdLeadMailboxOpenerWiringTest.
|
||||
final LeadMailbox leadMailbox = openLeadMailbox(cfg.coordinator(), System.getenv(), leadMailboxOpener());
|
||||
// CB-307: learn the primary's terminal from orchestration tool calls (or pin from config).
|
||||
// The pin also feeds CallerResolver below: a primary running inside a herdr pane would
|
||||
// otherwise resolve as a worker and be refused every orchestration tool.
|
||||
String pinnedPrimaryTerminal = cfg.primary() != null ? cfg.primary().terminal() : null;
|
||||
PrimaryRegistry primaryRegistry = new PrimaryRegistry(pinnedPrimaryTerminal);
|
||||
// CB-532: `primary.terminal` is superseded and no longer needed for either of its jobs —
|
||||
// identity comes from `leaders:`/`leadScan:`, and reply nudges now follow the delegating
|
||||
// lead. Say so once at startup rather than leaving a redundant pin to look load-bearing.
|
||||
if (pinnedPrimaryTerminal != null && !pinnedPrimaryTerminal.isBlank()) {
|
||||
log.warn("primary.terminal is DEPRECATED (CB-532) and can be deleted: identity now comes "
|
||||
+ "from leaders:/leadScan:, and reply nudges follow the lead that delegated. "
|
||||
+ "It still works, and is still the fallback nudge destination when a restart "
|
||||
+ "has lost the delegation map. Its pushReminders/pushBackoffMs stay valid.");
|
||||
}
|
||||
// CB-307: active push-to-primary loop — nudge the primary when replies land without an
|
||||
// open fleet_send. Uses its own lightweight scheduled executor, separate from the injector.
|
||||
int maxReminders = cfg.primary() != null ? cfg.primary().remindersOrDefault() : 5;
|
||||
long backoffMs = cfg.primary() != null ? cfg.primary().backoffMsOrDefault() : 15_000L;
|
||||
var pushScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-push-").unstarted(r));
|
||||
// CB-502: the registry is built before the service and the push loop so send/reply outcomes
|
||||
// are counted at their single funnel rather than at each of the two caller-facing surfaces.
|
||||
// CB-512: the push loop takes it too, so nudge outcomes (delivered|exhausted) are counted.
|
||||
Metrics metrics = FleetMetrics.create(sessions, replyInbox);
|
||||
var pushLoop = new ReplyPushLoop(primaryRegistry, router.leadAgents(), replyInbox,
|
||||
pushScheduler, maxReminders, backoffMs, metrics);
|
||||
// fleetd #201 Unit 5: point the forwarding holder captured by the backendErrorSink lambda
|
||||
// above at the real push loop, now that it exists.
|
||||
pushLoopRef.set(pushLoop);
|
||||
// CB-551: idle-lead heartbeat. Opt-in; absent `leadHeartbeat:` this is never constructed, so
|
||||
// an upgraded daemon cannot silently start spending subscription on nudging an idle lead.
|
||||
// It has its own single-thread scheduler and holds its own scheduler shutdown via close().
|
||||
final LeadHeartbeatLoop heartbeat;
|
||||
var heartbeatScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-heartbeat-").unstarted(r));
|
||||
if (cfg.leadHeartbeat() != null) {
|
||||
var hb = cfg.leadHeartbeat();
|
||||
// fleetd #609: own LeadContextGauge instance for the heartbeat loop — separate from the
|
||||
// one FleetMcp builds internally for fleet_list's context row. Each caches independently
|
||||
// (keyed by configDir+sessionId), so this costs at most one extra bounded tail read per
|
||||
// TTL window, never a shared-mutable-state hazard between the two callers.
|
||||
var leadContextGauge = new LeadContextGauge();
|
||||
heartbeat = new LeadHeartbeatLoop(primaryRegistry, router.leadAgents(), replyInbox, sessions::roster,
|
||||
pushLoop, heartbeatScheduler, System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(hb.idleAfterSeconds()), hb.backoffMs(), hb.quietNudgeCap(),
|
||||
metrics,
|
||||
leadContextSource(leadContextGauge, router.leadAgents(), leads,
|
||||
leadConfigDirLookup(() -> config.get().profiles(), leaders)),
|
||||
Boolean.TRUE.equals(hb.contextHighNudge()));
|
||||
heartbeat.start();
|
||||
} else {
|
||||
heartbeat = null;
|
||||
heartbeatScheduler.shutdownNow();
|
||||
}
|
||||
// fleetd #480: lead rollover. Opt-in; absent `leadRollover:` this is never constructed, so
|
||||
// an upgraded daemon cannot silently acquire the ability to clear the lead's own pane.
|
||||
// Unlike heartbeat above, this has no recurring scheduler of its own — nothing but an
|
||||
// explicit confirm() call (wired to an MCP tool by a later ticket; nothing calls it yet)
|
||||
// that passes every gate can ever schedule a roll. It does not take primaryRegistry: the
|
||||
// lead terminal to roll comes from the caller of open()/confirm() (resolved by the MCP
|
||||
// layer from the connection, the same way auth/CallerResolver#resolve builds a
|
||||
// Principal.leader(...)), never from a single-slot lookup — see LeadRollover's class
|
||||
// javadoc, fleetd #480 correction 2.
|
||||
LeadRollover leadRollover = leadRollover(cfg, router.leadAgents(), config, leads);
|
||||
MessageService messages = new MessageService(router, injector, rendezvous, replyInbox,
|
||||
pushLoop, metrics);
|
||||
|
||||
// Health is a slow whole-fleet observer. Keep it separate from the 250ms delivery poller.
|
||||
final FleetHealthMonitor healthMonitor;
|
||||
var healthScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-health-").unstarted(r));
|
||||
if (cfg.health() != null && cfg.health().isEnabled()) {
|
||||
// CB-580: a member found GONE/NEVER_READY must fail whatever ticket is waiting on it,
|
||||
// through the same idempotent target-wide operation CB-516 already uses on release.
|
||||
// fleetd #386: System::nanoTime freezes across a macOS sleep, so the stall check also
|
||||
// gets a wall-clock source to detect and correct for that freeze. Every other decision
|
||||
// in FleetHealthMonitor stays on the monotonic clock, unchanged.
|
||||
// fleetd #589 Group 3 (:591): failTarget extracted to healthFailTarget(...) — see
|
||||
// FleetdHealthFailTargetWiringTest.
|
||||
healthMonitor = new FleetHealthMonitor(agents, sessions::roster, messages, healthScheduler,
|
||||
System::nanoTime, () -> TimeUnit.MILLISECONDS.toNanos(System.currentTimeMillis()),
|
||||
cfg.health().intervalOrDefault(),
|
||||
cfg.health().workingSuspectAfterOrDefault(), healthFailTarget(messages));
|
||||
String coverage = FleetHealthMonitor.coverage(true,
|
||||
cfg.health().notifications() != null && cfg.health().notifications().configured());
|
||||
if ("detection-only".equals(coverage)) {
|
||||
log.warn("fleet health: {} (no notification sink configured)", coverage);
|
||||
} else {
|
||||
log.info("fleet health: {}", coverage);
|
||||
}
|
||||
healthMonitor.start();
|
||||
} else {
|
||||
healthMonitor = null;
|
||||
healthScheduler.shutdownNow();
|
||||
}
|
||||
|
||||
// CB-520: the reply inbox only consumes for agents this gateway owns. own on acquire,
|
||||
// release on teardown. Do this before CB-516 so the inbox is owned before any reply can land.
|
||||
sessions.onAcquire(replyInbox::own);
|
||||
// CB-516: releasing a worker must fail whatever send was waiting on it. Without this a
|
||||
// torn-down delegation kept reporting PENDING until the 30-minute async timeout, and never
|
||||
// reached /metrics — the delegation was unresolvable and nothing said so.
|
||||
// fleetd #589 Group 3 (:611-631): the whole cleanup lambda extracted to releaseCleanup(...)
|
||||
// — see FleetdReleaseCleanupWiringTest, and MessageService.abandon's javadoc for the
|
||||
// documented incident (a torn-down worker's rendezvous waiter left open) this lambda exists
|
||||
// to prevent.
|
||||
sessions.onRelease(releaseCleanup(messages, replyInbox, primaryRegistry));
|
||||
|
||||
// MCP server face (CB-105): fleet_send/fleet_reply/fleet_status, mounted at /mcp.
|
||||
// Caller identity is resolved from the connection (peer PID → herdr pane), not arguments.
|
||||
// CB-185: a caller's pane can live on either daemon (a lead's on the lead daemon, a
|
||||
// member's on the member daemon) — search both, lead first. Collapses to one scan when
|
||||
// memberHerdrSocket is unset (herdr == memberHerdr).
|
||||
ConnectionIdentity identity = new ConnectionIdentity(
|
||||
new PaneLocator(herdr, memberHerdr), new LsofPeerPidLookup(), new LsofProcessCwdLookup());
|
||||
|
||||
// CB-501: one resolver behind both entry paths. Worker identity still comes from the
|
||||
// connection and is never token-gated, so enabling token mode cannot lock the fleet out.
|
||||
final CallerResolver callers;
|
||||
if (cfg.auth().tokenMode()) {
|
||||
String token = System.getenv(cfg.auth().tokenEnv());
|
||||
if (token == null || token.isBlank()) {
|
||||
throw new IllegalStateException("auth.mode=token but env var " + cfg.auth().tokenEnv()
|
||||
+ " is unset or empty — export it before starting fleetd");
|
||||
}
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, true, token, leads, members);
|
||||
log.info("auth: token mode (bearer required for non-worker callers, env {})",
|
||||
cfg.auth().tokenEnv());
|
||||
} else {
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, false, null, leads, members);
|
||||
log.info("auth: loopback-trust (any loopback non-worker caller is the primary)");
|
||||
}
|
||||
|
||||
// fleetd #297: named once and reused verbatim below for FleetApp's GET /profiles, rather than
|
||||
// built a second time — two independently-constructed sources reading the SAME BackendQuarantine
|
||||
// / BackendOutagePolicy would still be able to drift (e.g. a future edit to the credentialIdFor
|
||||
// closure in only one of the two places), exactly the shape #284 was.
|
||||
FleetMcp.QuarantineSource quarantineSource = quarantineSource(config, quarantine,
|
||||
liveExhaustedPatterns, quarantineReasonByCredential);
|
||||
FleetMcp.OutageSource outageSource = new FleetMcp.OutageSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, outagePolicy);
|
||||
|
||||
FleetMcp.LoopHealthSource loopHealth = loopHealthSource(poller, reaper);
|
||||
FleetMcp mcp = new FleetMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, FleetMcp.AuthorizationMode.ENFORCED, metrics,
|
||||
capacitySource(config, cfg, profile -> liveCountRef.get().apply(profile)),
|
||||
healthCoverageSource(config),
|
||||
loopHealth,
|
||||
quarantineSource,
|
||||
leadMailbox,
|
||||
outageSource,
|
||||
new FleetMcp.LeadSeatSource(leadSeatLookup(() -> config.get().profiles(), leaders, leads)),
|
||||
// fleetd #602 gauge-wiring: threads each lead's configured configDir into the
|
||||
// context gauge — see leadConfigDirSource's own doc for why this, not a hardcoded
|
||||
// null, is what fleet_list's context row now reads.
|
||||
leadConfigDirSource(() -> config.get().profiles(), leaders),
|
||||
// fleetd #361: the operator-declared peers this daemon's fleet_list should try to
|
||||
// reach. Read from the SAME snapshot leadMailbox itself opened from (cfg.coordinator()),
|
||||
// not the live config.get() — coordinator wiring is already boot-time-fixed (see
|
||||
// leadMailbox above), so peers follows the same rule rather than half hot-reloading.
|
||||
cfg.coordinator() == null ? List.of() : cfg.coordinator().peers(),
|
||||
// fleetd #480 Unit C: the executor behind fleet_handover — null whenever
|
||||
// leadRollover: is not configured (see the leadRollover local above).
|
||||
leadRollover);
|
||||
|
||||
// CB-637: the receive half. Only constructed when a lead mailbox actually opened — with no
|
||||
// coordinator (or an unreachable one) there is nothing to deliver, so no scheduler is
|
||||
// created and no thread runs. It reads the SAME live lead supplier the injector's
|
||||
// deliverability gate does, so a lead found by the tab scan after startup is reachable
|
||||
// without a restart.
|
||||
final LeadCoordLoop leadCoordLoop;
|
||||
final ScheduledExecutorService leadCoordSchedulerRef;
|
||||
if (leadMailbox != null) {
|
||||
var leadCoordScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-leadcoord-").unstarted(r));
|
||||
leadCoordLoop = new LeadCoordLoop(leadMailbox, router.leadAgents(), leads, leadCoordScheduler,
|
||||
LEAD_COORD_INTERVAL_MS);
|
||||
leadCoordLoop.start();
|
||||
leadCoordSchedulerRef = leadCoordScheduler;
|
||||
} else {
|
||||
leadCoordLoop = null;
|
||||
leadCoordSchedulerRef = null;
|
||||
}
|
||||
|
||||
// CB-559: opt-in config reload. With no `configReload:` block nothing is constructed, so an
|
||||
// upgraded daemon behaves exactly as before — the file is read once at boot and never again.
|
||||
final ConfigWatcher configWatcher;
|
||||
if (cfg.configReload() != null && cfg.configReload().isEnabled()) {
|
||||
configWatcher = new ConfigWatcher(config, cfg.configReload().intervalSeconds());
|
||||
configWatcher.start();
|
||||
} else {
|
||||
configWatcher = null;
|
||||
}
|
||||
|
||||
// CB-303 part 3: single ordered shutdown hook. Drain sessions first while herdr is still
|
||||
// open (so releases reach the daemon), then stop poller/message/mcp/reaper, and close herdr
|
||||
// last. This replaces the earlier independent hooks that could race and close herdr early.
|
||||
Runtime.getRuntime().addShutdownHook(new Thread(() -> {
|
||||
sessions.close(cfg.lifecycle() != null ? cfg.lifecycle().drainTimeoutSeconds() : null);
|
||||
poller.stop();
|
||||
messages.close();
|
||||
pushLoop.close();
|
||||
if (heartbeat != null) heartbeat.close(); // CB-551: stop the idle-lead heartbeat scheduler
|
||||
if (leadCoordLoop != null) leadCoordLoop.close(); // CB-637: stop delivering peer-lead messages
|
||||
if (leadCoordSchedulerRef != null) leadCoordSchedulerRef.shutdownNow();
|
||||
if (healthMonitor != null) healthMonitor.stop();
|
||||
if (configWatcher != null) configWatcher.stop(); // CB-559: stop polling the config file
|
||||
mcp.close();
|
||||
if (reaper != null) reaper.stop();
|
||||
// Idle-sleep guard: release unconditionally, even though sessions.close() above already
|
||||
// drained every session (and each release already drove the live count to 0, which
|
||||
// releases the guard's assertion on its own) — this is the backstop for a drain that was
|
||||
// itself interrupted or threw, so no caffeinate child ever outlives the daemon.
|
||||
if (idleSleepGuard != null) idleSleepGuard.close();
|
||||
// Release the broker connection last among message resources (no-op for the in-memory inbox).
|
||||
if (replyInbox instanceof AutoCloseable closeable) {
|
||||
try {
|
||||
closeable.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("reply inbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
// CB-637: the coordination connection goes with it — after the loop that reads it has
|
||||
// stopped, so no tick can be mid-ack against a closed channel.
|
||||
if (leadMailbox != null) {
|
||||
try {
|
||||
leadMailbox.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("lead mailbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
router.close();
|
||||
}));
|
||||
|
||||
// CB-185: give FleetApp both daemons — /healthz must require both to answer and
|
||||
// GET /sessions must merge across both, or a down/unpolled member daemon is invisible.
|
||||
// fleetd #111: live (re-read-per-request) memberCredentials view for GET /member-credentials —
|
||||
// same hot-reload shape as the memberCredentials supplier passed to ClaudeCodeLauncher above.
|
||||
// fleetd #297: quarantineSource/outageSource are the SAME instances passed to FleetMcp above —
|
||||
// GET /profiles must report the identical quarantine/cool-off facts as fleet_profiles.
|
||||
Javalin app = new FleetApp(herdr, memberHerdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics, deliverable,
|
||||
() -> MemberCredentialPolicyView.of(config.get().memberCredentials()),
|
||||
quarantineSource, outageSource, loopHealth).build();
|
||||
app.start(cfg.bind().host(), cfg.bind().port());
|
||||
log.info("fleetd listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
// fleetd #612 A-gaps (gap 2): everything from here on — including the two post-validation
|
||||
// reports that used to run inline right here (reportRoleFallbackGaps,
|
||||
// assertChartersNameOnlyRegisteredTools) — now lives in FleetdAssembly.assembleAndStart,
|
||||
// built against a real ResourcePorts. Unit A originally moved only the socket/broker/HTTP
|
||||
// composition and left those two calls here, between validateAll() and the assembly call —
|
||||
// outside the boundary FleetdAssemblyLifecycleTest drives, so deleting either call
|
||||
// compiled clean and left the whole suite green. Moving the boundary to start immediately
|
||||
// after validateAll() (this line) puts both back under test, in the same relative order,
|
||||
// before either one does any I/O — see FleetdAssembly's javadoc for the full boot-order
|
||||
// contract this preserves exactly.
|
||||
// fleetd #625: the same `ports` the guard check above just used, not a second
|
||||
// ResourcePorts.system() instance — see this method's own javadoc.
|
||||
FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@link Injector}'s readiness gate (CB-534): a target is deliverable if it is a spawned
|
||||
* member whose agent has connected the bridge MCP, <em>or</em> a lead.
|
||||
* member whose agent has connected the bridge MCP, a lead, <em>or</em> a collaborator.
|
||||
*
|
||||
* <p>The gate exists for one reason — to hold a delivery out of a <em>spawned</em> member's boot
|
||||
* window, where herdr already reports {@code idle} but the TUI would drop an injected paste. That
|
||||
* hazard is a property of spawning. A lead is never spawned: the operator started it and named it
|
||||
* (or labelled its tab) only once it was up, so there is no boot window to guard.
|
||||
* (or labelled its tab) only once it was up, so there is no boot window to guard. A collaborator
|
||||
* is the same way — a person's own tab, matched to a configured name, never spawned.
|
||||
*
|
||||
* <p>A lead is also never enrolled in {@link MemberPresence} — {@code FleetMcp} marks presence
|
||||
* for every spawned member (worker and architect), deliberately, since that map doubles as the
|
||||
* member roster's availability signal and a lead counted there would show up as an available
|
||||
* member. So without the second disjunct a lead is permanently un-deliverable: every
|
||||
* lead→lead send sat on the gate for {@code READINESS_GRACE_POLLS} (~60s) and then failed
|
||||
* having never been typed into the pane.
|
||||
* <p>Neither a lead nor a collaborator is ever enrolled in {@link MemberPresence} — {@code
|
||||
* FleetMcp} marks presence only for a worker, an architect, or the unconfigured-pane floor,
|
||||
* never for a lead or a collaborator. So without the second and third disjuncts a lead or
|
||||
* collaborator is permanently un-deliverable: every send to one sat on the gate for
|
||||
* {@code READINESS_GRACE_POLLS} (~60s) and then failed having never been typed into the pane.
|
||||
*
|
||||
* <p>The lead set is read through the supplier on each call rather than snapshotted, so a lead
|
||||
* discovered by {@code leadScan} after startup becomes deliverable without a restart.
|
||||
* <p>Both sets are read through their supplier on each call rather than snapshotted, so a lead or
|
||||
* collaborator discovered by {@code leadScan} after startup becomes deliverable without a restart.
|
||||
*/
|
||||
static Predicate<String> deliverableTo(MemberPresence presence, Supplier<Map<String, String>> leads) {
|
||||
return target -> presence.isPresent(target) || leads.get().containsKey(target);
|
||||
public static Predicate<String> deliverableTo(MemberPresence presence, Supplier<Map<String, String>> leads,
|
||||
Supplier<Map<String, String>> collaborators) {
|
||||
return target -> presence.isPresent(target) || leads.get().containsKey(target)
|
||||
|| collaborators.get().containsKey(target);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -905,22 +336,6 @@ public final class Fleetd {
|
||||
}, reasonByCredential::get);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #415 (review follow-up): package-private factory for the CB-578 stage A {@code
|
||||
* exhaustedPattern} startup coverage line, paired explicitly with {@link
|
||||
* CompletionResolver.UnsetMeaning#OFF} — {@code exhaustedPattern} has no fallback, so a
|
||||
* profile with none configured really does have the classification off.
|
||||
*
|
||||
* <p>Extracted out of {@code main} for the same reason {@link #capacitySource} and {@link
|
||||
* #worktreeBranchLookup} were: {@code coverage()}'s own tests ({@code CompletionResolverTest})
|
||||
* prove it words {@code OFF} and {@link CompletionResolver.UnsetMeaning#BUILT_IN_DEFAULT}
|
||||
* correctly when a test supplies the meaning itself — they cannot prove {@code main} pairs the
|
||||
* right meaning with the right key, which is the actual fleetd #415 defect. <b>Measured:</b>
|
||||
* swapping the {@code UnsetMeaning} arguments between this method and {@link
|
||||
* #errorPatternCoverageLine} — recreating #415's defect with the two keys exchanged — compiled
|
||||
* with 0 errors and left all 1506 existing tests green before {@code
|
||||
* FleetdPatternCoverageLineTest} was added to catch exactly that swap.
|
||||
*/
|
||||
/**
|
||||
* fleetd #446 follow-up: the criterion-2 WARNING text — "name the fix, not just the fact" —
|
||||
* for a profile whose {@code model:} is configured. Extracted out of the {@code
|
||||
@@ -955,19 +370,22 @@ public final class Fleetd {
|
||||
+ "s quarantine above is the only thing keeping new spawns off it for now";
|
||||
}
|
||||
|
||||
/**
|
||||
* Package-private factory for the {@code exhaustedPattern} startup coverage line, paired
|
||||
* explicitly with {@link CompletionResolver.UnsetMeaning#OFF} — {@code exhaustedPattern} has
|
||||
* no fallback, so a profile with none configured really does have the classification off.
|
||||
*/
|
||||
static String exhaustedPatternCoverageLine(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
return CompletionResolver.coverage("exhaustedPattern", CompletionResolver.UnsetMeaning.OFF,
|
||||
allProfiles, configuredProfiles);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #415 (review follow-up): the {@code errorPattern} counterpart of {@link
|
||||
* #exhaustedPatternCoverageLine}, paired explicitly with {@link
|
||||
* CompletionResolver.UnsetMeaning#BUILT_IN_DEFAULT} — an unset {@code errorPattern} still runs
|
||||
* backend-error classification against {@code CompletionResolver}'s built-in {@code
|
||||
* BACKEND_ERROR} pattern, so the empty case is not "off". See {@link
|
||||
* #exhaustedPatternCoverageLine}'s javadoc for the measured swap mutation this pairing guards
|
||||
* against.
|
||||
* The {@code errorPattern} counterpart of {@link #exhaustedPatternCoverageLine}, paired
|
||||
* explicitly with {@link CompletionResolver.UnsetMeaning#BUILT_IN_DEFAULT} — an unset
|
||||
* {@code errorPattern} still runs backend-error classification against
|
||||
* {@code CompletionResolver}'s built-in {@code BACKEND_ERROR} pattern, so the empty case is
|
||||
* not "off".
|
||||
*/
|
||||
static String errorPatternCoverageLine(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
return CompletionResolver.coverage("errorPattern", CompletionResolver.UnsetMeaning.BUILT_IN_DEFAULT,
|
||||
@@ -1303,8 +721,8 @@ public final class Fleetd {
|
||||
* {@code CompletionResolver} constructor call — provably untested wiring, the whole reason
|
||||
* fleetd #248 exists: dropping that one argument (passing {@code _ -> null} instead) compiled
|
||||
* clean and left every test green. Extracted here, {@code main} now calls this factory instead
|
||||
* of building the lambda inline, and a source assertion on that call site
|
||||
* ({@code FleetdCompletionResolverWiringTest}) proves the argument is still actually passed.
|
||||
* of building the lambda inline, so the argument reaching the {@code CompletionResolver}
|
||||
* constructor is a named, directly testable call rather than an inline lambda.
|
||||
*
|
||||
* <p>Takes the roster as a plain {@link Supplier} — not a {@link SessionManager} — so this is
|
||||
* directly testable with a hand-built session list; no real {@code SessionManager} (launcher,
|
||||
@@ -1473,6 +891,32 @@ public final class Fleetd {
|
||||
throw (T) t;
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@link PrimaryRegistry} lookup for "which terminal currently hosts the lead named
|
||||
* {@code name}" — the inverse of {@code liveLeadTerminals} (terminal id → lead name), read live
|
||||
* on every call so a lead discovered, rolled, or lost since the last call is reflected without
|
||||
* a restart. Returns {@code null} when no currently recognised lead carries that name — a name
|
||||
* that is not a lead at all (an architect slot, a collaborator), or a lead whose tab the scan
|
||||
* cannot currently place (just rolled, off-host, non-herdr).
|
||||
*
|
||||
* @param liveLeadTerminals terminal id → lead name for every CURRENTLY recognised lead, normally
|
||||
* the same {@code leads} supplier {@code main} already builds for
|
||||
* {@code HerdrRouter}/{@link #leadSeatLookup}
|
||||
*/
|
||||
static Function<String, String> currentTerminalForName(Supplier<Map<String, String>> liveLeadTerminals) {
|
||||
return name -> {
|
||||
if (name == null) {
|
||||
return null;
|
||||
}
|
||||
for (var entry : liveLeadTerminals.get().entrySet()) {
|
||||
if (name.equals(entry.getValue())) {
|
||||
return entry.getKey();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #480: construct the {@link LeadRollover} executor only when {@code leadRollover:} is
|
||||
* present at startup — the same presence gate {@code leadHeartbeat:} uses just above this
|
||||
@@ -1510,21 +954,27 @@ public final class Fleetd {
|
||||
* cfg.leadHeartbeat()}
|
||||
* @param leadAgents the {@link AgentControl} instance that reaches the LEAD's pane (not
|
||||
* {@code memberAgents}), normally {@code router.leadAgents()}
|
||||
* @param leadSpaces the {@link WorkspaceControl} instance that reaches the LEAD's
|
||||
* workspace, normally {@code router.leadSpaces()} — used to tear down
|
||||
* a rolled lead's old pane and confirm it is gone
|
||||
* @param launcher starts the fresh lead a roll relaunches once the old one is gone
|
||||
* @param config the live {@link ConfigRef}, captured only inside the returned
|
||||
* supplier and the workspace lookup — never dereferenced here
|
||||
* supplier and the two lookups below — never dereferenced here
|
||||
* @param liveLeadTerminals terminal id → lead NAME for every CURRENTLY recognised lead, normally
|
||||
* the same {@code leads} supplier {@code main} already builds for
|
||||
* {@code HerdrRouter}/{@link #leadSeatLookup} — never a value snapshot
|
||||
* @return a constructed {@link LeadRollover}, or {@code null} when {@code leadRollover:} is
|
||||
* absent from the startup config
|
||||
*/
|
||||
static LeadRollover leadRollover(FleetConfig cfg, AgentControl leadAgents, ConfigRef config,
|
||||
static LeadRollover leadRollover(FleetConfig cfg, AgentControl leadAgents,
|
||||
WorkspaceControl leadSpaces, LeadLauncher launcher, ConfigRef config,
|
||||
Supplier<Map<String, String>> liveLeadTerminals) {
|
||||
if (cfg.leadRollover() == null) {
|
||||
return null;
|
||||
}
|
||||
Function<String, String> leadNameForTerminal = terminal -> liveLeadTerminals.get().get(terminal);
|
||||
Function<String, String> leadWorkspace = terminal -> {
|
||||
String leadName = liveLeadTerminals.get().get(terminal);
|
||||
String leadName = leadNameForTerminal.apply(terminal);
|
||||
if (leadName == null) {
|
||||
return null;
|
||||
}
|
||||
@@ -1532,7 +982,8 @@ public final class Fleetd {
|
||||
FleetConfig.Leader leader = fleet == null ? null : fleet.leaders().get(leadName);
|
||||
return leader == null ? null : leader.cwd();
|
||||
};
|
||||
return new LeadRollover(leadAgents, () -> config.get().leadRollover(), leadWorkspace);
|
||||
return new LeadRollover(leadAgents, leadSpaces, launcher, () -> config.get().leadRollover(),
|
||||
leadWorkspace, leadNameForTerminal, liveLeadTerminals);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1646,7 +1097,33 @@ public final class Fleetd {
|
||||
*/
|
||||
static FleetMcp.LeadConfigDirSource leadConfigDirSource(Supplier<Map<String, FleetConfig.Profile>> profiles,
|
||||
Map<String, FleetConfig.Leader> leaders) {
|
||||
return new FleetMcp.LeadConfigDirSource(leadConfigDirLookup(profiles, leaders));
|
||||
return new FleetMcp.LeadConfigDirSource(leadConfigDirLookup(profiles, leaders),
|
||||
leadContextWindowLookup(profiles, leaders));
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-lead-name factory for the effective auto-compact window {@link LeadContextGauge} scales
|
||||
* its HIGH threshold against — the same {@code fleet.leaders.<name>.profile} link {@link
|
||||
* #leadConfigDirLookup} already follows, one step further to that profile's own {@link
|
||||
* FleetConfig.Profile#effectiveAutoCompactWindow()}. A lead entry that names no
|
||||
* {@code profile:}, or whose named profile is not configured, or whose profile resolves no
|
||||
* window at all, returns {@code null} — {@link LeadContextGauge} then falls back to its own
|
||||
* fixed HIGH threshold.
|
||||
*/
|
||||
static Function<String, Long> leadContextWindowLookup(Supplier<Map<String, FleetConfig.Profile>> profiles,
|
||||
Map<String, FleetConfig.Leader> leaders) {
|
||||
return leadName -> {
|
||||
FleetConfig.Leader lead = leaders.get(leadName);
|
||||
if (lead == null || lead.profile() == null || lead.profile().isBlank()) {
|
||||
return null;
|
||||
}
|
||||
FleetConfig.Profile leadProfile = profiles.get().get(lead.profile());
|
||||
if (leadProfile == null) {
|
||||
return null;
|
||||
}
|
||||
Integer window = leadProfile.effectiveAutoCompactWindow();
|
||||
return window == null ? null : window.longValue();
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1668,22 +1145,30 @@ public final class Fleetd {
|
||||
* @param liveLeadTerminals terminal id → lead name for every CURRENTLY recognised lead
|
||||
* @param configDirForLeadName lead name → {@code configDir}, normally {@link
|
||||
* #leadConfigDirLookup}'s return
|
||||
* @param windowForLeadName lead name → that lead's profile's effective auto-compact window,
|
||||
* normally {@link #leadContextWindowLookup}'s return, and passed
|
||||
* through to {@link LeadContextGauge#read} so the heartbeat's own
|
||||
* HIGH reading scales with that lead's real window, or {@code null}
|
||||
* when it cannot be resolved — either way {@link LeadContextGauge}
|
||||
* falls back to its own fixed HIGH threshold
|
||||
*/
|
||||
static Function<String, LeadContextGauge.Reading> leadContextLookup(LeadContextGauge gauge, AgentControl agents,
|
||||
Supplier<Map<String, String>> liveLeadTerminals, Function<String, String> configDirForLeadName) {
|
||||
Supplier<Map<String, String>> liveLeadTerminals, Function<String, String> configDirForLeadName,
|
||||
Function<String, Long> windowForLeadName) {
|
||||
return terminal -> {
|
||||
String leadName = liveLeadTerminals.get().get(terminal);
|
||||
if (leadName == null) {
|
||||
return LeadContextGauge.Reading.unknown();
|
||||
}
|
||||
String configDir = configDirForLeadName.apply(leadName);
|
||||
Long effectiveWindow = windowForLeadName.apply(leadName);
|
||||
Agent live;
|
||||
try {
|
||||
live = agents.get(terminal);
|
||||
} catch (RuntimeException e) {
|
||||
return LeadContextGauge.Reading.unknown();
|
||||
}
|
||||
return gauge.read(configDir, live.sessionId(), live.agentType());
|
||||
return gauge.read(configDir, live.sessionId(), live.agentType(), effectiveWindow);
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1695,9 +1180,10 @@ public final class Fleetd {
|
||||
* (see {@code FleetdLeadConfigDirSourceWiringTest}'s javadoc for the measured gap this shape closes).
|
||||
*/
|
||||
static LeadHeartbeatLoop.LeadContextSource leadContextSource(LeadContextGauge gauge, AgentControl agents,
|
||||
Supplier<Map<String, String>> liveLeadTerminals, Function<String, String> configDirForLeadName) {
|
||||
Supplier<Map<String, String>> liveLeadTerminals, Function<String, String> configDirForLeadName,
|
||||
Function<String, Long> windowForLeadName) {
|
||||
return new LeadHeartbeatLoop.LeadContextSource(
|
||||
leadContextLookup(gauge, agents, liveLeadTerminals, configDirForLeadName));
|
||||
leadContextLookup(gauge, agents, liveLeadTerminals, configDirForLeadName, windowForLeadName));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1809,10 +1295,18 @@ public final class Fleetd {
|
||||
ReplyInbox open(String uri, int prefetch);
|
||||
}
|
||||
|
||||
/** Injection seam for {@link #openLeadMailbox}: production binds {@link LeadMailbox#open}. */
|
||||
/**
|
||||
* Injection seam for {@link #openLeadMailbox}: production binds {@link LeadMailbox#open}.
|
||||
*
|
||||
* <p>fleetd #612 A-gaps (gap 1): returns {@link LeadChannelHandle}, not the concrete {@link
|
||||
* LeadMailbox}, so a test can supply a fake closeable channel instead of a real broker
|
||||
* connection — see {@link LeadChannelHandle}'s own javadoc for why the narrower type existed
|
||||
* and what widening it to add {@code close()} costs (nothing: {@code LeadMailbox} already
|
||||
* implements it).
|
||||
*/
|
||||
@FunctionalInterface
|
||||
interface LeadMailboxOpener {
|
||||
LeadMailbox open(String uri, String selfCoordId, int prefetch);
|
||||
LeadChannelHandle open(String uri, String selfCoordId, int prefetch);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1835,7 +1329,7 @@ public final class Fleetd {
|
||||
* fleet still works exactly as it did before this feature existed.</li>
|
||||
* </ul>
|
||||
*/
|
||||
static LeadMailbox openLeadMailbox(FleetConfig.Coordinator coordinator, Map<String, String> env,
|
||||
static LeadChannelHandle openLeadMailbox(FleetConfig.Coordinator coordinator, Map<String, String> env,
|
||||
LeadMailboxOpener opener) {
|
||||
if (coordinator == null) {
|
||||
return null; // opt-in: nothing configured, nothing to say
|
||||
@@ -1855,7 +1349,7 @@ public final class Fleetd {
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
LeadMailbox mailbox = opener.open(uri, coordinator.selfId(), coordinator.prefetchOrDefault());
|
||||
LeadChannelHandle mailbox = opener.open(uri, coordinator.selfId(), coordinator.prefetchOrDefault());
|
||||
log.info("lead coordination: ON as coord-id {} (prefetch={})",
|
||||
coordinator.selfId(), coordinator.prefetchOrDefault());
|
||||
return mailbox;
|
||||
@@ -2235,16 +1729,20 @@ public final class Fleetd {
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #474: the one place both the startup call (right after {@code cfg.validateAll()} in
|
||||
* {@link #main}) and the reload call (wired into {@code config}'s {@code extraValidation} above,
|
||||
* via a method reference to this method) go through, so the two can never drift into checking
|
||||
* different things. Extracted only to give {@link ConfigRef}'s {@code Consumer<FleetConfig>}
|
||||
* hook a {@code FleetConfig -> void} shape to bind to — {@link CharterToolSurface} itself still
|
||||
* takes the raw charter map and knows nothing about {@code ConfigRef} or {@code Fleetd}.
|
||||
* fleetd #474: the one place both the startup call (fleetd #612 A-gaps gap 2: right after
|
||||
* {@code cfg.validateAll()} in {@link FleetdAssembly#assembleAndStart}, immediately after
|
||||
* {@code main} hands off to it — see that method's javadoc) and the reload call (wired into
|
||||
* {@code config}'s {@code extraValidation} above, via a method reference to this method) go
|
||||
* through, so the two can never drift into checking different things. Extracted only to give
|
||||
* {@link ConfigRef}'s {@code Consumer<FleetConfig>} hook a {@code FleetConfig -> void} shape to
|
||||
* bind to — {@link CharterToolSurface} itself still takes the raw charter map and knows nothing
|
||||
* about {@code ConfigRef} or {@code Fleetd}.
|
||||
*
|
||||
* <p>Package-private so a test can call it directly the same way the other startup-report
|
||||
* helpers above are tested, without needing to drive {@link #main} for a unit-level check;
|
||||
* {@code FleetdStartupValidationTest} proves the startup call site, and {@code
|
||||
* {@code FleetdStartupValidationTest} proves the startup call site by driving {@code main}
|
||||
* itself end to end (the throw still happens before {@code main} reaches any real socket or
|
||||
* broker work, since the assembly runs this before either), and {@code
|
||||
* FleetdConfigRefCharterToolSurfaceWiringTest} — by constructing {@code ConfigRef} with this
|
||||
* exact method reference, the same way {@code main} does above — proves the reload call site.
|
||||
* {@code dev.ltms.fleet.config.ConfigRefTest} pins the same reload behaviour too, through an
|
||||
@@ -2286,13 +1784,16 @@ public final class Fleetd {
|
||||
record HerdrAwaitOutcome(HerdrWaitResult result, long elapsedNanos) {}
|
||||
|
||||
/**
|
||||
* The real per-poll wait {@link #main} passes to {@link #awaitHerdr}: sleep
|
||||
* {@link #HERDR_WAIT_POLL_MILLIS}, and on interruption re-set the thread's interrupt flag
|
||||
* The real per-poll wait {@link FleetdAssembly#assembleAndStart} passes to {@link #awaitHerdr}:
|
||||
* sleep {@link #HERDR_WAIT_POLL_MILLIS}, and on interruption re-set the thread's interrupt flag
|
||||
* rather than throwing — {@link #awaitHerdr} detects an interruption by checking {@link
|
||||
* Thread#isInterrupted()} right after this returns, so a poller that swallowed the flag
|
||||
* instead of restoring it would make that check silently miss the interruption.
|
||||
*
|
||||
* <p>Package-private (fleetd #612 Unit A), not {@code private}: {@code FleetdAssembly}, a
|
||||
* different class in this package, needs a {@code Runnable} reference to this exact method.
|
||||
*/
|
||||
private static void sleepHerdrPoll() {
|
||||
static void sleepHerdrPoll() {
|
||||
try {
|
||||
Thread.sleep(HERDR_WAIT_POLL_MILLIS);
|
||||
} catch (InterruptedException ie) {
|
||||
|
||||
@@ -0,0 +1,586 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.auth.CallerResolver;
|
||||
import dev.ltms.fleet.auth.MemberRegistry;
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.ConfigWatcher;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.health.FleetHealthMonitor;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||
import dev.ltms.fleet.herdr.LeadTabScanner;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.herdr.UnixSocketHerdrClient;
|
||||
import dev.ltms.fleet.inject.BackendErrorPatternLookup;
|
||||
import dev.ltms.fleet.inject.BackendErrorSink;
|
||||
import dev.ltms.fleet.inject.CompletionResolver;
|
||||
import dev.ltms.fleet.inject.ExhaustedPatternLookup;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.inject.LiveExhaustedPatterns;
|
||||
import dev.ltms.fleet.inject.MemberPresence;
|
||||
import dev.ltms.fleet.inject.StatusPoller;
|
||||
import dev.ltms.fleet.inject.TurnListener;
|
||||
import dev.ltms.fleet.lead.LeadContextGauge;
|
||||
import dev.ltms.fleet.lead.LeadLauncher;
|
||||
import dev.ltms.fleet.lead.LeadRollover;
|
||||
import dev.ltms.fleet.mcp.ConnectionIdentity;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.mcp.LsofPeerPidLookup;
|
||||
import dev.ltms.fleet.mcp.LsofProcessCwdLookup;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.member.HerdrPeerLauncher;
|
||||
import dev.ltms.fleet.member.MemberCredentialPolicyView;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||
import dev.ltms.fleet.msg.LeadCoordLoop;
|
||||
import dev.ltms.fleet.msg.LeadHeartbeatLoop;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.power.CaffeinateSleepAssertionMechanism;
|
||||
import dev.ltms.fleet.power.IdleSleepGuard;
|
||||
import dev.ltms.fleet.rest.FleetApp;
|
||||
import dev.ltms.fleet.session.GitWorktrees;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.session.SessionReaper;
|
||||
import io.javalin.Javalin;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* fleetd #612 Unit A: the real boot assembly, extracted out of {@code Fleetd.main} so a test can
|
||||
* drive it directly. {@link #assembleAndStart} is <em>the same statements {@code main} used to run
|
||||
* inline</em>, in the same order, against a real {@link ResourcePorts} in production and a fake one
|
||||
* in a test — see {@code FleetdAssemblyLifecycleTest}. {@code Fleetd.main} keeps config loading and
|
||||
* {@code cfg.validateAll()}; everything from immediately after that call onward moved here —
|
||||
* including, since fleetd #612 A-gaps (gap 2), the two post-validation reports ({@code
|
||||
* reportRoleFallbackGaps}, {@code assertChartersNameOnlyRegisteredTools}) that Unit A originally
|
||||
* left behind in {@code main}. Those two calls do no I/O themselves, but leaving them outside this
|
||||
* method meant deleting either one compiled clean and left the whole suite green — nothing drove
|
||||
* {@code main} itself, so nothing could notice. They run first here, in the same relative order,
|
||||
* before the herdr socket or anything else that touches the outside world.
|
||||
*
|
||||
* <p><strong>Construction and start order is preserved exactly, on purpose.</strong> This is not
|
||||
* rebuilt into "construct everything, then start everything" — that would change boot timing. The
|
||||
* order recorded before any code moved (see the ticket and {@code FleetdAssemblyLifecycleTest}):
|
||||
* {@code SessionReaper} starts first (if {@code lifecycle.idleTtlSeconds} is configured), then
|
||||
* {@link StatusPoller}, then the optional {@link LeadHeartbeatLoop} and {@link FleetHealthMonitor},
|
||||
* then the optional {@link LeadCoordLoop} and {@link ConfigWatcher}, and the HTTP server starts
|
||||
* last of all. The close order (see {@link FleetdRuntime#close()}) is the mirror the original
|
||||
* shutdown hook always used.
|
||||
*
|
||||
* <p><strong>One statement could not move without reordering startup.</strong> {@code Fleetd.main}
|
||||
* registered its shutdown hook <em>before</em> building the Javalin {@code FleetApp} — the hook
|
||||
* itself never touched {@code app} (it still doesn't; see {@link FleetdRuntime#close()}), but the
|
||||
* hook needs a {@link FleetdRuntime} to close over, and the ticket asks that runtime to also own
|
||||
* {@code FleetApp}. Building the runtime before the app exists and mutating it afterward (via
|
||||
* {@link FleetdRuntime#attachApp}) preserves the exact original order — hook registered, then app
|
||||
* built, then HTTP started — without moving the app's construction earlier or the hook's
|
||||
* registration later. That is the one seam this ticket did not get to pin any other way.
|
||||
*
|
||||
* <p><strong>No inert variant.</strong> Deliberately, there is no overload of this method that
|
||||
* accepts a smaller/optional {@link ResourcePorts} or defaults one internally. A future edit that
|
||||
* wants to skip {@code FleetdAssembly} entirely and build its own graph is still possible — no
|
||||
* static analysis stops that — but it cannot do so by quietly swapping this call for an inert
|
||||
* substitute that still compiles, because none exists.
|
||||
*/
|
||||
final class FleetdAssembly {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(Fleetd.class);
|
||||
|
||||
/** CB-637: how often the lead coordination loop looks for peer messages — see {@code Fleetd}'s own constant. */
|
||||
private static final long LEAD_COORD_INTERVAL_MS = 3_000L;
|
||||
|
||||
private FleetdAssembly() {
|
||||
}
|
||||
|
||||
static FleetdRuntime assembleAndStart(AssemblyInputs inputs, ResourcePorts ports) {
|
||||
FleetConfig cfg = inputs.cfg();
|
||||
ConfigRef config = inputs.config();
|
||||
SubscriptionGuard guard = inputs.guard();
|
||||
|
||||
// fleetd #612 A-gaps (gap 2): moved in from Fleetd.main, immediately after cfg.validateAll()
|
||||
// there — the exact point main used to call these two, and still the first thing this
|
||||
// method does, before any socket or broker work below. See this class's javadoc and each
|
||||
// method's own for why they run here rather than in FleetConfig#validateAll() itself.
|
||||
Fleetd.reportRoleFallbackGaps(cfg);
|
||||
Fleetd.assertChartersNameOnlyRegisteredTools(cfg);
|
||||
|
||||
Path socket = cfg.herdrSocket() != null && !cfg.herdrSocket().isBlank()
|
||||
? Path.of(cfg.herdrSocket())
|
||||
: UnixSocketHerdrClient.defaultSocketPath();
|
||||
|
||||
HerdrClient herdr = ports.connectHerdr(socket);
|
||||
HerdrClient memberHerdr = cfg.memberHerdrSocket() != null && !cfg.memberHerdrSocket().isBlank()
|
||||
? ports.connectHerdr(Path.of(cfg.memberHerdrSocket()))
|
||||
: herdr;
|
||||
AtomicReference<Supplier<Map<String, String>>> leadsRef = new AtomicReference<>(Map::of);
|
||||
// fleetd #669 Unit E: a collaborator's pane is opened by a person, exactly like a lead's,
|
||||
// so its terminal must also route to the lead herdr daemon rather than the member one.
|
||||
AtomicReference<Supplier<Map<String, String>>> collaboratorTerminalsRef = new AtomicReference<>(Map::of);
|
||||
HerdrRouter router = new HerdrRouter(herdr, memberHerdr,
|
||||
target -> leadsRef.get().get().containsKey(target)
|
||||
|| collaboratorTerminalsRef.get().get().containsKey(target));
|
||||
// CB-402: one adapter per configured peer kind, fronted by a composite router. A profile's
|
||||
// `kind:` selects its adapter — claude-code (the default) and opencode partition the profile
|
||||
// set — and the composite dispatches each SPI call to the adapter that owns the profile/pane.
|
||||
Map<String, FleetConfig.Profile> claudeProfiles = new LinkedHashMap<>();
|
||||
Map<String, FleetConfig.Profile> opencodeProfiles = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, w) -> {
|
||||
if (w.isOpenCode()) {
|
||||
opencodeProfiles.put(name, w);
|
||||
} else {
|
||||
claudeProfiles.put(name, w);
|
||||
}
|
||||
});
|
||||
List<HerdrPeerLauncher> adapters = new ArrayList<>();
|
||||
// fleetd #175: the daemon's real ExhaustionSink can only be built once `sessions` exists
|
||||
// (below), but `sessions` needs `workers`, which needs the adapters built right here — a
|
||||
// genuine cycle. Break it exactly like liveCountRef below: a forwarding sink built now,
|
||||
// pointed at the real one once it exists.
|
||||
AtomicReference<ExhaustionSink> exhaustionSinkRef = new AtomicReference<>(ExhaustionSink.none());
|
||||
ExhaustionSink forwardingExhaustionSink = Fleetd.forwardingExhaustionSink(exhaustionSinkRef);
|
||||
// The claude-code adapter is the always-present default; keep it even with no profiles (so a
|
||||
// bridge configured with no workers, or opencode-only, still has a well-defined base adapter)
|
||||
// unless opencode is the only kind configured.
|
||||
if (!claudeProfiles.isEmpty() || opencodeProfiles.isEmpty()) {
|
||||
adapters.add(Fleetd.claudeCodeLauncher(router.memberAgents(), router.memberSpaces(), guard,
|
||||
claudeProfiles, cfg, config));
|
||||
}
|
||||
if (!opencodeProfiles.isEmpty()) {
|
||||
adapters.add(Fleetd.openCodeLauncher(router.memberAgents(), router.memberSpaces(),
|
||||
opencodeProfiles, cfg, config, forwardingExhaustionSink));
|
||||
}
|
||||
AtomicReference<Function<String, Integer>> liveCountRef = new AtomicReference<>(_ -> 0);
|
||||
// CB-578 stage B: one quarantine tracker for the whole daemon, shared between the launcher
|
||||
// (checked at spawn) and the exhaustion sink wired in below (written on BACKEND_EXHAUSTED).
|
||||
BackendQuarantine quarantine = BackendQuarantine.withEscalation(ports.nanoClock(),
|
||||
TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));
|
||||
// fleetd #201 Unit 5: one outage-cool-off tracker for the whole daemon, shared between the
|
||||
// launcher (checked at spawn, like `quarantine` above) and the backend-error sink wired in
|
||||
// below (written on a classified backend error).
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(ports.nanoClock());
|
||||
PeerLauncher workers = new CompositePeerLauncher(
|
||||
adapters,
|
||||
cfg.effectiveDefaultProfile(),
|
||||
config,
|
||||
profileName -> liveCountRef.get().apply(profileName),
|
||||
quarantine,
|
||||
outagePolicy);
|
||||
// fleetd #422 follow-up: say which of the three model-gate states the daemon booted into.
|
||||
log.info("model gate (fleetd #422): {}", Fleetd.modelGateCoverageLine(workers.modelGateState()));
|
||||
// CB-504: under supervision (launchd/systemd) fleetd can start before herdr's socket
|
||||
// exists. Wait, then degrade rather than die: serving with /healthz reporting "degraded" is
|
||||
// strictly more useful than exiting.
|
||||
Fleetd.HerdrAwaitOutcome herdrOutcome = Fleetd.awaitHerdr(herdr, ports.nanoClock(), ports.herdrPollWait());
|
||||
boolean herdrUp = Fleetd.logHerdrWaitOutcomeAndShouldReap(herdrOutcome);
|
||||
if (herdrUp) {
|
||||
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
||||
// with the previous process — reap those leaked orphans now, before we start serving.
|
||||
workers.reapOrphanWorkers();
|
||||
}
|
||||
|
||||
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
||||
// CB-301-ext: worktree provisioning seam, optionally rooted at a configured directory.
|
||||
// CB-303 part 2: context cap is opt-in and disabled (0) when absent/null.
|
||||
int contextCap = 0;
|
||||
if (cfg.lifecycle() != null && cfg.lifecycle().contextCap() != null
|
||||
&& cfg.lifecycle().contextCap() > 0) {
|
||||
contextCap = cfg.lifecycle().contextCap();
|
||||
}
|
||||
boolean clearAfterTurn = cfg.lifecycle() != null && cfg.lifecycle().clearAfterTurn();
|
||||
SessionManager sessions = new SessionManager(workers,
|
||||
new GitWorktrees(cfg.worktreeRoot(), cfg.worktreeGroup(), cfg.memberSkills()),
|
||||
ports.nanoClock(), contextCap, clearAfterTurn);
|
||||
liveCountRef.set(profileName -> Fleetd.liveSessionCount(sessions.roster(), profileName));
|
||||
|
||||
// Idle-sleep guard: hold an OS-level assertion against idle sleep while at least one
|
||||
// member is live. No-op (never constructed) off macOS or when idleSleepGuard.enabled is
|
||||
// explicitly false; the mechanism itself is additionally a no-op if 'caffeinate' cannot be
|
||||
// started, so this can never fail a spawn, a release, or startup.
|
||||
boolean idleSleepGuardEnabled = cfg.idleSleepGuard() == null || cfg.idleSleepGuard().isEnabled();
|
||||
final IdleSleepGuard idleSleepGuard;
|
||||
if (idleSleepGuardEnabled) {
|
||||
idleSleepGuard = new IdleSleepGuard(new CaffeinateSleepAssertionMechanism(), sessions::size);
|
||||
sessions.onAcquire(_ -> idleSleepGuard.recheck());
|
||||
sessions.onRelease(_ -> idleSleepGuard.recheck());
|
||||
} else {
|
||||
idleSleepGuard = null;
|
||||
}
|
||||
|
||||
// CB-303 part 1: idle-ttl reaper — only when configured, defaults to disabled. FIRST of the
|
||||
// recurring background loops to start — see this class's own javadoc for the full order.
|
||||
final SessionReaper reaper;
|
||||
if (cfg.lifecycle() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() > 0) {
|
||||
reaper = new SessionReaper(sessions, cfg.lifecycle().idleTtlSeconds());
|
||||
reaper.start();
|
||||
} else {
|
||||
reaper = null;
|
||||
}
|
||||
|
||||
// CB-530: every pane the config names as a lead, merged from `leaders:` and the legacy
|
||||
// singular pin. PrimaryRegistry below still tracks ONE terminal — it addresses the push
|
||||
// loop's nudges, which need a single destination — so it keeps the legacy pin.
|
||||
Map<String, String> leadTerminals = cfg.leaderTerminals();
|
||||
if (leadTerminals.size() > 1) {
|
||||
log.info("leads: {} panes recognised {}", leadTerminals.size(), leadTerminals.values());
|
||||
}
|
||||
// Discover leads by their tab labels, one scanner per configured lead's own space. fleetd
|
||||
// #669: the same scan also recognises a configured collaborator's tab, so one herdr pass
|
||||
// answers both.
|
||||
final Supplier<Map<String, String>> leads;
|
||||
final Supplier<Map<String, String>> collaboratorTerminals;
|
||||
var leaders = cfg.fleet().leaders();
|
||||
var collaboratorsConfig = cfg.fleet().collaborators();
|
||||
if (!leaders.isEmpty() || !collaboratorsConfig.isEmpty()) {
|
||||
Map<String, Map<String, String>> leadLabelsBySpace = new LinkedHashMap<>();
|
||||
Map<String, String> spaceByLeadName = new LinkedHashMap<>();
|
||||
leaders.forEach((name, leader) -> {
|
||||
if (leader == null) {
|
||||
return;
|
||||
}
|
||||
spaceByLeadName.put(name, leader.workspace());
|
||||
Map<String, String> labelsHere = leadLabelsBySpace
|
||||
.computeIfAbsent(leader.workspace(), k -> new LinkedHashMap<>());
|
||||
leader.acceptedLabels().forEach(label -> labelsHere.put(label, name));
|
||||
if (leader.tab() != null && !leader.tab().isBlank()) {
|
||||
log.warn("lead '{}' (fleet.leaders.{}) still configures tab: \"{}\" — deprecated, "
|
||||
+ "the lead tab label is now fixed to '{}'",
|
||||
name, name, leader.tab(), FleetConfig.Leader.LEAD_TAB_LABEL);
|
||||
}
|
||||
});
|
||||
Map<String, String> collaboratorTabToName = new LinkedHashMap<>();
|
||||
collaboratorsConfig.forEach((name, collaborator) -> {
|
||||
if (collaborator != null && collaborator.tab() != null && !collaborator.tab().isBlank()) {
|
||||
collaboratorTabToName.put(collaborator.tab(), name);
|
||||
}
|
||||
});
|
||||
// A collaborator-only fleet configures no `leaders:` entry to read a scan interval from
|
||||
// — FleetConfig.Collaborator carries no scanIntervalSeconds of its own. Falling back to
|
||||
// FleetConfig.Leader's own compact-constructor default keeps a collaborator-only
|
||||
// deployment on the same rescan cadence as the default lead cadence, instead of
|
||||
// inventing a second number for the same kind of scan.
|
||||
int scanIntervalSeconds = leaders.isEmpty()
|
||||
? 10
|
||||
: leaders.values().iterator().next().scanIntervalSeconds();
|
||||
// This must use the lead daemon: scanning member tabs would demote the lead to a worker.
|
||||
LeadTabScanner scanner = new LeadTabScanner(herdr, leadLabelsBySpace, collaboratorTabToName,
|
||||
Set.of(), TimeUnit.SECONDS.toNanos(scanIntervalSeconds), ports.nanoClock());
|
||||
leads = scanner;
|
||||
collaboratorTerminals = scanner::collaborators;
|
||||
log.info("lead/collaborator scan: space per lead {}, tabs {} host a collaborator "
|
||||
+ "(rescan every {}s)",
|
||||
spaceByLeadName, collaboratorTabToName.keySet(), scanIntervalSeconds);
|
||||
} else {
|
||||
leads = () -> leadTerminals;
|
||||
collaboratorTerminals = Map::of;
|
||||
}
|
||||
leadsRef.set(leads);
|
||||
collaboratorTerminalsRef.set(collaboratorTerminals);
|
||||
|
||||
// Constructed unconditionally — it is cheap and side-effect free — so a LeadRollover built
|
||||
// below can relaunch a lead even on a boot where herdr was down for the ensureLeads() call.
|
||||
LeadLauncher leadLauncher = new LeadLauncher(router.leadAgents(), router.leadSpaces(), cfg);
|
||||
|
||||
// CB-558: start any declared lead that is not already running. After the scanner is built,
|
||||
// and only when herdr answered — the launcher's whole safety property is that it can count
|
||||
// live leads first, and must never guess and risk a second orchestrator.
|
||||
if (herdrUp && !leaders.isEmpty()) {
|
||||
int launched = leadLauncher.ensureLeads();
|
||||
if (launched > 0) {
|
||||
log.info("lead auto-launch: {} lead(s) started", launched);
|
||||
}
|
||||
}
|
||||
|
||||
// CB-548: config-declared architect slots. Nothing here spawns a slot; the terminal → slot
|
||||
// binding is owned by the registry and empty at startup.
|
||||
MemberRegistry members = MemberRegistry.live(() -> config.get().fleet());
|
||||
sessions.setMemberLifecycle(members);
|
||||
if (!members.slots().isEmpty()) {
|
||||
log.info("member slots: {} configured {} — none bound yet (a slot is idle until the "
|
||||
+ "spawn lifecycle binds a live terminal to it)",
|
||||
members.slots().size(), members.slots().keySet());
|
||||
}
|
||||
|
||||
// Status-gated injector (CB-103): the single writer into workers, fed by a poller.
|
||||
// CB-106: a confirmed turn completion resolves a blocked send whose worker never replied.
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
// CB-578 stage A: classify a completion-fallback scrape that matches a profile's configured
|
||||
// usage-limit refusal as BACKEND_EXHAUSTED rather than handing it back as a real answer.
|
||||
LiveExhaustedPatterns liveExhaustedPatterns = Fleetd.liveExhaustedPatterns(config);
|
||||
ExhaustedPatternLookup exhaustedPatterns = Fleetd.exhaustedPatternLookup(sessions::roster, liveExhaustedPatterns);
|
||||
// The startup coverage line still reports the boot-time snapshot only.
|
||||
Set<String> exhaustedConfiguredAtStartup = cfg.profiles().entrySet().stream()
|
||||
.filter(e -> e.getValue().hasExhaustedPattern())
|
||||
.map(Map.Entry::getKey)
|
||||
.collect(Collectors.toCollection(LinkedHashSet::new));
|
||||
log.info("backend-exhausted classification (CB-578 stage A): {}",
|
||||
Fleetd.exhaustedPatternCoverageLine(cfg.profiles().keySet(), exhaustedConfiguredAtStartup));
|
||||
// fleetd #201 Unit 5: classify a completion-fallback scrape that matches a profile's
|
||||
// configured backend-error refusal (a credential outage, a provider 5xx) as a backend error
|
||||
// rather than handing it back as a real answer.
|
||||
Map<String, Pattern> errorPatternsByProfile = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (profile.hasErrorPattern()) {
|
||||
errorPatternsByProfile.put(name, Pattern.compile(profile.errorPattern()));
|
||||
}
|
||||
});
|
||||
BackendErrorPatternLookup backendErrorPatterns = Fleetd.backendErrorPatternLookup(sessions::roster,
|
||||
errorPatternsByProfile);
|
||||
log.info("backend-error classification (fleetd #201 Unit 5): {}",
|
||||
Fleetd.errorPatternCoverageLine(cfg.profiles().keySet(), errorPatternsByProfile.keySet()));
|
||||
// CB-578 stage B: on a classification that actually wins, quarantine the exhausted profile's
|
||||
// CREDENTIAL, so a profile sharing that credential is refused too, not just the one that
|
||||
// happened to report it.
|
||||
Map<String, String> quarantineReasonByCredential = new ConcurrentHashMap<>();
|
||||
// fleetd #175: point the forwarding sink handed to OpenCodeLauncher above at the real one,
|
||||
// now that `sessions` exists to resolve target -> session -> profile.
|
||||
ExhaustionSink exhaustionSink = Fleetd.publishExhaustionSink(exhaustionSinkRef, sessions, config,
|
||||
quarantine, quarantineReasonByCredential, cfg);
|
||||
// fleetd #201 Unit 5: the production BackendErrorSink needs `pushLoop` (built further below,
|
||||
// after `sessions`) to tell a lead about an incident or an unmapped target — the same
|
||||
// construction-order cycle `exhaustionSinkRef` breaks above, broken the same way: a mutable
|
||||
// holder set once `pushLoop` exists, read lazily from inside the lambda built here.
|
||||
AtomicReference<ReplyPushLoop> pushLoopRef = new AtomicReference<>();
|
||||
BackendErrorSink backendErrorSink = Fleetd.backendErrorSink(sessions, () -> config.get().profiles(),
|
||||
outagePolicy, pushLoopRef::get);
|
||||
AgentControl agents = router.memberAgents();
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous, exhaustedPatterns,
|
||||
exhaustionSink, backendErrorPatterns, backendErrorSink, ports.nanoClock(),
|
||||
Fleetd.worktreeBranchLookup(sessions::roster));
|
||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||
MemberPresence presence = sessions.asPresence();
|
||||
TurnListener turnListener = Fleetd.turnListener(completion, sessions);
|
||||
Predicate<String> deliverable = Fleetd.deliverableTo(presence, leads, collaboratorTerminals);
|
||||
// fleetd #556: registration is wired directly to `completion`, not folded into the
|
||||
// `turnListener` fan-out above — so it survives `sessions.onDelivered` (or any future
|
||||
// listener) throwing, regardless of call order.
|
||||
Injector injector = new Injector(router, turnListener, deliverable,
|
||||
presence::forget, Fleetd.turnRegistrar(completion));
|
||||
StatusPoller poller = new StatusPoller(router, injector, Injector.POLL_INTERVAL_MILLIS);
|
||||
poller.start(); // SECOND of the recurring background loops to start, after the reaper.
|
||||
|
||||
// CB-307: reply inbox. A broker: block selects the AMQP-backed durable adapter; absent (or
|
||||
// unusable), fleetd stays soft-state on the in-memory inbox. The AMQP inbox owns a broker
|
||||
// connection, so keep the reference to close it in the ordered shutdown hook.
|
||||
final ReplyInbox replyInbox = Fleetd.selectReplyInbox(cfg.broker(), ports.environment(),
|
||||
ports.replyInboxOpener());
|
||||
// CB-637: this daemon's lead-to-lead mailbox on the SHARED coordination vhost — a separate
|
||||
// broker from the reply inbox by design. Absent a coordinator: block this is null and every
|
||||
// lead path below is simply not wired, exactly the behaviour before this ticket. It owns a
|
||||
// broker connection, so keep the reference for the ordered shutdown hook.
|
||||
final LeadChannelHandle leadMailbox = Fleetd.openLeadMailbox(cfg.coordinator(), ports.environment(),
|
||||
ports.leadMailboxOpener());
|
||||
// CB-307: learn the primary's terminal from orchestration tool calls (or pin from config).
|
||||
String pinnedPrimaryTerminal = cfg.primary() != null ? cfg.primary().terminal() : null;
|
||||
PrimaryRegistry primaryRegistry = new PrimaryRegistry(pinnedPrimaryTerminal,
|
||||
Fleetd.currentTerminalForName(leads));
|
||||
// CB-532: `primary.terminal` is superseded and no longer needed for either of its jobs.
|
||||
if (pinnedPrimaryTerminal != null && !pinnedPrimaryTerminal.isBlank()) {
|
||||
log.warn("primary.terminal is DEPRECATED (CB-532) and can be deleted: identity now comes "
|
||||
+ "from leaders:/leadScan:, and reply nudges follow the lead that delegated. "
|
||||
+ "It still works, and is still the fallback nudge destination when a restart "
|
||||
+ "has lost the delegation map. Its pushReminders/pushBackoffMs stay valid.");
|
||||
}
|
||||
// CB-307: active push-to-primary loop — nudge the primary when replies land without an
|
||||
// open fleet_send. Uses its own lightweight scheduled executor, separate from the injector.
|
||||
int maxReminders = cfg.primary() != null ? cfg.primary().remindersOrDefault() : 5;
|
||||
long backoffMs = cfg.primary() != null ? cfg.primary().backoffMsOrDefault() : 15_000L;
|
||||
var pushScheduler = ports.newScheduler("bridge-push-");
|
||||
// CB-502: the registry is built before the service and the push loop so send/reply outcomes
|
||||
// are counted at their single funnel rather than at each of the two caller-facing surfaces.
|
||||
Metrics metrics = FleetMetrics.create(sessions, replyInbox);
|
||||
var pushLoop = new ReplyPushLoop(primaryRegistry, router.leadAgents(), replyInbox,
|
||||
pushScheduler, maxReminders, backoffMs, metrics);
|
||||
// fleetd #201 Unit 5: point the forwarding holder captured by the backendErrorSink lambda
|
||||
// above at the real push loop, now that it exists.
|
||||
pushLoopRef.set(pushLoop);
|
||||
// CB-551: idle-lead heartbeat. Opt-in; absent `leadHeartbeat:` this is never constructed, so
|
||||
// an upgraded daemon cannot silently start spending subscription on nudging an idle lead.
|
||||
// THIRD of the recurring background loops to start (optional).
|
||||
final LeadHeartbeatLoop heartbeat;
|
||||
var heartbeatScheduler = ports.newScheduler("bridge-heartbeat-");
|
||||
if (cfg.leadHeartbeat() != null) {
|
||||
var hb = cfg.leadHeartbeat();
|
||||
var leadContextGauge = new LeadContextGauge();
|
||||
// fleetd #621: the context-high notice's own wording must track this same effective
|
||||
// value — LeadRollover.confirm(...) already gates the roll on it (LeadRollover.java:480),
|
||||
// and absent `leadRollover:` entirely the roll is unusable regardless (NOT_CONFIGURED),
|
||||
// so `true` (the FleetConfig.LeadRollover default) is the safe, byte-identical fallback.
|
||||
// Carried in from Fleetd.main when #612 Unit A merged main: #622 added this line to the
|
||||
// block Unit A had already moved here, so the merge would otherwise have silently
|
||||
// dropped it — with a fully green suite, because nothing pins it (see the follow-up issue).
|
||||
boolean requireOperatorConfirm = cfg.leadRollover() == null || cfg.leadRollover().requireOperatorConfirm();
|
||||
heartbeat = new LeadHeartbeatLoop(primaryRegistry, router.leadAgents(), replyInbox, sessions::roster,
|
||||
pushLoop, heartbeatScheduler, ports.nanoClock(),
|
||||
TimeUnit.SECONDS.toNanos(hb.idleAfterSeconds()), hb.backoffMs(), hb.quietNudgeCap(),
|
||||
metrics,
|
||||
Fleetd.leadContextSource(leadContextGauge, router.leadAgents(), leads,
|
||||
Fleetd.leadConfigDirLookup(() -> config.get().profiles(), leaders),
|
||||
Fleetd.leadContextWindowLookup(() -> config.get().profiles(), leaders)),
|
||||
Boolean.TRUE.equals(hb.contextHighNudge()), requireOperatorConfirm);
|
||||
heartbeat.start();
|
||||
} else {
|
||||
heartbeat = null;
|
||||
heartbeatScheduler.shutdownNow();
|
||||
}
|
||||
// fleetd #480: lead rollover. Opt-in; absent `leadRollover:` this is never constructed.
|
||||
LeadRollover leadRollover = Fleetd.leadRollover(cfg, router.leadAgents(), router.leadSpaces(),
|
||||
leadLauncher, config, leads);
|
||||
MessageService messages = new MessageService(router, injector, rendezvous, replyInbox,
|
||||
pushLoop, metrics);
|
||||
|
||||
// Health is a slow whole-fleet observer. Keep it separate from the 250ms delivery poller.
|
||||
// FOURTH of the recurring background loops to start (optional).
|
||||
final FleetHealthMonitor healthMonitor;
|
||||
var healthScheduler = ports.newScheduler("bridge-health-");
|
||||
if (cfg.health() != null && cfg.health().isEnabled()) {
|
||||
// CB-580: a member found GONE/NEVER_READY must fail whatever ticket is waiting on it.
|
||||
healthMonitor = new FleetHealthMonitor(agents, sessions::roster, messages, healthScheduler,
|
||||
ports.nanoClock(), ports.wallClockNanos(),
|
||||
cfg.health().intervalOrDefault(),
|
||||
cfg.health().workingSuspectAfterOrDefault(), Fleetd.healthFailTarget(messages));
|
||||
String coverage = FleetHealthMonitor.coverage(true,
|
||||
cfg.health().notifications() != null && cfg.health().notifications().configured());
|
||||
if ("detection-only".equals(coverage)) {
|
||||
log.warn("fleet health: {} (no notification sink configured)", coverage);
|
||||
} else {
|
||||
log.info("fleet health: {}", coverage);
|
||||
}
|
||||
healthMonitor.start();
|
||||
} else {
|
||||
healthMonitor = null;
|
||||
healthScheduler.shutdownNow();
|
||||
}
|
||||
|
||||
// CB-520: the reply inbox only consumes for agents this gateway owns. own on acquire,
|
||||
// release on teardown. Do this before CB-516 so the inbox is owned before any reply can land.
|
||||
sessions.onAcquire(replyInbox::own);
|
||||
// CB-516: releasing a worker must fail whatever send was waiting on it.
|
||||
sessions.onRelease(Fleetd.releaseCleanup(messages, replyInbox, primaryRegistry));
|
||||
|
||||
// MCP server face (CB-105): fleet_send/fleet_reply/fleet_status, mounted at /mcp.
|
||||
// Caller identity is resolved from the connection (peer PID → herdr pane), not arguments.
|
||||
ConnectionIdentity identity = new ConnectionIdentity(
|
||||
new PaneLocator(herdr, memberHerdr), new LsofPeerPidLookup(), new LsofProcessCwdLookup());
|
||||
|
||||
// fleetd #669 Unit D: a live spawned member resolves as its own role, whatever a tab map
|
||||
// says about the same terminal. fleetd #702: SessionManager.spawnedMemberRole also answers
|
||||
// for a pane mid-teardown, not only one still in the registry — see its javadoc.
|
||||
Function<String, MemberRole> spawnedMemberRole = sessions::spawnedMemberRole;
|
||||
|
||||
// CB-501: one resolver behind both entry paths. Worker identity still comes from the
|
||||
// connection and is never token-gated, so enabling token mode cannot lock the fleet out.
|
||||
final CallerResolver callers;
|
||||
if (cfg.auth().tokenMode()) {
|
||||
String token = ports.environment().get(cfg.auth().tokenEnv());
|
||||
if (token == null || token.isBlank()) {
|
||||
throw new IllegalStateException("auth.mode=token but env var " + cfg.auth().tokenEnv()
|
||||
+ " is unset or empty — export it before starting fleetd");
|
||||
}
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, true, token, leads, members,
|
||||
spawnedMemberRole, collaboratorTerminals);
|
||||
log.info("auth: token mode (bearer required for non-worker callers, env {})",
|
||||
cfg.auth().tokenEnv());
|
||||
} else {
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, false, null, leads, members,
|
||||
spawnedMemberRole, collaboratorTerminals);
|
||||
log.info("auth: loopback-trust (any loopback non-worker caller is the primary)");
|
||||
}
|
||||
|
||||
FleetMcp.QuarantineSource quarantineSource = Fleetd.quarantineSource(config, quarantine,
|
||||
liveExhaustedPatterns, quarantineReasonByCredential);
|
||||
FleetMcp.OutageSource outageSource = new FleetMcp.OutageSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, outagePolicy);
|
||||
|
||||
FleetMcp.LoopHealthSource loopHealth = Fleetd.loopHealthSource(poller, reaper);
|
||||
FleetMcp mcp = new FleetMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, FleetMcp.AuthorizationMode.ENFORCED, metrics,
|
||||
Fleetd.capacitySource(config, cfg, profile -> liveCountRef.get().apply(profile)),
|
||||
Fleetd.healthCoverageSource(config),
|
||||
loopHealth,
|
||||
quarantineSource,
|
||||
leadMailbox,
|
||||
outageSource,
|
||||
new FleetMcp.LeadSeatSource(Fleetd.leadSeatLookup(() -> config.get().profiles(), leaders, leads)),
|
||||
Fleetd.leadConfigDirSource(() -> config.get().profiles(), leaders),
|
||||
cfg.coordinator() == null ? List.of() : cfg.coordinator().peers(),
|
||||
leadRollover);
|
||||
|
||||
// CB-637: the receive half. Only constructed when a lead mailbox actually opened.
|
||||
final LeadCoordLoop leadCoordLoop;
|
||||
final ScheduledExecutorService leadCoordSchedulerRef;
|
||||
if (leadMailbox != null) {
|
||||
var leadCoordScheduler = ports.newScheduler("bridge-leadcoord-");
|
||||
leadCoordLoop = new LeadCoordLoop(leadMailbox, router.leadAgents(), leads, leadCoordScheduler,
|
||||
LEAD_COORD_INTERVAL_MS);
|
||||
leadCoordLoop.start(); // FIFTH of the recurring background loops to start (optional).
|
||||
leadCoordSchedulerRef = leadCoordScheduler;
|
||||
} else {
|
||||
leadCoordLoop = null;
|
||||
leadCoordSchedulerRef = null;
|
||||
}
|
||||
|
||||
// CB-559: opt-in config reload. With no `configReload:` block nothing is constructed.
|
||||
// SIXTH of the recurring background loops to start (optional).
|
||||
final ConfigWatcher configWatcher;
|
||||
if (cfg.configReload() != null && cfg.configReload().isEnabled()) {
|
||||
configWatcher = new ConfigWatcher(config, cfg.configReload().intervalSeconds());
|
||||
configWatcher.start();
|
||||
} else {
|
||||
configWatcher = null;
|
||||
}
|
||||
|
||||
// CB-303 part 3: single ordered shutdown hook — see FleetdRuntime#close() for the statements
|
||||
// this used to be. Registered here, at the exact point `main` used to register it: after
|
||||
// configWatcher, before the Javalin app exists (see this class's own javadoc for why).
|
||||
FleetdRuntime runtime = new FleetdRuntime(cfg, sessions, router, poller, messages, pushLoop, heartbeat,
|
||||
leadCoordLoop, leadCoordSchedulerRef, healthMonitor, configWatcher, mcp, reaper, idleSleepGuard,
|
||||
replyInbox, leadMailbox, completion, injector);
|
||||
ports.addShutdownHook(runtime::close);
|
||||
|
||||
// CB-185: give FleetApp both daemons — /healthz must require both to answer and
|
||||
// GET /sessions must merge across both, or a down/unpolled member daemon is invisible.
|
||||
Javalin app = new FleetApp(herdr, memberHerdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics, deliverable,
|
||||
() -> MemberCredentialPolicyView.of(config.get().memberCredentials()),
|
||||
quarantineSource, outageSource, loopHealth).build();
|
||||
runtime.attachApp(app);
|
||||
ports.startHttp(app, cfg.bind().host(), cfg.bind().port()); // HTTP starts LAST, always.
|
||||
log.info("fleetd listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
return runtime;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,166 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigWatcher;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.health.FleetHealthMonitor;
|
||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||
import dev.ltms.fleet.inject.CompletionResolver;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.inject.StatusPoller;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||
import dev.ltms.fleet.msg.LeadCoordLoop;
|
||||
import dev.ltms.fleet.msg.LeadHeartbeatLoop;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||
import dev.ltms.fleet.power.IdleSleepGuard;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.session.SessionReaper;
|
||||
import io.javalin.Javalin;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
|
||||
/**
|
||||
* fleetd #612 Unit A: the assembled daemon. {@link FleetdAssembly#assembleAndStart} builds exactly
|
||||
* one of these and hands it to {@code ResourcePorts.addShutdownHook}; {@link #close()} is the
|
||||
* single ordered shutdown, moved verbatim out of {@code Fleetd.main}'s old shutdown-hook
|
||||
* {@code Thread} body — same statements, same order, see that method's javadoc.
|
||||
*
|
||||
* <p><strong>Owns the real production objects, never a copy.</strong> Every package-private
|
||||
* accessor below returns the identical instance the running daemon is using. That is the entire
|
||||
* point of this class existing (see fleetd #612's problem statement): a test that inspected a
|
||||
* snapshot built alongside the real objects could pass while production silently received
|
||||
* something else — the exact shape of the #602/#606 defect this ticket exists to stop from
|
||||
* recurring one call site at a time. Nothing here is rebuilt or copied for a test's benefit.
|
||||
*/
|
||||
final class FleetdRuntime implements AutoCloseable {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(FleetdRuntime.class);
|
||||
|
||||
private final FleetConfig cfg;
|
||||
private final SessionManager sessions;
|
||||
private final HerdrRouter router;
|
||||
private final StatusPoller poller;
|
||||
private final MessageService messages;
|
||||
private final ReplyPushLoop pushLoop;
|
||||
private final LeadHeartbeatLoop heartbeat; // nullable — leadHeartbeat: opt-in
|
||||
private final LeadCoordLoop leadCoordLoop; // nullable — coordinator: opt-in
|
||||
private final ScheduledExecutorService leadCoordScheduler; // nullable, paired with leadCoordLoop
|
||||
private final FleetHealthMonitor healthMonitor; // nullable — health.enabled opt-in
|
||||
private final ConfigWatcher configWatcher; // nullable — configReload.enabled opt-in
|
||||
private final FleetMcp mcp;
|
||||
private final SessionReaper reaper; // nullable — lifecycle.idleTtlSeconds opt-in
|
||||
private final IdleSleepGuard idleSleepGuard; // nullable — idleSleepGuard.enabled: false
|
||||
private final ReplyInbox replyInbox;
|
||||
private final LeadChannelHandle leadMailbox; // nullable — coordinator: opt-in
|
||||
private final CompletionResolver completion;
|
||||
private final Injector injector;
|
||||
/**
|
||||
* Not final: {@code Fleetd.main}'s shutdown hook was registered <em>before</em> the Javalin
|
||||
* {@code FleetApp} was built and started — see {@link FleetdAssembly#assembleAndStart}'s javadoc
|
||||
* for why that order could not be preserved AND have this constructor take {@code app}. {@link
|
||||
* #attachApp} is called immediately after the real app is built, still before HTTP starts
|
||||
* listening, so this is set long before any test or caller could observe it unset.
|
||||
*/
|
||||
private Javalin app;
|
||||
|
||||
FleetdRuntime(FleetConfig cfg, SessionManager sessions, HerdrRouter router, StatusPoller poller,
|
||||
MessageService messages, ReplyPushLoop pushLoop, LeadHeartbeatLoop heartbeat,
|
||||
LeadCoordLoop leadCoordLoop, ScheduledExecutorService leadCoordScheduler,
|
||||
FleetHealthMonitor healthMonitor, ConfigWatcher configWatcher, FleetMcp mcp,
|
||||
SessionReaper reaper, IdleSleepGuard idleSleepGuard, ReplyInbox replyInbox,
|
||||
LeadChannelHandle leadMailbox, CompletionResolver completion, Injector injector) {
|
||||
this.cfg = cfg;
|
||||
this.sessions = sessions;
|
||||
this.router = router;
|
||||
this.poller = poller;
|
||||
this.messages = messages;
|
||||
this.pushLoop = pushLoop;
|
||||
this.heartbeat = heartbeat;
|
||||
this.leadCoordLoop = leadCoordLoop;
|
||||
this.leadCoordScheduler = leadCoordScheduler;
|
||||
this.healthMonitor = healthMonitor;
|
||||
this.configWatcher = configWatcher;
|
||||
this.mcp = mcp;
|
||||
this.reaper = reaper;
|
||||
this.idleSleepGuard = idleSleepGuard;
|
||||
this.replyInbox = replyInbox;
|
||||
this.leadMailbox = leadMailbox;
|
||||
this.completion = completion;
|
||||
this.injector = injector;
|
||||
}
|
||||
|
||||
/** See the {@link #app} field doc for why this is a late-bound setter rather than a constructor arg. */
|
||||
void attachApp(Javalin app) {
|
||||
this.app = app;
|
||||
}
|
||||
|
||||
// --- package-private accessors: the SAME instances this runtime owns, never a copy ---------
|
||||
|
||||
SessionManager sessions() { return sessions; }
|
||||
HerdrRouter router() { return router; }
|
||||
StatusPoller poller() { return poller; }
|
||||
MessageService messages() { return messages; }
|
||||
ReplyPushLoop pushLoop() { return pushLoop; }
|
||||
LeadHeartbeatLoop heartbeat() { return heartbeat; }
|
||||
LeadCoordLoop leadCoordLoop() { return leadCoordLoop; }
|
||||
FleetHealthMonitor healthMonitor() { return healthMonitor; }
|
||||
ConfigWatcher configWatcher() { return configWatcher; }
|
||||
FleetMcp mcp() { return mcp; }
|
||||
SessionReaper reaper() { return reaper; }
|
||||
IdleSleepGuard idleSleepGuard() { return idleSleepGuard; }
|
||||
ReplyInbox replyInbox() { return replyInbox; }
|
||||
LeadChannelHandle leadMailbox() { return leadMailbox; }
|
||||
CompletionResolver completion() { return completion; }
|
||||
Injector injector() { return injector; }
|
||||
Javalin app() { return app; }
|
||||
|
||||
/**
|
||||
* CB-303 part 3: the single ordered shutdown. Moved verbatim out of {@code Fleetd.main}'s
|
||||
* shutdown-hook {@code Thread} body (fleetd #612 Unit A) — drain sessions first while herdr is
|
||||
* still open (so releases reach the daemon), then stop poller/message/mcp/reaper, and close
|
||||
* herdr last, exactly as before. {@code Fleetd.main} never calls this directly; it hands the
|
||||
* reference to {@code ResourcePorts.addShutdownHook} the moment this runtime exists, the same
|
||||
* point it used to register the hook {@code Thread} itself.
|
||||
*/
|
||||
@Override
|
||||
public void close() {
|
||||
sessions.close(cfg.lifecycle() != null ? cfg.lifecycle().drainTimeoutSeconds() : null);
|
||||
poller.stop();
|
||||
messages.close();
|
||||
pushLoop.close();
|
||||
if (heartbeat != null) heartbeat.close(); // CB-551: stop the idle-lead heartbeat scheduler
|
||||
if (leadCoordLoop != null) leadCoordLoop.close(); // CB-637: stop delivering peer-lead messages
|
||||
if (leadCoordScheduler != null) leadCoordScheduler.shutdownNow();
|
||||
if (healthMonitor != null) healthMonitor.stop();
|
||||
if (configWatcher != null) configWatcher.stop(); // CB-559: stop polling the config file
|
||||
mcp.close();
|
||||
if (reaper != null) reaper.stop();
|
||||
// Idle-sleep guard: release unconditionally, even though sessions.close() above already
|
||||
// drained every session (and each release already drove the live count to 0, which
|
||||
// releases the guard's assertion on its own) — this is the backstop for a drain that was
|
||||
// itself interrupted or threw, so no caffeinate child ever outlives the daemon.
|
||||
if (idleSleepGuard != null) idleSleepGuard.close();
|
||||
// Release the broker connection last among message resources (no-op for the in-memory inbox).
|
||||
if (replyInbox instanceof AutoCloseable closeable) {
|
||||
try {
|
||||
closeable.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("reply inbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
// CB-637: the coordination connection goes with it — after the loop that reads it has
|
||||
// stopped, so no tick can be mid-ack against a closed channel.
|
||||
if (leadMailbox != null) {
|
||||
try {
|
||||
leadMailbox.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("lead mailbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
router.close();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import io.javalin.Javalin;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
* fleetd #612 Unit A: every boot-time side effect {@link FleetdAssembly#assembleAndStart} performs
|
||||
* that a real daemon must do for real, and a test must not — read the process environment, connect
|
||||
* a herdr client, open a broker (the reply inbox, the lead mailbox), read a clock, start a
|
||||
* background scheduler, register the JVM shutdown hook, and bind the HTTP server.
|
||||
*
|
||||
* <p>{@link #system()} is the one production implementation ({@code SystemResourcePorts}), wired
|
||||
* verbatim from what {@code Fleetd.main} used to call directly at each of these call sites. A test
|
||||
* builds its own implementation instead of receiving an inert default from this interface —
|
||||
* deliberately, there is no {@code ResourcePorts.none()}. fleetd #612's whole problem is a call
|
||||
* site quietly swapped for an inert variant that still compiles; adding one here, even for tests,
|
||||
* would hand a future edit to {@code FleetdAssembly} the exact compiling substitute this ticket
|
||||
* exists to rule out. A test that wants an inert resource writes its own fake and owns that
|
||||
* decision explicitly.
|
||||
*/
|
||||
public interface ResourcePorts {
|
||||
|
||||
/** The process environment. Production: {@link System#getenv()}. */
|
||||
Map<String, String> environment();
|
||||
|
||||
/**
|
||||
* Connect a herdr client bound to {@code socketPath}. Production returns a real
|
||||
* {@code UnixSocketHerdrClient} — connection-per-call, so this itself never touches the socket.
|
||||
*/
|
||||
HerdrClient connectHerdr(Path socketPath);
|
||||
|
||||
/** The reply-inbox AMQP opener (CB-307). Production: {@link Fleetd#replyInboxOpener()}. */
|
||||
Fleetd.AmqpOpener replyInboxOpener();
|
||||
|
||||
/** The lead-mailbox AMQP opener (CB-637). Production: {@link Fleetd#leadMailboxOpener()}. */
|
||||
Fleetd.LeadMailboxOpener leadMailboxOpener();
|
||||
|
||||
/** A monotonic elapsed-time clock. Production: {@link System#nanoTime()}. */
|
||||
LongSupplier nanoClock();
|
||||
|
||||
/**
|
||||
* fleetd #629: the per-poll wait {@code FleetdAssembly#assembleAndStart} passes to {@code
|
||||
* Fleetd#awaitHerdr} while polling for herdr's socket. Production: {@link
|
||||
* Fleetd#sleepHerdrPoll()} — a real {@code Thread.sleep}. {@link #nanoClock()} alone is not
|
||||
* enough to make {@code awaitHerdr}'s deadline controllable: the old call site passed {@code
|
||||
* Fleetd::sleepHerdrPoll} directly, hardcoded, so a test that injected a fake clock still had
|
||||
* to wait out the real sleep between each poll to ever reach the deadline — the clock looked
|
||||
* injected and was not actually controllable. A test supplies a no-op that advances its own
|
||||
* injected {@link #nanoClock()} instead, so the deadline becomes reachable without any real
|
||||
* wall-clock time passing.
|
||||
*/
|
||||
Runnable herdrPollWait();
|
||||
|
||||
/**
|
||||
* A wall-clock reading, in nanoseconds. Production: {@code System.currentTimeMillis()}
|
||||
* converted to nanoseconds. Kept separate from {@link #nanoClock()} because {@link
|
||||
* dev.ltms.fleet.health.FleetHealthMonitor} needs both — one monotonic clock for elapsed-time
|
||||
* decisions, one wall clock to detect and correct for a macOS sleep freezing the monotonic one.
|
||||
*/
|
||||
LongSupplier wallClockNanos();
|
||||
|
||||
/** A dedicated single-thread scheduler; production names its (virtual) thread {@code purpose}. */
|
||||
ScheduledExecutorService newScheduler(String purpose);
|
||||
|
||||
/** Register a JVM shutdown hook that runs {@code hook} on JVM exit. */
|
||||
void addShutdownHook(Runnable hook);
|
||||
|
||||
/** Bind and start the HTTP server. */
|
||||
void startHttp(Javalin app, String host, int port);
|
||||
|
||||
/** The real production ports: a live herdr socket, a live broker, real threads, a real bind. */
|
||||
static ResourcePorts system() {
|
||||
return new SystemResourcePorts();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,71 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.UnixSocketHerdrClient;
|
||||
import io.javalin.Javalin;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
* fleetd #612 Unit A: the one production {@link ResourcePorts} — every method here is the exact
|
||||
* call {@code Fleetd.main} used to make directly at each of these sites before this ticket.
|
||||
* Package-private: obtained only through {@link ResourcePorts#system()}.
|
||||
*/
|
||||
final class SystemResourcePorts implements ResourcePorts {
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return System.getenv();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return UnixSocketHerdrClient.connect(socketPath, new ObjectMapper());
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return Fleetd.replyInboxOpener();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return Fleetd.leadMailboxOpener();
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
return Fleetd::sleepHerdrPoll;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return () -> TimeUnit.MILLISECONDS.toNanos(System.currentTimeMillis());
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor(r -> Thread.ofVirtual().name(purpose).unstarted(r));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
Runtime.getRuntime().addShutdownHook(new Thread(hook));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
app.start(host, port);
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,7 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import java.util.function.Predicate;
|
||||
|
||||
/**
|
||||
* The authorization table (CB-505), stated once and enforced on both entry paths.
|
||||
*
|
||||
@@ -20,16 +22,29 @@ public final class Authz {
|
||||
SPAWN,
|
||||
/** Tear a worker peer down. */
|
||||
STOP,
|
||||
/** Deliver a turn to a session (or answer a worker's question). */
|
||||
/** Deliver a turn to a local session, addressed by {@code sessionId}. */
|
||||
SEND,
|
||||
/** Resolve a worker's blocked question and resume its turn, addressed by {@code turnId}. */
|
||||
ANSWER,
|
||||
/** Address a peer lead on another daemon over the coordination broker, by {@code coordId}. */
|
||||
COORD_SEND,
|
||||
/** A worker's terminal reply for its own turn. */
|
||||
REPLY,
|
||||
/** A worker's mid-turn question to the primary. */
|
||||
ASK,
|
||||
/**
|
||||
* Collect the messages queued for the caller's OWN pane, instead of having them typed into
|
||||
* its terminal. Grouped with {@link #REPLY} and {@link #ASK} below as an only-as-itself
|
||||
* action: the pane is always the caller's connection-resolved terminal, never an argument,
|
||||
* so no caller can collect another pane's mail.
|
||||
*/
|
||||
INBOX,
|
||||
/** Collect held replies from a session's inbox. */
|
||||
DRAIN,
|
||||
/** Read-only observation: status, roster, profiles, task polling. */
|
||||
/** Read-only roster, profile, and identity observation: no ticket, task, or turn state. */
|
||||
READ,
|
||||
/** Poll a ticket, or read a session's status. */
|
||||
TASK_READ,
|
||||
/**
|
||||
* Read (never ack) this daemon's own held lead-to-lead coordination mail (fleetd #421).
|
||||
*
|
||||
@@ -54,43 +69,125 @@ public final class Authz {
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code caller} may perform {@code action} against {@code targetSession}.
|
||||
* The fail-closed classifier: answers no for every target, so a collaborator's {@code SEND}
|
||||
* is refused unless a caller supplies a real one. {@code CallerResolver#knownLeadOrCollaborator()}
|
||||
* is the real one, read from the same lead and collaborator maps {@code CallerResolver#resolve}
|
||||
* consults, so a target that classifier calls known is one {@code resolve} would actually
|
||||
* resolve as a lead or collaborator.
|
||||
*/
|
||||
public static final Predicate<String> NO_KNOWN_LEAD_OR_COLLABORATOR = target -> false;
|
||||
|
||||
/**
|
||||
* The fail-closed classifier for an observer's {@code SEND}: answers no for every target, so
|
||||
* the grant is refused unless a caller supplies a real one. {@code
|
||||
* CallerResolver#sendableObserverTarget()} is the real one, read from the same maps {@code
|
||||
* CallerResolver#resolve} consults, so a target that classifier calls known is one {@code
|
||||
* resolve} would actually resolve as {@link Role#OBSERVER}.
|
||||
*/
|
||||
public static final Predicate<String> NO_KNOWN_OBSERVER_TARGET = target -> false;
|
||||
|
||||
/**
|
||||
* Convenience form for a caller with no classifier to supply. Fails closed: a collaborator's
|
||||
* or an observer's {@code SEND} is refused, as if no terminal were a configured lead,
|
||||
* collaborator, or observer target — the same decision {@link #NO_KNOWN_LEAD_OR_COLLABORATOR}
|
||||
* and {@link #NO_KNOWN_OBSERVER_TARGET} give explicitly. Every other action's result is
|
||||
* identical to the five-argument form's, since none of them consult either classifier.
|
||||
*
|
||||
* @param targetSession the session id in the request path; only consulted for the worker-scoped
|
||||
* actions ({@code REPLY}, {@code ASK}), ignored otherwise, may be
|
||||
* {@code null}
|
||||
* <p>Its default classifiers deny every collaborator and every observer, so a caller
|
||||
* enforcing authorization must use the five-argument form instead.
|
||||
*/
|
||||
public static boolean permits(Principal caller, Action action, String targetSession) {
|
||||
return permits(caller, action, targetSession, NO_KNOWN_LEAD_OR_COLLABORATOR, NO_KNOWN_OBSERVER_TARGET);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #permits(Principal, Action, String)}, with a real classifier for a collaborator's
|
||||
* {@code SEND}. An observer's {@code SEND} still fails closed ({@link #NO_KNOWN_OBSERVER_TARGET}) —
|
||||
* a caller enforcing both grants must use the five-argument form.
|
||||
*/
|
||||
public static boolean permits(Principal caller, Action action, String targetSession,
|
||||
Predicate<String> knownLeadOrCollaborator) {
|
||||
return permits(caller, action, targetSession, knownLeadOrCollaborator, NO_KNOWN_OBSERVER_TARGET);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code caller} may perform {@code action} against {@code targetSession}.
|
||||
*
|
||||
* @param targetSession the session id in the request path; only consulted for the
|
||||
* worker-scoped actions ({@code REPLY}, {@code ASK},
|
||||
* {@code INBOX}), for a collaborator's {@code SEND}, and for an
|
||||
* observer's {@code SEND}, ignored otherwise, may be
|
||||
* {@code null}
|
||||
* @param knownLeadOrCollaborator whether a terminal is a configured lead or collaborator —
|
||||
* consulted only for a collaborator's {@code SEND}, to confine
|
||||
* it to another named peer and never a spawned member's
|
||||
* terminal
|
||||
* @param knownObserverTarget whether a terminal is one this daemon would itself resolve as
|
||||
* {@link Role#OBSERVER} — consulted only for an observer's
|
||||
* {@code SEND}, to confine it to another observer pane and never
|
||||
* a lead, a collaborator, or a spawned member
|
||||
*/
|
||||
public static boolean permits(Principal caller, Action action, String targetSession,
|
||||
Predicate<String> knownLeadOrCollaborator,
|
||||
Predicate<String> knownObserverTarget) {
|
||||
if (caller == null || caller.isAnonymous()) {
|
||||
return false; // authenticated as nothing ⇒ authorized for nothing
|
||||
}
|
||||
return switch (action) {
|
||||
// Fleet lifecycle is the primary's alone — spawn, stop, drain. An architect
|
||||
// deliberately does NOT get these (CB-548), so it cannot tear down or stand up workers
|
||||
// even though it coordinates them; and a worker driving any of these would be a worker
|
||||
// escalating into the orchestrator role.
|
||||
// Fleet lifecycle is the primary's alone — spawn, stop, drain. An architect and a
|
||||
// collaborator deliberately do NOT get these, so neither can tear down or stand up
|
||||
// workers even though one of them coordinates them; and a worker driving any of these
|
||||
// would be a worker escalating into the orchestrator role.
|
||||
case SPAWN, STOP, DRAIN, HANDOVER -> caller.isPrimary();
|
||||
|
||||
// Delivering a turn is open to the primary and the architect: an architect delegates
|
||||
// to workers (that is the role's point) but still has no lifecycle rights. A worker is
|
||||
// excluded — sending would be it escalating.
|
||||
case SEND -> caller.isPrimary() || caller.isArchitect();
|
||||
// Delivering a turn to a local session is open to the primary and the architect
|
||||
// unconditionally. A collaborator may reach only a target that is itself a configured
|
||||
// lead or collaborator, never a spawned member's terminal. An observer may reach only
|
||||
// a target that would itself resolve as an observer, never a lead, a collaborator, or
|
||||
// a spawned member. A worker is excluded from every case — sending would be it
|
||||
// escalating into the orchestrator role.
|
||||
case SEND -> caller.isPrimary() || caller.isArchitect()
|
||||
|| (caller.isCollaborator() && knownLeadOrCollaborator.test(targetSession))
|
||||
|| (caller.isObserver() && knownObserverTarget.test(targetSession));
|
||||
|
||||
// Resolving a worker's blocked question is part of delegating to it, open to the same
|
||||
// two roles that may stand up that delegation in the first place. Not a collaborator:
|
||||
// resuming another session's turn is lifecycle-adjacent, not peer messaging.
|
||||
case ANSWER -> caller.isPrimary() || caller.isArchitect();
|
||||
|
||||
// Leaves the daemon over the coordination broker rather than addressing a local
|
||||
// session, open to the same two roles as ANSWER. Not a collaborator: it is a
|
||||
// local-tab peer with no cross-host route.
|
||||
case COORD_SEND -> caller.isPrimary() || caller.isArchitect();
|
||||
|
||||
// The load-bearing rule: a caller acts only as the pane it occupies. CB-532 widened who
|
||||
// that can be — a lead answering another lead is replying for its OWN terminal, which
|
||||
// this already permits — while the rule itself is unchanged, and is what stops anyone
|
||||
// forging a reply for a rendezvous someone else is waiting on. An architect's own pane
|
||||
// passes through the same check, so it can answer a funnel that delegated to it. An
|
||||
// unnamed primary (token/loopback, no pane) owns nothing and is still excluded.
|
||||
case REPLY, ASK -> caller.ownsSession(targetSession);
|
||||
// forging a reply for a rendezvous someone else is waiting on. An architect's or a
|
||||
// collaborator's own pane passes through the same check, so each can answer a funnel
|
||||
// that delegated to it. An unnamed primary (token/loopback, no pane) owns nothing and
|
||||
// is still excluded.
|
||||
case REPLY, ASK, INBOX -> caller.ownsSession(targetSession);
|
||||
|
||||
// Observation is open to every authenticated role: a worker legitimately polls its own
|
||||
// status, and the roster carries no secrets.
|
||||
case READ, METRICS -> caller.isPrimary() || caller.isWorker() || caller.isArchitect();
|
||||
// READ is roster, profile, and identity observation — fleet_list, fleet_profiles, and
|
||||
// fleet_whoami — and carries no secrets: no ticket reply, no pending question, and no
|
||||
// other session's turn state. Those live under TASK_READ. METRICS is the separate
|
||||
// Prometheus scrape. Both are open to every authenticated role, including a
|
||||
// collaborator and the unconfigured-pane floor.
|
||||
case READ, METRICS -> caller.isPrimary() || caller.isWorker() || caller.isArchitect()
|
||||
|| caller.isCollaborator() || caller.isObserver();
|
||||
|
||||
// Ticket polling and session status, open to every role READ is open to except a
|
||||
// collaborator or an observer. MessageService compares a ticket's creator to the
|
||||
// caller on every read as well, so dropping this gate would not expose another
|
||||
// session's reply — it would move the refusal later and widen what a caller that
|
||||
// never orchestrates can probe.
|
||||
case TASK_READ -> caller.isPrimary() || caller.isWorker() || caller.isArchitect();
|
||||
|
||||
// fleetd #421: reading held lead-to-lead mail is the primary's alone. An architect
|
||||
// holds READ today (CB-548), so "not primary" must mean not-architect here too — this
|
||||
// is coordination between leads, not observation of the roster.
|
||||
// is coordination between leads, not observation of the roster. The same reasoning
|
||||
// excludes a collaborator.
|
||||
case COORD_READ -> caller.isPrimary();
|
||||
};
|
||||
}
|
||||
|
||||
@@ -7,6 +7,7 @@ import java.nio.charset.StandardCharsets;
|
||||
import java.security.MessageDigest;
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
@@ -19,6 +20,13 @@ import java.util.function.Supplier;
|
||||
*
|
||||
* <p><strong>Resolution order</strong> — connection identity first, token second, nothing third:
|
||||
* <ol>
|
||||
* <li>A loopback peer PID that maps to a pane this gateway itself spawned ⇒ that member's own
|
||||
* role: {@link Role#WORKER} for a dev, hunter, or reviewer; {@link Role#ARCHITECT} for an
|
||||
* architect, but only while the live slot role still confirms it (fleetd #424 — a slot
|
||||
* revoked from config demotes an already-bound session on its very next request, so the
|
||||
* roster's own role is never granted on its word alone). No tab map is consulted — a live
|
||||
* spawned member's identity comes from the registry that spawned it, never from a label a
|
||||
* pane could also carry.</li>
|
||||
* <li>A loopback peer PID that maps to a pane named by {@code leaders:}, by the legacy
|
||||
* {@code primary.terminal} pin, or by an operator-labelled lead tab (CB-307, CB-530, CB-531)
|
||||
* ⇒ {@link Role#PRIMARY}, carrying that lead's
|
||||
@@ -28,9 +36,11 @@ import java.util.function.Supplier;
|
||||
* so two leads can work as peers rather than one being demoted.</li>
|
||||
* <li>A loopback peer PID that maps to a pane bound to a CB-548 architect slot ⇒
|
||||
* {@link Role#ARCHITECT}, carrying the slot name. Just unforgeable as a worker's, and
|
||||
* resolved from the <em>live</em> terminal→slot binding (never a request argument), before
|
||||
* the generic worker fallback.</li>
|
||||
* <li>A loopback peer PID that maps to any other herdr pane ⇒ {@link Role#WORKER}. This is
|
||||
* resolved from the <em>live</em> terminal→slot binding (never a request argument). This is
|
||||
* the case the previous step does not catch: a binding with no live spawned-member session.</li>
|
||||
* <li>A loopback peer PID that maps to an operator-labelled collaborator tab ⇒
|
||||
* {@link Role#COLLABORATOR}, carrying that collaborator's name.</li>
|
||||
* <li>A loopback peer PID that maps to any other herdr pane ⇒ {@link Role#OBSERVER}. This is
|
||||
* unforgeable (the OS reports the PID, herdr owns the PID→pane map) and is honoured
|
||||
* regardless of auth mode, so enabling auth never breaks the fleet.</li>
|
||||
* <li>Otherwise, under {@code token} mode, a valid bearer token ⇒ {@link Role#PRIMARY}.</li>
|
||||
@@ -64,6 +74,20 @@ public final class CallerResolver {
|
||||
private final Supplier<Map<String, String>> architectTerminals;
|
||||
private final Function<String, MemberRole> memberSlotRoles;
|
||||
private final Function<String, String> memberSlotNames;
|
||||
/**
|
||||
* terminal_id → the role of the live spawned member occupying it, or {@code null} for a
|
||||
* terminal no spawned member occupies. Consulted first, ahead of every tab map: a live
|
||||
* spawned member's identity is its own, whatever a tab map says about the same terminal.
|
||||
* A function rather than the roster itself, so a resolve on the hot path never scans a list —
|
||||
* the lookup strategy is the caller's to choose.
|
||||
*/
|
||||
private final Function<String, MemberRole> spawnedMemberRole;
|
||||
/**
|
||||
* terminal_id → collaborator name; empty when none are configured. A supplier for the same
|
||||
* reason as {@link #leadTerminals}: a collaborator tab recognised after construction (the tab
|
||||
* scan discovering a newly-labelled tab) takes effect without a restart.
|
||||
*/
|
||||
private final Supplier<Map<String, String>> collaboratorTerminals;
|
||||
|
||||
/** Loopback-trust resolver: no token required, historical behaviour. Test-only. */
|
||||
CallerResolver(ConnectionIdentity identity) {
|
||||
@@ -124,17 +148,42 @@ public final class CallerResolver {
|
||||
/**
|
||||
* Live registry form that can confirm a bound slot is an architect slot.
|
||||
*
|
||||
* <p>This is the only public construction path. It keeps terminal bindings and slot roles in
|
||||
* the same {@link MemberRegistry}, so a configured architect can resolve as an architect.
|
||||
* <p>It keeps terminal bindings and slot roles in the same {@link MemberRegistry}, so a
|
||||
* configured architect can resolve as an architect. No spawned-member roster or collaborator
|
||||
* registry is consulted — equivalent to {@link #withLeadsAndMembers(ConnectionIdentity,
|
||||
* boolean, String, Supplier, MemberRegistry, Function, Supplier)} with both absent. Kept for
|
||||
* every caller that has neither to offer, so adding them did not churn every construction site.
|
||||
*/
|
||||
public static CallerResolver withLeadsAndMembers(ConnectionIdentity identity,
|
||||
boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
MemberRegistry members) {
|
||||
return withLeadsAndMembers(identity, tokenMode, token, leadTerminals, members, null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Live registry form that also resolves a live spawned member to its own role, and a
|
||||
* configured collaborator tab to {@link Role#COLLABORATOR}.
|
||||
*
|
||||
* <p>This is the only public construction path that exercises the full resolution order.
|
||||
*
|
||||
* @param spawnedMemberRole terminal_id → the role of the live spawned member occupying
|
||||
* it, or {@code null} for a terminal no spawned member occupies.
|
||||
* {@code null} here means no roster is consulted at all (every
|
||||
* terminal falls through to the tab maps), not that none matches.
|
||||
* @param collaboratorTerminals terminal_id → collaborator name, live like {@code leadTerminals}
|
||||
*/
|
||||
public static CallerResolver withLeadsAndMembers(ConnectionIdentity identity,
|
||||
boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
MemberRegistry members,
|
||||
Function<String, MemberRole> spawnedMemberRole,
|
||||
Supplier<Map<String, String>> collaboratorTerminals) {
|
||||
return new CallerResolver(identity, tokenMode, token, leadTerminals,
|
||||
members == null ? null : members::snapshot,
|
||||
members == null ? null : members::roleForSlot,
|
||||
members == null ? null : members::nameForSlot);
|
||||
members == null ? null : members::nameForSlot,
|
||||
spawnedMemberRole, collaboratorTerminals);
|
||||
}
|
||||
|
||||
private static Supplier<Map<String, String>> fixed(Map<String, String> leadTerminals) {
|
||||
@@ -160,6 +209,17 @@ public final class CallerResolver {
|
||||
Supplier<Map<String, String>> architectTerminals,
|
||||
Function<String, MemberRole> memberSlotRoles,
|
||||
Function<String, String> memberSlotNames) {
|
||||
this(identity, tokenMode, token, leadTerminals, architectTerminals, memberSlotRoles,
|
||||
memberSlotNames, null, null);
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals,
|
||||
Function<String, MemberRole> memberSlotRoles,
|
||||
Function<String, String> memberSlotNames,
|
||||
Function<String, MemberRole> spawnedMemberRole,
|
||||
Supplier<Map<String, String>> collaboratorTerminals) {
|
||||
if (tokenMode && (token == null || token.isBlank())) {
|
||||
throw new IllegalArgumentException(
|
||||
"auth.mode=token requires a non-empty token; check that the env var named by "
|
||||
@@ -172,6 +232,8 @@ public final class CallerResolver {
|
||||
this.architectTerminals = architectTerminals == null ? Map::of : architectTerminals;
|
||||
this.memberSlotRoles = memberSlotRoles == null ? _ -> null : memberSlotRoles;
|
||||
this.memberSlotNames = memberSlotNames == null ? Function.identity() : memberSlotNames;
|
||||
this.spawnedMemberRole = spawnedMemberRole == null ? _ -> null : spawnedMemberRole;
|
||||
this.collaboratorTerminals = collaboratorTerminals == null ? Map::of : collaboratorTerminals;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -188,14 +250,49 @@ public final class CallerResolver {
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-recognised architect slots, {@code terminal_id → slot name} (CB-548).
|
||||
* The currently-recognised collaborator tabs, {@code terminal_id → name}.
|
||||
*
|
||||
* <p>Read from the same supplier {@link #resolve} consults, so a slot that is <em>listed</em>
|
||||
* here but would not <em>resolve</em> (or the reverse) cannot drift apart. Live for the same
|
||||
* reason as {@link #leads()}.
|
||||
* <p>Read from the same supplier {@link #resolve} consults, for the reason given in
|
||||
* {@link #leads()}. Live for the same reason as {@link #leads()}.
|
||||
*/
|
||||
public Map<String, String> members() {
|
||||
return architectTerminals.get();
|
||||
public Map<String, String> collaborators() {
|
||||
return collaboratorTerminals.get();
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code target} names a terminal this resolver would resolve as a lead or a
|
||||
* collaborator — the classifier a collaborator's {@code SEND} is checked against, read from the
|
||||
* exact maps {@link #resolve} consults so a target that would resolve as a lead or collaborator
|
||||
* is never the one a collaborator is refused to reach, or the reverse.
|
||||
*/
|
||||
public Predicate<String> knownLeadOrCollaborator() {
|
||||
return target -> leadTerminals.get().containsKey(target)
|
||||
|| collaboratorTerminals.get().containsKey(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code target} names a terminal this resolver would itself resolve as {@link
|
||||
* Role#OBSERVER} — the classifier an observer's {@code SEND} is checked against, read from the
|
||||
* same maps and functions {@link #resolve} consults so a target this accepts is exactly one
|
||||
* {@code resolve} would hand back {@link Role#OBSERVER} for, and the reverse.
|
||||
*/
|
||||
public Predicate<String> sendableObserverTarget() {
|
||||
return target -> target != null
|
||||
&& spawnedMemberRole.apply(target) == null
|
||||
&& !leadTerminals.get().containsKey(target)
|
||||
&& !boundToArchitectSlot(target)
|
||||
&& !collaboratorTerminals.get().containsKey(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code terminal} is bound to a configured slot the live roster still confirms as an
|
||||
* architect — the one classifier {@link #sendableObserverTarget()} and {@code FleetMcp}'s
|
||||
* {@code panes} row both read, so a pane's reported role and its {@code SEND} reachability can
|
||||
* never drift apart.
|
||||
*/
|
||||
public boolean boundToArchitectSlot(String terminal) {
|
||||
String slot = architectTerminals.get().get(terminal);
|
||||
return slot != null && memberSlotRoles.apply(slot) == MemberRole.ARCHITECT;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -208,6 +305,25 @@ public final class CallerResolver {
|
||||
public Principal resolve(String remoteAddr, int remotePort, String authorizationHeader) {
|
||||
ConnectionIdentity.Caller c = identity.resolve(remoteAddr, remotePort);
|
||||
if (c.terminal() != null) {
|
||||
MemberRole spawnedRole = spawnedMemberRole.apply(c.terminal());
|
||||
if (spawnedRole != null) {
|
||||
// A live spawned member occupies this pane. Its identity is its own, whatever a tab
|
||||
// map says about the same terminal — checked before every tab map, consulting none
|
||||
// of them, so a tab label can never override a roster entry for the same terminal.
|
||||
if (spawnedRole == MemberRole.ARCHITECT) {
|
||||
// The roster only answers THAT this pane is a live spawned member; config still
|
||||
// decides WHAT that member's slot grants (fleetd #424). A slot revoked after the
|
||||
// bind must still demote this session on its very next request, so the roster's
|
||||
// own ARCHITECT role is confirmed against the live slot role, exactly as the
|
||||
// architect-slot step below confirms a binding with no live member session.
|
||||
String slot = architectTerminals.get().get(c.terminal());
|
||||
if (slot != null && memberSlotRoles.apply(slot) == MemberRole.ARCHITECT) {
|
||||
return Principal.architect(memberSlotNames.apply(slot), c.terminal(), c.pid());
|
||||
}
|
||||
return Principal.worker(c.terminal(), c.pid());
|
||||
}
|
||||
return Principal.worker(c.terminal(), c.pid());
|
||||
}
|
||||
String lead = leadTerminals.get().get(c.terminal());
|
||||
if (lead != null) {
|
||||
// The config names this pane as a lead's own. The pane mapping is exactly as
|
||||
@@ -221,11 +337,18 @@ public final class CallerResolver {
|
||||
// The config/live binding names this pane as an architect slot's own. Same
|
||||
// unforgeable pane mapping; the live binding, never a request argument, decides.
|
||||
// Check the slot role too: this defence in depth prevents a bad lifecycle bind from
|
||||
// escalating a dev, hunter or reviewer into an architect. Checked before
|
||||
// the worker fallback.
|
||||
// escalating a dev, hunter or reviewer into an architect. This is the case the
|
||||
// spawned-member step above does not catch: a binding with no live member session.
|
||||
return Principal.architect(memberSlotNames.apply(slot), c.terminal(), c.pid());
|
||||
}
|
||||
return Principal.worker(c.terminal(), c.pid()); // unforgeable; never token-gated
|
||||
String collaborator = collaboratorTerminals.get().get(c.terminal());
|
||||
if (collaborator != null) {
|
||||
// An operator-labelled collaborator tab, confirmed live by the same scan that
|
||||
// confirms a lead tab. Checked last among the tab maps so a pane also matching one
|
||||
// of the above keeps that stronger role.
|
||||
return Principal.collaborator(collaborator, c.terminal(), c.pid());
|
||||
}
|
||||
return Principal.observer(c.terminal(), c.pid()); // unforgeable; never token-gated
|
||||
}
|
||||
|
||||
if (tokenMode) {
|
||||
|
||||
@@ -11,7 +11,9 @@ package dev.ltms.fleet.auth;
|
||||
* @param pid the connecting process id, or {@code -1} when not resolvable (audit context)
|
||||
* @param name for a lead resolved from the CB-530 {@code leaders:} registry, which lead it is;
|
||||
* for an architect resolved from the CB-548 {@code architects:} registry, which
|
||||
* slot it occupies; {@code null} for every other caller, including an unnamed primary
|
||||
* slot it occupies; for a collaborator resolved from the {@code collaborators:}
|
||||
* registry, which collaborator it is; {@code null} for every other caller,
|
||||
* including an unnamed primary
|
||||
*/
|
||||
public record Principal(Role role, String terminal, long pid, String name) {
|
||||
|
||||
@@ -73,6 +75,26 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
||||
return new Principal(Role.ARCHITECT, terminal, pid, slotName);
|
||||
}
|
||||
|
||||
/**
|
||||
* A collaborator: a human-opened tab recognised by its exact label in the
|
||||
* {@code collaborators:} registry.
|
||||
*
|
||||
* <p>Carries {@link Role#COLLABORATOR}. {@code name} is reporting only — it lets
|
||||
* {@code fleet_whoami} say which collaborator is asking. Identity is the {@code terminal}:
|
||||
* like a worker's it comes from the connection, so {@code ownsSession} works exactly as it
|
||||
* does for a worker — a collaborator acts as its own pane and no other.
|
||||
*/
|
||||
public static Principal collaborator(String name, String terminal, long pid) {
|
||||
return new Principal(Role.COLLABORATOR, terminal, pid, name);
|
||||
}
|
||||
|
||||
/**
|
||||
* The unconfigured-pane floor: a loopback caller whose pane matched no other role.
|
||||
*/
|
||||
public static Principal observer(String terminal, long pid) {
|
||||
return new Principal(Role.OBSERVER, terminal, pid);
|
||||
}
|
||||
|
||||
public boolean isPrimary() {
|
||||
return role == Role.PRIMARY;
|
||||
}
|
||||
@@ -81,15 +103,23 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
||||
return role == Role.ARCHITECT;
|
||||
}
|
||||
|
||||
public boolean isCollaborator() {
|
||||
return role == Role.COLLABORATOR;
|
||||
}
|
||||
|
||||
public boolean isWorker() {
|
||||
return role == Role.WORKER;
|
||||
}
|
||||
|
||||
public boolean isObserver() {
|
||||
return role == Role.OBSERVER;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether this caller is a spawned member with its own pane.
|
||||
*
|
||||
* <p>Both workers and architects are spawned members. A lead is excluded because recording it
|
||||
* as present would count it as an available member in the roster.
|
||||
* <p>Both workers and architects are spawned members. A lead is not: it is a peer the
|
||||
* operator started and named, never a pane this daemon spawned.
|
||||
*/
|
||||
public boolean isSpawnedMember() {
|
||||
return role == Role.WORKER || role == Role.ARCHITECT;
|
||||
@@ -115,11 +145,33 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
||||
return terminal != null && terminal.equals(sessionId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Stable identity used to own tickets and open turns. The unnamed primary has no owner key —
|
||||
* {@code null} — and that is matched against a ticket's recorded owner the same way any other
|
||||
* key is: it owns a ticket another unnamed primary created, and nothing else.
|
||||
*/
|
||||
public String ownerKey() {
|
||||
return switch (role) {
|
||||
case PRIMARY -> name == null ? null : prefixed("leader", name);
|
||||
case WORKER -> prefixed("worker", terminal);
|
||||
case ARCHITECT -> prefixed("architect", terminal);
|
||||
case COLLABORATOR -> prefixed("collaborator", name);
|
||||
case OBSERVER -> prefixed("observer", terminal);
|
||||
case ANONYMOUS -> "anonymous";
|
||||
};
|
||||
}
|
||||
|
||||
private static String prefixed(String role, String identity) {
|
||||
return role + ":" + identity;
|
||||
}
|
||||
|
||||
/** Short, non-sensitive description for audit lines and error details. */
|
||||
public String describe() {
|
||||
return switch (role) {
|
||||
case WORKER -> "worker:" + terminal;
|
||||
case ARCHITECT -> "architect:" + name;
|
||||
case COLLABORATOR -> "collaborator:" + name;
|
||||
case OBSERVER -> "observer:" + terminal;
|
||||
case PRIMARY -> name == null ? "primary" : "leader:" + name;
|
||||
case ANONYMOUS -> "anonymous";
|
||||
};
|
||||
|
||||
@@ -12,9 +12,10 @@ package dev.ltms.fleet.auth;
|
||||
public enum Role {
|
||||
|
||||
/**
|
||||
* The orchestrating session. Established either by being a loopback caller that is not a
|
||||
* worker pane (under {@code loopback-trust}) or by presenting a valid bearer token (under
|
||||
* {@code token} mode).
|
||||
* The orchestrating session. Established either by being a loopback caller that resolves to
|
||||
* no herdr pane at all (under {@code loopback-trust}) or by presenting a valid bearer token
|
||||
* (under {@code token} mode). A loopback caller that does own a pane, but matches none of the
|
||||
* roles below, resolves to {@link #OBSERVER} instead.
|
||||
*/
|
||||
PRIMARY,
|
||||
|
||||
@@ -34,6 +35,29 @@ public enum Role {
|
||||
*/
|
||||
ARCHITECT,
|
||||
|
||||
/**
|
||||
* A config-declared, human-opened tab recognised by its exact label (the {@code
|
||||
* fleet.collaborators.<name>.tab} registry). Never spawned — identity comes from the
|
||||
* connection, never a request argument, exactly like {@link #WORKER} and {@link #ARCHITECT}.
|
||||
* May {@code SEND} only to a configured lead or collaborator, {@code REPLY}/{@code ASK} only
|
||||
* as its own pane, and {@code READ}/{@code METRICS}; may not {@code SPAWN}/{@code STOP}/
|
||||
* {@code DRAIN}/{@code HANDOVER}, poll a ticket ({@code TASK_READ}), or reach the
|
||||
* coordination broker ({@code COORD_SEND}/{@code COORD_READ}).
|
||||
*/
|
||||
COLLABORATOR,
|
||||
|
||||
/**
|
||||
* A loopback pane that resolved to none of the roles above: not a live spawned member, not a
|
||||
* configured lead, not a bound architect slot, not a configured collaborator tab. Unforgeable
|
||||
* like a worker's — derived from the connection's pane, never from a request argument, and
|
||||
* honoured regardless of auth mode. May {@code READ} and {@code METRICS}, {@code REPLY}/
|
||||
* {@code ASK} only as its own pane, and {@code SEND} only to a target that would itself
|
||||
* resolve as {@code OBSERVER}; may not {@code SPAWN}/{@code STOP}/{@code DRAIN}/
|
||||
* {@code HANDOVER}, poll a ticket ({@code TASK_READ}), or reach the coordination broker
|
||||
* ({@code COORD_SEND}/{@code COORD_READ}).
|
||||
*/
|
||||
OBSERVER,
|
||||
|
||||
/** Authenticated as nothing. Authorized for nothing but {@code /healthz}. */
|
||||
ANONYMOUS
|
||||
}
|
||||
|
||||
@@ -67,7 +67,7 @@ import java.util.function.Supplier;
|
||||
* {@code models:} above: {@code dev.ltms.fleet.lead.LeadRollover} holds a
|
||||
* {@code Supplier<FleetConfig.LeadRollover>} (the same {@code () -> config.get().x()} shape)
|
||||
* and reads {@code handoverPath}/{@code requireOperatorConfirm}/{@code maxDocAgeSeconds}/
|
||||
* {@code turnSettleSeconds}/{@code clearSettleSeconds}/{@code bootstrapText} fresh on every
|
||||
* {@code turnSettleSeconds}/{@code relaunchReadySeconds}/{@code bootstrapText} fresh on every
|
||||
* {@code open()}/{@code confirm()} call (and on the deferred post-{@code confirm()}
|
||||
* continuation fleetd #480's correction added — see {@code LeadRollover}'s class doc) rather
|
||||
* than capturing them into fields at construction — unlike its closest
|
||||
@@ -615,7 +615,7 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
// may bind to, AND what a slot already bound still grants) through its own instance of that
|
||||
// same supplier shape — see MemberRegistry.live and its class doc for the binding rule:
|
||||
// removing a slot revokes ARCHITECT on the bound pane's very next request, and only the slot
|
||||
// OCCUPANCY survives, so the demoted session keeps its slot key until it unbinds. Only
|
||||
// OCCUPANCY survives, so the demoted session keeps its slot key until it unbinds.
|
||||
// fleet.leaders is frozen (Fleetd.java:281 reads cfg.fleet().leaders() off the startup
|
||||
// snapshot to build both the LeadTabScanner's tab-label-to-name map, wired into
|
||||
// CallerResolver.withLeadsAndMembers at Fleetd.java:620/624, and — when herdr answered —
|
||||
@@ -635,6 +635,11 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
+ "live through that same supplier for placement AND through a separate supplier "
|
||||
+ "on MemberRegistry for spawn-time identity — both already applied");
|
||||
}
|
||||
if (!Objects.equals(collaboratorsOf(old), collaboratorsOf(fresh))) {
|
||||
changed.add("fleet: fleet.collaborators (each collaborator's tab) is read once at "
|
||||
+ "startup to build the LeadTabScanner's identity map, which is not rebuilt on "
|
||||
+ "reload, so a collaborator added, removed, or given a new tab: needs a restart");
|
||||
}
|
||||
// Kept in step with SPLIT_KEYS the same way changedColdKeys is kept in step with COLD_KEYS —
|
||||
// every message here must be traceable to one of the split keys the class doc documents.
|
||||
// NOTE what this does NOT prove, per the javadoc above: it does not catch a SPLIT_KEYS
|
||||
@@ -650,6 +655,11 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
return cfg.fleet() == null ? Map.of() : cfg.fleet().leaders();
|
||||
}
|
||||
|
||||
/** {@code cfg.fleet().collaborators()}, defensively, in case a caller hands in a non-defaulted config. */
|
||||
private static Map<String, FleetConfig.Collaborator> collaboratorsOf(FleetConfig cfg) {
|
||||
return cfg.fleet() == null ? Map.of() : cfg.fleet().collaborators();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@link FleetConfig.Profile} record components deliberately left out of
|
||||
* {@link #sameLaunchSettings} because they are read <em>live</em>, not baked in at spawn — see
|
||||
|
||||
@@ -28,6 +28,7 @@ import java.util.Comparator;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
@@ -798,6 +799,25 @@ public record FleetConfig(
|
||||
return isSubscription() ? SUBSCRIPTION_CREDENTIAL_ID : profile;
|
||||
}
|
||||
|
||||
/**
|
||||
* The auto-compaction window a launched Claude Code session actually runs on: {@code env:
|
||||
* CLAUDE_CODE_AUTO_COMPACT_WINDOW} when it parses as an integer, since that environment
|
||||
* variable wins over the {@code --autocompact} flag {@link #autoCompactWindow} produces (see
|
||||
* {@code ClaudeCodeArguments}); {@link #autoCompactWindow} otherwise. {@code null} when
|
||||
* neither resolves to a usable number.
|
||||
*/
|
||||
public Integer effectiveAutoCompactWindow() {
|
||||
String envValue = env.get(CLAUDE_CODE_AUTO_COMPACT_WINDOW_ENV);
|
||||
if (envValue == null) {
|
||||
return autoCompactWindow;
|
||||
}
|
||||
try {
|
||||
return Integer.valueOf(envValue.trim());
|
||||
} catch (NumberFormatException e) {
|
||||
return autoCompactWindow;
|
||||
}
|
||||
}
|
||||
|
||||
/** True when this profile's workers are granted a forge token to open their own PR (CB-302). */
|
||||
public boolean hasGitToken() {
|
||||
return gitTokenEnv != null && !gitTokenEnv.isBlank();
|
||||
@@ -1073,8 +1093,8 @@ public record FleetConfig(
|
||||
}
|
||||
|
||||
/**
|
||||
* One entry of the CB-530 {@code leaders:} registry — a pane that orchestrates rather than one
|
||||
* that is orchestrated.
|
||||
* One entry of the {@code leaders:} registry — a pane that orchestrates rather than one that is
|
||||
* orchestrated.
|
||||
*
|
||||
* <p>Why a registry and not a second {@code primary:}: {@code primary.terminal} is singular by
|
||||
* construction, so a session in any other pane resolves as a worker. That is correct while one
|
||||
@@ -1084,49 +1104,49 @@ public record FleetConfig(
|
||||
* <p>{@code kind} and {@code model} are descriptive only: they document what runs in the pane
|
||||
* and are reported back by {@code fleet_whoami}.
|
||||
*
|
||||
* <p><b>A lead is now also creatable (CB-557).</b> Before, nothing spawned one — a lead
|
||||
* pre-existed, which is why it had to be recognised by configuration rather than created. With
|
||||
* {@code profile} and {@code instances} the daemon may stand one up when none is live, so the
|
||||
* pane no longer has to exist before the daemon does. Recognition still comes first: a lead
|
||||
* already running in its configured {@code tab} is adopted, and only the shortfall is launched.
|
||||
* <p>A lead with {@code profile} and {@code instances} set may be launched by the daemon when
|
||||
* none is live; recognition always comes first, so only the shortfall is launched.
|
||||
*
|
||||
* <p><b>{@code tab} replaced {@code terminal} (CB-579).</b> A herdr {@code terminal_id} changes
|
||||
* every time the lead's session restarts, so pinning one cost a config edit and a daemon restart
|
||||
* per restart. A tab is stable: a human opens it once, it holds exactly one pane, and its label
|
||||
* survives restarts of the agent inside it — so identity is now the tab label alone.
|
||||
* <p>Every lead's tab is labelled {@link #LEAD_TAB_LABEL}, a fixed constant — not a per-entry
|
||||
* config value. {@code workspace} is therefore what tells one lead from another: two leaders
|
||||
* sharing one space would both resolve to the one tab named {@code lead} there, so only one
|
||||
* could ever be found. {@code tab} is a deprecated legacy label, still matched within this
|
||||
* lead's own space alongside the constant.
|
||||
*
|
||||
* @param profile the {@code profiles:} entry to launch this lead on when one must
|
||||
* be created; {@code null} ⇒ recognise-only, never create.
|
||||
* <p>fleetd #176: also the field {@code Fleetd.leadSeatLookup} reads
|
||||
* to learn which account this lead's own live session shares — set it
|
||||
* <p>Also the field {@code Fleetd.leadSeatLookup} reads to learn
|
||||
* which account this lead's own live session shares — set it
|
||||
* (safely, even on an already-running recognise-only lead: naming a
|
||||
* profile here never starts anything beyond {@code instances}) so a
|
||||
* {@code subscription: true} worker profile sharing its
|
||||
* {@code effectiveCredentialId()} has this lead's seat subtracted from
|
||||
* {@code fleet_list}'s {@code free}. {@code null} here also means this
|
||||
* lead's seat cannot be derived and is not counted.
|
||||
* @param tab the exact tab label hosting this lead, matched case-insensitively;
|
||||
* the only field identity depends on. Required — a lead with no
|
||||
* {@code tab} can never be discovered, launched or not
|
||||
* @param tab deprecated legacy tab label, matched case-insensitively within
|
||||
* this lead's own space alongside {@link #LEAD_TAB_LABEL}. Optional —
|
||||
* {@code null}/blank means only the constant is accepted
|
||||
* @param instances how many of this lead should be live (default 1). The daemon
|
||||
* launches only the shortfall, so a restart adopts rather than doubles
|
||||
* @param tabPrefix no longer used to find a lead's tab — {@code tab} is matched
|
||||
* exactly. Its only remaining job is the startup collision guard
|
||||
* ({@link #validateLeadTabPrefixes()}), which still uses it to refuse
|
||||
* a worker {@code tabLabel} template that could be misread as a lead.
|
||||
* Default {@code "lead:"}
|
||||
* @param tabPrefix lead-tab naming convention checked against member labels. Lead
|
||||
* identity uses {@link #acceptedLabels()}. Default {@code "lead:"}
|
||||
* @param scanIntervalSeconds how long a tab scan is cached before herdr is asked again; also the
|
||||
* worst case before a newly-labelled tab is recognised. Default 10
|
||||
* @param kind which agent runs there ({@code claude}, {@code opencode}, …)
|
||||
* @param model the model or selector it runs, for operators reading the roster
|
||||
* @param workspace the space this lead's tab lives in — the uniqueness boundary
|
||||
* identity now depends on. Default {@link #DEFAULT_WORKSPACE}
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record Leader(String profile, String tab, Integer instances, String tabPrefix,
|
||||
Integer scanIntervalSeconds, String kind, String model,
|
||||
String workspace, String cwd) {
|
||||
|
||||
/** The tab label every lead is found by, and an auto-launched instance is created with. */
|
||||
public static final String LEAD_TAB_LABEL = "lead";
|
||||
|
||||
/**
|
||||
* Where an auto-launched lead's tab is created (CB-558). It defaults to the SAME shared
|
||||
* Where an auto-launched lead's tab is created. It defaults to the SAME shared
|
||||
* {@code "fleet"} space the members use, so the operator sees one "session" with many tabs.
|
||||
* The scanner no longer excludes member spaces — it tells a lead from a member by the exact
|
||||
* tab label, so a lead sharing the members' space is still discovered (see LeadLauncher).
|
||||
@@ -1154,9 +1174,42 @@ public record FleetConfig(
|
||||
return profile != null && !profile.isBlank() && instances > 0;
|
||||
}
|
||||
|
||||
/** The tab label an auto-launched instance of this lead gets — its configured {@code tab}. */
|
||||
/** The tab label an auto-launched instance of this lead gets. */
|
||||
public String tabLabel() {
|
||||
return tab;
|
||||
return LEAD_TAB_LABEL;
|
||||
}
|
||||
|
||||
/**
|
||||
* The normalised labels (stripped, lower-cased) a tab in this lead's own space may carry to
|
||||
* be recognised as this lead: {@link #LEAD_TAB_LABEL} first, plus the deprecated {@code tab}
|
||||
* when configured and different. Every matcher in this class's callers reads this method —
|
||||
* none re-derives the set.
|
||||
*/
|
||||
public List<String> acceptedLabels() {
|
||||
String normalizedTab = (tab == null) ? null : tab.toLowerCase(Locale.ROOT);
|
||||
if (normalizedTab == null || normalizedTab.equals(LEAD_TAB_LABEL)) {
|
||||
return List.of(LEAD_TAB_LABEL);
|
||||
}
|
||||
return List.of(LEAD_TAB_LABEL, normalizedTab);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A tab fleetd recognises as a collaborator, keyed by name (fleetd #669).
|
||||
*
|
||||
* <p>Recognise-only: there is no {@code profile}, no {@code instances} and no {@code kind}.
|
||||
* Nothing here ever launches a pane.
|
||||
*
|
||||
* <p>{@code tabPrefix} is absent. Identity is matched on the exact {@code tab} alone.
|
||||
*
|
||||
* @param tab the exact tab label hosting this collaborator, matched case-insensitively; the
|
||||
* only field identity depends on. Required — an entry with no {@code tab} can
|
||||
* never be discovered.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record Collaborator(String tab) {
|
||||
public Collaborator {
|
||||
tab = (tab == null || tab.isBlank()) ? null : tab.strip();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1197,15 +1250,18 @@ public record FleetConfig(
|
||||
* is exactly compatible with that. The pool is also what replaced {@code defaultProfile:} — an
|
||||
* unqualified spawn names a role, and the role's pool supplies the candidates.
|
||||
*
|
||||
* @param leaders panes that orchestrate rather than are orchestrated, keyed by lead name
|
||||
* @param architects profiles the {@code architect} role may run on
|
||||
* @param developers profiles the {@code dev} role may run on
|
||||
* @param hunters profiles the {@code hunter} role may run on
|
||||
* @param reviewers profiles the {@code reviewer} role may run on
|
||||
* @param charters optional launch-charter text keyed by singular role wire name
|
||||
* @param tabLabel template for a member tab's label; {@code {role}}, {@code {profile}},
|
||||
* {@code {model}} and {@code {n}} (a per role+profile counter) are
|
||||
* substituted. Default {@link #DEFAULT_TAB_LABEL}
|
||||
* @param leaders panes that orchestrate rather than are orchestrated, keyed by lead name
|
||||
* @param architects profiles the {@code architect} role may run on
|
||||
* @param developers profiles the {@code dev} role may run on
|
||||
* @param hunters profiles the {@code hunter} role may run on
|
||||
* @param reviewers profiles the {@code reviewer} role may run on
|
||||
* @param charters optional launch-charter text keyed by singular role wire name
|
||||
* @param tabLabel template for a member tab's label; {@code {role}}, {@code {profile}},
|
||||
* {@code {model}} and {@code {n}} (a per role+profile counter) are
|
||||
* substituted. Default {@link #DEFAULT_TAB_LABEL}
|
||||
* @param collaborators tabs fleetd recognises as collaborators (fleetd #669), keyed by name.
|
||||
* Recognise-only, exactly like a {@code profile}-less {@link Leader}:
|
||||
* nothing here is ever auto-launched.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record Fleet(Map<String, Leader> leaders,
|
||||
@@ -1214,13 +1270,11 @@ public record FleetConfig(
|
||||
Map<String, Slot> hunters,
|
||||
Map<String, Slot> reviewers,
|
||||
Map<String, String> charters,
|
||||
String tabLabel) {
|
||||
String tabLabel,
|
||||
Map<String, Collaborator> collaborators) {
|
||||
|
||||
/**
|
||||
* Role first, so the tab bar reads as the fleet and so the label shares a namespace with a
|
||||
* lead's {@code tabPrefix}. Because {@code {role}} comes from a closed enum, a generated
|
||||
* member label can never begin with {@code "lead:"} — the clash that
|
||||
* {@link #validateLeadTabPrefixes()} used to have to check for is unrepresentable here.
|
||||
* Role first, so the tab bar identifies the member's fleet role.
|
||||
*/
|
||||
public static final String DEFAULT_TAB_LABEL = "{role}: {profile} #{n}";
|
||||
|
||||
@@ -1232,26 +1286,30 @@ public record FleetConfig(
|
||||
reviewers = unmodifiableOrEmpty(reviewers);
|
||||
charters = unmodifiableOrEmpty(charters);
|
||||
tabLabel = (tabLabel == null || tabLabel.isBlank()) ? DEFAULT_TAB_LABEL : tabLabel;
|
||||
collaborators = unmodifiableOrEmpty(collaborators);
|
||||
}
|
||||
|
||||
/**
|
||||
* A fleet with no configured launch charters — the shape every deployment had before
|
||||
* CB-566, and what most tests want.
|
||||
* A fleet with no configured launch charters and no collaborators — the shape every
|
||||
* deployment had before CB-566, and what most tests want.
|
||||
*
|
||||
* <p>Kept deliberately, even though an overload that drops a new field is normally the
|
||||
* shape to avoid. It is safe here because nothing <em>reads</em> a charter through a
|
||||
* constructor: the launcher reads {@code fleet.charters()} from the live config. Jackson
|
||||
* binds the canonical constructor, so this one cannot swallow an operator's YAML.
|
||||
* {@code collaborators} is dropped the same way and for the same reason: no caller of
|
||||
* this overload has ever needed to set it, so it defaults to empty here exactly as the
|
||||
* canonical constructor would default an absent YAML key.
|
||||
*/
|
||||
public Fleet(Map<String, Leader> leaders, Map<String, Slot> architects,
|
||||
Map<String, Slot> developers, Map<String, Slot> reviewers,
|
||||
Map<String, String> charters, String tabLabel) {
|
||||
this(leaders, architects, developers, null, reviewers, charters, tabLabel);
|
||||
this(leaders, architects, developers, null, reviewers, charters, tabLabel, null);
|
||||
}
|
||||
|
||||
public Fleet(Map<String, Leader> leaders, Map<String, Slot> architects,
|
||||
Map<String, Slot> developers, Map<String, Slot> reviewers, String tabLabel) {
|
||||
this(leaders, architects, developers, null, reviewers, null, tabLabel);
|
||||
this(leaders, architects, developers, null, reviewers, null, tabLabel, null);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1376,15 +1434,16 @@ public record FleetConfig(
|
||||
* before anything exists to call — the same fact already true of adding a brand-new
|
||||
* {@code profiles:} entry.
|
||||
*
|
||||
* <p><strong>{@code turnSettleSeconds} (fleetd #480 correction):</strong> {@code confirm()} is
|
||||
* called FROM the calling lead's own turn, so its pane is still {@code WORKING} the instant
|
||||
* {@code confirm()} validates every gate and schedules the roll. {@code
|
||||
* dev.ltms.fleet.lead.LeadRollover}'s deferred continuation waits up to this many seconds for
|
||||
* that SAME pane to report an injectable state again — i.e. for the calling turn to actually
|
||||
* end — before it sends {@code /clear} at all. If that wait times out, no {@code /clear} is
|
||||
* ever sent: a lead that never goes idle is still doing real work, and clearing it would
|
||||
* destroy live context. This is a separate wait from {@code clearSettleSeconds} below, which
|
||||
* bounds the SECOND wait, for the pane to re-settle AFTER {@code /clear} has already gone out.
|
||||
* <p><strong>{@code turnSettleSeconds}:</strong> {@code confirm()} is called FROM the calling
|
||||
* lead's own turn, so its pane is still {@code WORKING} the instant {@code confirm()} validates
|
||||
* every gate and schedules the roll. {@code dev.ltms.fleet.lead.LeadRollover}'s deferred
|
||||
* continuation waits up to this many seconds for that SAME pane to report {@code IDLE} or
|
||||
* {@code DONE} — i.e. for the calling turn to actually end — before it ends the old pane's
|
||||
* process at all. {@code BLOCKED} does not count: that is a live turn merely paused, not one
|
||||
* that has finished. If that wait times out, the old pane is never touched: a lead that never
|
||||
* goes idle is still doing real work, and the roll ends that pane's whole process — there is no
|
||||
* way back from this once it runs, so this wait is the only thing standing between "still
|
||||
* working" and "gone".
|
||||
*
|
||||
* @param handoverPath required when this block is present — where the handover file a fresh
|
||||
* lead session reads must live. There is no sane non-null default for an
|
||||
@@ -1403,37 +1462,51 @@ public record FleetConfig(
|
||||
* @param maxDocAgeSeconds default 3600 — refuse a handover file whose modified time is older
|
||||
* than this many seconds, so a stale leftover from an earlier rollover
|
||||
* attempt can never be mistaken for a fresh one.
|
||||
* @param turnSettleSeconds default 20 — bound on how long the deferred roll waits for the
|
||||
* CALLING lead's own turn to end (its pane to report injectable again)
|
||||
* before sending {@code /clear} at all. See the paragraph above.
|
||||
* @param clearSettleSeconds default 20 — bound on how long to wait for the lead's pane to
|
||||
* report an injectable state again after {@code /clear} before giving up. A
|
||||
* roll that times out here never sends {@code bootstrapText}.
|
||||
* @param turnSettleSeconds default 300 — bound on how long the deferred roll waits for the
|
||||
* CALLING lead's own turn to end (its pane to report {@code IDLE} or
|
||||
* {@code DONE}) before ending that pane's process at all. See the
|
||||
* paragraph above.
|
||||
* @param relaunchReadySeconds default 45 — bound on EACH of two separate waits that run after
|
||||
* the old lead's pane has been torn down and a fresh one launched: first,
|
||||
* for the fresh pane itself to reach a real turn boundary ({@code IDLE} or
|
||||
* {@code DONE}, never merely {@code BLOCKED}) — the safety gate, since
|
||||
* typing into a pane that has not finished booting loses the keystrokes;
|
||||
* second, for the fresh terminal to show up as a recognised lead, which is
|
||||
* bookkeeping rather than a safety gate, so a timeout on this second wait
|
||||
* does not withhold {@code bootstrapText} — it is sent once the pane is
|
||||
* ready regardless. Recognition comes from the same periodically-refreshed
|
||||
* scan {@code LeadTabScanner} already keeps ({@code scanIntervalSeconds},
|
||||
* 10s live), so a budget has to clear more than one scan interval to leave
|
||||
* any real margin for the CLI's own boot time; 20 was rejected for exactly
|
||||
* that reason — at a 10s scan interval it only buys two scans. 45 buys
|
||||
* roughly four. Only a timeout on the FIRST wait (the pane never becomes
|
||||
* ready) withholds {@code bootstrapText}.
|
||||
* @param bootstrapText default a sentence naming the RESOLVED handover path — sent to the
|
||||
* lead's pane once it settles after {@code /clear}, telling the fresh
|
||||
* session where to read the handover and carry on. Left {@code null} here
|
||||
* when the operator configures none: the default sentence cannot be built
|
||||
* at construction time because it must name the path AFTER {@code
|
||||
* dev.ltms.fleet.lead.LeadRollover#open} has resolved a relative {@code
|
||||
* handoverPath} against the calling lead's workspace, which this record has
|
||||
* no way to know — see {@link #bootstrapTextFor(String)}.
|
||||
* fresh lead's pane once it reaches a real turn boundary after relaunch,
|
||||
* telling the fresh session where to read the handover and carry on. Left
|
||||
* {@code null} here when the operator configures none: the default sentence
|
||||
* cannot be built at construction time because it must name the path AFTER
|
||||
* {@code dev.ltms.fleet.lead.LeadRollover#open} has resolved a relative
|
||||
* {@code handoverPath} against the calling lead's workspace, which this
|
||||
* record has no way to know — see {@link #bootstrapTextFor(String)}.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record LeadRollover(String handoverPath, Boolean requireOperatorConfirm,
|
||||
Integer maxDocAgeSeconds, Integer turnSettleSeconds,
|
||||
Integer clearSettleSeconds, String bootstrapText) {
|
||||
Integer relaunchReadySeconds, String bootstrapText) {
|
||||
public LeadRollover {
|
||||
requireOperatorConfirm = requireOperatorConfirm == null || requireOperatorConfirm;
|
||||
maxDocAgeSeconds = (maxDocAgeSeconds == null || maxDocAgeSeconds <= 0) ? 3600 : maxDocAgeSeconds;
|
||||
turnSettleSeconds = (turnSettleSeconds == null || turnSettleSeconds <= 0) ? 20 : turnSettleSeconds;
|
||||
clearSettleSeconds = (clearSettleSeconds == null || clearSettleSeconds <= 0) ? 20 : clearSettleSeconds;
|
||||
turnSettleSeconds = (turnSettleSeconds == null || turnSettleSeconds <= 0) ? 300 : turnSettleSeconds;
|
||||
relaunchReadySeconds = (relaunchReadySeconds == null || relaunchReadySeconds <= 0)
|
||||
? 45 : relaunchReadySeconds;
|
||||
bootstrapText = (bootstrapText == null || bootstrapText.isBlank()) ? null : bootstrapText;
|
||||
}
|
||||
|
||||
/**
|
||||
* The text actually sent to the lead's pane once it settles after {@code /clear}: the
|
||||
* operator's configured {@link #bootstrapText} when one is set, otherwise the default
|
||||
* sentence built from {@code resolvedHandoverPath}.
|
||||
* The text actually sent to the fresh lead's pane once it reaches a real turn boundary
|
||||
* after relaunch: the operator's configured {@link #bootstrapText} when one is set,
|
||||
* otherwise the default sentence built from {@code resolvedHandoverPath}.
|
||||
*
|
||||
* @param resolvedHandoverPath the ABSOLUTE path {@code dev.ltms.fleet.lead.LeadRollover
|
||||
* #open} already resolved — never the raw configured {@link
|
||||
@@ -1876,6 +1949,7 @@ public record FleetConfig(
|
||||
rejectNegativeMaxLoad(yaml);
|
||||
rejectAutoCompactWindowOutOfRange(yaml);
|
||||
warnConflictingAutoCompactWindows(yaml);
|
||||
warnRetiredClearSettleSecondsKey(yaml);
|
||||
rejectMalformedProfilePatterns(yaml);
|
||||
rejectUnknownKind(yaml);
|
||||
rejectUnknownAuthMode(yaml);
|
||||
@@ -1894,7 +1968,7 @@ public record FleetConfig(
|
||||
|
||||
/** The {@code fleet:} child blocks whose direct children are slot names. */
|
||||
private static final Set<String> FLEET_POOL_KEYS =
|
||||
Set.of("leaders", "architects", "developers", "hunters", "reviewers");
|
||||
Set.of("leaders", "architects", "developers", "hunters", "reviewers", "collaborators");
|
||||
|
||||
/**
|
||||
* Reject a {@code fleet:} role pool whose slot names repeat (CB-548, re-homed by CB-557).
|
||||
@@ -1904,7 +1978,7 @@ public record FleetConfig(
|
||||
* daemon would never know. Jackson's YAML parser does not fail on duplicate mapping keys by
|
||||
* default, so duplicates are caught here, at parse time, before the map is built.
|
||||
*
|
||||
* <p>Only the five pools <em>directly under the top-level {@code fleet:}</em> are considered,
|
||||
* <p>Only the six pools <em>directly under the top-level {@code fleet:}</em> are considered,
|
||||
* and only their direct child keys (the slot names). A nested field elsewhere, even one also
|
||||
* named {@code developers:}, is ignored, so parsing of the rest of the config is unaffected.
|
||||
*
|
||||
@@ -2039,13 +2113,6 @@ public record FleetConfig(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The top-level keys in {@code yaml} that this build does not understand, sorted. Package-private
|
||||
* so the guardrail is asserted directly rather than through a log appender.
|
||||
*
|
||||
* @return empty when everything is known, or when {@code yaml} is not a mapping at all (a
|
||||
* malformed file is {@code readValue}'s error to report, not this method's)
|
||||
*/
|
||||
/**
|
||||
* Top-level keys renamed by the member taxonomy, mapped old → new.
|
||||
*
|
||||
@@ -2184,6 +2251,12 @@ public record FleetConfig(
|
||||
static final int AUTO_COMPACT_WINDOW_MIN = 100_000;
|
||||
/** Highest {@code autoCompactWindow} Claude Code's {@code --autocompact <tokens>} flag accepts. */
|
||||
static final int AUTO_COMPACT_WINDOW_MAX = 1_000_000;
|
||||
/**
|
||||
* The {@code env:} key a launched Claude Code session reads for its auto-compaction window,
|
||||
* ahead of the {@code --autocompact} launch flag {@code autoCompactWindow} produces (see
|
||||
* {@link Profile#effectiveAutoCompactWindow()}).
|
||||
*/
|
||||
static final String CLAUDE_CODE_AUTO_COMPACT_WINDOW_ENV = "CLAUDE_CODE_AUTO_COMPACT_WINDOW";
|
||||
|
||||
/**
|
||||
* Reject a profile whose {@code autoCompactWindow:} is set but outside the token band Claude
|
||||
@@ -2265,13 +2338,13 @@ public record FleetConfig(
|
||||
if (!(entry.getValue() instanceof Map<?, ?> profile)
|
||||
|| !(profile.get("autoCompactWindow") instanceof Number window)
|
||||
|| !(profile.get("env") instanceof Map<?, ?> env)
|
||||
|| !env.containsKey("CLAUDE_CODE_AUTO_COMPACT_WINDOW")) {
|
||||
|| !env.containsKey(CLAUDE_CODE_AUTO_COMPACT_WINDOW_ENV)) {
|
||||
continue;
|
||||
}
|
||||
Object kind = profile.get("kind");
|
||||
boolean claudeCode = kind == null || String.valueOf(kind).isBlank()
|
||||
|| Profile.KIND_CLAUDE_CODE.equalsIgnoreCase(String.valueOf(kind));
|
||||
Object envValue = env.get("CLAUDE_CODE_AUTO_COMPACT_WINDOW");
|
||||
Object envValue = env.get(CLAUDE_CODE_AUTO_COMPACT_WINDOW_ENV);
|
||||
if (claudeCode && !String.valueOf(window).equals(String.valueOf(envValue))) {
|
||||
String name = String.valueOf(entry.getKey());
|
||||
names.add(name);
|
||||
@@ -2293,6 +2366,33 @@ public record FleetConfig(
|
||||
names, String.join(", ", detail));
|
||||
}
|
||||
|
||||
/**
|
||||
* Warn when a {@code leadRollover:} block still sets the retired {@code clearSettleSeconds}
|
||||
* key. {@link LeadRollover} carries {@code @JsonIgnoreProperties(ignoreUnknown = true)} and no
|
||||
* longer declares that component, so Jackson drops it with no signal of its own — this raw-YAML
|
||||
* check is the only place an operator's now-inert setting is reported at all; by the time a
|
||||
* {@link LeadRollover} instance exists to run a validator against, the key is already gone.
|
||||
*
|
||||
* @param yaml the raw config text
|
||||
*/
|
||||
static void warnRetiredClearSettleSecondsKey(String yaml) {
|
||||
Map<?, ?> raw;
|
||||
try {
|
||||
raw = YAML.readValue(yaml, Map.class);
|
||||
} catch (IOException | IllegalArgumentException e) {
|
||||
return;
|
||||
}
|
||||
if (raw == null || !(raw.get("leadRollover") instanceof Map<?, ?> leadRollover)) {
|
||||
return;
|
||||
}
|
||||
if (leadRollover.containsKey("clearSettleSeconds")) {
|
||||
log.warn("leadRollover.clearSettleSeconds is retired and no longer read. Set "
|
||||
+ "leadRollover.relaunchReadySeconds instead: it bounds how long to wait, after "
|
||||
+ "a lead is relaunched, for its pane to become ready and then for it to be "
|
||||
+ "recognised as a lead. Remove clearSettleSeconds from fleetd.yaml.");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Reject a profile whose {@code errorPattern} (fleetd #201 Unit 5) or {@code exhaustedPattern}
|
||||
* (CB-578 stage A) is not a valid Java regex, naming the profile, the key, and the parser's own
|
||||
@@ -2547,6 +2647,13 @@ public record FleetConfig(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The top-level keys in {@code yaml} that this build does not understand, sorted. Package-private
|
||||
* so the guardrail is asserted directly rather than through a log appender.
|
||||
*
|
||||
* @return empty when everything is known, or when {@code yaml} is not a mapping at all (a
|
||||
* malformed file is {@code readValue}'s error to report, not this method's)
|
||||
*/
|
||||
static List<String> unknownTopLevelKeys(String yaml) {
|
||||
Map<?, ?> raw;
|
||||
try {
|
||||
@@ -2657,59 +2764,243 @@ public record FleetConfig(
|
||||
}
|
||||
|
||||
/**
|
||||
* Reject a lead-scan convention that a worker tab would also satisfy (CB-531).
|
||||
* Reject a member tab-label template that could render as a configured lead or collaborator
|
||||
* tab, as the fixed lead tab label, or that matches a lead-tab naming convention; reject two
|
||||
* {@code fleet.leaders} entries that share one space; reject two {@code fleet.collaborators}
|
||||
* entries — or a lead and a collaborator — that share one exact tab; and reject a collaborator
|
||||
* tab equal to the fixed lead tab label.
|
||||
*
|
||||
* <p>The scan reads a tab label and concludes "a lead lives here". fleetd also <em>writes</em>
|
||||
* tab labels — every member gets one rendered into its tab. Choose a lead {@code tabPrefix} that
|
||||
* a member template matches and the daemon starts labelling its own members as leads, promoting
|
||||
* the entire fleet to {@link dev.ltms.fleet.auth.Role#PRIMARY} with no message and no diff.
|
||||
* The member-space exclusion in {@link dev.ltms.fleet.herdr.LeadTabScanner} already blocks the
|
||||
* realistic path, but defence that depends on one workspace label holding is not defence enough
|
||||
* for a privilege boundary.
|
||||
* <p>{@code fleet.collaborators} has no {@code tabPrefix}: identity is matched on the exact
|
||||
* {@code tab} alone, so only the exact-render check applies there, not the prefix check.
|
||||
*
|
||||
* <p>CB-557 shrank this check rather than removing it. The default template is
|
||||
* {@code "{role}: {profile} #{n}"} and {@code {role}} comes from a closed enum, so a
|
||||
* <em>generated</em> label can no longer collide by construction. What remains checkable is what
|
||||
* an operator still writes by hand: the {@code fleet.tabLabel} template and any per-profile
|
||||
* {@code tabLabel} override.
|
||||
*
|
||||
* <p>Fatal rather than a warning, unlike {@link #warnUnknownTopLevelKeys}: an unknown key means
|
||||
* a feature does nothing, while this means a feature does the opposite of what it says.
|
||||
*
|
||||
* @throws IllegalStateException when the fleet template or any profile's {@code tabLabel}
|
||||
* override starts with a configured lead prefix
|
||||
* @throws IllegalStateException when the fleet template or a profile {@code tabLabel} override
|
||||
* can render as a configured lead or collaborator tab, as the
|
||||
* fixed lead tab label, or match a lead-tab prefix; when two
|
||||
* leaders share one space; when two collaborators (or a lead and
|
||||
* a collaborator) carry the same exact {@code tab}
|
||||
* (case-insensitively); or when a collaborator's {@code tab}
|
||||
* equals the fixed lead tab label
|
||||
*/
|
||||
public void validateLeadTabPrefixes() {
|
||||
if (fleet == null || fleet.leaders().isEmpty()) {
|
||||
if (fleet == null) {
|
||||
return;
|
||||
}
|
||||
List<String> bad = new ArrayList<>();
|
||||
if (templateCanRenderAs(fleet.tabLabel(), Leader.LEAD_TAB_LABEL)) {
|
||||
bad.add("fleet.tabLabel=\"" + fleet.tabLabel() + "\" can render as \""
|
||||
+ Leader.LEAD_TAB_LABEL + "\", the fixed lead tab label");
|
||||
}
|
||||
profiles().entrySet().stream()
|
||||
.map(Map.Entry::getKey)
|
||||
.sorted()
|
||||
.forEach(p -> {
|
||||
String label = profiles().get(p).tabLabel();
|
||||
if (templateCanRenderAs(label, Leader.LEAD_TAB_LABEL)) {
|
||||
bad.add("profile '" + p + "' overrides tabLabel with \"" + label
|
||||
+ "\", which can render as \"" + Leader.LEAD_TAB_LABEL
|
||||
+ "\", the fixed lead tab label");
|
||||
}
|
||||
});
|
||||
fleet.leaders().forEach((leadName, leader) -> {
|
||||
if (leader == null) {
|
||||
return;
|
||||
}
|
||||
String tab = leader.tab();
|
||||
String prefix = leader.tabPrefix();
|
||||
// The fleet-wide template is checked once per prefix: it labels every member that has no
|
||||
// override, so one bad template promotes the entire fleet, not one profile.
|
||||
if (startsWithIgnoreCase(fleet.tabLabel(), prefix)) {
|
||||
if (templateCanRenderAs(fleet.tabLabel(), tab)) {
|
||||
bad.add("fleet.tabLabel=\"" + fleet.tabLabel() + "\" can render as the tab of "
|
||||
+ "lead '" + leadName + "' (\"" + tab + "\")");
|
||||
} else if (startsWithIgnoreCase(fleet.tabLabel(), prefix)) {
|
||||
bad.add("fleet.tabLabel=\"" + fleet.tabLabel() + "\" starts with the tabPrefix of "
|
||||
+ "lead '" + leadName + "' (\"" + prefix + "\")");
|
||||
}
|
||||
profiles().entrySet().stream()
|
||||
.filter(e -> startsWithIgnoreCase(e.getValue().tabLabel(), prefix))
|
||||
.map(Map.Entry::getKey)
|
||||
.sorted()
|
||||
.forEach(p -> bad.add("profile '" + p + "' overrides tabLabel with \""
|
||||
+ profiles().get(p).tabLabel() + "\", which starts with the tabPrefix of "
|
||||
+ "lead '" + leadName + "' (\"" + prefix + "\")"));
|
||||
.forEach(p -> {
|
||||
String label = profiles().get(p).tabLabel();
|
||||
if (templateCanRenderAs(label, tab)) {
|
||||
bad.add("profile '" + p + "' overrides tabLabel with \"" + label
|
||||
+ "\", which can render as the tab of lead '" + leadName
|
||||
+ "' (\"" + tab + "\")");
|
||||
} else if (startsWithIgnoreCase(label, prefix)) {
|
||||
bad.add("profile '" + p + "' overrides tabLabel with \"" + label
|
||||
+ "\", which starts with the tabPrefix of lead '" + leadName
|
||||
+ "' (\"" + prefix + "\")");
|
||||
}
|
||||
});
|
||||
});
|
||||
fleet.collaborators().forEach((collabName, collaborator) -> {
|
||||
if (collaborator == null) {
|
||||
return;
|
||||
}
|
||||
String tab = collaborator.tab();
|
||||
if (tab != null && tab.equalsIgnoreCase(Leader.LEAD_TAB_LABEL)) {
|
||||
bad.add("fleet.collaborators." + collabName + ".tab=\"" + tab + "\" is the fixed "
|
||||
+ "lead tab label — a collaborator there would shadow a lead");
|
||||
}
|
||||
if (templateCanRenderAs(fleet.tabLabel(), tab)) {
|
||||
bad.add("fleet.tabLabel=\"" + fleet.tabLabel() + "\" can render as the tab of "
|
||||
+ "collaborator '" + collabName + "' (\"" + tab + "\")");
|
||||
}
|
||||
profiles().entrySet().stream()
|
||||
.map(Map.Entry::getKey)
|
||||
.sorted()
|
||||
.forEach(p -> {
|
||||
String label = profiles().get(p).tabLabel();
|
||||
if (templateCanRenderAs(label, tab)) {
|
||||
bad.add("profile '" + p + "' overrides tabLabel with \"" + label
|
||||
+ "\", which can render as the tab of collaborator '"
|
||||
+ collabName + "' (\"" + tab + "\")");
|
||||
}
|
||||
});
|
||||
});
|
||||
if (!bad.isEmpty()) {
|
||||
throw new IllegalStateException("refusing to start: " + String.join("; ", bad)
|
||||
+ ". A member labelled that way, while its pane carries no entry in the "
|
||||
+ "spawned-member roster, is read back as a lead or collaborator and granted "
|
||||
+ "that identity's authority. Change one of the two so member tabs cannot be "
|
||||
+ "confused with a lead's or collaborator's tab.");
|
||||
}
|
||||
|
||||
List<String> collisions = new ArrayList<>();
|
||||
List<String> leadNames = fleet.leaders().keySet().stream().sorted().toList();
|
||||
for (int i = 0; i < leadNames.size(); i++) {
|
||||
String nameA = leadNames.get(i);
|
||||
Leader a = fleet.leaders().get(nameA);
|
||||
if (a == null) {
|
||||
continue;
|
||||
}
|
||||
for (int j = i + 1; j < leadNames.size(); j++) {
|
||||
String nameB = leadNames.get(j);
|
||||
Leader b = fleet.leaders().get(nameB);
|
||||
if (b == null) {
|
||||
continue;
|
||||
}
|
||||
if (a.workspace().equalsIgnoreCase(b.workspace())) {
|
||||
collisions.add("lead '" + nameA + "' and lead '" + nameB + "' share workspace \""
|
||||
+ a.workspace() + "\" — both would resolve to the tab named \""
|
||||
+ Leader.LEAD_TAB_LABEL + "\" in that space, so only one could ever be "
|
||||
+ "found");
|
||||
}
|
||||
}
|
||||
}
|
||||
List<String> collabNames = fleet.collaborators().keySet().stream().sorted().toList();
|
||||
for (int i = 0; i < collabNames.size(); i++) {
|
||||
String nameA = collabNames.get(i);
|
||||
Collaborator a = fleet.collaborators().get(nameA);
|
||||
if (a == null || a.tab() == null || a.tab().isBlank()) {
|
||||
continue;
|
||||
}
|
||||
for (int j = i + 1; j < collabNames.size(); j++) {
|
||||
String nameB = collabNames.get(j);
|
||||
Collaborator b = fleet.collaborators().get(nameB);
|
||||
if (b == null || b.tab() == null || b.tab().isBlank()) {
|
||||
continue;
|
||||
}
|
||||
if (a.tab().equalsIgnoreCase(b.tab())) {
|
||||
collisions.add("collaborator '" + nameA + "' and collaborator '" + nameB
|
||||
+ "' both use tab \"" + a.tab() + "\"");
|
||||
}
|
||||
}
|
||||
}
|
||||
for (String leadName : leadNames) {
|
||||
Leader lead = fleet.leaders().get(leadName);
|
||||
if (lead == null || lead.tab() == null || lead.tab().isBlank()) {
|
||||
continue;
|
||||
}
|
||||
for (String collabName : collabNames) {
|
||||
Collaborator collaborator = fleet.collaborators().get(collabName);
|
||||
if (collaborator == null || collaborator.tab() == null
|
||||
|| collaborator.tab().isBlank()) {
|
||||
continue;
|
||||
}
|
||||
if (lead.tab().equalsIgnoreCase(collaborator.tab())) {
|
||||
collisions.add("lead '" + leadName + "' and collaborator '" + collabName
|
||||
+ "' both use tab \"" + lead.tab() + "\"");
|
||||
}
|
||||
}
|
||||
}
|
||||
if (collisions.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
throw new IllegalStateException("refusing to start: " + String.join("; ", collisions)
|
||||
+ ". Identity is matched exactly, so only one of two entries sharing a space or a "
|
||||
+ "tab can ever be found — the other is silently unreachable. Give each lead its "
|
||||
+ "own space, and each collaborator its own exact tab.");
|
||||
}
|
||||
|
||||
private static boolean templateCanRenderAs(String template, String tab) {
|
||||
if (template == null || template.isBlank() || tab == null || tab.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
var placeholders = Pattern.compile("\\{(?:role|profile|model|n)}").matcher(template);
|
||||
StringBuilder expression = new StringBuilder("^");
|
||||
int literalStart = 0;
|
||||
while (placeholders.find()) {
|
||||
expression.append(Pattern.quote(template.substring(literalStart, placeholders.start())));
|
||||
expression.append(".*");
|
||||
literalStart = placeholders.end();
|
||||
}
|
||||
expression.append(Pattern.quote(template.substring(literalStart))).append("$");
|
||||
return Pattern.compile(expression.toString(), Pattern.CASE_INSENSITIVE).matcher(tab).matches();
|
||||
}
|
||||
|
||||
/** Case-insensitive prefix test that tolerates a null or blank label. */
|
||||
private static boolean startsWithIgnoreCase(String label, String prefix) {
|
||||
if (label == null || prefix == null || prefix.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
String stripped = label.strip();
|
||||
return stripped.regionMatches(true, 0, prefix, 0, prefix.length());
|
||||
}
|
||||
|
||||
/**
|
||||
* Reject a profile that places its members by {@code "pane"} while {@code fleet.leaders} has
|
||||
* any entry, or any {@code fleet.collaborators} entry names a {@code tab}. A pane-placed
|
||||
* member lands inside the focused tab rather than its own, so it can land inside a lead's or
|
||||
* collaborator's own labelled tab. {@link dev.ltms.fleet.herdr.LeadTabScanner} identifies a
|
||||
* lead or collaborator purely by that tab's label — it does not exclude the member space — so
|
||||
* a member that ends up there, while its pane carries no entry in the spawned-member roster,
|
||||
* is read back as that lead or collaborator and granted that identity's authority.
|
||||
*
|
||||
* <p>Every {@code fleet.leaders} entry is in scope regardless of its own {@code tab} field:
|
||||
* {@link Leader#acceptedLabels()} always includes {@link Leader#LEAD_TAB_LABEL}. Only a
|
||||
* collaborator with a non-blank {@code tab} is in scope: one with no {@code tab} feeds
|
||||
* nothing into {@link dev.ltms.fleet.herdr.LeadTabScanner}, so it creates no hazard here.
|
||||
*
|
||||
* @throws IllegalStateException when any {@code profiles:} entry is pane-placed while
|
||||
* {@code fleet.leaders} is non-empty, or any
|
||||
* {@code fleet.collaborators} entry names a non-blank
|
||||
* {@code tab}
|
||||
*/
|
||||
public void validatePanePlacementAgainstLeadTabs() {
|
||||
if (fleet == null) {
|
||||
return;
|
||||
}
|
||||
boolean anyLead = !fleet.leaders().isEmpty();
|
||||
boolean anyCollaboratorHasTab = fleet.collaborators().values().stream()
|
||||
.anyMatch(c -> c != null && c.tab() != null && !c.tab().isBlank());
|
||||
if (!anyLead && !anyCollaboratorHasTab) {
|
||||
return;
|
||||
}
|
||||
List<String> bad = new ArrayList<>();
|
||||
profiles().entrySet().stream()
|
||||
.filter(e -> !e.getValue().tabPlacement())
|
||||
.map(Map.Entry::getKey)
|
||||
.sorted()
|
||||
.forEach(bad::add);
|
||||
if (bad.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
throw new IllegalStateException("refusing to start: " + String.join("; ", bad)
|
||||
+ ". Every member labelled that way would be read back as a lead and granted "
|
||||
+ "spawn/stop/send on the whole fleet. Change one of the two so member tabs and "
|
||||
+ "lead tabs cannot be confused.");
|
||||
throw new IllegalStateException("refusing to start: profile(s) " + bad
|
||||
+ " use placement: pane while fleet.leaders or fleet.collaborators names a tab. A "
|
||||
+ "pane-placed member can land inside that labelled tab, and while its pane "
|
||||
+ "carries no entry in the spawned-member roster, it is read back as the lead or "
|
||||
+ "collaborator and granted that identity's authority. Set placement: tab for "
|
||||
+ "each named profile — the only fix when a lead triggered this, since a lead's "
|
||||
+ "tab label is fixed regardless of its own tab: field. A collaborator's tab can "
|
||||
+ "still be removed instead.");
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -2732,15 +3023,6 @@ public record FleetConfig(
|
||||
}
|
||||
}
|
||||
|
||||
/** Case-insensitive prefix test that tolerates a null/blank label. */
|
||||
private static boolean startsWithIgnoreCase(String label, String prefix) {
|
||||
if (label == null || prefix == null || prefix.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
String stripped = label.strip();
|
||||
return stripped.regionMatches(true, 0, prefix, 0, prefix.length());
|
||||
}
|
||||
|
||||
/**
|
||||
* Reject a subscription profile whose {@code env:} block tries to reseat the Anthropic binding
|
||||
* (CB-542).
|
||||
@@ -2825,8 +3107,13 @@ public record FleetConfig(
|
||||
* so duplicates are unrepresentable by construction once loaded — and {@link #load(Path)}
|
||||
* already rejects a duplicated slot name at parse time, before the map collapses.
|
||||
*
|
||||
* @throws IllegalStateException when a slot names no profile or an unknown one, or when a lead
|
||||
* can be neither found nor created, naming the offending entry
|
||||
* <p>A lead's {@code profile} is optional — a {@code profile}-less lead is still useful
|
||||
* recognise-only. Also rejects a {@code fleet.collaborators} entry with no (or a blank)
|
||||
* {@code tab}: a collaborator carries no other field at all, so a blank {@code tab} leaves
|
||||
* nothing for the entry to mean.
|
||||
*
|
||||
* @throws IllegalStateException when a slot or a lead references an unknown profile, or a
|
||||
* collaborator names no tab, naming the offending entry
|
||||
*/
|
||||
public void validateMembers() {
|
||||
if (fleet == null) {
|
||||
@@ -2859,9 +3146,15 @@ public record FleetConfig(
|
||||
+ "', which is not a configured profiles: entry (have: " + profiles.keySet()
|
||||
+ ").");
|
||||
}
|
||||
if (leader.tab() == null || leader.tab().isBlank()) {
|
||||
bad.add("fleet.leaders." + name + " has no tab: — a lead is now found (and, if "
|
||||
+ "auto-launched, labelled) purely by its tab, so every entry must name one.");
|
||||
});
|
||||
fleet.collaborators().forEach((name, collaborator) -> {
|
||||
if (collaborator == null) {
|
||||
return;
|
||||
}
|
||||
if (collaborator.tab() == null || collaborator.tab().isBlank()) {
|
||||
bad.add("fleet.collaborators." + name + " has no tab: — a collaborator is "
|
||||
+ "recognised purely by its tab, and carries no other field, so every "
|
||||
+ "entry must name one.");
|
||||
}
|
||||
});
|
||||
if (!bad.isEmpty()) {
|
||||
@@ -2912,11 +3205,11 @@ public record FleetConfig(
|
||||
* Runs every validator this class declares — found by reflection, not by name.
|
||||
*
|
||||
* <p>fleetd ticket "central allow-list of usable models", follow-up: mutation testing found
|
||||
* that although each of the six validators above was well pinned on its own, nothing proved
|
||||
* that although each validator above was well pinned on its own, nothing proved
|
||||
* either real caller ({@code Fleetd.main} and {@link ConfigRef#reload()}) still
|
||||
* invoked it — deleting a call site left the full suite green. The fix is not a seventh test
|
||||
* per caller; a hand-maintained list of six names here would have the exact same defect its
|
||||
* own javadoc would warn against: the seventh validator someone adds next month has no reason
|
||||
* invoked it — deleting a call site left the full suite green. The fix is not one more test
|
||||
* per caller; a hand-maintained list of names here would have the exact same defect its
|
||||
* own javadoc would warn against: the next validator someone adds has no reason
|
||||
* to be added to it. So this method does not name any validator. It sweeps {@link
|
||||
* #getClass()}'s own public, no-argument, {@code void} methods whose name starts with {@code
|
||||
* "validate"} (excluding itself) and invokes every one it finds, via {@link
|
||||
@@ -2925,7 +3218,7 @@ public record FleetConfig(
|
||||
* which it silently never runs.
|
||||
*
|
||||
* <p>{@code Fleetd.main} and {@link ConfigRef#reload()} each call this one method instead of
|
||||
* the six individually — see the comments at those two call sites for why
|
||||
* each validator individually — see the comments at those two call sites for why
|
||||
* each must run it.
|
||||
*
|
||||
* <p>Methods run in a fixed (alphabetical) order, so a config with more than one violation
|
||||
@@ -2942,9 +3235,9 @@ public record FleetConfig(
|
||||
/**
|
||||
* The reflective sweep behind {@link #validateAll()}, kept as its own method — taking any
|
||||
* {@code target}, not just {@code this} — so a test can prove the MECHANISM is generic (it
|
||||
* would sweep a seventh {@code validateXxx()} method added to any class, not just something
|
||||
* special-cased to today's six on {@link FleetConfig}) without needing to add a real, unwanted
|
||||
* seventh validator to this class just to exercise that claim. See {@code
|
||||
* would sweep any new {@code validateXxx()} method added to any class, not just something
|
||||
* special-cased to the set {@link FleetConfig} declares today) without needing to add a real,
|
||||
* unwanted extra validator to this class just to exercise that claim. See {@code
|
||||
* FleetConfigValidateAllTest} for that proof.
|
||||
*
|
||||
* @param target an object whose public, no-argument, {@code void} methods named {@code
|
||||
|
||||
@@ -137,6 +137,18 @@ public final class AgentControl {
|
||||
return result.path("read").path("text").asText("");
|
||||
}
|
||||
|
||||
/**
|
||||
* Read an agent's terminal with its ANSI styling kept, instead of the stripped text {@link
|
||||
* #read} returns. Needed when a caller must tell apart text the pane draws dim (a placeholder
|
||||
* hint) from text drawn plain (the operator's own typing).
|
||||
*
|
||||
* @param source one of {@code visible|recent|recent_unwrapped|detection}
|
||||
*/
|
||||
public String readWithStyling(String target, String source) {
|
||||
JsonNode result = agentCall("agent.read", target, Map.of("source", source, "strip_ansi", false));
|
||||
return result.path("read").path("text").asText("");
|
||||
}
|
||||
|
||||
/** Current agent record (status, session UUID, pane). */
|
||||
public Agent get(String target) {
|
||||
return Agent.from(agentCall("agent.get", target, Map.of()).get("agent"));
|
||||
|
||||
@@ -11,12 +11,18 @@ public final class HerdrRouter implements AutoCloseable {
|
||||
private final AgentControl memberAgents;
|
||||
private final WorkspaceControl leadSpaces;
|
||||
private final WorkspaceControl memberSpaces;
|
||||
private final Predicate<String> isLead;
|
||||
private final Predicate<String> routeToLead;
|
||||
|
||||
public HerdrRouter(HerdrClient lead, HerdrClient member, Predicate<String> isLead) {
|
||||
/**
|
||||
* @param routeToLead true for a terminal whose pane lives in the lead herdr daemon — a lead's
|
||||
* own pane or a configured collaborator's, both opened by a person at a
|
||||
* terminal rather than spawned, so both are found in the lead daemon rather
|
||||
* than the member one
|
||||
*/
|
||||
public HerdrRouter(HerdrClient lead, HerdrClient member, Predicate<String> routeToLead) {
|
||||
this.lead = Objects.requireNonNull(lead, "lead");
|
||||
this.member = member != null ? member : lead;
|
||||
this.isLead = Objects.requireNonNull(isLead, "isLead");
|
||||
this.routeToLead = Objects.requireNonNull(routeToLead, "routeToLead");
|
||||
leadAgents = new AgentControl(this.lead);
|
||||
memberAgents = this.member == this.lead ? leadAgents : new AgentControl(this.member);
|
||||
leadSpaces = new WorkspaceControl(this.lead);
|
||||
@@ -27,7 +33,7 @@ public final class HerdrRouter implements AutoCloseable {
|
||||
public WorkspaceControl leadSpaces() { return leadSpaces; }
|
||||
public AgentControl memberAgents() { return memberAgents; }
|
||||
public WorkspaceControl memberSpaces() { return memberSpaces; }
|
||||
public AgentControl agentsFor(String targetId) { return isLead.test(targetId) ? leadAgents : memberAgents; }
|
||||
public AgentControl agentsFor(String targetId) { return routeToLead.test(targetId) ? leadAgents : memberAgents; }
|
||||
|
||||
HerdrClient leadClient() { return lead; }
|
||||
HerdrClient memberClient() { return member; }
|
||||
|
||||
@@ -25,20 +25,21 @@ import java.util.function.Supplier;
|
||||
* by first starting the session and asking it. Scanning closes that loop: label the tab, and the
|
||||
* pane is recognised on the next resolve.
|
||||
*
|
||||
* <p><strong>CB-579 — matched by name, not prefix.</strong> This used to strip one shared
|
||||
* {@code tabPrefix} off a label to derive the lead's name, and merged a config-supplied
|
||||
* {@code terminal_id} pin over every scan result so the pin could never expire. Both are gone: each
|
||||
* lead now configures its own exact {@code tab} label ({@code fleet.leaders.<name>.tab}), so this
|
||||
* class is handed a {@code tab → name} map up front and matches labels against it exactly
|
||||
* (case-insensitively). There is no merge step — a scan result is the whole answer. That is the
|
||||
* fix for the bug this replaces: a {@code terminal_id} pin surviving in config after the pane it
|
||||
* named was gone, so the daemon kept treating a dead session as a live lead forever.
|
||||
* <p><strong>Matched by label within a space, not by a shared prefix.</strong> Each lead's accepted
|
||||
* labels (the fixed {@code lead} label, plus a deprecated {@code tab} when still configured) are
|
||||
* matched exactly (case-insensitively) against tabs in that lead's own space only — a tab named
|
||||
* {@code lead} in one space never resolves to another space's lead. A scan result is the whole
|
||||
* answer; nothing is merged in from configuration between scans, so a tab that is gone drops out on
|
||||
* the very next scan instead of lingering forever.
|
||||
*
|
||||
* <p><strong>Direction of trust.</strong> The label names the lead; it never <em>grants</em>
|
||||
* anything a pane could take for itself. Three properties keep that honest:
|
||||
* <ol>
|
||||
* <li>Worker spaces are excluded wholesale ({@code excludedWorkspaceLabels}), so a worker cannot
|
||||
* become a lead by being placed — as a split, say — inside a matching tab.</li>
|
||||
* <li>{@code excludedWorkspaceLabels} can filter a workspace out of the scan, but this class does
|
||||
* not by itself stop a worker from landing inside a matching tab — a caller may pass an empty
|
||||
* set, and the daemon does. The guard against that is {@code
|
||||
* FleetConfig.validatePanePlacementAgainstLeadTabs}: it refuses, at startup, any profile that
|
||||
* places members by pane while a lead names a tab.</li>
|
||||
* <li>A worker cannot rename a tab: {@code tab.rename} is reachable only through
|
||||
* {@link WorkspaceControl}, which no {@code fleet_*} tool exposes. The label is writable by
|
||||
* the human at the terminal and by nobody the bridge is defending against.</li>
|
||||
@@ -93,13 +94,20 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadTabScanner.class);
|
||||
|
||||
/** What a matched tab names: a lead or a collaborator. */
|
||||
private enum Kind { LEAD, COLLABORATOR }
|
||||
|
||||
/** One matched tab's name and what it names. */
|
||||
private record Entry(String name, Kind kind) {}
|
||||
|
||||
private final HerdrClient herdr;
|
||||
private final Map<String, String> tabToName;
|
||||
private final Map<String, Map<String, String>> leadLabelsBySpace;
|
||||
private final Map<String, String> collaboratorTabToName;
|
||||
private final Set<String> excludedWorkspaceLabels;
|
||||
private final long ttlNanos;
|
||||
private final LongSupplier clock;
|
||||
|
||||
private Map<String, String> cached = Map.of();
|
||||
private Map<String, Entry> cached = Map.of();
|
||||
private long scannedAtNanos;
|
||||
private boolean everScanned;
|
||||
|
||||
@@ -115,36 +123,82 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
/**
|
||||
* @param herdr the herdr client to query ({@code workspace.list},
|
||||
* {@code tab.list}, {@code pane.list} — all read-only)
|
||||
* @param tabToName every configured lead's exact tab label → its name
|
||||
* ({@code fleet.leaders.<name>.tab}), matched case-insensitively
|
||||
* @param leadLabelsBySpace each configured lead's accepted tab labels, keyed by the
|
||||
* lead's own space label, then by label, to its name — matched
|
||||
* case-insensitively on both the space and the label. A tab
|
||||
* matches a lead only within that lead's own space
|
||||
* @param excludedWorkspaceLabels workspaces never scanned — the configured worker spaces
|
||||
* @param ttlNanos how long a scan result is reused before the next one
|
||||
* @param clock nanosecond time source ({@code System::nanoTime} in production)
|
||||
*/
|
||||
public LeadTabScanner(HerdrClient herdr, Map<String, String> tabToName,
|
||||
public LeadTabScanner(HerdrClient herdr, Map<String, Map<String, String>> leadLabelsBySpace,
|
||||
Set<String> excludedWorkspaceLabels, long ttlNanos, LongSupplier clock) {
|
||||
this(herdr, leadLabelsBySpace, Map.of(), excludedWorkspaceLabels, ttlNanos, clock);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #LeadTabScanner(HerdrClient, Map, Set, long, LongSupplier)}, additionally scanning
|
||||
* for configured collaborator tabs in the same pass.
|
||||
*
|
||||
* @param collaboratorTabToName every configured collaborator's exact tab label → its name
|
||||
* ({@code fleet.collaborators.<name>.tab}), matched
|
||||
* case-insensitively in any space
|
||||
*/
|
||||
public LeadTabScanner(HerdrClient herdr, Map<String, Map<String, String>> leadLabelsBySpace,
|
||||
Map<String, String> collaboratorTabToName,
|
||||
Set<String> excludedWorkspaceLabels, long ttlNanos, LongSupplier clock) {
|
||||
this.herdr = herdr;
|
||||
this.tabToName = normalize(tabToName);
|
||||
this.leadLabelsBySpace = buildLeadIndex(leadLabelsBySpace);
|
||||
this.collaboratorTabToName = normalizedLabelMap(collaboratorTabToName);
|
||||
this.excludedWorkspaceLabels = excludedWorkspaceLabels == null
|
||||
? Set.of() : Set.copyOf(excludedWorkspaceLabels);
|
||||
this.ttlNanos = ttlNanos;
|
||||
this.clock = clock;
|
||||
}
|
||||
|
||||
/** Keys stripped and lower-cased once, so every lookup is a plain map hit. */
|
||||
private static Map<String, String> normalize(Map<String, String> tabToName) {
|
||||
if (tabToName == null || tabToName.isEmpty()) {
|
||||
/**
|
||||
* Space and label keys stripped and lower-cased once, so every lookup is a plain map hit. A
|
||||
* space with no usable labels is simply absent — {@link #leadLabelsFor} then finds nothing for
|
||||
* it, which is also what a space with a {@code null} label gets.
|
||||
*/
|
||||
private static Map<String, Map<String, String>> buildLeadIndex(
|
||||
Map<String, Map<String, String>> leadLabelsBySpace) {
|
||||
Map<String, Map<String, String>> out = new LinkedHashMap<>();
|
||||
if (leadLabelsBySpace == null) {
|
||||
return Map.of();
|
||||
}
|
||||
Map<String, String> out = new LinkedHashMap<>();
|
||||
tabToName.forEach((tab, name) -> {
|
||||
if (tab != null && !tab.isBlank() && name != null && !name.isBlank()) {
|
||||
out.put(tab.strip().toLowerCase(Locale.ROOT), name);
|
||||
leadLabelsBySpace.forEach((space, labelsToName) -> {
|
||||
if (space == null || space.isBlank()) {
|
||||
return;
|
||||
}
|
||||
Map<String, String> normalized = normalizedLabelMap(labelsToName);
|
||||
if (!normalized.isEmpty()) {
|
||||
out.put(space.strip().toLowerCase(Locale.ROOT), normalized);
|
||||
}
|
||||
});
|
||||
return Collections.unmodifiableMap(out);
|
||||
}
|
||||
|
||||
private static Map<String, String> normalizedLabelMap(Map<String, String> labelToName) {
|
||||
Map<String, String> out = new LinkedHashMap<>();
|
||||
if (labelToName != null) {
|
||||
labelToName.forEach((label, name) -> {
|
||||
if (label != null && !label.isBlank() && name != null && !name.isBlank()) {
|
||||
out.put(label.strip().toLowerCase(Locale.ROOT), name);
|
||||
}
|
||||
});
|
||||
}
|
||||
return Collections.unmodifiableMap(out);
|
||||
}
|
||||
|
||||
/** The accepted lead labels configured for {@code spaceLabel}, or an empty map for no match. */
|
||||
private Map<String, String> leadLabelsFor(String spaceLabel) {
|
||||
if (spaceLabel == null) {
|
||||
return Map.of();
|
||||
}
|
||||
return leadLabelsBySpace.getOrDefault(spaceLabel.strip().toLowerCase(Locale.ROOT), Map.of());
|
||||
}
|
||||
|
||||
/**
|
||||
* The current {@code terminal_id → lead name} map, rescanning when the cache has expired.
|
||||
*
|
||||
@@ -153,6 +207,29 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
*/
|
||||
@Override
|
||||
public synchronized Map<String, String> get() {
|
||||
return byKind(refresh(), Kind.LEAD);
|
||||
}
|
||||
|
||||
/**
|
||||
* The current {@code terminal_id → collaborator name} map, sharing the same scan and cache as
|
||||
* {@link #get()} — both kinds are matched in one pass, so this never costs a second herdr call.
|
||||
*/
|
||||
public synchronized Map<String, String> collaborators() {
|
||||
return byKind(refresh(), Kind.COLLABORATOR);
|
||||
}
|
||||
|
||||
private static Map<String, String> byKind(Map<String, Entry> entries, Kind kind) {
|
||||
Map<String, String> out = new LinkedHashMap<>();
|
||||
entries.forEach((terminal, entry) -> {
|
||||
if (entry.kind() == kind) {
|
||||
out.put(terminal, entry.name());
|
||||
}
|
||||
});
|
||||
return Collections.unmodifiableMap(out);
|
||||
}
|
||||
|
||||
/** Rescans if the cache has expired, otherwise returns the cached answer. */
|
||||
private Map<String, Entry> refresh() {
|
||||
long now = clock.getAsLong();
|
||||
if (everScanned && now - scannedAtNanos < ttlNanos) {
|
||||
return cached;
|
||||
@@ -162,43 +239,45 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
scannedAtNanos = now;
|
||||
everScanned = true;
|
||||
try {
|
||||
Map<String, String> fresh = scan();
|
||||
Map<String, Entry> fresh = scan();
|
||||
if (!fresh.equals(cached)) {
|
||||
log.info("lead panes: {}", fresh);
|
||||
log.info("lead/collaborator panes: {}", fresh);
|
||||
}
|
||||
cached = fresh;
|
||||
} catch (HerdrException e) {
|
||||
log.warn("lead-tab scan failed, keeping the {} lead(s) already known: {}",
|
||||
log.warn("lead-tab scan failed, keeping the {} entr(y/ies) already known: {}",
|
||||
cached.size(), e.getMessage());
|
||||
}
|
||||
return cached;
|
||||
}
|
||||
|
||||
/** One full pass: labelled tabs → live agents in them → those panes' terminals. */
|
||||
private Map<String, String> scan() {
|
||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
||||
private Map<String, Entry> scan() {
|
||||
Map<String, Entry> entryByTab = new LinkedHashMap<>();
|
||||
for (JsonNode w : herdr.call("workspace.list").path("workspaces")) {
|
||||
Workspace ws = Workspace.from(w);
|
||||
if (ws.workspaceId() == null || excludedWorkspaceLabels.contains(ws.label())) {
|
||||
continue;
|
||||
}
|
||||
Map<String, String> leadLabelsHere = leadLabelsFor(ws.label());
|
||||
for (JsonNode t : herdr.call("tab.list", Map.of("workspace_id", ws.workspaceId())).path("tabs")) {
|
||||
Tab tab = Tab.from(t);
|
||||
String name = leadNameOf(tab.label());
|
||||
if (name != null && tab.tabId() != null) {
|
||||
nameByTab.put(tab.tabId(), name);
|
||||
Entry entry = entryOf(tab.label(), leadLabelsHere);
|
||||
if (entry != null && tab.tabId() != null) {
|
||||
entryByTab.put(tab.tabId(), entry);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (nameByTab.isEmpty()) {
|
||||
if (entryByTab.isEmpty()) {
|
||||
gracedTerminals = Set.of();
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
// fleetd #359: a labelled tab is only a lead when herdr also reports a running agent in
|
||||
// it — the same liveness signal LeadLauncher.countLeads trusts for the identical purpose.
|
||||
// Without this, a tab left behind by a session that has since died reads as live forever.
|
||||
// fleetd #359: a labelled tab is only a lead (or collaborator) when herdr also reports a
|
||||
// running agent in it — the same liveness signal LeadLauncher.countLeads trusts for the
|
||||
// identical purpose. Without this, a tab left behind by a session that has since died reads
|
||||
// as live forever.
|
||||
Set<String> tabsWithAgent = new HashSet<>();
|
||||
for (JsonNode a : herdr.call("agent.list").path("agents")) {
|
||||
String tabId = a.path("tab_id").asText(null);
|
||||
@@ -207,18 +286,18 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
}
|
||||
}
|
||||
|
||||
Map<String, String> byTerminal = new LinkedHashMap<>();
|
||||
Map<String, Entry> byTerminal = new LinkedHashMap<>();
|
||||
Set<String> stillGraced = new HashSet<>();
|
||||
// One pane.list for every tab: panes carry tab_id, so the join is local.
|
||||
for (JsonNode p : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String tabId = p.path("tab_id").asText(null);
|
||||
String name = nameByTab.get(tabId);
|
||||
Entry entry = entryByTab.get(tabId);
|
||||
String terminal = p.path("terminal_id").asText(null);
|
||||
if (name == null || terminal == null || terminal.isBlank()) {
|
||||
if (entry == null || terminal == null || terminal.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
if (tabsWithAgent.contains(tabId)) {
|
||||
byTerminal.put(terminal, name);
|
||||
byTerminal.put(terminal, entry);
|
||||
continue;
|
||||
}
|
||||
// No agent reported for this tab, but its tab/pane are still here — this is the
|
||||
@@ -227,7 +306,7 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
// reported as live; a terminal we never reported live gets none, so the original #359
|
||||
// fix (a genuinely dead tab is never reported) is unaffected for the common case.
|
||||
if (cached.containsKey(terminal) && !gracedTerminals.contains(terminal)) {
|
||||
byTerminal.put(terminal, name);
|
||||
byTerminal.put(terminal, entry);
|
||||
stillGraced.add(terminal);
|
||||
}
|
||||
}
|
||||
@@ -236,18 +315,27 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
}
|
||||
|
||||
/**
|
||||
* The lead name a tab label declares, or {@code null} if it names none of the configured leads.
|
||||
* The entry a tab label declares within one space, or {@code null} if it names neither a lead
|
||||
* accepted in {@code leadLabelsHere} nor a configured collaborator.
|
||||
*
|
||||
* <p>Exact match (case-insensitive, ends stripped) against {@link #tabToName} — no prefix
|
||||
* stripping, so an operator's {@code "lead: something-else"} tab is never mistaken for a
|
||||
* configured lead just because it shares a prefix. The match strips a trailing
|
||||
* {@link PendingCloseMarker} first, so a tab {@code LeadLauncher} has flagged as maybe-dead but
|
||||
* not yet closed keeps resolving normally while that reconcile is pending.
|
||||
* <p>Exact match (case-insensitive, ends stripped) — no prefix stripping, so an operator's
|
||||
* {@code "lead: something-else"} tab is never mistaken for a configured lead just because it
|
||||
* shares a prefix. The match strips a trailing {@link PendingCloseMarker} first, so a tab
|
||||
* {@code LeadLauncher} has flagged as maybe-dead but not yet closed keeps resolving normally
|
||||
* while that reconcile is pending. A lead match wins over a collaborator match for the same
|
||||
* label — a lead can already do everything a collaborator can, and config validation refuses a
|
||||
* lead and a collaborator sharing one exact tab in the first place.
|
||||
*/
|
||||
private String leadNameOf(String label) {
|
||||
private Entry entryOf(String label, Map<String, String> leadLabelsHere) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
return tabToName.get(PendingCloseMarker.strip(label).toLowerCase(Locale.ROOT));
|
||||
String normalized = PendingCloseMarker.strip(label).toLowerCase(Locale.ROOT);
|
||||
String leadName = leadLabelsHere.get(normalized);
|
||||
if (leadName != null) {
|
||||
return new Entry(leadName, Kind.LEAD);
|
||||
}
|
||||
String collaboratorName = collaboratorTabToName.get(normalized);
|
||||
return collaboratorName == null ? null : new Entry(collaboratorName, Kind.COLLABORATOR);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ import com.fasterxml.jackson.databind.JsonNode;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
@@ -145,6 +146,52 @@ public final class PaneLocator {
|
||||
return ancestry;
|
||||
}
|
||||
|
||||
/**
|
||||
* Every tab herdr tracks across every searched daemon, keyed by tab id, to its display label —
|
||||
* the pane-discovery surface behind {@code GET /agents} and {@code fleet_list}'s {@code panes}
|
||||
* row. Collapses to one scan in the single-daemon deployment, the same as
|
||||
* {@link #terminalForPid}. A tab herdr reports with no label maps to a {@code null} value here;
|
||||
* a tab with no {@code tab_id} is skipped.
|
||||
*/
|
||||
public Map<String, String> tabLabelsByTabId() {
|
||||
Map<String, String> out = new LinkedHashMap<>();
|
||||
for (HerdrClient herdr : herdrs) {
|
||||
for (JsonNode w : herdr.call("workspace.list").path("workspaces")) {
|
||||
String workspaceId = w.path("workspace_id").asText(null);
|
||||
if (workspaceId == null) {
|
||||
continue;
|
||||
}
|
||||
for (JsonNode t : herdr.call("tab.list", Map.of("workspace_id", workspaceId)).path("tabs")) {
|
||||
Tab tab = Tab.from(t);
|
||||
if (tab.tabId() != null) {
|
||||
out.put(tab.tabId(), tab.label());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Every workspace ("space") herdr tracks across every searched daemon, keyed by workspace id,
|
||||
* to its display label — the human-readable name behind {@code fleet_list}'s {@code panes} row,
|
||||
* next to herdr's own internal {@code workspaceId}. Collapses to one scan in the single-daemon
|
||||
* deployment, the same as {@link #terminalForPid}. A workspace herdr reports with no label maps
|
||||
* to a {@code null} value here; a workspace with no {@code workspace_id} is skipped.
|
||||
*/
|
||||
public Map<String, String> workspaceLabelsByWorkspaceId() {
|
||||
Map<String, String> out = new LinkedHashMap<>();
|
||||
for (HerdrClient herdr : herdrs) {
|
||||
for (JsonNode w : herdr.call("workspace.list").path("workspaces")) {
|
||||
Workspace workspace = Workspace.from(w);
|
||||
if (workspace.workspaceId() != null) {
|
||||
out.put(workspace.workspaceId(), workspace.label());
|
||||
}
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Whether a pane owns one of the scanned pid's ancestors, or the check of it failed outright. */
|
||||
private enum Ownership { OWNS, DOES_NOT_OWN, UNKNOWN }
|
||||
|
||||
|
||||
@@ -0,0 +1,231 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Whether an agent pane's input box is clear for a delivery.
|
||||
*
|
||||
* <p>{@link AgentControl#send} pastes its text and submits it in the same call, so a delivery into a
|
||||
* pane whose input box already holds characters submits those characters too. {@link
|
||||
* AgentStatus#injectable()} cannot see that: it describes the agent, and an agent waiting at its
|
||||
* prompt reports the same status whether its box is empty or holds a half-typed line. This reads the
|
||||
* box itself.
|
||||
*
|
||||
* <p>Only a box that is positively empty clears the gate. A box with content, a pane this cannot
|
||||
* recognise, and a failed read all hold the delivery, because a held delivery is recoverable and a
|
||||
* submitted half-line is not. Every caller must therefore be a path that retries.
|
||||
*
|
||||
* <p>A pane that holds for {@link #HOLD_WARN_STREAK} consecutive checks gets one warning, so a box
|
||||
* that never clears is visible instead of silent. The warning repeats only after the box has cleared
|
||||
* again.
|
||||
*/
|
||||
public final class PromptBox {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(PromptBox.class);
|
||||
|
||||
/**
|
||||
* herdr {@code agent.read} source, read with its ANSI styling kept. The input box is always
|
||||
* drawn here, carrying transcript scrollback above it, which is why only the last box line is
|
||||
* the live one. Styling must survive the read because the pane draws a placeholder hint — the
|
||||
* pane's own last submitted prompt — in the same spot as unsubmitted text, dimmed; only the
|
||||
* escape codes tell the two apart.
|
||||
*/
|
||||
static final String PROBE_SOURCE = "visible";
|
||||
|
||||
/** Consecutive holds for one target before one warning is logged. */
|
||||
static final int HOLD_WARN_STREAK = 20;
|
||||
|
||||
/**
|
||||
* Input box markers, each matched only as a line's first characters once any leading ANSI
|
||||
* escape codes are skipped: the caret the current TUI draws, and the bordered box an older one
|
||||
* drew. A marker further along a line is transcript text, such as a caret inside something the
|
||||
* operator quoted.
|
||||
*/
|
||||
private static final List<String> BOX_MARKERS = List.of("❯", "│ >");
|
||||
|
||||
/** Marker of a turn that is still generating; a box drawn under it is not a settled prompt. */
|
||||
private static final String ACTIVE_TURN_MARKER = "esc to interrupt";
|
||||
|
||||
/** Block glyphs a terminal capture can leave in an otherwise empty box for the cursor cell. */
|
||||
private static final String CURSOR_GLYPHS = "█▉▊▋▌▍▎▏";
|
||||
|
||||
/** An SGR escape sequence, e.g. {@code ESC[2m} (faint) or {@code ESC[0m} (reset). */
|
||||
private static final Pattern SGR = Pattern.compile("\u001b\\[([0-9;]*)m");
|
||||
|
||||
/** The SGR code that dims text — herdr's placeholder hint is drawn inside a span of this. */
|
||||
private static final String FAINT_CODE = "2";
|
||||
|
||||
/** The SGR code (or an empty code list) that clears every attribute, including faint. */
|
||||
private static final List<String> RESET_CODES = List.of("", "0");
|
||||
|
||||
/** What a box holds: nothing, unsubmitted characters, or a pane this cannot read as a box. */
|
||||
public enum State { EMPTY, DRAFT, UNREADABLE }
|
||||
|
||||
/** A box reading: its state, and how many characters it holds ({@code 0} unless {@code DRAFT}). */
|
||||
public record Reading(State state, int characters) {
|
||||
}
|
||||
|
||||
private final AgentControl agents;
|
||||
|
||||
/** Consecutive holds per target, so a box that never clears can be warned about once. */
|
||||
private final Map<String, Integer> holdStreaks = new ConcurrentHashMap<>();
|
||||
|
||||
public PromptBox(AgentControl agents) {
|
||||
this.agents = agents;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code target}'s input box is empty, so a delivery would submit only its own text.
|
||||
* {@code false} means hold and come back; it never means the delivery failed.
|
||||
*/
|
||||
public boolean clearToSubmit(String target) {
|
||||
Reading reading = inspect(target);
|
||||
if (reading.state() == State.EMPTY) {
|
||||
holdStreaks.remove(target);
|
||||
return true;
|
||||
}
|
||||
int streak = holdStreaks.merge(target, 1, Integer::sum);
|
||||
if (streak == HOLD_WARN_STREAK) {
|
||||
log.warn("prompt box of {} has held a delivery {} times in a row ({}, {} character(s) in the box)"
|
||||
+ " — nothing is lost, delivery resumes once the box is empty",
|
||||
target, streak, reading.state(), reading.characters());
|
||||
} else {
|
||||
log.debug("prompt box of {} is {} ({} character(s)), holding delivery {}",
|
||||
target, reading.state(), reading.characters(), streak);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/** Read and classify {@code target}'s pane. A read failure reads as {@link State#UNREADABLE}. */
|
||||
private Reading inspect(String target) {
|
||||
String pane;
|
||||
try {
|
||||
pane = agents.readWithStyling(target, PROBE_SOURCE);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("prompt box read for {} failed, holding delivery: {}", target, e.getMessage());
|
||||
return new Reading(State.UNREADABLE, 0);
|
||||
}
|
||||
return classify(pane);
|
||||
}
|
||||
|
||||
/**
|
||||
* Classify a Claude Code TUI pane region. Pure, so it is unit-testable without herdr.
|
||||
*
|
||||
* <p>{@link State#EMPTY} needs two positive signals: the pane's last box line holds nothing after
|
||||
* its marker, and nothing below that line says a turn is still generating. Everything else is
|
||||
* {@link State#UNREADABLE} — a blank capture, or a region with no box line at all — so a pane this
|
||||
* does not understand holds the delivery rather than guessing it is safe.
|
||||
*
|
||||
* <p>The generating marker is looked for only from the box line down. Above it is scrollback, where
|
||||
* an earlier turn's marker survives; treating that as a live turn would make {@link State#EMPTY}
|
||||
* unreachable and hold every delivery forever.
|
||||
*
|
||||
* <p>Whitespace, a trailing box border and a cursor block count as nothing. A placeholder hint —
|
||||
* text the pane draws faint, in the same spot as unsubmitted text — also counts as nothing: only
|
||||
* a character drawn outside a faint span is the operator's own typing.
|
||||
*/
|
||||
static Reading classify(String pane) {
|
||||
if (pane == null || pane.isBlank()) return new Reading(State.UNREADABLE, 0);
|
||||
int box = lastBoxLineStart(pane);
|
||||
if (box < 0) return new Reading(State.UNREADABLE, 0);
|
||||
String fromBox = pane.substring(box);
|
||||
if (fromBox.toLowerCase().contains(ACTIVE_TURN_MARKER)) return new Reading(State.UNREADABLE, 0);
|
||||
String content = boxContent(firstLine(fromBox));
|
||||
return content.isEmpty() ? new Reading(State.EMPTY, 0) : new Reading(State.DRAFT, content.length());
|
||||
}
|
||||
|
||||
/** Offset of the last line starting with a box marker, or {@code -1} if the region has none. */
|
||||
private static int lastBoxLineStart(String pane) {
|
||||
int found = -1;
|
||||
for (int start = 0; start <= pane.length(); ) {
|
||||
int end = pane.indexOf('\n', start);
|
||||
String line = pane.substring(start, end < 0 ? pane.length() : end);
|
||||
if (markerLength(line) > 0) found = start;
|
||||
if (end < 0) break;
|
||||
start = end + 1;
|
||||
}
|
||||
return found;
|
||||
}
|
||||
|
||||
/**
|
||||
* Length of the marker prefix — any leading SGR escape codes, then a box marker — this line
|
||||
* starts with, or {@code 0} if it starts with neither. The colour drawn on the caret itself
|
||||
* (e.g. an empty box's grey) sits before the marker glyph, so it must be skipped before the
|
||||
* marker can match.
|
||||
*/
|
||||
private static int markerLength(String line) {
|
||||
int skip = leadingEscapeLength(line);
|
||||
for (String marker : BOX_MARKERS) {
|
||||
if (line.startsWith(marker, skip)) return skip + marker.length();
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/** Length of the run of SGR escape codes starting at the beginning of {@code line}. */
|
||||
private static int leadingEscapeLength(String line) {
|
||||
Matcher m = SGR.matcher(line);
|
||||
int pos = 0;
|
||||
while (m.find(pos) && m.start() == pos) pos = m.end();
|
||||
return pos;
|
||||
}
|
||||
|
||||
private static String firstLine(String text) {
|
||||
int newline = text.indexOf('\n');
|
||||
return newline < 0 ? text : text.substring(0, newline);
|
||||
}
|
||||
|
||||
/** One rendered character of a box line, and whether it was drawn inside a faint (dim) span. */
|
||||
private record Glyph(char c, boolean faint) {
|
||||
}
|
||||
|
||||
/**
|
||||
* The text the box holds: its own line after the marker, with border, padding, cursor and any
|
||||
* faint (placeholder-hint) text left out — only a character drawn outside a faint span is the
|
||||
* operator's own typing.
|
||||
*/
|
||||
private static String boxContent(String boxLine) {
|
||||
List<Glyph> glyphs = renderedGlyphs(boxLine.substring(markerLength(boxLine)));
|
||||
int end = glyphs.size();
|
||||
while (end > 0 && isBoxPadding(glyphs.get(end - 1).c())) end--;
|
||||
if (end > 0 && glyphs.get(end - 1).c() == '│') end--;
|
||||
StringBuilder content = new StringBuilder();
|
||||
for (int i = 0; i < end; i++) {
|
||||
Glyph glyph = glyphs.get(i);
|
||||
if (glyph.faint() || isBoxPadding(glyph.c())) continue;
|
||||
content.append(glyph.c());
|
||||
}
|
||||
return content.toString();
|
||||
}
|
||||
|
||||
/** Decode {@code text} into its rendered characters, tracking the faint (SGR 2) span each sits in. */
|
||||
private static List<Glyph> renderedGlyphs(String text) {
|
||||
List<Glyph> glyphs = new ArrayList<>();
|
||||
Matcher m = SGR.matcher(text);
|
||||
boolean faint = false;
|
||||
int i = 0;
|
||||
while (i < text.length()) {
|
||||
if (m.find(i) && m.start() == i) {
|
||||
String codes = m.group(1);
|
||||
if (RESET_CODES.contains(codes)) faint = false;
|
||||
else if (FAINT_CODE.equals(codes)) faint = true;
|
||||
i = m.end();
|
||||
continue;
|
||||
}
|
||||
glyphs.add(new Glyph(text.charAt(i), faint));
|
||||
i++;
|
||||
}
|
||||
return glyphs;
|
||||
}
|
||||
|
||||
private static boolean isBoxPadding(char c) {
|
||||
return Character.isWhitespace(c) || Character.isSpaceChar(c) || CURSOR_GLYPHS.indexOf(c) >= 0;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,146 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.List;
|
||||
import java.util.function.IntFunction;
|
||||
|
||||
/**
|
||||
* The three protections every {@code agent.start} caller needs against herdr's pane-typed launch
|
||||
* surface (fleetd #220, #727): a byte-limit check on the assembled command line, a bounded retry
|
||||
* on {@code agent_pane_busy} (the target pane's shell has not reached its prompt yet), and a
|
||||
* bounded retry on {@code agent_name_taken} with a fresh name each attempt. One implementation —
|
||||
* every caller of {@code agent.start}, lead or member, goes through this seam rather than carrying
|
||||
* its own copy.
|
||||
*/
|
||||
public final class ResilientAgentLaunch {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ResilientAgentLaunch.class);
|
||||
|
||||
private ResilientAgentLaunch() {
|
||||
}
|
||||
|
||||
/**
|
||||
* The pty line buffer herdr types a launch command into: BSD/macOS {@code MAX_CANON}. Not a
|
||||
* fleetd choice and not configurable — see {@link #checkFits}.
|
||||
*/
|
||||
public static final int PANE_COMMAND_BYTE_LIMIT = 1024;
|
||||
|
||||
/** Per-argument allowance for the separating space and a shell quote pair fleetd cannot see. */
|
||||
private static final int QUOTING_OVERHEAD_PER_ARG = 3;
|
||||
|
||||
/** herdr rejects a duplicate agent {@code name}; a caller retries a bumped name this many times. */
|
||||
public static final int NAME_RETRIES = 8;
|
||||
|
||||
/**
|
||||
* Retries for {@code agent.start} against a seed pane whose shell has not reached its prompt
|
||||
* yet — {@code tab.create}/{@code pane.split} return as soon as the pane exists, and herdr
|
||||
* refuses to start an agent in a pane that is not "an available shell" ({@code agent_pane_busy}).
|
||||
*/
|
||||
public static final int SHELL_READY_RETRIES = 20;
|
||||
|
||||
/** Raised by {@link #checkFits} when the assembled command cannot fit the pane line. */
|
||||
public static final class TooLargeException extends RuntimeException {
|
||||
public TooLargeException(String message) {
|
||||
super(message);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Verify the assembled launch line fits the pane herdr types it into. herdr does not exec the
|
||||
* launch command — it TYPES it into the pane as one line, and a pty line buffer holds only
|
||||
* {@value #PANE_COMMAND_BYTE_LIMIT} bytes. Everything past that byte is dropped with no error
|
||||
* anywhere: herdr answers "agent started", the backend exits on the mangled argument it was
|
||||
* handed, the pane closes, and the only symptom is a readiness timeout with no reason. That is
|
||||
* how fleetd #214 broke every claude-code spawn — one 50-byte flag pushed a 978-byte command to
|
||||
* 1028, and the tail that got cut was {@code --autocompact 250000}.
|
||||
*
|
||||
* <p>So measure it here and refuse, loudly and immediately, rather than start something that
|
||||
* cannot work. The estimate is deliberately conservative: fleetd cannot see herdr's quoting, so
|
||||
* every argument is charged its own bytes plus a separator and a quote pair. An over-estimate
|
||||
* costs a clear error at a length that was already unsafe; an under-estimate would let the
|
||||
* silent truncation back in.
|
||||
*
|
||||
* @param label names the launch in the refusal message (a profile name)
|
||||
* @param argv the full argv, including the executable at index 0
|
||||
* @throws TooLargeException naming the limit, the estimate, and the longest argument
|
||||
*/
|
||||
public static void checkFits(String label, List<String> argv) {
|
||||
int bytes = 0;
|
||||
String longest = null;
|
||||
int longestBytes = 0;
|
||||
for (String arg : argv) {
|
||||
int argBytes = arg == null ? 0 : arg.getBytes(StandardCharsets.UTF_8).length;
|
||||
bytes += argBytes + QUOTING_OVERHEAD_PER_ARG;
|
||||
if (argBytes > longestBytes) {
|
||||
longestBytes = argBytes;
|
||||
longest = arg;
|
||||
}
|
||||
}
|
||||
if (bytes <= PANE_COMMAND_BYTE_LIMIT) {
|
||||
return;
|
||||
}
|
||||
String culprit = longest == null ? "<none>"
|
||||
: longest.substring(0, Math.min(longest.length(), 60)) + (longest.length() > 60 ? "…" : "");
|
||||
throw new TooLargeException(
|
||||
"launch command for " + label + " is about " + bytes + " bytes, over the "
|
||||
+ PANE_COMMAND_BYTE_LIMIT + "-byte limit of the pane line herdr types it into. "
|
||||
+ "The pty would drop the tail silently and the backend would exit on a mangled "
|
||||
+ "argument. Longest argument is " + longestBytes + " bytes: " + culprit
|
||||
+ " — move it off the command line (a file flag) or shorten it.");
|
||||
}
|
||||
|
||||
/**
|
||||
* Start an agent into {@code paneId}, retrying {@code agent_pane_busy} up to {@code retries}
|
||||
* times with {@code sleeper} run between attempts.
|
||||
*
|
||||
* @throws HerdrException the last {@code agent_pane_busy} failure once {@code retries} is
|
||||
* spent, or immediately for any other herdr failure
|
||||
*/
|
||||
public static Agent startAwaitingShellPrompt(AgentControl agents, String name, String kind,
|
||||
List<String> args, String paneId,
|
||||
int retries, Runnable sleeper) {
|
||||
HerdrException busy = null;
|
||||
for (int attempt = 0; attempt < retries; attempt++) {
|
||||
try {
|
||||
return agents.start(name, kind, args, paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_pane_busy".equals(e.code())) throw e;
|
||||
log.debug("pane {} not at its shell prompt yet, retrying agent.start", paneId);
|
||||
busy = e;
|
||||
sleeper.run();
|
||||
}
|
||||
}
|
||||
throw busy;
|
||||
}
|
||||
|
||||
/**
|
||||
* Start an agent under a freshly generated name each attempt, retrying {@code agent_name_taken}
|
||||
* up to {@code nameRetries} times — herdr refuses a duplicate {@code name} outright, so a stale
|
||||
* registry entry (a crashed session, a name the registry has not yet released) must not block a
|
||||
* legitimate relaunch. Each attempt also carries its own {@link #startAwaitingShellPrompt} retry.
|
||||
*
|
||||
* @param nameForAttempt called once per attempt (0-based) to produce that attempt's name
|
||||
* @throws HerdrException the last {@code agent_name_taken} failure once {@code nameRetries} is
|
||||
* spent, or immediately for any other herdr failure
|
||||
*/
|
||||
public static Agent startUniquelyNamed(AgentControl agents, String kind, List<String> args,
|
||||
String paneId, IntFunction<String> nameForAttempt,
|
||||
int nameRetries, int shellReadyRetries, Runnable sleeper) {
|
||||
HerdrException last = null;
|
||||
for (int attempt = 0; attempt < nameRetries; attempt++) {
|
||||
String name = nameForAttempt.apply(attempt);
|
||||
try {
|
||||
return startAwaitingShellPrompt(agents, name, kind, args, paneId,
|
||||
shellReadyRetries, sleeper);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_name_taken".equals(e.code())) throw e;
|
||||
log.debug("agent name '{}' taken, retrying", name);
|
||||
last = e;
|
||||
}
|
||||
}
|
||||
throw last;
|
||||
}
|
||||
}
|
||||
@@ -94,6 +94,17 @@ public final class Injector {
|
||||
*/
|
||||
private static final int READINESS_GRACE_POLLS = 240;
|
||||
|
||||
/**
|
||||
* How many consecutive polls a message may sit queued with no delivery attempt at all before
|
||||
* it is failed and the queue cleared — covers every reason the head of the queue is never
|
||||
* reached, including a target that stays busy ({@code working}) or unclassifiable
|
||||
* ({@code unknown}) for the whole window. Set well above an ordinary turn so a worker
|
||||
* genuinely mid-task is never cut off, and below a caller's own overall timeout so a target
|
||||
* that never frees up fails with this specific reason instead of riding out that longer wait
|
||||
* silently.
|
||||
*/
|
||||
private static final int QUEUE_WAIT_GRACE_POLLS = 4800;
|
||||
|
||||
/**
|
||||
* The single source for the injector poll cadence — how often the {@link StatusPoller} drives
|
||||
* {@link #onStatus} at. {@code Fleetd} passes this to every {@link StatusPoller} it constructs,
|
||||
@@ -121,6 +132,12 @@ public final class Injector {
|
||||
* already uses for the same purpose.
|
||||
*/
|
||||
private final LongSupplier nowMillis;
|
||||
/**
|
||||
* Mail offered to panes that collect it themselves. Owned here because this is the single
|
||||
* writer of delivery state, and the offer must be made and taken back under the same target
|
||||
* monitor that guards the queue the message is still sitting on.
|
||||
*/
|
||||
private final PaneInbox paneInbox;
|
||||
private final ConcurrentHashMap<String, Target> targets = new ConcurrentHashMap<>();
|
||||
|
||||
/** Delivery only; completion signalling is a no-op and every target is treated as available. */
|
||||
@@ -192,6 +209,7 @@ public final class Injector {
|
||||
this.ready = ready;
|
||||
this.forget = forget;
|
||||
this.nowMillis = nowMillis;
|
||||
this.paneInbox = new PaneInbox(nowMillis);
|
||||
}
|
||||
|
||||
public Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
||||
@@ -221,6 +239,7 @@ public final class Injector {
|
||||
this.ready = ready;
|
||||
this.forget = forget;
|
||||
this.nowMillis = nowMillis;
|
||||
this.paneInbox = new PaneInbox(nowMillis);
|
||||
}
|
||||
|
||||
private AgentControl agentsFor(String target) {
|
||||
@@ -296,13 +315,16 @@ public final class Injector {
|
||||
final String text;
|
||||
final TurnToken token;
|
||||
final CompletableFuture<Void> delivered;
|
||||
final long enqueuedAtMillis;
|
||||
volatile State state = State.QUEUED; // written under the owning Target monitor
|
||||
|
||||
Pending(String target, String text, TurnToken token, CompletableFuture<Void> delivered) {
|
||||
Pending(String target, String text, TurnToken token, CompletableFuture<Void> delivered,
|
||||
long enqueuedAtMillis) {
|
||||
this.target = target;
|
||||
this.text = text;
|
||||
this.token = token;
|
||||
this.delivered = delivered;
|
||||
this.enqueuedAtMillis = enqueuedAtMillis;
|
||||
}
|
||||
|
||||
String text() {
|
||||
@@ -329,10 +351,15 @@ public final class Injector {
|
||||
int unknownSincePostTurn; // the same, for the post-turn housekeeping phase (fleetd #306)
|
||||
int notReadySincePoll; // consecutive injectable samples a queued message waited on the readiness gate (CB-114)
|
||||
long notReadySinceMillis; // wall-clock time of the FIRST non-ready sample in the current notReadySincePoll streak (fleetd #501); reset alongside it
|
||||
int queueWaitSincePoll; // consecutive polls the queue has held an undelivered message with no attempt made
|
||||
boolean postTurnPending; // completion observed; adapter housekeeping has not started yet
|
||||
boolean awaitingPostTurnPickup;
|
||||
boolean postTurnObserved;
|
||||
int injectableSincePostTurnPickup;
|
||||
/** The head of {@link #queue} as offered to a mod-served pane, or {@code null}. */
|
||||
PaneInbox.Entry inboxOffer;
|
||||
/** Whether the delivery the pickup latch is waiting on was collected rather than typed. */
|
||||
boolean deliveredViaInbox;
|
||||
|
||||
synchronized void add(Pending p) {
|
||||
queue.add(p);
|
||||
@@ -349,7 +376,7 @@ public final class Injector {
|
||||
*/
|
||||
public Delivery enqueue(String target, String text, TurnToken token) {
|
||||
CompletableFuture<Void> delivered = new CompletableFuture<>();
|
||||
Pending p = new Pending(target, text, token, delivered);
|
||||
Pending p = new Pending(target, text, token, delivered, nowMillis.getAsLong());
|
||||
targets.compute(target, (_, existing) -> {
|
||||
Target t = (existing != null) ? existing : new Target();
|
||||
t.add(p); // synchronized on the Target monitor — atomic with a concurrent drop
|
||||
@@ -361,7 +388,8 @@ public final class Injector {
|
||||
/**
|
||||
* Cancel this exact queued delivery. The target monitor serializes this operation with
|
||||
* {@link #onStatus}: if delivery wins that race, this returns {@link Cancellation#DELIVERED}
|
||||
* rather than claiming the message remained queued.
|
||||
* rather than claiming the message remained queued. A message a mod-served pane has already
|
||||
* collected answers the same way, even though no poll has recorded that delivery yet.
|
||||
*/
|
||||
public Cancellation cancel(Delivery delivery) {
|
||||
Pending p = delivery.pending;
|
||||
@@ -370,7 +398,22 @@ public final class Injector {
|
||||
return cancellationOf(p);
|
||||
}
|
||||
synchronized (t) {
|
||||
if (p.state != Pending.State.QUEUED || !t.queue.remove(p)) {
|
||||
if (p.state != Pending.State.QUEUED) {
|
||||
return cancellationOf(p);
|
||||
}
|
||||
if (t.queue.peek() == p && t.inboxOffer != null) {
|
||||
// This exact entry is the one offered to a mod-served pane. The pane takes an
|
||||
// offer on its own thread, so withdraw first and then read the outcome: a taken
|
||||
// offer means the pane already holds this text, and the next poll records that
|
||||
// delivery. Cancelling it would tell the caller nothing arrived while the pane
|
||||
// acts on it.
|
||||
paneInbox.withdrawAll(p.target);
|
||||
if (t.inboxOffer.taken()) {
|
||||
return Cancellation.DELIVERED;
|
||||
}
|
||||
t.inboxOffer = null;
|
||||
}
|
||||
if (!t.queue.remove(p)) {
|
||||
return cancellationOf(p);
|
||||
}
|
||||
p.state = Pending.State.CANCELLED;
|
||||
@@ -415,6 +458,7 @@ public final class Injector {
|
||||
boolean resubmit = false;
|
||||
boolean startPostTurn = false;
|
||||
List<Pending> notReady = null; // queued messages failed because the worker never became ready
|
||||
List<Pending> queueStalled = null; // queued messages failed because the queue never drained
|
||||
synchronized (t) {
|
||||
if (status == AgentStatus.WORKING) {
|
||||
if (t.awaitingPostTurnPickup) {
|
||||
@@ -454,10 +498,14 @@ public final class Injector {
|
||||
t.injectableSincePickup = 0;
|
||||
t.awaitingCompletion = false;
|
||||
t.turnObserved = false;
|
||||
} else {
|
||||
} else if (!t.deliveredViaInbox) {
|
||||
// Delivered but still idle → the worker hasn't picked it up; the submit
|
||||
// keystroke likely raced the paste (esp. right as the TUI became ready).
|
||||
// Re-nudge Enter (CB-113) until the worker starts (WORKING) or the grace ends.
|
||||
//
|
||||
// A pane that collected the message submits it itself, and nothing was
|
||||
// typed into it. Pressing Enter there would submit whatever its operator
|
||||
// has in the prompt box instead.
|
||||
resubmit = true;
|
||||
}
|
||||
}
|
||||
@@ -482,45 +530,77 @@ public final class Injector {
|
||||
if (p != null && ready.test(target)) {
|
||||
t.notReadySincePoll = 0;
|
||||
t.notReadySinceMillis = 0;
|
||||
// fleetd #551: poll and record BEFORE the irreversible send, not after.
|
||||
// The entry comes off the queue and its state is set to ATTEMPTED here,
|
||||
// unconditionally — so a Throwable escaping the send call below (caught or
|
||||
// not) can never leave the entry QUEUED at the head of t.queue (the fleetd
|
||||
// #546 hazard, since peek() alone would let the next onStatus round re-enter
|
||||
// this block and send the same text again), and no path can write a
|
||||
// confident DELIVERED or NOT_DELIVERED before we actually know which one
|
||||
// happened.
|
||||
t.queue.poll();
|
||||
p.state = Pending.State.ATTEMPTED;
|
||||
try {
|
||||
agentsFor(target).send(target, p.text());
|
||||
p.state = Pending.State.DELIVERED;
|
||||
t.awaitingPickup = true;
|
||||
t.awaitingCompletion = true;
|
||||
t.turnObserved = false;
|
||||
t.injectableSincePickup = 0;
|
||||
sent = p;
|
||||
} catch (Throwable e) {
|
||||
// fleetd #551: leave p.state == ATTEMPTED (recorded above, before the
|
||||
// call) rather than downgrading it to NOT_DELIVERED here — reaching this
|
||||
// catch does not prove the text never reached the pane. Three of the
|
||||
// four HerdrException throw sites in HerdrCodec fire only after herdr
|
||||
// has already replied (so it processed the request), and the fourth (a
|
||||
// transport IOException) leaves it genuinely unknown whether herdr even
|
||||
// received the bytes — see #551 comment 16867. The one exception is a
|
||||
// herdr `*_not_found` error: that family is already read as "definitely
|
||||
// absent, not merely inconclusive" everywhere else in this codebase
|
||||
// (StatusPoller, AgentControl's own retry, WorkspaceControl,
|
||||
// HerdrPeerLauncher, FleetApp, ReplyPushLoop) because it means the
|
||||
// target pane/agent does not exist at all, so nothing could have been
|
||||
// pasted anywhere — #551 keeps the new state consistent with that
|
||||
// existing vocabulary rather than inventing a second one.
|
||||
if (e instanceof HerdrException he && he.code() != null
|
||||
&& he.code().endsWith("_not_found")) {
|
||||
p.state = Pending.State.NOT_DELIVERED;
|
||||
boolean modServed = paneInbox.isModServed(target);
|
||||
if (t.inboxOffer != null && !modServed) {
|
||||
// The pane stopped collecting its mail, so take the offer back and
|
||||
// fall through to the terminal route below. withdrawAll leaves an
|
||||
// entry the pane took first alone, so the branch under it still sees
|
||||
// that as the delivery it is.
|
||||
paneInbox.withdrawAll(target);
|
||||
if (!t.inboxOffer.taken()) {
|
||||
t.inboxOffer = null;
|
||||
}
|
||||
}
|
||||
if (t.inboxOffer == null && modServed) {
|
||||
t.inboxOffer = paneInbox.offer(target, p.text());
|
||||
}
|
||||
if (t.inboxOffer != null) {
|
||||
// The message stays at the head of the queue until the pane takes
|
||||
// it: nothing has reached the pane yet, so nothing may be recorded
|
||||
// as delivered and nothing may be failed.
|
||||
if (t.inboxOffer.taken()) {
|
||||
t.inboxOffer = null;
|
||||
t.queue.poll();
|
||||
p.state = Pending.State.DELIVERED;
|
||||
t.awaitingPickup = true;
|
||||
t.awaitingCompletion = true;
|
||||
t.turnObserved = false;
|
||||
t.injectableSincePickup = 0;
|
||||
t.deliveredViaInbox = true;
|
||||
sent = p;
|
||||
}
|
||||
} else {
|
||||
// fleetd #551: poll and record BEFORE the irreversible send, not after.
|
||||
// The entry comes off the queue and its state is set to ATTEMPTED here,
|
||||
// unconditionally — so a Throwable escaping the send call below (caught or
|
||||
// not) can never leave the entry QUEUED at the head of t.queue (the fleetd
|
||||
// #546 hazard, since peek() alone would let the next onStatus round re-enter
|
||||
// this block and send the same text again), and no path can write a
|
||||
// confident DELIVERED or NOT_DELIVERED before we actually know which one
|
||||
// happened.
|
||||
t.queue.poll();
|
||||
p.state = Pending.State.ATTEMPTED;
|
||||
try {
|
||||
agentsFor(target).send(target, p.text());
|
||||
p.state = Pending.State.DELIVERED;
|
||||
t.awaitingPickup = true;
|
||||
t.awaitingCompletion = true;
|
||||
t.turnObserved = false;
|
||||
t.injectableSincePickup = 0;
|
||||
t.deliveredViaInbox = false;
|
||||
sent = p;
|
||||
} catch (Throwable e) {
|
||||
// fleetd #551: leave p.state == ATTEMPTED (recorded above, before the
|
||||
// call) rather than downgrading it to NOT_DELIVERED here — reaching this
|
||||
// catch does not prove the text never reached the pane. Three of the
|
||||
// four HerdrException throw sites in HerdrCodec fire only after herdr
|
||||
// has already replied (so it processed the request), and the fourth (a
|
||||
// transport IOException) leaves it genuinely unknown whether herdr even
|
||||
// received the bytes — see #551 comment 16867. The one exception is a
|
||||
// herdr `*_not_found` error: that family is already read as "definitely
|
||||
// absent, not merely inconclusive" everywhere else in this codebase
|
||||
// (StatusPoller, AgentControl's own retry, WorkspaceControl,
|
||||
// HerdrPeerLauncher, FleetApp, ReplyPushLoop) because it means the
|
||||
// target pane/agent does not exist at all, so nothing could have been
|
||||
// pasted anywhere — #551 keeps the new state consistent with that
|
||||
// existing vocabulary rather than inventing a second one.
|
||||
if (e instanceof HerdrException he && he.code() != null
|
||||
&& he.code().endsWith("_not_found")) {
|
||||
p.state = Pending.State.NOT_DELIVERED;
|
||||
}
|
||||
sent = p;
|
||||
sendError = e;
|
||||
}
|
||||
sent = p;
|
||||
sendError = e;
|
||||
}
|
||||
} else if (p != null) {
|
||||
// fleetd #501: stamp the wall-clock time of the FIRST non-ready sample in
|
||||
@@ -539,6 +619,8 @@ public final class Injector {
|
||||
for (Pending pending : notReady) {
|
||||
pending.state = Pending.State.NOT_DELIVERED;
|
||||
}
|
||||
paneInbox.withdrawAll(target);
|
||||
t.inboxOffer = null;
|
||||
// fleetd #501: t.notReadySincePoll — the loop's own counter, already in
|
||||
// scope — is printed here instead of the READINESS_GRACE_POLLS constant.
|
||||
// On this branch the counter has JUST reached the threshold, so the two
|
||||
@@ -606,6 +688,30 @@ public final class Injector {
|
||||
}
|
||||
}
|
||||
|
||||
// A message still queued and never attempted this poll is bounded on its own, whatever
|
||||
// the reason the head of the queue was never reached — a target stuck WORKING or
|
||||
// UNKNOWN for the whole window hits this even though neither branch above ever looks at
|
||||
// the queue. Any poll that did attempt the head (`sent != null`, success or failure
|
||||
// alike) counts as progress and resets the streak, even if messages remain behind it.
|
||||
if (!t.queue.isEmpty() && sent == null) {
|
||||
if (++t.queueWaitSincePoll >= QUEUE_WAIT_GRACE_POLLS) {
|
||||
queueStalled = new ArrayList<>(t.queue);
|
||||
for (Pending pending : queueStalled) {
|
||||
pending.state = Pending.State.NOT_DELIVERED;
|
||||
}
|
||||
paneInbox.withdrawAll(target);
|
||||
t.inboxOffer = null;
|
||||
log.warn("queue for {} never drained after {} polls (limit={} polls/{}s): "
|
||||
+ "failing {} queued message(s) that were never attempted",
|
||||
target, t.queueWaitSincePoll, QUEUE_WAIT_GRACE_POLLS,
|
||||
QUEUE_WAIT_GRACE_POLLS * POLL_INTERVAL_MILLIS / 1000, queueStalled.size());
|
||||
t.queue.clear();
|
||||
t.queueWaitSincePoll = 0;
|
||||
}
|
||||
} else {
|
||||
t.queueWaitSincePoll = 0;
|
||||
}
|
||||
|
||||
// Reclaim the entry once the worker is fully quiescent (nothing queued, no pickup or
|
||||
// completion awaited), so the map cannot grow without bound across short-lived workers.
|
||||
if (isQuiescent(t)) {
|
||||
@@ -676,6 +782,19 @@ public final class Injector {
|
||||
forget.accept(target);
|
||||
turnListener.onTurnFailed(target);
|
||||
}
|
||||
if (queueStalled != null) {
|
||||
// The target is not gone — it may still be genuinely busy — so this does not call
|
||||
// forget.accept: that would clear presence/readiness state for a worker that is
|
||||
// simply taking a long turn. It still resolves the awaiting send's own waiter via
|
||||
// onTurnFailed (mirroring notReady above), so a caller learns this specific message
|
||||
// never reached the pane instead of riding out its own much longer timeout.
|
||||
RuntimeException cause = new IllegalStateException(
|
||||
target + " never freed up to receive this message within the queue wait grace");
|
||||
for (Pending p : queueStalled) {
|
||||
p.delivered().completeExceptionally(cause);
|
||||
}
|
||||
turnListener.onTurnFailed(target, cause.getMessage());
|
||||
}
|
||||
if (turnCompleted) {
|
||||
if (startPostTurn) {
|
||||
// fleetd #553: the listener call is wrapped so `t.postTurnPending` (set true inside
|
||||
@@ -795,6 +914,24 @@ public final class Injector {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Hand {@code terminal} every message held for it and stamp it as collecting its own mail.
|
||||
* While that stamp is fresh this injector offers that pane's messages instead of typing them;
|
||||
* once it goes stale the pane's queued mail takes the terminal route again.
|
||||
*
|
||||
* <p>Returns the messages in the order they were queued, and an empty list when there are
|
||||
* none — an empty collection still counts as collecting, so a pane that polls on a timer stays
|
||||
* mod-served between messages.
|
||||
*/
|
||||
public List<String> collectInbox(String terminal) {
|
||||
return paneInbox.drain(terminal);
|
||||
}
|
||||
|
||||
/** Whether {@code terminal} has collected its mail recently enough to be offered the next one. */
|
||||
public boolean isModServed(String terminal) {
|
||||
return paneInbox.isModServed(terminal);
|
||||
}
|
||||
|
||||
/**
|
||||
* Targets the poller must keep sampling: those with a queued message, an awaited pickup, or an
|
||||
* awaited turn completion (so the {@code working → idle} boundary is observed).
|
||||
@@ -812,6 +949,23 @@ public final class Injector {
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* How long the oldest still-queued, never-attempted message for {@code target} has been
|
||||
* waiting, or {@code null} when nothing is queued (including when the head has already been
|
||||
* attempted or delivered). A caller uses this to tell a message that genuinely never reached
|
||||
* the pane apart from one that was delivered and is now simply being worked on.
|
||||
*/
|
||||
public Long queuedWaitMillis(String target) {
|
||||
Target t = targets.get(target);
|
||||
if (t == null) {
|
||||
return null;
|
||||
}
|
||||
synchronized (t) {
|
||||
Pending head = t.queue.peek();
|
||||
return head != null ? nowMillis.getAsLong() - head.enqueuedAtMillis : null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Forget a target whose worker is gone, failing every still-queued message so awaiting callers
|
||||
* unblock instead of hanging forever. If a message had already been <em>delivered</em> but its
|
||||
@@ -831,6 +985,11 @@ public final class Injector {
|
||||
p.state = Pending.State.NOT_DELIVERED;
|
||||
}
|
||||
t.queue.clear();
|
||||
// The pane is gone, so drop its offered mail and its poll stamp together: a terminal
|
||||
// id can be reused, and a stale stamp would make the next pane under it look mod-served
|
||||
// before it has ever collected anything.
|
||||
paneInbox.forget(target);
|
||||
t.inboxOffer = null;
|
||||
hadDeliveredTurn = t.awaitingCompletion;
|
||||
t.awaitingCompletion = false;
|
||||
t.awaitingPickup = false;
|
||||
|
||||
@@ -4,16 +4,17 @@ import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* Tracks which workers are <em>available</em> — their Claude has booted and connected its MCP client
|
||||
* to the bridge (CB-113). This is the reliable readiness signal, unlike herdr's {@code agent_status},
|
||||
* which reports {@code idle} for a worker whose Claude is still booting. Delivering into that boot
|
||||
* window pastes into a not-yet-ready TUI (the text is lost) and wedges the worker's delivery state,
|
||||
* so the {@link Injector} holds the first delivery until the worker is present here.
|
||||
* Tracks which peers are <em>available</em> — their Claude has booted and connected its MCP
|
||||
* client to the bridge. For a spawned member this is the reliable readiness signal,
|
||||
* unlike herdr's {@code agent_status}, which reports {@code idle} while its Claude is still
|
||||
* booting. Delivering into that boot window pastes into a not-yet-ready TUI (the text is lost)
|
||||
* and wedges that member's delivery state, so the {@link Injector} holds a spawned member's
|
||||
* first delivery until it is present here.
|
||||
*
|
||||
* <p>Populated from the MCP transport: any MCP request whose connection resolves to a worker terminal
|
||||
* marks that worker present (its {@code initialize} is the first such contact). A worker that never
|
||||
* mounts the bridge MCP is never marked present — its sends stay queued until they time out, which is
|
||||
* correct (it could not have replied anyway).
|
||||
* <p>Populated from the MCP transport, for the peers whose deliverability rests on proving a live
|
||||
* MCP contact rather than on a configured registry entry. A peer that never mounts the bridge MCP
|
||||
* is never marked present — its sends stay queued until they time out, which is correct (it could
|
||||
* not have replied anyway).
|
||||
*/
|
||||
public class MemberPresence {
|
||||
|
||||
|
||||
@@ -0,0 +1,141 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import java.util.ArrayDeque;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Deque;
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
* Mail held for a pane that collects it itself instead of having it typed into its terminal.
|
||||
*
|
||||
* <p>A pane becomes <em>mod-served</em> by calling {@code fleet_inbox}: {@link #drain} stamps the
|
||||
* pane as polling, and {@link #isModServed} answers {@code true} while that stamp is younger than
|
||||
* {@link #MOD_SERVED_WINDOW_MILLIS}. Nothing else sets it, so a pane that has never polled is
|
||||
* never mod-served and its mail takes the terminal route.
|
||||
*
|
||||
* <p>An offered entry is removed exactly once, by {@link #drain} or by {@link #withdrawAll}, and
|
||||
* both run under the owning pane's monitor. So an entry the pane took is never also withdrawn, and
|
||||
* an entry that was withdrawn can never still be collected — which is what lets the {@link
|
||||
* Injector} keep one message both offered here and queued for the terminal without risking two
|
||||
* deliveries of it.
|
||||
*
|
||||
* <p>This class holds no queue of its own beyond what is currently offered: the {@link Injector}
|
||||
* keeps the message on its own queue until the pane takes it, so a pane that stops polling strands
|
||||
* nothing.
|
||||
*/
|
||||
public class PaneInbox {
|
||||
|
||||
/**
|
||||
* How long after a {@link #drain} a pane still counts as mod-served. It must exceed the mod's
|
||||
* own poll interval by enough that a few missed polls are not read as a pane that stopped,
|
||||
* while staying short enough that a pane which really stopped falls back to the terminal route
|
||||
* promptly.
|
||||
*/
|
||||
public static final long MOD_SERVED_WINDOW_MILLIS = 15_000;
|
||||
|
||||
/** One message held for a pane until that pane collects it. */
|
||||
public static final class Entry {
|
||||
private final String text;
|
||||
private boolean taken; // written under the owning pane's monitor
|
||||
|
||||
private Entry(String text) {
|
||||
this.text = text;
|
||||
}
|
||||
|
||||
/** The message text, as it will be handed to the pane. */
|
||||
public String text() {
|
||||
return text;
|
||||
}
|
||||
|
||||
/** Whether the pane has collected this entry. Once {@code true} it never goes back. */
|
||||
public synchronized boolean taken() {
|
||||
return taken;
|
||||
}
|
||||
|
||||
private synchronized void markTaken() {
|
||||
taken = true;
|
||||
}
|
||||
}
|
||||
|
||||
private final LongSupplier nowMillis;
|
||||
private final ConcurrentHashMap<String, Long> lastPolledAtMillis = new ConcurrentHashMap<>();
|
||||
private final ConcurrentHashMap<String, Deque<Entry>> offered = new ConcurrentHashMap<>();
|
||||
|
||||
public PaneInbox() {
|
||||
this(System::currentTimeMillis);
|
||||
}
|
||||
|
||||
public PaneInbox(LongSupplier nowMillis) {
|
||||
this.nowMillis = Objects.requireNonNull(nowMillis, "nowMillis");
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code terminal} collected its mail within {@link #MOD_SERVED_WINDOW_MILLIS}. A
|
||||
* terminal that has never collected any is never mod-served.
|
||||
*/
|
||||
public boolean isModServed(String terminal) {
|
||||
Long at = terminal == null ? null : lastPolledAtMillis.get(terminal);
|
||||
return at != null && nowMillis.getAsLong() - at <= MOD_SERVED_WINDOW_MILLIS;
|
||||
}
|
||||
|
||||
/** Hold {@code text} for {@code terminal} to collect, and return the entry holding it. */
|
||||
public Entry offer(String terminal, String text) {
|
||||
Entry entry = new Entry(text);
|
||||
Deque<Entry> queue = offered.computeIfAbsent(terminal, _ -> new ArrayDeque<>());
|
||||
synchronized (queue) {
|
||||
queue.add(entry);
|
||||
}
|
||||
return entry;
|
||||
}
|
||||
|
||||
/**
|
||||
* Take back every entry {@code terminal} has not collected. An entry the pane took first stays
|
||||
* taken — this never un-delivers one.
|
||||
*/
|
||||
public void withdrawAll(String terminal) {
|
||||
Deque<Entry> queue = terminal == null ? null : offered.get(terminal);
|
||||
if (queue == null) {
|
||||
return;
|
||||
}
|
||||
synchronized (queue) {
|
||||
queue.removeIf(e -> !e.taken());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Collect everything held for {@code terminal}, in the order it was offered, and stamp the pane
|
||||
* as polling. Each returned entry is marked taken, so the {@link Injector} can tell a message
|
||||
* the pane really has from one it merely offered.
|
||||
*/
|
||||
public List<String> drain(String terminal) {
|
||||
if (terminal == null || terminal.isBlank()) {
|
||||
return List.of();
|
||||
}
|
||||
lastPolledAtMillis.put(terminal, nowMillis.getAsLong());
|
||||
Deque<Entry> queue = offered.get(terminal);
|
||||
if (queue == null) {
|
||||
return List.of();
|
||||
}
|
||||
List<String> collected = new ArrayList<>();
|
||||
synchronized (queue) {
|
||||
for (Entry e : queue) {
|
||||
e.markTaken();
|
||||
collected.add(e.text());
|
||||
}
|
||||
queue.clear();
|
||||
}
|
||||
return collected;
|
||||
}
|
||||
|
||||
/** Forget a pane that is gone, so neither its poll stamp nor its offered mail lingers. */
|
||||
public void forget(String terminal) {
|
||||
if (terminal == null) {
|
||||
return;
|
||||
}
|
||||
lastPolledAtMillis.remove(terminal);
|
||||
offered.remove(terminal);
|
||||
}
|
||||
}
|
||||
@@ -68,8 +68,10 @@ import java.util.function.LongSupplier;
|
||||
* (never the whole 52 MB a long-lived transcript reaches on the host this was measured on), and
|
||||
* {@link #DEFAULT_CACHE_TTL_MILLIS} bounds how often that bounded read actually happens — a burst
|
||||
* of {@code fleet_list} calls inside one TTL window reads the file once. One instance's cache is
|
||||
* keyed by {@code (configDir, sessionId)}, so it is safe to share across every lead a single
|
||||
* {@code fleet_list} call reports on.
|
||||
* keyed by {@code (configDir, sessionId, highThreshold)}, so it is safe to share across every lead
|
||||
* a single {@code fleet_list} call reports on, and a call that resolves a different effective
|
||||
* window for the same lead never reads back a state computed against the other window's
|
||||
* threshold.
|
||||
*/
|
||||
public final class LeadContextGauge {
|
||||
|
||||
@@ -97,14 +99,19 @@ public final class LeadContextGauge {
|
||||
static final long DEFAULT_CACHE_TTL_MILLIS = 5_000;
|
||||
|
||||
/**
|
||||
* Live tokens at or above this count report {@link State#HIGH}. On the host this was measured
|
||||
* on, auto-compaction actually fires around 267,000–270,000 tokens, but the point of a HIGH
|
||||
* state is to warn before that happens, not at it — 200,000 is the standard Claude context
|
||||
* window size and a sensible built-in default: no config key is required to pick it, and a
|
||||
* lead crossing it is already deep enough into its window that a compaction is foreseeable.
|
||||
* Fallback HIGH threshold used when a caller resolves no effective auto-compact window for the
|
||||
* lead being read (see {@link #read(String, String, String, Long)}) — the built-in default so
|
||||
* no config key is required to get a warning at all.
|
||||
*/
|
||||
static final long HIGH_THRESHOLD_TOKENS = 200_000;
|
||||
|
||||
/**
|
||||
* The fraction of a resolved effective auto-compact window that HIGH warns at, so the warning
|
||||
* margin scales with the window instead of only ever meaning something against the fixed
|
||||
* {@link #HIGH_THRESHOLD_TOKENS} fallback.
|
||||
*/
|
||||
static final double HIGH_THRESHOLD_FRACTION = 2.0 / 3.0;
|
||||
|
||||
/** The only peer kind this reader understands ({@code Agent.agentType()}'s wire value). */
|
||||
private static final String CLAUDE_AGENT_TYPE = "claude";
|
||||
|
||||
@@ -165,8 +172,13 @@ public final class LeadContextGauge {
|
||||
* {@code "claude"} (including {@code null}, meaning undetected) reports
|
||||
* {@link State#UNKNOWN} — this reader only understands Claude Code's own
|
||||
* transcript format
|
||||
* @param effectiveWindowTokens the caller's resolved effective auto-compact window for this
|
||||
* lead's own profile, or {@code null} when it cannot be resolved.
|
||||
* HIGH fires at {@link #HIGH_THRESHOLD_FRACTION} of this value;
|
||||
* {@code null} (or a non-positive value) falls back to the fixed
|
||||
* {@link #HIGH_THRESHOLD_TOKENS}
|
||||
*/
|
||||
public Reading read(String configDir, String sessionId, String agentType) {
|
||||
public Reading read(String configDir, String sessionId, String agentType, Long effectiveWindowTokens) {
|
||||
if (sessionId == null || sessionId.isBlank()) {
|
||||
return Reading.unknown();
|
||||
}
|
||||
@@ -176,18 +188,27 @@ public final class LeadContextGauge {
|
||||
String base = (configDir == null || configDir.isBlank())
|
||||
? System.getProperty("user.home") + "/.claude"
|
||||
: configDir;
|
||||
String cacheKey = base + '\u0000' + sessionId;
|
||||
long highThreshold = highThreshold(effectiveWindowTokens);
|
||||
String cacheKey = base + '\u0000' + sessionId + '\u0000' + highThreshold;
|
||||
long now = clock.getAsLong();
|
||||
CacheEntry cached = cache.get(cacheKey);
|
||||
if (cached != null && now - cached.readAtMillis() < ttlMillis) {
|
||||
return cached.reading();
|
||||
}
|
||||
Reading fresh = readUncached(base, sessionId);
|
||||
Reading fresh = readUncached(base, sessionId, highThreshold);
|
||||
cache.put(cacheKey, new CacheEntry(fresh, now));
|
||||
return fresh;
|
||||
}
|
||||
|
||||
private Reading readUncached(String base, String sessionId) {
|
||||
/** {@link #HIGH_THRESHOLD_FRACTION} of {@code effectiveWindowTokens}, or the fixed fallback. */
|
||||
private static long highThreshold(Long effectiveWindowTokens) {
|
||||
if (effectiveWindowTokens == null || effectiveWindowTokens <= 0) {
|
||||
return HIGH_THRESHOLD_TOKENS;
|
||||
}
|
||||
return (long) (effectiveWindowTokens * HIGH_THRESHOLD_FRACTION);
|
||||
}
|
||||
|
||||
private Reading readUncached(String base, String sessionId, long highThreshold) {
|
||||
diskReads.incrementAndGet();
|
||||
Path file = findTranscript(base, sessionId);
|
||||
if (file == null) {
|
||||
@@ -199,7 +220,7 @@ public final class LeadContextGauge {
|
||||
} catch (IOException e) {
|
||||
return Reading.unknown();
|
||||
}
|
||||
return parse(tail);
|
||||
return parse(tail, highThreshold);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -258,7 +279,7 @@ public final class LeadContextGauge {
|
||||
* read raced — see the "torn final line" section of the class javadoc) is skipped, not fatal.
|
||||
* Only when none of the remaining lines parse does this report {@link State#UNKNOWN}.
|
||||
*/
|
||||
private Reading parse(TailRead tail) {
|
||||
private Reading parse(TailRead tail, long highThreshold) {
|
||||
String text = new String(tail.bytes(), StandardCharsets.UTF_8);
|
||||
List<String> lines = new ArrayList<>(List.of(text.split("\n", -1)));
|
||||
if (!lines.isEmpty() && lines.get(lines.size() - 1).isEmpty()) {
|
||||
@@ -299,7 +320,7 @@ public final class LeadContextGauge {
|
||||
if (tokens == null) {
|
||||
return new Reading(State.UNKNOWN, null, compactions);
|
||||
}
|
||||
State state = tokens >= HIGH_THRESHOLD_TOKENS ? State.HIGH : State.OK;
|
||||
State state = tokens >= highThreshold ? State.HIGH : State.OK;
|
||||
return new Reading(state, tokens, compactions);
|
||||
}
|
||||
|
||||
|
||||
@@ -5,6 +5,7 @@ import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.PendingCloseMarker;
|
||||
import dev.ltms.fleet.herdr.ResilientAgentLaunch;
|
||||
import dev.ltms.fleet.herdr.Tab;
|
||||
import dev.ltms.fleet.herdr.Workspace;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
@@ -12,13 +13,16 @@ import dev.ltms.fleet.launch.ClaudeCodeArguments;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.security.SecureRandom;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
|
||||
/**
|
||||
@@ -80,9 +84,19 @@ public final class LeadLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadLauncher.class);
|
||||
|
||||
/** Attempts {@link #relaunch(String)} makes before giving up and returning {@code null}. */
|
||||
static final int RELAUNCH_ATTEMPTS = 3;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final WorkspaceControl spaces;
|
||||
private final FleetConfig cfg;
|
||||
private final Runnable sleeper;
|
||||
|
||||
// Per-process token mixed into each lead agent name so a fresh daemon process (seq back at 0)
|
||||
// cannot collide with a same-name lead that outlived a restart — the same scheme
|
||||
// HerdrPeerLauncher uses for members (fleetd #727).
|
||||
private final String nameNonce = String.format("%06x", new SecureRandom().nextInt(1 << 24));
|
||||
private final AtomicLong nameSeq = new AtomicLong();
|
||||
|
||||
/**
|
||||
* @param agents herdr agent control (start, list)
|
||||
@@ -90,9 +104,27 @@ public final class LeadLauncher {
|
||||
* @param cfg the loaded config — {@code fleet.leaders}, {@code profiles} and each lead's tab
|
||||
*/
|
||||
public LeadLauncher(AgentControl agents, WorkspaceControl spaces, FleetConfig cfg) {
|
||||
this(agents, spaces, cfg, () -> sleepUninterruptibly(300));
|
||||
}
|
||||
|
||||
/**
|
||||
* Test seam: as above, plus an injectable {@code sleeper} for the {@code agent_pane_busy}
|
||||
* retry (fleetd #727), so a test can prove the retry budget without a real sleep.
|
||||
*/
|
||||
LeadLauncher(AgentControl agents, WorkspaceControl spaces, FleetConfig cfg, Runnable sleeper) {
|
||||
this.agents = agents;
|
||||
this.spaces = spaces;
|
||||
this.cfg = cfg;
|
||||
this.sleeper = sleeper;
|
||||
}
|
||||
|
||||
/** Uninterruptible sleep — the production {@link #sleeper} between {@code agent_pane_busy} retries. */
|
||||
private static void sleepUninterruptibly(long ms) {
|
||||
try {
|
||||
Thread.sleep(ms);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -173,24 +205,14 @@ public final class LeadLauncher {
|
||||
log.info("lead '{}': {} live, {} wanted — nothing to start", name, running, wanted);
|
||||
continue;
|
||||
}
|
||||
if (!lead.isCreatable()) {
|
||||
// A lead with a `tab:` but no `profile:` is recognise-only by design: the operator
|
||||
// opens it by hand. Say so once rather than looking like a silent failure.
|
||||
log.info("lead '{}' is not live, and names no profile — it can be recognised but not "
|
||||
+ "launched. Add `profile:` under fleet.leaders.{} to have fleetd start it.",
|
||||
name, name);
|
||||
continue;
|
||||
}
|
||||
|
||||
FleetConfig.Profile profile = cfg.profiles().get(lead.profile());
|
||||
if (profile == null) {
|
||||
log.warn("lead '{}' names profile '{}', which is not configured — not launching",
|
||||
name, lead.profile());
|
||||
ResolvedLead resolved = resolveLaunchable(name);
|
||||
if (resolved == null) {
|
||||
continue;
|
||||
}
|
||||
|
||||
for (int i = running; i < wanted; i++) {
|
||||
if (launch(name, lead, profile)) {
|
||||
if (launch(name, resolved.lead(), resolved.profile()) != null) {
|
||||
started++;
|
||||
}
|
||||
}
|
||||
@@ -198,17 +220,89 @@ public final class LeadLauncher {
|
||||
return started;
|
||||
}
|
||||
|
||||
/** A declared lead paired with the profile it launches on — {@link #resolveLaunchable}'s result. */
|
||||
private record ResolvedLead(FleetConfig.Leader lead, FleetConfig.Profile profile) {
|
||||
}
|
||||
|
||||
/**
|
||||
* The declared {@code Leader} and its {@code Profile} for {@code name}, read from the config
|
||||
* snapshot this launcher was constructed with.
|
||||
*
|
||||
* @return the resolved pair, or {@code null} (having logged) if {@code name} is not declared
|
||||
* under {@code fleet.leaders}, that lead names no {@code profile:} (a {@code tab:}-only,
|
||||
* recognise-only lead), or its {@code profile:} is not configured. Shared by
|
||||
* {@link #ensureLeads()} and {@link #relaunch(String)} so the three refusals and their
|
||||
* wording live in one place.
|
||||
*/
|
||||
private ResolvedLead resolveLaunchable(String name) {
|
||||
FleetConfig.Leader lead = cfg.fleet().leaders().get(name);
|
||||
if (lead == null) {
|
||||
log.warn("lead '{}' is not declared under fleet.leaders — not launching", name);
|
||||
return null;
|
||||
}
|
||||
if (!lead.isCreatable()) {
|
||||
// A lead with a `tab:` but no `profile:` is recognise-only by design: the operator
|
||||
// opens it by hand. Say so once rather than looking like a silent failure.
|
||||
log.info("lead '{}' names no profile — it can be recognised but not launched. Add "
|
||||
+ "`profile:` under fleet.leaders.{} to have fleetd start it.", name, name);
|
||||
return null;
|
||||
}
|
||||
|
||||
FleetConfig.Profile profile = cfg.profiles().get(lead.profile());
|
||||
if (profile == null) {
|
||||
log.warn("lead '{}' names profile '{}', which is not configured — not launching",
|
||||
name, lead.profile());
|
||||
return null;
|
||||
}
|
||||
return new ResolvedLead(lead, profile);
|
||||
}
|
||||
|
||||
/**
|
||||
* Start the named lead from the config snapshot this launcher was constructed with — not a
|
||||
* live read, so a lead's {@code profile:} or {@code tab:} edited in config needs a daemon
|
||||
* restart to take effect here — outside of {@link #ensureLeads()}'s {@code instances}
|
||||
* bookkeeping.
|
||||
*
|
||||
* @return the started {@link Agent}, or {@code null} if {@code name} is not declared under
|
||||
* {@code fleet.leaders}, that lead names no {@code profile:} (a {@code tab:}-only,
|
||||
* recognise-only lead), its {@code profile:} is not configured, or every attempt up to
|
||||
* {@link #RELAUNCH_ATTEMPTS} failed to start it. Never throws.
|
||||
*
|
||||
* <p>Does not count how many instances of this lead are already live. {@link #ensureLeads()}'s
|
||||
* count exists to avoid starting a second orchestrator; the caller of this method has already
|
||||
* decided to replace the lead and owns that decision.
|
||||
*
|
||||
* <p>Retries the whole launch attempt — not only the {@code agent_name_taken}/
|
||||
* {@code agent_pane_busy} cases {@link ResilientAgentLaunch} already retries inside one
|
||||
* {@code agents.start} call — up to {@link #RELAUNCH_ATTEMPTS} times, sleeping via the
|
||||
* injected sleeper between attempts, and returns the agent from the first attempt that
|
||||
* succeeds.
|
||||
*/
|
||||
public Agent relaunch(String name) {
|
||||
ResolvedLead resolved = resolveLaunchable(name);
|
||||
if (resolved == null) {
|
||||
return null;
|
||||
}
|
||||
|
||||
for (int attempt = 1; attempt <= RELAUNCH_ATTEMPTS; attempt++) {
|
||||
Agent started = launch(name, resolved.lead(), resolved.profile());
|
||||
if (started != null) {
|
||||
return started;
|
||||
}
|
||||
if (attempt < RELAUNCH_ATTEMPTS) {
|
||||
sleeper.run();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* How many live leads exist per configured name, and which of that name's labelled tabs are
|
||||
* <em>not</em> live: a running agent in a tab labelled with that lead's exact {@code tab}
|
||||
* (CB-579). Member workspaces are excluded, exactly as the scanner excludes them: a member must
|
||||
* not be counted as a lead because it happens to sit in a matching tab.
|
||||
*
|
||||
* <p>There used to be a second path here — a running agent on the terminal a
|
||||
* {@code fleet.leaders.<name>.terminal} pin named, for a lead opened and pinned by hand. That
|
||||
* pin is retired: {@code tab} is now the only field identity depends on, and {@link Agent}
|
||||
* already carries {@link Agent#tabId()} directly, so a hand-opened lead is found the same way an
|
||||
* auto-launched one is — by labelling its tab to match.
|
||||
* <em>not</em> live: a running agent in a tab, in that lead's own space, carrying one of its
|
||||
* {@link FleetConfig.Leader#acceptedLabels()}. A member sitting in the same shared workspace is
|
||||
* not counted as a lead because its tab carries a different label, not because any workspace is
|
||||
* excluded from this count. A tab matching a lead's label in a <em>different</em> space is not
|
||||
* counted either — space is the uniqueness boundary between leads.
|
||||
*
|
||||
* <p>fleetd #359 review finding 1: a labelled tab with nothing running in it is split into
|
||||
* {@code toClose} (already flagged pending-close by a previous reconcile, and still dead — two
|
||||
@@ -222,12 +316,11 @@ public final class LeadLauncher {
|
||||
}
|
||||
|
||||
private Map<String, LeadCount> countLeads(Map<String, FleetConfig.Leader> leaders) {
|
||||
// A lead and the members share ONE workspace now (the operator asked for a single "session"
|
||||
// with many tabs), so a workspace can no longer be excluded wholesale — the lead lives in the
|
||||
// member workspace by design. The sole discriminator is the exact tab label: a lead carries
|
||||
// its configured `fleet.leaders.<name>.tab` ("lead: opus"), while a member carries its
|
||||
// profile's `worker: {profile} #{n}` template. These never collide, so an exact-label match
|
||||
// separates them without needing to know which workspace anyone is in.
|
||||
// A lead and the members share ONE workspace (the operator asked for a single "session" with
|
||||
// many tabs), so a workspace can no longer be excluded wholesale — the lead lives in the
|
||||
// member workspace by design. The discriminator is the tab label together with the space: a
|
||||
// member's tab never carries one of a lead's accepted labels, and a lead's own label only
|
||||
// counts within that lead's configured space.
|
||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
||||
Set<String> flaggedTabIds = new LinkedHashSet<>();
|
||||
for (Workspace ws : spaces.listWorkspaces()) {
|
||||
@@ -235,7 +328,7 @@ public final class LeadLauncher {
|
||||
continue;
|
||||
}
|
||||
for (Tab tab : spaces.listTabs(ws.workspaceId())) {
|
||||
String declared = leadNameOf(tab.label(), leaders);
|
||||
String declared = leadNameOf(tab.label(), ws.label(), leaders);
|
||||
if (declared != null && tab.tabId() != null) {
|
||||
nameByTab.put(tab.tabId(), declared);
|
||||
if (PendingCloseMarker.isFlagged(tab.label())) {
|
||||
@@ -285,29 +378,41 @@ public final class LeadLauncher {
|
||||
}
|
||||
|
||||
/**
|
||||
* The configured lead a tab label names, or {@code null} for a label that names none.
|
||||
* The configured lead a tab names, or {@code null} for a label or space that names none.
|
||||
*
|
||||
* <p>Matched exactly (case-insensitively) against each lead's configured {@code tab}, so an
|
||||
* operator's {@code "lead: something-else"} tab is not mistaken for a configured lead. A
|
||||
* trailing {@link PendingCloseMarker} is stripped first, so a tab this class flagged on a
|
||||
* previous reconcile is still recognised as the same lead's tab on this one.
|
||||
* <p>A match requires both: the label (case-insensitively, trailing {@link PendingCloseMarker}
|
||||
* stripped) must be one of the lead's {@link FleetConfig.Leader#acceptedLabels()}, and {@code
|
||||
* space} must be that lead's own {@link FleetConfig.Leader#workspace()}. The same label in a
|
||||
* different space names no lead — space is the uniqueness boundary between leads.
|
||||
*/
|
||||
private String leadNameOf(String label, Map<String, FleetConfig.Leader> leaders) {
|
||||
if (label == null) {
|
||||
private String leadNameOf(String label, String space, Map<String, FleetConfig.Leader> leaders) {
|
||||
if (label == null || space == null) {
|
||||
return null;
|
||||
}
|
||||
String l = PendingCloseMarker.strip(label);
|
||||
String l = PendingCloseMarker.strip(label).toLowerCase(Locale.ROOT);
|
||||
for (Map.Entry<String, FleetConfig.Leader> e : leaders.entrySet()) {
|
||||
String tab = e.getValue().tabLabel();
|
||||
if (tab != null && l.equalsIgnoreCase(tab.strip())) {
|
||||
FleetConfig.Leader lead = e.getValue();
|
||||
if (lead == null || !lead.workspace().equalsIgnoreCase(space)) {
|
||||
continue;
|
||||
}
|
||||
if (lead.acceptedLabels().contains(l)) {
|
||||
return e.getKey();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Start one lead. Returns false (having logged) rather than throwing on any failure. */
|
||||
private boolean launch(String name, FleetConfig.Leader lead, FleetConfig.Profile profile) {
|
||||
/**
|
||||
* Start one lead. Returns null (having logged) rather than throwing on any failure.
|
||||
*
|
||||
* <p>Goes through the same {@link ResilientAgentLaunch} seam every member spawn uses
|
||||
* (fleetd #727): the assembled argv is refused outright if it cannot fit the pane line herdr
|
||||
* types it into, a stale {@code agent_name_taken} (a crashed session's name the registry has
|
||||
* not yet released) is retried under a fresh per-attempt name rather than refusing the whole
|
||||
* relaunch, and a seed pane whose shell has not reached its prompt yet ({@code
|
||||
* agent_pane_busy}) is retried rather than failing on the first miss.
|
||||
*/
|
||||
private Agent launch(String name, FleetConfig.Leader lead, FleetConfig.Profile profile) {
|
||||
String label = lead.tabLabel();
|
||||
String cwd = (lead.cwd() == null || lead.cwd().isBlank())
|
||||
? System.getProperty("user.dir") : lead.cwd();
|
||||
@@ -323,8 +428,12 @@ public final class LeadLauncher {
|
||||
// Same shape as the member launchers: herdr resolves the executable from `kind`, so
|
||||
// argv[0] (the configured launcher, e.g. `ccs`) is dropped and only the rest is passed.
|
||||
List<String> argv = leadArgv(profile);
|
||||
Agent started = agents.start("lead-" + name, herdrKind(profile),
|
||||
argv.isEmpty() ? argv : argv.subList(1, argv.size()), tab.rootPaneId());
|
||||
ResilientAgentLaunch.checkFits(profile.profile(), argv);
|
||||
List<String> args = argv.isEmpty() ? argv : argv.subList(1, argv.size());
|
||||
Agent started = ResilientAgentLaunch.startUniquelyNamed(agents, herdrKind(profile), args,
|
||||
tab.rootPaneId(),
|
||||
attempt -> "lead-" + name + "-" + nameNonce + "-" + nameSeq.incrementAndGet(),
|
||||
ResilientAgentLaunch.NAME_RETRIES, ResilientAgentLaunch.SHELL_READY_RETRIES, sleeper);
|
||||
|
||||
// Label AFTER the start succeeds. A label written before would survive a failed start
|
||||
// and then read back as a live lead on the next boot, which is the exact staleness the
|
||||
@@ -334,7 +443,7 @@ public final class LeadLauncher {
|
||||
log.info("lead '{}' launched: profile={} tab={} pane={} terminal={} label='{}' cwd={}",
|
||||
name, profile.profile(), tab.tab().tabId(), started.paneId(),
|
||||
started.terminalId(), label, cwd);
|
||||
return true;
|
||||
return started;
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("lead '{}' failed to launch on profile '{}': {}",
|
||||
name, profile.profile(), e.getMessage());
|
||||
@@ -346,7 +455,7 @@ public final class LeadLauncher {
|
||||
tab.tab().tabId(), cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
return false;
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,8 +1,11 @@
|
||||
package dev.ltms.fleet.lead;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -24,64 +27,73 @@ import java.util.function.Supplier;
|
||||
/**
|
||||
* fleetd #480: replace a lead session that has decided it is ready to be rolled over, without an
|
||||
* operator doing it by hand. A lead writes a handover file, calls {@link #open}, and then — once
|
||||
* every gate ({@link #confirm}'s own checks) has passed — a deferred, single-shot continuation
|
||||
* clears the lead's own pane and bootstraps a fresh session against that file.
|
||||
* every gate ({@link #confirm}'s own checks) has passed — a deferred, single-shot continuation ends
|
||||
* the lead's own pane, launches a fresh one, and bootstraps that fresh session against the file.
|
||||
*
|
||||
* <p>This is the executor behind the {@code fleet_handover} MCP tool ({@code
|
||||
* dev.ltms.fleet.mcp.FleetMcp#handover}), which drives {@link #open}, {@link #confirm}, {@link
|
||||
* #cancel}, and {@link #status} from a tool call — wired in fleetd #480 Unit C. <strong>An earlier
|
||||
* version of this paragraph said nothing called this class at all; that stopped being true once
|
||||
* that unit landed, and this correction exists so the javadoc does not go on claiming it.</strong>
|
||||
* #cancel}, and {@link #status} from a tool call.
|
||||
*
|
||||
* <p><strong>{@code confirm()} cannot roll inline — a fleetd #480 correction.</strong> The first
|
||||
* version of this class called {@code agents.send(lead, "/clear")} directly from inside {@code
|
||||
* confirm()}, then polled for the pane to become injectable again. That is wrong, because {@code
|
||||
* confirm()} is called BY the lead, FROM the lead's own turn: the lead's pane is {@code WORKING}
|
||||
* for the whole duration of that call and cannot possibly report injectable until {@code confirm()}
|
||||
* itself returns. The poll always timed out — but only after the {@code /clear} had already been
|
||||
* sent and queued in the pane, where it fired the instant the turn ended anyway. The result was the
|
||||
* worst outcome this feature can produce: a silently destroyed lead context with no fresh session
|
||||
* ever started, and a refusal return value that claimed nothing had happened.
|
||||
*
|
||||
* <p>The fix: {@link #confirm} validates every gate, then does no I/O against the lead's own pane
|
||||
* at all — it only records that the request is approved and hands a one-shot continuation to
|
||||
* {@code continuationRunner} before returning. That continuation is what actually touches the pane,
|
||||
* once the calling turn has ended, in this order:
|
||||
* <p><strong>{@code confirm()} cannot roll inline.</strong> {@code confirm()} is called BY the
|
||||
* lead, FROM the lead's own turn: the lead's pane is {@code WORKING} for the whole duration of that
|
||||
* call and cannot possibly report a real turn boundary until {@code confirm()} itself returns. So
|
||||
* {@link #confirm} validates every gate, then does no I/O against the lead's own pane at all — it
|
||||
* only records that the request is approved and hands a one-shot continuation to {@code
|
||||
* continuationRunner} before returning. That continuation is what actually touches the pane, once
|
||||
* the calling turn has ended, in this order:
|
||||
* <ol>
|
||||
* <li>wait for the lead's own pane to report a real turn boundary — {@code IDLE} or {@code
|
||||
* DONE}, never merely {@code BLOCKED} — i.e. wait for the very {@code confirm()} call that
|
||||
* approved this roll to finish its turn — bounded by {@code turnSettleSeconds}. <strong>If
|
||||
* this never happens, nothing else in this list runs: no {@code /clear} is ever sent.</strong>
|
||||
* A lead that never goes idle is a lead still doing real work, and clearing it would throw
|
||||
* away live context — exactly the failure this correction exists to prevent.</li>
|
||||
* <li>{@code agents.send(lead, "/clear")}</li>
|
||||
* <li>wait for {@code /clear} to be picked up and settle, bounded by {@code clearSettleSeconds}
|
||||
* (fleetd #489: no longer a plain re-check of the same boundary — {@code /clear} starts no
|
||||
* turn of its own, so this instead nudges the submit keystroke while no pickup has been seen,
|
||||
* then waits for a real {@code WORKING} → {@code IDLE}/{@code DONE} boundary once one has;
|
||||
* see {@link #waitForClearPickupAndSettle})</li>
|
||||
* <li>{@code agents.send(lead, cfg.bootstrapTextFor(p.handoverPath()))}</li>
|
||||
* this never happens, nothing else in this list runs: the old pane is never touched.</strong>
|
||||
* A lead that never goes idle is a lead still doing real work, and tearing it down would throw
|
||||
* away live context.</li>
|
||||
* <li>capture the old pane id (and, through it, the old tab) from {@link AgentControl#get}, with
|
||||
* a bounded retry — the terminal-to-pane lookup it goes through can itself report a genuinely
|
||||
* live agent as not found (see {@code AgentControl#agentCall}'s own re-resolve-once
|
||||
* behaviour), and one false negative here must not abort an otherwise-healthy roll. Neither id
|
||||
* is ever re-resolved from the terminal again after this — once the pane below is closed there
|
||||
* is nothing left to resolve it from.</li>
|
||||
* <li>resolve the lead's configured name from its terminal, for the relaunch step below.</li>
|
||||
* <li>end the old session: close the pane (an already-gone pane counts as success; any other
|
||||
* failure propagates), then close its tab only when the pane was that tab's sole occupant —
|
||||
* the same pane-then-tab teardown {@code HerdrPeerLauncher#stop} uses for a member.</li>
|
||||
* <li>confirm the old pane is actually gone by polling {@link
|
||||
* dev.ltms.fleet.herdr.WorkspaceControl#locatePane} for a {@code null} result — never {@link
|
||||
* AgentControl#status}, and never the live-lead terminal map, each of which answers a
|
||||
* different question. <strong>If the old pane is never confirmed gone, no relaunch is
|
||||
* attempted</strong> — see {@link RollState#OLD_PANE_NEVER_DIED}.</li>
|
||||
* <li>launch a fresh lead with {@code LeadLauncher#relaunch}. <strong>If every attempt fails,
|
||||
* {@code bootstrapText} is never sent</strong> — see {@link RollState#RELAUNCH_FAILED}.</li>
|
||||
* <li>wait for the fresh pane to reach a real turn boundary ({@code IDLE} or {@code DONE},
|
||||
* never merely {@code BLOCKED}), bounded by {@code relaunchReadySeconds}. This is the
|
||||
* safety gate: typing into a pane that has not actually finished booting loses the
|
||||
* keystrokes. <strong>If the pane never becomes ready, {@code bootstrapText} is never
|
||||
* sent</strong> — see {@link RollState#RELAUNCH_NEVER_READY}.</li>
|
||||
* <li>wait for the fresh terminal to be recognised as a live lead — present in the live-lead
|
||||
* terminal map — bounded by {@code relaunchReadySeconds}. This is bookkeeping, not a
|
||||
* safety gate: {@code bootstrapText} is sent either way once the pane is ready, whether or
|
||||
* not this wait itself times out — see {@link RollState#RELAUNCH_NOT_RECOGNISED}.</li>
|
||||
* <li>{@code agents.send(newTerminal, cfg.bootstrapTextFor(p.handoverPath()))} — sent to the
|
||||
* FRESH terminal, never the one that was just torn down.</li>
|
||||
* </ol>
|
||||
* A {@link #confirm} that returns {@link RollDecision#approved()} therefore means <em>"every gate
|
||||
* passed and the roll is scheduled"</em>, never <em>"the pane has been cleared"</em> — the pane may
|
||||
* still be mid-turn, possibly for a long time, when the caller gets that answer back.
|
||||
* passed and the roll is scheduled"</em>, never <em>"the lead has already been replaced"</em> — the
|
||||
* old pane may still be mid-turn, possibly for a long time, when the caller gets that answer back.
|
||||
*
|
||||
* <p><strong>The safety invariant survives this change, restated precisely.</strong> The ticket
|
||||
* that first defined this class required "no timer, no scheduler, no background thread" so that
|
||||
* nothing but an explicit {@link #confirm} call could ever cause a {@code /clear}. That invariant
|
||||
* is about INITIATIVE, not about synchronicity, and this correction keeps it: {@code
|
||||
* continuationRunner} launches a single-shot task that exists only because one specific,
|
||||
* <p><strong>The safety invariant.</strong> "No timer, no scheduler, no background thread" means
|
||||
* that nothing but an explicit {@link #confirm} call can ever tear a lead's pane down.
|
||||
* {@code continuationRunner} launches a single-shot task that exists only because one specific,
|
||||
* already-approved {@link #confirm} call created it — it is not recurring, it is not started at
|
||||
* construction time or on any schedule, and no two invocations of it ever share state. A recurring
|
||||
* heartbeat or timer that could decide on its own initiative to roll a pane is still, and will
|
||||
* always be, absent from this class. <strong>Nothing but an explicit {@link #confirm} call that
|
||||
* passes every gate can ever cause a {@code /clear} — that call may simply finish its own work
|
||||
* slightly later than the method return, as a continuation of the same approved request, rather
|
||||
* than entirely inside the method body.</strong>
|
||||
* heartbeat or timer that could decide on its own initiative to roll a pane is absent from this
|
||||
* class. <strong>Nothing but an explicit {@link #confirm} call that passes every gate can ever tear
|
||||
* a pane down — that call may simply finish its own work slightly later than the method return, as
|
||||
* a continuation of the same approved request, rather than entirely inside the method body.</strong>
|
||||
*
|
||||
* <p><strong>Identity is resolved by the caller, never looked up here — a second fleetd #480
|
||||
* correction.</strong> The first version resolved the pane to clear via {@code
|
||||
* PrimaryRegistry#primaryTerminal()}. That is correct for a background loop with no caller (see
|
||||
* PrimaryRegistry#currentPrimaryTerminal()}. That is correct for a background loop with no caller (see
|
||||
* {@code dev.ltms.fleet.msg.LeadHeartbeatLoop}), but wrong here and a violation of this project's
|
||||
* own charter invariant 3 — "identity comes from the connection, never an argument." This daemon
|
||||
* can hold more than one labelled lead tab (see {@code LeadLauncher}'s fleetd #359 two-reading
|
||||
@@ -115,19 +127,17 @@ public final class LeadRollover {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadRollover.class);
|
||||
|
||||
/** Poll interval while waiting for the lead's pane to settle after {@code /clear}. */
|
||||
static final long SETTLE_POLL_MS = 250;
|
||||
/** Poll interval shared by every bounded wait in this class. */
|
||||
static final long POLL_INTERVAL_MS = 250;
|
||||
|
||||
/**
|
||||
* How many consecutive not-yet-picked-up polls {@link #waitForClearPickupAndSettle} allows
|
||||
* before releasing rather than wedging the roll — the same constant and the same
|
||||
* release-not-wedge choice {@link dev.ltms.fleet.inject.Injector} already makes for its own
|
||||
* post-turn {@code /clear} housekeeping (fleetd #306). <strong>This bounds the number of
|
||||
* consecutive polls, not the number of nudges:</strong> the first {@code PICKUP_GRACE_POLLS - 1}
|
||||
* of those polls each send a nudge, and the {@code PICKUP_GRACE_POLLS}th releases instead of
|
||||
* nudging again — so 8 polls produce 7 nudges, not 8.
|
||||
* How long {@link #waitUntilPaneGone} polls {@link WorkspaceControl#locatePane} before giving
|
||||
* up on ever seeing the old pane disappear. Not configurable: once {@link #endOldSession} has
|
||||
* closed the pane (and, usually, its tab), herdr dropping the pane from its own bookkeeping is
|
||||
* expected to show up within one or two polls, not on an operator-tunable timescale the way a
|
||||
* CLI boot is.
|
||||
*/
|
||||
static final int PICKUP_GRACE_POLLS = 8;
|
||||
static final int PANE_DEATH_TIMEOUT_SECONDS = 10;
|
||||
|
||||
/**
|
||||
* One request opened by {@link #open}, pending its {@link #confirm} (or {@link #cancel}).
|
||||
@@ -139,9 +149,17 @@ public final class LeadRollover {
|
||||
* calling lead's workspace, before storing it here; see this class's
|
||||
* javadoc. This is the value the MCP layer hands back to the lead as
|
||||
* "write your file here", so callers may rely on it always being absolute.
|
||||
* @param rolloverKey the key {@link #confirm}'s single-flight claim is taken under. {@link
|
||||
* #open} resolves this once, here, from {@code leadTerminal} while the
|
||||
* calling lead is certainly still live: the lead's configured name, or
|
||||
* {@code leadTerminal} itself when no name resolves. Carried rather than
|
||||
* recomputed at release time, because by the time a roll's continuation
|
||||
* releases its claim the OLD terminal may no longer resolve to any name at
|
||||
* all — recomputing there would release a different key than the one the
|
||||
* claim was taken under.
|
||||
*/
|
||||
public record PendingRollover(String token, String leadTerminal, String handoverPath,
|
||||
long requestedAtMillis) {}
|
||||
long requestedAtMillis, String rolloverKey) {}
|
||||
|
||||
/** Which check refused a {@link #confirm} call, named so a caller can act on it. */
|
||||
public enum RefusalReason {
|
||||
@@ -164,15 +182,21 @@ public final class LeadRollover {
|
||||
* The handover file's modified time is not after {@link #open}'s request timestamp, or is
|
||||
* older than {@code maxDocAgeSeconds}.
|
||||
*/
|
||||
HANDOVER_STALE
|
||||
HANDOVER_STALE,
|
||||
/**
|
||||
* This lead already has a roll running: an earlier {@link #confirm} call claimed its
|
||||
* single-flight key (see {@link PendingRollover#rolloverKey}) and that roll's continuation
|
||||
* has not released it yet. {@code detail} names the key and the token that holds the claim.
|
||||
*/
|
||||
ROLL_ALREADY_RUNNING
|
||||
}
|
||||
|
||||
/**
|
||||
* The outcome of a {@link #confirm} call. {@link #approved()} means every gate passed and the
|
||||
* roll has been handed to a one-shot continuation — <strong>not</strong> that the pane has been
|
||||
* cleared; the continuation may still be waiting for the calling turn to end when this returns.
|
||||
* Whether the deferred roll itself later goes on to clear the pane, refuse for never going
|
||||
* idle, or refuse for never re-settling after {@code /clear} is logged only (see this class's
|
||||
* roll has been handed to a one-shot continuation — <strong>not</strong> that the lead has
|
||||
* already been replaced; the continuation may still be waiting for the calling turn to end when
|
||||
* this returns. Whether the deferred roll itself later goes on to tear the old pane down and
|
||||
* relaunch the lead, or refuses at any of its own steps, is logged only (see this class's
|
||||
* javadoc) — there is deliberately no synchronous caller left by that point to hand a result to.
|
||||
*/
|
||||
public record RollDecision(boolean accepted, RefusalReason reason, String detail) {
|
||||
@@ -208,7 +232,7 @@ public final class LeadRollover {
|
||||
|
||||
/**
|
||||
* What is known about one token, right now — the answer {@link #status} gives. Distinguishes
|
||||
* three terminal outcomes an approved roll can finish with, one in-flight outcome for a roll
|
||||
* five terminal outcomes an approved roll can finish with, one in-flight outcome for a roll
|
||||
* that has been approved but has not finished yet, and two answers for a token that names no
|
||||
* active work at all: still pending confirmation, or nothing known about this token at all.
|
||||
*/
|
||||
@@ -231,41 +255,64 @@ public final class LeadRollover {
|
||||
* #status} could wrongly answer {@link #UNKNOWN} ("nothing was ever requested") for a roll
|
||||
* that is, in fact, actively running. This is not sticky: the deferred continuation
|
||||
* overwrites this same entry with a terminal state ({@link #ROLLED}, {@link
|
||||
* #TURN_NEVER_SETTLED}, {@link #CLEAR_NEVER_SETTLED}, or {@link #FAILED}) once it finishes
|
||||
* — including by throwing, which fleetd #615's catch in {@link #runRollover} now turns into
|
||||
* {@link #FAILED} instead of leaving this entry stuck forever.
|
||||
* #TURN_NEVER_SETTLED}, {@link #OLD_PANE_NEVER_DIED}, {@link #RELAUNCH_FAILED}, {@link
|
||||
* #RELAUNCH_NEVER_READY}, {@link #RELAUNCH_NOT_RECOGNISED}, or {@link #FAILED}) once it
|
||||
* finishes — including by throwing, which {@link #runRollover}'s catch turns into {@link
|
||||
* #FAILED} instead of leaving this entry stuck forever.
|
||||
*/
|
||||
IN_PROGRESS,
|
||||
/**
|
||||
* {@link #confirm} was approved and the deferred continuation completed the entire roll:
|
||||
* the calling lead's turn settled, {@code /clear} was sent and settled, and {@code
|
||||
* bootstrapText} was sent.
|
||||
* {@link #confirm} was approved and the deferred continuation completed the entire roll: the
|
||||
* calling lead's turn settled, the old pane was torn down and confirmed gone, a fresh lead
|
||||
* was launched and recognised, and {@code bootstrapText} was sent to it.
|
||||
*/
|
||||
ROLLED,
|
||||
/**
|
||||
* {@link #confirm} was approved, but the calling lead's own turn never reached a boundary
|
||||
* (IDLE or DONE) within {@code turnSettleSeconds} — no {@code /clear} was ever sent, at
|
||||
* all. This is the branch the fleetd #480 correction exists to make safe, and the one this
|
||||
* status exists to make VISIBLE: before this, a lead that hit this case had no way to find
|
||||
* out, and would carry on believing it was about to be replaced. See this class's javadoc.
|
||||
* (IDLE or DONE) within {@code turnSettleSeconds} — the old pane was never touched at all.
|
||||
* This is the state that makes a lead's own stuck turn VISIBLE: without it, a lead that hit
|
||||
* this case would have no way to find out, and would carry on believing it was about to be
|
||||
* replaced. See this class's javadoc.
|
||||
*/
|
||||
TURN_NEVER_SETTLED,
|
||||
/**
|
||||
* {@link #confirm} was approved and {@code /clear} was sent, but the pane never re-settled
|
||||
* within {@code clearSettleSeconds} — {@code bootstrapText} was never sent.
|
||||
* {@link #confirm} was approved and the calling lead's turn settled, the old pane was closed
|
||||
* (and its tab, if it was the sole occupant), but {@link
|
||||
* dev.ltms.fleet.herdr.WorkspaceControl#locatePane} kept reporting it as still present for
|
||||
* the whole pane-death timeout. No relaunch was ever attempted, and {@code bootstrapText}
|
||||
* was never sent.
|
||||
*/
|
||||
CLEAR_NEVER_SETTLED,
|
||||
OLD_PANE_NEVER_DIED,
|
||||
/**
|
||||
* fleetd #615: the deferred continuation threw a {@link RuntimeException} — most likely a
|
||||
* {@link dev.ltms.fleet.herdr.HerdrException} out of one of the two unwrapped {@code
|
||||
* agents.send} calls in {@link #runRollover} — and the continuation thread died with it.
|
||||
* Before this state existed, that throw left {@link #outcomes} holding {@link #IN_PROGRESS}
|
||||
* forever, because the production {@code continuationRunner} is a bare virtual thread with
|
||||
* no uncaught-exception handler and nothing downstream of the throw ever ran to write a
|
||||
* terminal outcome. {@code detail} names the exception, so a reader has something to act on
|
||||
* — the same diagnostic style as {@link #TURN_NEVER_SETTLED} and {@link
|
||||
* #CLEAR_NEVER_SETTLED}. The roll is dead at this point and does not retry itself; a stuck
|
||||
* lead must {@link #open} a fresh request.
|
||||
* The old pane was confirmed gone, but {@code LeadLauncher#relaunch} returned {@code null}
|
||||
* — every launch attempt failed. {@code bootstrapText} was never sent, and no fresh terminal
|
||||
* exists for this roll to have recognised.
|
||||
*/
|
||||
RELAUNCH_FAILED,
|
||||
/**
|
||||
* A fresh lead was launched, but its pane never reached a real turn boundary ({@code IDLE}
|
||||
* or {@code DONE}, never merely {@code BLOCKED}) within {@code relaunchReadySeconds} — the
|
||||
* CLI never finished booting, or it stayed paused on a startup prompt. {@code bootstrapText}
|
||||
* was never sent: typing into a pane that is not actually ready to accept input loses the
|
||||
* keystrokes.
|
||||
*/
|
||||
RELAUNCH_NEVER_READY,
|
||||
/**
|
||||
* A fresh lead was launched and its pane reached a real turn boundary, so {@code
|
||||
* bootstrapText} WAS sent to it, but the terminal was never recognised as a live lead —
|
||||
* present in the live-lead terminal map — within {@code relaunchReadySeconds}. The session
|
||||
* itself is alive and bootstrapped; only the daemon's own bookkeeping has not caught up, and
|
||||
* an operator should check why the tab was not recognised.
|
||||
*/
|
||||
RELAUNCH_NOT_RECOGNISED,
|
||||
/**
|
||||
* The deferred continuation threw a {@link RuntimeException} and the continuation thread
|
||||
* died with it. Without this state, that throw would leave {@link #outcomes} holding {@link
|
||||
* #IN_PROGRESS} forever, because the production {@code continuationRunner} is a bare virtual
|
||||
* thread with no uncaught-exception handler and nothing downstream of the throw ever runs to
|
||||
* write a terminal outcome. {@code detail} names the exception, so a reader has something to
|
||||
* act on. The roll is dead at this point and does not retry itself; a stuck lead must
|
||||
* {@link #open} a fresh request.
|
||||
*/
|
||||
FAILED,
|
||||
/**
|
||||
@@ -286,6 +333,10 @@ public final class LeadRollover {
|
||||
public record RollStatus(RollState state, String detail) {}
|
||||
|
||||
private final AgentControl agents;
|
||||
/** Workspace/tab/pane control — used to tear down the old pane and confirm it is gone. */
|
||||
private final WorkspaceControl spaces;
|
||||
/** Starts the fresh lead that replaces the one this roll tears down. */
|
||||
private final LeadLauncher launcher;
|
||||
private final Supplier<FleetConfig.LeadRollover> configSupplier;
|
||||
/**
|
||||
* Terminal id → that lead's configured workspace directory (their {@code
|
||||
@@ -295,8 +346,24 @@ public final class LeadRollover {
|
||||
* daemon-cwd bug this parameter exists to fix.
|
||||
*/
|
||||
private final Function<String, String> leadWorkspace;
|
||||
/**
|
||||
* Terminal id → that lead's configured name under {@code fleet.leaders}, or {@code null} when
|
||||
* the terminal names no currently-recognised lead. {@link #open} calls this on the calling
|
||||
* lead's own terminal, while it is certainly still live, to resolve {@link
|
||||
* PendingRollover#rolloverKey}. The deferred continuation also calls this, on the OLD terminal,
|
||||
* before tearing it down, so it knows which lead to pass to {@link LeadLauncher#relaunch} — by
|
||||
* that point the live roster may no longer contain the old terminal, so this lookup can return
|
||||
* {@code null} here even though {@link #open}'s earlier call against the same terminal did not.
|
||||
*/
|
||||
private final Function<String, String> leadNameForTerminal;
|
||||
/**
|
||||
* The daemon's current terminal id → lead name map, read fresh on every poll. The deferred
|
||||
* continuation polls this for the FRESH terminal {@link LeadLauncher#relaunch} returns, to
|
||||
* learn when that terminal has been recognised as a live lead — see this class's javadoc.
|
||||
*/
|
||||
private final Supplier<Map<String, String>> liveLeadTerminals;
|
||||
private final LongSupplier nowMillis;
|
||||
private final Runnable settleSleeper;
|
||||
private final Runnable pollSleeper;
|
||||
/**
|
||||
* Launches the post-{@code confirm()} continuation. Production uses a single unstarted virtual
|
||||
* thread per confirmed request — see this class's javadoc for why that is a single-shot task,
|
||||
@@ -305,6 +372,15 @@ public final class LeadRollover {
|
||||
*/
|
||||
private final Consumer<Runnable> continuationRunner;
|
||||
private final Map<String, PendingRollover> pending = new ConcurrentHashMap<>();
|
||||
/**
|
||||
* {@link PendingRollover#rolloverKey} → the token of the roll currently holding that key
|
||||
* exclusive, for {@link #confirm}'s single-flight claim. {@link #confirm} claims an entry here
|
||||
* with an atomic put-if-absent once every other gate has passed, refusing with {@link
|
||||
* RefusalReason#ROLL_ALREADY_RUNNING} when a claim is already held; {@link #runRollover}
|
||||
* releases it in a {@code finally}, on both the success and the thrown-exception path. A key
|
||||
* absent from this map has no roll currently in flight for it.
|
||||
*/
|
||||
private final Map<String, String> rollingByLead = new ConcurrentHashMap<>();
|
||||
/**
|
||||
* Finished tokens → what actually happened, for {@link #status}. Bounded by {@link
|
||||
* #OUTCOME_HISTORY_CAP}, oldest evicted first ({@code removeEldestEntry} on an insertion-order
|
||||
@@ -323,29 +399,40 @@ public final class LeadRollover {
|
||||
}
|
||||
});
|
||||
|
||||
/** Production constructor — wall clock, real sleep between settle polls, a real virtual thread. */
|
||||
public LeadRollover(AgentControl agents, Supplier<FleetConfig.LeadRollover> configSupplier,
|
||||
Function<String, String> leadWorkspace) {
|
||||
this(agents, configSupplier, leadWorkspace, System::currentTimeMillis,
|
||||
() -> sleepUninterruptibly(SETTLE_POLL_MS),
|
||||
/** Production constructor — wall clock, real sleep between polls, a real virtual thread. */
|
||||
public LeadRollover(AgentControl agents, WorkspaceControl spaces, LeadLauncher launcher,
|
||||
Supplier<FleetConfig.LeadRollover> configSupplier,
|
||||
Function<String, String> leadWorkspace,
|
||||
Function<String, String> leadNameForTerminal,
|
||||
Supplier<Map<String, String>> liveLeadTerminals) {
|
||||
this(agents, spaces, launcher, configSupplier, leadWorkspace, leadNameForTerminal,
|
||||
liveLeadTerminals, System::currentTimeMillis,
|
||||
() -> sleepUninterruptibly(POLL_INTERVAL_MS),
|
||||
r -> Thread.ofVirtual().name("lead-rollover-continuation-").start(r));
|
||||
}
|
||||
|
||||
/**
|
||||
* Full constructor — an injectable wall-clock supplier, settle-poll sleeper, and continuation
|
||||
* runner, for tests. {@code nowMillis} MUST be a wall-clock source (e.g. {@code
|
||||
* Full constructor — an injectable wall-clock supplier, poll sleeper, and continuation runner,
|
||||
* for tests. {@code nowMillis} MUST be a wall-clock source (e.g. {@code
|
||||
* System.currentTimeMillis()}), never {@code System.nanoTime()}: the freshness check compares
|
||||
* against a file's modified time, which only a wall clock is comparable to, and {@code
|
||||
* nanoTime} freezes while the host sleeps (fleetd #386).
|
||||
* nanoTime} freezes while the host sleeps.
|
||||
*/
|
||||
LeadRollover(AgentControl agents, Supplier<FleetConfig.LeadRollover> configSupplier,
|
||||
Function<String, String> leadWorkspace, LongSupplier nowMillis,
|
||||
Runnable settleSleeper, Consumer<Runnable> continuationRunner) {
|
||||
LeadRollover(AgentControl agents, WorkspaceControl spaces, LeadLauncher launcher,
|
||||
Supplier<FleetConfig.LeadRollover> configSupplier,
|
||||
Function<String, String> leadWorkspace,
|
||||
Function<String, String> leadNameForTerminal,
|
||||
Supplier<Map<String, String>> liveLeadTerminals,
|
||||
LongSupplier nowMillis, Runnable pollSleeper, Consumer<Runnable> continuationRunner) {
|
||||
this.agents = agents;
|
||||
this.spaces = spaces;
|
||||
this.launcher = launcher;
|
||||
this.configSupplier = configSupplier;
|
||||
this.leadWorkspace = leadWorkspace;
|
||||
this.leadNameForTerminal = leadNameForTerminal;
|
||||
this.liveLeadTerminals = liveLeadTerminals;
|
||||
this.nowMillis = nowMillis;
|
||||
this.settleSleeper = settleSleeper;
|
||||
this.pollSleeper = pollSleeper;
|
||||
this.continuationRunner = continuationRunner;
|
||||
}
|
||||
|
||||
@@ -362,8 +449,9 @@ public final class LeadRollover {
|
||||
/**
|
||||
* The lead says it is ready to be replaced. Generates a token and records the resolved
|
||||
* handover path, this moment's wall-clock timestamp (the baseline {@link #confirm} checks the
|
||||
* handover file's modified time against), and {@code leadTerminal} — only that exact terminal
|
||||
* may later {@link #confirm} this token.
|
||||
* handover file's modified time against), {@code leadTerminal} — only that exact terminal may
|
||||
* later {@link #confirm} this token — and {@link PendingRollover#rolloverKey}, resolved here
|
||||
* from {@code leadTerminal} while the calling lead is certainly still live.
|
||||
*
|
||||
* @param leadTerminal the calling lead's terminal id, resolved by the MCP layer from the
|
||||
* connection (see this class's javadoc) — never a client-supplied value
|
||||
@@ -383,7 +471,9 @@ public final class LeadRollover {
|
||||
String token = UUID.randomUUID().toString();
|
||||
long requestedAt = nowMillis.getAsLong();
|
||||
String resolvedPath = resolveHandoverPath(cfg.handoverPath(), leadTerminal);
|
||||
PendingRollover p = new PendingRollover(token, leadTerminal, resolvedPath, requestedAt);
|
||||
String leadName = leadNameForTerminal.apply(leadTerminal);
|
||||
String rolloverKey = (leadName == null || leadName.isBlank()) ? leadTerminal : leadName;
|
||||
PendingRollover p = new PendingRollover(token, leadTerminal, resolvedPath, requestedAt, rolloverKey);
|
||||
pending.put(token, p);
|
||||
if (resolvedPath.equals(cfg.handoverPath())) {
|
||||
log.info("lead-rollover: open token={} lead={} handoverPath={} reason={}",
|
||||
@@ -487,6 +577,15 @@ public final class LeadRollover {
|
||||
return docCheck;
|
||||
}
|
||||
|
||||
// Single-flight claim: atomic put-if-absent, taken only after every other gate has
|
||||
// passed, so a refused confirm() never takes it. A non-null previous value means a
|
||||
// different, still-running roll already holds this lead's claim.
|
||||
String holder = rollingByLead.putIfAbsent(p.rolloverKey(), token);
|
||||
if (holder != null) {
|
||||
return RollDecision.refused(RefusalReason.ROLL_ALREADY_RUNNING,
|
||||
"lead '" + p.rolloverKey() + "' already has a roll running under token " + holder);
|
||||
}
|
||||
|
||||
// Record IN_PROGRESS BEFORE removing from `pending` — see RollState#IN_PROGRESS and
|
||||
// OUTCOME_HISTORY_CAP's javadoc. This ordering means `token` is written into `outcomes`
|
||||
// while it is STILL present in `pending`; status() checks `outcomes` first (see that
|
||||
@@ -494,34 +593,48 @@ public final class LeadRollover {
|
||||
// remove-then-put ordering would leave in which the token is in neither map.
|
||||
outcomes.put(token, new RollStatus(RollState.IN_PROGRESS,
|
||||
"confirm() approved this roll and handed it to the deferred continuation; it has "
|
||||
+ "not finished yet — still waiting for the calling turn to settle, for "
|
||||
+ "/clear to be sent and settle, or for bootstrapText to be sent"));
|
||||
+ "not finished yet — still waiting for the calling turn to settle, for the "
|
||||
+ "old pane to be torn down and confirmed gone, for the fresh lead to be "
|
||||
+ "recognised, or for bootstrapText to be sent"));
|
||||
pending.remove(token);
|
||||
log.info("lead-rollover: confirmed token={} lead={} — roll scheduled once the calling turn ends",
|
||||
token, callerTerminal);
|
||||
continuationRunner.accept(() -> runRollover(p, cfg));
|
||||
try {
|
||||
continuationRunner.accept(() -> runRollover(p, cfg));
|
||||
} catch (RuntimeException e) {
|
||||
// continuationRunner can reject the hand-off itself (e.g. a bounded executor's
|
||||
// RejectedExecutionException) before runRollover ever starts, so runRollover's own
|
||||
// finally — the only other place that releases rollingByLead — never runs either.
|
||||
// Release the claim here and overwrite the IN_PROGRESS entry with a terminal outcome,
|
||||
// or this lead could never be rolled again and status() would report IN_PROGRESS
|
||||
// forever for a roll that in fact never started.
|
||||
log.warn("lead-rollover: continuationRunner rejected token={} lead={}: {} — the roll "
|
||||
+ "never started; releasing its claim and reporting it as FAILED",
|
||||
token, callerTerminal, e.toString(), e);
|
||||
rollingByLead.remove(p.rolloverKey(), token);
|
||||
outcomes.put(token, new RollStatus(RollState.FAILED,
|
||||
"continuationRunner rejected this roll before it ever started: " + e.toString()
|
||||
+ " — the roll never ran; open() a fresh rollover request"));
|
||||
}
|
||||
return RollDecision.approved();
|
||||
}
|
||||
|
||||
/**
|
||||
* The single-shot continuation {@link #confirm} hands to {@code continuationRunner}. Runs
|
||||
* entirely after {@link #confirm} has returned to its caller — see this class's javadoc for the
|
||||
* four-step order. There is no result to return to by this point, so every outcome is logged
|
||||
* only.
|
||||
* full order. There is no result to return to by this point, so every outcome is logged only.
|
||||
*
|
||||
* <p><strong>fleetd #615 — the whole body is wrapped in one {@code try}.</strong> The two {@code
|
||||
* agents.send} calls below are not wrapped individually: {@code send} → {@code agentCall} →
|
||||
* {@code herdr.call} can throw an unchecked {@link dev.ltms.fleet.herdr.HerdrException} (see
|
||||
* {@code AgentControl.java}), and the production {@code continuationRunner} is a bare virtual
|
||||
* thread with no uncaught-exception handler (see this class's public constructor). Before this
|
||||
* fix, either throw killed the continuation thread silently, leaving the {@link
|
||||
* RollState#IN_PROGRESS} entry {@link #confirm} wrote at hand-off stuck forever — {@link
|
||||
* #status} had no way to tell a dead roll from one still genuinely running. The {@code catch}
|
||||
* below is scoped to the method body rather than to each {@code send} call individually, so it
|
||||
* also covers anything else added to this continuation later, not just today's two call sites —
|
||||
* the same reasoning that put the write-a-terminal-outcome step at each of this method's other
|
||||
* exits (see the {@link RollState#TURN_NEVER_SETTLED} and {@link RollState#CLEAR_NEVER_SETTLED}
|
||||
* branches below) rather than inside the helpers that detect them.</p>
|
||||
* <p><strong>The whole body is wrapped in one {@code try}.</strong> Several calls below —
|
||||
* {@code agents.get}, {@code agents.close}, {@code agents.send} — can throw an unchecked {@link
|
||||
* dev.ltms.fleet.herdr.HerdrException} (see {@code AgentControl.java}), and the production
|
||||
* {@code continuationRunner} is a bare virtual thread with no uncaught-exception handler (see
|
||||
* this class's public constructor). An uncaught throw would kill the continuation thread
|
||||
* silently, leaving the {@link RollState#IN_PROGRESS} entry {@link #confirm} wrote at hand-off
|
||||
* stuck forever — {@link #status} would have no way to tell a dead roll from one still
|
||||
* genuinely running. The {@code catch} below is scoped to the method body rather than to each
|
||||
* call individually, so it also covers every call in this continuation, not a fixed list of
|
||||
* call sites — the same reasoning that put the write-a-terminal-outcome step at each of this
|
||||
* method's other exits rather than inside the helpers that detect them.</p>
|
||||
*
|
||||
* <p>Only {@link RuntimeException} is caught, matching the local convention {@link
|
||||
* #waitUntilAtTurnBoundary} already set around its own {@code agents.status} call — not the
|
||||
@@ -539,6 +652,12 @@ public final class LeadRollover {
|
||||
"the roll's continuation threw " + e.toString() + " — the roll is dead and will "
|
||||
+ "not retry itself; check the daemon log for the stack trace, then open() "
|
||||
+ "a fresh rollover request"));
|
||||
} finally {
|
||||
// Release the single-flight claim on both the normal return and the thrown-exception
|
||||
// path above — a release only on success would leave this lead unrollable forever
|
||||
// after one failure. The conditional two-argument remove only clears the entry this
|
||||
// roll itself holds, never a different roll's claim on the same key.
|
||||
rollingByLead.remove(p.rolloverKey(), p.token());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -546,53 +665,237 @@ public final class LeadRollover {
|
||||
private void runRolloverUnguarded(PendingRollover p, FleetConfig.LeadRollover cfg) {
|
||||
String lead = p.leadTerminal();
|
||||
long rollStartMillis = nowMillis.getAsLong();
|
||||
|
||||
TurnSettleResult turnResult = waitUntilAtTurnBoundary(lead, cfg.turnSettleSeconds());
|
||||
if (!turnResult.settled()) {
|
||||
// fleetd #494 follow-up: this line had the SAME defect as the /clear-timeout line below
|
||||
// — cfg.turnSettleSeconds() is the CONFIGURED budget, not how long this wait actually
|
||||
// ran. Print the measured elapsed time alongside it, labelled, exactly like the /clear
|
||||
// path already does.
|
||||
log.warn("lead-rollover: pane {} did not reach a turn boundary (IDLE or DONE) after "
|
||||
+ "confirm() — refusing to send /clear at all; the calling lead's own "
|
||||
+ "turn is still live and clearing it now would destroy live context "
|
||||
+ "confirm() — the old pane is never touched; the calling lead's own "
|
||||
+ "turn is still live and tearing it down now would destroy live context "
|
||||
+ "(token={}, configured={}s elapsed={}ms)",
|
||||
lead, p.token(), cfg.turnSettleSeconds(), turnResult.elapsedMillis());
|
||||
outcomes.put(p.token(), new RollStatus(RollState.TURN_NEVER_SETTLED,
|
||||
"the calling lead's own turn never reached a boundary (IDLE or DONE) within "
|
||||
+ "turnSettleSeconds=" + cfg.turnSettleSeconds() + "s (measured elapsed="
|
||||
+ turnResult.elapsedMillis() + "ms) — no /clear was ever sent. If this "
|
||||
+ "keeps happening, raise turnSettleSeconds in fleetd.yaml"));
|
||||
+ turnResult.elapsedMillis() + "ms) — the old pane was never touched. If "
|
||||
+ "this keeps happening, raise turnSettleSeconds in fleetd.yaml"));
|
||||
return;
|
||||
}
|
||||
|
||||
// This deliberately bypasses Injector, exactly like ClaudeCodeLauncher#clearContext:
|
||||
// /clear is housekeeping, not a delegated turn, and routing it through Injector wedges the
|
||||
// pane forever (see this class's javadoc).
|
||||
agents.send(lead, "/clear");
|
||||
ClearSettleResult clearResult = waitForClearPickupAndSettle(lead, cfg.clearSettleSeconds());
|
||||
if (!clearResult.settled()) {
|
||||
// fleetd #494: cfg.clearSettleSeconds() is the CONFIGURED budget, not how long the wait
|
||||
// actually ran — an operator reading only that number wrongly believes it is a measured
|
||||
// duration. Print the measured elapsed time and nudge count alongside it, each labelled,
|
||||
// so the two can be compared at a glance.
|
||||
log.warn("lead-rollover: pane {} did not reach a turn boundary (IDLE or DONE) after "
|
||||
+ "/clear — NOT sending bootstrapText (token={}, configured={}s "
|
||||
+ "elapsed={}ms nudges={})",
|
||||
lead, p.token(), cfg.clearSettleSeconds(), clearResult.elapsedMillis(),
|
||||
clearResult.nudges());
|
||||
outcomes.put(p.token(), new RollStatus(RollState.CLEAR_NEVER_SETTLED,
|
||||
"/clear was sent, but the pane never re-settled within clearSettleSeconds="
|
||||
+ cfg.clearSettleSeconds() + "s (measured elapsed=" + clearResult.elapsedMillis()
|
||||
+ "ms, nudges=" + clearResult.nudges() + ") — bootstrapText was never sent"));
|
||||
// Captured once, here, and never re-resolved from `lead` again below: once the pane is
|
||||
// closed there is nothing left for a terminal lookup to find.
|
||||
Agent oldAgent = captureAgentWithRetry(lead);
|
||||
String oldPaneId = oldAgent.paneId();
|
||||
String leadName = leadNameForTerminal.apply(lead);
|
||||
|
||||
endOldSession(oldPaneId);
|
||||
DeathResult deathResult = waitUntilPaneGone(oldPaneId);
|
||||
if (!deathResult.gone()) {
|
||||
log.warn("lead-rollover: old pane {} for lead {} was never confirmed gone after being "
|
||||
+ "closed — not attempting a relaunch (token={}, timeout={}s "
|
||||
+ "elapsed={}ms)",
|
||||
oldPaneId, lead, p.token(), PANE_DEATH_TIMEOUT_SECONDS, deathResult.elapsedMillis());
|
||||
outcomes.put(p.token(), new RollStatus(RollState.OLD_PANE_NEVER_DIED,
|
||||
"the old pane was closed, but locatePane kept reporting it as still present "
|
||||
+ "after a pane-death timeout=" + PANE_DEATH_TIMEOUT_SECONDS
|
||||
+ "s (measured elapsed=" + deathResult.elapsedMillis() + "ms) — no "
|
||||
+ "relaunch was attempted"));
|
||||
return;
|
||||
}
|
||||
agents.send(lead, cfg.bootstrapTextFor(p.handoverPath()));
|
||||
|
||||
Agent newAgent = launcher.relaunch(leadName);
|
||||
if (newAgent == null) {
|
||||
log.warn("lead-rollover: relaunch of lead '{}' (old terminal {}) failed every attempt "
|
||||
+ "— bootstrapText was never sent (token={})", leadName, lead, p.token());
|
||||
outcomes.put(p.token(), new RollStatus(RollState.RELAUNCH_FAILED,
|
||||
"lead '" + leadName + "' could not be relaunched — every attempt failed; "
|
||||
+ "bootstrapText was never sent"));
|
||||
return;
|
||||
}
|
||||
|
||||
ReadinessResult readinessResult = waitUntilPaneReady(newAgent.terminalId(),
|
||||
cfg.relaunchReadySeconds());
|
||||
if (!readinessResult.ready()) {
|
||||
log.warn("lead-rollover: fresh pane for lead '{}' (terminal {}) never reached a real "
|
||||
+ "turn boundary — bootstrapText was never sent (token={}, configured={}s "
|
||||
+ "elapsed={}ms)",
|
||||
leadName, newAgent.terminalId(), p.token(), cfg.relaunchReadySeconds(),
|
||||
readinessResult.elapsedMillis());
|
||||
outcomes.put(p.token(), new RollStatus(RollState.RELAUNCH_NEVER_READY,
|
||||
"fresh terminal " + newAgent.terminalId() + " never reached a real turn "
|
||||
+ "boundary (IDLE or DONE) within relaunchReadySeconds="
|
||||
+ cfg.relaunchReadySeconds() + "s (measured elapsed="
|
||||
+ readinessResult.elapsedMillis() + "ms) — bootstrapText was never "
|
||||
+ "sent"));
|
||||
return;
|
||||
}
|
||||
|
||||
IdentityResult identityResult = waitUntilRecognisedAsLead(newAgent.terminalId(),
|
||||
cfg.relaunchReadySeconds());
|
||||
agents.send(newAgent.terminalId(), cfg.bootstrapTextFor(p.handoverPath()));
|
||||
if (!identityResult.ready()) {
|
||||
log.warn("lead-rollover: fresh terminal {} for lead '{}' is alive and bootstrapped, but "
|
||||
+ "was never recognised as a live lead — an operator should check why "
|
||||
+ "the tab was not recognised (token={}, configured={}s elapsed={}ms)",
|
||||
newAgent.terminalId(), leadName, p.token(), cfg.relaunchReadySeconds(),
|
||||
identityResult.elapsedMillis());
|
||||
outcomes.put(p.token(), new RollStatus(RollState.RELAUNCH_NOT_RECOGNISED,
|
||||
"bootstrapText was sent to fresh terminal " + newAgent.terminalId() + ", but "
|
||||
+ "it was never recognised as a live lead within relaunchReadySeconds="
|
||||
+ cfg.relaunchReadySeconds() + "s (measured elapsed="
|
||||
+ identityResult.elapsedMillis() + "ms) — check why the tab was not "
|
||||
+ "recognised"));
|
||||
return;
|
||||
}
|
||||
|
||||
long rollElapsedMillis = nowMillis.getAsLong() - rollStartMillis;
|
||||
log.info("lead-rollover: rolled token={} lead={} elapsedMs={}", p.token(), lead, rollElapsedMillis);
|
||||
log.info("lead-rollover: rolled token={} oldLead={} newTerminal={} elapsedMs={}",
|
||||
p.token(), lead, newAgent.terminalId(), rollElapsedMillis);
|
||||
outcomes.put(p.token(), new RollStatus(RollState.ROLLED,
|
||||
"rolled successfully in " + rollElapsedMillis + "ms"));
|
||||
"rolled successfully in " + rollElapsedMillis + "ms; new terminal="
|
||||
+ newAgent.terminalId()));
|
||||
}
|
||||
|
||||
/** Attempts {@link #captureAgentWithRetry} makes before letting the failure propagate. */
|
||||
static final int CAPTURE_RETRIES = 3;
|
||||
|
||||
/**
|
||||
* {@link AgentControl#get} for {@code lead}, retried up to {@link #CAPTURE_RETRIES} times. The
|
||||
* terminal-to-pane lookup it goes through can report a genuinely live agent as not found (see
|
||||
* {@code AgentControl#agentCall}'s own re-resolve-once behaviour), and one such false negative
|
||||
* must not abort an otherwise-healthy roll. The result is captured once by the caller and never
|
||||
* looked up again — see this class's javadoc.
|
||||
*
|
||||
* @throws RuntimeException the last failure, if every attempt fails — {@link #runRollover}'s
|
||||
* catch turns that into {@link RollState#FAILED}
|
||||
*/
|
||||
private Agent captureAgentWithRetry(String lead) {
|
||||
RuntimeException last = null;
|
||||
for (int attempt = 1; attempt <= CAPTURE_RETRIES; attempt++) {
|
||||
try {
|
||||
return agents.get(lead);
|
||||
} catch (RuntimeException e) {
|
||||
last = e;
|
||||
log.debug("lead-rollover: agents.get({}) failed on attempt {}/{}: {}",
|
||||
lead, attempt, CAPTURE_RETRIES, e.toString());
|
||||
if (attempt < CAPTURE_RETRIES) {
|
||||
pollSleeper.run();
|
||||
}
|
||||
}
|
||||
}
|
||||
throw last;
|
||||
}
|
||||
|
||||
/**
|
||||
* End the old lead's session: close its pane, then close its tab only when the pane was that
|
||||
* tab's sole occupant — the same pane-then-tab teardown {@code HerdrPeerLauncher#stop} uses for
|
||||
* a member. An already-gone pane counts as success; any other {@code agents.close} failure
|
||||
* propagates, so a genuinely failed teardown is never reported as done. A failing
|
||||
* {@code spaces.closeTab} never propagates — by the time it runs the pane is already closed, so
|
||||
* it is cosmetic tidying, not a real teardown failure.
|
||||
*/
|
||||
private void endOldSession(String paneId) {
|
||||
WorkspaceControl.PaneLocation loc = spaces.locatePane(paneId);
|
||||
try {
|
||||
agents.close(paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!isAlreadyGone(e)) {
|
||||
throw e;
|
||||
}
|
||||
log.debug("lead-rollover: pane.close({}) ignored — already gone: {}", paneId, e.getMessage());
|
||||
}
|
||||
if (loc != null && loc.tabPaneCount() == 1) {
|
||||
try {
|
||||
spaces.closeTab(loc.tabId());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("lead-rollover: tab.close({}) failed — the pane is already torn down, so "
|
||||
+ "continuing; the tab may need manual cleanup: {}", loc.tabId(), e.getMessage());
|
||||
}
|
||||
} else if (loc != null) {
|
||||
log.debug("lead-rollover: not closing tab {} — it holds {} panes (not a dedicated lead "
|
||||
+ "tab)", loc.tabId(), loc.tabPaneCount());
|
||||
}
|
||||
}
|
||||
|
||||
/** True when a herdr error means the target is already gone (safe to treat as done). */
|
||||
private static boolean isAlreadyGone(HerdrException e) {
|
||||
return e.code() != null && e.code().endsWith("_not_found");
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll {@link WorkspaceControl#locatePane} for {@code paneId} until it reports {@code null}
|
||||
* (the pane is gone) or {@link #PANE_DEATH_TIMEOUT_SECONDS} elapses. Deliberately never calls
|
||||
* {@link AgentControl#status} and never reads the live-lead terminal map — both answer a
|
||||
* different question (whether an AGENT is live, not whether this PANE still exists) and
|
||||
* {@code locatePane} alone catches a {@link HerdrException} from the underlying {@code
|
||||
* pane.get} and turns it into {@code null} — see this class's javadoc.
|
||||
*/
|
||||
private DeathResult waitUntilPaneGone(String paneId) {
|
||||
long startMillis = nowMillis.getAsLong();
|
||||
long deadline = startMillis + TimeUnit.SECONDS.toMillis(PANE_DEATH_TIMEOUT_SECONDS);
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
if (spaces.locatePane(paneId) == null) {
|
||||
return new DeathResult(true, nowMillis.getAsLong() - startMillis);
|
||||
}
|
||||
pollSleeper.run();
|
||||
}
|
||||
return new DeathResult(false, nowMillis.getAsLong() - startMillis);
|
||||
}
|
||||
|
||||
/** The measured outcome of {@link #waitUntilPaneGone}. */
|
||||
private record DeathResult(boolean gone, long elapsedMillis) {}
|
||||
|
||||
/**
|
||||
* Poll until {@code newTerminal}'s own pane reaches a real turn boundary ({@link
|
||||
* AgentStatus#IDLE} or {@link AgentStatus#DONE}, never merely {@link AgentStatus#BLOCKED}) —
|
||||
* the same exclusion {@link #waitUntilAtTurnBoundary} applies to the calling lead's own turn,
|
||||
* applied here to the fresh one, so {@code bootstrapText} is never typed into a pane that has
|
||||
* not actually finished booting — or {@code readySeconds} elapses. A failed status read
|
||||
* degrades to "not yet ready" and is retried on the next poll.
|
||||
*/
|
||||
private ReadinessResult waitUntilPaneReady(String newTerminal, int readySeconds) {
|
||||
long startMillis = nowMillis.getAsLong();
|
||||
long deadline = startMillis + TimeUnit.SECONDS.toMillis(readySeconds);
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
AgentStatus status;
|
||||
try {
|
||||
status = agents.status(newTerminal);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("lead-rollover: status check failed while waiting for {} to be ready: {}",
|
||||
newTerminal, e.toString());
|
||||
status = null;
|
||||
}
|
||||
if (status == AgentStatus.IDLE || status == AgentStatus.DONE) {
|
||||
return new ReadinessResult(true, nowMillis.getAsLong() - startMillis);
|
||||
}
|
||||
pollSleeper.run();
|
||||
}
|
||||
return new ReadinessResult(false, nowMillis.getAsLong() - startMillis);
|
||||
}
|
||||
|
||||
/** The measured outcome of {@link #waitUntilPaneReady}. */
|
||||
private record ReadinessResult(boolean ready, long elapsedMillis) {}
|
||||
|
||||
/**
|
||||
* Poll until {@code newTerminal} is present in {@link #liveLeadTerminals} or {@code
|
||||
* readySeconds} elapses. This is bookkeeping, not a safety gate: the pane's own readiness (see
|
||||
* {@link #waitUntilPaneReady}) is what decides whether {@code bootstrapText} is safe to send —
|
||||
* a timeout here only means the daemon's own lead-discovery scan has not caught up yet.
|
||||
*/
|
||||
private IdentityResult waitUntilRecognisedAsLead(String newTerminal, int readySeconds) {
|
||||
long startMillis = nowMillis.getAsLong();
|
||||
long deadline = startMillis + TimeUnit.SECONDS.toMillis(readySeconds);
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
if (liveLeadTerminals.get().containsKey(newTerminal)) {
|
||||
return new IdentityResult(true, nowMillis.getAsLong() - startMillis);
|
||||
}
|
||||
pollSleeper.run();
|
||||
}
|
||||
return new IdentityResult(false, nowMillis.getAsLong() - startMillis);
|
||||
}
|
||||
|
||||
/** The measured outcome of {@link #waitUntilRecognisedAsLead}. */
|
||||
private record IdentityResult(boolean ready, long elapsedMillis) {}
|
||||
|
||||
/** Drop a pending request without rolling. @return whether a pending request existed for {@code token} */
|
||||
public boolean cancel(String token) {
|
||||
return pending.remove(token) != null;
|
||||
@@ -682,15 +985,12 @@ public final class LeadRollover {
|
||||
|
||||
/**
|
||||
* Poll {@link AgentControl#status} until {@code target} reports a real turn boundary — {@link
|
||||
* AgentStatus#IDLE} or {@link AgentStatus#DONE} — bounded by {@code settleSeconds}. Used once by
|
||||
* {@link #runRollover}, to wait for the CALLING turn's own pane to settle before {@code /clear}
|
||||
* is ever sent at all — the {@code turnSettleSeconds} gate that makes this correction safe. The
|
||||
* SECOND wait, after {@code /clear}, is {@link #waitForClearPickupAndSettle} instead (fleetd
|
||||
* #489) — a plain boundary check is not enough there, because {@code /clear} starts no turn of
|
||||
* its own, so this method would (wrongly) report "settled" on its very first poll whether or not
|
||||
* {@code /clear} was actually picked up. A failed status read degrades to "not yet settled" and
|
||||
* is retried on the next poll, the same posture {@code LeadHeartbeatLoop} and {@code
|
||||
* HerdrPeerLauncher}'s readiness gate already take toward an unreadable status.
|
||||
* AgentStatus#IDLE} or {@link AgentStatus#DONE} — bounded by {@code settleSeconds}. Used by
|
||||
* {@link #runRollover} to wait for the CALLING turn's own pane to settle before the old pane is
|
||||
* touched at all — the {@code turnSettleSeconds} gate that makes tearing it down safe. A failed
|
||||
* status read degrades to "not yet settled" and is retried on the next poll, the same posture
|
||||
* {@code LeadHeartbeatLoop} and {@code HerdrPeerLauncher}'s readiness gate already take toward
|
||||
* an unreadable status.
|
||||
*
|
||||
* <p><strong>Deliberately not {@link AgentStatus#injectable()}.</strong> {@code injectable()}
|
||||
* answers the {@code Injector}'s question — "may I deliver a message without stepping on a live
|
||||
@@ -698,16 +998,16 @@ public final class LeadRollover {
|
||||
* an approval prompt is safe to queue a message behind. This class asks a stricter question —
|
||||
* "has the turn actually ended" — and {@code BLOCKED} answers no: it is a live turn that is
|
||||
* merely paused, not one that has finished. Reusing {@code injectable()} here would let this
|
||||
* wait fire {@code /clear} while the lead's own {@code confirm()}-calling turn is still live and
|
||||
* paused on a prompt — exactly the live-context-destroying failure the {@code turnSettleSeconds}
|
||||
* gate exists to prevent. Do not "simplify" this back to {@code injectable()}. ({@link
|
||||
* #waitForClearPickupAndSettle} keeps the same exclusion of {@code BLOCKED}, for the same
|
||||
* reason, on the second wait.)
|
||||
* wait tear the old pane down while the lead's own {@code confirm()}-calling turn is still live
|
||||
* and paused on a prompt — exactly the live-context-destroying failure {@code turnSettleSeconds}
|
||||
* exists to prevent. Do not "simplify" this back to {@code injectable()}. ({@link
|
||||
* #waitUntilPaneReady} applies the same exclusion of {@code BLOCKED} to the fresh lead's own
|
||||
* turn.)
|
||||
*
|
||||
* @return a {@link TurnSettleResult} whose {@code settled()} is {@code true} once a real
|
||||
* boundary was observed, {@code false} if {@code settleSeconds} elapses first.
|
||||
* {@code elapsedMillis()} is a MEASURED value from the injected {@link #nowMillis}
|
||||
* clock, never the configured {@code settleSeconds} budget (fleetd #494 follow-up).
|
||||
* clock, never the configured {@code settleSeconds} budget.
|
||||
*/
|
||||
private TurnSettleResult waitUntilAtTurnBoundary(String target, int settleSeconds) {
|
||||
long startMillis = nowMillis.getAsLong();
|
||||
@@ -724,142 +1024,11 @@ public final class LeadRollover {
|
||||
if (status == AgentStatus.IDLE || status == AgentStatus.DONE) {
|
||||
return new TurnSettleResult(true, nowMillis.getAsLong() - startMillis);
|
||||
}
|
||||
settleSleeper.run();
|
||||
pollSleeper.run();
|
||||
}
|
||||
return new TurnSettleResult(false, nowMillis.getAsLong() - startMillis);
|
||||
}
|
||||
|
||||
/**
|
||||
* The measured outcome of {@link #waitUntilAtTurnBoundary} — fleetd #494 follow-up. The sibling
|
||||
* of {@link ClearSettleResult} for the FIRST wait, which never nudges, so it carries no nudge
|
||||
* count.
|
||||
*/
|
||||
/** The measured outcome of {@link #waitUntilAtTurnBoundary}. */
|
||||
private record TurnSettleResult(boolean settled, long elapsedMillis) {}
|
||||
|
||||
/**
|
||||
* The SECOND wait in {@link #runRollover} — after {@code /clear} has been sent, waits for it to
|
||||
* settle, bounded by {@code settleSeconds}. <strong>fleetd #489 — the paste-race fix.</strong>
|
||||
* {@code /clear} does not start a real turn of its own, so a pane with no submit race simply
|
||||
* stays {@link AgentStatus#IDLE} the whole time: {@link #waitUntilAtTurnBoundary} would (wrongly)
|
||||
* call that "settled" on its very first poll, whether or not the {@code /clear} Enter actually
|
||||
* landed. That was Fault 1, measured live on 2026-09-12 — the second gate was a no-op, so a
|
||||
* {@code bootstrapText} send followed immediately, racing Fault 2: {@link AgentControl#submit}'s
|
||||
* own javadoc already records that the submit accompanying a delivery "can race the paste —
|
||||
* especially right as the worker's TUI becomes interactive — leaving the text unsubmitted"
|
||||
* (CB-113). Because {@code runRollover} deliberately bypasses {@code Injector} for {@code
|
||||
* /clear} (see this class's javadoc), it inherited none of {@code Injector}'s nudging — so the
|
||||
* lost {@code /clear} Enter sat in the input box and {@code bootstrapText} was typed right after
|
||||
* it, landing as one concatenated line.
|
||||
*
|
||||
* <p>This method copies the pickup-nudge pattern {@link dev.ltms.fleet.inject.Injector} already
|
||||
* ships for exactly this, on its own post-turn {@code /clear} housekeeping (fleetd #306; see
|
||||
* {@code Injector.java:288-340} and {@code Injector.java:437-442}):
|
||||
* <ul>
|
||||
* <li>an {@link AgentStatus#WORKING} sample means {@code /clear} was picked up as a real
|
||||
* turn;</li>
|
||||
* <li>until that happens, each poll that still reports {@link AgentStatus#IDLE} or {@link
|
||||
* AgentStatus#DONE} re-sends the submit keystroke ({@link AgentControl#submit}) to nudge
|
||||
* the raced Enter — for the first {@code PICKUP_GRACE_POLLS - 1} of {@link
|
||||
* #PICKUP_GRACE_POLLS} consecutive such polls (i.e. {@code PICKUP_GRACE_POLLS - 1}
|
||||
* nudges: 7, not 8, given {@code PICKUP_GRACE_POLLS = 8}). A second Enter on an empty
|
||||
* Claude Code prompt is a no-op, so repeating it is safe;</li>
|
||||
* <li>the {@code PICKUP_GRACE_POLLS}th consecutive such poll, with {@code WORKING} still never
|
||||
* observed, releases rather than wedges the roll instead of nudging again — the same
|
||||
* choice {@code Injector} makes — and returns {@code settled() == true} anyway, logged at
|
||||
* {@code warn} with the measured elapsed time (fleetd #494) so an operator can see which
|
||||
* path ran and how long it actually took;</li>
|
||||
* <li>once {@code WORKING} has been observed, nudging stops and this instead waits for a real
|
||||
* {@code working → IDLE/DONE} completion boundary before returning {@code true}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>{@link AgentStatus#BLOCKED} is deliberately excluded from both the nudge and the
|
||||
* boundary check</strong> — the same reasoning as {@link #waitUntilAtTurnBoundary}'s own
|
||||
* javadoc: a paused live turn is not a settled one, and re-sending Enter into an open approval
|
||||
* prompt could wrongly answer it. A {@code BLOCKED} sample (or an unreadable/{@link
|
||||
* AgentStatus#UNKNOWN} one) simply keeps this polling, with no nudge and no release, until either
|
||||
* a real boundary is reached or {@code settleSeconds} runs out.
|
||||
*
|
||||
* <p>{@link AgentControl#submit} can itself throw; a {@link RuntimeException} from it is
|
||||
* swallowed and logged at {@code debug}, exactly like {@code Injector.java:437-442} — a failed
|
||||
* nudge must not abort the roll.
|
||||
*
|
||||
* @return a {@link ClearSettleResult} whose {@code settled()} is {@code true} once {@code
|
||||
* /clear} has settled, or once the nudge budget was exhausted with no pickup ever
|
||||
* observed (released rather than wedged); {@code false} if {@code settleSeconds} elapses
|
||||
* first — the caller must NOT send {@code bootstrapText} in that case, exactly as before
|
||||
* this fix. {@code elapsedMillis()} and {@code nudges()} are MEASURED values (from the
|
||||
* injected {@link #nowMillis} clock and an actual nudge count), never the configured
|
||||
* {@code settleSeconds} budget (fleetd #494).
|
||||
*/
|
||||
private ClearSettleResult waitForClearPickupAndSettle(String target, int settleSeconds) {
|
||||
long startMillis = nowMillis.getAsLong();
|
||||
long deadline = startMillis + TimeUnit.SECONDS.toMillis(settleSeconds);
|
||||
boolean pickedUp = false; // a WORKING sample has been observed since /clear was sent
|
||||
int idlePollsAwaitingPickup = 0;
|
||||
int nudges = 0;
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
AgentStatus status;
|
||||
try {
|
||||
status = agents.status(target);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("lead-rollover: status check failed while waiting for {} to settle after "
|
||||
+ "/clear: {}", target, e.toString());
|
||||
status = null;
|
||||
}
|
||||
if (status == AgentStatus.WORKING) {
|
||||
pickedUp = true;
|
||||
} else if (status == AgentStatus.IDLE || status == AgentStatus.DONE) {
|
||||
if (pickedUp) {
|
||||
// a real WORKING -> IDLE/DONE completion boundary
|
||||
return new ClearSettleResult(true, nowMillis.getAsLong() - startMillis, nudges);
|
||||
}
|
||||
if (++idlePollsAwaitingPickup >= PICKUP_GRACE_POLLS) {
|
||||
long elapsedMillis = nowMillis.getAsLong() - startMillis;
|
||||
// fleetd #494: this release trades a possibly-unsubmitted /clear for progress
|
||||
// instead of wedging the roll — that trade is deliberate and stays. But it is
|
||||
// also exactly the case that reported false success in the real incident (the
|
||||
// whole roll "succeeded" after 438ms of a 20s budget), so raise it to WARN and
|
||||
// print the MEASURED elapsed time next to the target pane, not just the count.
|
||||
//
|
||||
// fleetd #494 follow-up (2nd pass): BOTH numbers in this line must come from
|
||||
// the loop's own counters, never from the PICKUP_GRACE_POLLS constant.
|
||||
// `idlePollsAwaitingPickup` and `nudges` each have exactly one write site in
|
||||
// this loop, on the same branch, so on this branch they cannot differ from
|
||||
// PICKUP_GRACE_POLLS / PICKUP_GRACE_POLLS - 1 today — no test can prove the
|
||||
// difference on this line, and printing the counters does not change that.
|
||||
// What it does buy: one source of truth instead of two, so a later change to
|
||||
// the loop (an early return, a second increment site, a different exit
|
||||
// condition) cannot leave this message reporting a number the loop no longer
|
||||
// produces. The place where `nudges` genuinely varies with the run — and is
|
||||
// covered by a test that can tell it apart from a constant — is the
|
||||
// /clear-timeout warn in runRollover, which prints clearResult.nudges().
|
||||
log.warn("lead-rollover: /clear on {} was never observed as WORKING after {} "
|
||||
+ "consecutive IDLE/DONE polls ({} of those were nudged) — "
|
||||
+ "releasing rather than wedging the roll (elapsed={}ms)",
|
||||
target, idlePollsAwaitingPickup, nudges, elapsedMillis);
|
||||
return new ClearSettleResult(true, elapsedMillis, nudges);
|
||||
}
|
||||
try {
|
||||
agents.submit(target); // nudge a raced Enter (CB-113) so /clear actually submits
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("lead-rollover: resubmit to {} failed (will retry next poll): {}",
|
||||
target, e.getMessage());
|
||||
} finally {
|
||||
nudges++; // an attempted nudge, whether or not the submit call itself threw
|
||||
}
|
||||
}
|
||||
// AgentStatus.BLOCKED or UNKNOWN (or an unreadable status, above): neither a pickup
|
||||
// signal nor a boundary — keep polling without nudging or releasing.
|
||||
settleSleeper.run();
|
||||
}
|
||||
return new ClearSettleResult(false, nowMillis.getAsLong() - startMillis, nudges);
|
||||
}
|
||||
|
||||
/**
|
||||
* The measured outcome of {@link #waitForClearPickupAndSettle} — fleetd #494. Carries the
|
||||
* MEASURED elapsed time (from the injected {@link #nowMillis} clock) and nudge count alongside
|
||||
* the settle/timeout decision, so callers can log them instead of the configured budget, which
|
||||
* is not how long the wait actually ran.
|
||||
*/
|
||||
private record ClearSettleResult(boolean settled, long elapsedMillis, int nudges) {}
|
||||
}
|
||||
|
||||
@@ -5,13 +5,13 @@ import dev.ltms.fleet.herdr.PaneLocator;
|
||||
/**
|
||||
* Resolves <em>who is calling</em> an MCP tool from the connection alone — the anti-spoofing
|
||||
* identity model of the MCP contract. It ties the connection's loopback peer PID (from the OS)
|
||||
* to a herdr agent pane (from herdr), yielding the caller's worker {@code terminal_id}. A caller
|
||||
* that maps to no worker pane — the primary, or an off-host client — resolves to {@code null}.
|
||||
* to a herdr agent pane (from herdr), yielding that pane's {@code terminal_id}. A connection that
|
||||
* maps to no pane resolves to {@code null}; this class assigns no role to either outcome — {@link
|
||||
* dev.ltms.fleet.auth.CallerResolver} does that.
|
||||
*
|
||||
* <p>Both sources are authoritative and unforgeable: the OS reports the real connecting PID, and
|
||||
* herdr owns the PID→pane mapping. A worker cannot claim to be another worker, nor the primary.
|
||||
* Single-host only (the herd shares the {@code fleetd} host); the token path is the split-host
|
||||
* fallback.
|
||||
* herdr owns the PID→pane mapping, so a caller cannot claim to be at another pane. Single-host
|
||||
* only (the herd shares the {@code fleetd} host); the token path is the split-host fallback.
|
||||
*/
|
||||
public final class ConnectionIdentity {
|
||||
|
||||
@@ -32,9 +32,24 @@ public final class ConnectionIdentity {
|
||||
}
|
||||
|
||||
/**
|
||||
* The caller resolved from the connection: its worker {@code terminal} (or {@code null} for the
|
||||
* primary / an off-host client), its {@code pid} (or {@code -1} if not resolvable), and whether
|
||||
* the pane scan behind {@code terminal} ran to completion ({@link #scanComplete}).
|
||||
* The {@link PaneLocator} this identity resolves callers against — fleetd #612 CB-185: lets a
|
||||
* test drive the exact {@link PaneLocator} a real assembly wired up (e.g. {@code
|
||||
* FleetdAssembly}'s {@code new ConnectionIdentity(new PaneLocator(herdr, memberHerdr), ...)})
|
||||
* directly with a chosen pid, bypassing the OS-dependent {@link PeerPidLookup} that {@link
|
||||
* #resolve} otherwise goes through. A full HTTP round trip cannot exercise this: {@code
|
||||
* LsofPeerPidLookup} excludes its own pid, and an in-process test client and server share one
|
||||
* JVM pid, so {@code pidForLocalPort} always returns {@code -1} and {@link PaneLocator} never
|
||||
* gets called at all.
|
||||
*/
|
||||
public PaneLocator panes() {
|
||||
return panes;
|
||||
}
|
||||
|
||||
/**
|
||||
* The caller resolved from the connection: the {@code terminal} of the pane it connects from
|
||||
* (or {@code null} when the connection maps to no pane), its {@code pid} (or {@code -1} if not
|
||||
* resolvable), and whether the pane scan behind {@code terminal} ran to completion
|
||||
* ({@link #scanComplete}).
|
||||
*/
|
||||
public record Caller(String terminal, long pid, boolean scanComplete) {
|
||||
|
||||
@@ -73,8 +88,8 @@ public final class ConnectionIdentity {
|
||||
}
|
||||
|
||||
/**
|
||||
* The calling worker's {@code terminal_id}, or {@code null} if the caller is not a known
|
||||
* on-host worker (treat as the primary).
|
||||
* The terminal id of the pane the caller connects from, or {@code null} if the connection
|
||||
* maps to no pane.
|
||||
*/
|
||||
public String callerTerminal(String remoteAddr, int remotePort) {
|
||||
return resolve(remoteAddr, remotePort).terminal();
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -44,7 +44,8 @@ public enum FleetTool {
|
||||
STOP("fleet_stop"),
|
||||
PROFILES("fleet_profiles"),
|
||||
WHOAMI("fleet_whoami"),
|
||||
HANDOVER("fleet_handover");
|
||||
HANDOVER("fleet_handover"),
|
||||
INBOX("fleet_inbox");
|
||||
|
||||
private final String wireName;
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ import org.slf4j.LoggerFactory;
|
||||
import java.util.Optional;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
|
||||
/**
|
||||
* Single-slot, thread-safe registry for the primary's herdr {@code terminal_id}.
|
||||
@@ -18,13 +19,25 @@ import java.util.concurrent.atomic.AtomicReference;
|
||||
* <p>The push loop ({@code ReplyPushLoop}) uses {@link #isKnown()} to decide
|
||||
* whether active nudging is possible; an empty registry means the primary is
|
||||
* off-host or non-herdr and delivery falls back to pull.
|
||||
*
|
||||
* <p><strong>A learned terminal can go stale; a configured lead's name cannot.</strong> A lead that
|
||||
* is rolled (a fresh pane replacing the old one) keeps its name but gets a new {@code terminal_id}.
|
||||
* So every terminal this class learns — the singleton and each per-target delegation — is recorded
|
||||
* together with the delegating lead's name, when the caller carries one. {@link
|
||||
* #currentPrimaryTerminal()} and {@link #nudgeTargetFor(String)} resolve that name back to a
|
||||
* terminal through the live {@code currentTerminalForName} lookup before falling back to the
|
||||
* terminal that was actually recorded. A caller with no name (an unnamed primary, an architect, a
|
||||
* collaborator — none of those are leads a lookup keyed on lead names can resolve) is tracked by
|
||||
* terminal alone, exactly as before this indirection existed.
|
||||
*/
|
||||
public final class PrimaryRegistry {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(PrimaryRegistry.class);
|
||||
|
||||
private final AtomicReference<String> terminal = new AtomicReference<>();
|
||||
private final AtomicReference<String> primaryName = new AtomicReference<>();
|
||||
private final boolean pinned;
|
||||
private final Function<String, String> currentTerminalForName;
|
||||
|
||||
/**
|
||||
* CB-532: worker terminal → the lead that delegated to it. The single slot above answers "who is
|
||||
@@ -33,12 +46,32 @@ public final class PrimaryRegistry {
|
||||
* other lead's delegations. This map answers the question that actually matters — "who is
|
||||
* waiting on THIS worker" — and is what lets {@code primary.terminal} be retired.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, String> leadByTarget = new ConcurrentHashMap<>();
|
||||
private final ConcurrentHashMap<String, Delegation> leadByTarget = new ConcurrentHashMap<>();
|
||||
|
||||
/** A recorded delegator: the terminal learned from call traffic, and its name, if it has one. */
|
||||
private record Delegation(String terminal, String name) {
|
||||
}
|
||||
|
||||
/**
|
||||
* @param pinnedTerminal an optional pinned terminal from config ({@code null}/blank = unpinned)
|
||||
*/
|
||||
public PrimaryRegistry(String pinnedTerminal) {
|
||||
this(pinnedTerminal, name -> null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, with a live {@code lead name → current terminal} lookup — normally the inverse of
|
||||
* the same {@code terminal_id → lead name} supplier {@code CallerResolver} and the lead-tab
|
||||
* scan already read. A lookup that cannot place a name (it is not a currently recognised lead,
|
||||
* or no lookup is wired) returns {@code null}, and every resolution here falls back to the
|
||||
* terminal that was actually recorded.
|
||||
*
|
||||
* @param pinnedTerminal an optional pinned terminal from config ({@code null}/blank =
|
||||
* unpinned)
|
||||
* @param currentTerminalForName lead name → its current terminal, or {@code null} if that name
|
||||
* is not a currently recognised lead
|
||||
*/
|
||||
public PrimaryRegistry(String pinnedTerminal, Function<String, String> currentTerminalForName) {
|
||||
if (pinnedTerminal != null && !pinnedTerminal.isBlank()) {
|
||||
this.terminal.set(pinnedTerminal);
|
||||
this.pinned = true;
|
||||
@@ -46,19 +79,31 @@ public final class PrimaryRegistry {
|
||||
} else {
|
||||
this.pinned = false;
|
||||
}
|
||||
this.currentTerminalForName = currentTerminalForName != null ? currentTerminalForName : name -> null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Record a terminal_id. No-op when:
|
||||
* Record a terminal_id, with no lead name. No-op when:
|
||||
* <ul>
|
||||
* <li>the registry is pinned (config override),
|
||||
* <li>{@code terminalId} is {@code null} or blank (non-herdr caller).
|
||||
* </ul>
|
||||
*/
|
||||
public void record(String terminalId) {
|
||||
record(terminalId, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #record(String)}, additionally recording the caller's name — present for a
|
||||
* configured lead, {@code null} for an unnamed primary. The name is what lets {@link
|
||||
* #currentPrimaryTerminal()} keep nudging the same lead across a roll even though its terminal
|
||||
* changed.
|
||||
*/
|
||||
public void record(String terminalId, String name) {
|
||||
if (pinned) return;
|
||||
if (terminalId == null || terminalId.isBlank()) return;
|
||||
String prev = terminal.getAndSet(terminalId);
|
||||
primaryName.set(blankToNull(name));
|
||||
if (prev == null) {
|
||||
log.debug("primary terminal learned: {}", terminalId);
|
||||
} else if (!prev.equals(terminalId)) {
|
||||
@@ -67,7 +112,8 @@ public final class PrimaryRegistry {
|
||||
}
|
||||
|
||||
/**
|
||||
* Record that {@code leadTerminal} owns the accepted delegation of worker {@code target} (CB-532).
|
||||
* Record that {@code leadTerminal} owns the accepted delegation of worker {@code target}
|
||||
* (CB-532), with no lead name.
|
||||
*
|
||||
* <p>Called from the {@code MessageService} accepted-delivery hook — only after a send has won
|
||||
* the session's send lock and queued delivery — where both halves are known (CB-548). It is
|
||||
@@ -77,10 +123,19 @@ public final class PrimaryRegistry {
|
||||
* lead that most recently delegated to it, which is the one waiting.
|
||||
*/
|
||||
public void recordDelegation(String target, String leadTerminal) {
|
||||
recordDelegation(target, leadTerminal, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #recordDelegation(String, String)}, additionally recording the delegating lead's
|
||||
* name when the caller carries one. See {@link #record(String, String)} for why the name
|
||||
* matters.
|
||||
*/
|
||||
public void recordDelegation(String target, String leadTerminal, String leadName) {
|
||||
if (target == null || target.isBlank() || leadTerminal == null || leadTerminal.isBlank()) {
|
||||
return;
|
||||
}
|
||||
leadByTarget.put(target, leadTerminal);
|
||||
leadByTarget.put(target, new Delegation(leadTerminal, blankToNull(leadName)));
|
||||
}
|
||||
|
||||
/** Forget a worker's delegating lead — call on release, so a torn-down session leaks nothing. */
|
||||
@@ -99,19 +154,59 @@ public final class PrimaryRegistry {
|
||||
* recorded delegation there is no right answer, so this returns empty rather than guessing —
|
||||
* delivery degrades to pull, which is exactly what the durable inbox is for, instead of
|
||||
* interrupting the wrong lead with someone else's result.
|
||||
*
|
||||
* <p>A delegation recorded with a name is resolved to that lead's <em>current</em> terminal
|
||||
* first — see {@link #currentTerminalForName} — so a lead that has since been rolled is still
|
||||
* reachable here, not just the pane that delegated the work originally.
|
||||
*/
|
||||
public Optional<String> nudgeTargetFor(String target) {
|
||||
String lead = target == null ? null : leadByTarget.get(target);
|
||||
return lead != null ? Optional.of(lead) : Optional.ofNullable(terminal.get());
|
||||
Delegation delegation = target == null ? null : leadByTarget.get(target);
|
||||
if (delegation != null) {
|
||||
return Optional.of(resolveCurrent(delegation.terminal(), delegation.name()));
|
||||
}
|
||||
return currentPrimaryTerminal();
|
||||
}
|
||||
|
||||
/** The known primary terminal, or empty if not yet learned (and not pinned). */
|
||||
/**
|
||||
* The known primary terminal, or empty if not yet learned (and not pinned) — the raw value as
|
||||
* it was recorded, with no attempt to resolve a named lead's current pane. Callers that need a
|
||||
* nudge destination which survives a lead roll want {@link #currentPrimaryTerminal()} instead.
|
||||
*/
|
||||
public Optional<String> primaryTerminal() {
|
||||
return Optional.ofNullable(terminal.get());
|
||||
}
|
||||
|
||||
/**
|
||||
* The terminal to nudge for the singleton primary right now: the recorded name resolved to its
|
||||
* current terminal when one was recorded and is still a recognised lead, otherwise the terminal
|
||||
* that was actually recorded — empty only when nothing has been learned or pinned at all.
|
||||
*/
|
||||
public Optional<String> currentPrimaryTerminal() {
|
||||
String learned = terminal.get();
|
||||
if (learned == null) {
|
||||
return Optional.empty();
|
||||
}
|
||||
return Optional.of(resolveCurrent(learned, primaryName.get()));
|
||||
}
|
||||
|
||||
/** {@code true} once a terminal has been recorded (or was pinned at construction). */
|
||||
public boolean isKnown() {
|
||||
return terminal.get() != null;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code learnedTerminal}, unless {@code name} is non-null and {@code currentTerminalForName}
|
||||
* currently places that name at a different, live terminal — in which case the live one wins.
|
||||
*/
|
||||
private String resolveCurrent(String learnedTerminal, String name) {
|
||||
if (name == null) {
|
||||
return learnedTerminal;
|
||||
}
|
||||
String current = currentTerminalForName.apply(name);
|
||||
return current != null && !current.isBlank() ? current : learnedTerminal;
|
||||
}
|
||||
|
||||
private static String blankToNull(String s) {
|
||||
return s == null || s.isBlank() ? null : s;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -445,27 +445,6 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
+ " distinct candidate(s): " + String.join(", ", unreachable));
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse an explicit-profile spawn when the profile is at its {@code maxLoad} cap.
|
||||
*
|
||||
* <p>maxLoad is a documented, unconditional capacity limit (see {@code FleetConfig.Profile#maxLoad}),
|
||||
* and the charter makes explicit-profile spawns the normal path — so enforcing it only in placement
|
||||
* ({@code PlacementPolicyUtil}, package-private, hence not linked) would leave the cap dead config
|
||||
* on every call that names a profile. Same rule as placement: {@code live >= cap} is at capacity.
|
||||
*
|
||||
* <p>Deliberately no fallback to another profile: the caller named {@code profile} for a cost/model
|
||||
* reason, and silently re-routing a paid-tier (subscription) request elsewhere is worse than
|
||||
* refusing it. A caller that wants placement should omit the profile and let the policy pick.
|
||||
*
|
||||
* <p>Known TOCTOU limitation — documented, not fixed. {@link #liveCount} is read outside any lock and
|
||||
* {@code SessionManager} registers a session only after {@code launcher.spawn} returns, so two
|
||||
* genuinely concurrent spawns can both pass this check. The race already exists on the placement
|
||||
* path. Closing it needs slot reservation in the registry; serializing spawn here would block on
|
||||
* the readiness gate and is a far worse trade.
|
||||
*
|
||||
* @param profile the profile the caller explicitly named
|
||||
* @throws PlacementException when the profile is at capacity
|
||||
*/
|
||||
/**
|
||||
* Refuse an explicit-profile spawn whose credential is quarantined (CB-578 stage B): a prior
|
||||
* {@code BACKEND_EXHAUSTED} classification on this profile, or on another profile sharing its
|
||||
@@ -527,6 +506,27 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse an explicit-profile spawn when the profile is at its {@code maxLoad} cap.
|
||||
*
|
||||
* <p>maxLoad is a documented, unconditional capacity limit (see {@code FleetConfig.Profile#maxLoad}),
|
||||
* and the charter makes explicit-profile spawns the normal path — so enforcing it only in placement
|
||||
* ({@code PlacementPolicyUtil}, package-private, hence not linked) would leave the cap dead config
|
||||
* on every call that names a profile. Same rule as placement: {@code live >= cap} is at capacity.
|
||||
*
|
||||
* <p>No fallback to another profile: the caller named {@code profile} for a cost/model
|
||||
* reason, and silently re-routing a paid-tier (subscription) request elsewhere is worse than
|
||||
* refusing it. A caller that wants placement should omit the profile and let the policy pick.
|
||||
*
|
||||
* <p>Known TOCTOU limitation — documented, not fixed. {@link #liveCount} is read outside any lock and
|
||||
* {@code SessionManager} registers a session only after {@code launcher.spawn} returns, so two
|
||||
* genuinely concurrent spawns can both pass this check. The race already exists on the placement
|
||||
* path. Closing it needs slot reservation in the registry; serializing spawn here would block on
|
||||
* the readiness gate and is a far worse trade.
|
||||
*
|
||||
* @param profile the profile the caller explicitly named
|
||||
* @throws PlacementException when the profile is at capacity
|
||||
*/
|
||||
private void enforceMaxLoad(String profile) {
|
||||
// Absent config, or a config whose maxLoad normalized to null (ABSENT ⇒ unlimited at load),
|
||||
// means no cap — never cap what wasn't configured. Note "non-positive ⇒ unlimited" was true
|
||||
|
||||
@@ -474,7 +474,6 @@ public final class EnvAllowListScrub {
|
||||
}
|
||||
}
|
||||
|
||||
/** Best-effort recursive delete; failures are swallowed — JVM-exit cleanup is the backstop. */
|
||||
/**
|
||||
* Remove generated directories left behind by an earlier daemon process.
|
||||
*
|
||||
@@ -511,6 +510,7 @@ public final class EnvAllowListScrub {
|
||||
}
|
||||
}
|
||||
|
||||
/** Best-effort recursive delete; failures are swallowed — JVM-exit cleanup is the backstop. */
|
||||
static void deleteRecursively(Path dir) {
|
||||
if (dir == null || !Files.exists(dir)) {
|
||||
return;
|
||||
|
||||
@@ -6,6 +6,7 @@ import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.ResilientAgentLaunch;
|
||||
import dev.ltms.fleet.herdr.Tab;
|
||||
import dev.ltms.fleet.herdr.Workspace;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
@@ -67,16 +68,6 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
|
||||
/** herdr rejects a duplicate agent {@code name}; we retry a bumped name this many times. */
|
||||
private static final int NAME_RETRIES = 8;
|
||||
|
||||
/**
|
||||
* Retries for {@code agent.start} against a seed pane whose shell has not reached its prompt
|
||||
* yet — {@code tab.create}/{@code pane.split} return as soon as the pane exists, and herdr
|
||||
* refuses to start an agent in a pane that is not "an available shell" ({@code agent_pane_busy}).
|
||||
*/
|
||||
private static final int SHELL_READY_RETRIES = 20;
|
||||
|
||||
private final String namePrefix; // label prefix: naming + reap scheme
|
||||
private final AgentControl agents;
|
||||
private final WorkspaceControl spaces;
|
||||
@@ -766,90 +757,26 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
// Protocol 19 resolves the executable from the agent kind (== namePrefix here), so
|
||||
// argv[0] — the configured executable — is dropped and only the extra args are passed.
|
||||
List<String> args = argv.isEmpty() ? argv : argv.subList(1, argv.size());
|
||||
checkPaneCommandFits(cfg, argv);
|
||||
HerdrException last = null;
|
||||
for (int attempt = 0; attempt < NAME_RETRIES; attempt++) {
|
||||
long seq = nameSeq.incrementAndGet();
|
||||
String name = namePrefix + "-" + cfg.profile() + "-" + nameNonce + "-" + seq;
|
||||
try {
|
||||
return new Started(startAwaitingShellPrompt(name, args, paneId), seq);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_name_taken".equals(e.code())) throw e;
|
||||
log.debug("peer name '{}' taken, retrying", name);
|
||||
last = e;
|
||||
}
|
||||
try {
|
||||
ResilientAgentLaunch.checkFits(cfg.profile(), argv);
|
||||
} catch (ResilientAgentLaunch.TooLargeException e) {
|
||||
throw new PeerUnreachableException(e.getMessage());
|
||||
}
|
||||
throw last;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #220: herdr does not exec the launch command — it TYPES it into the pane as one line,
|
||||
* and a pty line buffer holds only {@value #PANE_COMMAND_BYTE_LIMIT} bytes (BSD/macOS {@code
|
||||
* MAX_CANON}). Everything past that byte is dropped. Nothing reports it: herdr answers "agent
|
||||
* started", the backend exits on the mangled argument it was handed, the pane closes, and the
|
||||
* only symptom is {@link #waitUntilInjectableOrThrow} timing out 20 seconds later with no
|
||||
* reason. That is exactly how #214 broke every claude-code spawn — one 50-byte flag pushed a
|
||||
* 978-byte command to 1028, and the tail that got cut was {@code --autocompact 250000}.
|
||||
*
|
||||
* <p>So measure it here and refuse, loudly and immediately, rather than spawn something that
|
||||
* cannot work. The estimate is deliberately conservative: fleetd cannot see herdr's quoting, so
|
||||
* every argument is charged its own bytes plus a separator and a quote pair. An over-estimate
|
||||
* costs a clear error at a length that was already unsafe; an under-estimate would let the
|
||||
* silent truncation back in.
|
||||
*
|
||||
* @throws PeerUnreachableException when the command cannot fit — the same failure the spawn
|
||||
* would have hit anyway, named at the point it is still
|
||||
* explainable
|
||||
*/
|
||||
private void checkPaneCommandFits(FleetConfig.Profile cfg, List<String> argv) {
|
||||
int bytes = 0;
|
||||
String longest = null;
|
||||
int longestBytes = 0;
|
||||
for (String arg : argv) {
|
||||
int argBytes = arg == null ? 0 : arg.getBytes(java.nio.charset.StandardCharsets.UTF_8).length;
|
||||
bytes += argBytes + QUOTING_OVERHEAD_PER_ARG;
|
||||
if (argBytes > longestBytes) {
|
||||
longestBytes = argBytes;
|
||||
longest = arg;
|
||||
}
|
||||
}
|
||||
if (bytes <= PANE_COMMAND_BYTE_LIMIT) {
|
||||
return;
|
||||
}
|
||||
String culprit = longest == null ? "<none>"
|
||||
: longest.substring(0, Math.min(longest.length(), 60)) + (longest.length() > 60 ? "…" : "");
|
||||
throw new PeerUnreachableException(
|
||||
"launch command for profile " + cfg.profile() + " is about " + bytes + " bytes, over the "
|
||||
+ PANE_COMMAND_BYTE_LIMIT + "-byte limit of the pane line herdr types it into. "
|
||||
+ "The pty would drop the tail silently and the backend would exit on a mangled "
|
||||
+ "argument. Longest argument is " + longestBytes + " bytes: " + culprit
|
||||
+ " — move it off the command line (a file flag) or shorten it.");
|
||||
long[] lastSeq = {0};
|
||||
Agent agent = ResilientAgentLaunch.startUniquelyNamed(agents, namePrefix, args, paneId,
|
||||
attempt -> {
|
||||
lastSeq[0] = nameSeq.incrementAndGet();
|
||||
return namePrefix + "-" + cfg.profile() + "-" + nameNonce + "-" + lastSeq[0];
|
||||
},
|
||||
ResilientAgentLaunch.NAME_RETRIES, ResilientAgentLaunch.SHELL_READY_RETRIES, sleeper);
|
||||
return new Started(agent, lastSeq[0]);
|
||||
}
|
||||
|
||||
/**
|
||||
* The pty line buffer herdr types a launch command into: BSD/macOS {@code MAX_CANON}. Not a
|
||||
* fleetd choice and not configurable — see {@link #checkPaneCommandFits}.
|
||||
* fleetd choice and not configurable — see {@link ResilientAgentLaunch#checkFits}.
|
||||
*/
|
||||
static final int PANE_COMMAND_BYTE_LIMIT = 1024;
|
||||
|
||||
/** Per-argument allowance for the separating space and a shell quote pair fleetd cannot see. */
|
||||
private static final int QUOTING_OVERHEAD_PER_ARG = 3;
|
||||
|
||||
/** Start the agent into {@code paneId}, waiting out the seed shell's boot with the sleeper. */
|
||||
private Agent startAwaitingShellPrompt(String name, List<String> args, String paneId) {
|
||||
HerdrException busy = null;
|
||||
for (int attempt = 0; attempt < SHELL_READY_RETRIES; attempt++) {
|
||||
try {
|
||||
return agents.start(name, namePrefix, args, paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_pane_busy".equals(e.code())) throw e;
|
||||
log.debug("pane {} not at its shell prompt yet, retrying agent.start", paneId);
|
||||
busy = e;
|
||||
sleeper.run();
|
||||
}
|
||||
}
|
||||
throw busy;
|
||||
}
|
||||
static final int PANE_COMMAND_BYTE_LIMIT = ResilientAgentLaunch.PANE_COMMAND_BYTE_LIMIT;
|
||||
|
||||
// --- discovery + reap ----------------------------------------------------------------------
|
||||
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
/**
|
||||
* fleetd #612 A-gaps (gap 1): a {@link LeadChannel} that its owner can also close.
|
||||
*
|
||||
* <p>{@link LeadChannel}'s own javadoc says plainly that {@code close()} is deliberately left out
|
||||
* of that interface — draining is a caller convenience nobody uses, and closing is the
|
||||
* <em>owner's</em> job. This interface is that owner's own, wider view: whoever opens the
|
||||
* coordination mailbox (the assembly that builds the daemon) also needs to close it from the
|
||||
* shutdown path, and a test standing in for a real broker connection needs a fake it can mark
|
||||
* closed, without ever holding a live connection. Every ordinary consumer ({@code FleetMcp},
|
||||
* {@link LeadCoordLoop}) keeps taking the narrower {@link LeadChannel} exactly as before — only
|
||||
* the owner speaks this wider one.
|
||||
*
|
||||
* <p>{@link LeadMailbox} is still the only production implementation. This only generalises the
|
||||
* TYPE its owner holds it as (previously the concrete class), so a test can substitute a fake
|
||||
* closeable channel instead of a real AMQP connection.
|
||||
*/
|
||||
public interface LeadChannelHandle extends LeadChannel, AutoCloseable {
|
||||
|
||||
/** Release the underlying connection. Declared with no checked exception, unlike the plain {@link AutoCloseable#close()}. */
|
||||
@Override
|
||||
void close();
|
||||
}
|
||||
@@ -2,6 +2,7 @@ package dev.ltms.fleet.msg;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.PromptBox;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -23,7 +24,8 @@ import java.util.function.Supplier;
|
||||
* <p><strong>Status-gated, exactly like {@link ReplyPushLoop}.</strong> A pane may only be injected
|
||||
* into at a turn boundary ({@link AgentStatus#injectable()} — idle, blocked or done); pasting into
|
||||
* a live turn corrupts it. So a tick that finds the lead busy simply does nothing and comes back
|
||||
* later.
|
||||
* later. The same holds for a lead whose prompt box holds unsubmitted text ({@link PromptBox}) —
|
||||
* delivering there would submit the operator's half-typed line along with the message.
|
||||
*
|
||||
* <p><strong>Ack only after delivery.</strong> A message is acked — removed from the broker — only
|
||||
* once {@link AgentControl#send} has actually put it in the pane. Anything not delivered (no lead
|
||||
@@ -52,6 +54,7 @@ public final class LeadCoordLoop {
|
||||
|
||||
private final LeadChannel channel;
|
||||
private final AgentControl agents;
|
||||
private final PromptBox promptBox;
|
||||
private final Supplier<Map<String, String>> leads;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final long intervalMs;
|
||||
@@ -73,6 +76,7 @@ public final class LeadCoordLoop {
|
||||
ScheduledExecutorService scheduler, long intervalMs) {
|
||||
this.channel = channel;
|
||||
this.agents = agents;
|
||||
this.promptBox = new PromptBox(agents);
|
||||
this.leads = leads;
|
||||
this.scheduler = scheduler;
|
||||
this.intervalMs = intervalMs;
|
||||
@@ -149,6 +153,11 @@ public final class LeadCoordLoop {
|
||||
lead, status, held.size());
|
||||
return;
|
||||
}
|
||||
if (!promptBox.clearToSubmit(lead)) {
|
||||
log.debug("lead coordination: lead {} has unsubmitted text in its prompt box, holding {} message(s)",
|
||||
lead, held.size());
|
||||
return;
|
||||
}
|
||||
try {
|
||||
agents.send(lead, DELIVERY_FORMAT.formatted(msg.from(), msg.content()));
|
||||
} catch (RuntimeException e) {
|
||||
|
||||
@@ -2,6 +2,7 @@ package dev.ltms.fleet.msg;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.PromptBox;
|
||||
import dev.ltms.fleet.lead.LeadContextGauge;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
@@ -33,7 +34,7 @@ import java.util.function.Supplier;
|
||||
* no such block it is never constructed, so upgrading the daemon cannot silently acquire a behaviour
|
||||
* that spends the operator's model subscription on its own initiative (constraint 1).
|
||||
*
|
||||
* <p>Four invariants keep it from becoming a runaway subscription burner:
|
||||
* <p>Five invariants keep it from becoming a runaway subscription burner:
|
||||
* <ol>
|
||||
* <li><b>Status-gated</b> — a {@code WORKING} lead is making progress and is never touched; only an
|
||||
* injectable (idle/done/blocked) lead is even considered (constraint 2).</li>
|
||||
@@ -45,6 +46,8 @@ import java.util.function.Supplier;
|
||||
* <li><b>Never races {@link ReplyPushLoop}</b> — while that loop is actively nudging any target this
|
||||
* loop stands down, so two competing injections never start two turns in the same pane
|
||||
* (constraint 6).</li>
|
||||
* <li><b>Never submits the operator's draft</b> — a nudge is held while the lead's prompt box holds
|
||||
* unsubmitted text ({@link PromptBox}), because the delivery pastes and submits in one call.</li>
|
||||
* </ol>
|
||||
*
|
||||
* <p><b>fleetd #609 — context-high notice.</b> Optionally ({@code contextHighNudge}, opt-in like the
|
||||
@@ -65,6 +68,7 @@ public final class LeadHeartbeatLoop {
|
||||
|
||||
private final PrimaryRegistry primaryRegistry;
|
||||
private final AgentControl agents;
|
||||
private final PromptBox promptBox;
|
||||
private final ReplyInbox inbox;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final ReplyPushLoop pushLoop;
|
||||
@@ -76,6 +80,7 @@ public final class LeadHeartbeatLoop {
|
||||
private final Metrics metrics; // CB-512 pattern: nullable — no registry in unit tests
|
||||
private final LeadContextSource contextSource; // fleetd #609
|
||||
private final boolean contextHighNudge; // fleetd #609: opt-in, like the loop itself
|
||||
private final boolean requireOperatorConfirm; // fleetd #621: mirrors leadRollover.requireOperatorConfirm
|
||||
|
||||
/** When the current idle stretch began (nanos), or {@link #NOT_IDLE}. Single scheduler thread only. */
|
||||
private long idleSinceNanos = NOT_IDLE;
|
||||
@@ -106,14 +111,35 @@ public final class LeadHeartbeatLoop {
|
||||
* fleetd #609: as above, plus the lead's own context source and whether a HIGH reading should
|
||||
* append a hand-over notice to the loop's nudge. Pass {@link LeadContextSource#none()} and
|
||||
* {@code false} to keep the pre-#609 behaviour exactly (both existing public constructors do).
|
||||
*
|
||||
* <p>fleetd #621: delegates to the full constructor with {@code requireOperatorConfirm=true} —
|
||||
* the pre-#621 wording ("ask the operator ... only the operator can approve the roll") assumed
|
||||
* the config default, so every caller of this overload keeps that text byte-identical.
|
||||
*/
|
||||
public LeadHeartbeatLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
Supplier<List<MemberSession>> roster, ReplyPushLoop pushLoop,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock,
|
||||
long idleAfterNanos, long backoffMs, int quietNudgeCap, Metrics metrics,
|
||||
LeadContextSource contextSource, boolean contextHighNudge) {
|
||||
this(primaryRegistry, agents, inbox, roster, pushLoop, scheduler, clock,
|
||||
idleAfterNanos, backoffMs, quietNudgeCap, metrics, contextSource, contextHighNudge, true);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #621: as above, plus the daemon's effective {@code leadRollover.requireOperatorConfirm}
|
||||
* value — threaded into {@link #contextNotice(boolean, LeadContextGauge.Reading, boolean, boolean)}
|
||||
* so the notice's wording tracks the config the daemon actually enforces (see {@code
|
||||
* LeadRollover.confirm}) instead of always asserting the operator gate is on.
|
||||
*/
|
||||
public LeadHeartbeatLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
Supplier<List<MemberSession>> roster, ReplyPushLoop pushLoop,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock,
|
||||
long idleAfterNanos, long backoffMs, int quietNudgeCap, Metrics metrics,
|
||||
LeadContextSource contextSource, boolean contextHighNudge,
|
||||
boolean requireOperatorConfirm) {
|
||||
this.primaryRegistry = primaryRegistry;
|
||||
this.agents = agents;
|
||||
this.promptBox = new PromptBox(agents);
|
||||
this.inbox = inbox;
|
||||
this.roster = roster;
|
||||
this.pushLoop = pushLoop;
|
||||
@@ -125,6 +151,7 @@ public final class LeadHeartbeatLoop {
|
||||
this.metrics = metrics;
|
||||
this.contextSource = contextSource;
|
||||
this.contextHighNudge = contextHighNudge;
|
||||
this.requireOperatorConfirm = requireOperatorConfirm;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -165,7 +192,9 @@ public final class LeadHeartbeatLoop {
|
||||
/** Idle past the quiet period with nothing pending and the cap exhausted — stop until new state appears. */
|
||||
QUIET_DONE,
|
||||
/** {@link ReplyPushLoop} is actively nudging — stand aside rather than start a competing turn. */
|
||||
STAND_DOWN
|
||||
STAND_DOWN,
|
||||
/** The lead's prompt box holds unsubmitted text — hold the nudge rather than submit that text. */
|
||||
DRAFT_HELD
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -285,12 +314,12 @@ public final class LeadHeartbeatLoop {
|
||||
* (mirroring {@link ReplyPushLoop#tick(String)}) so tests can drive it directly with a fake clock and a
|
||||
* fake {@link AgentControl} instead of racing the scheduler thread. */
|
||||
void tick() {
|
||||
boolean leadKnown = primaryRegistry.primaryTerminal().isPresent();
|
||||
boolean leadKnown = primaryRegistry.currentPrimaryTerminal().isPresent();
|
||||
FleetState fleet = snapshot(inbox, roster);
|
||||
AgentStatus status = AgentStatus.UNKNOWN;
|
||||
LeadContextGauge.Reading reading = LeadContextGauge.Reading.unknown();
|
||||
if (leadKnown) {
|
||||
String leadTerminal = primaryRegistry.primaryTerminal().orElseThrow();
|
||||
String leadTerminal = primaryRegistry.currentPrimaryTerminal().orElseThrow();
|
||||
try {
|
||||
status = agents.status(leadTerminal);
|
||||
} catch (RuntimeException e) {
|
||||
@@ -308,6 +337,7 @@ public final class LeadHeartbeatLoop {
|
||||
idleSinceNanos == NOT_IDLE ? null : idleSinceNanos,
|
||||
quietCount, status, pushLoop.isActive(), leadKnown, fleet,
|
||||
reading.state(), contextNotified);
|
||||
d = holdIfOperatorIsTyping(d);
|
||||
applyDecision(d);
|
||||
switch (d.action()) {
|
||||
case INJECT -> injectNudge(d, fleet, reading);
|
||||
@@ -315,11 +345,33 @@ public final class LeadHeartbeatLoop {
|
||||
countNudge("exhausted");
|
||||
contextNotified = d.contextNotified();
|
||||
}
|
||||
case WAIT_IDLE, LEAD_BUSY, STAND_DOWN -> contextNotified = d.contextNotified();
|
||||
case WAIT_IDLE, LEAD_BUSY, STAND_DOWN, DRAFT_HELD -> contextNotified = d.contextNotified();
|
||||
}
|
||||
scheduleNext();
|
||||
}
|
||||
|
||||
/**
|
||||
* Turn a decision to inject into {@link Action#DRAFT_HELD} when the lead's prompt box holds text
|
||||
* the operator has not submitted. The pane read happens only for a decision that would otherwise
|
||||
* send, so a busy or debouncing lead costs no extra herdr call.
|
||||
*
|
||||
* <p>The held decision carries this tick's idle window but the <em>pre-tick</em> quiet count and
|
||||
* context latch: nothing reached the pane, so neither the quiet budget nor the one context notice
|
||||
* per stretch may be spent on it.
|
||||
*/
|
||||
private Decision holdIfOperatorIsTyping(Decision d) {
|
||||
if (d.action() != Action.INJECT) {
|
||||
return d;
|
||||
}
|
||||
var lead = primaryRegistry.currentPrimaryTerminal();
|
||||
if (lead.isEmpty() || promptBox.clearToSubmit(lead.get())) {
|
||||
return d;
|
||||
}
|
||||
log.debug("idle-heartbeat: lead {} has unsubmitted text in its prompt box, holding the nudge",
|
||||
lead.get());
|
||||
return new Decision(Action.DRAFT_HELD, d.idleSinceNanos(), quietCount, contextNotified);
|
||||
}
|
||||
|
||||
/**
|
||||
* Persist the idle/quiet state a decision returned, so the next tick starts from it.
|
||||
*
|
||||
@@ -344,8 +396,8 @@ public final class LeadHeartbeatLoop {
|
||||
// d.contextNotified() is the value to persist once delivery is confirmed, not the value the text
|
||||
// itself should be built from. Otherwise a HIGH stretch that is still latched would never see the
|
||||
// notice at all, defeating the very check this fixes.
|
||||
String notice = contextNotice(contextHighNudge, reading, contextNotified);
|
||||
var lead = primaryRegistry.primaryTerminal();
|
||||
String notice = contextNotice(contextHighNudge, reading, contextNotified, requireOperatorConfirm);
|
||||
var lead = primaryRegistry.currentPrimaryTerminal();
|
||||
boolean sent = lead.isPresent() && trySend(lead.get(), fleet.nudgeText() + notice, notice);
|
||||
// The latch becomes true only when all three hold: decide() chose to notify, a notice was
|
||||
// actually included in the text, and the send reached the pane without throwing. Whenever no
|
||||
@@ -381,8 +433,9 @@ public final class LeadHeartbeatLoop {
|
||||
/**
|
||||
* fleetd #609: the text appended to a nudge when the lead's own context is full — {@code ""}
|
||||
* whenever the notice does not apply, so callers can unconditionally append this without an extra
|
||||
* branch. Wording stays plain (CEFR B1) and honest that only the operator approves a roll — this
|
||||
* loop only ever prints text, it never calls {@code fleet_handover} itself.
|
||||
* branch. Wording stays plain (CEFR B1) and honest about who actually gates the roll — see the
|
||||
* {@code requireOperatorConfirm} overload (fleetd #621) for which check that is. This loop only
|
||||
* ever prints text, it never calls {@code fleet_handover} itself.
|
||||
*
|
||||
* @param enabled the {@code leadHeartbeat.contextHighNudge} config flag
|
||||
* @param reading the lead's current {@link LeadContextGauge} reading
|
||||
@@ -401,9 +454,33 @@ public final class LeadHeartbeatLoop {
|
||||
* closing sentence ("You will not be told again until your context reads ok.") false. {@link
|
||||
* #injectNudge} is the only caller that passes a non-default {@code alreadyNotified}.
|
||||
*
|
||||
* <p>fleetd #621: delegates with {@code requireOperatorConfirm=true} — the pre-#621 default and the
|
||||
* value every existing caller of this overload (including every test written before #621) already
|
||||
* assumed, so the text this overload returns stays byte-identical.
|
||||
*
|
||||
* @param alreadyNotified whether the lead has already been told about the current HIGH stretch
|
||||
*/
|
||||
static String contextNotice(boolean enabled, LeadContextGauge.Reading reading, boolean alreadyNotified) {
|
||||
return contextNotice(enabled, reading, alreadyNotified, true);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #621: as {@link #contextNotice(boolean, LeadContextGauge.Reading, boolean)}, but the closing
|
||||
* instructions also track the daemon's effective {@code leadRollover.requireOperatorConfirm} value,
|
||||
* instead of always asserting that only the operator can approve the roll.
|
||||
*
|
||||
* <p>{@code LeadRollover.confirm(...)} already honours this flag: when it is {@code false}, the daemon
|
||||
* itself gates the roll on the three handover-file checks alone (exists, modified after the {@code
|
||||
* open()} request, and no older than {@code maxDocAgeSeconds}) and never consults {@code
|
||||
* operatorConfirmed}. Before this parameter existed, this notice told the lead to ask the operator
|
||||
* regardless — so a lead that followed its own instructions asked anyway, and setting the config knob
|
||||
* to {@code false} stopped the daemon refusing the roll without stopping the operator being
|
||||
* interrupted. This parameter is how the text is kept honest about which gate is actually live.
|
||||
*
|
||||
* @param requireOperatorConfirm the effective {@code leadRollover.requireOperatorConfirm} value
|
||||
*/
|
||||
static String contextNotice(boolean enabled, LeadContextGauge.Reading reading, boolean alreadyNotified,
|
||||
boolean requireOperatorConfirm) {
|
||||
if (!enabled || alreadyNotified || reading.state() != LeadContextGauge.State.HIGH) {
|
||||
return "";
|
||||
}
|
||||
@@ -414,16 +491,24 @@ public final class LeadHeartbeatLoop {
|
||||
.append(reading.compactions()).append(' ').append(compactionWord).append(" so far.");
|
||||
} else {
|
||||
// A HIGH reading always carries a non-null token count today: LeadContextGauge only
|
||||
// reaches HIGH by comparing a number against HIGH_THRESHOLD_TOKENS. That invariant
|
||||
// lives in another class and nothing asserts it, so this branch does not rely on it —
|
||||
// it drops the token clause rather than printing "null tokens".
|
||||
// reaches HIGH by comparing a number against a threshold. That invariant lives in
|
||||
// another class and nothing asserts it, so this branch does not rely on it — it drops
|
||||
// the token clause rather than printing "null tokens".
|
||||
sb.append(" (").append(reading.compactions()).append(' ').append(compactionWord)
|
||||
.append(" so far).");
|
||||
}
|
||||
sb.append(" A fresh session would work better. To hand over: call fleet_handover(action=\"open\"), "
|
||||
+ "write the file it names, ask the operator, then call fleet_handover(action=\"confirm\", "
|
||||
+ "token, operatorConfirmed). Only the operator can approve the roll. You will not be told "
|
||||
+ "again until your context reads ok.");
|
||||
if (requireOperatorConfirm) {
|
||||
sb.append(" A fresh session would work better. To hand over: call fleet_handover(action=\"open\"), "
|
||||
+ "write the file it names, ask the operator, then call fleet_handover(action=\"confirm\", "
|
||||
+ "token, operatorConfirmed). Only the operator can approve the roll. You will not be told "
|
||||
+ "again until your context reads ok.");
|
||||
} else {
|
||||
sb.append(" A fresh session would work better. To hand over: call fleet_handover(action=\"open\"), "
|
||||
+ "write the file it names, then call fleet_handover(action=\"confirm\", token). Decide for "
|
||||
+ "yourself when to confirm: the roll goes through if the handover file exists, was "
|
||||
+ "changed after you opened it, and is not older than maxDocAgeSeconds. You will not be "
|
||||
+ "told again until your context reads ok.");
|
||||
}
|
||||
return sb.toString();
|
||||
}
|
||||
|
||||
|
||||
@@ -63,7 +63,7 @@ import java.util.concurrent.TimeoutException;
|
||||
* which messages reached a lead. Any publish still awaiting its confirm is failed rather than left to idle out
|
||||
* the confirm timeout against a sequence number that means nothing on the new channel.
|
||||
*/
|
||||
public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
public final class LeadMailbox implements LeadChannelHandle {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadMailbox.class);
|
||||
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import dev.ltms.fleet.auth.Principal;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||
@@ -11,6 +12,7 @@ import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.CompletionException;
|
||||
@@ -87,7 +89,7 @@ public final class MessageService {
|
||||
/**
|
||||
* The worker paused mid-turn to ask the primary a question (CB-205); {@code text} is the
|
||||
* question and {@code turnId} correlates the answer. Not terminal — the primary answers with
|
||||
* {@link #answer(String, String, long)} and the turn resumes.
|
||||
* {@link #answer(String, String, long, String)} and the turn resumes.
|
||||
*/
|
||||
QUESTION,
|
||||
/** Timed out after the message was delivered — the worker is still working. */
|
||||
@@ -120,10 +122,16 @@ public final class MessageService {
|
||||
/** Another send to this session was in flight for the whole window. */
|
||||
BUSY,
|
||||
/**
|
||||
* An answer ({@link #answer(String, String, long)}) referenced a {@code turnId} that is no
|
||||
* longer open — the worker's {@code fleet_ask} already timed out or was answered.
|
||||
* An answer ({@link #answer(String, String, long, String)}) referenced a {@code turnId}
|
||||
* that is no longer open — the worker's {@code fleet_ask} already timed out or was answered.
|
||||
*/
|
||||
STALE_TURN
|
||||
STALE_TURN,
|
||||
/**
|
||||
* An answer ({@link #answer(String, String, long, String)}) named a {@code turnId} that is
|
||||
* still open, but the answering caller is not the caller whose accepted delegation opened
|
||||
* it. Distinct from {@link #STALE_TURN} so a refusal is never reported as a lapsed turn.
|
||||
*/
|
||||
NOT_TURN_OWNER
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -133,7 +141,7 @@ public final class MessageService {
|
||||
* {@link Outcome#COMPLETED_UNREPLIED}), or the question for {@link Outcome#QUESTION},
|
||||
* else {@code null}
|
||||
* @param turnId correlation id for a {@link Outcome#QUESTION} (answered via
|
||||
* {@link #answer(String, String, long)}), else {@code null}
|
||||
* {@link #answer(String, String, long, String)}), else {@code null}
|
||||
*/
|
||||
public record Reply(Outcome outcome, String text, String turnId) {
|
||||
/** A reply with no correlation id (the common terminal outcomes). */
|
||||
@@ -275,10 +283,16 @@ public final class MessageService {
|
||||
* {@link #abandon}) can never match again regardless of this flag's value.
|
||||
*/
|
||||
private volatile boolean askTimedOut;
|
||||
/**
|
||||
* The owner key of the caller whose {@code fleet_send{wait:false}} created this ticket, or
|
||||
* {@code null} for the unnamed primary and overloads that do not record a caller.
|
||||
*/
|
||||
private final String creatorOwner;
|
||||
|
||||
private Task(String ticket, String target, LongSupplier nowNanos) {
|
||||
private Task(String ticket, String target, LongSupplier nowNanos, String creatorOwner) {
|
||||
this.ticket = ticket;
|
||||
this.target = target;
|
||||
this.creatorOwner = creatorOwner;
|
||||
this.createdNanos = nowNanos.getAsLong();
|
||||
future.whenComplete((reply, ex) -> completedNanos = nowNanos.getAsLong());
|
||||
}
|
||||
@@ -340,6 +354,14 @@ public final class MessageService {
|
||||
*/
|
||||
private final ConcurrentHashMap<String, Boolean> queuedDeliveries = new ConcurrentHashMap<>();
|
||||
private final AtomicLong ticketSeq = new AtomicLong();
|
||||
/**
|
||||
* Minted once per {@code MessageService} instance and folded into every ticket id (see
|
||||
* {@link #sendAsync(String, String, Runnable, Principal)}). {@link #ticketSeq} alone restarts at
|
||||
* zero for every instance, so without this a ticket id can be reused across instances and
|
||||
* resolve to an unrelated {@link Task} with no error; this nonce makes that impossible, because
|
||||
* an id minted by one instance can never match the id space of another.
|
||||
*/
|
||||
private final String ticketBootNonce = UUID.randomUUID().toString().substring(0, 6);
|
||||
private final ExecutorService asyncExecutor = Executors.newThreadPerTaskExecutor(
|
||||
Thread.ofVirtual().name("bridge-async-", 0).factory());
|
||||
|
||||
@@ -492,6 +514,20 @@ public final class MessageService {
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Hand {@code session} every message queued for it that it has not collected yet, and record
|
||||
* that it collects its own mail. While that record is fresh, delivery to that session is
|
||||
* offered for collection instead of typed into its terminal; once it goes stale, the terminal
|
||||
* route takes over again with nothing lost.
|
||||
*
|
||||
* <p>The messages are returned in the order they were queued, and are removed by this call.
|
||||
* An empty list is an ordinary answer: a session polling on a timer keeps itself collecting
|
||||
* between messages.
|
||||
*/
|
||||
public List<String> collectInbox(String session) {
|
||||
return injector.collectInbox(session);
|
||||
}
|
||||
|
||||
/**
|
||||
* Route a worker's explicit {@code fleet_reply}: resolve an open send, complete an async ticket
|
||||
* still parked waiting on this exact turn's answer, or — only once neither applies — queue it in
|
||||
@@ -656,7 +692,7 @@ public final class MessageService {
|
||||
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, TIMED_OUT_UNCONFIRMED, BUSY -> "timeout";
|
||||
case WORKER_FAILED -> "failed";
|
||||
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
||||
case STALE_TURN, QUESTION -> null; // not a completed delegation
|
||||
case STALE_TURN, QUESTION, NOT_TURN_OWNER -> null; // not a completed delegation
|
||||
};
|
||||
}
|
||||
|
||||
@@ -915,14 +951,17 @@ public final class MessageService {
|
||||
|
||||
/**
|
||||
* Deliver {@code content} to {@code target} (a herdr {@code terminal_id}) and block until the
|
||||
* worker replies via {@link Rendezvous} or {@code timeoutMillis} elapses.
|
||||
* worker replies via {@link Rendezvous} or {@code timeoutMillis} elapses. {@code callerOwner}
|
||||
* identifies the caller making this call and is recorded as the turn's owner. It is the only
|
||||
* caller {@link #answer(String, String, long, String)} will
|
||||
* later accept an answer from if the worker pauses mid-turn to ask.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis) {
|
||||
return send(target, content, timeoutMillis, null);
|
||||
public Reply send(String target, String content, long timeoutMillis, String callerOwner) {
|
||||
return send(target, content, timeoutMillis, null, callerOwner);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #send(String, String, long)}, but with an accepted-delivery hook.
|
||||
* As {@link #send(String, String, long, String)}, but with an accepted-delivery hook.
|
||||
*
|
||||
* <p>{@code onAccepted} is invoked exactly once, once this send has won {@code target}'s send
|
||||
* lock and so become the <em>accepted target turn</em> — it runs <em>before</em> delivery is
|
||||
@@ -933,12 +972,13 @@ public final class MessageService {
|
||||
* acceptance means a concurrent sender that times out {@code BUSY} can never steal ownership it
|
||||
* never earned. {@code null} disables the hook.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis, Runnable onAccepted) {
|
||||
return send(target, content, timeoutMillis, onAccepted, null);
|
||||
public Reply send(String target, String content, long timeoutMillis, Runnable onAccepted, String callerOwner) {
|
||||
return send(target, content, timeoutMillis, onAccepted, null, callerOwner);
|
||||
}
|
||||
|
||||
/** Run a send, optionally stopping an async task that teardown already failed before acceptance. */
|
||||
private Reply send(String target, String content, long timeoutMillis, Runnable onAccepted, Task task) {
|
||||
private Reply send(String target, String content, long timeoutMillis, Runnable onAccepted, Task task,
|
||||
String callerOwner) {
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(target, _ -> new ReentrantLock());
|
||||
|
||||
@@ -958,7 +998,7 @@ public final class MessageService {
|
||||
// open race). Opening first also means a throwing onAccepted (fired before enqueue) or an
|
||||
// enqueue failure is safely closed by the finally below: nothing is left queued, and the
|
||||
// failed send leaves no stale waiter behind.
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(target);
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(target, Rendezvous.Owner.of(callerOwner));
|
||||
// CB-640: this send now owns target's delivery, so any earlier stranded-reply or
|
||||
// still-queued fact no longer describes the live state — clear both rather than let
|
||||
// them outlive the send that supersedes them.
|
||||
@@ -1137,19 +1177,30 @@ public final class MessageService {
|
||||
* mid-turn (already picked up), so the answer flows back through its own open {@code fleet_ask}
|
||||
* call, not a new status-gated delivery. The forward waiter is opened <em>before</em> the worker
|
||||
* is unblocked so a reply that lands the instant it resumes is not lost.
|
||||
*
|
||||
* <p>{@code callerOwner} identifies the caller making this call. It is checked against the
|
||||
* turn's recorded owner (the caller whose
|
||||
* accepted delegation opened it, see {@link #send(String, String, long, String)} and
|
||||
* {@link #sendAsync(String, String, Runnable, Principal)}) before anything else runs: a mismatch,
|
||||
* including a turn with no owner on record at all, returns {@link Outcome#NOT_TURN_OWNER}
|
||||
* without touching the rendezvous, the session lock, or any async task bookkeeping.
|
||||
*/
|
||||
public Reply answer(String turnId, String content, long timeoutMillis) {
|
||||
public Reply answer(String turnId, String content, long timeoutMillis, String callerOwner) {
|
||||
String workerSession = rendezvous.askSession(turnId);
|
||||
if (workerSession == null) {
|
||||
return new Reply(Outcome.STALE_TURN, null); // the ask lapsed (timed out or already answered)
|
||||
}
|
||||
Rendezvous.Owner owner = rendezvous.askOwner(turnId);
|
||||
if (!Rendezvous.Owner.permits(owner, callerOwner)) {
|
||||
return new Reply(Outcome.NOT_TURN_OWNER, null);
|
||||
}
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(workerSession, _ -> new ReentrantLock());
|
||||
if (!tryLock(lock, remainingMillis(deadlineNanos))) {
|
||||
return new Reply(Outcome.BUSY, null);
|
||||
}
|
||||
try {
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession);
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession, owner);
|
||||
// fleetd #575: this try used to open below, AFTER the Task lookup/registration and the
|
||||
// STALE_TURN early return that follows it — so that return was covered only by a
|
||||
// hand-rolled copy of the finally's own cleanup pair, not the finally itself. Widening the
|
||||
@@ -1279,8 +1330,20 @@ public final class MessageService {
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content, Runnable onAccepted) {
|
||||
String ticket = "task-" + ticketSeq.incrementAndGet();
|
||||
Task task = new Task(ticket, target, nowNanos);
|
||||
return sendAsync(target, content, onAccepted, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #sendAsync(String, String, Runnable)}, recording {@code creator}'s owner key as this
|
||||
* ticket's owner. The key is derived here from the resolved principal so callers cannot pass a
|
||||
* terminal address where an owner identity is required.
|
||||
*
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content, Runnable onAccepted, Principal creator) {
|
||||
String ticket = "task-" + ticketBootNonce + "-" + ticketSeq.incrementAndGet();
|
||||
String creatorOwner = creator == null ? null : creator.ownerKey();
|
||||
Task task = new Task(ticket, target, nowNanos, creatorOwner);
|
||||
tasks.put(ticket, task);
|
||||
if (pushLoop != null) {
|
||||
// CB-588: task.future only ever completes on a terminal phase (DONE or a failure) — a
|
||||
@@ -1311,7 +1374,7 @@ public final class MessageService {
|
||||
}
|
||||
asyncExecutor.submit(() -> {
|
||||
try {
|
||||
Reply result = send(target, content, ASYNC_TIMEOUT_MS, onAccepted, task);
|
||||
Reply result = send(target, content, ASYNC_TIMEOUT_MS, onAccepted, task, creatorOwner);
|
||||
if (result.outcome() == Outcome.QUESTION) {
|
||||
// Keep the accepted owner until answer() finishes it. markAsyncQuestion may run
|
||||
// just after resolveQuestion wakes this thread.
|
||||
@@ -1343,15 +1406,33 @@ public final class MessageService {
|
||||
}
|
||||
|
||||
/**
|
||||
* Snapshot the state of an async delegation. Returns {@code null} for an unknown/expired ticket;
|
||||
* otherwise a {@link Phase#PENDING} view (with the live worker status as detail), a
|
||||
* {@link Phase#DONE} view carrying the reply, or a {@link Phase#FAILED} view with the reason.
|
||||
* As {@link #poll(String, String)}, but bypasses the ownership check entirely via
|
||||
* {@link #INTERNAL_NO_OWNER_CHECK}. No production code calls this overload — it exists for
|
||||
* tests that only need the ticket's state and have no caller identity to pass.
|
||||
*/
|
||||
public TaskView poll(String ticket) {
|
||||
return poll(ticket, INTERNAL_NO_OWNER_CHECK);
|
||||
}
|
||||
|
||||
/**
|
||||
* Snapshot the state of an async delegation. Returns {@code null} for an unknown/expired ticket.
|
||||
* Refuses a {@code callerOwner} that differs from the owner that created the ticket (see
|
||||
* {@link #sendAsync(String, String, Runnable, Principal)}) with a {@link Phase#FAILED} view that
|
||||
* carries no reply text. The unnamed primary's owner key is {@code null}, matched the same way
|
||||
* as any other key — it reads a ticket another unnamed primary created, and is refused on a
|
||||
* ticket a named caller created. Otherwise returns a {@link Phase#PENDING} view (with the live
|
||||
* worker status as detail), a {@link Phase#DONE} view carrying the reply, or a
|
||||
* {@link Phase#FAILED} view with the reason.
|
||||
*/
|
||||
public TaskView poll(String ticket, String callerOwner) {
|
||||
Task task = tasks.get(ticket);
|
||||
if (task == null) {
|
||||
return null;
|
||||
}
|
||||
if (!ownsTicket(task, callerOwner)) {
|
||||
return new TaskView(ticket, Phase.FAILED, null, null,
|
||||
"forbidden: this ticket was created by a different session", null);
|
||||
}
|
||||
CompletableFuture<Reply> f = task.future;
|
||||
if (!f.isDone()) {
|
||||
Reply question = task.question;
|
||||
@@ -1359,7 +1440,7 @@ public final class MessageService {
|
||||
return new TaskView(ticket, Phase.ASKING, question.text(), null,
|
||||
"worker is waiting for your answer", question.turnId());
|
||||
}
|
||||
return new TaskView(ticket, Phase.PENDING, null, null, "worker " + liveStatus(task.target), null);
|
||||
return new TaskView(ticket, Phase.PENDING, null, null, pendingDetail(task.target), null);
|
||||
}
|
||||
// CB-588: the ticket is terminal and being handed to the caller right here — tell the push
|
||||
// loop it is collected so a later tick's nudge never names a ticket the lead already has.
|
||||
@@ -1387,6 +1468,30 @@ public final class MessageService {
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, detail, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Marker passed as {@code callerOwner} to bypass the ownership check entirely. No
|
||||
* {@link Principal#ownerKey()} ever produces this value — every real key is either
|
||||
* {@code null} (the unnamed primary) or prefixed with its role, such as {@code "worker:"} or
|
||||
* {@code "leader:"}. {@link #poll(String)} passes it; {@link #pendingAsk} has no matching
|
||||
* no-check overload, so this stays package-private for the test that drives the bypass
|
||||
* directly.
|
||||
*/
|
||||
static final String INTERNAL_NO_OWNER_CHECK = "internal:no-owner-check";
|
||||
|
||||
/**
|
||||
* Whether {@code callerOwner} may read {@code task}'s state. {@code callerOwner} is matched
|
||||
* against the task's recorded owner key by equality, including a {@code null} match — the
|
||||
* unnamed primary's owner key is {@code null}, so it owns a ticket another unnamed primary
|
||||
* created and nothing else, the same rule every other role follows. The only caller that
|
||||
* reads any ticket is {@link #INTERNAL_NO_OWNER_CHECK}. This differs from
|
||||
* {@link Rendezvous.Owner#permits}: a missing rendezvous owner is not an authenticated
|
||||
* unnamed primary, so that gate refuses every caller when no owner was recorded.
|
||||
*/
|
||||
private static boolean ownsTicket(Task task, String callerOwner) {
|
||||
return INTERNAL_NO_OWNER_CHECK.equals(callerOwner)
|
||||
|| Objects.equals(callerOwner, task.creatorOwner);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test seam only — carries no production behaviour, and nothing in this class calls it;
|
||||
* {@link #pruneTerminalTickets} still reads {@link Task#completedNanos} directly.
|
||||
@@ -1408,6 +1513,21 @@ public final class MessageService {
|
||||
return task != null && task.completedNanos != null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Detail text for a {@link Phase#PENDING} poll of a plain (non-asking) delegation.
|
||||
* Distinguishes a message still sitting in the injector's queue, never delivered, from one
|
||||
* that already reached the pane and is simply being worked on — so a caller cannot read
|
||||
* "worker working" as "received" when it was not.
|
||||
*/
|
||||
private String pendingDetail(String target) {
|
||||
Long queuedMillis = injector.queuedWaitMillis(target);
|
||||
if (queuedMillis != null) {
|
||||
return "queued, not yet delivered (target is " + liveStatus(target) + "; queued "
|
||||
+ (queuedMillis / 1000) + "s)";
|
||||
}
|
||||
return "worker " + liveStatus(target);
|
||||
}
|
||||
|
||||
/** Best-effort live worker status for a pending poll; never throws (a lookup error is just noise). */
|
||||
private String liveStatus(String target) {
|
||||
try {
|
||||
@@ -1706,22 +1826,84 @@ public final class MessageService {
|
||||
}
|
||||
|
||||
/**
|
||||
* The question {@code workerSession} is currently paused on via {@code fleet_ask}, if any
|
||||
* (CB-582) — {@code fleet_status} uses this to show a pending question without the caller
|
||||
* needing the ticket. {@code null} when the session has no open async question (including a
|
||||
* session mid a <em>blocking</em> {@code fleet_ask}, which has no {@link Task} to look up — see
|
||||
* {@link PendingAsk}).
|
||||
* The question {@code workerSession} is currently paused on via {@code fleet_ask}, if any —
|
||||
* {@code fleet_status} uses this to show a pending question without the caller needing the
|
||||
* ticket. {@code null} when the session has no open async question (including a session mid a
|
||||
* <em>blocking</em> {@code fleet_ask}, which has no {@link Task} to look up — see
|
||||
* {@link PendingAsk}), or when {@code callerOwner} does not own the task the question
|
||||
* belongs to (see {@link #ownsTicket(Task, String)}).
|
||||
*/
|
||||
public PendingAsk pendingAsk(String workerSession) {
|
||||
public PendingAsk pendingAsk(String workerSession, String callerOwner) {
|
||||
for (Task task : tasks.values()) {
|
||||
Reply q = task.question;
|
||||
if (q != null && workerSession.equals(task.target)) {
|
||||
if (q != null && workerSession.equals(task.target) && ownsTicket(task, callerOwner)) {
|
||||
return new PendingAsk(task.ticket, q.text(), q.turnId());
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* One ticket {@code callerOwner} created, still present in {@link #tasks}, surfaced by
|
||||
* {@link #outstanding} so a lead can carry its id into a handover file. {@link #phase} is the
|
||||
* same value {@link #poll} would report right now, terminal phases included: a {@code DONE} or
|
||||
* {@code FAILED} ticket stays in {@link #tasks} — and so stays reported here — until
|
||||
* {@link #pruneTerminalTickets} evicts it.
|
||||
*/
|
||||
public record OutstandingTicket(String ticket, Phase phase, String target) {
|
||||
}
|
||||
|
||||
/**
|
||||
* One worker session paused in {@code fleet_ask}, with the {@code turnId} that answers it,
|
||||
* surfaced by {@link #outstanding} alongside {@link OutstandingTicket}.
|
||||
*/
|
||||
public record OutstandingAsk(String ticket, String turnId, String workerSession) {
|
||||
}
|
||||
|
||||
/** The outstanding tickets and open asks a single call to {@link #outstanding} reports. */
|
||||
public record Outstanding(List<OutstandingTicket> tickets, List<OutstandingAsk> asks) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Every ticket {@code callerOwner} created that is still in {@link #tasks} — including a
|
||||
* finished one nobody has polled yet, since {@link #pruneTerminalTickets} discards its reply
|
||||
* on a timer and a lead that does not carry its id forward can no longer read it after losing
|
||||
* its session's context — plus the subset of those whose worker is paused in
|
||||
* {@code fleet_ask}. Filtered by the same ownership rule as {@link #poll}:
|
||||
* {@link #ownsTicket(Task, String)}.
|
||||
*/
|
||||
public Outstanding outstanding(String callerOwner) {
|
||||
List<OutstandingTicket> tickets = new ArrayList<>();
|
||||
List<OutstandingAsk> asks = new ArrayList<>();
|
||||
for (Task task : tasks.values()) {
|
||||
if (!ownsTicket(task, callerOwner)) {
|
||||
continue;
|
||||
}
|
||||
Reply question = task.question;
|
||||
Phase phase;
|
||||
if (task.future.isDone()) {
|
||||
phase = terminalPhase(task.future);
|
||||
} else if (question != null) {
|
||||
phase = Phase.ASKING;
|
||||
asks.add(new OutstandingAsk(task.ticket, question.turnId(), task.target));
|
||||
} else {
|
||||
phase = Phase.PENDING;
|
||||
}
|
||||
tickets.add(new OutstandingTicket(task.ticket, phase, task.target));
|
||||
}
|
||||
return new Outstanding(tickets, asks);
|
||||
}
|
||||
|
||||
/** As {@link #poll}'s own terminal-result handling, reduced to just the {@link Phase}. */
|
||||
private static Phase terminalPhase(CompletableFuture<Reply> future) {
|
||||
try {
|
||||
Reply r = future.getNow(null);
|
||||
return r != null && r.completed() ? Phase.DONE : Phase.FAILED;
|
||||
} catch (CompletionException | java.util.concurrent.CancellationException e) {
|
||||
return Phase.FAILED;
|
||||
}
|
||||
}
|
||||
|
||||
/** Release the async executor. */
|
||||
public void close() {
|
||||
asyncExecutor.shutdown();
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
@@ -60,7 +61,7 @@ public final class Rendezvous {
|
||||
}
|
||||
|
||||
/** A worker's open mid-turn question: the worker session it belongs to and the answer future. */
|
||||
private record AskWaiter(String session, CompletableFuture<String> answer) {
|
||||
private record AskWaiter(String session, CompletableFuture<String> answer, Owner owner) {
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -70,11 +71,44 @@ public final class Rendezvous {
|
||||
public record AskTicket(String turnId, CompletableFuture<String> answer, boolean fresh) {
|
||||
}
|
||||
|
||||
private final ConcurrentHashMap<String, CompletableFuture<Resolution>> waiters = new ConcurrentHashMap<>();
|
||||
/**
|
||||
* The caller whose accepted delegation opened a turn — the only caller allowed to answer it.
|
||||
* A {@code null} owner key means the unnamed primary.
|
||||
*/
|
||||
public record Owner(String ownerKey) {
|
||||
public static final Owner UNNAMED_PRIMARY = new Owner(null);
|
||||
|
||||
public static Owner of(String ownerKey) {
|
||||
return ownerKey == null ? UNNAMED_PRIMARY : new Owner(ownerKey);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code callerOwner} matches {@code owner}. A {@code null} owner means no owner was
|
||||
* recorded, so it matches no caller. {@link #UNNAMED_PRIMARY} records the unnamed primary
|
||||
* with an owner object whose key is {@code null}.
|
||||
*/
|
||||
public static boolean permits(Owner owner, String callerOwner) {
|
||||
return owner != null && java.util.Objects.equals(owner.ownerKey(), callerOwner);
|
||||
}
|
||||
}
|
||||
|
||||
/** A registered forward waiter together with the owner its delegation was opened under. */
|
||||
private record ForwardWaiter(Owner owner, CompletableFuture<Resolution> future) {
|
||||
}
|
||||
|
||||
private final ConcurrentHashMap<String, ForwardWaiter> waiters = new ConcurrentHashMap<>();
|
||||
|
||||
/** Reverse rendezvous (CB-205): worker questions awaiting the primary's answer, keyed by {@code turnId}. */
|
||||
private final ConcurrentHashMap<String, AskWaiter> asks = new ConcurrentHashMap<>();
|
||||
private final AtomicLong askSeq = new AtomicLong();
|
||||
/**
|
||||
* Minted once per {@code Rendezvous} instance and folded into every {@code turnId} (see
|
||||
* {@link #openAsk(String)}). {@link #askSeq} alone restarts at zero for every instance, so
|
||||
* without this a {@code turnId} minted by one instance could be minted again by another and
|
||||
* resolve to an unrelated ask with no error; this nonce makes that impossible, because an id
|
||||
* minted by one instance can never match the id space of another.
|
||||
*/
|
||||
private final String askBootNonce = UUID.randomUUID().toString().substring(0, 6);
|
||||
/** Per-session index of the currently-open ask, so duplicate fleet_ask calls coalesce onto one turn. */
|
||||
private final ConcurrentHashMap<String, String> openAsksBySession = new ConcurrentHashMap<>();
|
||||
|
||||
@@ -89,8 +123,17 @@ public final class Rendezvous {
|
||||
* code a double open is impossible; this is a tripwire for the day that no longer holds.
|
||||
*/
|
||||
public CompletableFuture<Resolution> open(String session) {
|
||||
return open(session, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Same as {@link #open(String)}, additionally recording {@code owner} as the caller whose
|
||||
* delegation opened this waiter. A {@code null} owner records no owner at all — the
|
||||
* fail-closed default {@link Owner#permits} refuses to everyone.
|
||||
*/
|
||||
public CompletableFuture<Resolution> open(String session, Owner owner) {
|
||||
CompletableFuture<Resolution> waiter = new CompletableFuture<>();
|
||||
CompletableFuture<Resolution> existing = waiters.putIfAbsent(session, waiter);
|
||||
ForwardWaiter existing = waiters.putIfAbsent(session, new ForwardWaiter(owner, waiter));
|
||||
if (existing != null) {
|
||||
throw new IllegalStateException(
|
||||
"rendezvous double-open for session " + session + " — a waiter is already registered");
|
||||
@@ -105,7 +148,13 @@ public final class Rendezvous {
|
||||
* successful {@code open} after a finished turn requires this close to have happened first).
|
||||
*/
|
||||
public void close(String session, CompletableFuture<Resolution> waiter) {
|
||||
waiters.remove(session, waiter);
|
||||
waiters.computeIfPresent(session, (s, w) -> w.future() == waiter ? null : w);
|
||||
}
|
||||
|
||||
/** The owner recorded for {@code session}'s open waiter, or {@code null} if none is open. */
|
||||
public Owner ownerOf(String session) {
|
||||
ForwardWaiter w = waiters.get(session);
|
||||
return w == null ? null : w.owner();
|
||||
}
|
||||
|
||||
/** Whether a send is currently awaiting a resolution for {@code session}. */
|
||||
@@ -119,7 +168,8 @@ public final class Rendezvous {
|
||||
* send (see the CB-116 note above) rather than whichever send happens to be waiting when they fire.
|
||||
*/
|
||||
public CompletableFuture<Resolution> currentWaiter(String session) {
|
||||
return waiters.get(session);
|
||||
ForwardWaiter w = waiters.get(session);
|
||||
return w == null ? null : w.future();
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -145,9 +195,9 @@ public final class Rendezvous {
|
||||
while (true) {
|
||||
AskWaiter[] minted = { null };
|
||||
String turnId = openAsksBySession.computeIfAbsent(session, _ -> {
|
||||
String newTurnId = session + "#" + askSeq.incrementAndGet();
|
||||
String newTurnId = session + "#" + askBootNonce + "-" + askSeq.incrementAndGet();
|
||||
CompletableFuture<String> answer = new CompletableFuture<>();
|
||||
AskWaiter waiter = new AskWaiter(session, answer);
|
||||
AskWaiter waiter = new AskWaiter(session, answer, ownerOf(session));
|
||||
asks.put(newTurnId, waiter);
|
||||
minted[0] = waiter;
|
||||
return newTurnId;
|
||||
@@ -183,6 +233,16 @@ public final class Rendezvous {
|
||||
return w == null ? null : w.session();
|
||||
}
|
||||
|
||||
/**
|
||||
* The owner recorded for {@code turnId} when its ask turn was freshly opened — the caller
|
||||
* whose delegation {@link #answerAsk} must match. {@code null} if {@code turnId} is unknown or
|
||||
* lapsed, or if the ask opened with no forward waiter owner on record.
|
||||
*/
|
||||
public Owner askOwner(String turnId) {
|
||||
AskWaiter w = asks.get(turnId);
|
||||
return w == null ? null : w.owner();
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a worker's blocked {@code fleet_ask} with the primary's {@code answer}, unblocking it
|
||||
* to resume its turn.
|
||||
@@ -245,7 +305,7 @@ public final class Rendezvous {
|
||||
}
|
||||
|
||||
private boolean complete(String session, Resolution resolution) {
|
||||
CompletableFuture<Resolution> waiter = waiters.get(session);
|
||||
return waiter != null && waiter.complete(resolution);
|
||||
ForwardWaiter waiter = waiters.get(session);
|
||||
return waiter != null && waiter.future().complete(resolution);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@ package dev.ltms.fleet.msg;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.PromptBox;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
@@ -50,6 +51,11 @@ import java.util.stream.Collectors;
|
||||
* exhausting its cap does not stop nudges about the others (post-CB-590 regression fix; see
|
||||
* {@link #decide}) — whichever the durable inbox / pending set doesn't already answer via
|
||||
* {@code STOP}.
|
||||
*
|
||||
* <p>A lead that is injectable is nudged only when its prompt box is also empty
|
||||
* ({@link PromptBox}): the delivery pastes and submits in one call, so a nudge into a box holding
|
||||
* the operator's half-typed line would submit that line too. A nudge held for that reason waits for
|
||||
* the next tick like any other, and the pending work is re-read then.
|
||||
*/
|
||||
public final class ReplyPushLoop {
|
||||
|
||||
@@ -81,6 +87,7 @@ public final class ReplyPushLoop {
|
||||
|
||||
private final PrimaryRegistry primaryRegistry;
|
||||
private final AgentControl agents;
|
||||
private final PromptBox promptBox;
|
||||
private final ReplyInbox inbox;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final int maxReminders;
|
||||
@@ -125,6 +132,7 @@ public final class ReplyPushLoop {
|
||||
int maxReminders, long backoffMs, Metrics metrics) {
|
||||
this.primaryRegistry = primaryRegistry;
|
||||
this.agents = agents;
|
||||
this.promptBox = new PromptBox(agents);
|
||||
this.inbox = inbox;
|
||||
this.scheduler = scheduler;
|
||||
this.maxReminders = maxReminders;
|
||||
@@ -398,11 +406,15 @@ public final class ReplyPushLoop {
|
||||
log.debug("push: status check failed for lead {}, will retry", lead, e);
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
if (status.injectable()) {
|
||||
return Action.INJECT;
|
||||
if (!status.injectable()) {
|
||||
log.debug("push: lead {} is {} (not injectable), waiting", lead, status);
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
log.debug("push: lead {} is {} (not injectable), waiting", lead, status);
|
||||
return Action.WAIT_BUSY;
|
||||
if (!promptBox.clearToSubmit(lead)) {
|
||||
log.debug("push: lead {} has unsubmitted text in its prompt box, waiting", lead);
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
return Action.INJECT;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -439,6 +451,15 @@ public final class ReplyPushLoop {
|
||||
* — timeout, transport error, a decode error — is treated as still live and the binding is left
|
||||
* alone, because guessing wrong here is unrecoverable while guessing "live" merely costs one more
|
||||
* retry on the next tick, which {@link #decide} already tolerates.
|
||||
*
|
||||
* <p><strong>The fallback is probed too.</strong> {@code PrimaryRegistry.nudgeTargetFor} already
|
||||
* resolves a named delegator to its current terminal before this method ever sees it, which
|
||||
* keeps a rolled lead's per-target binding live. What that resolution cannot fix is a caller
|
||||
* that was never recorded with a name at all — an unnamed primary, or a lead whose tab the
|
||||
* scanner cannot currently see — where the fallback it returns is still the raw terminal last
|
||||
* learned from call traffic. This method returns that fallback only after the same liveness
|
||||
* check, and gives up for this tick (an empty result, exactly like "no lead known at all") rather
|
||||
* than hand a caller a second stale address un-probed.
|
||||
*/
|
||||
private Optional<String> resolveLiveLead(String target) {
|
||||
Optional<String> lead = primaryRegistry.nudgeTargetFor(target);
|
||||
@@ -448,7 +469,12 @@ public final class ReplyPushLoop {
|
||||
log.debug("push: lead {} delegated to for {} is no longer live, forgetting the stale binding "
|
||||
+ "and falling back", lead.get(), target);
|
||||
primaryRegistry.forgetDelegation(target);
|
||||
return primaryRegistry.nudgeTargetFor(target);
|
||||
Optional<String> fallback = primaryRegistry.nudgeTargetFor(target);
|
||||
if (fallback.isEmpty() || isLive(fallback.get())) {
|
||||
return fallback;
|
||||
}
|
||||
log.debug("push: fallback lead {} for {} is also not live, skipping this tick", fallback.get(), target);
|
||||
return Optional.empty();
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
package dev.ltms.fleet.peer;
|
||||
|
||||
import java.util.Locale;
|
||||
import java.util.stream.Collectors;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
/**
|
||||
* What a member is <em>for</em> — the contract it runs under.
|
||||
@@ -100,6 +102,11 @@ public enum MemberRole {
|
||||
return null;
|
||||
}
|
||||
|
||||
/** The wire name of every role, joined with {@code ", "} in declaration order. */
|
||||
public static String wireNames() {
|
||||
return Stream.of(values()).map(MemberRole::wireName).collect(Collectors.joining(", "));
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a config/wire spelling, case-insensitively.
|
||||
*
|
||||
@@ -118,14 +125,7 @@ public enum MemberRole {
|
||||
}
|
||||
}
|
||||
}
|
||||
StringBuilder valid = new StringBuilder();
|
||||
for (MemberRole r : values()) {
|
||||
if (!valid.isEmpty()) {
|
||||
valid.append(", ");
|
||||
}
|
||||
valid.append(r.wireName());
|
||||
}
|
||||
throw new IllegalArgumentException(
|
||||
"unknown member role '" + s + "'; valid roles are: " + valid);
|
||||
"unknown member role '" + s + "'; valid roles are: " + wireNames());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,6 +13,7 @@ import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.inject.MemberPresence;
|
||||
import dev.ltms.fleet.member.MemberCredentialPolicyView;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
@@ -49,22 +50,52 @@ import java.util.stream.Collectors;
|
||||
*/
|
||||
public final class FleetApp {
|
||||
|
||||
/** The authorization action the matching route handler hands to {@link #allow}. */
|
||||
/**
|
||||
* The authorization action the matching route handler hands to {@link #allow}, for a route
|
||||
* whose action does not depend on the request body.
|
||||
*/
|
||||
static Authz.Action routeAction(String route) {
|
||||
return routeAction(route, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, plus the one route whose action depends on the body: {@code POST
|
||||
* /sessions/{id}/message} carries a {@code turnId} (the answer-a-blocked-worker shape) or not
|
||||
* (a plain delivery), mirroring {@code FleetMcp#sendAction}'s split of the same two call
|
||||
* shapes over MCP. {@code turnId} is ignored by every other route.
|
||||
*
|
||||
* @param turnId the request body's {@code turnId}, or {@code null}/blank when absent or not
|
||||
* applicable to this route
|
||||
*/
|
||||
static Authz.Action routeAction(String route, String turnId) {
|
||||
return switch (route) {
|
||||
case "GET /metrics" -> Authz.Action.METRICS;
|
||||
case "POST /members" -> Authz.Action.SPAWN;
|
||||
case "DELETE /members/{paneId}" -> Authz.Action.STOP;
|
||||
case "POST /sessions/{id}/message" -> Authz.Action.SEND;
|
||||
case "POST /sessions/{id}/message" -> turnId == null || turnId.isBlank()
|
||||
? Authz.Action.SEND : Authz.Action.ANSWER;
|
||||
case "POST /sessions/{id}/reply" -> Authz.Action.REPLY;
|
||||
case "GET /sessions/{id}/replies" -> Authz.Action.DRAIN;
|
||||
case "POST /sessions/{id}/ask" -> Authz.Action.ASK;
|
||||
case "GET /sessions", "GET /agents", "GET /members", "GET /profiles",
|
||||
"GET /member-credentials", "GET /sessions/{id}/status", "GET /tasks/{ticket}" -> Authz.Action.READ;
|
||||
"GET /member-credentials" -> Authz.Action.READ;
|
||||
case "GET /sessions/{id}/status", "GET /tasks/{ticket}" -> Authz.Action.TASK_READ;
|
||||
default -> throw new IllegalArgumentException("route has no authorization gate: " + route);
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* The second gate for {@code POST /sessions/{id}/message}: checked only when {@code turnId}
|
||||
* is present and non-blank, against {@link Authz.Action#ANSWER}. A request with no {@code
|
||||
* turnId} passes this gate unconditionally, without consulting {@code permit} at all, having
|
||||
* already cleared the coarse {@link Authz.Action#SEND} grant checked ahead of it.
|
||||
*
|
||||
* @param permit reports whether the caller holds the named grant
|
||||
*/
|
||||
static boolean answerGatePasses(String turnId, Predicate<Authz.Action> permit) {
|
||||
return turnId == null || turnId.isBlank() || permit.test(Authz.Action.ANSWER);
|
||||
}
|
||||
|
||||
/** Default blocking window for a message; kept under typical HTTP idle timeouts. */
|
||||
private static final long DEFAULT_MESSAGE_TIMEOUT_MS = 25_000;
|
||||
private static final long MAX_MESSAGE_TIMEOUT_MS = 120_000;
|
||||
@@ -236,6 +267,33 @@ public final class FleetApp {
|
||||
return app;
|
||||
}
|
||||
|
||||
/**
|
||||
* The authorization decision behind {@link #allow}, taking the caller directly rather than
|
||||
* pulling it from a servlet {@link Context} — unit-testable without fabricating a live
|
||||
* request, the same reason {@code FleetMcp#denyFor} is split from {@code FleetMcp#deny}.
|
||||
*
|
||||
* @param knownLeadOrCollaborator the classifier a collaborator's {@code SEND} is checked
|
||||
* against; pass {@link #auth}'s own {@code
|
||||
* knownLeadOrCollaborator()} to exercise the real production
|
||||
* gate, as {@link #allow} does
|
||||
*/
|
||||
static boolean permitsFor(Principal caller, Authz.Action action, String target,
|
||||
Predicate<String> knownLeadOrCollaborator) {
|
||||
return Authz.permits(caller, action, target, knownLeadOrCollaborator);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #permitsFor(Principal, Authz.Action, String, Predicate)}, also threading the
|
||||
* classifier an observer's {@code SEND} is checked against; pass {@link #auth}'s own
|
||||
* {@code sendableObserverTarget()} to exercise the real production gate, as {@link #allow}
|
||||
* does.
|
||||
*/
|
||||
static boolean permitsFor(Principal caller, Authz.Action action, String target,
|
||||
Predicate<String> knownLeadOrCollaborator,
|
||||
Predicate<String> knownObserverTarget) {
|
||||
return Authz.permits(caller, action, target, knownLeadOrCollaborator, knownObserverTarget);
|
||||
}
|
||||
|
||||
/**
|
||||
* Gate a handler on the CB-505 authorization table. Returns {@code true} when the request may
|
||||
* proceed; otherwise writes the error response and returns {@code false}.
|
||||
@@ -249,8 +307,9 @@ public final class FleetApp {
|
||||
return true; // legacy: authorization not enforced
|
||||
}
|
||||
Principal caller = ctx.attribute(CALLER);
|
||||
if (Authz.permits(caller, action, target)) {
|
||||
if (action != Authz.Action.READ && action != Authz.Action.METRICS) {
|
||||
if (permitsFor(caller, action, target, auth.knownLeadOrCollaborator(), auth.sendableObserverTarget())) {
|
||||
if (action != Authz.Action.READ && action != Authz.Action.METRICS
|
||||
&& action != Authz.Action.TASK_READ) {
|
||||
AuditLog.allowed(caller, action, target); // reads would drown the trail
|
||||
}
|
||||
return true;
|
||||
@@ -390,14 +449,27 @@ public final class FleetApp {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Tab id → its herdr display label, or an empty map on a {@code workspace.list}/{@code
|
||||
* tab.list} failure — a missing label must not cost the agent roster.
|
||||
*/
|
||||
private Map<String, String> tabLabelsOrEmpty() {
|
||||
try {
|
||||
return new PaneLocator(herdr, memberHerdr).tabLabelsByTabId();
|
||||
} catch (HerdrException e) {
|
||||
return Map.of();
|
||||
}
|
||||
}
|
||||
|
||||
/** Discovery: every agent herdr tracks, keyed by its Claude session UUID. */
|
||||
private void agents(Context ctx) {
|
||||
if (!allow(ctx, routeAction("GET /agents"), null)) {
|
||||
return;
|
||||
}
|
||||
final Map<String, String> tabLabels = tabLabelsOrEmpty();
|
||||
try {
|
||||
ctx.status(200).json(Map.of("agents",
|
||||
workers.list().stream().map(Agent.class::cast).map(FleetApp::view).toList()));
|
||||
workers.list().stream().map(Agent.class::cast).map(a -> view(a, tabLabels)).toList()));
|
||||
} catch (HerdrException e) {
|
||||
// fleetd #297: workers.list() reaches herdr — a transport failure must land in the same
|
||||
// {error, detail} envelope every other failure path here uses, not escape as a bare
|
||||
@@ -603,47 +675,64 @@ public final class FleetApp {
|
||||
* status-gated injector and block until the worker returns a structured {@code fleet_reply}.
|
||||
* Times out with a typed 202 (working / queued / busy) rather than an error — the message may
|
||||
* still land.
|
||||
*
|
||||
* <p>Two call shapes share this route, exactly as {@code fleet_send} does over MCP (see
|
||||
* {@code FleetMcp#sendAction}): a plain delivery to {@code id}, and -- when the body carries
|
||||
* {@code turnId} -- resolving a worker's blocked question. The coarse {@link
|
||||
* Authz.Action#SEND} grant is checked first, before the body is read at all; only once that
|
||||
* passes is the body parsed, and a present {@code turnId} is then checked again against
|
||||
* {@link Authz.Action#ANSWER}. A body that fails to parse is rejected with 400 and reaches
|
||||
* neither {@code messages.answer} nor {@code messages.send}.
|
||||
*/
|
||||
private void sendMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/message"), id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
String turnId;
|
||||
long timeout;
|
||||
boolean wait;
|
||||
Principal caller = ctx.attribute(CALLER);
|
||||
String callerOwner = caller == null ? null : caller.ownerKey();
|
||||
JsonNode body;
|
||||
try {
|
||||
JsonNode body = mapper.readTree(ctx.body());
|
||||
content = body.path("content").asText("");
|
||||
turnId = body.path("turnId").asText(null);
|
||||
timeout = body.path("timeoutMs").asLong(DEFAULT_MESSAGE_TIMEOUT_MS);
|
||||
wait = body.path("wait").asBoolean(true); // default: block for the reply (CB-104)
|
||||
body = mapper.readTree(ctx.body());
|
||||
} catch (Exception e) {
|
||||
body = null;
|
||||
}
|
||||
if (body == null) {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "body must be JSON"));
|
||||
return;
|
||||
}
|
||||
String turnId = body.path("turnId").asText(null);
|
||||
if (!answerGatePasses(turnId, action -> allow(ctx, action, id))) {
|
||||
return;
|
||||
}
|
||||
String content = body.path("content").asText("");
|
||||
long timeout = body.path("timeoutMs").asLong(DEFAULT_MESSAGE_TIMEOUT_MS);
|
||||
boolean wait = body.path("wait").asBoolean(true); // default: block for the reply (CB-104)
|
||||
if (content.isBlank()) {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "content is required"));
|
||||
return;
|
||||
}
|
||||
// An observer's SEND reaches a pane that cannot otherwise distinguish this from a human
|
||||
// paste (see FleetMcp#attributeIfObserver, the same rule on the MCP entry path); every
|
||||
// other caller's content passes through unchanged.
|
||||
content = FleetMcp.attributeIfObserver(caller, content);
|
||||
timeout = Math.clamp(timeout, 1, MAX_MESSAGE_TIMEOUT_MS);
|
||||
|
||||
// Answering a worker's fleet_ask (CB-205): always blocks, and derives the worker from turnId.
|
||||
if (turnId != null && !turnId.isBlank()) {
|
||||
writeReply(ctx, id, messages.answer(turnId, content, timeout), timeout);
|
||||
writeReply(ctx, id, messages.answer(turnId, content, timeout, callerOwner), timeout);
|
||||
return;
|
||||
}
|
||||
|
||||
if (!wait) {
|
||||
// Fire-and-poll (CB-107): return a ticket immediately; the caller polls GET /tasks/{ticket}.
|
||||
String ticket = messages.sendAsync(id, content);
|
||||
String ticket = messages.sendAsync(id, content, null, caller);
|
||||
ctx.status(202).json(Map.of("sessionId", id, "ticket", ticket, "status", "accepted"));
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
writeReply(ctx, id, messages.send(id, content, timeout), timeout);
|
||||
writeReply(ctx, id, messages.send(id, content, timeout, callerOwner), timeout);
|
||||
} catch (HerdrException e) {
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
@@ -662,6 +751,10 @@ public final class FleetApp {
|
||||
case STALE_TURN -> ctx.status(409).json(Map.of(
|
||||
"sessionId", id, "error", "stale_turn",
|
||||
"detail", "that question is no longer open (timed out or already answered)"));
|
||||
case NOT_TURN_OWNER -> ctx.status(403).json(Map.of(
|
||||
"sessionId", id, "error", "not_turn_owner",
|
||||
"detail", "this turn belongs to a different delegation — only the caller that "
|
||||
+ "opened it may answer it"));
|
||||
case REPLIED, COMPLETED_UNREPLIED -> {
|
||||
// replySource distinguishes a structured fleet_reply from the CB-106 completion
|
||||
// fallback (a scrape of the worker's transcript when it finished without replying).
|
||||
@@ -676,9 +769,10 @@ public final class FleetApp {
|
||||
// silent fall-through. That is exactly the bug this ticket exists to fix:
|
||||
// `default -> "done"` used to sit here and would have told a REST caller the
|
||||
// delegation completed for TIMED_OUT_UNCONFIRMED, the one outcome where delivery
|
||||
// is unknown. REPLIED, COMPLETED_UNREPLIED, QUESTION and STALE_TURN can never
|
||||
// actually reach this inner switch — the outer switch above always dispatches
|
||||
// them first — but they still need an arm to keep this switch exhaustive.
|
||||
// is unknown. REPLIED, COMPLETED_UNREPLIED, QUESTION, STALE_TURN and
|
||||
// NOT_TURN_OWNER can never actually reach this inner switch — the outer switch
|
||||
// above always dispatches them first — but they still need an arm to keep this
|
||||
// switch exhaustive.
|
||||
"status", switch (reply.outcome()) {
|
||||
case TIMED_OUT_WORKING -> "working";
|
||||
case TIMED_OUT_QUEUED -> "queued";
|
||||
@@ -688,7 +782,7 @@ public final class FleetApp {
|
||||
case BUSY -> "busy";
|
||||
case WORKER_FAILED -> "failed";
|
||||
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
||||
case REPLIED, COMPLETED_UNREPLIED, QUESTION, STALE_TURN -> "done"; // unreachable
|
||||
case REPLIED, COMPLETED_UNREPLIED, QUESTION, STALE_TURN, NOT_TURN_OWNER -> "done"; // unreachable
|
||||
},
|
||||
"detail", reply.outcome() == MessageService.Outcome.TIMED_OUT_UNCONFIRMED
|
||||
? "no reply within " + timeout + "ms; delivery is unconfirmed — the "
|
||||
@@ -817,10 +911,11 @@ public final class FleetApp {
|
||||
body.put("sessionId", id);
|
||||
body.put("status", messages.status(id).name().toLowerCase());
|
||||
body.put("ready", deliverable.test(id));
|
||||
// CB-582: a worker paused mid-turn in an async fleet_ask is otherwise invisible to a
|
||||
// status poll — surface the open question and how to answer it, same as fleet_poll's
|
||||
// Phase.ASKING view.
|
||||
MessageService.PendingAsk ask = messages.pendingAsk(id);
|
||||
// A worker paused mid-turn in an async fleet_ask is otherwise invisible to a status
|
||||
// poll — surface the open question and how to answer it, same as fleet_poll's
|
||||
// Phase.ASKING view, but only to the caller whose owner key created that delegation.
|
||||
Principal caller = ctx.attribute(CALLER);
|
||||
MessageService.PendingAsk ask = messages.pendingAsk(id, caller == null ? null : caller.ownerKey());
|
||||
if (ask != null) {
|
||||
body.put("question", ask.question());
|
||||
body.put("turnId", ask.turnId());
|
||||
@@ -837,7 +932,8 @@ public final class FleetApp {
|
||||
if (!allow(ctx, routeAction("GET /tasks/{ticket}"), null)) {
|
||||
return;
|
||||
}
|
||||
MessageService.TaskView v = messages.poll(ctx.pathParam("ticket"));
|
||||
Principal caller = ctx.attribute(CALLER);
|
||||
MessageService.TaskView v = messages.poll(ctx.pathParam("ticket"), caller == null ? null : caller.ownerKey());
|
||||
if (v == null) {
|
||||
ctx.status(404).json(Map.of("error", "unknown_ticket", "detail", "no such task (or it has expired)"));
|
||||
return;
|
||||
@@ -869,13 +965,19 @@ public final class FleetApp {
|
||||
}
|
||||
}
|
||||
|
||||
/** Stable JSON projection of an agent (null-safe for the start-time shape). */
|
||||
private static Map<String, Object> view(Agent a) {
|
||||
/**
|
||||
* Stable JSON projection of an agent (null-safe for the start-time shape).
|
||||
*
|
||||
* @param tabLabels tab id → its herdr display label; a tab absent from this map, or carrying
|
||||
* a {@code null} label itself, projects as a {@code null} "label"
|
||||
*/
|
||||
private static Map<String, Object> view(Agent a, Map<String, String> tabLabels) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("terminalId", a.terminalId());
|
||||
m.put("paneId", a.paneId());
|
||||
m.put("workspaceId", a.workspaceId());
|
||||
m.put("tabId", a.tabId());
|
||||
m.put("label", tabLabels.get(a.tabId()));
|
||||
m.put("sessionId", a.sessionId());
|
||||
m.put("agentType", a.agentType());
|
||||
m.put("status", a.status().name().toLowerCase());
|
||||
|
||||
@@ -442,38 +442,6 @@ public final class GitWorktrees implements Worktrees {
|
||||
ENVIRONMENT_CREDENTIAL_HELPER);
|
||||
}
|
||||
|
||||
/**
|
||||
* {@link #configureEnvironmentCredentialHelper} only ever fires for an HTTPS origin — Git never
|
||||
* consults a {@code credential.helper} for an SSH transport. This repo's own origin is
|
||||
* {@code ssh://git@git.ltms.dev:2224/fleet/fleetd.git}, so a member sitting on that origin never
|
||||
* reaches the helper and the repo-scoped {@code WORKER_GITEA_TOKEN} is simply not used.
|
||||
*
|
||||
* <p>An earlier version of this javadoc justified the rewrite by claiming a member <em>cannot</em>
|
||||
* push once {@code memberCredentials.policy: allow-list} blocks {@code SSH_AUTH_SOCK}, because
|
||||
* "there is no private key file on this host, only an ssh-agent socket". That premise is false
|
||||
* (fleetd #184): {@code ssh -G} resolves a readable, passphrase-free {@code IdentityFile} outside
|
||||
* {@code ~/.ssh}, and a member — same OS user — pushes over SSH with the socket blanked. The
|
||||
* rewrite is still worth having, but for the reason below rather than that one: it routes the
|
||||
* member through its own scoped token instead of the operator's ssh identity, which is what makes
|
||||
* a member's pushes attributable and revocable.
|
||||
*
|
||||
* <p>The fix is a <em>worktree-scoped</em> URL rewrite: {@code url.<https-base>.insteadOf
|
||||
* <ssh-base>}, set with {@code --worktree} so it lands only in
|
||||
* {@code <worktree>/.git/worktrees/<name>/config.worktree} (enabled by
|
||||
* {@code extensions.worktreeConfig}, already turned on above) and never touches the shared
|
||||
* repo-level config the primary checkout also reads. {@code insteadOf} — not
|
||||
* {@code pushInsteadOf} — because a member may also need to fetch or rebase, and both should go
|
||||
* through the member's own token for the same reason.
|
||||
*
|
||||
* <p>The host (and, for the rewrite's SSH-side match, the port) come from parsing the origin
|
||||
* itself — never a hardcoded forge host, which is exactly what #177 removed. An origin that is
|
||||
* already {@code https://} is left alone; the credential helper already covers it. An origin
|
||||
* that is neither {@code ssh://} nor {@code https://} — including the scp-like shorthand
|
||||
* ({@code git@host:path}, no scheme) — is left untouched deliberately: that shorthand's
|
||||
* {@code host:path} split is defined by the user's ssh_config aliases, not by URI syntax, so
|
||||
* guessing at it risks rewriting to the wrong place. A repo provisioned from that form keeps
|
||||
* today's (broken, if the policy blocks the agent) SSH-only behaviour rather than a wrong rewrite.
|
||||
*/
|
||||
/**
|
||||
* Blank the user-info of a remote URL before it reaches a log. A remote URL is not obviously a
|
||||
* credential channel, which is exactly why one has leaked here three times ({@code git remote -v}
|
||||
@@ -485,6 +453,30 @@ public final class GitWorktrees implements Worktrees {
|
||||
return url == null ? null : url.replaceAll("://[^@/]*@", "://<redacted>@");
|
||||
}
|
||||
|
||||
/**
|
||||
* {@link #configureEnvironmentCredentialHelper} only fires for an HTTPS origin — Git never
|
||||
* consults a {@code credential.helper} for an SSH transport. This repo's own origin is
|
||||
* {@code ssh://git@git.ltms.dev:2224/fleet/fleetd.git}, so a member on that origin never
|
||||
* reaches the helper, and the repo-scoped token goes unused without a separate rewrite.
|
||||
*
|
||||
* <p>This method routes the member through its own scoped token instead of the operator's ssh
|
||||
* identity, which is what makes a member's pushes attributable and revocable.
|
||||
*
|
||||
* <p>The fix is a <em>worktree-scoped</em> URL rewrite: {@code url.<https-base>.insteadOf
|
||||
* <ssh-base>}, set with {@code --worktree} so it lands only in
|
||||
* {@code <worktree>/.git/worktrees/<name>/config.worktree} and never touches the shared
|
||||
* repo-level config the primary checkout also reads. {@code insteadOf} — not
|
||||
* {@code pushInsteadOf} — because a member may also need to fetch or rebase through its own
|
||||
* token.
|
||||
*
|
||||
* <p>The host (and, for the rewrite's SSH-side match, the port) come from parsing the origin
|
||||
* itself, never a hardcoded forge host. An origin already {@code https://} is left alone; the
|
||||
* credential helper already covers it. An origin that is neither {@code ssh://} nor
|
||||
* {@code https://} — including the scp-like shorthand ({@code git@host:path}, no scheme) — is
|
||||
* left untouched: that shorthand's {@code host:path} split is defined by the user's ssh_config
|
||||
* aliases, not by URI syntax, so guessing at it risks rewriting to the wrong place, and that
|
||||
* origin keeps SSH-only push behaviour instead.
|
||||
*/
|
||||
private void configureHttpsUrlRewriteForSshOrigin(String repoRoot, String worktreePath) {
|
||||
if (exitCode("git", "-C", repoRoot, "config", "--get", "remote.origin.url") != 0) {
|
||||
return;
|
||||
|
||||
@@ -26,6 +26,7 @@ import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* Authoritative in-daemon registry of the worker sessions this {@code fleetd} process spawned.
|
||||
@@ -57,6 +58,18 @@ public final class SessionManager implements TurnListener {
|
||||
* Populated on every spawn path, removed on {@link #release}.
|
||||
*/
|
||||
private final ConcurrentHashMap<String /*paneId*/, PeerHandle> handles = new ConcurrentHashMap<>();
|
||||
/**
|
||||
* fleetd #702: a pane mid-teardown, keyed by paneId, held from just before its registry entry
|
||||
* is removed until {@link #releaseRemoved} finishes. {@link #spawnedMemberRole} consults this
|
||||
* alongside the registry, so a caller resolving the pane's terminal during that window still
|
||||
* sees a live member and never falls through to a tab map.
|
||||
*
|
||||
* <p>Depth-counted rather than a plain set: two threads can be tearing down the same pane at
|
||||
* once (the CAS in {@link #releaseIfCurrent} exists for exactly that race), and with a set the
|
||||
* loser's {@code finally} would unmark the pane while the winner is still mid-teardown,
|
||||
* reopening the window this exists to close.
|
||||
*/
|
||||
private final ConcurrentHashMap<String /*paneId*/, Releasing> releasing = new ConcurrentHashMap<>();
|
||||
private final MemberPresence presence;
|
||||
private final SecureRandom nonceRandom = new SecureRandom();
|
||||
private final AtomicLong nonceSeq = new AtomicLong();
|
||||
@@ -242,6 +255,10 @@ public final class SessionManager implements TurnListener {
|
||||
handle.id(), handle.terminalId(), resolvedProfile, actualRole, cwd, ownerTerminal, now, now, 0,
|
||||
MemberSession.State.SPAWNING, null, null, handle.charterReceipt(), handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
// A presence contact that already arrived for this terminal found no registry
|
||||
// entry to transition and gave up silently. Retry it now that one exists; remove
|
||||
// this call and such a session stays in SPAWNING even though it is present.
|
||||
reconcilePresence(handle.terminalId());
|
||||
handles.put(handle.id(), handle);
|
||||
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
||||
@@ -303,9 +320,11 @@ public final class SessionManager implements TurnListener {
|
||||
* is a logged path an operator can reclaim, the cost of a deleted one is unrecoverable work.
|
||||
*/
|
||||
private MemberSession release(String paneId, ReleaseCause cause) {
|
||||
MemberSession removed = registry.remove(paneId);
|
||||
releaseRemoved(paneId, removed, handles.remove(paneId), cause);
|
||||
return removed;
|
||||
return releaseWindow(paneId, registry.get(paneId), () -> {
|
||||
MemberSession removed = registry.remove(paneId);
|
||||
releaseRemoved(paneId, removed, handles.remove(paneId), cause);
|
||||
return removed;
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -314,16 +333,85 @@ public final class SessionManager implements TurnListener {
|
||||
* DONE record from stopping a worker that delivery has made BUSY.
|
||||
*/
|
||||
private boolean releaseIfCurrent(MemberSession expected, ReleaseCause cause) {
|
||||
if (!registry.remove(expected.paneId(), expected)) {
|
||||
// A lifecycle transition replaced the record between the caller's check and this remove.
|
||||
// Log it: this race is by definition unobservable otherwise, and a reaper that silently
|
||||
// declines to reap is the hardest kind of behaviour to diagnose after the fact.
|
||||
log.debug("skipping reap of pane={}: its registry record changed after the idle check "
|
||||
+ "(most likely a delivery made it BUSY)", expected.paneId());
|
||||
return false;
|
||||
return releaseWindow(expected.paneId(), expected, () -> {
|
||||
if (!registry.remove(expected.paneId(), expected)) {
|
||||
// A lifecycle transition replaced the record between the caller's check and this
|
||||
// remove. Log it: this race is by definition unobservable otherwise, and a reaper
|
||||
// that silently declines to reap is the hardest kind of behaviour to diagnose
|
||||
// after the fact.
|
||||
log.debug("skipping reap of pane={}: its registry record changed after the idle "
|
||||
+ "check (most likely a delivery made it BUSY)", expected.paneId());
|
||||
return false;
|
||||
}
|
||||
releaseRemoved(expected.paneId(), expected, handles.remove(expected.paneId()), cause);
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #702: mark {@code paneId} as mid-teardown — using {@code known}'s terminal/role when
|
||||
* it is available — for the whole of {@code teardown}, which removes the registry entry and
|
||||
* then runs {@link #releaseRemoved}. Shared by both registry-removal sites ({@link #release}'s
|
||||
* unconditional remove and {@link #releaseIfCurrent}'s CAS remove) so neither can leave the
|
||||
* other's window unmarked.
|
||||
*
|
||||
* <p>The mark is written before {@code teardown} runs — so it covers the removal itself, not
|
||||
* only what comes after it — and cleared in a {@code finally}, so an unchecked throw out of
|
||||
* {@code teardown} (including one from {@link PeerLauncher#stop}, which declares nothing) can
|
||||
* never leave the pane marked for the rest of the daemon's life.
|
||||
*/
|
||||
private <T> T releaseWindow(String paneId, MemberSession known, Supplier<T> teardown) {
|
||||
releasing.compute(paneId, (_, prior) -> Releasing.enter(prior, known));
|
||||
try {
|
||||
return teardown.get();
|
||||
} finally {
|
||||
releasing.compute(paneId, (_, prior) -> prior == null ? null : prior.leave());
|
||||
}
|
||||
releaseRemoved(expected.paneId(), expected, handles.remove(expected.paneId()), cause);
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Depth count plus the terminal/role a mid-teardown pane belongs to, for
|
||||
* {@link #spawnedMemberRole}. The terminal/role come from whichever call into
|
||||
* {@link #releaseWindow} first knew them: a call that finds the registry entry already gone
|
||||
* passes a {@code null} session, and must not blank out what the first call recorded.
|
||||
*/
|
||||
record Releasing(int depth, String terminalId, MemberRole role) {
|
||||
static Releasing enter(Releasing prior, MemberSession known) {
|
||||
int depth = (prior == null ? 0 : prior.depth()) + 1;
|
||||
String terminalId = known != null ? known.terminalId() : prior == null ? null : prior.terminalId();
|
||||
MemberRole role = known != null ? known.role() : prior == null ? null : prior.role();
|
||||
return new Releasing(depth, terminalId, role);
|
||||
}
|
||||
|
||||
Releasing leave() {
|
||||
return depth <= 1 ? null : new Releasing(depth - 1, terminalId, role);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The role of the live spawned member occupying {@code terminal} — whether it is currently in
|
||||
* the registry, or mid-teardown between {@link #release} removing its registry entry and
|
||||
* {@link #releaseRemoved} actually stopping its pane (fleetd #702). {@code null} for a terminal
|
||||
* that is neither: this method is the one reader a caller resolver consults before any tab
|
||||
* map, so a live or releasing member's identity never falls back to a tab label.
|
||||
*
|
||||
* <p>Checks the registry directly via {@link #findByTerminal} rather than {@link #roster()},
|
||||
* so this hot-path lookup (consulted on every resolve) never pays for a list copy or a stream.
|
||||
*/
|
||||
public MemberRole spawnedMemberRole(String terminal) {
|
||||
MemberSession session = findByTerminal(terminal);
|
||||
if (session != null) {
|
||||
return session.role();
|
||||
}
|
||||
if (terminal == null) {
|
||||
return null;
|
||||
}
|
||||
for (Releasing r : releasing.values()) {
|
||||
if (terminal.equals(r.terminalId())) {
|
||||
return r.role();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private void releaseRemoved(String paneId, MemberSession removed, PeerHandle removedHandle,
|
||||
@@ -382,6 +470,12 @@ public final class SessionManager implements TurnListener {
|
||||
MemberSession resolved = resolveAgentSessionId(removed, removedHandle);
|
||||
notifyReleased(new ReleaseDetail(resolved.terminalId(), resolved.worktree(),
|
||||
resolved.branch(), snapshotRef, resolved.agentSessionId()));
|
||||
String terminal = removed.terminalId();
|
||||
if (terminal != null && !terminal.isBlank()) {
|
||||
// Without this, a terminal stays marked present after its pane is gone, so a
|
||||
// later send to the same id would read as deliverable instead of refused.
|
||||
presence.forget(terminal);
|
||||
}
|
||||
}
|
||||
}
|
||||
// CB-581: the pane must always stop, even if the dirty check above threw. A session removed
|
||||
@@ -725,6 +819,10 @@ public final class SessionManager implements TurnListener {
|
||||
handle.charterReceipt(),
|
||||
handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
// A presence contact that already arrived for this terminal found no registry entry to
|
||||
// transition and gave up silently. Retry it now that one exists; remove this call and
|
||||
// such a session stays in SPAWNING even though it is present.
|
||||
reconcilePresence(handle.terminalId());
|
||||
handles.put(handle.id(), handle);
|
||||
log.debug("acquired worktree session id={} terminal={} profile={} branch={} path={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.branch(), session.worktree());
|
||||
@@ -899,6 +997,20 @@ public final class SessionManager implements TurnListener {
|
||||
transitionByTerminal(terminalId, MemberSession.State.SPAWNING, MemberSession.State.READY);
|
||||
}
|
||||
|
||||
/**
|
||||
* Completes a newly registered session's {@code SPAWNING -> READY} transition when {@code
|
||||
* terminalId} was already marked present before this ran. A terminal never marked present is
|
||||
* left in {@code SPAWNING}; it reaches {@code READY} normally through {@link #onReady} once
|
||||
* its own contact arrives. Callers must run this only once the session's registry entry is
|
||||
* already visible — {@link #onReady}'s transition matches against that entry, and reconciling
|
||||
* before the entry exists finds nothing to transition.
|
||||
*/
|
||||
private void reconcilePresence(String terminalId) {
|
||||
if (terminalId != null && !terminalId.isBlank() && presence.isPresent(terminalId)) {
|
||||
onReady(terminalId);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Lifecycle hook: a message was delivered into the worker — it is now busy on a turn.
|
||||
* The turn count is bumped and the activity timestamp is refreshed. A {@code DONE} session
|
||||
|
||||
@@ -14,10 +14,12 @@ import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-534: the injector's readiness gate must open for a lead as well as for a present worker.
|
||||
* fleetd #669 follow-up: the same gate must also open for a collaborator, which — like a lead —
|
||||
* is never enrolled in {@link MemberPresence} and never discovered by the lead scan.
|
||||
*
|
||||
* <p>The bug these cover was silent and slow: a lead was never marked present (only workers are), so
|
||||
* every lead→lead delivery sat on the gate for the full readiness grace and failed ~60s later without
|
||||
* a keystroke ever reaching the pane.
|
||||
* <p>The bug these cover was silent and slow: a lead (and later a collaborator) was never marked
|
||||
* present (only workers are) and never counted as a lead, so every send to one sat on the gate for
|
||||
* the full readiness grace and failed ~60s later without a keystroke ever reaching the pane.
|
||||
*/
|
||||
class FleetDeliverabilityTest {
|
||||
|
||||
@@ -25,19 +27,25 @@ class FleetDeliverabilityTest {
|
||||
return () -> m;
|
||||
}
|
||||
|
||||
private static Supplier<Map<String, String>> collaborators(Map<String, String> m) {
|
||||
return () -> m;
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a worker that has connected its MCP is deliverable")
|
||||
void presentWorkerIsDeliverable() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
presence.markPresent("term_worker");
|
||||
|
||||
assertTrue(Fleetd.deliverableTo(presence, leads(Map.of())).test("term_worker"));
|
||||
assertTrue(Fleetd.deliverableTo(presence, leads(Map.of()), collaborators(Map.of()))
|
||||
.test("term_worker"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a worker still in its boot window is held back")
|
||||
void absentWorkerIsNotDeliverable() {
|
||||
assertFalse(Fleetd.deliverableTo(new MemberPresence(), leads(Map.of())).test("term_booting"));
|
||||
assertFalse(Fleetd.deliverableTo(new MemberPresence(), leads(Map.of()), collaborators(Map.of()))
|
||||
.test("term_booting"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -45,19 +53,31 @@ class FleetDeliverabilityTest {
|
||||
void leadIsDeliverableWithoutPresence() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
Predicate<String> deliverable =
|
||||
Fleetd.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")));
|
||||
Fleetd.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")), collaborators(Map.of()));
|
||||
|
||||
assertFalse(presence.isPresent("term_lead"), "a lead is never enrolled in worker presence");
|
||||
assertTrue(deliverable.test("term_lead"), "…and must be deliverable anyway");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an unknown terminal is deliverable to neither")
|
||||
@DisplayName("a collaborator is deliverable without ever being marked present or scanned as a lead")
|
||||
void collaboratorIsDeliverableWithoutPresenceOrLeadStatus() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
Predicate<String> deliverable = Fleetd.deliverableTo(presence, leads(Map.of()),
|
||||
collaborators(Map.of("term_collab", "kevin")));
|
||||
|
||||
assertFalse(presence.isPresent("term_collab"), "a collaborator is never enrolled in worker presence");
|
||||
assertTrue(deliverable.test("term_collab"), "…and must be deliverable anyway");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an unknown terminal is deliverable to none of presence, leads, or collaborators")
|
||||
void strangerIsNotDeliverable() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
presence.markPresent("term_worker");
|
||||
|
||||
assertFalse(Fleetd.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")))
|
||||
assertFalse(Fleetd.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")),
|
||||
collaborators(Map.of("term_collab", "kevin")))
|
||||
.test("term_stranger"));
|
||||
}
|
||||
|
||||
@@ -65,22 +85,47 @@ class FleetDeliverabilityTest {
|
||||
@DisplayName("a lead discovered after startup becomes deliverable with no restart")
|
||||
void leadSetIsReadThroughOnEveryCall() {
|
||||
Map<String, String> discovered = new HashMap<>();
|
||||
Predicate<String> deliverable = Fleetd.deliverableTo(new MemberPresence(), leads(discovered));
|
||||
Predicate<String> deliverable =
|
||||
Fleetd.deliverableTo(new MemberPresence(), leads(discovered), collaborators(Map.of()));
|
||||
|
||||
assertFalse(deliverable.test("term_late"));
|
||||
discovered.put("term_late", "gpt-sol-5.6"); // leadScan picks up a newly labelled tab
|
||||
assertTrue(deliverable.test("term_late"), "the supplier must be re-read, not snapshotted");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a collaborator discovered after startup becomes deliverable with no restart")
|
||||
void collaboratorSetIsReadThroughOnEveryCall() {
|
||||
Map<String, String> discovered = new HashMap<>();
|
||||
Predicate<String> deliverable =
|
||||
Fleetd.deliverableTo(new MemberPresence(), leads(Map.of()), collaborators(discovered));
|
||||
|
||||
assertFalse(deliverable.test("term_late_collab"));
|
||||
discovered.put("term_late_collab", "kevin"); // the same tab scan picks up a newly labelled collaborator tab
|
||||
assertTrue(deliverable.test("term_late_collab"), "the supplier must be re-read, not snapshotted");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("forgetting a torn-down worker does not strip a lead of its deliverability")
|
||||
void forgetDoesNotDisarmALead() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
Predicate<String> deliverable =
|
||||
Fleetd.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")));
|
||||
Fleetd.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")), collaborators(Map.of()));
|
||||
|
||||
presence.forget("term_lead"); // the injector's cleanup path runs against every target
|
||||
|
||||
assertTrue(deliverable.test("term_lead"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("forgetting a torn-down worker does not strip a collaborator of its deliverability")
|
||||
void forgetDoesNotDisarmACollaborator() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
Predicate<String> deliverable = Fleetd.deliverableTo(presence, leads(Map.of()),
|
||||
collaborators(Map.of("term_collab", "kevin")));
|
||||
|
||||
presence.forget("term_collab"); // the injector's cleanup path runs against every target
|
||||
|
||||
assertTrue(deliverable.test("term_collab"));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,183 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.fleet.msg.LeadChannel;
|
||||
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||
import dev.ltms.fleet.msg.LeadMessage;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertSame;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 step 4 ranks 3 and 8: the assembled daemon must use the AMQP openers from {@link
|
||||
* ResourcePorts}, and its startup report must describe the object the runtime actually owns. These
|
||||
* fakes never open a socket.
|
||||
*/
|
||||
class FleetdAssemblyAmqpOpenersTest {
|
||||
|
||||
private static final String COORD_ID = "assembly-test";
|
||||
|
||||
private static final class DurableReplyInbox implements ReplyInbox {
|
||||
@Override public void own(String target) { }
|
||||
@Override public void release(String target) { }
|
||||
@Override public void publish(String target, String msgId, String content) { }
|
||||
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||
@Override public boolean ack(String target, String msgId) { return false; }
|
||||
}
|
||||
|
||||
private static final class DurableLeadMailbox implements LeadChannelHandle {
|
||||
@Override public void publish(String toCoordId, LeadMessage message) { }
|
||||
@Override public List<LeadMessage> peek() { return List.of(); }
|
||||
@Override public void ack(String msgId) { }
|
||||
@Override public String selfCoordId() { return COORD_ID; }
|
||||
@Override public boolean heldDurable() { return true; }
|
||||
@Override public MailboxState inspect(String coordId) { return MailboxState.unknown(coordId); }
|
||||
@Override public void close() { }
|
||||
}
|
||||
|
||||
private static final class RecordingPorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
final DurableReplyInbox replyInbox = new DurableReplyInbox();
|
||||
final DurableLeadMailbox leadMailbox = new DurableLeadMailbox();
|
||||
final AtomicInteger replyOpenCalls = new AtomicInteger();
|
||||
final AtomicInteger mailboxOpenCalls = new AtomicInteger();
|
||||
final boolean openSucceeds;
|
||||
|
||||
RecordingPorts(boolean openSucceeds) {
|
||||
this.openSucceeds = openSucceeds;
|
||||
}
|
||||
|
||||
@Override public Map<String, String> environment() { return Map.of(); }
|
||||
@Override public HerdrClient connectHerdr(Path socketPath) { return herdr; }
|
||||
@Override public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> {
|
||||
replyOpenCalls.incrementAndGet();
|
||||
if (!openSucceeds) throw new IllegalStateException("fake reply broker is down");
|
||||
return replyInbox;
|
||||
};
|
||||
}
|
||||
@Override public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfId, prefetch) -> {
|
||||
mailboxOpenCalls.incrementAndGet();
|
||||
if (!openSucceeds) throw new IllegalStateException("fake coordination broker is down");
|
||||
return leadMailbox;
|
||||
};
|
||||
}
|
||||
@Override public LongSupplier nanoClock() { return System::nanoTime; }
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
@Override public LongSupplier wallClockNanos() { return System::nanoTime; }
|
||||
@Override public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
@Override public void addShutdownHook(Runnable hook) { }
|
||||
@Override public void startHttp(Javalin app, String host, int port) { }
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Files.createDirectories(dir);
|
||||
Path config = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(config, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-reply-broker/vh"
|
||||
coordinator:
|
||||
uri: "amqp://fake-coordination-broker/vh"
|
||||
selfId: "assembly-test"
|
||||
""");
|
||||
return FleetConfig.load(config);
|
||||
}
|
||||
|
||||
private static FleetdRuntime assemble(Path dir, RecordingPorts ports) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
return FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, new ConfigRef(dir.resolve("fleetd.yaml"), cfg),
|
||||
new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||
}
|
||||
|
||||
private static boolean reportContains(ListAppender<ILoggingEvent> appender, String text) {
|
||||
return appender.list.stream().map(ILoggingEvent::getFormattedMessage).anyMatch(message -> message.contains(text));
|
||||
}
|
||||
|
||||
@Test
|
||||
void assembledAmqpOpenersAndTheirReportsAgreeOnDurableAndFallbackStates(@TempDir Path dir) throws Exception {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
Level oldLevel = logger.getLevel();
|
||||
ListAppender<ILoggingEvent> reports = new ListAppender<>();
|
||||
reports.start();
|
||||
logger.setLevel(Level.INFO);
|
||||
logger.addAppender(reports);
|
||||
try {
|
||||
RecordingPorts durablePorts = new RecordingPorts(true);
|
||||
FleetdRuntime durable = assemble(dir.resolve("durable"), durablePorts);
|
||||
try {
|
||||
// Control: this fails loudly if the assembly did not run or used an inert opener.
|
||||
assertEquals(1, durablePorts.replyOpenCalls.get(), "assembly must call replyInboxOpener once");
|
||||
assertEquals(1, durablePorts.mailboxOpenCalls.get(), "assembly must call leadMailboxOpener once");
|
||||
assertSame(durablePorts.replyInbox, durable.replyInbox(),
|
||||
"the durable reply report must describe the exact inbox the runtime owns");
|
||||
assertSame(durablePorts.leadMailbox, durable.leadMailbox(),
|
||||
"the coordination-on report must describe the exact mailbox the runtime owns");
|
||||
assertNotNull(durable.leadCoordLoop(), "a durable mailbox must start lead coordination");
|
||||
assertTrue(reportContains(reports, "reply inbox: AMQP broker (durable)"));
|
||||
assertTrue(reportContains(reports, "lead coordination: ON as coord-id " + COORD_ID));
|
||||
} finally {
|
||||
durable.close();
|
||||
}
|
||||
|
||||
reports.list.clear();
|
||||
RecordingPorts fallbackPorts = new RecordingPorts(false);
|
||||
FleetdRuntime fallback = assemble(dir.resolve("fallback"), fallbackPorts);
|
||||
try {
|
||||
assertEquals(1, fallbackPorts.replyOpenCalls.get(), "assembly must call the failing reply opener once");
|
||||
assertEquals(1, fallbackPorts.mailboxOpenCalls.get(), "assembly must call the failing mailbox opener once");
|
||||
assertTrue(fallback.replyInbox() instanceof InMemoryReplyInbox,
|
||||
"a failed reply opener must make the runtime own the in-memory fallback");
|
||||
assertNull(fallback.leadMailbox(), "a failed mailbox opener must leave coordination off");
|
||||
assertNull(fallback.leadCoordLoop(), "coordination must not start without a mailbox");
|
||||
assertTrue(reportContains(reports, "reply inbox: in-memory (soft-state)"));
|
||||
assertTrue(reportContains(reports, "lead-to-lead messaging is OFF"));
|
||||
} finally {
|
||||
fallback.close();
|
||||
}
|
||||
} finally {
|
||||
logger.detachAppender(reports);
|
||||
logger.setLevel(oldLevel);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,166 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.auth.Authz;
|
||||
import dev.ltms.fleet.auth.Principal;
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Method;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* Asserts that the {@link FleetMcp} built by {@link FleetdAssembly#assembleAndStart} applies the
|
||||
* authorization table: a worker is refused {@code SPAWN}, and the primary is allowed it.
|
||||
*/
|
||||
class FleetdAssemblyAuthorizationModeTest {
|
||||
|
||||
private static final class TestResourcePorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> new ReplyInbox() {
|
||||
@Override public void own(String target) { }
|
||||
@Override public void release(String target) { }
|
||||
@Override public void publish(String target, String msgId, String content) { }
|
||||
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||
@Override public boolean ack(String target, String msgId) { return false; }
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException(
|
||||
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Binding a real port would clash with any daemon already listening on it.
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("FakeHerdr is healthy; no poll wait is expected");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private TestResourcePorts ports;
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
if (ports != null && ports.shutdownHook != null) {
|
||||
ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
health:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-test-broker/vh"
|
||||
""");
|
||||
return FleetConfig.load(file);
|
||||
}
|
||||
|
||||
private FleetMcp assemble(Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ports = new TestResourcePorts();
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg,
|
||||
new ConfigRef(dir.resolve("fleetd.yaml"), cfg), new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||
return runtime.mcp();
|
||||
}
|
||||
|
||||
/**
|
||||
* Invokes {@code FleetMcp#denyFor}, which is package-private to {@code dev.ltms.fleet.mcp}
|
||||
* while this test is in {@code dev.ltms.fleet}. Nothing here catches a missing method: if
|
||||
* {@code denyFor} is renamed or removed, {@link NoSuchMethodException} propagates and the
|
||||
* test fails.
|
||||
*/
|
||||
private static McpSchema.CallToolResult denyFor(FleetMcp mcp, Principal caller, Authz.Action action,
|
||||
String target) throws Exception {
|
||||
Method m = FleetMcp.class.getDeclaredMethod("denyFor", Principal.class, Authz.Action.class, String.class);
|
||||
m.setAccessible(true);
|
||||
return (McpSchema.CallToolResult) m.invoke(mcp, caller, action, target);
|
||||
}
|
||||
|
||||
@Test
|
||||
void productionBootPathRefusesAnUnauthorizedCallerThroughTheAssembledFleetMcp(@TempDir Path dir)
|
||||
throws Exception {
|
||||
FleetMcp mcp = assemble(dir);
|
||||
|
||||
McpSchema.CallToolResult deniedForWorker = denyFor(mcp, Principal.worker("term_a", 200),
|
||||
Authz.Action.SPAWN, "term_a");
|
||||
assertNotNull(deniedForWorker,
|
||||
"a worker must not be able to fleet_spawn through the assembled FleetMcp");
|
||||
assertTrue(deniedForWorker.isError(), "a refusal is returned as an MCP tool error");
|
||||
|
||||
McpSchema.CallToolResult allowedForPrimary = denyFor(mcp, Principal.primary(100),
|
||||
Authz.Action.SPAWN, "term_a");
|
||||
assertNull(allowedForPrimary,
|
||||
"control: the primary must still be allowed to fleet_spawn — otherwise the worker "
|
||||
+ "refusal above would pass even with the gate wired backwards");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,165 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertSame;
|
||||
|
||||
/**
|
||||
* fleetd #669 Unit E. Reaches the real {@link dev.ltms.fleet.herdr.HerdrRouter} that {@link
|
||||
* FleetdAssembly#assembleAndStart} builds and wires — not a copy built for this test — and proves
|
||||
* that a configured collaborator's terminal routes to the LEAD herdr daemon.
|
||||
*
|
||||
* <p>Two distinct {@link FakeHerdr} instances are required, the same pattern {@code
|
||||
* FleetdAssemblyConnectionIdentityTest} and {@code FleetdLeadRolloverAssemblyTest} already use:
|
||||
* with one client shared between {@code herdrSocket} and {@code memberHerdrSocket},
|
||||
* {@code HerdrRouter} folds {@code leadAgents} and {@code memberAgents} into the same instance
|
||||
* (see its constructor), and {@code agentsFor} would return that one object regardless of whether
|
||||
* the collaborator map was ever consulted — invisible to a mutation of the predicate this ticket
|
||||
* fixes. This test's two sockets resolve to two different fakes, so the assertion only passes when
|
||||
* the collaborator's terminal is actually recognised and routed to the lead one.
|
||||
*/
|
||||
class FleetdAssemblyCollaboratorHerdrRoutingTest {
|
||||
|
||||
private static final Path LEAD_SOCKET = Path.of("/fake/lead-herdr.sock");
|
||||
private static final Path MEMBER_SOCKET = Path.of("/fake/member-herdr.sock");
|
||||
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
final Map<Path, HerdrClient> herdrsBySocket = new LinkedHashMap<>();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
HerdrClient client = herdrsBySocket.get(socketPath);
|
||||
if (client == null) {
|
||||
throw new IllegalStateException("no fake herdr registered for socket " + socketPath);
|
||||
}
|
||||
return client;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> new ReplyInbox() {
|
||||
@Override public void own(String target) { }
|
||||
@Override public void release(String target) { }
|
||||
@Override public void publish(String target, String msgId, String content) { }
|
||||
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||
@Override public boolean ack(String target, String msgId) { return false; }
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException(
|
||||
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Do not bind a real port in this assembly test.
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("FakeHerdr is healthy; no poll wait is expected");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: "%s"
|
||||
memberHerdrSocket: "%s"
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
fleet:
|
||||
collaborators:
|
||||
reviewer-alex:
|
||||
tab: "collab: alex"
|
||||
profiles:
|
||||
sonnet:
|
||||
subscription: true
|
||||
argv: ["ccs", "sonnet"]
|
||||
""".formatted(LEAD_SOCKET, MEMBER_SOCKET));
|
||||
return FleetConfig.load(file);
|
||||
}
|
||||
|
||||
@Test
|
||||
void assembledRouterRoutesACollaboratorTerminalToTheLeadDaemon(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
|
||||
// The fixed FakeHerdr fixture already ties terminal "term_a" to a live agent on tab
|
||||
// "w2:t7" (pane "w2:p7") — seeding only the tab LABEL to match the configured collaborator
|
||||
// is enough to make LeadTabScanner resolve "term_a" as that collaborator. Seeded on the
|
||||
// LEAD fake only: a collaborator's pane lives in the lead daemon, exactly like a lead's.
|
||||
FakeHerdr lead = new FakeHerdr().withTab("w2", "w2:t7", "collab: alex");
|
||||
FakeHerdr member = new FakeHerdr();
|
||||
ports.herdrsBySocket.put(LEAD_SOCKET, lead);
|
||||
ports.herdrsBySocket.put(MEMBER_SOCKET, member);
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg,
|
||||
new ConfigRef(dir.resolve("fleetd.yaml"), cfg), new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||
try {
|
||||
assertSame(runtime.router().leadAgents(), runtime.router().agentsFor("term_a"),
|
||||
"a configured collaborator's terminal must route to the LEAD daemon — "
|
||||
+ "FleetdAssembly must wire the collaborator map into the router's "
|
||||
+ "predicate, not just LeadTabScanner.get()");
|
||||
assertSame(runtime.router().memberAgents(), runtime.router().agentsFor("term_shell"),
|
||||
"control: a terminal naming neither a lead nor a collaborator (term_shell, on "
|
||||
+ "the unlabelled tab w2:t8) must still route to the member daemon");
|
||||
} finally {
|
||||
assertNotNull(ports.shutdownHook, "control: assembly must capture its shutdown hook");
|
||||
ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,230 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
|
||||
/**
|
||||
* fleetd #612 step 2, unit B2 (CB-185, identity half). Replaces the deleted
|
||||
* {@code FleetdConnectionIdentityConstructionTest}, which pinned this claim by reading {@code
|
||||
* Fleetd.java}'s source text for {@code "new PaneLocator(herdr, memberHerdr)"}. That claim moved
|
||||
* to {@code FleetdAssembly.java} (fleetd #612 Unit A) and is pinned here instead, by driving the
|
||||
* real {@link ConnectionIdentity} — via {@code runtime.mcp().identity()}, not a copy — that {@link
|
||||
* FleetdAssembly#assembleAndStart} built.
|
||||
*
|
||||
* <p><strong>What this guards against</strong> (from the deleted test's own javadoc): pinning
|
||||
* {@code PaneLocator} to {@code memberHerdr} alone leaves every LEAD's own MCP connection
|
||||
* unresolvable ({@code callerTerminal == null}) the moment {@code memberHerdrSocket} names a
|
||||
* second daemon, which breaks {@code fleet_reply}/{@code fleet_ask}/{@code fleet_whoami} for a
|
||||
* lead. {@code PaneLocatorTest} already proves {@link PaneLocator} itself can search two clients
|
||||
* given two — the gap this pins is that the assembly actually passes it two, and in the right
|
||||
* order (lead first).
|
||||
*
|
||||
* <p><strong>Why this cannot be driven through a real MCP/HTTP round trip.</strong> The natural
|
||||
* way to observe {@code ConnectionIdentity} would be a real {@code fleet_whoami} call over the
|
||||
* built {@code FleetMcp}, the way {@code FleetMcpContextExtractorTest} drives its own
|
||||
* hand-built one. That does not work for the REAL assembly, because {@code FleetdAssembly} wires
|
||||
* {@code ConnectionIdentity} with a hardcoded {@code new LsofPeerPidLookup()} (see {@code
|
||||
* FleetdAssembly.java:444}), and {@code LsofPeerPidLookup} explicitly excludes its own PID — see
|
||||
* its javadoc: "we exclude our own PID and take the other end". In a JUnit test the HTTP client
|
||||
* and the daemon under test run in the very same JVM, so the "client" and "server" ends of the
|
||||
* loopback connection ARE the same PID, and {@code pidForLocalPort} always returns {@code -1}
|
||||
* before {@link PaneLocator} is ever reached — proving nothing about which daemon(s) got searched.
|
||||
* This test instead reaches the real {@link PaneLocator} the assembly built (through {@link
|
||||
* ConnectionIdentity#panes()}, added for exactly this) and drives it with a chosen pid directly,
|
||||
* bypassing the OS-dependent PID lookup entirely — a legitimate substitute, since the pid lookup
|
||||
* is not what CB-185 is about.
|
||||
*/
|
||||
class FleetdAssemblyConnectionIdentityTest {
|
||||
|
||||
/** Same shape as {@code FleetdAssemblyLifecycleTest}'s fake, but keys {@code connectHerdr} by
|
||||
* socket path so the lead and member daemons can be two DIFFERENT {@link FakeHerdr}s. */
|
||||
private static final class TwoHerdrResourcePorts implements ResourcePorts {
|
||||
|
||||
final Map<Path, HerdrClient> herdrsBySocket = new LinkedHashMap<>();
|
||||
final CopyOnWriteArrayList<ScheduledExecutorService> schedulers = new CopyOnWriteArrayList<>();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
HerdrClient client = herdrsBySocket.get(socketPath);
|
||||
if (client == null) {
|
||||
throw new IllegalStateException("no fake herdr registered for socket " + socketPath);
|
||||
}
|
||||
return client;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> new dev.ltms.fleet.msg.InMemoryReplyInbox();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException(
|
||||
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
ScheduledExecutorService scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
schedulers.add(scheduler);
|
||||
return scheduler;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Deliberately never bind — this test never issues a real HTTP request.
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private FleetdRuntime runtime;
|
||||
private TwoHerdrResourcePorts ports;
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
if (ports != null && ports.shutdownHook != null) {
|
||||
ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
|
||||
private static final Path LEAD_SOCKET = Path.of("/fake/lead-herdr.sock");
|
||||
private static final Path MEMBER_SOCKET = Path.of("/fake/member-herdr.sock");
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: "%s"
|
||||
memberHerdrSocket: "%s"
|
||||
lifecycle:
|
||||
idleTtlSeconds: 600
|
||||
health:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-test-broker/vh"
|
||||
""".formatted(LEAD_SOCKET, MEMBER_SOCKET));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
private FleetdRuntime assemble(Path dir, FakeHerdr lead, FakeHerdr member) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
ports = new TwoHerdrResourcePorts();
|
||||
ports.herdrsBySocket.put(LEAD_SOCKET, lead);
|
||||
ports.herdrsBySocket.put(MEMBER_SOCKET, member);
|
||||
runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
return runtime;
|
||||
}
|
||||
|
||||
/**
|
||||
* The pin. {@code lead} carries the one pane {@link FakeHerdr}'s canned {@code
|
||||
* pane.process_info} ties to {@link FakeHerdr#WORKER_PID} (pane {@code w2:p7}); {@code member}
|
||||
* reports NO panes at all ({@link FakeHerdr#withNoPanes()}) — modelling a second daemon that
|
||||
* simply does not host the caller's pane, exactly the CB-185 javadoc's scenario for a lead's
|
||||
* own connection. If {@code PaneLocator} only ever searches the member daemon (the bug), this
|
||||
* pid resolves to nothing, because the pane that owns it lives on the LEAD daemon the bug
|
||||
* skips.
|
||||
*/
|
||||
@Test
|
||||
void connectionIdentitySearchesTheLeadDaemonNotJustTheMemberOne(@TempDir Path dir) throws Exception {
|
||||
FakeHerdr lead = new FakeHerdr();
|
||||
FakeHerdr member = new FakeHerdr().withNoPanes();
|
||||
|
||||
assemble(dir, lead, member);
|
||||
|
||||
PaneLocator panes = runtime.mcp().identity().panes();
|
||||
PaneLocator.Lookup lookup = panes.terminalForPid(FakeHerdr.WORKER_PID);
|
||||
|
||||
assertEquals("term_a", lookup.terminal(),
|
||||
"the pane owning WORKER_PID lives on the LEAD daemon only (the member fake reports "
|
||||
+ "no panes) — PaneLocator must still find it, which is only possible if it "
|
||||
+ "searches the lead client and not just the member one");
|
||||
}
|
||||
|
||||
/**
|
||||
* The mirror control: when the pane instead lives ONLY on the member daemon (the lead reports
|
||||
* no panes), the lookup must still find it — proving the member client is genuinely searched
|
||||
* too, not merely tolerated as a second, always-losing argument.
|
||||
*/
|
||||
@Test
|
||||
void connectionIdentityAlsoSearchesTheMemberDaemon(@TempDir Path dir) throws Exception {
|
||||
FakeHerdr lead = new FakeHerdr().withNoPanes();
|
||||
FakeHerdr member = new FakeHerdr();
|
||||
|
||||
assemble(dir, lead, member);
|
||||
|
||||
PaneLocator panes = runtime.mcp().identity().panes();
|
||||
PaneLocator.Lookup lookup = panes.terminalForPid(FakeHerdr.WORKER_PID);
|
||||
|
||||
assertEquals("term_a", lookup.terminal(),
|
||||
"the pane owning WORKER_PID lives on the MEMBER daemon only — PaneLocator must "
|
||||
+ "find it there too");
|
||||
}
|
||||
|
||||
/** Sanity control: a pid nobody owns resolves to nothing on either daemon. */
|
||||
@Test
|
||||
void aPidNoPaneOwnsResolvesToNoTerminalOnEitherDaemon(@TempDir Path dir) throws Exception {
|
||||
FakeHerdr lead = new FakeHerdr();
|
||||
FakeHerdr member = new FakeHerdr();
|
||||
|
||||
assemble(dir, lead, member);
|
||||
|
||||
PaneLocator panes = runtime.mcp().identity().panes();
|
||||
PaneLocator.Lookup lookup = panes.terminalForPid(999_999L);
|
||||
|
||||
assertNull(lookup.terminal());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,219 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.msg.LeadChannel;
|
||||
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||
import dev.ltms.fleet.msg.LeadMessage;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertSame;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 A-gaps (gap 1): {@code FleetdAssemblyLifecycleTest}'s own class javadoc says plainly
|
||||
* that it leaves {@code coordinator:} unset, so {@code leadMailbox} and {@code leadCoordLoop} stay
|
||||
* {@code null} throughout — the configured-coordinator path is never exercised by Unit A's own
|
||||
* test. This class drives that path instead: a real {@code coordinator:} block, a fake {@link
|
||||
* Fleetd.LeadMailboxOpener} returning a fake closeable channel (never a real broker connection),
|
||||
* and proof that {@link FleetdAssembly#assembleAndStart} both builds it and, on shutdown, closes it.
|
||||
*
|
||||
* <p>Made possible by generalising {@code Fleetd.LeadMailboxOpener}'s return type (and {@code
|
||||
* FleetdRuntime}'s field) from the concrete {@code LeadMailbox} to {@link LeadChannelHandle} — a
|
||||
* {@link LeadChannel} its owner can also close. {@code FleetMcp} and {@code LeadCoordLoop} already
|
||||
* consumed the narrower {@link LeadChannel}; this only widens the one seam that owns and closes it.
|
||||
*/
|
||||
class FleetdAssemblyCoordinatorLifecycleTest {
|
||||
|
||||
private static final String SELF_COORD_ID = "test-lead";
|
||||
|
||||
/** A fake {@link LeadChannelHandle}: never touches a broker, and records whether it was closed. */
|
||||
private static final class FakeLeadChannel implements LeadChannelHandle {
|
||||
volatile boolean closed = false;
|
||||
|
||||
@Override
|
||||
public void publish(String toCoordId, LeadMessage m) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<LeadMessage> peek() {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String msgId) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public String selfCoordId() {
|
||||
return SELF_COORD_ID;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean heldDurable() {
|
||||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public MailboxState inspect(String coordId) {
|
||||
return MailboxState.unknown(coordId);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
closed = true;
|
||||
}
|
||||
}
|
||||
|
||||
/** Minimal fake {@link ResourcePorts}: a real herdr fake, a fake reply inbox, and a real, offered fake lead channel. */
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||
final FakeLeadChannel leadChannel = new FakeLeadChannel();
|
||||
String offeredUri;
|
||||
String offeredSelfId;
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> replyInbox;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
offeredUri = uri;
|
||||
offeredSelfId = selfCoordId;
|
||||
return leadChannel;
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// No real HTTP bind in a unit test.
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private static final class SentinelReplyInbox implements ReplyInbox {
|
||||
@Override
|
||||
public void own(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean ack(String target, String msgId) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
coordinator:
|
||||
uri: "amqp://fake-lead-broker/vh"
|
||||
selfId: "test-lead"
|
||||
""");
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
@Test
|
||||
void configuredCoordinatorIsBuiltByTheAssemblyAndClosedOnShutdown(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
|
||||
// --- the assembly actually calls the configured opener and builds the coordinator path ---
|
||||
assertEquals("amqp://fake-lead-broker/vh", ports.offeredUri,
|
||||
"the assembly must open the mailbox at the configured broker uri");
|
||||
assertEquals(SELF_COORD_ID, ports.offeredSelfId,
|
||||
"the assembly must open the mailbox under the configured selfId");
|
||||
assertSame(ports.leadChannel, runtime.leadMailbox(),
|
||||
"FleetdRuntime must own the exact LeadChannelHandle the opener returned, not a copy");
|
||||
assertNotNull(runtime.leadCoordLoop(),
|
||||
"a configured coordinator: block must build the receiving LeadCoordLoop too");
|
||||
assertFalse(ports.leadChannel.closed, "the channel must still be open while the daemon is running");
|
||||
|
||||
// --- shutting the assembly down closes it -------------------------------------------------
|
||||
assertNotNull(ports.shutdownHook, "FleetdAssembly must have registered a shutdown hook");
|
||||
ports.shutdownHook.run();
|
||||
|
||||
assertTrue(ports.leadChannel.closed,
|
||||
"FleetdRuntime.close() must close the configured LeadChannelHandle");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,278 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.Timeout;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.net.URI;
|
||||
import java.net.http.HttpClient;
|
||||
import java.net.http.HttpRequest;
|
||||
import java.net.http.HttpResponse;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* fleetd #612 step 2, unit B2 (CB-185, {@code FleetApp} half). Replaces the deleted {@code
|
||||
* FleetdFleetAppConstructionTest}, which pinned this claim by reading {@code Fleetd.java}'s
|
||||
* source text for {@code "new FleetApp(herdr, memberHerdr, workers,"}. That claim moved to {@code
|
||||
* FleetdAssembly.java} (fleetd #612 Unit A) and is pinned here instead, by driving the real {@code
|
||||
* Javalin} app — via {@code runtime.app()}, not a copy — that {@link
|
||||
* FleetdAssembly#assembleAndStart} built and handed to {@link FleetdRuntime}.
|
||||
*
|
||||
* <p><strong>What this guards against</strong> (from the deleted test's own javadoc): constructing
|
||||
* {@code FleetApp} with the lead-only {@code herdr} client (dropping {@code memberHerdr}) makes
|
||||
* {@code GET /healthz} report green while the MEMBER daemon is down — so every spawn fails
|
||||
* invisibly — and silently drops every member workspace from {@code GET /sessions}. {@code
|
||||
* FleetAppTwoDaemonTest} already proves {@code FleetApp} itself merges/gates correctly given two
|
||||
* clients; the gap this pins is that the assembly actually passes it two.
|
||||
*
|
||||
* <p><strong>Both directions, not just one</strong> (fleetd #612 issue comment 17525): the deleted
|
||||
* guard's positive assertion required the exact pair {@code "new FleetApp(herdr, memberHerdr,
|
||||
* workers,"}, which does not survive EITHER daemon being dropped. An earlier version of this class
|
||||
* only proved the member-dropped direction, which left {@code new FleetApp(memberHerdr,
|
||||
* memberHerdr, ...)} — the symmetric bug, {@code /healthz} green while the LEAD daemon is down —
|
||||
* an undetected regression. {@link #healthzGoesRedWhenTheLeadDaemonIsDownEvenThoughTheMemberIsUp}
|
||||
* closes that.
|
||||
*
|
||||
* <p>Unlike the {@code ConnectionIdentity} half of CB-185 ({@code
|
||||
* FleetdAssemblyConnectionIdentityTest}), {@code /healthz} needs no caller identity at all, so
|
||||
* this test can bind {@link FleetdRuntime#app()} to a REAL ephemeral port (exactly {@code
|
||||
* FleetAppTwoDaemonTest} does for its own hand-built {@code FleetApp}) and drive it with a real
|
||||
* {@code HttpClient} — no accessor needed for this half.
|
||||
*
|
||||
* <p><strong>{@code GET /sessions} could not be driven the same way</strong>, so this class does
|
||||
* not pin the merge half of the deleted test's javadoc. This class configures no {@code auth:}
|
||||
* block, so it runs under the default {@code loopback-trust} mode ({@code FleetConfig}). Under
|
||||
* that mode, {@code /sessions} requires {@code Authz.Action.READ}, which — through the REAL
|
||||
* assembly's real {@code CallerResolver}/{@code ConnectionIdentity} (built with a hardcoded
|
||||
* {@code new LsofPeerPidLookup()}) — needs {@code Caller.resolved()}, i.e. a real positive pid
|
||||
* from {@code lsof}. {@code LsofPeerPidLookup} excludes its own pid (see its javadoc), and a
|
||||
* JUnit test's HTTP client and the daemon under test share one JVM pid, so the resolved pid is
|
||||
* always {@code -1} and every such request is refused as {@code ANONYMOUS} (fleetd #317's
|
||||
* fail-closed rule) before the route handler — and its {@code memberHerdr} merge — is ever
|
||||
* reached. Verified directly: driving {@code GET /sessions} here returns {@code 401
|
||||
* unauthenticated}, not the merged body. {@code FleetAppTwoDaemonTest} avoids this because it
|
||||
* builds {@code FleetApp} with {@code callers: null}, which is not what the real assembly
|
||||
* passes. The {@code /healthz} pin below is what this class relies on for CB-185's {@code
|
||||
* FleetApp} half; {@code FleetAppTwoDaemonTest} remains the full behavioural proof that
|
||||
* {@code FleetApp} itself merges {@code /sessions} correctly once handed two clients.
|
||||
*
|
||||
* <p><strong>This refusal is {@code loopback-trust}-specific, not a property of {@code
|
||||
* CallerResolver} in general.</strong> Under {@code auth.mode: token}, {@code
|
||||
* CallerResolver#resolve} returns before ever consulting {@code Caller.resolved()} or {@code
|
||||
* Caller.scanComplete()}: a request carrying a valid bearer token in its {@code Authorization}
|
||||
* header resolves to {@code Role#PRIMARY} with no pid lookup at all, so the same-JVM-pid
|
||||
* exclusion above never comes into play. {@code FleetdQuarantineOutageDualWindowAssemblyTest}
|
||||
* and {@code FleetdListReportingSourcesAssemblyTest} both drive {@code Authz.Action.READ} this
|
||||
* way, over a real {@code McpSyncClient}/{@code HttpClient} against a real {@code
|
||||
* FleetdAssembly#assembleAndStart}, and both get the real response rather than a refusal.
|
||||
*
|
||||
* <p><strong>fleetd #629 follow-up.</strong> The fix below (see {@link TwoHerdrResourcePorts})
|
||||
* makes {@link #healthzGoesRedWhenTheLeadDaemonIsDownEvenThoughTheMemberIsUp}'s fake {@code
|
||||
* nanoClock()} frozen unless {@code herdrPollWait()} itself advances it. That is a sharper pin
|
||||
* than an assertion — if a future edit to {@code FleetdAssembly} ever bypasses {@code
|
||||
* ports.herdrPollWait()} again (e.g. reverting to a hardcoded {@code Thread.sleep}), the clock
|
||||
* never advances, {@code Fleetd#awaitHerdr}'s deadline is never reached, and this test hangs
|
||||
* forever instead of failing — proven by deliberately reintroducing that exact regression while
|
||||
* fixing this ticket. {@code @Timeout} turns that silent hang into a bounded, named test failure:
|
||||
* {@code SEPARATE_THREAD} so JUnit's timeout governor can actually interrupt a thread stuck in a
|
||||
* real {@code Thread.sleep} loop (the default {@code SAME_THREAD} mode cannot — it only measures
|
||||
* elapsed time after the test method returns on its own, which never happens here). 10 seconds is
|
||||
* roughly 150x the real passing times measured here (~0.06s), so a slow CI machine has no reason
|
||||
* to flake, and it is still 3x faster than discovering the regression by burning a CI job's whole
|
||||
* wall-clock budget.
|
||||
*/
|
||||
@Timeout(value = 10, unit = TimeUnit.SECONDS, threadMode = Timeout.ThreadMode.SEPARATE_THREAD)
|
||||
class FleetdAssemblyFleetAppTest {
|
||||
|
||||
private static final class TwoHerdrResourcePorts implements ResourcePorts {
|
||||
|
||||
final Map<Path, HerdrClient> herdrsBySocket = new LinkedHashMap<>();
|
||||
final CopyOnWriteArrayList<ScheduledExecutorService> schedulers = new CopyOnWriteArrayList<>();
|
||||
// fleetd #629: a fake, advanceable clock — NOT System::nanoTime. awaitHerdr's poll wait
|
||||
// (herdrPollWait() below) advances this on every poll instead of sleeping for real, so the
|
||||
// down-lead test below reaches awaitHerdr's deadline without burning real wall-clock time.
|
||||
final AtomicLong nowNanos = new AtomicLong(1_000_000_000L); // arbitrary non-zero start
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
HerdrClient client = herdrsBySocket.get(socketPath);
|
||||
if (client == null) {
|
||||
throw new IllegalStateException("no fake herdr registered for socket " + socketPath);
|
||||
}
|
||||
return client;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> new dev.ltms.fleet.msg.InMemoryReplyInbox();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException(
|
||||
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// fleetd #629: advance the fake clock instead of a real Thread.sleep, so awaitHerdr's
|
||||
// deadline is reached in real time regardless of the configured poll interval.
|
||||
return () -> nowNanos.addAndGet(TimeUnit.SECONDS.toNanos(1));
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
ScheduledExecutorService scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
schedulers.add(scheduler);
|
||||
return scheduler;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Deliberately never bind here — this test binds runtime.app() itself, for real, below.
|
||||
}
|
||||
}
|
||||
|
||||
private static final Path LEAD_SOCKET = Path.of("/fake/lead-herdr.sock");
|
||||
private static final Path MEMBER_SOCKET = Path.of("/fake/member-herdr.sock");
|
||||
|
||||
private final HttpClient http = HttpClient.newHttpClient();
|
||||
private FleetdRuntime runtime;
|
||||
private TwoHerdrResourcePorts ports;
|
||||
private Javalin boundApp;
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
if (boundApp != null) {
|
||||
boundApp.stop();
|
||||
}
|
||||
if (ports != null && ports.shutdownHook != null) {
|
||||
ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: "%s"
|
||||
memberHerdrSocket: "%s"
|
||||
lifecycle:
|
||||
idleTtlSeconds: 600
|
||||
health:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-test-broker/vh"
|
||||
""".formatted(LEAD_SOCKET, MEMBER_SOCKET));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
/** Assembles the real graph, then binds the real {@code Javalin app} to an ephemeral port. */
|
||||
private int assembleAndBind(Path dir, FakeHerdr lead, FakeHerdr member) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
ports = new TwoHerdrResourcePorts();
|
||||
ports.herdrsBySocket.put(LEAD_SOCKET, lead);
|
||||
ports.herdrsBySocket.put(MEMBER_SOCKET, member);
|
||||
runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
boundApp = runtime.app().start("127.0.0.1", 0);
|
||||
return boundApp.port();
|
||||
}
|
||||
|
||||
private HttpResponse<String> get(int port, String path) throws Exception {
|
||||
HttpRequest req = HttpRequest.newBuilder(URI.create("http://127.0.0.1:" + port + path)).GET().build();
|
||||
return http.send(req, HttpResponse.BodyHandlers.ofString());
|
||||
}
|
||||
|
||||
/**
|
||||
* The pin. The MEMBER daemon is down; the LEAD daemon is healthy. If the assembly built
|
||||
* {@code FleetApp} with only the lead client (the bug: passing {@code herdr} where {@code
|
||||
* memberHerdr} is expected), the down member is invisible and {@code /healthz} stays 200.
|
||||
*/
|
||||
@Test
|
||||
void healthzGoesRedWhenTheMemberDaemonIsDownEvenThoughTheLeadIsUp(@TempDir Path dir) throws Exception {
|
||||
FakeHerdr lead = new FakeHerdr();
|
||||
FakeHerdr member = new FakeHerdr().healthy(false);
|
||||
|
||||
int port = assembleAndBind(dir, lead, member);
|
||||
|
||||
HttpResponse<String> res = get(port, "/healthz");
|
||||
assertEquals(503, res.statusCode(),
|
||||
"a down MEMBER daemon must not be masked by a healthy lead: " + res.body());
|
||||
}
|
||||
|
||||
/** Sanity control: both daemons healthy must still be green through the real assembly. */
|
||||
@Test
|
||||
void healthzIsGreenWhenBothDaemonsAreUp(@TempDir Path dir) throws Exception {
|
||||
FakeHerdr lead = new FakeHerdr();
|
||||
FakeHerdr member = new FakeHerdr();
|
||||
|
||||
int port = assembleAndBind(dir, lead, member);
|
||||
|
||||
assertEquals(200, get(port, "/healthz").statusCode());
|
||||
}
|
||||
|
||||
/**
|
||||
* The symmetric pin (fleetd #612 issue comment 17525): the LEAD daemon is down; the MEMBER
|
||||
* daemon is healthy. If the assembly built {@code FleetApp} with only the member client
|
||||
* (dropping {@code herdr} — the mirror of the bug above, {@code new FleetApp(memberHerdr,
|
||||
* memberHerdr, ...)}), the down LEAD is invisible and {@code /healthz} stays 200. Without this
|
||||
* case the pair above is one-directional and does not cover the deleted guard's positive
|
||||
* assertion (it required BOTH {@code herdr,} and {@code memberHerdr,} in that order).
|
||||
*/
|
||||
@Test
|
||||
void healthzGoesRedWhenTheLeadDaemonIsDownEvenThoughTheMemberIsUp(@TempDir Path dir) throws Exception {
|
||||
FakeHerdr lead = new FakeHerdr().healthy(false);
|
||||
FakeHerdr member = new FakeHerdr();
|
||||
|
||||
int port = assembleAndBind(dir, lead, member);
|
||||
|
||||
HttpResponse<String> res = get(port, "/healthz");
|
||||
assertEquals(503, res.statusCode(),
|
||||
"a down LEAD daemon must not be masked by a healthy member: " + res.body());
|
||||
}
|
||||
}
|
||||
+198
@@ -0,0 +1,198 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.health.FleetHealthMonitor;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 rank 7 — {@code FleetdAssembly.java:429} wires {@link FleetHealthMonitor}'s {@code
|
||||
* failTarget} callback with {@code Fleetd.healthFailTarget(messages)}. {@link
|
||||
* FleetdHealthFailTargetWiringTest} already pins that the FACTORY itself delegates to {@code
|
||||
* messages::abandon}, but it calls {@code Fleetd.healthFailTarget} directly — it never drives {@code
|
||||
* FleetdAssembly.assembleAndStart} and so cannot see whether the real call site at {@code :429}
|
||||
* still passes it the real, assembled {@link MessageService}. Swapping that argument for a no-op
|
||||
* {@code (a, b) -> {}} compiles clean and leaves the whole suite — including the factory-level test
|
||||
* — green: a dead member's waiting ticket then sits {@code PENDING} for the full 30-minute async
|
||||
* timeout instead of failing immediately.
|
||||
*
|
||||
* <p>This test assembles the real daemon with {@code health.enabled: true}, pulls the REAL {@code
|
||||
* failTarget} {@link BiConsumer} out of the REAL, assembled {@link FleetHealthMonitor} (via
|
||||
* reflection — the field is package-private to {@code dev.ltms.fleet.health}, and nothing public
|
||||
* exposes it; {@code StatusPollerResilienceTest} already uses the same technique in this suite), and
|
||||
* invokes it directly against the REAL {@link MessageService} {@link FleetdRuntime#messages()}
|
||||
* returns. A no-op lambda swapped in at the call site leaves the ticket {@code PENDING} forever,
|
||||
* which this test catches; the real one fails it.
|
||||
*/
|
||||
class FleetdAssemblyHealthFailTargetBehaviouralTest {
|
||||
|
||||
private static final String TARGET = "term_a";
|
||||
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no broker: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
health:
|
||||
enabled: true
|
||||
intervalSeconds: 30
|
||||
""");
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the real assembled FleetHealthMonitor's failTarget reaches the real "
|
||||
+ "MessageService.abandon, not a no-op")
|
||||
void assembledHealthFailTargetReachesRealMessages(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
assertNotNull(ports.shutdownHook, "FleetdAssembly must have registered a shutdown hook");
|
||||
|
||||
// Surefire runs the whole suite in one JVM fork, so the scheduler/loops this assembly starts
|
||||
// (SessionReaper, StatusPoller, the health monitor) must be torn down here, on the failure
|
||||
// path too — hence the try/finally, not just a statement at the end of the happy path.
|
||||
try {
|
||||
FleetHealthMonitor healthMonitor = runtime.healthMonitor();
|
||||
assertNotNull(healthMonitor, "health.enabled: true in this test's config, so "
|
||||
+ "FleetdAssembly.assembleAndStart must have built a real FleetHealthMonitor");
|
||||
|
||||
Field field = FleetHealthMonitor.class.getDeclaredField("failTarget");
|
||||
field.setAccessible(true);
|
||||
BiConsumer<String, String> failTarget = (BiConsumer<String, String>) field.get(healthMonitor);
|
||||
assertNotNull(failTarget, "FleetHealthMonitor's failTarget must never be null — the "
|
||||
+ "constructor itself requires it");
|
||||
|
||||
MessageService messages = runtime.messages();
|
||||
|
||||
// --- loud control: prove the assembled MessageService is actually wired up and a ticket is
|
||||
// genuinely PENDING before failTarget ever runs. If this fails, the test below would pass
|
||||
// vacuously on a MessageService that never got a ticket in the first place. TARGET has no
|
||||
// live agent behind it (no session was ever acquired), so nothing resolves this ticket on
|
||||
// its own — it stays PENDING until failTarget (or a timeout) ends it.
|
||||
String ticket = messages.sendAsync(TARGET, "long task");
|
||||
MessageService.TaskView before = messages.poll(ticket);
|
||||
assertEquals(MessageService.Phase.PENDING, before.phase(),
|
||||
"control: the async ticket must be PENDING before failTarget runs");
|
||||
|
||||
failTarget.accept(TARGET, "member unreachable (health monitor)");
|
||||
|
||||
MessageService.TaskView after = awaitTerminal(messages, ticket);
|
||||
assertEquals(MessageService.Phase.FAILED, after.phase(),
|
||||
"FleetdAssembly.java:429 must pass Fleetd.healthFailTarget(messages) built from the "
|
||||
+ "SAME assembled MessageService — a no-op BiConsumer at that call site leaves "
|
||||
+ "this ticket PENDING for the full 30-minute async timeout instead of failing it");
|
||||
assertTrue(after.detail() != null && after.detail().contains("member unreachable"),
|
||||
"the failure reason passed to failTarget.accept must reach MessageService.abandon and "
|
||||
+ "end up in the ticket's detail");
|
||||
} finally {
|
||||
// Proof the teardown actually ran, not just an assurance that a finally was added: the
|
||||
// captured shutdown hook's close order (FleetdAssemblyLifecycleTest) closes the herdr
|
||||
// client last, so ports.herdr.closed flips to true only if this hook really executed.
|
||||
ports.shutdownHook.run();
|
||||
assertTrue(ports.herdr.closed, "the captured shutdown hook must have run and closed herdr — "
|
||||
+ "proof this test's assembled background loops/scheduler were torn down");
|
||||
}
|
||||
}
|
||||
|
||||
private static MessageService.TaskView awaitTerminal(MessageService messages, String ticket)
|
||||
throws InterruptedException {
|
||||
long deadline = System.currentTimeMillis() + 5000;
|
||||
MessageService.TaskView view = messages.poll(ticket);
|
||||
while (view.phase() == MessageService.Phase.PENDING && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(10);
|
||||
view = messages.poll(ticket);
|
||||
}
|
||||
return view;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,200 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.LeadTabScanner;
|
||||
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||
import dev.ltms.fleet.msg.LeadCoordLoop;
|
||||
import dev.ltms.fleet.msg.LeadMessage;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertInstanceOf;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #670 — pins the {@code excludedWorkspaceLabels} argument {@link FleetdAssembly}'s
|
||||
* production boot path passes to {@link LeadTabScanner} ({@code Set.of()}).
|
||||
*
|
||||
* <p>{@code LeadTabScannerTest} already covers this constructor parameter, but it builds its own
|
||||
* {@link LeadTabScanner} with its own set, so it tests the seam and proves nothing about the
|
||||
* producer. This test instead reaches the exact object {@link FleetdAssembly#assembleAndStart}
|
||||
* builds: a {@code fleet.leaders:} block makes the assembly construct a real
|
||||
* {@link LeadTabScanner} for its local {@code leads} supplier, and a {@code coordinator:} block
|
||||
* makes it hand that same supplier instance to {@link LeadCoordLoop} (fleetd #637), which stores
|
||||
* it as a field. Reflection recovers it from there, and then from the scanner itself, so the
|
||||
* assertion is against the real production argument rather than a copy built for this test.
|
||||
*/
|
||||
class FleetdAssemblyLeadTabScannerExclusionTest {
|
||||
|
||||
private static final class FakeLeadChannel implements LeadChannelHandle {
|
||||
@Override
|
||||
public void publish(String toCoordId, LeadMessage message) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<LeadMessage> peek() {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String msgId) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public String selfCoordId() {
|
||||
return "test-lead";
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean heldDurable() {
|
||||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public MailboxState inspect(String coordId) {
|
||||
return MailboxState.unknown(coordId);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
private static final class TestResourcePorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> new ReplyInbox() {
|
||||
@Override public void own(String target) { }
|
||||
@Override public void release(String target) { }
|
||||
@Override public void publish(String target, String msgId, String content) { }
|
||||
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||
@Override public boolean ack(String target, String msgId) { return false; }
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> new FakeLeadChannel();
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Do not bind a real port in this assembly test.
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("FakeHerdr is healthy; no poll wait is expected");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
coordinator:
|
||||
uri: "amqp://fake-lead-broker/vh"
|
||||
selfId: "test-lead"
|
||||
fleet:
|
||||
leaders:
|
||||
primary:
|
||||
tab: "lead: primary"
|
||||
profile: sonnet
|
||||
profiles:
|
||||
sonnet:
|
||||
subscription: true
|
||||
argv: ["ccs", "sonnet"]
|
||||
""");
|
||||
return FleetConfig.load(file);
|
||||
}
|
||||
|
||||
@Test
|
||||
void productionBootPathPassesNoExcludedWorkspaceLabels(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
TestResourcePorts ports = new TestResourcePorts();
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg,
|
||||
new ConfigRef(dir.resolve("fleetd.yaml"), cfg), new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||
try {
|
||||
LeadCoordLoop coordLoop = runtime.leadCoordLoop();
|
||||
assertNotNull(coordLoop, "control: a configured coordinator: block must build LeadCoordLoop");
|
||||
|
||||
Field leadsField = LeadCoordLoop.class.getDeclaredField("leads");
|
||||
leadsField.setAccessible(true);
|
||||
@SuppressWarnings("unchecked")
|
||||
Supplier<Map<String, String>> leads = (Supplier<Map<String, String>>) leadsField.get(coordLoop);
|
||||
|
||||
assertInstanceOf(LeadTabScanner.class, leads,
|
||||
"control: a non-empty fleet.leaders: block must make FleetdAssembly build a real "
|
||||
+ "LeadTabScanner for its `leads` supplier, not the Map::of fallback — "
|
||||
+ "otherwise this test would pass for the wrong reason");
|
||||
|
||||
Field excludedField = LeadTabScanner.class.getDeclaredField("excludedWorkspaceLabels");
|
||||
excludedField.setAccessible(true);
|
||||
Set<?> excluded = (Set<?>) excludedField.get(leads);
|
||||
|
||||
assertTrue(excluded.isEmpty(),
|
||||
"FleetdAssembly must pass an empty excludedWorkspaceLabels to "
|
||||
+ "LeadTabScanner — scanning member tabs would demote the lead to a worker");
|
||||
} finally {
|
||||
assertNotNull(ports.shutdownHook, "control: assembly must capture its shutdown hook");
|
||||
ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,266 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 Unit A: proves {@link FleetdAssembly#assembleAndStart} — not a copy of its logic —
|
||||
* against a real {@link FleetConfig}, a {@link FakeHerdr} and a fake {@link ResourcePorts}, with no
|
||||
* real herdr socket, no real broker, and no real HTTP bind.
|
||||
*
|
||||
* <p><strong>The canonical order this test asserts against was recorded from {@code Fleetd.java}
|
||||
* BEFORE any code moved</strong> (fleetd #612 Unit A's mandated order of work), by reading the
|
||||
* original {@code main}'s body and its shutdown-hook {@code Thread}:
|
||||
*
|
||||
* <p>Start order: {@code SessionReaper.start()} → {@code StatusPoller.start()} →
|
||||
* {@code LeadHeartbeatLoop.start()} (opt-in) → {@code FleetHealthMonitor.start()} (opt-in) →
|
||||
* {@code LeadCoordLoop.start()} (opt-in) → {@code ConfigWatcher.start()} (opt-in) →
|
||||
* {@code app.start()} (HTTP), always last.
|
||||
*
|
||||
* <p>Close order (from the original shutdown hook body): {@code sessions.close(drainTimeoutSeconds)}
|
||||
* → {@code poller.stop()} → {@code messages.close()} → {@code pushLoop.close()} →
|
||||
* {@code heartbeat.close()} (if present) → {@code leadCoordLoop.close()} (if present) →
|
||||
* {@code leadCoordScheduler.shutdownNow()} (if present) → {@code healthMonitor.stop()} (if present)
|
||||
* → {@code configWatcher.stop()} (if present) → {@code mcp.close()} → {@code reaper.stop()} (if
|
||||
* present) → {@code idleSleepGuard.close()} (if present) → {@code replyInbox.close()} (if
|
||||
* {@code AutoCloseable}) → {@code leadMailbox.close()} (if present) → {@code router.close()}.
|
||||
*
|
||||
* <p>This test's config deliberately leaves {@code coordinator:} unset, so {@code leadMailbox} and
|
||||
* {@code leadCoordLoop} stay {@code null} throughout — the lead-mailbox resource-ledger criterion is
|
||||
* NOT exercised here; see the class-level caveat in the implementer's hand-off. {@code
|
||||
* idleSleepGuard.enabled: false} is set for the same kind of reason: it would otherwise try to spawn
|
||||
* a real {@code caffeinate} subprocess, which is not one of the resources the ticket's acceptance
|
||||
* criteria names (scheduler/inbox/mailbox/loop/MCP server/herdr router).
|
||||
*/
|
||||
class FleetdAssemblyLifecycleTest {
|
||||
|
||||
/** The one {@link HerdrClient} both {@code herdrSocket} and {@code memberHerdrSocket} resolve to. */
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||
final List<String> ledger = new CopyOnWriteArrayList<>();
|
||||
final List<ScheduledExecutorService> schedulers = new CopyOnWriteArrayList<>();
|
||||
final List<String> schedulerPurposes = new ArrayList<>();
|
||||
Runnable shutdownHook;
|
||||
Javalin startedApp;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
ledger.add("connectHerdr");
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
ledger.add("replyInboxOpener");
|
||||
return (uri, prefetch) -> replyInbox;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
// Never invoked: this test's config has no `coordinator:` block, so
|
||||
// Fleetd.openLeadMailbox returns null before calling the opener at all.
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException(
|
||||
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
ledger.add("newScheduler:" + purpose);
|
||||
schedulerPurposes.add(purpose);
|
||||
ScheduledExecutorService scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
schedulers.add(scheduler);
|
||||
return scheduler;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
ledger.add("addShutdownHook");
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Deliberately never call app.start(host, port): no real HTTP bind in a unit test.
|
||||
ledger.add("startHttp");
|
||||
this.startedApp = app;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
/** A fake {@link ReplyInbox} that is also {@link AutoCloseable}, so the ledger can prove it closes. */
|
||||
private static final class SentinelReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
volatile boolean closed = false;
|
||||
|
||||
@Override
|
||||
public void own(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean ack(String target, String msgId) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
closed = true;
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
lifecycle:
|
||||
idleTtlSeconds: 600
|
||||
health:
|
||||
enabled: true
|
||||
intervalSeconds: 30
|
||||
leadHeartbeat:
|
||||
idleAfterSeconds: 600
|
||||
backoffMs: 15000
|
||||
quietNudgeCap: 5
|
||||
configReload:
|
||||
enabled: true
|
||||
intervalSeconds: 30
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-test-broker/vh"
|
||||
""");
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
@Test
|
||||
void assemblesTheRealBootGraphWithoutTouchingAnyRealSocketBrokerOrPort(@TempDir Path dir)
|
||||
throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
|
||||
// --- start order: the herdr connect, then the recurring background loops, in the recorded
|
||||
// order, then the shutdown hook is registered, then (finally) HTTP "starts" -------------
|
||||
assertTrue(ports.ledger.indexOf("connectHerdr") < ports.ledger.indexOf("newScheduler:bridge-push-"),
|
||||
"herdr must connect before the push scheduler is created: " + ports.ledger);
|
||||
assertEquals(List.of("bridge-push-", "bridge-heartbeat-", "bridge-health-"), ports.schedulerPurposes,
|
||||
"the three always-created schedulers must be requested in exactly this order: "
|
||||
+ ports.schedulerPurposes);
|
||||
assertTrue(ports.ledger.indexOf("addShutdownHook") < ports.ledger.indexOf("startHttp"),
|
||||
"the shutdown hook must be registered before HTTP starts — the one statement that "
|
||||
+ "could not be reordered without changing FleetdRuntime's constructor shape, "
|
||||
+ "see FleetdAssembly's javadoc: " + ports.ledger);
|
||||
assertEquals(ports.ledger.size() - 1, ports.ledger.indexOf("startHttp"),
|
||||
"HTTP must start LAST of everything this fake observes: " + ports.ledger);
|
||||
assertNotNull(ports.startedApp, "FleetdAssembly must have built and handed off a real FleetApp");
|
||||
|
||||
// No real HTTP bind and no real herdr socket: this call returning at all, plus the ledger
|
||||
// above, is the proof — a real bind or a real UnixSocketHerdrClient.connect would have
|
||||
// thrown or hung against the sockets/ports this test never opened.
|
||||
assertNotNull(runtime.app(), "FleetdRuntime must own the same Javalin app that was built");
|
||||
assertEquals(ports.startedApp, runtime.app(), "attachApp must hand FleetdRuntime the SAME instance");
|
||||
|
||||
// --- the runtime owns the real, live objects — not a copy --------------------------------
|
||||
assertEquals(LoopWatchdog.State.RUNNING, runtime.reaper().health(),
|
||||
"SessionReaper must be running: lifecycle.idleTtlSeconds is configured");
|
||||
assertEquals(LoopWatchdog.State.RUNNING, runtime.poller().health(), "StatusPoller must be running");
|
||||
assertNotNull(runtime.heartbeat(), "leadHeartbeat: is configured, so the loop must be built and started");
|
||||
assertNotNull(runtime.healthMonitor(), "health.enabled: true, so the monitor must be built and started");
|
||||
assertNotNull(runtime.configWatcher(), "configReload.enabled: true, so the watcher must be built and started");
|
||||
// Documented gap (see class javadoc): no coordinator: block, so these stay null.
|
||||
assertEquals(null, runtime.leadCoordLoop(), "no coordinator: block — leadCoordLoop must stay unbuilt");
|
||||
assertEquals(null, runtime.leadMailbox(), "no coordinator: block — leadMailbox must stay unbuilt");
|
||||
assertEquals(runtime.replyInbox(), ports.replyInbox,
|
||||
"FleetdRuntime must own the exact ReplyInbox instance this fake's AmqpOpener returned");
|
||||
assertFalse(ports.herdr.closed, "herdr must still be open while the daemon is running");
|
||||
assertFalse(ports.replyInbox.closed, "the reply inbox must still be open while the daemon is running");
|
||||
for (ScheduledExecutorService scheduler : ports.schedulers) {
|
||||
assertFalse(scheduler.isShutdown(), "a scheduler must still be running while the daemon is up");
|
||||
}
|
||||
|
||||
// --- close order: invoke the captured shutdown-hook Runnable directly (no real JVM shutdown
|
||||
// happens in a unit test) and prove every resource this fake can observe is released --------
|
||||
assertNotNull(ports.shutdownHook, "FleetdAssembly must have registered a shutdown hook");
|
||||
ports.shutdownHook.run();
|
||||
|
||||
assertEquals(LoopWatchdog.State.STOPPED, runtime.reaper().health(), "SessionReaper must stop on close");
|
||||
assertEquals(LoopWatchdog.State.STOPPED, runtime.poller().health(), "StatusPoller must stop on close");
|
||||
assertTrue(ports.herdr.closed, "router.close() must close the herdr client last");
|
||||
assertTrue(ports.replyInbox.closed, "the AutoCloseable reply inbox must be closed");
|
||||
for (int i = 0; i < ports.schedulers.size(); i++) {
|
||||
assertTrue(ports.schedulers.get(i).isShutdown(),
|
||||
"scheduler for '" + ports.schedulerPurposes.get(i) + "' must be shut down by close(): "
|
||||
+ "LeadHeartbeatLoop/FleetHealthMonitor/ReplyPushLoop each call "
|
||||
+ "scheduler.shutdownNow() on the exact instance ports.newScheduler(...) handed them");
|
||||
}
|
||||
|
||||
// Calling the captured hook a second time must never happen for a real JVM shutdown hook,
|
||||
// but nothing above should have thrown — that already proves every accessed field's close()
|
||||
// tolerated running once, in the recorded order, without an exception escaping.
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,180 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.net.URI;
|
||||
import java.net.http.HttpClient;
|
||||
import java.net.http.HttpRequest;
|
||||
import java.net.http.HttpResponse;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 Shape A, rank 10: {@link FleetdAssembly} creates one {@link FleetMcp.LoopHealthSource}
|
||||
* from the real started {@code StatusPoller} and {@code SessionReaper}, then gives it to two operator
|
||||
* windows. These tests reach the real assembled objects through {@link FleetdRuntime}, rather than
|
||||
* building a second source beside them. A hardcoded {@code RUNNING} source would pass a simple
|
||||
* "running" test, so the mutation proof also mis-wires the source to never-started loops: both
|
||||
* windows must then report {@code STOPPED} and these assertions go red.
|
||||
*/
|
||||
class FleetdAssemblyLoopHealthTest {
|
||||
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> new dev.ltms.fleet.msg.InMemoryReplyInbox();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no coordinator is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("healthy FakeHerdr must not be polled");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Bind runtime.app() to an ephemeral port only in the REST assertion below.
|
||||
}
|
||||
}
|
||||
|
||||
private final HttpClient http = HttpClient.newHttpClient();
|
||||
private RecordingResourcePorts ports;
|
||||
private FleetdRuntime runtime;
|
||||
private Javalin boundApp;
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
if (boundApp != null) {
|
||||
boundApp.stop();
|
||||
}
|
||||
if (runtime != null) {
|
||||
runtime.close();
|
||||
}
|
||||
if (ports != null) {
|
||||
assertNotNull(ports.shutdownHook, "the assembly must register its shutdown hook");
|
||||
assertTrue(ports.herdr.closed, "FleetdRuntime.close must close the real assembled herdr client");
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
lifecycle:
|
||||
idleTtlSeconds: 600
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
health:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-test-broker/vh"
|
||||
""");
|
||||
return FleetConfig.load(file);
|
||||
}
|
||||
|
||||
private void assemble(Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ports = new RecordingResourcePorts();
|
||||
runtime = FleetdAssembly.assembleAndStart(
|
||||
new AssemblyInputs(cfg, new ConfigRef(dir.resolve("fleetd.yaml"), cfg),
|
||||
new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||
}
|
||||
|
||||
@Test
|
||||
void fleetListUsesTheRunningLoopsInTheRealAssembledMcp(@TempDir Path dir) throws Exception {
|
||||
assemble(dir);
|
||||
|
||||
// The field is the exact source captured by FleetMcp's fleet_list handler. Reflection is
|
||||
// necessary because FleetMcp has no public source accessor; it is not a source-text check.
|
||||
FleetMcp.LoopHealthSource loopHealth = loopHealthOf(runtime.mcp());
|
||||
assertRunning(loopHealth, "FleetMcp's real fleet_list source");
|
||||
}
|
||||
|
||||
@Test
|
||||
void healthzUsesTheRunningLoopsInTheRealAssembledApp(@TempDir Path dir) throws Exception {
|
||||
assemble(dir);
|
||||
boundApp = runtime.app().start("127.0.0.1", 0);
|
||||
|
||||
HttpRequest request = HttpRequest.newBuilder(
|
||||
URI.create("http://127.0.0.1:" + boundApp.port() + "/healthz")).GET().build();
|
||||
HttpResponse<String> response = http.send(request, HttpResponse.BodyHandlers.ofString());
|
||||
assertEquals(200, response.statusCode(), response.body());
|
||||
assertTrue(response.body().contains("\"statusPoller\":\"RUNNING\""), response.body());
|
||||
assertTrue(response.body().contains("\"sessionReaper\":\"RUNNING\""), response.body());
|
||||
}
|
||||
|
||||
private static FleetMcp.LoopHealthSource loopHealthOf(FleetMcp mcp) throws Exception {
|
||||
Field field = FleetMcp.class.getDeclaredField("loopHealth");
|
||||
field.setAccessible(true);
|
||||
return (FleetMcp.LoopHealthSource) field.get(mcp);
|
||||
}
|
||||
|
||||
private static void assertRunning(FleetMcp.LoopHealthSource source, String consumer) {
|
||||
assertEquals(LoopWatchdog.State.RUNNING, source.statusPoller().get(),
|
||||
consumer + " must report the started real StatusPoller as RUNNING");
|
||||
assertEquals(LoopWatchdog.State.RUNNING, source.sessionReaper().get(),
|
||||
consumer + " must report the started real SessionReaper as RUNNING");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,232 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 rank 6 — {@code FleetdAssembly.java:447} wires {@code
|
||||
* sessions.onRelease(Fleetd.releaseCleanup(messages, replyInbox, primaryRegistry))}. {@link
|
||||
* FleetdReleaseCleanupWiringTest} already pins that the FACTORY {@code Fleetd.releaseCleanup}
|
||||
* itself reaches all three collaborators — but it calls the factory directly, never {@code
|
||||
* FleetdAssembly.assembleAndStart}, so it cannot see whether the real call site at {@code :447}
|
||||
* still registers it (as opposed to a no-op {@code detail -> { }}) or still passes it the REAL,
|
||||
* assembled {@code messages}/{@code replyInbox}/{@code primaryRegistry}. Swapping the registered
|
||||
* listener for a no-op at that call site compiles clean and leaves the whole suite — including the
|
||||
* factory-level test — green: EVERY teardown then leaks a stuck rendezvous waiter, an unreleased
|
||||
* reply-inbox consumer, and a stale lead binding, all three at once.
|
||||
*
|
||||
* <p>This test assembles the real daemon with {@code idleSleepGuard.enabled: false} — the ONLY
|
||||
* other {@code onRelease} registration in {@code FleetdAssembly} (see {@code
|
||||
* dev.ltms.fleet.power.IdleSleepGuard}'s own wiring at {@code FleetdAssembly.java:228}) — so the
|
||||
* real {@link SessionManager}'s release-listener list holds exactly the one listener this call site
|
||||
* registers. It pulls that REAL listener out via reflection (the list itself is private, like
|
||||
* {@code StatusPollerResilienceTest}'s use of the same technique elsewhere in this suite), invokes
|
||||
* it directly, and asserts all three collaborator effects against the REAL, assembled {@link
|
||||
* MessageService} ({@link FleetdRuntime#messages()}), the REAL {@link ReplyInbox} ({@link
|
||||
* FleetdRuntime#replyInbox()}), and the REAL {@link PrimaryRegistry} — reached through {@link
|
||||
* FleetdRuntime#pushLoop()}, the only other accessor that was handed the same {@code
|
||||
* primaryRegistry} instance ({@code FleetdAssembly.java:382}), since {@code FleetMcp} never exposes
|
||||
* it.
|
||||
*/
|
||||
class FleetdAssemblyReleaseCleanupBehaviouralTest {
|
||||
|
||||
private static final String TARGET = "term_a";
|
||||
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no broker: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
""");
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the real assembled release listener reaches messages.abandon, "
|
||||
+ "replyInbox.release, AND primaryRegistry.forgetDelegation — all three leaks at once")
|
||||
void assembledReleaseListenerReachesAllThreeCollaborators(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
assertNotNull(ports.shutdownHook, "FleetdAssembly must have registered a shutdown hook");
|
||||
|
||||
// Surefire runs the whole suite in one JVM fork, so the scheduler/loops this assembly starts
|
||||
// must be torn down here, on the failure path too — hence the try/finally, not just a
|
||||
// statement at the end of the happy path.
|
||||
try {
|
||||
// --- reach into SessionManager's private release-listener list. idleSleepGuard.enabled:
|
||||
// false above means FleetdAssembly.java:228 never registers, so this list must hold EXACTLY
|
||||
// the one listener :447 registers.
|
||||
Field listenersField = SessionManager.class.getDeclaredField("releaseListeners");
|
||||
listenersField.setAccessible(true);
|
||||
List<Consumer<SessionManager.ReleaseDetail>> releaseListeners =
|
||||
(List<Consumer<SessionManager.ReleaseDetail>>) listenersField.get(runtime.sessions());
|
||||
assertEquals(1, releaseListeners.size(), "control: with idleSleepGuard.enabled: false, "
|
||||
+ "FleetdAssembly.java:447 must be the ONLY onRelease registration — a different "
|
||||
+ "count means this test is no longer isolating the call site it claims to pin");
|
||||
Consumer<SessionManager.ReleaseDetail> releaseListener = releaseListeners.get(0);
|
||||
|
||||
MessageService messages = runtime.messages();
|
||||
ReplyInbox replyInbox = runtime.replyInbox();
|
||||
|
||||
// primaryRegistry is never exposed by FleetdRuntime directly — ReplyPushLoop is the other
|
||||
// collaborator FleetdAssembly.java:382 hands the SAME instance to, so reach it from there.
|
||||
Field primaryRegistryField = ReplyPushLoop.class.getDeclaredField("primaryRegistry");
|
||||
primaryRegistryField.setAccessible(true);
|
||||
PrimaryRegistry primaryRegistry = (PrimaryRegistry) primaryRegistryField.get(runtime.pushLoop());
|
||||
assertNotNull(primaryRegistry, "control: the assembled ReplyPushLoop must hold a real "
|
||||
+ "PrimaryRegistry instance");
|
||||
|
||||
// --- loud controls: set up the "before" state each collaborator's effect is measured
|
||||
// against, against the REAL assembled objects. If any of these three fails, the test below
|
||||
// would pass vacuously because the subject it claims to observe never existed in the first
|
||||
// place.
|
||||
String ticket = messages.sendAsync(TARGET, "long task");
|
||||
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket).phase(),
|
||||
"control: the async ticket must be PENDING before the release listener runs");
|
||||
|
||||
replyInbox.own(TARGET);
|
||||
replyInbox.publish(TARGET, "msg-1", "hello");
|
||||
assertEquals(1, replyInbox.peek(TARGET).size(),
|
||||
"control: the reply inbox must own TARGET and hold one message before the release "
|
||||
+ "listener runs");
|
||||
|
||||
primaryRegistry.recordDelegation(TARGET, "lead-1");
|
||||
assertEquals("lead-1", primaryRegistry.nudgeTargetFor(TARGET).orElse(null),
|
||||
"control: the delegation must be recorded before the release listener runs");
|
||||
|
||||
// --- the one call under test: invoke the REAL, assembled release listener directly, the
|
||||
// same way SessionManager.release(...) would on a real teardown.
|
||||
releaseListener.accept(new SessionManager.ReleaseDetail(TARGET, null, null, null, null));
|
||||
|
||||
MessageService.TaskView after = awaitTerminal(messages, ticket);
|
||||
assertEquals(MessageService.Phase.FAILED, after.phase(),
|
||||
"FleetdAssembly.java:447 must register a listener that calls messages.abandon(...) "
|
||||
+ "on the SAME assembled MessageService — an inert listener leaves this "
|
||||
+ "ticket PENDING for the full 30-minute async timeout");
|
||||
assertTrue(after.detail() != null && after.detail().contains("released"),
|
||||
"the abandon reason must say the worker session was released");
|
||||
|
||||
assertTrue(replyInbox.peek(TARGET).isEmpty(),
|
||||
"FleetdAssembly.java:447 must register a listener that calls replyInbox.release(...) "
|
||||
+ "— an inert listener leaves the inbox still owning TARGET with its message");
|
||||
|
||||
assertTrue(primaryRegistry.nudgeTargetFor(TARGET).isEmpty(),
|
||||
"FleetdAssembly.java:447 must register a listener that calls "
|
||||
+ "primaryRegistry.forgetDelegation(...) — an inert listener leaves the stale "
|
||||
+ "delegation in place");
|
||||
} finally {
|
||||
// Proof the teardown actually ran, not just an assurance that a finally was added: the
|
||||
// captured shutdown hook's close order (FleetdAssemblyLifecycleTest) closes the herdr
|
||||
// client last, so ports.herdr.closed flips to true only if this hook really executed.
|
||||
ports.shutdownHook.run();
|
||||
assertTrue(ports.herdr.closed, "the captured shutdown hook must have run and closed herdr — "
|
||||
+ "proof this test's assembled background loops/scheduler were torn down");
|
||||
}
|
||||
}
|
||||
|
||||
private static MessageService.TaskView awaitTerminal(MessageService messages, String ticket)
|
||||
throws InterruptedException {
|
||||
long deadline = System.currentTimeMillis() + 5000;
|
||||
MessageService.TaskView view = messages.poll(ticket);
|
||||
while (view.phase() == MessageService.Phase.PENDING && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(10);
|
||||
view = messages.poll(ticket);
|
||||
}
|
||||
return view;
|
||||
}
|
||||
}
|
||||
+241
@@ -0,0 +1,241 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.lead.LeadContextGauge;
|
||||
import dev.ltms.fleet.msg.LeadHeartbeatLoop;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.lang.reflect.Method;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #630, and fleetd #612's ranks for this call site. {@code FleetdAssembly.java:402}
|
||||
* computes {@code requireOperatorConfirm} from the effective {@code leadRollover.requireOperatorConfirm}
|
||||
* config, and {@code :409} threads it as the 14th argument into the full {@link LeadHeartbeatLoop}
|
||||
* constructor. Measured on 26f1986: dropping that one argument so the 13-argument overload is
|
||||
* selected instead (it delegates with {@code true} hardcoded — see that overload's own javadoc,
|
||||
* fleetd #621) compiles with 0 errors and leaves all 1883 tests green, both with and without the
|
||||
* argument. In production this means the daemon keeps starting and keeps nudging, but the
|
||||
* context-high notice silently goes back to telling EVERY lead to ask the operator before a
|
||||
* context roll — on a host that set {@code requireOperatorConfirm: false} specifically so it would
|
||||
* not have to. That is the operator's own fix silently reverting, with a fully green suite.
|
||||
*
|
||||
* <p>{@code LeadHeartbeatLoopTest} already proves {@link LeadHeartbeatLoop}'s package-private
|
||||
* {@code contextNotice(boolean, LeadContextGauge.Reading, boolean, boolean)} branches correctly on
|
||||
* its own {@code requireOperatorConfirm} argument — that the METHOD works. It says nothing about
|
||||
* which value {@code FleetdAssembly} actually passes into the constructed loop, so it is not
|
||||
* reused here as coverage for the call site.
|
||||
*
|
||||
* <p>This test assembles the real daemon TWICE — once with {@code leadRollover.requireOperatorConfirm:
|
||||
* false}, once with {@code true} — pulls the REAL {@code requireOperatorConfirm} field out of the
|
||||
* REAL, assembled {@link LeadHeartbeatLoop} each time (reflection: the field, and {@code
|
||||
* contextNotice} itself, are package-private to {@code dev.ltms.fleet.msg}, and nothing public
|
||||
* exposes either — the same technique {@code StatusPollerResilienceTest} already uses in this
|
||||
* suite), and calls the REAL {@code contextNotice} method with that field's value to produce the
|
||||
* actual notice text the assembled loop would append to a nudge. Both directions are asserted: a
|
||||
* one-directional test here would pass on a constant.
|
||||
*/
|
||||
class FleetdAssemblyRequireOperatorConfirmBehaviouralTest {
|
||||
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no broker: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir, boolean requireOperatorConfirm) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
leadHeartbeat:
|
||||
idleAfterSeconds: 600
|
||||
backoffMs: 15000
|
||||
quietNudgeCap: 5
|
||||
leadRollover:
|
||||
handoverPath: handover.md
|
||||
requireOperatorConfirm: %s
|
||||
""".formatted(requireOperatorConfirm));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
/** Carries both the assembled loop under test AND its {@link RecordingResourcePorts}, so the
|
||||
* caller can tear the assembly down (this test assembles the real daemon TWICE — see the class
|
||||
* javadoc — and each assembly needs its own teardown, not just the last one). */
|
||||
private record Assembled(LeadHeartbeatLoop heartbeat, RecordingResourcePorts ports) {
|
||||
}
|
||||
|
||||
private static Assembled assembleHeartbeat(Path dir, boolean requireOperatorConfirm) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir, requireOperatorConfirm);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
assertNotNull(ports.shutdownHook, "FleetdAssembly must have registered a shutdown hook");
|
||||
|
||||
LeadHeartbeatLoop heartbeat = runtime.heartbeat();
|
||||
assertNotNull(heartbeat, "control: leadHeartbeat: is configured, so FleetdAssembly.assembleAndStart "
|
||||
+ "must have built a real LeadHeartbeatLoop");
|
||||
return new Assembled(heartbeat, ports);
|
||||
}
|
||||
|
||||
/** Pulls the REAL {@code requireOperatorConfirm} field off the REAL, assembled loop. */
|
||||
private static boolean assembledRequireOperatorConfirm(LeadHeartbeatLoop heartbeat) throws Exception {
|
||||
Field field = LeadHeartbeatLoop.class.getDeclaredField("requireOperatorConfirm");
|
||||
field.setAccessible(true);
|
||||
return field.getBoolean(heartbeat);
|
||||
}
|
||||
|
||||
/** Calls the REAL, package-private {@code contextNotice(boolean, Reading, boolean, boolean)} via reflection. */
|
||||
private static String contextNotice(boolean enabled, LeadContextGauge.Reading reading, boolean alreadyNotified,
|
||||
boolean requireOperatorConfirm) throws Exception {
|
||||
Method method = LeadHeartbeatLoop.class.getDeclaredMethod("contextNotice", boolean.class,
|
||||
LeadContextGauge.Reading.class, boolean.class, boolean.class);
|
||||
method.setAccessible(true);
|
||||
return (String) method.invoke(null, enabled, reading, alreadyNotified, requireOperatorConfirm);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the real assembled LeadHeartbeatLoop's context-high notice tracks "
|
||||
+ "leadRollover.requireOperatorConfirm — BOTH directions")
|
||||
void assembledRequireOperatorConfirmControlsNoticeWording(@TempDir Path dir) throws Exception {
|
||||
LeadContextGauge.Reading highReading = new LeadContextGauge.Reading(LeadContextGauge.State.HIGH,
|
||||
250_000L, 2);
|
||||
|
||||
String noticeFalse;
|
||||
String noticeTrue;
|
||||
|
||||
// --- direction 1: requireOperatorConfirm: false -----------------------------------------
|
||||
Path falseDir = dir.resolve("false");
|
||||
Files.createDirectories(falseDir);
|
||||
Assembled assembledFalse = assembleHeartbeat(falseDir, false);
|
||||
// Surefire runs the whole suite in one JVM fork, so each assembly's scheduler/loops must be
|
||||
// torn down here, on the failure path too — hence try/finally per assembly (this test
|
||||
// assembles TWICE, so both need their own teardown, not just the last one).
|
||||
try {
|
||||
boolean fieldFalse = assembledRequireOperatorConfirm(assembledFalse.heartbeat());
|
||||
assertFalse(fieldFalse, "FleetdAssembly.java:402/:409 must thread leadRollover."
|
||||
+ "requireOperatorConfirm: false into the assembled LeadHeartbeatLoop's own field — "
|
||||
+ "dropping the 14th constructor argument selects the 13-argument overload, which "
|
||||
+ "hardcodes true regardless of config (fleetd #621), and this would read true instead");
|
||||
|
||||
noticeFalse = contextNotice(true, highReading, false, fieldFalse);
|
||||
assertTrue(noticeFalse.contains("Decide for yourself when to confirm"),
|
||||
"with requireOperatorConfirm: false, the assembled loop's own notice must tell the "
|
||||
+ "lead it can decide for itself — got: " + noticeFalse);
|
||||
assertFalse(noticeFalse.contains("ask the operator") || noticeFalse.contains("Only the operator"),
|
||||
"with requireOperatorConfirm: false, the assembled loop's own notice must NOT ask the "
|
||||
+ "operator — got: " + noticeFalse);
|
||||
} finally {
|
||||
// Proof the teardown actually ran, not just an assurance that a finally was added: the
|
||||
// captured shutdown hook's close order (FleetdAssemblyLifecycleTest) closes the herdr
|
||||
// client last, so ports.herdr.closed flips to true only if this hook really executed.
|
||||
assembledFalse.ports().shutdownHook.run();
|
||||
assertTrue(assembledFalse.ports().herdr.closed, "the captured shutdown hook must have run "
|
||||
+ "and closed herdr — proof this assembly's background loops/scheduler were torn down");
|
||||
}
|
||||
|
||||
// --- direction 2: requireOperatorConfirm: true -------------------------------------------
|
||||
Path trueDir = dir.resolve("true");
|
||||
Files.createDirectories(trueDir);
|
||||
Assembled assembledTrue = assembleHeartbeat(trueDir, true);
|
||||
try {
|
||||
boolean fieldTrue = assembledRequireOperatorConfirm(assembledTrue.heartbeat());
|
||||
assertTrue(fieldTrue, "FleetdAssembly.java:402/:409 must thread leadRollover."
|
||||
+ "requireOperatorConfirm: true into the assembled LeadHeartbeatLoop's own field");
|
||||
|
||||
noticeTrue = contextNotice(true, highReading, false, fieldTrue);
|
||||
assertTrue(noticeTrue.contains("ask the operator") && noticeTrue.contains("Only the operator can approve the roll"),
|
||||
"with requireOperatorConfirm: true, the assembled loop's own notice must ask the "
|
||||
+ "operator — got: " + noticeTrue);
|
||||
assertFalse(noticeTrue.contains("Decide for yourself when to confirm"),
|
||||
"with requireOperatorConfirm: true, the assembled loop's own notice must NOT tell "
|
||||
+ "the lead it can decide for itself — got: " + noticeTrue);
|
||||
} finally {
|
||||
assembledTrue.ports().shutdownHook.run();
|
||||
assertTrue(assembledTrue.ports().herdr.closed, "the captured shutdown hook must have run "
|
||||
+ "and closed herdr — proof this assembly's background loops/scheduler were torn down");
|
||||
}
|
||||
|
||||
// --- the two directions must actually differ: a constant return would pass both assertion
|
||||
// blocks above vacuously if they happened to share wording, so compare them directly too.
|
||||
assertTrue(!noticeFalse.equals(noticeTrue),
|
||||
"the two directions must produce genuinely different notice text — got the same "
|
||||
+ "text for both: " + noticeFalse);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,173 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import dev.ltms.fleet.testing.CapturedLog;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 A-gaps (gap 2): {@code Fleetd.java:185} used to call {@code
|
||||
* reportRoleFallbackGaps(cfg)} <em>outside</em> the boundary {@code FleetdAssembly.assembleAndStart}
|
||||
* — the assembly call itself sat at line 201, after it — so nothing that drives the assembly (the
|
||||
* seam {@code FleetdAssemblyLifecycleTest} exercises) could ever notice the call being deleted.
|
||||
* {@code Fleetd.main} still refuses a bad config end to end (see {@code
|
||||
* FleetdStartupValidationTest}), but that only pins {@code validateAll()} and {@code
|
||||
* assertChartersNameOnlyRegisteredTools} — a validator that <em>throws</em>. {@code
|
||||
* reportRoleFallbackGaps} only logs; nothing about {@code main} throwing or not throwing can
|
||||
* observe whether that particular call ran.
|
||||
*
|
||||
* <p>This drives {@link FleetdAssembly#assembleAndStart} directly — never a copy of its logic —
|
||||
* with a config that has no {@code fleet:} pools or charters configured for any role, so every role
|
||||
* trips both of {@code reportRoleFallbackGaps}' log branches, and asserts on the real log line a
|
||||
* {@code ListAppender} attached to the shared {@code Fleetd}/{@code FleetdAssembly} logger
|
||||
* captures. Deleting the call from {@code FleetdAssembly} (verified by hand, see the ticket) turns
|
||||
* this test red; deleting it from {@code Fleetd.main} instead (its old location) would not, which
|
||||
* is exactly the gap this test closes.
|
||||
*/
|
||||
class FleetdAssemblyRoleFallbackBoundaryTest {
|
||||
|
||||
/** Minimal fake {@link ResourcePorts}: enough for {@code assembleAndStart} to run with no real I/O. */
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> replyInbox;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException(
|
||||
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// No real HTTP bind in a unit test.
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private static final class SentinelReplyInbox implements ReplyInbox {
|
||||
@Override
|
||||
public void own(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean ack(String target, String msgId) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
// Deliberately no `fleet:` block at all: every MemberRole has neither a pool nor a
|
||||
// charter, so reportRoleFallbackGaps' "role fallback: no fleet.<role>s: pool for ..."
|
||||
// branch is guaranteed to log something to assert on.
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
""");
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
@Test
|
||||
void assembleAndStartItselfReportsTheRoleFallbackGaps(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
|
||||
try (var captured = CapturedLog.at(Fleetd.class, Level.INFO)) {
|
||||
FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
|
||||
String infoLines = captured.events().stream()
|
||||
.filter(e -> e.getLevel() == Level.INFO)
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.reduce("", (a, b) -> a + "\n" + b);
|
||||
assertTrue(infoLines.contains("role fallback: no fleet.<role>s: pool for"),
|
||||
() -> "FleetdAssembly.assembleAndStart itself must call reportRoleFallbackGaps "
|
||||
+ "(fleetd #612 A-gaps gap 2) — captured INFO lines: " + infoLines);
|
||||
} finally {
|
||||
if (ports.shutdownHook != null) {
|
||||
ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,190 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.inject.StatusPoller;
|
||||
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||
import dev.ltms.fleet.msg.LeadCoordLoop;
|
||||
import dev.ltms.fleet.msg.LeadMessage;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
|
||||
/**
|
||||
* Asserts that the assembled loops use the production reminder, coordination, and delivery timing
|
||||
* defaults when no {@code primary:} block configures the reply-push values.
|
||||
*/
|
||||
class FleetdAssemblyTimingDefaultsTest {
|
||||
|
||||
private static final class FakeLeadChannel implements LeadChannelHandle {
|
||||
@Override
|
||||
public void publish(String toCoordId, LeadMessage message) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<LeadMessage> peek() {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String msgId) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public String selfCoordId() {
|
||||
return "test-lead";
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean heldDurable() {
|
||||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public MailboxState inspect(String coordId) {
|
||||
return MailboxState.unknown(coordId);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
private static final class TestResourcePorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> new ReplyInbox() {
|
||||
@Override public void own(String target) { }
|
||||
@Override public void release(String target) { }
|
||||
@Override public void publish(String target, String msgId, String content) { }
|
||||
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||
@Override public boolean ack(String target, String msgId) { return false; }
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> new FakeLeadChannel();
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("FakeHerdr is healthy; no poll wait is expected");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private TestResourcePorts ports;
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
if (ports != null && ports.shutdownHook != null) {
|
||||
ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
coordinator:
|
||||
uri: "amqp://fake-lead-broker/vh"
|
||||
selfId: "test-lead"
|
||||
""");
|
||||
return FleetConfig.load(file);
|
||||
}
|
||||
|
||||
private FleetdRuntime assemble(Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ports = new TestResourcePorts();
|
||||
return FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg,
|
||||
new ConfigRef(dir.resolve("fleetd.yaml"), cfg), new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||
}
|
||||
|
||||
private static long longField(Object target, String name) throws Exception {
|
||||
Field field = target.getClass().getDeclaredField(name);
|
||||
field.setAccessible(true);
|
||||
return field.getLong(target);
|
||||
}
|
||||
|
||||
@Test
|
||||
void productionBootPathUsesTheExpectedLoopTimingDefaults(@TempDir Path dir) throws Exception {
|
||||
FleetdRuntime runtime = assemble(dir);
|
||||
|
||||
ReplyPushLoop pushLoop = runtime.pushLoop();
|
||||
assertEquals(5, longField(pushLoop, "maxReminders"),
|
||||
"without primary:, ReplyPushLoop must stop after five reminder attempts");
|
||||
assertEquals(15_000L, longField(pushLoop, "backoffMs"),
|
||||
"without primary:, ReplyPushLoop must wait fifteen seconds before the next reminder");
|
||||
|
||||
LeadCoordLoop leadCoordLoop = runtime.leadCoordLoop();
|
||||
assertNotNull(leadCoordLoop, "control: coordinator: must build LeadCoordLoop");
|
||||
assertEquals(3_000L, longField(leadCoordLoop, "intervalMs"),
|
||||
"LeadCoordLoop must poll for peer-lead mail every three seconds");
|
||||
|
||||
StatusPoller poller = runtime.poller();
|
||||
assertEquals(Injector.POLL_INTERVAL_MILLIS, longField(poller, "intervalMillis"),
|
||||
"StatusPoller must use Injector's delivery poll interval");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,185 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.inject.TurnListener;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 rank 12 — {@code FleetdAssembly} supplies the {@link Injector}'s {@code TurnRegistrar}
|
||||
* with {@code Fleetd.turnRegistrar(completion)}. The normal delivery path cannot distinguish that
|
||||
* registrar from {@code TurnRegistrar.NOOP}: {@link dev.ltms.fleet.inject.CompletionResolver#onDelivered}
|
||||
* registers the same turn shortly afterwards. The distinction matters when a delivery listener throws
|
||||
* after the pane received the message but before normal completion runs. The registrar must already have
|
||||
* registered the waiter with the REAL assembled resolver, so a later completion can still resolve it.
|
||||
*
|
||||
* <p>The test installs a throwing wrapper around the real assembled listener. It does not replace the
|
||||
* registrar or resolver. This constructs the narrow failure condition without a sleep, then observes the
|
||||
* real resolver through {@link FleetdRuntime#completion()}. An inert registrar, or a registrar wired to a
|
||||
* throwaway resolver, leaves this waiter's turn absent from the real resolver and makes the final assertion
|
||||
* fail.
|
||||
*/
|
||||
class FleetdAssemblyTurnRegistrarBehaviouralTest {
|
||||
|
||||
private static final String TARGET = "term_a";
|
||||
|
||||
private static final class ControllableResourcePorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
final AtomicLong nowNanos = new AtomicLong(1_000_000_000L);
|
||||
Runnable shutdownHook;
|
||||
|
||||
void advanceSeconds(long seconds) {
|
||||
nowNanos.addAndGet(TimeUnit.SECONDS.toNanos(seconds));
|
||||
}
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no broker: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Do not bind a real port in this assembly test.
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("FakeHerdr is healthy; no poll wait is expected");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
fleet:
|
||||
leaders:
|
||||
primary:
|
||||
tab: "lead: primary"
|
||||
profile: sonnet
|
||||
workspace: "ltms"
|
||||
profiles:
|
||||
sonnet:
|
||||
subscription: true
|
||||
argv: ["ccs", "sonnet"]
|
||||
""");
|
||||
return FleetConfig.load(file);
|
||||
}
|
||||
|
||||
@Test
|
||||
void realAssembledResolverStillResolvesAfterDeliveredListenerThrows(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ControllableResourcePorts ports = new ControllableResourcePorts();
|
||||
ports.herdr.withTab("w2", "w2:t7", "lead: primary");
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg,
|
||||
new ConfigRef(dir.resolve("fleetd.yaml"), cfg), new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||
try {
|
||||
Injector injector = runtime.injector();
|
||||
installThrowingDeliveredListener(injector);
|
||||
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = new CompletableFuture<>();
|
||||
injector.enqueue(TARGET, "brief", new TurnToken(TARGET, waiter));
|
||||
|
||||
IllegalStateException thrown = assertThrows(IllegalStateException.class,
|
||||
() -> injector.onStatus(TARGET, AgentStatus.IDLE),
|
||||
"control: delivery must reach the installed listener and it must throw after delivery");
|
||||
assertTrue(thrown.getMessage().contains("listener failure"), thrown::getMessage);
|
||||
|
||||
ports.herdr.readText("worker report after the listener failure");
|
||||
ports.advanceSeconds(3);
|
||||
runtime.completion().resolveBeforePostAction(TARGET);
|
||||
|
||||
assertTrue(waiter.isDone(),
|
||||
"FleetdAssembly must wire the Injector registrar to this runtime's real CompletionResolver: "
|
||||
+ "after a delivered listener throws, resolveBeforePostAction must still find and "
|
||||
+ "resolve the registered waiter");
|
||||
} finally {
|
||||
assertTrue(ports.shutdownHook != null, "control: assembly must capture its shutdown hook");
|
||||
ports.shutdownHook.run();
|
||||
assertTrue(ports.herdr.closed, "teardown control: the captured shutdown hook must close herdr");
|
||||
}
|
||||
}
|
||||
|
||||
private static void installThrowingDeliveredListener(Injector injector) throws Exception {
|
||||
Field field = Injector.class.getDeclaredField("turnListener");
|
||||
field.setAccessible(true);
|
||||
field.set(injector, new TurnListener() {
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target, TurnToken token) {
|
||||
throw new IllegalStateException("listener failure after delivery");
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -91,6 +91,11 @@ class FleetdBackendErrorSinkTest {
|
||||
return MAPPER.createObjectNode().set("agent", MAPPER.createObjectNode()
|
||||
.put("terminal_id", "term_primary").put("agent_status", "idle"));
|
||||
}
|
||||
if ("agent.read".equals(method)) {
|
||||
// The lead-nudge paths read the input box before pasting into it.
|
||||
return MAPPER.createObjectNode().set("read",
|
||||
MAPPER.createObjectNode().put("text", FakeHerdr.IDLE_PROMPT_CARET));
|
||||
}
|
||||
if ("agent.prompt".equals(method)) {
|
||||
prompts.add(params);
|
||||
sendLatch.countDown();
|
||||
|
||||
@@ -0,0 +1,207 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 B3 — replaces {@code FleetdBackendQuarantineWiringTest} (fleetd #466), a source-text
|
||||
* test that scraped {@code Fleetd.java} (now {@code FleetdAssembly.java}, moved there by fleetd #612
|
||||
* Unit A) for the {@code BackendQuarantine.withEscalation(...)} call, and separately asserted the
|
||||
* flat two-argument constructor's text was ABSENT. That proves the right method NAME appears in
|
||||
* source; it proves nothing about what the constructed object actually DOES.
|
||||
*
|
||||
* <p>This test instead drives the REAL {@link BackendQuarantine} the real {@link
|
||||
* FleetdAssembly#assembleAndStart} builds — reached through {@link
|
||||
* dev.ltms.fleet.mcp.FleetMcp#quarantineSource()} on the real, live {@code FleetMcp} {@code
|
||||
* FleetdRuntime} owns — and asserts the ONE behavioural difference {@code withEscalation} and the
|
||||
* flat constructor actually produce (see {@link BackendQuarantine}'s own class doc, "Mechanism"):
|
||||
* quarantining the same credential twice in a row, within one base cooldown of the first deadline,
|
||||
* must escalate the second cooldown past the first. A flat instance reports the identical cooldown
|
||||
* both times.
|
||||
*/
|
||||
class FleetdBackendQuarantineAssemblyTest {
|
||||
|
||||
/** Base cooldown used throughout — long enough that rounding never blurs the 2x escalation. */
|
||||
private static final int COOLDOWN_SECONDS = 100;
|
||||
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||
final AtomicLong clockNanos = new AtomicLong(0L);
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> replyInbox;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException(
|
||||
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
// Controllable: the SAME LongSupplier instance BackendQuarantine.withEscalation(...) is
|
||||
// built with, so advancing clockNanos after assembly moves the quarantine tracker's own
|
||||
// clock, with no real sleep needed to observe escalation.
|
||||
return clockNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return clockNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private static final class SentinelReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
@Override
|
||||
public void own(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean ack(String target, String msgId) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-test-broker/vh"
|
||||
quarantineCooldownSeconds: %d
|
||||
""".formatted(COOLDOWN_SECONDS));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the real assembled BackendQuarantine escalates a repeated exhaustion, "
|
||||
+ "which the flat two-argument constructor can never do")
|
||||
void assembledQuarantineEscalatesOnARepeatedExhaustion(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
|
||||
BackendQuarantine quarantine = runtime.mcp().quarantineSource().quarantine();
|
||||
|
||||
// First exhaustion, at clock=0: a fresh occurrence, blocked for exactly the base cooldown.
|
||||
quarantine.quarantine("cred-x");
|
||||
BackendQuarantine.Status first = quarantine.status("cred-x").orElseThrow(
|
||||
() -> new AssertionError("credential must be quarantined immediately after quarantine()"));
|
||||
assertEquals(1, first.repeatCount(), "the first call is repeat #1");
|
||||
assertEquals(COOLDOWN_SECONDS, first.remainingSeconds(),
|
||||
"a fresh quarantine blocks for exactly the base cooldown");
|
||||
|
||||
// Second exhaustion, arriving just after the first deadline — well within one base cooldown
|
||||
// of it, so this is a CONTINUATION of the same streak (repeat #2), not a fresh occurrence.
|
||||
long firstDeadlineNanos = COOLDOWN_SECONDS * 1_000_000_000L;
|
||||
ports.clockNanos.set(firstDeadlineNanos + 1);
|
||||
quarantine.quarantine("cred-x");
|
||||
BackendQuarantine.Status second = quarantine.status("cred-x").orElseThrow(
|
||||
() -> new AssertionError("credential must be quarantined immediately after the second "
|
||||
+ "quarantine() call"));
|
||||
assertEquals(2, second.repeatCount(), "the second call, arriving within one base cooldown of "
|
||||
+ "the first deadline, continues the streak as repeat #2");
|
||||
|
||||
// The one behavioural difference: withEscalation doubles the cooldown on repeat #2 (capped
|
||||
// well above this at 12x base), the flat two-argument constructor never grows past the base
|
||||
// cooldown no matter how many times quarantine() is called in a row.
|
||||
assertEquals(2 * COOLDOWN_SECONDS, second.remainingSeconds(),
|
||||
"withEscalation's default backoff doubles the cooldown on the second consecutive "
|
||||
+ "exhaustion — this is the exact call FleetdAssembly.java makes at the "
|
||||
+ "BackendQuarantine.withEscalation(...) call site");
|
||||
assertTrue(second.remainingSeconds() > first.remainingSeconds(),
|
||||
"the flat two-argument BackendQuarantine constructor would report the SAME remaining "
|
||||
+ "seconds both times — this inequality is what a mutation to the flat "
|
||||
+ "constructor at that call site must fail");
|
||||
|
||||
// Also confirm isQuarantined/remainingSeconds agree, exercising the accessors a real caller
|
||||
// (fleet_profiles / fleet_list, per BackendQuarantine's own class doc) actually reads.
|
||||
assertTrue(quarantine.isQuarantined("cred-x"));
|
||||
OptionalLong remaining = quarantine.remainingSeconds("cred-x");
|
||||
assertTrue(remaining.isPresent());
|
||||
assertEquals(2 * COOLDOWN_SECONDS, remaining.getAsLong());
|
||||
}
|
||||
}
|
||||
@@ -1,65 +0,0 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #466 follow-up: {@code Fleetd.main} builds the daemon's one {@code BackendQuarantine}
|
||||
* from {@link dev.ltms.fleet.placement.BackendQuarantine#withEscalation(java.util.function.LongSupplier,
|
||||
* long)} — the escalating factory — rather than the plain two-argument constructor, which is still a
|
||||
* flat cooldown (kept for backward compatibility, see that class's doc). {@code
|
||||
* BackendQuarantineTest} proves {@code withEscalation} itself escalates, is ceilinged, and resets;
|
||||
* it says nothing about which one {@code main} actually calls.
|
||||
*
|
||||
* <p>Measured directly: reverting {@code main} to {@code new BackendQuarantine(System::nanoTime,
|
||||
* TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()))} — the pre-#466 flat call — compiles
|
||||
* with 0 errors and leaves the entire 1608-test suite (including every {@code BackendQuarantineTest}
|
||||
* case) green, because no other test constructs its {@code BackendQuarantine} through {@code main};
|
||||
* every one of them builds its own instance directly. That silent regression is exactly the shape
|
||||
* {@link FleetdLeadSeatWiringTest} and {@link FleetdCompletionResolverWiringTest} already guard
|
||||
* against for their own constructor arguments — this is the same class of gap for fleetd #466's
|
||||
* factory choice, following their approach.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||
* BackendQuarantine} and never runs {@code main} — a green result here proves only that the exact
|
||||
* text {@code main} calls {@code BackendQuarantine.withEscalation(...)} rather than the flat
|
||||
* constructor. It does not prove that call actually executes at startup (no test here starts the
|
||||
* daemon), and it does not prove the escalation reaches a real backend or credential — only
|
||||
* {@code BackendQuarantineTest} proves the factory's own behaviour, and only a live daemon proves
|
||||
* the wiring runs.
|
||||
*/
|
||||
class FleetdBackendQuarantineWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] main's BackendQuarantine local is still built from BackendQuarantine.withEscalation(...)")
|
||||
void mainStillWiresTheEscalatingQuarantineFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"BackendQuarantine quarantine = BackendQuarantine.withEscalation(System::nanoTime,\n"
|
||||
+ " TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));"),
|
||||
"Fleetd.main's BackendQuarantine local must still be built from "
|
||||
+ "BackendQuarantine.withEscalation(System::nanoTime, "
|
||||
+ "TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds())). Reverting to the flat "
|
||||
+ "two-argument constructor (fleetd #466's measured regression) compiles with 0 errors "
|
||||
+ "and leaves the whole suite green, including every BackendQuarantineTest case that "
|
||||
+ "proves the escalation itself works — this source check is what must go red instead. "
|
||||
+ "A reverted daemon would go back to retrying a weekly subscription limit on every "
|
||||
+ "flat ~30-minute cooldown, about 336 times across the week.");
|
||||
|
||||
// Negative form of the same check: the pre-#466 flat call, if it ever reappears at this
|
||||
// declaration, must not be mistaken for the escalating one by a looser positive-only check.
|
||||
assertFalse(source.contains(
|
||||
"BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,\n"
|
||||
+ " TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));"),
|
||||
"main's BackendQuarantine local must never regress to the flat two-argument constructor");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,315 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.inject.CompletionResolver;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.WorktreeRequest;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 step 2, Unit B1 — replaces {@code FleetdCompletionResolverWiringTest} (deleted in
|
||||
* this same commit), whose four tests read {@code Fleetd.java}'s source text and asserted the
|
||||
* {@code CompletionResolver} construction call still named the right arguments. That proved the
|
||||
* call site's spelling, never that the assembled resolver actually behaves differently when an
|
||||
* argument is dropped.
|
||||
*
|
||||
* <p>These tests drive {@link FleetdAssembly#assembleAndStart} — the real boot composition,
|
||||
* fleetd #612 Unit A — and read {@link FleetdRuntime#completion()}: the exact {@link
|
||||
* CompletionResolver} instance the assembled daemon uses, never a copy built alongside it for the
|
||||
* test's benefit. Two behaviours are pinned, matching the ticket's own two measured mutations:
|
||||
*
|
||||
* <ul>
|
||||
* <li>the 8th constructor argument ({@code Fleetd.worktreeBranchLookup(sessions::roster)}) —
|
||||
* {@link #assembledResolverReportsTheMembersWorktreeAndBranchInAFallbackReport}; and</li>
|
||||
* <li>the 5th/6th arguments ({@code backendErrorPatterns}, {@code backendErrorSink}, both
|
||||
* assigned from {@code Fleetd}'s extracted factories rather than an inline lambda) —
|
||||
* {@link #assembledResolverClassifiesAndCoolsOffOnAConfiguredBackendErrorPattern}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Both tests bypass {@link dev.ltms.fleet.inject.StatusPoller} and drive {@link
|
||||
* CompletionResolver#onDelivered} / {@link CompletionResolver#resolveBeforePostAction} directly —
|
||||
* the same public, synchronous entry points {@code CompletionResolverTest} uses — with a
|
||||
* hand-built {@link CompletableFuture} waiter, so no real poller loop or herdr status poll is
|
||||
* needed. The pane scrape comes from {@link FakeHerdr#readText}; the elapsed-time floor
|
||||
* ({@code CompletionResolver.MIN_TURN_NANOS}) is controlled via a fake, advanceable {@link
|
||||
* ResourcePorts#nanoClock()} rather than a real sleep.
|
||||
*/
|
||||
class FleetdCompletionResolverAssemblyTest {
|
||||
|
||||
/** Same shape as {@code FleetdAssemblyLifecycleTest}'s fake, plus a nanoClock this test can advance. */
|
||||
private static final class ControllableResourcePorts implements ResourcePorts {
|
||||
|
||||
final FakeHerdr herdr;
|
||||
final AtomicLong nowNanos = new AtomicLong(1_000_000_000L); // arbitrary non-zero start
|
||||
Runnable shutdownHook;
|
||||
|
||||
ControllableResourcePorts(FakeHerdr herdr) {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
void advanceSeconds(long seconds) {
|
||||
nowNanos.addAndGet(TimeUnit.SECONDS.toNanos(seconds));
|
||||
}
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
// Never invoked: this test's config has no `broker:` block, so Fleetd.selectReplyInbox
|
||||
// returns the in-memory inbox before calling the opener at all.
|
||||
return (uri, prefetch) -> {
|
||||
throw new UnsupportedOperationException("replyInboxOpener must not be called — no broker: block");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
// Never invoked: no `coordinator:` block configured either.
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException("leadMailboxOpener must not be called — no coordinator: block");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Deliberately never bind a real port.
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir, String profilesYaml, String extraGuardHost,
|
||||
String worktreeRootYamlLine) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
%s
|
||||
profiles:
|
||||
%s
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- %s
|
||||
""".formatted(worktreeRootYamlLine == null ? "" : worktreeRootYamlLine, profilesYaml, extraGuardHost));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
private static void gitQuiet(Path cwd, String... args) throws Exception {
|
||||
List<String> cmd = new java.util.ArrayList<>(List.of("git"));
|
||||
cmd.addAll(List.of(args));
|
||||
Process p = new ProcessBuilder(cmd).directory(cwd.toFile()).redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git timed out: git " + String.join(" ", args));
|
||||
assertEquals(0, p.exitValue(), "git " + String.join(" ", args) + " failed:\n" + out);
|
||||
}
|
||||
|
||||
private static Path initRepo(Path dir) throws Exception {
|
||||
Files.createDirectories(dir);
|
||||
gitQuiet(dir, "init", "-q", "-b", "main");
|
||||
gitQuiet(dir, "config", "user.email", "test@example.invalid");
|
||||
gitQuiet(dir, "config", "user.name", "Test");
|
||||
Files.writeString(dir.resolve("README.md"), "seed\n");
|
||||
gitQuiet(dir, "add", "README.md");
|
||||
gitQuiet(dir, "commit", "-q", "-m", "seed");
|
||||
return dir;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #248's first measured mutation: replacing {@code CompletionResolver}'s 8th constructor
|
||||
* argument with the inert {@code _ -> null} compiles clean and leaves every existing test green
|
||||
* — it silently drops fleetd #241's fallback-report location. This drives the real assembled
|
||||
* resolver through a member echoing its own injected brief back (no {@code fleet_reply}), which
|
||||
* resolves via {@code noReportMessage(target)}, and proves the real member's {@code branch} —
|
||||
* only obtainable via {@code Fleetd.worktreeBranchLookup(sessions::roster)} reading the real,
|
||||
* worktree-provisioned {@link MemberSession} — appears in the reported text.
|
||||
*/
|
||||
@Test
|
||||
void assembledResolverReportsTheMembersWorktreeAndBranchInAFallbackReport(@TempDir Path dir) throws Exception {
|
||||
Path repo = initRepo(dir.resolve("repo"));
|
||||
FleetConfig cfg = writeConfig(dir, """
|
||||
wtprofile:
|
||||
baseUrl: http://wthost.local:8000
|
||||
model: sonnet
|
||||
""", "wthost.local", "worktreeRoot: " + dir.resolve("wts"));
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
ControllableResourcePorts ports = new ControllableResourcePorts(new FakeHerdr());
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
try {
|
||||
MemberSession session = runtime.sessions().acquire("wtprofile", repo.toString(), repo.toString(),
|
||||
null, new WorktreeRequest("fleetd-612-b1", null));
|
||||
String target = session.terminalId();
|
||||
String branch = session.branch();
|
||||
assertTrue(branch != null && branch.startsWith("worker/"),
|
||||
"sanity: a worktree-provisioned session must carry a real branch, got: " + branch);
|
||||
|
||||
CompletionResolver completion = runtime.completion();
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = new CompletableFuture<>();
|
||||
String echoedBrief = "z".repeat(450); // >= CompletionResolver.ECHO_MIN_CHARS normalised chars
|
||||
|
||||
ports.herdr.readText("idle, nothing yet");
|
||||
completion.onDelivered(target, new TurnToken(target, waiter, echoedBrief));
|
||||
|
||||
ports.herdr.readText(echoedBrief); // the pane just echoes the injected brief back — no real report
|
||||
ports.advanceSeconds(3); // clear CompletionResolver.MIN_TURN_NANOS (2s) without a real sleep
|
||||
completion.resolveBeforePostAction(target);
|
||||
|
||||
Rendezvous.Resolution resolution = waiter.getNow(null);
|
||||
assertTrue(resolution != null, "the waiter must have resolved synchronously");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, resolution.kind());
|
||||
assertTrue(resolution.text().contains(CompletionResolver.NO_REPORT_PREFIX),
|
||||
"sanity: must have gone down the noReportMessage sub-path: " + resolution.text());
|
||||
assertTrue(resolution.text().contains("branch=" + branch),
|
||||
"the assembled resolver must report the member's real branch (fleetd #241 via "
|
||||
+ "fleetd #248's worktreeBranchLookup wiring); got: " + resolution.text());
|
||||
} finally {
|
||||
if (ports.shutdownHook != null) ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #248's second measured mutation, and fleetd#201 Unit 5's own gap: replacing {@code
|
||||
* backendErrorPatterns}/{@code backendErrorSink} with {@code BackendErrorPatternLookup.legacy()}
|
||||
* / {@code BackendErrorSink.none()} compiles clean and leaves every existing behavioural test
|
||||
* green.
|
||||
*
|
||||
* <p>Classification proof: this test's profile configures {@code errorPattern: "credential
|
||||
* outage"} — text the built-in {@code (?i)\bAPI Error\s*:} fallback ({@code legacy()}'s only
|
||||
* behaviour) never matches. So a real {@code Fleetd.backendErrorPatternLookup(...)} wiring
|
||||
* classifies the send as {@code FAILED}; {@code legacy()} would fall through to the plain
|
||||
* completion path instead ({@code Kind.COMPLETION}).
|
||||
*
|
||||
* <p>Cool-off proof: two distinct targets on the same profile/credential each classified as a
|
||||
* backend error inside the 60s window must cool the credential off ({@link
|
||||
* dev.ltms.fleet.placement.BackendOutagePolicy}, fleetd#201 Unit 5) — observable two ways: (1)
|
||||
* the real {@code Fleetd.backendErrorSink(...)} marks each session {@code BACKEND_ERROR} (only
|
||||
* the real sink calls {@code sessions.onBackendError}; {@code BackendErrorSink.none()} never
|
||||
* does), and (2) a third explicit-profile spawn attempt is refused with a {@link
|
||||
* PlacementException} naming the cool-off — only reachable because the real sink's {@code
|
||||
* outagePolicy.record(...)} call actually ran.
|
||||
*/
|
||||
@Test
|
||||
void assembledResolverClassifiesAndCoolsOffOnAConfiguredBackendErrorPattern(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir, """
|
||||
coolprofile:
|
||||
baseUrl: http://coolhost.local:8000
|
||||
model: sonnet
|
||||
errorPattern: "credential outage"
|
||||
""", "coolhost.local", null);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
ControllableResourcePorts ports = new ControllableResourcePorts(new FakeHerdr());
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
try {
|
||||
MemberSession session1 = runtime.sessions().acquire("coolprofile", null, dir.toString(), null);
|
||||
MemberSession session2 = runtime.sessions().acquire("coolprofile", null, dir.toString(), null);
|
||||
String target1 = session1.terminalId();
|
||||
String target2 = session2.terminalId();
|
||||
assertTrue(!target1.equals(target2), "sanity: the two spawns must be distinct targets");
|
||||
|
||||
CompletionResolver completion = runtime.completion();
|
||||
|
||||
CompletableFuture<Rendezvous.Resolution> waiter1 = new CompletableFuture<>();
|
||||
ports.herdr.readText("idle 1");
|
||||
completion.onDelivered(target1, new TurnToken(target1, waiter1, null));
|
||||
ports.herdr.readText("credential outage: upstream 503");
|
||||
ports.advanceSeconds(3);
|
||||
completion.resolveBeforePostAction(target1);
|
||||
Rendezvous.Resolution resolution1 = waiter1.getNow(null);
|
||||
assertTrue(resolution1 != null, "target1's waiter must have resolved synchronously");
|
||||
assertEquals(Rendezvous.Kind.FAILED, resolution1.kind(),
|
||||
"a configured errorPattern the built-in fallback never matches must classify as "
|
||||
+ "a backend error, not a plain completion; got: " + resolution1);
|
||||
assertTrue(resolution1.text().contains("credential outage: upstream 503"), resolution1.text());
|
||||
|
||||
CompletableFuture<Rendezvous.Resolution> waiter2 = new CompletableFuture<>();
|
||||
ports.herdr.readText("idle 2");
|
||||
completion.onDelivered(target2, new TurnToken(target2, waiter2, null));
|
||||
ports.herdr.readText("credential outage: upstream 503 again");
|
||||
ports.advanceSeconds(3);
|
||||
completion.resolveBeforePostAction(target2);
|
||||
Rendezvous.Resolution resolution2 = waiter2.getNow(null);
|
||||
assertTrue(resolution2 != null, "target2's waiter must have resolved synchronously");
|
||||
assertEquals(Rendezvous.Kind.FAILED, resolution2.kind());
|
||||
|
||||
List<MemberSession> roster = runtime.sessions().roster();
|
||||
assertTrue(roster.stream().anyMatch(s -> target1.equals(s.terminalId())
|
||||
&& s.state() == MemberSession.State.BACKEND_ERROR),
|
||||
"the real backendErrorSink must have transitioned target1 to BACKEND_ERROR: " + roster);
|
||||
assertTrue(roster.stream().anyMatch(s -> target2.equals(s.terminalId())
|
||||
&& s.state() == MemberSession.State.BACKEND_ERROR),
|
||||
"the real backendErrorSink must have transitioned target2 to BACKEND_ERROR: " + roster);
|
||||
|
||||
PlacementException coolOff = assertThrows(PlacementException.class,
|
||||
() -> runtime.sessions().acquire("coolprofile", null, dir.toString(), null),
|
||||
"two distinct targets classified within the 60s window must cool the credential "
|
||||
+ "off (BackendOutagePolicy), refusing a third explicit-profile spawn");
|
||||
assertTrue(coolOff.getMessage().contains("cooling off"), coolOff.getMessage());
|
||||
} finally {
|
||||
if (ports.shutdownHook != null) ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,89 +0,0 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #248: this is the test that was actually missing. {@code Fleetd.main} builds its {@code
|
||||
* CompletionResolver} from an 8-argument constructor, and the ticket's own measurement proved two
|
||||
* ways to silently unwire it — both compiled with 0 errors and left every existing test green:
|
||||
*
|
||||
* <ul>
|
||||
* <li>replacing the worktree/branch argument (the 8th) with {@code _ -> null} — drops
|
||||
* fleetd#241's fallback-report location entirely;</li>
|
||||
* <li>replacing {@code backendErrorPatterns, backendErrorSink} (5th/6th) with {@code
|
||||
* BackendErrorPatternLookup.legacy(), BackendErrorSink.none()} — drops fleetd#201 Unit 5's
|
||||
* backend-error classification and cool-off entirely.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Neither mutation could be caught by any test that constructs its own {@code
|
||||
* CompletionResolver} (every test before this one did exactly that) or by a test of {@link
|
||||
* Fleetd#worktreeBranchLookup}, {@link Fleetd#backendErrorPatternLookup}, or {@link
|
||||
* Fleetd#backendErrorSink} in isolation (see {@code FleetdWorktreeBranchLookupTest}, {@code
|
||||
* FleetdBackendErrorPatternLookupTest}, {@code FleetdBackendErrorSinkTest}) — those prove the
|
||||
* factories work, never that {@code main} still calls them. This class is a plain source-text
|
||||
* assertion on {@code Fleetd.java} — crude, but honest about what it checks, and it turns red the
|
||||
* instant the wiring is dropped, mirroring the same fallback shape {@link
|
||||
* FleetdFleetAppConstructionTest} already uses for a different constructor argument.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||
* CompletionResolver} and never runs {@code main}.
|
||||
*/
|
||||
class FleetdCompletionResolverWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] CompletionResolver's construction call still names backendErrorPatterns and backendErrorSink")
|
||||
void backendErrorArgumentsAreStillNamedAtTheCallSite() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"exhaustionSink, backendErrorPatterns, backendErrorSink, System::nanoTime,"),
|
||||
"CompletionResolver's construction call must still pass backendErrorPatterns and "
|
||||
+ "backendErrorSink as its 5th/6th arguments. Replacing them with "
|
||||
+ "BackendErrorPatternLookup.legacy()/BackendErrorSink.none() (fleetd #248's measured "
|
||||
+ "mutation) compiles with 0 errors and leaves every behavioural test green — this "
|
||||
+ "source check is what must go red instead.");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] CompletionResolver's construction call still passes worktreeBranchLookup(sessions::roster)")
|
||||
void worktreeBranchLookupIsStillPassedAtTheCallSite() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains("worktreeBranchLookup(sessions::roster)"),
|
||||
"CompletionResolver's construction call must still pass worktreeBranchLookup(sessions::roster) "
|
||||
+ "as its 8th (last) argument. Replacing it with the inert `_ -> null` (fleetd #248's "
|
||||
+ "other measured mutation) compiles with 0 errors and leaves every behavioural test "
|
||||
+ "green — this source check is what must go red instead.");
|
||||
assertFalse(source.contains("System::nanoTime,\n _ -> null"),
|
||||
"the worktree/branch argument must never regress to the inert `_ -> null` literal");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] backendErrorPatterns is assigned from the extracted backendErrorPatternLookup(...) factory")
|
||||
void backendErrorPatternsComesFromTheFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"BackendErrorPatternLookup backendErrorPatterns = backendErrorPatternLookup(sessions::roster,"),
|
||||
"backendErrorPatterns must be assigned from Fleetd.backendErrorPatternLookup(...), not an "
|
||||
+ "inline lambda that a source check on the CompletionResolver call alone cannot see "
|
||||
+ "through");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] backendErrorSink is assigned from the extracted backendErrorSink(...) factory")
|
||||
void backendErrorSinkComesFromTheFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"BackendErrorSink backendErrorSink = backendErrorSink(sessions, () -> config.get().profiles(),"),
|
||||
"backendErrorSink must be assigned from Fleetd.backendErrorSink(...), not an inline lambda "
|
||||
+ "that a source check on the CompletionResolver call alone cannot see through");
|
||||
}
|
||||
}
|
||||
@@ -24,8 +24,8 @@ import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
* {@code ConfigRefTest} and {@code FleetdConfigRefCharterToolSurfaceWiringTest} case — because
|
||||
* neither of those tests constructs its {@code ConfigRef} through {@code main}; both build their own
|
||||
* instance directly, wired with the check by hand. That silent regression is exactly the shape
|
||||
* {@link FleetdBackendQuarantineWiringTest}, {@link FleetdLeadSeatWiringTest} and {@link
|
||||
* FleetdCompletionResolverWiringTest} already guard against for their own constructor arguments —
|
||||
* {@link FleetdBackendQuarantineAssemblyTest}, {@link FleetdLeadSeatAssemblyTest} and {@link
|
||||
* FleetdCompletionResolverAssemblyTest} already guard against for their own constructor arguments —
|
||||
* this class is the same class of gap for fleetd #474's {@code extraValidation} argument, following
|
||||
* their approach.
|
||||
*
|
||||
|
||||
@@ -1,31 +0,0 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-185: {@code ConnectionIdentity} must resolve a caller's pane on EITHER herdr daemon (a
|
||||
* lead's MCP connection resolves against the lead daemon; a member's against the member daemon).
|
||||
* Pinning {@code PaneLocator} to {@code memberHerdr} alone — the bug this guards against — leaves
|
||||
* every lead's own connection unresolvable ({@code callerTerminal == null}) the moment
|
||||
* {@code memberHerdrSocket} names a second daemon, which breaks {@code fleet_reply}/{@code
|
||||
* fleet_ask} and {@code fleet_whoami} for a lead. A unit test on {@link
|
||||
* dev.ltms.fleet.herdr.PaneLocator} alone (see {@code PaneLocatorTest}) proves the class CAN
|
||||
* search two clients, but not that {@code Fleetd.main} actually wires it that way — hence this
|
||||
* source-level assertion, the same technique {@code FleetdHerdrControlConstructionTest} uses.
|
||||
*/
|
||||
class FleetdConnectionIdentityConstructionTest {
|
||||
@Test
|
||||
void connectionIdentitySearchesBothDaemonsNotJustTheMemberOne() throws Exception {
|
||||
String source = Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
assertFalse(source.contains("new PaneLocator(memberHerdr)"),
|
||||
"PaneLocator must not be pinned to the member daemon alone — a lead's own "
|
||||
+ "connection resolves against the LEAD daemon and would never be found");
|
||||
assertTrue(source.contains("new PaneLocator(herdr, memberHerdr)"),
|
||||
"PaneLocator must search the lead daemon first, then the member daemon");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,218 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.inject.CompletionResolver;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 step 4, ranks 1 and 2 (publish side) — {@link FleetdAssembly} lines
|
||||
* {@code liveExhaustedPatterns}/{@code exhaustedPatterns} (CB-578 stage A, the ticket's own "worst
|
||||
* consequence in the whole sweep": a genuine usage-limit refusal handed back to a waiting caller
|
||||
* AS REAL COMPLETED WORK) and {@code Fleetd.publishExhaustionSink(...)} (CB-578 stage B: the
|
||||
* credential that hit the limit is never quarantined). None of these three lines is driven by an
|
||||
* existing test through the real assembly: {@code FleetdExhaustedPatternLookupWiringTest} and
|
||||
* {@code FleetdLiveExhaustedPatternsWiringTest} (fleetd #589) call {@code Fleetd.liveExhaustedPatterns}
|
||||
* / {@code Fleetd.exhaustedPatternLookup} directly as factories, never through {@link
|
||||
* FleetdAssembly#assembleAndStart} — they prove the FACTORY classifies correctly, never that THIS
|
||||
* call site is the one that actually got wired into the running {@link CompletionResolver}. {@link
|
||||
* FleetdBackendQuarantineAssemblyTest} drives {@code BackendQuarantine.withEscalation(...)}
|
||||
* directly, a different call site from {@code publishExhaustionSink} here.
|
||||
*
|
||||
* <p>This test drives the REAL assembled {@link CompletionResolver} ({@link
|
||||
* FleetdRuntime#completion()}) with a profile carrying a configured {@code exhaustedPattern},
|
||||
* through a pane scrape that matches it, and asserts both halves of the production consequence:
|
||||
* (1) the resolution is {@link Rendezvous.Kind#BACKEND_EXHAUSTED}, never a plain completion handed
|
||||
* back as real work, and (2) the profile's credential is actually quarantined afterward, through
|
||||
* the REAL {@link BackendQuarantine} the same assembly built ({@link
|
||||
* FleetdRuntime#mcp()}{@code .quarantineSource().quarantine()}) — never a copy.
|
||||
*
|
||||
* <p>Same {@code ControllableResourcePorts} shape as {@code FleetdCompletionResolverAssemblyTest}:
|
||||
* a fake, advanceable {@code nanoClock} so {@code CompletionResolver.MIN_TURN_NANOS} clears without
|
||||
* a real sleep, and {@link FakeHerdr#readText} to drive the pane scrape.
|
||||
*/
|
||||
class FleetdExhaustedPatternAssemblyTest {
|
||||
|
||||
private static final class ControllableResourcePorts implements ResourcePorts {
|
||||
|
||||
final FakeHerdr herdr;
|
||||
final AtomicLong nowNanos = new AtomicLong(1_000_000_000L); // arbitrary non-zero start
|
||||
Runnable shutdownHook;
|
||||
|
||||
ControllableResourcePorts(FakeHerdr herdr) {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
void advanceSeconds(long seconds) {
|
||||
nowNanos.addAndGet(TimeUnit.SECONDS.toNanos(seconds));
|
||||
}
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> {
|
||||
throw new UnsupportedOperationException("replyInboxOpener must not be called — no broker: block");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException("leadMailboxOpener must not be called — no coordinator: block");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Deliberately never bind a real port.
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir, int cooldownSeconds) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
quarantineCooldownSeconds: %d
|
||||
profiles:
|
||||
exhaustprofile:
|
||||
baseUrl: http://exhausthost.local:8000
|
||||
model: sonnet
|
||||
exhaustedPattern: "usage limit reached"
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- exhausthost.local
|
||||
""".formatted(cooldownSeconds));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #589's own description of this gap ({@code Fleetd#exhaustedPatternLookup}'s javadoc):
|
||||
* "the worst consequence in the whole #589 sweep" — a genuine usage-limit refusal stops being
|
||||
* classified as {@code BACKEND_EXHAUSTED} and is handed back to a waiting {@code fleet_send} as
|
||||
* if it were real completed work. Pins {@code FleetdAssembly}'s {@code liveExhaustedPatterns}
|
||||
* AND {@code exhaustedPatterns} lines (rank 1) together with {@code publishExhaustionSink}
|
||||
* (rank 2, the non-OpenCode half) in one flow: classify, then quarantine.
|
||||
*/
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] a scrape matching the profile's exhaustedPattern resolves "
|
||||
+ "BACKEND_EXHAUSTED (never a plain completion) and quarantines the credential")
|
||||
void assembledResolverClassifiesExhaustionAndQuarantinesTheCredential(@TempDir Path dir) throws Exception {
|
||||
int cooldownSeconds = 120;
|
||||
FleetConfig cfg = writeConfig(dir, cooldownSeconds);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
ControllableResourcePorts ports = new ControllableResourcePorts(new FakeHerdr());
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
try {
|
||||
MemberSession session = runtime.sessions().acquire("exhaustprofile", null, dir.toString(), null);
|
||||
String target = session.terminalId();
|
||||
|
||||
CompletionResolver completion = runtime.completion();
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = new CompletableFuture<>();
|
||||
|
||||
ports.herdr.readText("idle, nothing yet");
|
||||
completion.onDelivered(target, new TurnToken(target, waiter, null));
|
||||
// The matched text must START the pane line (CompletionResolver.startsWithExhaustion) —
|
||||
// no preceding sentence — for the quarantine side-effect to fire, same as production.
|
||||
ports.herdr.readText("usage limit reached: try again in a few hours");
|
||||
ports.advanceSeconds(3); // clear CompletionResolver.MIN_TURN_NANOS (2s), no real sleep
|
||||
completion.resolveBeforePostAction(target);
|
||||
|
||||
// CONTROL: the waiter must have resolved synchronously at all — if the assembled
|
||||
// CompletionResolver were never actually driven (e.g. a wiring break upstream silently
|
||||
// left the resolver unreachable), this fails loudly before the real assertions below
|
||||
// ever run, rather than passing on an untouched waiter.
|
||||
Rendezvous.Resolution resolution = waiter.getNow(null);
|
||||
assertTrue(resolution != null, "CONTROL: the waiter must have resolved synchronously — "
|
||||
+ "if this is null, the assembled resolver was never actually exercised");
|
||||
|
||||
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, resolution.kind(),
|
||||
"a scrape matching the profile's configured exhaustedPattern must classify as "
|
||||
+ "BACKEND_EXHAUSTED, not a plain completion handed back as real work — "
|
||||
+ "replacing FleetdAssembly's liveExhaustedPatterns/exhaustedPatterns "
|
||||
+ "lines with their inert forms (Map.of() / target -> null) must fail "
|
||||
+ "this assertion; got: " + resolution);
|
||||
assertTrue(resolution.text().contains("usage limit reached"), resolution.text());
|
||||
|
||||
BackendQuarantine quarantine = runtime.mcp().quarantineSource().quarantine();
|
||||
assertTrue(quarantine.isQuarantined("exhaustprofile"),
|
||||
"the real publishExhaustionSink-built sink must have quarantined the profile's "
|
||||
+ "credential (effectiveCredentialId() == the profile name here, no "
|
||||
+ "credentialId configured) — replacing FleetdAssembly's "
|
||||
+ "publishExhaustionSink call site with a hardcoded ExhaustionSink.none() "
|
||||
+ "must fail this assertion, since nothing would ever call "
|
||||
+ "quarantine.quarantine(...)");
|
||||
OptionalLong remaining = quarantine.remainingSeconds("exhaustprofile");
|
||||
assertTrue(remaining.isPresent() && remaining.getAsLong() > 0
|
||||
&& remaining.getAsLong() <= cooldownSeconds,
|
||||
"a fresh quarantine must block for at most the configured base cooldown: " + remaining);
|
||||
} finally {
|
||||
if (ports.shutdownHook != null) ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,29 +0,0 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-185: {@code FleetApp} must be constructed with BOTH herdr clients (the lead's and the
|
||||
* member's), never the raw lead-only {@code herdr}. Passing only {@code herdr} — the bug this
|
||||
* guards against — makes {@code GET /healthz} green while the member daemon is down (so every
|
||||
* spawn fails invisibly) and silently drops every member workspace from {@code GET /sessions}.
|
||||
* A behavioural test on {@code FleetApp} alone (see {@code FleetAppTwoDaemonTest}) proves the
|
||||
* class merges/gates correctly when given two clients, but not that {@code Fleetd.main} actually
|
||||
* passes it two — hence this source-level assertion, mirroring
|
||||
* {@code FleetdHerdrControlConstructionTest}.
|
||||
*/
|
||||
class FleetdFleetAppConstructionTest {
|
||||
@Test
|
||||
void fleetAppIsConstructedWithBothHerdrDaemons() throws Exception {
|
||||
String source = Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
assertFalse(source.contains("new FleetApp(herdr, workers,"),
|
||||
"FleetApp must not be constructed with the lead-only herdr client");
|
||||
assertTrue(source.contains("new FleetApp(herdr, memberHerdr, workers,"),
|
||||
"FleetApp must be constructed with both the lead and the member herdr client");
|
||||
}
|
||||
}
|
||||
@@ -2,16 +2,67 @@ package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* {@code AgentControl} caches {@code paneByTerminal}, so {@code HerdrRouter} must be its only
|
||||
* production factory — a second instance means a second cache; the same reasoning applies to
|
||||
* {@code WorkspaceControl}. {@code HerdrRouter}'s constructor is the one place both are built.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||
* HerdrRouter} and never runs {@code FleetdAssembly.assembleAndStart} — a green result proves only
|
||||
* that neither watched file's text contains {@code new AgentControl(} or {@code new
|
||||
* WorkspaceControl(}. It does not prove the instances {@code HerdrRouter} does build are the ones
|
||||
* actually wired through the rest of the daemon, and it does not cover a bypass written into a
|
||||
* production file other than the two this test reads.
|
||||
*/
|
||||
class FleetdHerdrControlConstructionTest {
|
||||
|
||||
private static String source(String relativePath) throws Exception {
|
||||
return Files.readString(Path.of(relativePath));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] Fleetd.java never constructs AgentControl or WorkspaceControl directly")
|
||||
void fleetdDelegatesStatefulControlsToTheRouter() throws Exception {
|
||||
// AgentControl caches paneByTerminal, so the router must be its only production factory.
|
||||
String source = Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
assertFalse(source.contains("new AgentControl("));
|
||||
assertFalse(source.contains("new WorkspaceControl("));
|
||||
String source = source("src/main/java/dev/ltms/fleet/Fleetd.java");
|
||||
|
||||
// A broken read (wrong working directory, wrong path, a file that came back empty) would
|
||||
// make the assertFalse checks below pass vacuously — a "clean" negative check that actually
|
||||
// checked nothing. Guard against that first, with an anchor that has nothing to do with
|
||||
// this mutation, so a bad read fails loudly here instead of silently proving nothing below.
|
||||
assertTrue(source.contains("public final class Fleetd"),
|
||||
"the read of Fleetd.java did not come back containing its own class declaration — "
|
||||
+ "the assertFalse checks below would pass vacuously on a broken read; fix the "
|
||||
+ "read before trusting this test.");
|
||||
|
||||
assertFalse(source.contains("new AgentControl("),
|
||||
"Fleetd.java must not construct AgentControl directly — HerdrRouter is its only "
|
||||
+ "production factory");
|
||||
assertFalse(source.contains("new WorkspaceControl("),
|
||||
"Fleetd.java must not construct WorkspaceControl directly — HerdrRouter is its only "
|
||||
+ "production factory");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] FleetdAssembly.java never constructs AgentControl or WorkspaceControl directly")
|
||||
void fleetdAssemblyDelegatesStatefulControlsToTheRouter() throws Exception {
|
||||
String source = source("src/main/java/dev/ltms/fleet/FleetdAssembly.java");
|
||||
|
||||
assertTrue(source.contains("final class FleetdAssembly"),
|
||||
"the read of FleetdAssembly.java did not come back containing its own class "
|
||||
+ "declaration — the assertFalse checks below would pass vacuously on a broken "
|
||||
+ "read; fix the read before trusting this test.");
|
||||
|
||||
assertFalse(source.contains("new AgentControl("),
|
||||
"FleetdAssembly.java must not construct AgentControl directly — HerdrRouter is its "
|
||||
+ "only production factory");
|
||||
assertFalse(source.contains("new WorkspaceControl("),
|
||||
"FleetdAssembly.java must not construct WorkspaceControl directly — HerdrRouter is "
|
||||
+ "its only production factory");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,213 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
|
||||
/**
|
||||
* fleetd #612 Shape A, unit r5 — {@code FleetdAssembly.java:488} wires {@link
|
||||
* FleetMcp.LeadConfigDirSource} with {@code Fleetd.leadConfigDirSource(() -> config.get().profiles(),
|
||||
* leaders)}. {@link FleetdLeadConfigDirSourceWiringTest} already pins that the FACTORY itself
|
||||
* delegates to the real {@link Fleetd#leadConfigDirLookup} — but, by its own javadoc, it "does not
|
||||
* and structurally cannot cover" whether the real call site in {@code FleetdAssembly} still calls
|
||||
* that factory at all. Measured there: swapping that one-line call for a bare {@code
|
||||
* FleetMcp.LeadConfigDirSource.none()} compiles with 0 errors and leaves the full suite green.
|
||||
*
|
||||
* <p>This is the literal fleetd #602/#606 defect, one call site away from its own fix: {@code main}
|
||||
* (now {@code FleetdAssembly}) used to build {@code LeadConfigDirSource.none()} inline, the whole
|
||||
* suite passed, and the live daemon reported {@code "state":"unknown"} for every lead's context,
|
||||
* forever, with no test noticing. The fix extracted the factory; this test is the one that proves
|
||||
* {@code FleetdAssembly}'s own call site still reaches it.
|
||||
*
|
||||
* <p>This test drives the REAL {@link FleetMcp} the real {@link FleetdAssembly#assembleAndStart}
|
||||
* builds, reached through {@link FleetdRuntime#mcp()}, and reads the {@code leadConfigDirs} field it
|
||||
* was constructed with via reflection — {@code FleetMcp} exposes no public accessor for it (unlike
|
||||
* {@code quarantineSource()}/{@code leadSeatSource()}), so there is no non-reflective route to the
|
||||
* live instance. The assertion resolves a REAL lead name against a REAL configured {@code
|
||||
* configDir:}: {@link FleetMcp.LeadConfigDirSource#none()} (the historical defect, and the
|
||||
* mis-wire this test's mutation cycles reintroduce) always returns {@code null} regardless of the
|
||||
* input, so a non-null, config-matching answer is a property {@code none()} can never produce by
|
||||
* accident.
|
||||
*/
|
||||
class FleetdLeadConfigDirSourceAssemblyTest {
|
||||
|
||||
private static final String LEAD_NAME = "opus";
|
||||
private static final String LEAD_TAB = "lead: opus";
|
||||
private static final String LEAD_PROFILE = "sonnet";
|
||||
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> replyInbox;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException(
|
||||
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private static final class SentinelReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
@Override
|
||||
public void own(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean ack(String target, String msgId) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir, String configDir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-test-broker/vh"
|
||||
fleet:
|
||||
leaders:
|
||||
%s:
|
||||
tab: "%s"
|
||||
profile: %s
|
||||
profiles:
|
||||
%s:
|
||||
subscription: true
|
||||
argv: ["ccs", "sonnet"]
|
||||
configDir: "%s"
|
||||
""".formatted(LEAD_NAME, LEAD_TAB, LEAD_PROFILE, LEAD_PROFILE, configDir));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static FleetMcp.LeadConfigDirSource leadConfigDirSourceOf(FleetMcp mcp) throws Exception {
|
||||
Field field = FleetMcp.class.getDeclaredField("leadConfigDirs");
|
||||
field.setAccessible(true);
|
||||
return (FleetMcp.LeadConfigDirSource) field.get(mcp);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the real assembled LeadConfigDirSource resolves a lead's REAL "
|
||||
+ "configured configDir, not the none() stand-in's hardcoded null")
|
||||
void assembledLeadConfigDirSourceResolvesTheRealConfiguredConfigDir(@TempDir Path dir) throws Exception {
|
||||
String configuredConfigDir = "/mnt/fake-lead-configdir";
|
||||
FleetConfig cfg = writeConfig(dir, configuredConfigDir);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
// Label FakeHerdr's own default pane's tab (term_a / w2:p7 / w2:t7, already carrying a live
|
||||
// agent) to match fleet.leaders.opus.tab exactly, so LeadLauncher.ensureLeads() sees the
|
||||
// lead as already live and does not try to auto-launch a second one.
|
||||
ports.herdr.withTab("w2", "w2:t7", LEAD_TAB);
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
try {
|
||||
FleetMcp.LeadConfigDirSource source = leadConfigDirSourceOf(runtime.mcp());
|
||||
|
||||
assertEquals(configuredConfigDir, source.configDirFor().apply(LEAD_NAME),
|
||||
"fleet.leaders." + LEAD_NAME + ".profile (" + LEAD_PROFILE + ") configures "
|
||||
+ "configDir: " + configuredConfigDir + " — the real assembled source must "
|
||||
+ "resolve it. FleetMcp.LeadConfigDirSource.none() (the inert stand-in "
|
||||
+ "this test's mutation cycles swap the call site for, and the historical "
|
||||
+ "fleetd #602/#606 defect) always reports null here, whatever the input");
|
||||
|
||||
// A lead name the config does not recognise still resolves to null, not a crash — the
|
||||
// same source, applied to an input that must stay at the inert answer even on the real,
|
||||
// non-inert instance.
|
||||
assertNull(source.configDirFor().apply("no-such-lead"));
|
||||
} finally {
|
||||
// Surefire runs the whole suite in one JVM fork (fleetd/pom.xml sets no forkCount /
|
||||
// reuseForks), so the scheduler/loops this assembly starts must be torn down here, on the
|
||||
// failure path too — hence try/finally rather than a bare statement at the end.
|
||||
runtime.close();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
|
||||
/**
|
||||
* Pins {@link Fleetd#leadConfigDirSource}'s own wiring of the window lookup into the returned
|
||||
* {@link FleetMcp.LeadConfigDirSource}, not only the detached {@link Fleetd#leadContextWindowLookup}
|
||||
* factory it delegates to. Calls the producer directly, with real {@link FleetConfig.Profile}/
|
||||
* {@link FleetConfig.Leader} fixtures, and asserts on {@code windowFor()} — the companion of
|
||||
* {@link FleetdLeadConfigDirSourceWiringTest}, which pins the same factory's {@code configDirFor()}.
|
||||
*/
|
||||
class FleetdLeadConfigDirSourceWindowWiringTest {
|
||||
|
||||
private static FleetConfig.Profile profileWithWindow(String name, Integer autoCompactWindow) {
|
||||
return new FleetConfig.Profile(name, null, "claude-sonnet-5", null, null, null,
|
||||
"tab", "fleet", "w #{n}", null, null, null, null, null, null, null,
|
||||
null, null, true, null, null, null, null, null, autoCompactWindow, null);
|
||||
}
|
||||
|
||||
private static FleetConfig.Leader leadOnProfile(String profile) {
|
||||
return new FleetConfig.Leader(profile, "lead: primary", 1, "lead:", 10, "claude", "claude-sonnet-5");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("the returned source resolves the lead's REAL configured effective window, not a hardcoded null")
|
||||
void resolvesTheRealConfiguredWindow() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("opus", profileWithWindow("opus", 250_000));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||
|
||||
FleetMcp.LeadConfigDirSource source = Fleetd.leadConfigDirSource(() -> profiles, leaders);
|
||||
|
||||
assertEquals(250_000L, source.windowFor().apply("primary"),
|
||||
"windowFor must delegate to the real leadContextWindowLookup, not a stub that always "
|
||||
+ "returns null");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a lead on a profile with no window configured still resolves to null, not a crash")
|
||||
void leadWithNoWindowConfiguredResolvesToNull() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("opus", profileWithWindow("opus", null));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||
|
||||
FleetMcp.LeadConfigDirSource source = Fleetd.leadConfigDirSource(() -> profiles, leaders);
|
||||
|
||||
assertNull(source.windowFor().apply("primary"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an unrecognised lead name resolves to null, not a thrown exception")
|
||||
void unrecognisedLeadNameResolvesToNull() {
|
||||
FleetMcp.LeadConfigDirSource source = Fleetd.leadConfigDirSource(Map::of, Map.of());
|
||||
|
||||
assertNull(source.windowFor().apply("ghost-lead"));
|
||||
}
|
||||
}
|
||||
@@ -99,7 +99,7 @@ class FleetdLeadContextLookupTest {
|
||||
@DisplayName("an unrecognised terminal resolves to UNKNOWN, not a thrown exception")
|
||||
void unrecognisedTerminalResolvesToUnknown() {
|
||||
Function<String, LeadContextGauge.Reading> lookup = Fleetd.leadContextLookup(
|
||||
new LeadContextGauge(), throwingAgentControl(), Map::of, name -> null);
|
||||
new LeadContextGauge(), throwingAgentControl(), Map::of, name -> null, name -> null);
|
||||
|
||||
LeadContextGauge.Reading reading = assertDoesNotThrow(() -> lookup.apply("ghost-terminal"));
|
||||
|
||||
@@ -112,7 +112,7 @@ class FleetdLeadContextLookupTest {
|
||||
void agentsGetThrowingDegradesToUnknown() {
|
||||
Map<String, String> liveLeadTerminals = Map.of(LEAD_TERMINAL, LEAD_NAME);
|
||||
Function<String, LeadContextGauge.Reading> lookup = Fleetd.leadContextLookup(
|
||||
new LeadContextGauge(), throwingAgentControl(), () -> liveLeadTerminals, name -> null);
|
||||
new LeadContextGauge(), throwingAgentControl(), () -> liveLeadTerminals, name -> null, name -> null);
|
||||
|
||||
LeadContextGauge.Reading reading = assertDoesNotThrow(() -> lookup.apply(LEAD_TERMINAL));
|
||||
|
||||
@@ -128,7 +128,7 @@ class FleetdLeadContextLookupTest {
|
||||
AgentControl agents = agentControlStub(SESSION_ID, "claude", "idle");
|
||||
Function<String, LeadContextGauge.Reading> lookup = Fleetd.leadContextLookup(
|
||||
new LeadContextGauge(), agents, () -> liveLeadTerminals,
|
||||
name -> LEAD_NAME.equals(name) ? tmp.toString() : null);
|
||||
name -> LEAD_NAME.equals(name) ? tmp.toString() : null, name -> null);
|
||||
|
||||
LeadContextGauge.Reading reading = lookup.apply(LEAD_TERMINAL);
|
||||
|
||||
@@ -146,7 +146,7 @@ class FleetdLeadContextLookupTest {
|
||||
AgentControl agents = agentControlStub(SESSION_ID, "opencode", "idle");
|
||||
Function<String, LeadContextGauge.Reading> lookup = Fleetd.leadContextLookup(
|
||||
new LeadContextGauge(), agents, () -> liveLeadTerminals,
|
||||
name -> LEAD_NAME.equals(name) ? tmp.toString() : null);
|
||||
name -> LEAD_NAME.equals(name) ? tmp.toString() : null, name -> null);
|
||||
|
||||
LeadContextGauge.Reading reading = lookup.apply(LEAD_TERMINAL);
|
||||
|
||||
|
||||
@@ -0,0 +1,224 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.lead.LeadContextGauge;
|
||||
import dev.ltms.fleet.msg.LeadHeartbeatLoop;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.lang.reflect.Field;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* fleetd #659: {@code FleetdAssembly.java} wires {@code Fleetd.leadContextSource}'s window-lookup
|
||||
* argument with {@code Fleetd.leadContextWindowLookup(() -> config.get().profiles(), leaders)} —
|
||||
* but nothing called the real assembled {@link LeadHeartbeatLoop} far enough to prove that
|
||||
* argument is the one the live heartbeat reads through. Measured: swapping that one call-site
|
||||
* argument for {@code _ -> null} compiles with 0 errors and leaves the full suite green.
|
||||
*
|
||||
* <p>This test drives the REAL {@link LeadHeartbeatLoop} the real {@link
|
||||
* FleetdAssembly#assembleAndStart} builds, reached through {@link FleetdRuntime#heartbeat()}, and
|
||||
* reads its private {@code contextSource} field via reflection — the loop exposes no public
|
||||
* accessor for it, the same reason {@link FleetdLeadConfigDirSourceAssemblyTest} reflects on
|
||||
* {@code FleetMcp.leadConfigDirs}. The configured profile's {@code autoCompactWindow: 100000}
|
||||
* resolves a HIGH threshold of {@code 66666} ({@link LeadContextGauge}'s {@code 2/3} fraction) —
|
||||
* far below the fixed {@code 200000} fallback a lost window argument would silently revert to.
|
||||
* {@code 90000} live tokens sits between the two: HIGH under the real window, OK under the
|
||||
* fallback — a property the fallback can never produce by accident.
|
||||
*/
|
||||
class FleetdLeadContextSourceWindowAssemblyTest {
|
||||
|
||||
private static final String LEAD_NAME = "opus";
|
||||
private static final String LEAD_TAB = "lead: opus";
|
||||
private static final String LEAD_PROFILE = "sonnet";
|
||||
/** {@code FakeHerdr}'s own default {@code agent.list} entry: terminal {@code term_a}, session {@code sess-1111}. */
|
||||
private static final String LEAD_TERMINAL = "term_a";
|
||||
private static final String LEAD_SESSION_ID = "sess-1111";
|
||||
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> replyInbox;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException(
|
||||
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private static final class SentinelReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
@Override
|
||||
public void own(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean ack(String target, String msgId) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
private static String usageLine(long tokens) {
|
||||
return "{\"type\":\"assistant\",\"message\":{\"role\":\"assistant\",\"usage\":{"
|
||||
+ "\"input_tokens\":" + tokens + ",\"cache_read_input_tokens\":0,\"cache_creation_input_tokens\":0}}}";
|
||||
}
|
||||
|
||||
/** Lays out {@code <configDir>/projects/<anySlug>/<sessionId>.jsonl} carrying one usage record. */
|
||||
private static void writeTranscript(Path configDir, String sessionId, long tokens) throws IOException {
|
||||
Path projectDir = configDir.resolve("projects").resolve("some-project-slug");
|
||||
Files.createDirectories(projectDir);
|
||||
Files.writeString(projectDir.resolve(sessionId + ".jsonl"), usageLine(tokens) + "\n", StandardCharsets.UTF_8);
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir, String configDir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-test-broker/vh"
|
||||
leadHeartbeat:
|
||||
idleAfterSeconds: 600
|
||||
backoffMs: 15000
|
||||
quietNudgeCap: 5
|
||||
fleet:
|
||||
leaders:
|
||||
%s:
|
||||
tab: "%s"
|
||||
profile: %s
|
||||
workspace: "ltms"
|
||||
profiles:
|
||||
%s:
|
||||
subscription: true
|
||||
argv: ["ccs", "sonnet"]
|
||||
configDir: "%s"
|
||||
autoCompactWindow: 100000
|
||||
""".formatted(LEAD_NAME, LEAD_TAB, LEAD_PROFILE, LEAD_PROFILE, configDir));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static LeadHeartbeatLoop.LeadContextSource contextSourceOf(LeadHeartbeatLoop heartbeat) throws Exception {
|
||||
Field field = LeadHeartbeatLoop.class.getDeclaredField("contextSource");
|
||||
field.setAccessible(true);
|
||||
return (LeadHeartbeatLoop.LeadContextSource) field.get(heartbeat);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the real assembled heartbeat loop resolves HIGH against the lead's "
|
||||
+ "REAL configured window, not the fixed 200000 fallback a lost window argument reverts to")
|
||||
void assembledHeartbeatContextSourceResolvesTheRealConfiguredWindow(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir, dir.toString());
|
||||
writeTranscript(dir, LEAD_SESSION_ID, 90_000);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
// Label FakeHerdr's own default pane's tab (term_a / w2:p7 / w2:t7, already carrying a live
|
||||
// agent on session sess-1111) to match fleet.leaders.opus.tab exactly, so LeadTabScanner
|
||||
// recognises it as the live "opus" lead without a second auto-launched pane.
|
||||
ports.herdr.withTab("w2", "w2:t7", LEAD_TAB).agentSessionId(LEAD_SESSION_ID);
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
try {
|
||||
LeadHeartbeatLoop.LeadContextSource source = contextSourceOf(runtime.heartbeat());
|
||||
|
||||
LeadContextGauge.Reading reading = source.readingFor().apply(LEAD_TERMINAL);
|
||||
|
||||
assertEquals(LeadContextGauge.State.HIGH, reading.state(),
|
||||
"profiles." + LEAD_PROFILE + ".autoCompactWindow: 100000 resolves a HIGH threshold "
|
||||
+ "of 66666 tokens — 90000 live tokens must read HIGH against it. Mutating "
|
||||
+ "FleetdAssembly's window-lookup argument to `_ -> null` falls back to the "
|
||||
+ "fixed 200000 threshold, under which 90000 reads OK instead: " + reading);
|
||||
} finally {
|
||||
// Surefire runs the whole suite in one JVM fork (fleetd/pom.xml sets no forkCount /
|
||||
// reuseForks), so the scheduler/loops this assembly starts must be torn down here, on the
|
||||
// failure path too — hence try/finally rather than a bare statement at the end.
|
||||
runtime.close();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -87,7 +87,8 @@ class FleetdLeadContextSourceWiringTest {
|
||||
Map<String, String> liveLeadTerminals = Map.of(LEAD_TERMINAL, LEAD_NAME);
|
||||
|
||||
LeadHeartbeatLoop.LeadContextSource source = Fleetd.leadContextSource(new LeadContextGauge(),
|
||||
agentControlStub(), () -> liveLeadTerminals, name -> LEAD_NAME.equals(name) ? tmp.toString() : null);
|
||||
agentControlStub(), () -> liveLeadTerminals,
|
||||
name -> LEAD_NAME.equals(name) ? tmp.toString() : null, name -> null);
|
||||
|
||||
LeadContextGauge.Reading reading = source.readingFor().apply(LEAD_TERMINAL);
|
||||
|
||||
@@ -101,7 +102,7 @@ class FleetdLeadContextSourceWiringTest {
|
||||
@DisplayName("an unrecognised lead terminal resolves to UNKNOWN, not a thrown exception")
|
||||
void unrecognisedTerminalResolvesToUnknown() {
|
||||
LeadHeartbeatLoop.LeadContextSource source = Fleetd.leadContextSource(new LeadContextGauge(),
|
||||
agentControlStub(), Map::of, name -> null);
|
||||
agentControlStub(), Map::of, name -> null, name -> null);
|
||||
|
||||
assertEquals(LeadContextGauge.State.UNKNOWN, source.readingFor().apply("ghost-terminal").state());
|
||||
}
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
|
||||
/**
|
||||
* {@link Fleetd#leadContextWindowLookup} is the factory wired into {@code
|
||||
* FleetMcp.LeadConfigDirSource} and {@code LeadHeartbeatLoop.LeadContextSource} so {@link
|
||||
* dev.ltms.fleet.lead.LeadContextGauge} scales its HIGH threshold against a lead's own profile's
|
||||
* effective auto-compact window instead of always the gauge's fixed fallback — the same {@code
|
||||
* fleet.leaders.<name>.profile} link {@link Fleetd#leadConfigDirLookup} already follows, one step
|
||||
* further to {@link FleetConfig.Profile#effectiveAutoCompactWindow()}.
|
||||
*/
|
||||
class FleetdLeadContextWindowLookupTest {
|
||||
|
||||
private static FleetConfig.Profile profileWithWindow(String name, Integer autoCompactWindow,
|
||||
Map<String, String> env) {
|
||||
return new FleetConfig.Profile(name, null, "claude-sonnet-5", null, null, null,
|
||||
"tab", "fleet", "w #{n}", null, null, null, null, null, null, env,
|
||||
null, null, true, null, null, null, null, null, autoCompactWindow, null);
|
||||
}
|
||||
|
||||
private static FleetConfig.Leader leadOnProfile(String profile) {
|
||||
return new FleetConfig.Leader(profile, "lead: primary", 1, "lead:", 10, "claude", "claude-sonnet-5");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a lead on a profile that sets autoCompactWindow resolves to that window")
|
||||
void leadOnAProfileWithAutoCompactWindowResolvesToIt() {
|
||||
Map<String, FleetConfig.Profile> profiles =
|
||||
Map.of("opus", profileWithWindow("opus", 250_000, Map.of()));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||
Function<String, Long> lookup = Fleetd.leadContextWindowLookup(() -> profiles, leaders);
|
||||
|
||||
assertEquals(250_000L, lookup.apply("primary"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("yaml and env disagree: the lookup resolves the env value, not the yaml one")
|
||||
void yamlAndEnvDisagreeLookupResolvesTheEnvValue() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("opus",
|
||||
profileWithWindow("opus", 250_000, Map.of("CLAUDE_CODE_AUTO_COMPACT_WINDOW", "150000")));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||
Function<String, Long> lookup = Fleetd.leadContextWindowLookup(() -> profiles, leaders);
|
||||
|
||||
assertEquals(150_000L, lookup.apply("primary"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a lead entry with no `profile:` resolves to null, not a thrown exception")
|
||||
void recogniseOnlyLeadWithNoProfileResolvesToNull() {
|
||||
Map<String, FleetConfig.Profile> profiles =
|
||||
Map.of("opus", profileWithWindow("opus", 250_000, Map.of()));
|
||||
FleetConfig.Leader recogniseOnly = new FleetConfig.Leader(null, "lead: primary", 1, "lead:", 10,
|
||||
"claude", "claude-sonnet-5");
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", recogniseOnly);
|
||||
Function<String, Long> lookup = Fleetd.leadContextWindowLookup(() -> profiles, leaders);
|
||||
|
||||
assertNull(lookup.apply("primary"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a lead naming a profile that is not configured resolves to null, not a thrown exception")
|
||||
void leadOnAnUnconfiguredProfileResolvesToNull() {
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("ghost-profile"));
|
||||
Function<String, Long> lookup = Fleetd.leadContextWindowLookup(Map::of, leaders);
|
||||
|
||||
assertNull(lookup.apply("primary"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a lead on a profile that resolves no window at all resolves to null")
|
||||
void leadOnAProfileWithNoWindowResolvesToNull() {
|
||||
Map<String, FleetConfig.Profile> profiles =
|
||||
Map.of("opus", profileWithWindow("opus", null, Map.of()));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||
Function<String, Long> lookup = Fleetd.leadContextWindowLookup(() -> profiles, leaders);
|
||||
|
||||
assertNull(lookup.apply("primary"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an unrecognised lead name resolves to null, not a thrown exception")
|
||||
void unrecognisedLeadNameResolvesToNull() {
|
||||
Function<String, Long> lookup = Fleetd.leadContextWindowLookup(Map::of, Map.of());
|
||||
|
||||
assertNull(lookup.apply("ghost-lead"));
|
||||
}
|
||||
}
|
||||
@@ -3,6 +3,7 @@ package dev.ltms.fleet;
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||
import dev.ltms.fleet.msg.LeadMailbox;
|
||||
import dev.ltms.fleet.testing.CapturedLog;
|
||||
import org.junit.jupiter.api.Test;
|
||||
@@ -35,7 +36,7 @@ class FleetdLeadMailboxSelectionTest {
|
||||
boolean unreachable;
|
||||
|
||||
@Override
|
||||
public LeadMailbox open(String uri, String selfCoordId, int prefetch) {
|
||||
public LeadChannelHandle open(String uri, String selfCoordId, int prefetch) {
|
||||
this.offeredUri = uri;
|
||||
this.offeredSelfId = selfCoordId;
|
||||
this.offeredPrefetch = prefetch;
|
||||
|
||||
@@ -0,0 +1,409 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.lead.LeadLauncher;
|
||||
import dev.ltms.fleet.lead.LeadRollover;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
import static org.junit.jupiter.api.Assertions.fail;
|
||||
|
||||
/**
|
||||
* fleetd #612 B3 — replaces {@code FleetdLeadRolloverWiringTest} (fleetd #480). That class was a
|
||||
* source-text test scraping {@code Fleetd.java} (now {@code FleetdAssembly.java}, moved there by
|
||||
* fleetd #612 Unit A) with three methods: {@code unrelatedAnchorStillPresent} (a scaffold anchor,
|
||||
* not an independent claim — needs no replacement of its own), {@code
|
||||
* mainStillCallsTheLeadRolloverFactory} (the call-site pin replaced by {@link
|
||||
* #assembledLeadRolloverEndsTheOldPaneThroughTheRealHerdrRouter}), and {@code
|
||||
* factoryGatesOnConfigPresence} (the absent-config claim replaced by {@link
|
||||
* #absentLeadRolloverConfigMeansNoRolloverIsBuilt} — a claim this ticket found was NOT actually
|
||||
* covered behaviourally anywhere else: {@code LeadRolloverTest}'s only related assertion is
|
||||
* vacuous, {@code assertNull(null)}, and never calls the real factory).
|
||||
*
|
||||
* <p><strong>fleetd #612 B3 correction (ticket comment 17553):</strong> the first version of this
|
||||
* test configured a single shared {@link FakeHerdr} for both the lead and member herdr sockets.
|
||||
* {@code FleetdAssembly.java:140-142} falls back to {@code memberHerdr = herdr} whenever no
|
||||
* distinct {@code memberHerdrSocket} is configured, so with one fake, {@code
|
||||
* router.leadAgents()} and {@code router.memberAgents()} wrapped the identical client — a
|
||||
* mutation swapping {@code Fleetd.leadRollover(cfg, router.leadAgents(), config, leads)} for
|
||||
* {@code ..., router.memberAgents(), ...} at {@code FleetdAssembly.java:408} was therefore
|
||||
* invisible to this test, even though the two are genuinely different daemons in production. This
|
||||
* version configures two distinct sockets and two distinct {@link FakeHerdr} instances (the same
|
||||
* pattern {@code FleetdAssemblyConnectionIdentityTest}, fleetd #612 B2, already uses to separate
|
||||
* lead from member) and asserts the roll's {@code pane.close} call lands on the LEAD fake and
|
||||
* never on the MEMBER one.
|
||||
*/
|
||||
class FleetdLeadRolloverAssemblyTest {
|
||||
|
||||
private static final Path LEAD_SOCKET = Path.of("/fake/lead-herdr.sock");
|
||||
private static final Path MEMBER_SOCKET = Path.of("/fake/member-herdr.sock");
|
||||
|
||||
/** Keys {@code connectHerdr} by socket path so the lead and member daemons can be two
|
||||
* DIFFERENT {@link FakeHerdr}s — same shape as B2's {@code FleetdAssemblyConnectionIdentityTest
|
||||
* .TwoHerdrResourcePorts}. */
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
|
||||
final Map<Path, HerdrClient> herdrsBySocket = new LinkedHashMap<>();
|
||||
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
HerdrClient client = herdrsBySocket.get(socketPath);
|
||||
if (client == null) {
|
||||
throw new IllegalStateException("no fake herdr registered for socket " + socketPath);
|
||||
}
|
||||
return client;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> replyInbox;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException(
|
||||
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private static final class SentinelReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
@Override
|
||||
public void own(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean ack(String target, String msgId) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir, Path leadCwd) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: "%s"
|
||||
memberHerdrSocket: "%s"
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-test-broker/vh"
|
||||
fleet:
|
||||
leaders:
|
||||
opus:
|
||||
tab: "lead: opus"
|
||||
cwd: "%s"
|
||||
workspace: "ltms"
|
||||
leadRollover:
|
||||
handoverPath: handover.md
|
||||
requireOperatorConfirm: false
|
||||
""".formatted(LEAD_SOCKET, MEMBER_SOCKET, leadCwd.toString()));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
/**
|
||||
* Unlike {@link #writeConfig}, this names a {@code profile:} for the lead and declares it
|
||||
* under {@code profiles:}, so {@code LeadLauncher#relaunch} can actually start a fresh agent
|
||||
* instead of refusing with "names no profile". {@code relaunchReadySeconds} is cut to 2s so
|
||||
* the recognition wait (expected to time out — see the test) does not cost real test seconds.
|
||||
*/
|
||||
private static FleetConfig writeConfigWithRelaunchableLead(Path dir, Path leadCwd) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: "%s"
|
||||
memberHerdrSocket: "%s"
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-test-broker/vh"
|
||||
fleet:
|
||||
leaders:
|
||||
opus:
|
||||
tab: "lead: opus"
|
||||
cwd: "%s"
|
||||
profile: opus
|
||||
workspace: "ltms"
|
||||
profiles:
|
||||
opus:
|
||||
subscription: true
|
||||
argv: ["ccs", "opus"]
|
||||
leadRollover:
|
||||
handoverPath: handover.md
|
||||
requireOperatorConfirm: false
|
||||
relaunchReadySeconds: 2
|
||||
""".formatted(LEAD_SOCKET, MEMBER_SOCKET, leadCwd.toString()));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the real assembled LeadRollover runs the open/confirm/continuation "
|
||||
+ "sequence through the real herdr router — ending the old pane, then giving up once it "
|
||||
+ "never reports gone")
|
||||
void assembledLeadRolloverEndsTheOldPaneThroughTheRealHerdrRouter(@TempDir Path dir) throws Exception {
|
||||
Path leadCwd = dir.resolve("lead-workspace");
|
||||
Files.createDirectories(leadCwd);
|
||||
FleetConfig cfg = writeConfig(dir, leadCwd);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
// Two DISTINCT fakes — one per configured socket — so leadAgents()/memberAgents() wrap
|
||||
// genuinely different clients, exactly like production when memberHerdrSocket is set.
|
||||
FakeHerdr lead = new FakeHerdr();
|
||||
lead.withTab("w2", "w2:t7", "lead: opus");
|
||||
FakeHerdr member = new FakeHerdr();
|
||||
ports.herdrsBySocket.put(LEAD_SOCKET, lead);
|
||||
ports.herdrsBySocket.put(MEMBER_SOCKET, member);
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
|
||||
LeadRollover rollover = runtime.mcp().leadRollover();
|
||||
assertNotNull(rollover, "leadRollover: is present in this test's config, so "
|
||||
+ "FleetdAssembly.assembleAndStart must have built a real LeadRollover through the "
|
||||
+ "Fleetd.leadRollover(...) call site — a mutation to `LeadRollover leadRollover = "
|
||||
+ "null;` at that call site can never pass this");
|
||||
|
||||
// assembleAndStart's own boot work (the orphan-worker reap) makes a real call on the
|
||||
// member daemon before the roll ever starts. Clear it here so the assertion below measures
|
||||
// only what the roll itself does, not what daemon startup does.
|
||||
member.calls.clear();
|
||||
|
||||
LeadRollover.PendingRollover pending = rollover.open("term_a", "fleetd #612 B3 test");
|
||||
String expectedHandoverPath = leadCwd.resolve("handover.md").normalize().toString();
|
||||
assertEquals(expectedHandoverPath, pending.handoverPath());
|
||||
|
||||
// Ensure the handover file's mtime lands strictly AFTER open()'s requestedAtMillis —
|
||||
// LeadRollover.checkHandover refuses on mtime <= requestedAt (HANDOVER_STALE).
|
||||
Thread.sleep(50);
|
||||
Files.writeString(Path.of(pending.handoverPath()), "handover content for fleetd #612 B3");
|
||||
|
||||
LeadRollover.RollDecision decision = rollover.confirm("term_a", pending.token(), true);
|
||||
assertTrue(decision.accepted(), "confirm() must approve: requireOperatorConfirm is false, "
|
||||
+ "the caller terminal matches open()'s, and the handover file exists, is non-empty "
|
||||
+ "and fresh — got: " + decision);
|
||||
|
||||
// The production LeadRollover constructor always runs the post-confirm continuation on a
|
||||
// real virtual thread (see Fleetd.leadRollover, which never passes the package-private test
|
||||
// constructor), so this polls the real FleetMcp.leadRollover() instance's status(token)
|
||||
// until the real continuation finishes.
|
||||
LeadRollover.RollStatus status = pollUntilTerminal(rollover, pending.token());
|
||||
|
||||
// FakeHerdr's pane.get is a fixed canned response that never reports a pane as gone, so the
|
||||
// real router's death poll runs out its whole budget and the roll stops here — proving the
|
||||
// real teardown call landed on the real LEAD pane without ever reaching a relaunch or a send.
|
||||
assertEquals(LeadRollover.RollState.OLD_PANE_NEVER_DIED, status.state(),
|
||||
"the old pane never reports gone against this fake, so the roll must stop with "
|
||||
+ "OLD_PANE_NEVER_DIED rather than ever relaunching or sending anything — "
|
||||
+ "detail: " + status.detail());
|
||||
|
||||
// Prove the real herdr router actually closed the real LEAD pane — this is the one thing a
|
||||
// source-text pin on the call site could never show.
|
||||
boolean closedOldPane = lead.calls.stream()
|
||||
.anyMatch(c -> c.method().equals("pane.close")
|
||||
&& c.params() instanceof Map<?, ?> m && "w2:p7".equals(m.get("pane_id")));
|
||||
assertTrue(closedOldPane, "endOldSession must close the real old pane (w2:p7) through the "
|
||||
+ "real LEAD herdr client, got calls: " + lead.calls);
|
||||
|
||||
// No agent.prompt is ever sent on this path: the roll stops at the pane-death wait, strictly
|
||||
// before the relaunch and the final send step.
|
||||
List<FakeHerdr.Call> prompts = lead.calls.stream()
|
||||
.filter(c -> c.method().equals("agent.prompt"))
|
||||
.toList();
|
||||
assertTrue(prompts.isEmpty(), "a roll that stops at OLD_PANE_NEVER_DIED must never reach the "
|
||||
+ "send step, got agent.prompt call(s) on the LEAD daemon: " + prompts);
|
||||
|
||||
// fleetd #612 B3 correction: prove the roll never touches the MEMBER daemon. A mutation
|
||||
// swapping router.leadAgents() for router.memberAgents() at the real call site would move
|
||||
// the pane.close call above onto `member` instead, which this assertion catches — the thing
|
||||
// the single-fake version of this test could never see, because both wrapped the same client.
|
||||
assertTrue(member.calls.isEmpty(), "the roll must be wired to the LEAD daemon only — got "
|
||||
+ member.calls.size() + " call(s) recorded on the MEMBER daemon since the roll began: "
|
||||
+ member.calls);
|
||||
}
|
||||
|
||||
/**
|
||||
* Exercises the relaunch site {@link #assembledLeadRolloverEndsTheOldPaneThroughTheRealHerdrRouter}
|
||||
* never reaches: with the old pane confirmed gone, the roll relaunches a fresh lead, and
|
||||
* {@code bootstrapText} must reach it even though recognition times out (FakeHerdr's
|
||||
* {@code tab.list} is a fixed canned response that never reflects the relaunch's own
|
||||
* {@code tab.rename}, so the fresh terminal is never recognised as a live lead). Same
|
||||
* dual-socket shape as the sibling test: two distinct {@link FakeHerdr} instances, so a
|
||||
* {@code bootstrapText} send wired to the wrong daemon is visible.
|
||||
*/
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] bootstrapText reaches the fresh LEAD terminal even when recognition "
|
||||
+ "times out, and the MEMBER daemon never sees it")
|
||||
void bootstrapTextReachesTheFreshLeadTerminalEvenWhenRecognitionTimesOut(@TempDir Path dir) throws Exception {
|
||||
Path leadCwd = dir.resolve("lead-workspace");
|
||||
Files.createDirectories(leadCwd);
|
||||
FleetConfig cfg = writeConfigWithRelaunchableLead(dir, leadCwd);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
FakeHerdr lead = new FakeHerdr();
|
||||
lead.withTab("w2", "w2:t7", "lead: opus");
|
||||
// Lets the old pane (w2:p7) report gone once pane.close actually reaches it, so the roll
|
||||
// proceeds to relaunch instead of stopping at OLD_PANE_NEVER_DIED.
|
||||
lead.paneGoneAfterClose("w2:p7");
|
||||
FakeHerdr member = new FakeHerdr();
|
||||
ports.herdrsBySocket.put(LEAD_SOCKET, lead);
|
||||
ports.herdrsBySocket.put(MEMBER_SOCKET, member);
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
|
||||
LeadRollover rollover = runtime.mcp().leadRollover();
|
||||
assertNotNull(rollover, "leadRollover: is present in this test's config, so a real "
|
||||
+ "LeadRollover must have been built");
|
||||
|
||||
LeadRollover.PendingRollover pending = rollover.open("term_a", "bootstrapText relaunch test");
|
||||
Thread.sleep(50);
|
||||
Files.writeString(Path.of(pending.handoverPath()), "handover content for bootstrapText test");
|
||||
|
||||
LeadRollover.RollDecision decision = rollover.confirm("term_a", pending.token(), true);
|
||||
assertTrue(decision.accepted(), "confirm() must approve — got: " + decision);
|
||||
|
||||
LeadRollover.RollStatus status = pollUntilTerminal(rollover, pending.token());
|
||||
assertEquals(LeadRollover.RollState.RELAUNCH_NOT_RECOGNISED, status.state(),
|
||||
"the fresh terminal is never recognised against this fake's static tab.list, so the "
|
||||
+ "roll must reach RELAUNCH_NOT_RECOGNISED — not an earlier failure state and "
|
||||
+ "not ROLLED — detail: " + status.detail());
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
List<FakeHerdr.Call> leadPrompts = lead.calls.stream()
|
||||
.filter(c -> c.method().equals("agent.prompt"))
|
||||
.toList();
|
||||
assertEquals(1, leadPrompts.size(), "exactly one bootstrapText send is expected, on the LEAD "
|
||||
+ "daemon, once recognition gives up — got: " + leadPrompts);
|
||||
Object text = ((Map<String, Object>) leadPrompts.get(0).params()).get("text");
|
||||
assertTrue(text instanceof String && ((String) text).contains("handover.md"),
|
||||
"the send must be bootstrapText naming the resolved handover path, got: " + text);
|
||||
|
||||
// Scoped to agent.prompt specifically, not every MEMBER call: the orphan-worker reap also
|
||||
// talks to the MEMBER daemon once, unconditionally, at daemon boot — unrelated to this roll.
|
||||
List<FakeHerdr.Call> memberPrompts = member.calls.stream()
|
||||
.filter(c -> c.method().equals("agent.prompt"))
|
||||
.toList();
|
||||
assertTrue(memberPrompts.isEmpty(), "bootstrapText must never be sent to the MEMBER daemon, "
|
||||
+ "got: " + memberPrompts);
|
||||
}
|
||||
|
||||
private static LeadRollover.RollStatus pollUntilTerminal(LeadRollover rollover, String token)
|
||||
throws InterruptedException {
|
||||
long deadline = System.nanoTime() + java.util.concurrent.TimeUnit.SECONDS.toNanos(15);
|
||||
while (System.nanoTime() < deadline) {
|
||||
LeadRollover.RollStatus status = rollover.status(token);
|
||||
if (status.state() != LeadRollover.RollState.PENDING
|
||||
&& status.state() != LeadRollover.RollState.IN_PROGRESS) {
|
||||
return status;
|
||||
}
|
||||
Thread.sleep(50);
|
||||
}
|
||||
fail("the real continuation did not reach a terminal state within 10s — last status: "
|
||||
+ rollover.status(token));
|
||||
throw new AssertionError("unreachable");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] Fleetd.leadRollover(...) returns null when leadRollover: is absent "
|
||||
+ "from config — the opt-in gate FleetdLeadRolloverWiringTest's "
|
||||
+ "factoryGatesOnConfigPresence pinned by source text alone")
|
||||
void absentLeadRolloverConfigMeansNoRolloverIsBuilt(@TempDir Path dir) throws Exception {
|
||||
Path yaml = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
""");
|
||||
ConfigRef config = new ConfigRef(yaml, FleetConfig.load(yaml));
|
||||
AgentControl agents = new AgentControl(new FakeHerdr());
|
||||
WorkspaceControl spaces = new WorkspaceControl(new FakeHerdr());
|
||||
LeadLauncher launcher = new LeadLauncher(agents, spaces, config.get());
|
||||
|
||||
LeadRollover rollover = Fleetd.leadRollover(config.get(), agents, spaces, launcher, config, Map::of);
|
||||
|
||||
assertNull(rollover, "leadRollover: is absent from this config, so the factory's opt-in "
|
||||
+ "gate (`if (cfg.leadRollover() == null) return null;`) must fire and no "
|
||||
+ "LeadRollover must be constructed at all");
|
||||
}
|
||||
}
|
||||
@@ -1,89 +0,0 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #480 Unit A, hard requirement 6: pin {@code Fleetd.main}'s construction of {@link
|
||||
* dev.ltms.fleet.lead.LeadRollover} with a source-text assertion, mirroring {@code
|
||||
* FleetdCompletionResolverWiringTest}'s pattern — five log-only reporters in {@code Fleetd.main}
|
||||
* already survived mutation batteries this exact way (fleetd #415's extraction antidote note).
|
||||
*
|
||||
* <p>What this class still covers, and what it never claimed to. {@code LeadRolloverTest}
|
||||
* constructs its own {@code LeadRollover} directly (as every prior test of an extracted factory
|
||||
* does) with a hand-built lookup, so a mutation that deletes the {@code leadRollover(...)} call
|
||||
* from {@code main} — or replaces one of its arguments with something that still compiles, e.g.
|
||||
* {@code router.leadAgents()} swapped for {@code null}, or the whole assignment swapped for a bare
|
||||
* {@code null} literal — leaves every behavioural test green. This is a plain string read, guarded
|
||||
* by an unrelated anchor assertion so a broken or empty file read cannot pass as a real change.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||
* LeadRollover} and never runs {@code main}. It pins the {@code leadRollover(...)} CALL SITE's
|
||||
* argument list — that {@code main} still passes {@code leads} at all — never what the factory
|
||||
* DOES with that argument once inside its own body.
|
||||
*
|
||||
* <p><b>Correction (fleetd #480 relative-handover-path follow-up): that gap used to be real, and
|
||||
* now is not — but not here.</b> This class's javadoc previously claimed "no behavioural test can
|
||||
* catch this wiring dropping out" for the whole factory, including the lambda {@code
|
||||
* leadRollover(...)} builds internally (terminal → lead name → {@code Leader.cwd()}). That claim
|
||||
* was proven true at the time — mutating that lambda's body to {@code String leadName = null;}
|
||||
* (always "no lead found", which silently reintroduces the daemon-cwd bug this ticket fixes) left
|
||||
* the full suite green, {@code Tests run: 1669, Failures: 0}. It is no longer true: {@code
|
||||
* FleetdLeadRolloverWorkspaceLookupTest} now calls {@code Fleetd.leadRollover(...)} directly with a
|
||||
* real {@link dev.ltms.fleet.config.ConfigRef} built from a temp {@code fleetd.yaml}, and fails
|
||||
* against that exact one-line mutation. So: THIS class still covers only the call site's argument
|
||||
* list; {@code FleetdLeadRolloverWorkspaceLookupTest} is what now covers the lambda's body. Neither
|
||||
* one subsumes the other — keep both.
|
||||
*/
|
||||
class FleetdLeadRolloverWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] unrelated anchor: Fleetd.java still declares the Fleetd class")
|
||||
void unrelatedAnchorStillPresent() throws Exception {
|
||||
// Guards the two assertions below: without this, a bad read (empty string, wrong file,
|
||||
// truncated file) could vacuously fail to contain the leadRollover(...) call too, and a
|
||||
// test that only asserts "contains X" would report a false pass for the wrong reason if X
|
||||
// happened to match. Asserting an unrelated, structurally distant string first proves the
|
||||
// read actually pulled real file content.
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains("public final class Fleetd"),
|
||||
"sanity anchor failed — the file read did not return real Fleetd.java source; the "
|
||||
+ "leadRollover(...) wiring assertions below cannot be trusted until this passes");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] main still constructs LeadRollover via the leadRollover(...) factory, exactly as heartbeat is constructed")
|
||||
void mainStillCallsTheLeadRolloverFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"LeadRollover leadRollover = leadRollover(cfg, router.leadAgents(), config, leads);"),
|
||||
"Fleetd.main must still assign `LeadRollover leadRollover = leadRollover(cfg, "
|
||||
+ "router.leadAgents(), config, leads);`. Dropping this call, or swapping one of "
|
||||
+ "its arguments for something that still compiles (e.g. null in place of "
|
||||
+ "router.leadAgents()), leaves every behavioural test green — this source check is "
|
||||
+ "what must go red instead. fleetd #480 correction 2 deliberately dropped "
|
||||
+ "primaryRegistry from this call — see LeadRollover's class javadoc for why a "
|
||||
+ "single-slot lookup was wrong here. The fleetd #480 relative-handover-path "
|
||||
+ "follow-up added `leads` (terminal → lead name) so the factory can resolve a "
|
||||
+ "relative handoverPath against the calling lead's own workspace.");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] the leadRollover(...) factory itself gates construction on cfg.leadRollover() != null")
|
||||
void factoryGatesOnConfigPresence() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains("if (cfg.leadRollover() == null) {"),
|
||||
"Fleetd.leadRollover(...) must refuse to construct a LeadRollover when the "
|
||||
+ "leadRollover: block is absent — an upgraded daemon must never silently acquire "
|
||||
+ "the ability to clear the lead's own pane. See LeadHeartbeatLoop's construction "
|
||||
+ "gate (cfg.leadHeartbeat() != null) for the pattern this mirrors.");
|
||||
}
|
||||
}
|
||||
@@ -4,6 +4,8 @@ import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.lead.LeadLauncher;
|
||||
import dev.ltms.fleet.lead.LeadRollover;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
@@ -54,6 +56,11 @@ class FleetdLeadRolloverWorkspaceLookupTest {
|
||||
return new AgentControl(new FakeHerdr());
|
||||
}
|
||||
|
||||
/** None of this class's tests reach the deferred continuation, so a plain fake is enough. */
|
||||
private static LeadLauncher fakeLauncher(FleetConfig cfg) {
|
||||
return new LeadLauncher(fakeAgents(), new WorkspaceControl(new FakeHerdr()), cfg);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] Fleetd.leadRollover(...) resolves a relative handoverPath against "
|
||||
+ "the CALLING lead's configured cwd, not the daemon's own working directory")
|
||||
@@ -74,7 +81,8 @@ class FleetdLeadRolloverWorkspaceLookupTest {
|
||||
""".formatted(leadCwd.toString()));
|
||||
ConfigRef config = new ConfigRef(yaml, FleetConfig.load(yaml));
|
||||
|
||||
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(), config,
|
||||
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(),
|
||||
new WorkspaceControl(new FakeHerdr()), fakeLauncher(config.get()), config,
|
||||
() -> Map.of("term_opus", "opus"));
|
||||
assertNotNull(rollover, "leadRollover: is present in the loaded config, so the factory "
|
||||
+ "must construct an object");
|
||||
@@ -105,7 +113,8 @@ class FleetdLeadRolloverWorkspaceLookupTest {
|
||||
|
||||
// No lead has been discovered yet — exactly the real shape of a lead the live tab scan
|
||||
// has not yet scanned, or one with no fleet.leaders entry at all.
|
||||
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(), config, Map::of);
|
||||
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(),
|
||||
new WorkspaceControl(new FakeHerdr()), fakeLauncher(config.get()), config, Map::of);
|
||||
assertNotNull(rollover);
|
||||
|
||||
LeadRollover.PendingRollover pending = rollover.open("term_unknown", "test");
|
||||
@@ -144,7 +153,8 @@ class FleetdLeadRolloverWorkspaceLookupTest {
|
||||
// below — exactly the natural mistake to make, since leads are discovered by a live tab
|
||||
// scan that runs AFTER this factory is constructed at startup.
|
||||
Map<String, String> liveLeadTerminals = new HashMap<>();
|
||||
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(), config,
|
||||
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(),
|
||||
new WorkspaceControl(new FakeHerdr()), fakeLauncher(config.get()), config,
|
||||
() -> liveLeadTerminals);
|
||||
assertNotNull(rollover);
|
||||
|
||||
|
||||
@@ -0,0 +1,181 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* fleetd #612 B3 — replaces {@code FleetdLeadSeatWiringTest} (fleetd #176), a source-text test that
|
||||
* scraped {@code Fleetd.java} (now {@code FleetdAssembly.java}, moved there by fleetd #612 Unit A)
|
||||
* for the exact {@code new FleetMcp.LeadSeatSource(Fleetd.leadSeatLookup(...))} constructor-call
|
||||
* text. That proves the right symbols appear in source; it proves nothing about what the daemon's
|
||||
* live {@code fleet_list} actually reports.
|
||||
*
|
||||
* <p>This test instead drives the REAL {@link FleetMcp.LeadSeatSource} the real {@link
|
||||
* FleetdAssembly#assembleAndStart} builds — including the REAL {@code LeadTabScanner} it wires
|
||||
* {@code Fleetd.leadSeatLookup} through — reached via {@link FleetMcp#leadSeatSource()} on the
|
||||
* live {@code FleetMcp} {@code FleetdRuntime} owns. It seeds one FakeHerdr tab labelled to match a
|
||||
* configured {@code fleet.leaders.opus.tab}, with a live agent already in it (FakeHerdr's own
|
||||
* default {@code agent.list}/{@code pane.list} entries for {@code term_a}/{@code w2:p7}/{@code
|
||||
* w2:t7} — no FakeHerdr change needed), and asserts the assembled seat source reports exactly the
|
||||
* seat {@link FleetMcp.LeadSeatSource#none()} (the inert stand-in) could never produce: 1, not 0.
|
||||
*/
|
||||
class FleetdLeadSeatAssemblyTest {
|
||||
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> replyInbox;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException(
|
||||
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private static final class SentinelReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
@Override
|
||||
public void own(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean ack(String target, String msgId) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-test-broker/vh"
|
||||
fleet:
|
||||
leaders:
|
||||
opus:
|
||||
tab: "lead: opus"
|
||||
profile: sonnet
|
||||
workspace: "ltms"
|
||||
profiles:
|
||||
sonnet:
|
||||
subscription: true
|
||||
argv: ["ccs", "sonnet"]
|
||||
""");
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the real assembled LeadSeatSource, backed by the real LeadTabScanner, "
|
||||
+ "reports a live lead's seat against its own subscription profile")
|
||||
void assembledLeadSeatSourceReportsALiveLeadsSeat(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
// Label FakeHerdr's own default pane's tab (term_a / w2:p7 / w2:t7, already carrying a live
|
||||
// agent) to match fleet.leaders.opus.tab exactly — no FakeHerdr change needed at all.
|
||||
ports.herdr.withTab("w2", "w2:t7", "lead: opus");
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
|
||||
FleetMcp.LeadSeatSource seatSource = runtime.mcp().leadSeatSource();
|
||||
assertEquals(1, seatSource.seatsFor().apply("sonnet"),
|
||||
"the real LeadTabScanner recognises the labelled tab as a live 'opus' lead on "
|
||||
+ "profile 'sonnet' (same credential, matched by Fleetd.leadSeatLookup), so "
|
||||
+ "subscription profile 'sonnet' must be charged one seat — "
|
||||
+ "FleetMcp.LeadSeatSource.none() (the inert stand-in this test's mutation "
|
||||
+ "swaps the call site for) always reports 0, whatever the input");
|
||||
|
||||
// A profile no lead is running on gets no seat charged — the same seat source, applied to
|
||||
// an input that must stay at the inert answer even on the real, non-inert instance.
|
||||
assertEquals(0, seatSource.seatsFor().apply("no-such-profile"));
|
||||
}
|
||||
}
|
||||
@@ -1,43 +0,0 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #176: {@code Fleetd.main} builds its {@code FleetMcp} from a 14-argument constructor whose
|
||||
* last argument is a {@code FleetMcp.LeadSeatSource} wrapping {@link Fleetd#leadSeatLookup}. That
|
||||
* argument is exactly the kind of wiring fleetd #248 warned about: dropping it (or swapping it for
|
||||
* the inert {@code FleetMcp.LeadSeatSource.none()}) compiles with 0 errors and leaves every test
|
||||
* that builds its own {@code FleetMcp}/{@code CapacitySource} directly — every test that predates
|
||||
* this ticket — green, because none of them go through {@code main} at all.
|
||||
*
|
||||
* <p>{@link FleetdLeadSeatLookupTest} proves the factory's own matching logic; this class is the
|
||||
* plain source-text assertion that proves {@code main} still passes its result in, mirroring
|
||||
* {@code FleetdCompletionResolverWiringTest}'s approach for the same class of gap.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a
|
||||
* {@code FleetMcp} and never runs {@code main}.
|
||||
*/
|
||||
class FleetdLeadSeatWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] FleetMcp's construction call still passes a LeadSeatSource built from leadSeatLookup(...)")
|
||||
void fleetMcpConstructionStillWiresLeadSeatLookup() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains("new FleetMcp.LeadSeatSource(leadSeatLookup(() -> config.get().profiles(), "
|
||||
+ "leaders, leads))"),
|
||||
"FleetMcp's construction call must still pass a LeadSeatSource built from "
|
||||
+ "Fleetd.leadSeatLookup(...). Dropping it or swapping in "
|
||||
+ "FleetMcp.LeadSeatSource.none() (fleetd #176's would-be silent regression, the same "
|
||||
+ "shape as fleetd #248's measured mutations) compiles with 0 errors and leaves every "
|
||||
+ "existing behavioural test green — this source check is what must go red instead.");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,243 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||
import dev.ltms.fleet.msg.LeadMessage;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import io.modelcontextprotocol.client.McpClient;
|
||||
import io.modelcontextprotocol.client.McpSyncClient;
|
||||
import io.modelcontextprotocol.client.transport.HttpClientStreamableHttpTransport;
|
||||
import io.modelcontextprotocol.spec.McpClientTransport;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.Timeout;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.net.http.HttpRequest;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 Shape A ranks 9 and 11: drives the real {@code fleet_list} MCP route through a
|
||||
* token-authenticated primary caller. The assertions read the response from the exact {@link
|
||||
* dev.ltms.fleet.mcp.FleetMcp} instance assembled by {@link FleetdAssembly}, rather than a source
|
||||
* scrape or a separately built reporting source.
|
||||
*
|
||||
* <p>The coordinator fixture contains a peer on purpose. Both a missing coordinator and an
|
||||
* incorrectly wired peers argument can render as an empty list, so the non-empty peer assertion
|
||||
* distinguishes the real call-site value from that inert result. Its fake mailbox never contacts a
|
||||
* broker.
|
||||
*/
|
||||
class FleetdListReportingSourcesAssemblyTest {
|
||||
|
||||
private static final String TOKEN = "fleetd-r9-r11-test-token";
|
||||
private static final String TOKEN_ENV = "FLEETD_R9_TEST_TOKEN";
|
||||
private static final String PROFILE = "capacity-profile";
|
||||
private static final String PEER = "peer-fleet";
|
||||
|
||||
private static final class FakeLeadChannel implements LeadChannelHandle {
|
||||
@Override
|
||||
public void publish(String toCoordId, LeadMessage message) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<LeadMessage> peek() {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String msgId) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public String selfCoordId() {
|
||||
return "test-lead";
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean heldDurable() {
|
||||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public MailboxState inspect(String coordId) {
|
||||
return MailboxState.unknown(coordId);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
private static final class TestResourcePorts implements ResourcePorts {
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
private Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of(TOKEN_ENV, TOKEN);
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> new ReplyInbox() {
|
||||
@Override public void own(String target) { }
|
||||
@Override public void release(String target) { }
|
||||
@Override public void publish(String target, String msgId, String content) { }
|
||||
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||
@Override public boolean ack(String target, String msgId) { return false; }
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> new FakeLeadChannel();
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// The test binds runtime.app() itself, below.
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("healthy FakeHerdr must not poll");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private FleetdRuntime runtime;
|
||||
private TestResourcePorts ports;
|
||||
private Javalin boundApp;
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
if (boundApp != null) {
|
||||
boundApp.stop();
|
||||
}
|
||||
if (ports != null && ports.shutdownHook != null) {
|
||||
ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
auth:
|
||||
mode: token
|
||||
tokenEnv: %s
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
health:
|
||||
enabled: true
|
||||
notifications:
|
||||
mode: webhook
|
||||
profiles:
|
||||
%s:
|
||||
baseUrl: http://capacity.test:8000
|
||||
model: test-model
|
||||
maxLoad: 7
|
||||
coordinator:
|
||||
uri: amqp://fake-coordinator/vh
|
||||
selfId: test-lead
|
||||
peers:
|
||||
- %s
|
||||
""".formatted(TOKEN_ENV, PROFILE, PEER));
|
||||
return FleetConfig.load(file);
|
||||
}
|
||||
|
||||
private int assembleAndBind(Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
ports = new TestResourcePorts();
|
||||
runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config,
|
||||
new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||
boundApp = runtime.app().start("127.0.0.1", 0);
|
||||
return boundApp.port();
|
||||
}
|
||||
|
||||
private static String fleetList(int port) {
|
||||
HttpRequest.Builder requestTemplate = HttpRequest.newBuilder()
|
||||
.header("Authorization", "Bearer " + TOKEN);
|
||||
McpClientTransport transport = HttpClientStreamableHttpTransport.builder("http://127.0.0.1:" + port)
|
||||
.endpoint("/mcp")
|
||||
.requestBuilder(requestTemplate)
|
||||
.build();
|
||||
try (McpSyncClient client = McpClient.sync(transport).build()) {
|
||||
client.initialize();
|
||||
McpSchema.CallToolResult response = client.callTool(McpSchema.CallToolRequest.builder("fleet_list")
|
||||
.arguments(Map.of()).build());
|
||||
return ((McpSchema.TextContent) response.content().getFirst()).text();
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@Timeout(value = 15, unit = TimeUnit.SECONDS, threadMode = Timeout.ThreadMode.SEPARATE_THREAD)
|
||||
void fleetListReportsTheAssembledCapacitySource(@TempDir Path dir) throws Exception {
|
||||
String response = fleetList(assembleAndBind(dir));
|
||||
|
||||
assertTrue(response.contains("\"capacity\""), response);
|
||||
assertTrue(response.contains("\"" + PROFILE + "\""), response);
|
||||
assertTrue(response.contains("\"maxLoad\":7"), response);
|
||||
}
|
||||
|
||||
@Test
|
||||
@Timeout(value = 15, unit = TimeUnit.SECONDS, threadMode = Timeout.ThreadMode.SEPARATE_THREAD)
|
||||
void fleetListReportsTheAssembledHealthCoverageSource(@TempDir Path dir) throws Exception {
|
||||
String response = fleetList(assembleAndBind(dir));
|
||||
|
||||
assertTrue(response.contains("\"healthCoverage\":\"full\""), response);
|
||||
}
|
||||
|
||||
@Test
|
||||
@Timeout(value = 15, unit = TimeUnit.SECONDS, threadMode = Timeout.ThreadMode.SEPARATE_THREAD)
|
||||
void fleetListReportsTheAssembledCoordinatorPeers(@TempDir Path dir) throws Exception {
|
||||
String response = fleetList(assembleAndBind(dir));
|
||||
|
||||
assertTrue(response.contains("\"coordinator\""), response);
|
||||
assertTrue(response.contains("\"coordId\":\"" + PEER + "\""), response);
|
||||
}
|
||||
}
|
||||
+197
@@ -0,0 +1,197 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.member.HerdrPeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 step 4, rank 2 (OpenCode half) — {@link FleetdAssembly}'s {@code
|
||||
* forwardingExhaustionSink} line ({@code Fleetd.forwardingExhaustionSink(exhaustionSinkRef)}),
|
||||
* handed to {@link dev.ltms.fleet.member.OpenCodeLauncher} so its fleetd #175 model-mismatch check
|
||||
* can quarantine a credential before {@code sessions} exists to build the real sink (the
|
||||
* construction-order cycle documented at that call site). The ticket calls this independent from
|
||||
* {@code publishExhaustionSink} (pinned by {@link FleetdExhaustedPatternAssemblyTest}): a credential
|
||||
* that hits a usage limit through THIS path is never quarantined if {@code forwardingExhaustionSink}
|
||||
* is swapped for a hardcoded {@link ExhaustionSink#none()} at that call site — the OpenCode
|
||||
* launcher's own quarantine check keeps compiling and keeps "running", but it permanently talks to
|
||||
* a sink that does nothing, independent of whatever {@code publishExhaustionSink} does later.
|
||||
*
|
||||
* <p>{@code FleetdExhaustionSinkForwardingWiringTest} (fleetd #589) already proves {@code
|
||||
* Fleetd.forwardingExhaustionSink(ref)} forwards to whatever {@code ref} holds — as a bare factory
|
||||
* call, never through {@link FleetdAssembly#assembleAndStart}. It proves nothing about whether
|
||||
* THIS call site is the one FleetdAssembly actually wires into the real {@code OpenCodeLauncher}
|
||||
* it builds, which is exactly the #602/#606-shaped gap this ticket exists to close.
|
||||
*
|
||||
* <p>No accessor on {@link FleetdRuntime} reaches the adapter instances (by design — see that
|
||||
* class's own javadoc: only the final collaborators it owns directly are exposed), so this test
|
||||
* reaches the REAL, assembled {@code OpenCodeLauncher}'s {@code exhaustionSink} field the same way
|
||||
* {@code SessionManager}/{@code CompositePeerLauncher} wire it internally: a short, targeted
|
||||
* reflective walk ({@code SessionManager.launcher} → {@code CompositePeerLauncher.byProfile} →
|
||||
* {@code OpenCodeLauncher.exhaustionSink}) onto the exact object the assembly built — never a copy,
|
||||
* and never a read of the source text. Reflection is used the same way elsewhere in this suite
|
||||
* (e.g. {@code StatusPollerWatchdogTest}) to reach a private collaborator a production constructor
|
||||
* intentionally does not expose a public accessor for.
|
||||
*/
|
||||
class FleetdOpenCodeExhaustionForwardingAssemblyTest {
|
||||
|
||||
private static final class ControllableResourcePorts implements ResourcePorts {
|
||||
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
final AtomicLong nowNanos = new AtomicLong(1_000_000_000L);
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> {
|
||||
throw new UnsupportedOperationException("replyInboxOpener must not be called — no broker: block");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException("leadMailboxOpener must not be called — no coordinator: block");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Deliberately never bind a real port.
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir, int cooldownSeconds) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
quarantineCooldownSeconds: %d
|
||||
profiles:
|
||||
gemini:
|
||||
kind: opencode
|
||||
model: google/gemini-2.5-pro
|
||||
""".formatted(cooldownSeconds));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
/** Reach a declared field by name on {@code target}'s runtime class, bypassing the access check. */
|
||||
private static Object readField(Object target, Class<?> declaringClass, String fieldName) throws Exception {
|
||||
Field field = declaringClass.getDeclaredField(fieldName);
|
||||
field.setAccessible(true);
|
||||
return field.get(target);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the real assembled OpenCodeLauncher's exhaustionSink field forwards "
|
||||
+ "an onExhausted call into the real daemon's BackendQuarantine")
|
||||
void assembledOpenCodeLauncherExhaustionSinkQuarantinesTheCredential(@TempDir Path dir) throws Exception {
|
||||
int cooldownSeconds = 90;
|
||||
FleetConfig cfg = writeConfig(dir, cooldownSeconds);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
ControllableResourcePorts ports = new ControllableResourcePorts();
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
try {
|
||||
PeerLauncher launcherField = (PeerLauncher) readField(runtime.sessions(),
|
||||
runtime.sessions().getClass(), "launcher");
|
||||
// CONTROL: the composite launcher must actually be the real production type with a
|
||||
// "gemini" -> OpenCodeLauncher entry — if this fails, nothing below exercised the real
|
||||
// assembly at all, rather than silently passing on an empty/wrong object.
|
||||
assertTrue(launcherField instanceof CompositePeerLauncher,
|
||||
"CONTROL: SessionManager.launcher must be the real CompositePeerLauncher the "
|
||||
+ "assembly built, got: " + launcherField);
|
||||
@SuppressWarnings("unchecked")
|
||||
Map<String, HerdrPeerLauncher> byProfile = (Map<String, HerdrPeerLauncher>)
|
||||
readField(launcherField, CompositePeerLauncher.class, "byProfile");
|
||||
HerdrPeerLauncher adapter = byProfile.get("gemini");
|
||||
assertTrue(adapter != null && adapter.getClass().getSimpleName().equals("OpenCodeLauncher"),
|
||||
"CONTROL: the 'gemini' profile must resolve to a real OpenCodeLauncher adapter, "
|
||||
+ "got: " + adapter);
|
||||
|
||||
ExhaustionSink sink = (ExhaustionSink) readField(adapter, adapter.getClass(), "exhaustionSink");
|
||||
assertTrue(sink != null, "CONTROL: OpenCodeLauncher.exhaustionSink must never be null");
|
||||
|
||||
// The exact call OpenCodeLauncher.SessionAwareHandle#checkModelMatch makes on a real
|
||||
// model mismatch (fleetd #175): target, reason, and its own already-known profile name.
|
||||
sink.onExhausted("term_gemini_1", "opencode model mismatch (test)", "gemini");
|
||||
|
||||
BackendQuarantine quarantine = runtime.mcp().quarantineSource().quarantine();
|
||||
assertTrue(quarantine.isQuarantined("gemini"),
|
||||
"the real forwardingExhaustionSink-wired field must have delegated into the "
|
||||
+ "published production sink, which quarantines the profile's credential "
|
||||
+ "('gemini' here — no credentialId configured) — replacing "
|
||||
+ "FleetdAssembly's forwardingExhaustionSink call site with a hardcoded "
|
||||
+ "ExhaustionSink.none() must fail this assertion, since the field read "
|
||||
+ "above would then BE the inert no-op and nothing would ever reach "
|
||||
+ "quarantine.quarantine(...)");
|
||||
assertEquals(cooldownSeconds, quarantine.remainingSeconds("gemini").orElseThrow(
|
||||
() -> new AssertionError("credential must report a remaining cooldown")));
|
||||
} finally {
|
||||
if (ports.shutdownHook != null) ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
}
|
||||
+368
@@ -0,0 +1,368 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.inject.CompletionResolver;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import io.javalin.Javalin;
|
||||
import io.modelcontextprotocol.client.McpClient;
|
||||
import io.modelcontextprotocol.client.McpSyncClient;
|
||||
import io.modelcontextprotocol.client.transport.HttpClientStreamableHttpTransport;
|
||||
import io.modelcontextprotocol.spec.McpClientTransport;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.Timeout;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.net.URI;
|
||||
import java.net.http.HttpClient;
|
||||
import java.net.http.HttpRequest;
|
||||
import java.net.http.HttpResponse;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 Shape A, unit r4 — {@link FleetdAssembly}'s {@code quarantineSource} (lines 471-472)
|
||||
* and {@code outageSource} (lines 473-476 at {@code main} = {@code 141ae3b}), EACH of which feeds
|
||||
* two separate consumers: {@code FleetMcp} ({@code fleet_profiles}) and {@code FleetApp}
|
||||
* ({@code GET /profiles}), one call site ({@code :484}/{@code :486}) into the MCP constructor and
|
||||
* the SAME shared local again ({@code :529}) into the REST constructor.
|
||||
*
|
||||
* <p><strong>The property pinned here</strong> (from the ticket): when the real assembly has built
|
||||
* a real {@link dev.ltms.fleet.placement.BackendQuarantine} holding a quarantined credential, and a
|
||||
* real {@link dev.ltms.fleet.placement.BackendOutagePolicy} holding a cooling-off credential, BOTH
|
||||
* operator windows must report that state — the real assembled {@code FleetMcp} (reached through
|
||||
* {@link FleetdRuntime#mcp()}) and the real assembled REST surface (reached through {@link
|
||||
* FleetdRuntime#app()}). A mutation that starves one consumer while leaving the other wired must
|
||||
* make only that consumer's assertion go red.
|
||||
*
|
||||
* <p><strong>No source-text assertion anywhere in this file.</strong> Both windows are read off the
|
||||
* REAL running objects: {@code fleet_profiles} is called through a real MCP client over a real
|
||||
* HTTP connection to the servlet {@link FleetdAssembly} actually mounted, and {@code GET /profiles}
|
||||
* is called through a real {@link java.net.http.HttpClient} against the real bound {@link
|
||||
* FleetdRuntime#app()}. Neither is a copy built alongside the assembly for this test's benefit.
|
||||
*
|
||||
* <p><strong>How this gets past CB-185's own pid-resolution dead end.</strong> {@code
|
||||
* FleetdAssemblyFleetAppTest}'s class javadoc explains that {@code GET /sessions} cannot be driven
|
||||
* over real HTTP here because {@code LsofPeerPidLookup} excludes its own pid and an in-process test
|
||||
* client/server share one JVM pid — every such request resolves {@code ANONYMOUS} and is refused
|
||||
* before the handler runs. {@code GET /profiles} and {@code fleet_profiles} sit behind the exact
|
||||
* same {@code Authz.Action.READ} gate. This test sidesteps the dead end instead of hitting it:
|
||||
* {@code auth.mode: token} (see {@link dev.ltms.fleet.auth.CallerResolver#resolve}) resolves a
|
||||
* caller to {@code PRIMARY} from a valid {@code Authorization: Bearer} header ALONE, with no pid
|
||||
* resolution involved at all — the same technique {@code FleetMcpContextExtractorTest} already uses
|
||||
* to drive a real {@code fleet_whoami} call through the real transport.
|
||||
*
|
||||
* <p><strong>How the quarantined/cooling-off state is set up.</strong> Both {@code BackendQuarantine}
|
||||
* and {@code BackendOutagePolicy} are private to the collaborators the assembly wires them into, and
|
||||
* (unlike {@code quarantineSource()}) neither {@code FleetMcp} nor {@code FleetApp} exposes a public
|
||||
* accessor for the live {@code BackendOutagePolicy} instance. Rather than add one (acceptance
|
||||
* criterion 1: no production change), this test drives the REAL production classification path —
|
||||
* exactly the recipe {@code FleetdExhaustedPatternAssemblyTest} (quarantine) and {@code
|
||||
* FleetdCompletionResolverAssemblyTest} (cool-off) already proved works end to end against this same
|
||||
* {@link FleetdAssembly#assembleAndStart}: acquire a real {@link MemberSession}, feed the real {@link
|
||||
* CompletionResolver} a pane scrape matching the profile's configured {@code exhaustedPattern} /
|
||||
* {@code errorPattern}, and let the real {@code exhaustionSink}/{@code backendErrorSink} write into
|
||||
* the real, shared tracker. Each setup step asserts its own {@link Rendezvous.Kind} as a CONTROL —
|
||||
* if the resolver were never actually exercised, the setup itself fails loudly before either window
|
||||
* is ever read.
|
||||
*/
|
||||
class FleetdQuarantineOutageDualWindowAssemblyTest {
|
||||
|
||||
private static final String TOKEN = "s3cret-r4-token";
|
||||
private static final String TOKEN_ENV = "FLEETD_R4_TEST_TOKEN";
|
||||
|
||||
private static final class ControllableResourcePorts implements ResourcePorts {
|
||||
|
||||
final FakeHerdr herdr;
|
||||
final AtomicLong nowNanos = new AtomicLong(1_000_000_000L); // arbitrary non-zero start
|
||||
Runnable shutdownHook;
|
||||
|
||||
ControllableResourcePorts(FakeHerdr herdr) {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
void advanceSeconds(long seconds) {
|
||||
nowNanos.addAndGet(TimeUnit.SECONDS.toNanos(seconds));
|
||||
}
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of(TOKEN_ENV, TOKEN);
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
// Never invoked: this test's config has no `broker:` block.
|
||||
return (uri, prefetch) -> {
|
||||
throw new UnsupportedOperationException("replyInboxOpener must not be called — no broker: block");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
// Never invoked: this test's config has no `coordinator:` block.
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException(
|
||||
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Deliberately never bind here — this test binds runtime.app() itself, for real, below.
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private FleetdRuntime runtime;
|
||||
private ControllableResourcePorts ports;
|
||||
private Javalin boundApp;
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
if (boundApp != null) {
|
||||
boundApp.stop();
|
||||
}
|
||||
if (ports != null && ports.shutdownHook != null) {
|
||||
ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir, int cooldownSeconds) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
auth:
|
||||
mode: token
|
||||
tokenEnv: %s
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
quarantineCooldownSeconds: %d
|
||||
profiles:
|
||||
exhaustprofile:
|
||||
baseUrl: http://exhausthost.local:8000
|
||||
model: sonnet
|
||||
exhaustedPattern: "usage limit reached"
|
||||
coolprofile:
|
||||
baseUrl: http://coolhost.local:8000
|
||||
model: sonnet
|
||||
errorPattern: "credential outage"
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- exhausthost.local
|
||||
- coolhost.local
|
||||
""".formatted(TOKEN_ENV, cooldownSeconds));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
/** Assembles the real graph, then binds the real {@code Javalin app} to an ephemeral port. */
|
||||
private int assembleAndBind(Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir, 120);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
ports = new ControllableResourcePorts(new FakeHerdr());
|
||||
runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
boundApp = runtime.app().start("127.0.0.1", 0);
|
||||
return boundApp.port();
|
||||
}
|
||||
|
||||
/**
|
||||
* Drives the real assembled {@link CompletionResolver} through a scrape matching {@code
|
||||
* exhaustprofile}'s configured {@code exhaustedPattern}, exactly {@code
|
||||
* FleetdExhaustedPatternAssemblyTest}'s own recipe, so the real {@code exhaustionSink} quarantines
|
||||
* the credential ({@code effectiveCredentialId() == "exhaustprofile"}, no explicit credentialId
|
||||
* configured).
|
||||
*/
|
||||
private void quarantineExhaustProfile(Path dir) {
|
||||
MemberSession session = runtime.sessions().acquire("exhaustprofile", null, dir.toString(), null);
|
||||
String target = session.terminalId();
|
||||
|
||||
CompletionResolver completion = runtime.completion();
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = new CompletableFuture<>();
|
||||
|
||||
ports.herdr.readText("idle, nothing yet");
|
||||
completion.onDelivered(target, new TurnToken(target, waiter, null));
|
||||
ports.herdr.readText("usage limit reached: try again in a few hours");
|
||||
ports.advanceSeconds(3); // clear CompletionResolver.MIN_TURN_NANOS (2s), no real sleep
|
||||
completion.resolveBeforePostAction(target);
|
||||
|
||||
Rendezvous.Resolution resolution = waiter.getNow(null);
|
||||
assertTrue(resolution != null && resolution.kind() == Rendezvous.Kind.BACKEND_EXHAUSTED,
|
||||
"CONTROL: setup must classify as BACKEND_EXHAUSTED before either window is read — "
|
||||
+ "if this fails, the assembled resolver was never actually exercised: " + resolution);
|
||||
}
|
||||
|
||||
/**
|
||||
* Drives the real assembled {@link CompletionResolver} with TWO distinct targets on {@code
|
||||
* coolprofile}, each matching its configured {@code errorPattern}, exactly {@code
|
||||
* FleetdCompletionResolverAssemblyTest}'s own recipe, so the real {@code backendErrorSink} cools
|
||||
* the credential off ({@code effectiveCredentialId() == "coolprofile"}).
|
||||
*/
|
||||
private void coolOffCoolProfile(Path dir) {
|
||||
MemberSession s1 = runtime.sessions().acquire("coolprofile", null, dir.toString(), null);
|
||||
MemberSession s2 = runtime.sessions().acquire("coolprofile", null, dir.toString(), null);
|
||||
CompletionResolver completion = runtime.completion();
|
||||
|
||||
String t1 = s1.terminalId();
|
||||
CompletableFuture<Rendezvous.Resolution> w1 = new CompletableFuture<>();
|
||||
ports.herdr.readText("idle 1");
|
||||
completion.onDelivered(t1, new TurnToken(t1, w1, null));
|
||||
ports.herdr.readText("credential outage: upstream 503");
|
||||
ports.advanceSeconds(3);
|
||||
completion.resolveBeforePostAction(t1);
|
||||
Rendezvous.Resolution r1 = w1.getNow(null);
|
||||
assertTrue(r1 != null && r1.kind() == Rendezvous.Kind.FAILED,
|
||||
"CONTROL: target1's setup must classify FAILED (backend error): " + r1);
|
||||
|
||||
String t2 = s2.terminalId();
|
||||
CompletableFuture<Rendezvous.Resolution> w2 = new CompletableFuture<>();
|
||||
ports.herdr.readText("idle 2");
|
||||
completion.onDelivered(t2, new TurnToken(t2, w2, null));
|
||||
ports.herdr.readText("credential outage: upstream 503 again");
|
||||
ports.advanceSeconds(3);
|
||||
completion.resolveBeforePostAction(t2);
|
||||
Rendezvous.Resolution r2 = w2.getNow(null);
|
||||
assertTrue(r2 != null && r2.kind() == Rendezvous.Kind.FAILED,
|
||||
"CONTROL: target2's setup must classify FAILED (backend error) — two distinct "
|
||||
+ "targets are required to cross BackendOutagePolicy.THRESHOLD: " + r2);
|
||||
}
|
||||
|
||||
/** Calls the real {@code fleet_profiles} tool over a real MCP client, token-authenticated as PRIMARY. */
|
||||
private static McpSchema.CallToolResult callProfilesViaMcp(int port) {
|
||||
HttpRequest.Builder requestTemplate = HttpRequest.newBuilder().header("Authorization", "Bearer " + TOKEN);
|
||||
McpClientTransport transport = HttpClientStreamableHttpTransport.builder("http://127.0.0.1:" + port)
|
||||
.endpoint("/mcp")
|
||||
.requestBuilder(requestTemplate)
|
||||
.build();
|
||||
try (McpSyncClient client = McpClient.sync(transport).build()) {
|
||||
client.initialize();
|
||||
return client.callTool(McpSchema.CallToolRequest.builder("fleet_profiles").arguments(Map.of()).build());
|
||||
}
|
||||
}
|
||||
|
||||
private static String textOf(McpSchema.CallToolResult r) {
|
||||
return ((McpSchema.TextContent) r.content().getFirst()).text();
|
||||
}
|
||||
|
||||
/** Calls the real {@code GET /profiles} route over a real {@link HttpClient}, same token. */
|
||||
private static String getProfilesViaRest(int port) throws Exception {
|
||||
HttpClient http = HttpClient.newHttpClient();
|
||||
HttpRequest req = HttpRequest.newBuilder(URI.create("http://127.0.0.1:" + port + "/profiles"))
|
||||
.header("Authorization", "Bearer " + TOKEN)
|
||||
.GET().build();
|
||||
HttpResponse<String> res = http.send(req, HttpResponse.BodyHandlers.ofString());
|
||||
assertEquals(200, res.statusCode(), "GET /profiles must succeed with the real token: " + res.body());
|
||||
return res.body();
|
||||
}
|
||||
|
||||
// --- quarantineSource (FleetdAssembly.java :471-472) --------------------------------------
|
||||
|
||||
@Test
|
||||
@Timeout(value = 15, unit = TimeUnit.SECONDS, threadMode = Timeout.ThreadMode.SEPARATE_THREAD)
|
||||
void quarantinedCredentialIsReportedByTheRealAssembledFleetMcp(@TempDir Path dir) throws Exception {
|
||||
int port = assembleAndBind(dir);
|
||||
quarantineExhaustProfile(dir);
|
||||
|
||||
String out = textOf(callProfilesViaMcp(port));
|
||||
|
||||
assertTrue(out.contains("\"quarantined\""), "fleet_profiles must report a quarantined "
|
||||
+ "section once the real BackendQuarantine holds a quarantined credential: " + out);
|
||||
assertTrue(out.contains("\"exhaustprofile\""), out);
|
||||
assertTrue(out.contains("\"quarantinedForSeconds\""), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
@Timeout(value = 15, unit = TimeUnit.SECONDS, threadMode = Timeout.ThreadMode.SEPARATE_THREAD)
|
||||
void quarantinedCredentialIsReportedByTheRealAssembledFleetApp(@TempDir Path dir) throws Exception {
|
||||
int port = assembleAndBind(dir);
|
||||
quarantineExhaustProfile(dir);
|
||||
|
||||
String out = getProfilesViaRest(port);
|
||||
|
||||
assertTrue(out.contains("\"quarantined\""), "GET /profiles must report a quarantined "
|
||||
+ "section once the real BackendQuarantine holds a quarantined credential: " + out);
|
||||
assertTrue(out.contains("\"exhaustprofile\""), out);
|
||||
assertTrue(out.contains("\"quarantinedForSeconds\""), out);
|
||||
}
|
||||
|
||||
// --- outageSource (FleetdAssembly.java :473-476) -------------------------------------------
|
||||
|
||||
@Test
|
||||
@Timeout(value = 15, unit = TimeUnit.SECONDS, threadMode = Timeout.ThreadMode.SEPARATE_THREAD)
|
||||
void coolingOffCredentialIsReportedByTheRealAssembledFleetMcp(@TempDir Path dir) throws Exception {
|
||||
int port = assembleAndBind(dir);
|
||||
coolOffCoolProfile(dir);
|
||||
|
||||
String out = textOf(callProfilesViaMcp(port));
|
||||
|
||||
assertTrue(out.contains("\"coolingOff\""), "fleet_profiles must report a coolingOff "
|
||||
+ "section once the real BackendOutagePolicy holds a cooling-off credential: " + out);
|
||||
assertTrue(out.contains("\"coolprofile\""), out);
|
||||
assertTrue(out.contains("\"coolingOffForSeconds\""), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
@Timeout(value = 15, unit = TimeUnit.SECONDS, threadMode = Timeout.ThreadMode.SEPARATE_THREAD)
|
||||
void coolingOffCredentialIsReportedByTheRealAssembledFleetApp(@TempDir Path dir) throws Exception {
|
||||
int port = assembleAndBind(dir);
|
||||
coolOffCoolProfile(dir);
|
||||
|
||||
String out = getProfilesViaRest(port);
|
||||
|
||||
assertTrue(out.contains("\"coolingOff\""), "GET /profiles must report a coolingOff "
|
||||
+ "section once the real BackendOutagePolicy holds a cooling-off credential: " + out);
|
||||
assertTrue(out.contains("\"coolprofile\""), out);
|
||||
assertTrue(out.contains("\"coolingOffForSeconds\""), out);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,202 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.guard.GuardException;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #625: pins {@link dev.ltms.fleet.guard.SubscriptionGuard#assertPrimaryClean}'s call site
|
||||
* in {@link Fleetd#main(String[])} — the ONE place it runs at startup, and the check behind the
|
||||
* bridge charter's invariant 1 (never let the primary carry {@code ANTHROPIC_BASE_URL}). Nothing
|
||||
* pinned it before this ticket: deleting {@code guard.assertPrimaryClean(...)} from {@code main}
|
||||
* left the full suite green, because the call site read the real process environment ({@code
|
||||
* System.getenv()}), which a test cannot taint from inside the JVM.
|
||||
*
|
||||
* <p>{@link Fleetd#main(String[], ResourcePorts)} (added by this ticket) is the literal production
|
||||
* sequence — not a copy of it — driven here with a {@link ResourcePorts} whose {@link
|
||||
* ResourcePorts#environment()} is a plain {@code Map} a test controls. The guard itself was
|
||||
* already pinned by {@code SubscriptionGuardTest}, directly, with a {@code Map} — that proves the
|
||||
* method's behaviour, not that {@code main} still calls it at the right point. This class pins the
|
||||
* call site and, separately, the ORDER: the guard must still run before {@code cfg.validateAll()}
|
||||
* and before {@code FleetdAssembly.assembleAndStart} touches a socket, a broker, or HTTP — not just
|
||||
* be present somewhere in {@code main}.
|
||||
*
|
||||
* <p>Presence alone is not enough (a fix that pins only presence trades an invisible deletion for
|
||||
* an invisible reordering), so each test below is built so that EITHER deleting the guard call OR
|
||||
* moving it later makes the <em>same</em> test fail — with a different exception type than the one
|
||||
* asserted, not a vacuous pass. See each test's own javadoc for how.
|
||||
*/
|
||||
class FleetdSubscriptionGuardOrderingTest {
|
||||
|
||||
private static final Map<String, String> TAINTED_ENV =
|
||||
Map.of("ANTHROPIC_BASE_URL", "http://tainted.example");
|
||||
private static final Map<String, String> CLEAN_ENV = Map.of("PATH", "/usr/bin");
|
||||
|
||||
/**
|
||||
* A {@link ResourcePorts} whose {@link #environment()} is fixed to whatever the test hands it,
|
||||
* and whose every other method refuses to be called at all. That refusal is the ordering pin:
|
||||
* if {@code main} ever reaches {@link FleetdAssembly#assembleAndStart} before the guard has had
|
||||
* a chance to throw, the very first thing the assembly does with {@code ports} is {@link
|
||||
* #connectHerdr} — so a test that expects {@link GuardException} and instead observes {@link
|
||||
* UnsupportedOperationException} has just caught the guard running too late (or not at all).
|
||||
*/
|
||||
private static final class FixedEnvPorts implements ResourcePorts {
|
||||
private final Map<String, String> env;
|
||||
|
||||
FixedEnvPorts(Map<String, String> env) {
|
||||
this.env = env;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return env;
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
throw new UnsupportedOperationException("connectHerdr must not be called before the guard runs");
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
throw new UnsupportedOperationException("replyInboxOpener must not be called before the guard runs");
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
throw new UnsupportedOperationException("leadMailboxOpener must not be called before the guard runs");
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
throw new UnsupportedOperationException("nanoClock must not be called before the guard runs");
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
throw new UnsupportedOperationException("wallClockNanos must not be called before the guard runs");
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
throw new UnsupportedOperationException("newScheduler must not be called before the guard runs");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
throw new UnsupportedOperationException("addShutdownHook must not be called before the guard runs");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
throw new UnsupportedOperationException("startHttp must not be called before the guard runs");
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called before the guard runs");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Otherwise-invalid: {@code bind.host: 0.0.0.0} with no {@code auth.mode: token} fails {@code
|
||||
* cfg.validateAll()} (CB-501's auth-exposure check — the same fixture {@code
|
||||
* FleetdStartupValidationTest#mainRefusesANonLoopbackBindWithoutTokenMode} uses), with an
|
||||
* {@link IllegalStateException}. That is deliberate: it is what {@code main} would throw INSTEAD
|
||||
* of {@link GuardException} if the guard call were deleted, or moved to run after {@code
|
||||
* validateAll()} — a different, distinguishable exception type.
|
||||
*/
|
||||
private static Path invalidConfig(Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 0.0.0.0
|
||||
port: 8765
|
||||
""");
|
||||
return f;
|
||||
}
|
||||
|
||||
/** Passes {@code cfg.validateAll()} cleanly — nothing here trips any of its checks. */
|
||||
private static Path validConfig(Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
""");
|
||||
return f;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pins the order against {@code cfg.validateAll()}. The environment is tainted and the config
|
||||
* is otherwise invalid (see {@link #invalidConfig}). If the guard runs first (the required
|
||||
* order), {@code main} throws {@link GuardException} before {@code validateAll()} is ever
|
||||
* reached. If the guard were deleted, or reordered to run after {@code validateAll()}, {@code
|
||||
* validateAll()} throws {@link IllegalStateException} instead and this assertion fails on the
|
||||
* wrong exception type.
|
||||
*/
|
||||
@Test
|
||||
void mainRefusesATaintedEnvironmentBeforeValidatingTheConfig(@TempDir Path dir) throws Exception {
|
||||
Path config = invalidConfig(dir);
|
||||
FixedEnvPorts ports = new FixedEnvPorts(TAINTED_ENV);
|
||||
|
||||
GuardException ex = assertThrows(GuardException.class,
|
||||
() -> Fleetd.main(new String[]{config.toString()}, ports));
|
||||
assertTrue(ex.getMessage().contains("tainted"),
|
||||
"expected the primary-taint message, got: " + ex.getMessage());
|
||||
}
|
||||
|
||||
/**
|
||||
* Pins the order against {@code FleetdAssembly.assembleAndStart}. The environment is tainted
|
||||
* and the config is otherwise VALID (see {@link #validConfig}), so {@code cfg.validateAll()}
|
||||
* passes silently and the next thing that could possibly run is the assembly's first socket
|
||||
* call. If the guard runs first (the required order), {@code main} throws {@link
|
||||
* GuardException} before assembly starts. If the guard were deleted, or reordered to run after
|
||||
* assembly begins touching {@code ports}, {@link FixedEnvPorts#connectHerdr} throws {@link
|
||||
* UnsupportedOperationException} instead and this assertion fails on the wrong exception type.
|
||||
*/
|
||||
@Test
|
||||
void mainRefusesATaintedEnvironmentBeforeAssemblyTouchesAnyPort(@TempDir Path dir) throws Exception {
|
||||
Path config = validConfig(dir);
|
||||
FixedEnvPorts ports = new FixedEnvPorts(TAINTED_ENV);
|
||||
|
||||
GuardException ex = assertThrows(GuardException.class,
|
||||
() -> Fleetd.main(new String[]{config.toString()}, ports));
|
||||
assertTrue(ex.getMessage().contains("tainted"),
|
||||
"expected the primary-taint message, got: " + ex.getMessage());
|
||||
}
|
||||
|
||||
/**
|
||||
* The CONTROL for the two tests above. Same otherwise-valid config, same {@link FixedEnvPorts}
|
||||
* whose every method but {@code environment()} refuses to be called — but a CLEAN environment.
|
||||
* Without this, a guard that always threw {@link GuardException} regardless of input (the
|
||||
* opposite bug — e.g. the check inverted) would make the two tests above pass for the wrong
|
||||
* reason: not because they actually drove a real taint through a real guard, but because
|
||||
* anything would have thrown {@code GuardException}. Here, with nothing to taint, the guard
|
||||
* must let {@code main} proceed into {@code cfg.validateAll()} and on into the real assembly,
|
||||
* which reaches {@code ports.connectHerdr} — and THAT throws. A loud, positive assertion: if
|
||||
* the boot path never actually ran this far, there is no {@link UnsupportedOperationException}
|
||||
* to catch, only a quiet, unexpected hang or an unrelated early failure.
|
||||
*/
|
||||
@Test
|
||||
void mainProceedsPastTheGuardOnACleanEnvironment(@TempDir Path dir) throws Exception {
|
||||
Path config = validConfig(dir);
|
||||
FixedEnvPorts ports = new FixedEnvPorts(CLEAN_ENV);
|
||||
|
||||
UnsupportedOperationException ex = assertThrows(UnsupportedOperationException.class,
|
||||
() -> Fleetd.main(new String[]{config.toString()}, ports));
|
||||
assertTrue(ex.getMessage().contains("connectHerdr"),
|
||||
"expected forward progress to reach the assembly's first port call, got: " + ex.getMessage());
|
||||
}
|
||||
}
|
||||
@@ -1,96 +1,174 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import com.tngtech.archunit.base.DescribedPredicate;
|
||||
import com.tngtech.archunit.core.domain.Dependency;
|
||||
import com.tngtech.archunit.core.domain.JavaClass;
|
||||
import com.tngtech.archunit.core.domain.JavaClass.Predicates;
|
||||
import com.tngtech.archunit.core.domain.JavaClasses;
|
||||
import com.tngtech.archunit.core.importer.ClassFileImporter;
|
||||
import com.tngtech.archunit.core.importer.ImportOption;
|
||||
import com.tngtech.archunit.library.dependencies.SliceRule;
|
||||
import com.tngtech.archunit.library.dependencies.SlicesRuleDefinition;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.fail;
|
||||
|
||||
/**
|
||||
* fleetd #131 (CB-627): enforce package boundaries with an ArchUnit test instead of a
|
||||
* Maven module split.
|
||||
* Enforces package boundaries between the top-level {@code dev.ltms.fleet.*} packages.
|
||||
*
|
||||
* <p>This test fails the build the moment a NEW cycle appears between the top-level
|
||||
* {@code dev.ltms.fleet.*} packages. Today's cycles are recorded below as explicit,
|
||||
* narrow exceptions: each one ignores dependencies between exactly the two named
|
||||
* packages, in both directions, and nothing else. A cycle through any other pair of
|
||||
* packages -- or a brand new pair -- still fails this test.
|
||||
* <p>{@link #BASELINE_EDGES} names the exact {@code origin class -> target class}
|
||||
* dependencies allowed to cross a top-level package boundary. Any dependency between two
|
||||
* top-level packages that is not in that set fails this test, including a brand new
|
||||
* dependency between a pair of packages that already has other baselined edges. A baseline
|
||||
* entry whose dependency no longer exists in the code also fails this test, so the baseline
|
||||
* always names exactly today's exceptions and nothing more.
|
||||
*
|
||||
* <p><b>Main code only.</b> The import excludes test classes
|
||||
* ({@link ImportOption.Predefined#DO_NOT_INCLUDE_TESTS}). Test code legitimately wires
|
||||
* across many packages for setup and mocking; that is not part of the shipped
|
||||
* architecture this rule protects. Verified: importing test classes too pulls in a much
|
||||
* larger, noisier cycle set -- {@code herdr}, {@code member}, {@code peer}, {@code
|
||||
* config}, {@code guard} and {@code placement} all show up in cycles that disappear the
|
||||
* moment test classes are excluded. Scanning off the classpath via {@code
|
||||
* importPackages(...)} (not a hardcoded {@code target/classes} path) also keeps this
|
||||
* test correct regardless of the working directory the build is invoked from.
|
||||
*
|
||||
* <p><b>No package moves here</b> -- ticket #131 is explicit that removing a cycle is
|
||||
* its own, later PR. See the comment on each exception below for which ticket step
|
||||
* removes it.
|
||||
* ({@link ImportOption.Predefined#DO_NOT_INCLUDE_TESTS}). Scanning off the classpath via
|
||||
* {@code importPackages(...)} keeps this test correct regardless of the working directory
|
||||
* the build is invoked from.
|
||||
*/
|
||||
class PackageCyclesTest {
|
||||
|
||||
/**
|
||||
* Exact {@code "origin -> target"} class dependencies allowed to cross a top-level
|
||||
* package boundary. Each entry is one directed edge between two specific classes; a
|
||||
* two-way relationship between a pair of packages is listed as two separate entries,
|
||||
* one per direction.
|
||||
*/
|
||||
private static final Set<String> BASELINE_EDGES = Set.of(
|
||||
"dev.ltms.fleet.auth.CallerResolver -> dev.ltms.fleet.mcp.ConnectionIdentity",
|
||||
"dev.ltms.fleet.auth.CallerResolver -> dev.ltms.fleet.mcp.ConnectionIdentity$Caller",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.auth.AuditLog",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.auth.Authz",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.auth.Authz$Action",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.auth.CallerResolver",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.auth.Principal",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.auth.Role",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.msg.LeadChannel",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.msg.LeadChannel$MailboxState",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.msg.LeadMessage",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.msg.MessageService",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.msg.MessageService$AskOutcome",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.msg.MessageService$AskResult",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.msg.MessageService$Outcome",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.msg.MessageService$Outstanding",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.msg.MessageService$PendingAsk",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.msg.MessageService$Phase",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.msg.MessageService$Reply",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.msg.MessageService$ReplyOutcome",
|
||||
"dev.ltms.fleet.mcp.FleetMcp -> dev.ltms.fleet.msg.MessageService$TaskView",
|
||||
"dev.ltms.fleet.mcp.FleetMcp$1 -> dev.ltms.fleet.msg.MessageService$AskOutcome",
|
||||
"dev.ltms.fleet.mcp.FleetMcp$1 -> dev.ltms.fleet.msg.MessageService$Outcome",
|
||||
"dev.ltms.fleet.mcp.FleetMcp$1 -> dev.ltms.fleet.msg.MessageService$Phase",
|
||||
"dev.ltms.fleet.mcp.FleetMcp$CoordinationSource -> dev.ltms.fleet.msg.LeadChannel",
|
||||
"dev.ltms.fleet.msg.LeadHeartbeatLoop -> dev.ltms.fleet.mcp.PrimaryRegistry",
|
||||
"dev.ltms.fleet.msg.ReplyPushLoop -> dev.ltms.fleet.mcp.PrimaryRegistry",
|
||||
"dev.ltms.fleet.inject.CompletionResolver -> dev.ltms.fleet.msg.Rendezvous",
|
||||
"dev.ltms.fleet.inject.CompletionResolver -> dev.ltms.fleet.msg.Rendezvous$Resolution",
|
||||
"dev.ltms.fleet.inject.CompletionResolver -> dev.ltms.fleet.msg.TurnToken",
|
||||
"dev.ltms.fleet.inject.CompletionResolver$InFlight -> dev.ltms.fleet.msg.Rendezvous$Resolution",
|
||||
"dev.ltms.fleet.inject.Injector -> dev.ltms.fleet.msg.TurnToken",
|
||||
"dev.ltms.fleet.inject.Injector$Pending -> dev.ltms.fleet.msg.TurnToken",
|
||||
"dev.ltms.fleet.inject.TurnListener -> dev.ltms.fleet.msg.TurnToken",
|
||||
"dev.ltms.fleet.inject.TurnRegistrar -> dev.ltms.fleet.msg.TurnToken",
|
||||
"dev.ltms.fleet.msg.MessageService -> dev.ltms.fleet.inject.Injector",
|
||||
"dev.ltms.fleet.msg.MessageService -> dev.ltms.fleet.inject.Injector$Cancellation",
|
||||
"dev.ltms.fleet.msg.MessageService -> dev.ltms.fleet.inject.Injector$Delivery",
|
||||
"dev.ltms.fleet.metrics.FleetMetrics -> dev.ltms.fleet.msg.ReplyInbox",
|
||||
"dev.ltms.fleet.msg.LeadHeartbeatLoop -> dev.ltms.fleet.metrics.Metrics",
|
||||
"dev.ltms.fleet.msg.MessageService -> dev.ltms.fleet.metrics.Metrics",
|
||||
"dev.ltms.fleet.msg.ReplyPushLoop -> dev.ltms.fleet.metrics.Metrics",
|
||||
"dev.ltms.fleet.msg.LeadHeartbeatLoop -> dev.ltms.fleet.session.MemberSession",
|
||||
"dev.ltms.fleet.msg.LeadHeartbeatLoop -> dev.ltms.fleet.session.MemberSession$State",
|
||||
"dev.ltms.fleet.session.SessionManager -> dev.ltms.fleet.msg.TurnToken"
|
||||
);
|
||||
|
||||
private static final String ROOT_PACKAGE = "dev.ltms.fleet.";
|
||||
|
||||
@Test
|
||||
void packagesAreFreeOfCycles() {
|
||||
var classes = new ClassFileImporter()
|
||||
JavaClasses classes = new ClassFileImporter()
|
||||
.withImportOption(ImportOption.Predefined.DO_NOT_INCLUDE_TESTS)
|
||||
.importPackages("dev.ltms.fleet");
|
||||
|
||||
checkBaselineMatchesTodaysEdges(classes);
|
||||
|
||||
SliceRule rule = SlicesRuleDefinition.slices()
|
||||
.matching("dev.ltms.fleet.(*)..")
|
||||
.should().beFreeOfCycles();
|
||||
|
||||
// fleetd #131 step 1: move ConnectionIdentity so authz stops depending on the
|
||||
// MCP layer. Evidence: auth/CallerResolver.java:3 imports mcp.ConnectionIdentity;
|
||||
// mcp/FleetMcp.java:3-7 imports auth.AuditLog, Authz, CallerResolver, Principal,
|
||||
// Role.
|
||||
rule = ignoreCycle(rule, "auth", "mcp");
|
||||
|
||||
// fleetd #131 step 2: PrimaryRegistry is used by loops in msg; move it, or put
|
||||
// an interface between msg and mcp. Evidence: msg/ReplyPushLoop.java:5 and
|
||||
// msg/LeadHeartbeatLoop.java:5 import mcp.PrimaryRegistry; mcp/FleetMcp.java:15-18
|
||||
// imports msg.LeadChannel, LeadMessage, MessageService, Rendezvous.
|
||||
rule = ignoreCycle(rule, "mcp", "msg");
|
||||
|
||||
// fleetd #131 -- found while implementing this test, NOT one of the ticket's
|
||||
// original three; it names its own follow-up step before removal. Evidence:
|
||||
// inject/CompletionResolver.java:4-5, inject/Injector.java:6 and
|
||||
// inject/TurnListener.java:3 import msg.Rendezvous / msg.TurnToken;
|
||||
// msg/MessageService.java:6 imports inject.Injector.
|
||||
rule = ignoreCycle(rule, "inject", "msg");
|
||||
|
||||
// fleetd #131 -- same as above, its own follow-up. Evidence:
|
||||
// metrics/FleetMetrics.java:3 imports msg.ReplyInbox; msg/MessageService.java:7-8,
|
||||
// msg/LeadHeartbeatLoop.java:6-7 and msg/ReplyPushLoop.java:6-7 import
|
||||
// metrics.FleetMetrics / metrics.Metrics.
|
||||
rule = ignoreCycle(rule, "metrics", "msg");
|
||||
|
||||
// fleetd #131 -- same as above, its own follow-up. Evidence:
|
||||
// session/SessionManager.java:7 imports msg.TurnToken;
|
||||
// msg/LeadHeartbeatLoop.java:8 imports session.MemberSession.
|
||||
rule = ignoreCycle(rule, "msg", "session");
|
||||
|
||||
for (String edge : BASELINE_EDGES) {
|
||||
String[] originAndTarget = edge.split(" -> ");
|
||||
rule = rule.ignoreDependency(originAndTarget[0], originAndTarget[1]);
|
||||
}
|
||||
rule.check(classes);
|
||||
}
|
||||
|
||||
/**
|
||||
* Accepts today's known cycle between two top-level packages, and nothing else.
|
||||
* Ignoring both directions removes exactly this pair from cycle detection; every
|
||||
* other dependency -- including any new one added later, between these same two
|
||||
* packages or any other pair -- is still checked.
|
||||
* Fails with the exact offending edge when the live code and {@link #BASELINE_EDGES}
|
||||
* disagree: a dependency crossing a baselined package pair that is not in the baseline,
|
||||
* or a baseline entry whose dependency no longer exists.
|
||||
*/
|
||||
private static SliceRule ignoreCycle(SliceRule rule, String packageA, String packageB) {
|
||||
return rule
|
||||
.ignoreDependency(residesIn(packageA), residesIn(packageB))
|
||||
.ignoreDependency(residesIn(packageB), residesIn(packageA));
|
||||
private static void checkBaselineMatchesTodaysEdges(JavaClasses classes) {
|
||||
Set<String> baselinedPackagePairs = new TreeSet<>();
|
||||
for (String edge : BASELINE_EDGES) {
|
||||
String[] originAndTarget = edge.split(" -> ");
|
||||
baselinedPackagePairs.add(unorderedPair(
|
||||
topLevelPackageOf(originAndTarget[0]), topLevelPackageOf(originAndTarget[1])));
|
||||
}
|
||||
|
||||
Set<String> liveEdgesInBaselinedPairs = new TreeSet<>();
|
||||
for (JavaClass javaClass : classes) {
|
||||
for (Dependency dependency : javaClass.getDirectDependenciesFromSelf()) {
|
||||
JavaClass origin = dependency.getOriginClass();
|
||||
JavaClass target = dependency.getTargetClass();
|
||||
String originPackage = topLevelPackageOf(origin.getFullName());
|
||||
String targetPackage = topLevelPackageOf(target.getFullName());
|
||||
if (originPackage.isEmpty() || targetPackage.isEmpty() || originPackage.equals(targetPackage)) {
|
||||
continue;
|
||||
}
|
||||
if (baselinedPackagePairs.contains(unorderedPair(originPackage, targetPackage))) {
|
||||
liveEdgesInBaselinedPairs.add(origin.getFullName() + " -> " + target.getFullName());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
List<String> problems = new ArrayList<>();
|
||||
for (String liveEdge : liveEdgesInBaselinedPairs) {
|
||||
if (!BASELINE_EDGES.contains(liveEdge)) {
|
||||
String[] originAndTarget = liveEdge.split(" -> ");
|
||||
problems.add("new dependency not in the baseline: " + liveEdge
|
||||
+ " (packages " + topLevelPackageOf(originAndTarget[0])
|
||||
+ " -> " + topLevelPackageOf(originAndTarget[1]) + ")");
|
||||
}
|
||||
}
|
||||
for (String baselineEdge : BASELINE_EDGES) {
|
||||
if (!liveEdgesInBaselinedPairs.contains(baselineEdge)) {
|
||||
String[] originAndTarget = baselineEdge.split(" -> ");
|
||||
problems.add("stale baseline entry, no such dependency exists: " + baselineEdge
|
||||
+ " (packages " + topLevelPackageOf(originAndTarget[0])
|
||||
+ " -> " + topLevelPackageOf(originAndTarget[1]) + ")");
|
||||
}
|
||||
}
|
||||
|
||||
if (!problems.isEmpty()) {
|
||||
fail("PackageCyclesTest baseline is out of date:\n " + String.join("\n ", problems));
|
||||
}
|
||||
}
|
||||
|
||||
private static DescribedPredicate<JavaClass> residesIn(String topLevelPackage) {
|
||||
return Predicates.resideInAPackage("dev.ltms.fleet." + topLevelPackage + "..");
|
||||
private static String unorderedPair(String packageA, String packageB) {
|
||||
return packageA.compareTo(packageB) <= 0 ? packageA + "|" + packageB : packageB + "|" + packageA;
|
||||
}
|
||||
|
||||
private static String topLevelPackageOf(String fullyQualifiedClassName) {
|
||||
if (!fullyQualifiedClassName.startsWith(ROOT_PACKAGE)) {
|
||||
return "";
|
||||
}
|
||||
String rest = fullyQualifiedClassName.substring(ROOT_PACKAGE.length());
|
||||
int dot = rest.indexOf('.');
|
||||
return dot < 0 ? "" : rest.substring(0, dot);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -14,6 +14,8 @@ class AuthzTest {
|
||||
private static final Principal ANON = Principal.anonymous();
|
||||
private static final Principal ARCH_DESIGN = Principal.architect("lead-designer", "term_design", 400);
|
||||
private static final Principal ARCH_OTHER = Principal.architect("reviewer", "term_review", 500);
|
||||
private static final Principal COLLABORATOR = Principal.collaborator("ops", "term_collab", 600);
|
||||
private static final Principal OBSERVER = Principal.observer("term_observer", 700);
|
||||
|
||||
@Test
|
||||
void anonymousIsAuthorizedForNothing() {
|
||||
@@ -29,6 +31,44 @@ class AuthzTest {
|
||||
assertTrue(Authz.isUnauthenticated(null));
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code MessageService.answer}'s turn-ownership check treats a caller with no terminal as
|
||||
* matching a turn recorded for the unnamed primary. That rule only stays safe because an
|
||||
* unauthenticated caller — whose terminal is also {@code null} — never reaches {@code answer}
|
||||
* at all: {@link #anonymousIsAuthorizedForNothing} already covers every action including
|
||||
* {@code ANSWER}, but this test names the exact coupling so a future change to either side
|
||||
* cannot drift without turning this test red.
|
||||
*/
|
||||
@Test
|
||||
void anAnonymousCallerIsRefusedAnswerSoItCanNeverBeMistakenForTheUnnamedPrimary() {
|
||||
assertFalse(Authz.permits(ANON, ANSWER, null),
|
||||
"an anonymous caller, whose terminal is also null, must never reach answer() — the "
|
||||
+ "turn-ownership check's null-terminal match for the unnamed primary owner "
|
||||
+ "relies on this gate refusing it first");
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code fleet_inbox} collects the mail queued for the caller's own pane, so it is gated on
|
||||
* terminal ownership and on nothing else: every role may do it for itself, and no role may do
|
||||
* it for another pane.
|
||||
*/
|
||||
@Test
|
||||
void collectingAnInboxIsOnlyEverForTheCallersOwnPane() {
|
||||
for (Principal self : new Principal[]{WORKER_A, ARCH_DESIGN, COLLABORATOR, OBSERVER}) {
|
||||
assertTrue(Authz.permits(self, INBOX, self.terminal()),
|
||||
self.role() + " must be able to collect the mail for its own pane");
|
||||
assertFalse(Authz.permits(self, INBOX, "term_someone_else"),
|
||||
self.role() + " must not be able to collect another pane's mail");
|
||||
}
|
||||
// A lead carries a pane too, so it collects its own mail on the same rule.
|
||||
assertTrue(Authz.permits(Principal.leader("opus", "term_lead", 800), INBOX, "term_lead"));
|
||||
// The unnamed primary owns no pane, so there is no inbox it could be asking for.
|
||||
assertFalse(Authz.permits(PRIMARY, INBOX, "term_a"),
|
||||
"a caller with no pane of its own has no inbox to collect");
|
||||
assertFalse(Authz.permits(WORKER_A, INBOX, null),
|
||||
"a missing terminal must never match an owner");
|
||||
}
|
||||
|
||||
@Test
|
||||
void orchestrationBelongsToThePrimaryAlone() {
|
||||
for (Authz.Action a : new Authz.Action[]{SPAWN, STOP, SEND, DRAIN}) {
|
||||
@@ -38,6 +78,37 @@ class AuthzTest {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code fleet_send} is three call shapes behind one action name until {@code
|
||||
* FleetMcp#sendAction} picks one: a plain local {@link Authz.Action#SEND}, the {@code coordId}
|
||||
* route ({@link Authz.Action#COORD_SEND}), and the {@code turnId} answer form ({@link
|
||||
* Authz.Action#ANSWER}). All three carry the same grant as the undivided action did — a worker
|
||||
* is excluded from every one, exactly as it was excluded from the one combined action before.
|
||||
*/
|
||||
@Test
|
||||
void theThreeSendShapesCarryTheSameGrantAsTheOldUndividedAction() {
|
||||
for (Authz.Action a : new Authz.Action[]{SEND, COORD_SEND, ANSWER}) {
|
||||
assertTrue(Authz.permits(PRIMARY, a, "term_a"), "the primary may " + a);
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, a, "term_a"), "an architect may " + a);
|
||||
assertFalse(Authz.permits(WORKER_A, a, "term_a"),
|
||||
"a worker performing " + a + " would be escalating into the orchestrator role");
|
||||
assertFalse(Authz.permits(ANON, a, "term_a"));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code fleet_poll{ticket}} and {@code fleet_status} are {@link Authz.Action#TASK_READ}, split
|
||||
* out of the roster-only {@link Authz.Action#READ} (fleetd #678). The grant is unchanged from
|
||||
* what the undivided {@code READ} action gave every one of these callers.
|
||||
*/
|
||||
@Test
|
||||
void taskReadCarriesTheSameGrantReadDidBeforeTheSplit() {
|
||||
assertTrue(Authz.permits(PRIMARY, TASK_READ, null));
|
||||
assertTrue(Authz.permits(WORKER_A, TASK_READ, null));
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, TASK_READ, null));
|
||||
assertFalse(Authz.permits(ANON, TASK_READ, null));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerMayReplyAndAskOnlyAsItself() {
|
||||
assertTrue(Authz.permits(WORKER_A, REPLY, "term_a"));
|
||||
@@ -135,4 +206,184 @@ class AuthzTest {
|
||||
assertFalse(Authz.isUnauthenticated(WORKER_A));
|
||||
assertFalse(Authz.isUnauthenticated(PRIMARY));
|
||||
}
|
||||
|
||||
// ── the collaborator matrix ─────────────────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* {@code SEND} for a collaborator is the one grant that is conditional rather than fixed:
|
||||
* flipping only the classifier's answer for the target flips only this outcome.
|
||||
*/
|
||||
@Test
|
||||
void aCollaboratorMaySendOnlyWhenTheClassifierAcceptsTheTarget() {
|
||||
assertTrue(Authz.permits(COLLABORATOR, SEND, "term_lead", target -> true),
|
||||
"the classifier accepting the target must grant SEND");
|
||||
assertFalse(Authz.permits(COLLABORATOR, SEND, "term_lead", target -> false),
|
||||
"the classifier refusing the target must deny SEND");
|
||||
assertFalse(Authz.permits(COLLABORATOR, SEND, "term_lead"),
|
||||
"the real production classifier recognises no terminal yet, so SEND is refused today");
|
||||
}
|
||||
|
||||
/**
|
||||
* Control for the test above: every other action's result for a collaborator does not move
|
||||
* when the classifier does. Only {@code SEND} is wired to it.
|
||||
*/
|
||||
@Test
|
||||
void theClassifierMovesOnlySendForACollaborator() {
|
||||
for (Authz.Action a : Authz.Action.values()) {
|
||||
if (a == SEND) {
|
||||
continue;
|
||||
}
|
||||
assertEquals(
|
||||
Authz.permits(COLLABORATOR, a, "term_lead"),
|
||||
Authz.permits(COLLABORATOR, a, "term_lead", target -> true),
|
||||
a + " must not depend on the classifier at all");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aCollaboratorMayReadAndScrapeMetrics() {
|
||||
assertTrue(Authz.permits(COLLABORATOR, READ, null));
|
||||
assertTrue(Authz.permits(COLLABORATOR, METRICS, null));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aCollaboratorMayReplyAndAskOnlyAsItsOwnPane() {
|
||||
assertTrue(Authz.permits(COLLABORATOR, REPLY, "term_collab"),
|
||||
"its own pane is its own");
|
||||
assertTrue(Authz.permits(COLLABORATOR, ASK, "term_collab"));
|
||||
|
||||
assertFalse(Authz.permits(COLLABORATOR, REPLY, "term_design"),
|
||||
"a collaborator must not reply on another pane");
|
||||
assertFalse(Authz.permits(COLLABORATOR, REPLY, null),
|
||||
"an absent target must not pass the own-session rule");
|
||||
}
|
||||
|
||||
/**
|
||||
* Every action denied to a collaborator, asserted denied even when the classifier would
|
||||
* accept any target — proving none of these is actually gated on the classifier at all.
|
||||
*/
|
||||
@Test
|
||||
void aCollaboratorIsDeniedLifecycleCoordinationAndTicketPolling() {
|
||||
for (Authz.Action a : new Authz.Action[]{SPAWN, STOP, DRAIN, HANDOVER, ANSWER, COORD_SEND,
|
||||
COORD_READ, TASK_READ}) {
|
||||
assertFalse(Authz.permits(COLLABORATOR, a, "term_lead", target -> true),
|
||||
"a collaborator must not " + a + " even when the classifier accepts every target");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aCollaboratorIsNotCountedAsPrimaryWorkerOrArchitect() {
|
||||
assertFalse(COLLABORATOR.isPrimary());
|
||||
assertFalse(COLLABORATOR.isWorker());
|
||||
assertFalse(COLLABORATOR.isArchitect());
|
||||
assertTrue(COLLABORATOR.isCollaborator());
|
||||
}
|
||||
|
||||
/**
|
||||
* A collaborator is never spawned, so it must not be enrolled in the presence map as an
|
||||
* available member. Control: both a worker and an architect — which ARE spawned — still are.
|
||||
*/
|
||||
@Test
|
||||
void isSpawnedMemberIsFalseForACollaboratorButTrueForAWorkerAndAnArchitect() {
|
||||
assertFalse(COLLABORATOR.isSpawnedMember());
|
||||
assertTrue(WORKER_A.isSpawnedMember());
|
||||
assertTrue(ARCH_DESIGN.isSpawnedMember());
|
||||
}
|
||||
|
||||
// ── the observer matrix ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void anObserverMayReadAndScrapeMetrics() {
|
||||
assertTrue(Authz.permits(OBSERVER, READ, null));
|
||||
assertTrue(Authz.permits(OBSERVER, METRICS, null));
|
||||
}
|
||||
|
||||
@Test
|
||||
void anObserverMayReplyAndAskOnlyAsItsOwnPane() {
|
||||
assertTrue(Authz.permits(OBSERVER, REPLY, "term_observer"), "its own pane is its own");
|
||||
assertTrue(Authz.permits(OBSERVER, ASK, "term_observer"));
|
||||
|
||||
assertFalse(Authz.permits(OBSERVER, REPLY, "term_design"),
|
||||
"an observer must not reply on another pane");
|
||||
assertFalse(Authz.permits(OBSERVER, REPLY, null),
|
||||
"an absent target must not pass the own-session rule");
|
||||
}
|
||||
|
||||
/**
|
||||
* Every action beyond READ/METRICS/REPLY/ASK/INBOX/SEND, asserted denied for an observer —
|
||||
* including {@code TASK_READ}, which is the entire point of this role: an unconfigured pane
|
||||
* must not be able to poll a ticket or read another session's status. The exempt set is the
|
||||
* three only-as-itself actions plus the two open reads. {@code SEND} is excluded here and
|
||||
* given its own matrix below, since — unlike every action in this loop — its grant is
|
||||
* conditional on the target, not fixed.
|
||||
*/
|
||||
@Test
|
||||
void anObserverIsDeniedEverythingBeyondReadMetricsReplyAskInboxAndSend() {
|
||||
for (Authz.Action a : Authz.Action.values()) {
|
||||
if (a == READ || a == METRICS || a == REPLY || a == ASK || a == INBOX || a == SEND) {
|
||||
continue;
|
||||
}
|
||||
assertFalse(Authz.permits(OBSERVER, a, "term_observer", target -> true),
|
||||
"an observer must not " + a + " even when the classifier accepts every target");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void anObserverIsNotCountedAsAnyOtherRole() {
|
||||
assertFalse(OBSERVER.isPrimary());
|
||||
assertFalse(OBSERVER.isWorker());
|
||||
assertFalse(OBSERVER.isArchitect());
|
||||
assertFalse(OBSERVER.isCollaborator());
|
||||
assertFalse(OBSERVER.isSpawnedMember());
|
||||
assertTrue(OBSERVER.isObserver());
|
||||
}
|
||||
|
||||
// ── the observer SEND matrix ────────────────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* {@code SEND} for an observer is the one grant that is conditional rather than fixed, exactly
|
||||
* like a collaborator's: flipping only the observer-target classifier's answer flips only this
|
||||
* outcome.
|
||||
*/
|
||||
@Test
|
||||
void anObserverMaySendOnlyWhenTheClassifierAcceptsTheTargetAsAnObserver() {
|
||||
assertTrue(Authz.permits(OBSERVER, SEND, "term_other_observer",
|
||||
Authz.NO_KNOWN_LEAD_OR_COLLABORATOR, target -> true),
|
||||
"the classifier accepting the target as an observer must grant SEND");
|
||||
assertFalse(Authz.permits(OBSERVER, SEND, "term_other_observer",
|
||||
Authz.NO_KNOWN_LEAD_OR_COLLABORATOR, target -> false),
|
||||
"the classifier refusing the target must deny SEND");
|
||||
assertFalse(Authz.permits(OBSERVER, SEND, "term_other_observer"),
|
||||
"the real production classifier recognises no terminal as an observer target yet, "
|
||||
+ "so SEND is refused by the three- and four-argument convenience forms");
|
||||
}
|
||||
|
||||
/**
|
||||
* Control for the test above: every other action's result for an observer does not move when
|
||||
* the observer-target classifier does. Only {@code SEND} is wired to it.
|
||||
*/
|
||||
@Test
|
||||
void theObserverTargetClassifierMovesOnlySendForAnObserver() {
|
||||
for (Authz.Action a : Authz.Action.values()) {
|
||||
if (a == SEND) {
|
||||
continue;
|
||||
}
|
||||
assertEquals(
|
||||
Authz.permits(OBSERVER, a, "term_observer"),
|
||||
Authz.permits(OBSERVER, a, "term_observer",
|
||||
Authz.NO_KNOWN_LEAD_OR_COLLABORATOR, target -> true),
|
||||
a + " must not depend on the observer-target classifier at all");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* An observer's {@code SEND} is gated on a different classifier than a collaborator's: the
|
||||
* collaborator classifier accepting every target must not itself grant an observer's SEND.
|
||||
*/
|
||||
@Test
|
||||
void anObserversSendDoesNotMoveOnTheCollaboratorClassifier() {
|
||||
assertFalse(Authz.permits(OBSERVER, SEND, "term_lead", target -> true),
|
||||
"an observer's SEND must consult the observer-target classifier, never the "
|
||||
+ "lead-or-collaborator one");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,13 +1,25 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.mcp.ConnectionIdentity;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.session.WorktreeRequest;
|
||||
import dev.ltms.fleet.session.Worktrees;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
@@ -45,17 +57,23 @@ class CallerResolverTest {
|
||||
return members;
|
||||
}
|
||||
|
||||
/**
|
||||
* With no roster wired up at all (the simple constructor), a loopback pane that owns a herdr
|
||||
* pane but is not recognised as a live spawned member lands on the {@link Role#OBSERVER} floor
|
||||
* — unforgeable and never token-gated, exactly like a worker's own identity, because it comes
|
||||
* from the same connection-derived pane mapping.
|
||||
*/
|
||||
@Test
|
||||
void aLoopbackWorkerPaneResolvesToWorkerRegardlessOfAuthMode() {
|
||||
void aLoopbackPaneWithNoLiveRosterResolvesToObserverRegardlessOfAuthMode() {
|
||||
Principal underTrust = new CallerResolver(workerIdentity()).resolve("127.0.0.1", 42, null);
|
||||
Principal underToken = new CallerResolver(workerIdentity(), true, "s3cret")
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, underTrust.role());
|
||||
assertEquals(Role.OBSERVER, underTrust.role());
|
||||
assertEquals("term_a", underTrust.terminal());
|
||||
assertEquals(Role.WORKER, underToken.role(),
|
||||
"worker identity is unforgeable and must never be token-gated — otherwise enabling "
|
||||
+ "auth would lock the whole fleet out of fleet_reply");
|
||||
assertEquals(Role.OBSERVER, underToken.role(),
|
||||
"the floor is unforgeable and must never be token-gated — otherwise enabling auth "
|
||||
+ "would lock every unconfigured pane out of even READ");
|
||||
assertEquals("term_a", underToken.terminal());
|
||||
}
|
||||
|
||||
@@ -79,20 +97,20 @@ class CallerResolverTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void otherPanesRemainWorkersWhenAPinIsSet() {
|
||||
void otherPanesRemainAtTheFloorWhenAPinIsSet() {
|
||||
Principal p = CallerResolver.pinnedTo(workerIdentity(), false, null, "term_someone_else")
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role());
|
||||
assertEquals(Role.OBSERVER, p.role());
|
||||
assertEquals("term_a", p.terminal());
|
||||
}
|
||||
|
||||
/** The pin is optional config, so an absent or whitespace one must change nothing at all. */
|
||||
@Test
|
||||
void aBlankPinLeavesWorkerResolutionUntouched() {
|
||||
assertEquals(Role.WORKER,
|
||||
void aBlankPinLeavesFloorResolutionUntouched() {
|
||||
assertEquals(Role.OBSERVER,
|
||||
CallerResolver.pinnedTo(workerIdentity(), false, null, " ").resolve("127.0.0.1", 42, null).role());
|
||||
assertEquals(Role.WORKER,
|
||||
assertEquals(Role.OBSERVER,
|
||||
CallerResolver.pinnedTo(workerIdentity(), false, null, null).resolve("127.0.0.1", 42, null).role());
|
||||
}
|
||||
|
||||
@@ -213,13 +231,13 @@ class CallerResolverTest {
|
||||
* mid-scan teardown into a refusal — the real match is still found and resolves as a worker.
|
||||
*/
|
||||
@Test
|
||||
void aHerdrErrorOnANonOwningPaneStillResolvesTheRealWorker() {
|
||||
void aHerdrErrorOnANonOwningPaneStillResolvesTheRealPane() {
|
||||
FakeHerdr vanishedElsewhere = new FakeHerdr().processInfoFailsForPane("w2:p9", "pane_not_found");
|
||||
ConnectionIdentity id = new ConnectionIdentity(new PaneLocator(vanishedElsewhere), _ -> FakeHerdr.WORKER_PID);
|
||||
|
||||
Principal p = new CallerResolver(id).resolve("127.0.0.1", 55555, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role());
|
||||
assertEquals(Role.OBSERVER, p.role());
|
||||
assertEquals("term_a", p.terminal());
|
||||
}
|
||||
|
||||
@@ -263,12 +281,12 @@ class CallerResolverTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPaneAbsentFromTheRegistryIsStillAWorker() {
|
||||
void aPaneAbsentFromTheRegistryFallsToTheObserverFloor() {
|
||||
Principal p = new CallerResolver(workerIdentity(), false, null,
|
||||
Map.of("term_elsewhere", "gpt-sol-5.6"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role());
|
||||
assertEquals(Role.OBSERVER, p.role());
|
||||
assertEquals("term_a", p.terminal());
|
||||
assertNull(p.name());
|
||||
}
|
||||
@@ -294,16 +312,16 @@ class CallerResolverTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void anEmptyRegistryLeavesEveryPaneAWorker() {
|
||||
void anEmptyRegistryLeavesEveryPaneAtTheObserverFloor() {
|
||||
Map<String, String> noLeads = null;
|
||||
assertEquals(Role.WORKER,
|
||||
assertEquals(Role.OBSERVER,
|
||||
new CallerResolver(workerIdentity(), false, null, Map.of())
|
||||
.resolve("127.0.0.1", 42, null).role());
|
||||
assertEquals(Role.WORKER,
|
||||
assertEquals(Role.OBSERVER,
|
||||
new CallerResolver(workerIdentity(), false, null, noLeads)
|
||||
.resolve("127.0.0.1", 42, null).role());
|
||||
// CB-531: and the same for the live-registry form, whose supplier may also be absent.
|
||||
assertEquals(Role.WORKER,
|
||||
// And the same for the live-registry form, whose supplier may also be absent.
|
||||
assertEquals(Role.OBSERVER,
|
||||
CallerResolver.withLeads(workerIdentity(), false, null, null)
|
||||
.resolve("127.0.0.1", 42, null).role());
|
||||
}
|
||||
@@ -376,7 +394,7 @@ class CallerResolverTest {
|
||||
Map<String, String> live = new java.util.HashMap<>();
|
||||
CallerResolver r = CallerResolver.withLeads(workerIdentity(), false, null, () -> live);
|
||||
|
||||
assertEquals(Role.WORKER, r.resolve("127.0.0.1", 42, null).role());
|
||||
assertEquals(Role.OBSERVER, r.resolve("127.0.0.1", 42, null).role());
|
||||
|
||||
live.put("term_a", "gpt-sol-5.6"); // the scanner sees a newly-labelled tab
|
||||
|
||||
@@ -393,7 +411,7 @@ class CallerResolverTest {
|
||||
|
||||
mutable.put("term_a", "sneaky");
|
||||
|
||||
assertEquals(Role.WORKER, r.resolve("127.0.0.1", 42, null).role());
|
||||
assertEquals(Role.OBSERVER, r.resolve("127.0.0.1", 42, null).role());
|
||||
}
|
||||
|
||||
// ── CB-548: architect slots ─────────────────────────────────────────────────────────────────
|
||||
@@ -423,13 +441,13 @@ class CallerResolverTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void anUnboundPaneStillResolvesAsAWorker() {
|
||||
void anUnboundPaneResolvesToTheObserverFloor() {
|
||||
MemberRegistry members = new MemberRegistry(new FleetConfig.Fleet(Map.of(),
|
||||
Map.of("lead-designer", new FleetConfig.Slot("sonnet")), Map.of(), Map.of(), null));
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, members).resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role());
|
||||
assertEquals(Role.OBSERVER, p.role());
|
||||
assertNull(p.name());
|
||||
}
|
||||
|
||||
@@ -455,12 +473,11 @@ class CallerResolverTest {
|
||||
CallerResolver r = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, members);
|
||||
|
||||
assertEquals(Role.WORKER, r.resolve("127.0.0.1", 42, null).role());
|
||||
assertEquals(Role.OBSERVER, r.resolve("127.0.0.1", 42, null).role());
|
||||
|
||||
assertTrue(members.bind("architect:lead-designer", "term_a")); // the later lifecycle binds the slot
|
||||
|
||||
assertEquals(Role.ARCHITECT, r.resolve("127.0.0.1", 42, null).role());
|
||||
assertEquals("architect:lead-designer", r.members().get("term_a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -483,12 +500,13 @@ class CallerResolverTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBoundNonArchitectSlotStillResolvesAsAWorker() {
|
||||
void aBoundNonArchitectSlotResolvesToTheObserverFloorNotArchitect() {
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, boundMembers("dev:builder", MemberRole.DEV))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role(), "a dev binding must never grant architect rights");
|
||||
assertEquals(Role.OBSERVER, p.role(), "a dev binding must never grant architect rights, "
|
||||
+ "and this construction path wires no roster to recognise it as the live dev it is");
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -499,17 +517,17 @@ class CallerResolverTest {
|
||||
assertThrows(IllegalArgumentException.class, () -> new CallerResolver(id, true, " "));
|
||||
}
|
||||
@Test
|
||||
void aWorkerOnAnyLoopbackSourceAddressIsStillAWorkerNotThePrimary() {
|
||||
// fleetd #305: the escalation. ConnectionIdentity used to accept only 127.0.0.1, so a
|
||||
// worker connecting from 127.0.0.2 resolved to no terminal, and this resolver's own
|
||||
// (wider) loopback check then made it the PRIMARY — granting spawn, stop, send and drain.
|
||||
// Measured on the Linux fleet host: binding a source of 127.0.0.2 succeeds there, so the
|
||||
// path is real and not theoretical.
|
||||
void aPaneOnAnyLoopbackSourceAddressIsStillAtTheFloorNotThePrimary() {
|
||||
// fleetd #305: the escalation this guards against. ConnectionIdentity used to accept only
|
||||
// 127.0.0.1, so a pane connecting from 127.0.0.2 resolved to no terminal, and this
|
||||
// resolver's own (wider) loopback check then made it the PRIMARY — granting spawn, stop,
|
||||
// send and drain. Measured on the Linux fleet host: binding a source of 127.0.0.2 succeeds
|
||||
// there, so the path is real and not theoretical.
|
||||
CallerResolver r = new CallerResolver(workerIdentity(), false, null);
|
||||
for (String src : new String[]{"127.0.0.1", "127.0.0.2", "127.1.2.3", "::ffff:127.0.0.2"}) {
|
||||
Principal p = r.resolve(src, 55555, null);
|
||||
assertEquals(Role.WORKER, p.role(), "a worker must stay a worker from source " + src);
|
||||
assertEquals("term_a", p.terminal(), "worker terminal from source " + src);
|
||||
assertEquals(Role.OBSERVER, p.role(), "the pane must stay off PRIMARY from source " + src);
|
||||
assertEquals("term_a", p.terminal(), "pane terminal from source " + src);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -523,4 +541,443 @@ class CallerResolverTest {
|
||||
}
|
||||
}
|
||||
|
||||
// ── fleetd #669 Unit D: a live spawned member outranks every tab map ────────────────────────
|
||||
|
||||
/**
|
||||
* Criterion 1: a terminal present in BOTH the spawned-member roster AND the lead tab map
|
||||
* resolves as its member role, not as a lead — the roster is checked first, consulting no tab
|
||||
* map at all when it matches.
|
||||
*/
|
||||
@Test
|
||||
void aSpawnedMemberWinsOverALeadTabForTheSamePane() {
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
() -> Map.of("term_a", "opus-5.0"), new MemberRegistry(null),
|
||||
t -> "term_a".equals(t) ? MemberRole.DEV : null, Map::of)
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role(),
|
||||
"a live spawned member's own identity must win over a tab map naming the same pane a lead");
|
||||
assertEquals("term_a", p.terminal());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #702: between {@link SessionManager#release} removing the registry entry and the
|
||||
* pane actually stopping, a resolve for that terminal must still see the live member and
|
||||
* never fall through to a tab map — wired through the real {@link SessionManager}, not a
|
||||
* hand-rolled stand-in for its {@code spawnedMemberRole}.
|
||||
*
|
||||
* <p>Reuses the ticket's own test idea: an injected {@code hasUncommitted} resolves the
|
||||
* releasing terminal from inside {@code release}'s window — a real call landing inside the
|
||||
* window, so no sleep and no race.
|
||||
*
|
||||
* <p>The property asserted is "no tab map is consulted", never "the same role is returned".
|
||||
* The mandatory control is the lead tab map: it names this exact terminal, and the same
|
||||
* resolve taken <em>outside</em> the window (before release runs) must still return the lead
|
||||
* role — without that control, the in-window assertion would also pass on an empty map and
|
||||
* prove nothing.
|
||||
*/
|
||||
@Test
|
||||
void aPaneMidTeardownResolvesAsItsOwnRoleConsultingNoTabMap() {
|
||||
FakeHerdr herdr = new FakeHerdr().pinNextStarts(1, "term_a", "w2:p7");
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
|
||||
AtomicReference<Principal> duringWindow = new AtomicReference<>();
|
||||
AtomicReference<CallerResolver> resolverRef = new AtomicReference<>();
|
||||
Worktrees worktrees = new Worktrees() {
|
||||
@Override
|
||||
public String add(String repoRoot, String branch, String baseRef) {
|
||||
return "/wt/" + branch.replace('/', '_');
|
||||
}
|
||||
|
||||
@Override
|
||||
public void remove(String repoRoot, String worktreePath) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void deleteBranch(String repoRoot, String branch) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasUncommitted(String worktreePath) {
|
||||
// Runs from INSIDE release()'s git-status shell-out: the registry entry is
|
||||
// already gone, but the pane has not stopped yet.
|
||||
duringWindow.set(resolverRef.get().resolve("127.0.0.1", 42, null));
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void overlayParity(String repoRoot, String worktreePath, List<String> overlay) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public String repoRoot(String cwd) {
|
||||
return "/repo";
|
||||
}
|
||||
|
||||
@Override
|
||||
public Optional<String> snapshot(String worktreePath, String branch, String message) {
|
||||
return Optional.empty();
|
||||
}
|
||||
|
||||
@Override
|
||||
public WipRefStats wipRefs(String repoRoot) {
|
||||
return new WipRefStats(0, 0L);
|
||||
}
|
||||
|
||||
@Override
|
||||
public int pruneWipRefs(String repoRoot, long minAgeMillis) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void shareWithGroup(String repoRoot, String worktreePath) {
|
||||
}
|
||||
};
|
||||
SessionManager sessions = new SessionManager(workers, worktrees);
|
||||
|
||||
// The lead tab map names "term_a" before anything is ever spawned onto it — the control
|
||||
// this test needs. Built up front so the SAME resolver answers every resolve() call below.
|
||||
ConnectionIdentity identity = new ConnectionIdentity(new PaneLocator(herdr), _ -> FakeHerdr.WORKER_PID);
|
||||
CallerResolver resolver = CallerResolver.withLeadsAndMembers(identity, false, null,
|
||||
() -> Map.of("term_a", "the-lead"), null, sessions::spawnedMemberRole, Map::of);
|
||||
resolverRef.set(resolver);
|
||||
|
||||
Principal before = resolver.resolve("127.0.0.1", 42, null);
|
||||
assertEquals(Role.PRIMARY, before.role(),
|
||||
"control: with no live or releasing member on this terminal, the lead tab map must "
|
||||
+ "win — this is what proves the in-window assertion below is not passing on "
|
||||
+ "an empty map");
|
||||
assertEquals("the-lead", before.name());
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("fleetd-702", null));
|
||||
assertEquals("term_a", s.terminalId(), "sanity: the spawn resolved to the pinned pane");
|
||||
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertNotNull(duringWindow.get(), "the dirty check must have run and captured a resolve");
|
||||
assertEquals(Role.WORKER, duringWindow.get().role(),
|
||||
"inside the window the pane must resolve as its own live-member role, consulting no "
|
||||
+ "tab map — a lead tab naming the same terminal must not win");
|
||||
assertEquals("term_a", duringWindow.get().terminal());
|
||||
}
|
||||
|
||||
/**
|
||||
* {@link SessionManager#releaseRemoved} unbinds the architect slot before the git-status
|
||||
* shell-out that opens the teardown window, so a resolve landing inside that window must see
|
||||
* the slot already unbound and resolve {@link Role#WORKER} — never {@link Role#ARCHITECT},
|
||||
* and never by checking role equality against the live session, which would hold even if the
|
||||
* unbind ran too late.
|
||||
*
|
||||
* <p>The control is the same resolve taken outside the window, while the slot is still bound,
|
||||
* which must return {@link Role#ARCHITECT} — without it this test would also pass against a
|
||||
* slot that was never bound, and prove nothing.
|
||||
*/
|
||||
@Test
|
||||
void aReleasingArchitectIsDemotedToWorkerInsideTheTeardownWindow() {
|
||||
FakeHerdr herdr = new FakeHerdr().pinNextStarts(1, "term_a", "w2:p7");
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
|
||||
AtomicReference<Principal> duringWindow = new AtomicReference<>();
|
||||
AtomicReference<CallerResolver> resolverRef = new AtomicReference<>();
|
||||
Worktrees worktrees = new Worktrees() {
|
||||
@Override
|
||||
public String add(String repoRoot, String branch, String baseRef) {
|
||||
return "/wt/" + branch.replace('/', '_');
|
||||
}
|
||||
|
||||
@Override
|
||||
public void remove(String repoRoot, String worktreePath) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void deleteBranch(String repoRoot, String branch) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasUncommitted(String worktreePath) {
|
||||
// Runs from INSIDE release()'s git-status shell-out: the architect slot is already
|
||||
// unbound by this point, but the pane has not stopped yet.
|
||||
duringWindow.set(resolverRef.get().resolve("127.0.0.1", 42, null));
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void overlayParity(String repoRoot, String worktreePath, List<String> overlay) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public String repoRoot(String cwd) {
|
||||
return "/repo";
|
||||
}
|
||||
|
||||
@Override
|
||||
public Optional<String> snapshot(String worktreePath, String branch, String message) {
|
||||
return Optional.empty();
|
||||
}
|
||||
|
||||
@Override
|
||||
public WipRefStats wipRefs(String repoRoot) {
|
||||
return new WipRefStats(0, 0L);
|
||||
}
|
||||
|
||||
@Override
|
||||
public int pruneWipRefs(String repoRoot, long minAgeMillis) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void shareWithGroup(String repoRoot, String worktreePath) {
|
||||
}
|
||||
};
|
||||
SessionManager sessions = new SessionManager(workers, worktrees);
|
||||
|
||||
MemberRegistry members = new MemberRegistry(new FleetConfig.Fleet(Map.of(),
|
||||
Map.of("lead-designer", new FleetConfig.Slot("ltms-local")), Map.of(), Map.of(), null));
|
||||
sessions.setMemberLifecycle(members);
|
||||
|
||||
ConnectionIdentity identity = new ConnectionIdentity(new PaneLocator(herdr), _ -> FakeHerdr.WORKER_PID);
|
||||
CallerResolver resolver = CallerResolver.withLeadsAndMembers(identity, false, null,
|
||||
Map::of, members, sessions::spawnedMemberRole, Map::of);
|
||||
resolverRef.set(resolver);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", MemberRole.ARCHITECT, null, "/caller/proj", null,
|
||||
new WorktreeRequest("fleetd-702d", null));
|
||||
assertEquals("term_a", s.terminalId(), "sanity: the spawn resolved to the pinned pane");
|
||||
assertEquals(MemberRole.ARCHITECT, s.role(), "sanity: the slot bind succeeded");
|
||||
|
||||
Principal before = resolver.resolve("127.0.0.1", 42, null);
|
||||
assertEquals(Role.ARCHITECT, before.role(),
|
||||
"control: with the slot still bound, the pane must resolve as an architect — this "
|
||||
+ "is what proves the in-window assertion below is not passing against a "
|
||||
+ "slot that was never bound");
|
||||
assertEquals("lead-designer", before.name());
|
||||
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertNotNull(duringWindow.get(), "the dirty check must have run and captured a resolve");
|
||||
assertEquals(Role.WORKER, duringWindow.get().role(),
|
||||
"inside the window the architect slot is already unbound, so the result must be "
|
||||
+ "WORKER — asserting role-equality with the live session here would tempt "
|
||||
+ "moving the unbind earlier or later, which would be wrong either way");
|
||||
assertEquals("term_a", duringWindow.get().terminal());
|
||||
}
|
||||
|
||||
/** A spawned architect in the roster resolves ARCHITECT, carrying its bound slot's name. */
|
||||
@Test
|
||||
void aSpawnedArchitectInTheRosterResolvesArchitectWithItsSlotName() {
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null, Map::of,
|
||||
boundMembers("architect:lead-designer", MemberRole.ARCHITECT),
|
||||
t -> "term_a".equals(t) ? MemberRole.ARCHITECT : null, Map::of)
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.ARCHITECT, p.role());
|
||||
assertEquals("lead-designer", p.name());
|
||||
assertEquals("term_a", p.terminal());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #424 regression: the roster only answers THAT a pane is a live spawned member; config
|
||||
* still decides WHAT that member's slot grants. A slot revoked after the bind must still demote
|
||||
* the session on its very next request, exactly as it would for a pane with no roster entry at
|
||||
* all — the roster's own ARCHITECT role must never be granted on its word alone.
|
||||
*
|
||||
* <p>{@code bind} refuses an unconfigured slot, so the revoked state can only be reached by
|
||||
* binding while the slot is configured and then swapping the config out from under it, the way
|
||||
* a live reload does.
|
||||
*/
|
||||
@Test
|
||||
void aRevokedArchitectSlotDemotesALiveSpawnedArchitectToWorker() {
|
||||
FleetConfig.Fleet configured = new FleetConfig.Fleet(Map.of(),
|
||||
Map.of("lead-designer", new FleetConfig.Slot("sonnet")), Map.of(), Map.of(), null);
|
||||
java.util.concurrent.atomic.AtomicReference<FleetConfig.Fleet> live =
|
||||
new java.util.concurrent.atomic.AtomicReference<>(configured);
|
||||
MemberRegistry members = MemberRegistry.live(live::get);
|
||||
assertTrue(members.bind("architect:lead-designer", "term_a"));
|
||||
|
||||
live.set(new FleetConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(), null)); // slot revoked
|
||||
|
||||
// Setup controls: the slot is really gone from config, but the occupancy is still there —
|
||||
// otherwise this test would pass for the wrong reason.
|
||||
assertNull(members.roleForSlot("architect:lead-designer"), "setup control: the slot must be gone from config");
|
||||
assertEquals("architect:lead-designer", members.snapshot().get("term_a"),
|
||||
"setup control: the binding itself must still be there");
|
||||
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null, Map::of,
|
||||
members, t -> "term_a".equals(t) ? MemberRole.ARCHITECT : null, Map::of)
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role(),
|
||||
"a revoked slot must demote a live spawned architect on its very next request");
|
||||
}
|
||||
|
||||
/** Criterion 3: a configured collaborator tab that is not a spawned member resolves COLLABORATOR. */
|
||||
@Test
|
||||
void aConfiguredCollaboratorTabResolvesToCollaboratorCarryingItsName() {
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null, Map::of,
|
||||
new MemberRegistry(null), t -> null, () -> Map.of("term_a", "ops"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.COLLABORATOR, p.role());
|
||||
assertEquals("ops", p.name());
|
||||
assertEquals("term_a", p.terminal());
|
||||
}
|
||||
|
||||
/** Regression: an empty collaborator registry leaves every pane at the unconfigured-pane floor. */
|
||||
@Test
|
||||
void anEmptyCollaboratorRegistryLeavesEveryPaneAtTheObserverFloor() {
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null, Map::of,
|
||||
new MemberRegistry(null), t -> null, Map::of)
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.OBSERVER, p.role());
|
||||
assertNull(p.name());
|
||||
}
|
||||
|
||||
@Test
|
||||
void describeNamesTheCollaborator() {
|
||||
assertEquals("collaborator:ops", Principal.collaborator("ops", "term_a", 1).describe());
|
||||
}
|
||||
|
||||
// ── fleetd #669 Unit D: knownLeadOrCollaborator() reads the same maps resolve() does ───────────
|
||||
|
||||
@Test
|
||||
void knownLeadOrCollaboratorIsTrueForALeadTerminal() {
|
||||
CallerResolver r = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
() -> Map.of("term_lead", "opus-5.0"), new MemberRegistry(null), t -> null, Map::of);
|
||||
|
||||
assertTrue(r.knownLeadOrCollaborator().test("term_lead"));
|
||||
assertFalse(r.knownLeadOrCollaborator().test("term_other"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void knownLeadOrCollaboratorIsTrueForACollaboratorTerminal() {
|
||||
CallerResolver r = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null, Map::of,
|
||||
new MemberRegistry(null), t -> null, () -> Map.of("term_collab", "ops"));
|
||||
|
||||
assertTrue(r.knownLeadOrCollaborator().test("term_collab"));
|
||||
assertFalse(r.knownLeadOrCollaborator().test("term_other"));
|
||||
}
|
||||
|
||||
// ── fleetd #705: narrowing the unconfigured-pane floor to OBSERVER ──────────────────────────
|
||||
|
||||
/**
|
||||
* The case this ticket exists for: a pane the resolver cannot place as a live spawned member,
|
||||
* a lead, a bound architect slot, or a configured collaborator must land on the narrow
|
||||
* {@link Role#OBSERVER} floor, never the {@link Role#WORKER} the old fallback granted.
|
||||
*
|
||||
* <p>The second assertion is the control the ticket requires: a terminal the roster DOES
|
||||
* recognise as a live spawned member must still resolve its own role. Without it, this test
|
||||
* would also pass if the fix accidentally turned every caller into an observer.
|
||||
*/
|
||||
@Test
|
||||
void anUnconfiguredPaneResolvesObserverButARegisteredMemberStillResolvesItsOwnRole() {
|
||||
Principal unconfigured = new CallerResolver(workerIdentity()).resolve("127.0.0.1", 42, null);
|
||||
assertEquals(Role.OBSERVER, unconfigured.role(),
|
||||
"a pane matching none of the configured or live-roster roles must fall to the "
|
||||
+ "floor, not WORKER");
|
||||
assertEquals("term_a", unconfigured.terminal());
|
||||
|
||||
Principal registered = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, new MemberRegistry(null),
|
||||
t -> "term_a".equals(t) ? MemberRole.DEV : null, Map::of)
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
assertEquals(Role.WORKER, registered.role(),
|
||||
"control: a live spawned member must keep resolving its own role, never the "
|
||||
+ "unconfigured-pane floor");
|
||||
}
|
||||
|
||||
@Test
|
||||
void describeNamesTheObserverByItsPane() {
|
||||
assertEquals("observer:term_a", Principal.observer("term_a", 1).describe());
|
||||
}
|
||||
|
||||
@Test
|
||||
void knownLeadOrCollaboratorIsFalseForASpawnedMembersTerminal() {
|
||||
// The exact scenario a collaborator's SEND must never reach: a live spawned member's own
|
||||
// terminal, which is neither a configured lead nor a configured collaborator.
|
||||
CallerResolver r = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null, Map::of,
|
||||
new MemberRegistry(null), t -> "term_a".equals(t) ? MemberRole.DEV : null, Map::of);
|
||||
|
||||
assertFalse(r.knownLeadOrCollaborator().test("term_a"));
|
||||
}
|
||||
|
||||
// ── fleetd #743: sendableObserverTarget() reads the same maps and functions resolve() does ────
|
||||
|
||||
/**
|
||||
* A terminal this resolver recognises as none of the privileged roles is exactly the one
|
||||
* {@code resolve} would itself hand back {@link Role#OBSERVER} for.
|
||||
*/
|
||||
@Test
|
||||
void sendableObserverTargetIsTrueForATerminalKnownAsNoOtherRole() {
|
||||
CallerResolver r = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
() -> Map.of("term_lead", "opus-5.0"), new MemberRegistry(null),
|
||||
t -> "term_worker".equals(t) ? MemberRole.DEV : null,
|
||||
() -> Map.of("term_collab", "ops"));
|
||||
|
||||
assertTrue(r.sendableObserverTarget().test("term_other"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendableObserverTargetIsFalseForALeadTerminal() {
|
||||
CallerResolver r = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
() -> Map.of("term_lead", "opus-5.0"), new MemberRegistry(null), t -> null, Map::of);
|
||||
|
||||
assertFalse(r.sendableObserverTarget().test("term_lead"),
|
||||
"a lead's own terminal must never be a sendable observer target");
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendableObserverTargetIsFalseForACollaboratorTerminal() {
|
||||
CallerResolver r = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null, Map::of,
|
||||
new MemberRegistry(null), t -> null, () -> Map.of("term_collab", "ops"));
|
||||
|
||||
assertFalse(r.sendableObserverTarget().test("term_collab"),
|
||||
"a collaborator's own terminal must never be a sendable observer target");
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendableObserverTargetIsFalseForALiveSpawnedMembersTerminal() {
|
||||
// Covers both a worker and an architect: spawnedMemberRole.apply(target) is non-null for
|
||||
// either, and resolve() never falls through to OBSERVER once it is.
|
||||
CallerResolver r = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null, Map::of,
|
||||
new MemberRegistry(null),
|
||||
t -> switch (t) {
|
||||
case "term_worker" -> MemberRole.DEV;
|
||||
case "term_architect" -> MemberRole.ARCHITECT;
|
||||
default -> null;
|
||||
}, Map::of);
|
||||
|
||||
assertFalse(r.sendableObserverTarget().test("term_worker"));
|
||||
assertFalse(r.sendableObserverTarget().test("term_architect"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendableObserverTargetIsFalseForABoundArchitectSlotWithNoLiveMember() {
|
||||
// The edge case resolve() itself carries: a terminal bound to a configured architect slot
|
||||
// but with no live spawned-member session yet.
|
||||
CallerResolver r = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null, Map::of,
|
||||
boundMembers("architect:lead-designer", MemberRole.ARCHITECT), t -> null, Map::of);
|
||||
|
||||
assertFalse(r.sendableObserverTarget().test("term_a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendableObserverTargetIsFalseForANullTarget() {
|
||||
CallerResolver r = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null, Map::of,
|
||||
new MemberRegistry(null), t -> null, Map::of);
|
||||
|
||||
assertFalse(r.sendableObserverTarget().test(null));
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -295,9 +295,10 @@ class MemberRegistryLiveTest {
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
Principal after = resolver.resolve("127.0.0.1", 42, null);
|
||||
assertEquals(Role.WORKER, after.role(),
|
||||
"removing the slot from config must demote the bound session to worker on its "
|
||||
+ "NEXT request — this is the ticket's whole point");
|
||||
assertEquals(Role.OBSERVER, after.role(),
|
||||
"removing the slot from config must demote the bound session on its NEXT request — "
|
||||
+ "this harness wires no live roster for term_a, so the demotion lands on "
|
||||
+ "the unconfigured-pane floor");
|
||||
assertEquals("term_a", after.terminal(), "same pane, same terminal — only the role changed");
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
|
||||
class PrincipalTest {
|
||||
|
||||
@Test
|
||||
void ownerKeyCoversEveryRole() {
|
||||
assertEquals("leader:opus", Principal.leader("opus", "term_lead", 1).ownerKey());
|
||||
assertNull(Principal.primary(2).ownerKey());
|
||||
assertEquals("worker:term_worker", Principal.worker("term_worker", 3).ownerKey());
|
||||
assertEquals("architect:term_arch", Principal.architect("opus", "term_arch", 4).ownerKey());
|
||||
assertEquals("collaborator:ops", Principal.collaborator("ops", "term_collab", 5).ownerKey());
|
||||
assertEquals("observer:term_observer", Principal.observer("term_observer", 6).ownerKey());
|
||||
assertEquals("anonymous", Principal.anonymous().ownerKey());
|
||||
}
|
||||
|
||||
@Test
|
||||
void rolePrefixesKeepLeadAndArchitectKeysDistinct() {
|
||||
String lead = Principal.leader("opus", "term_lead", 1).ownerKey();
|
||||
String architect = Principal.architect("design", "opus", 2).ownerKey();
|
||||
|
||||
assertEquals("leader:opus", lead);
|
||||
assertEquals("architect:opus", architect);
|
||||
assertNotEquals(lead, architect);
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user