Compare commits
349 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 9d653e86df | |||
| f6d1131d7a | |||
| a0505dc614 | |||
| d2f30f1654 | |||
| cd1f04cbb4 | |||
| 73137f198f | |||
| 4ffe49f3bb | |||
| efd9cdb983 | |||
| aabecce901 | |||
| 6754b4edbc | |||
| 787ae0ed7a | |||
| e3050efe8b | |||
| 11998cd626 | |||
| 8e5394f63f | |||
| 428a12af62 | |||
| a332dfdb2c | |||
| d0f4ae057b | |||
| 7b3beaa209 | |||
| b3b2bf3da6 | |||
| 8bb2aa0be4 | |||
| 4ce3149bfd | |||
| 1fc9e85bf1 | |||
| 38544d467c | |||
| 809b7d9b20 | |||
| 8cf7215d56 | |||
| 1ef93e57cc | |||
| cf0c9b9316 | |||
| 337dbd491e | |||
| 7cf6075b79 | |||
| fbdcd709c9 | |||
| 8d3f10d291 | |||
| 38f4fd64ee | |||
| 3fc39b981d | |||
| 70328ca0f8 | |||
| efab9b8c49 | |||
| 2374de28e4 | |||
| dd18bd1f38 | |||
| efeffb4ab7 | |||
| ed4f4b08ad | |||
| 28a1f3d6f5 | |||
| b12d70716b | |||
| 0f5985b419 | |||
| 6e06058b07 | |||
| e5f4fb81ab | |||
| d28ab0968b | |||
| 9a64d42599 | |||
| f4e0ca41e6 | |||
| 29a2f97c25 | |||
| d2db8c7dc9 | |||
| 95311c6e8e | |||
| 10ab58e4fc | |||
| 0ba597e394 | |||
| c468953963 | |||
| 25d53e6ef7 | |||
| a9a37af957 | |||
| 7d497aa423 | |||
| b205bcc2aa | |||
| 11eccc3a1b | |||
| 4939d40362 | |||
| e7a7711a4e | |||
| 1fd2cfa716 | |||
| 3e8e314656 | |||
| 20fc42b572 | |||
| f82073717a | |||
| 16b52fac6c | |||
| 13b6ae2628 | |||
| 7e838ba8b9 | |||
| c3e3554bde | |||
| b5bc5d4ab5 | |||
| d83972ede2 | |||
| 02c6909546 | |||
| b92a669ddc | |||
| bb29b001e4 | |||
| f0ff25221e | |||
| 780cb342ad | |||
| 5c563f02c8 | |||
| ad593c9bb9 | |||
| 736fd9cf4b | |||
| 05244a82b3 | |||
| ef4996a01e | |||
| edbd8d816a | |||
| 9425a9b696 | |||
| d0688c8a60 | |||
| 2c467c2553 | |||
| c6430d8edd | |||
| bfee23acc3 | |||
| 7dec74f1b4 | |||
| 482598e2a6 | |||
| 5f5d16fbd4 | |||
| 7df7985a16 | |||
| 724b35b46e | |||
| 2eb2d6112e | |||
| 7c458e8bf2 | |||
| 133f03e428 | |||
| 804279175d | |||
| 9dea289975 | |||
| cb4a6869b9 | |||
| 28b45d97e5 | |||
| 1a397e962e | |||
| 209e1231ea | |||
| 5d8b9d365c | |||
| a6aeda39e7 | |||
| 31b3c24caa | |||
| d105da978d | |||
| 5051a06443 | |||
| a134eccc57 | |||
| 6f275227d2 | |||
| 283ccf8423 | |||
| b96fba4a03 | |||
| 6d97d210b4 | |||
| 37b23cd704 | |||
| 367facf6a6 | |||
| 7f9a9c09f9 | |||
| e854957247 | |||
| b4b7cf5155 | |||
| cbb35ad947 | |||
| 656588f597 | |||
| e33377b2ca | |||
| 4b4a8688c2 | |||
| 7e48d4b86c | |||
| 41cc785534 | |||
| 60fa86a107 | |||
| f288cee2bb | |||
| a5d6ce1a37 | |||
| a52ca35d34 | |||
| 9417de1123 | |||
| 03d92be751 | |||
| 136bec8e28 | |||
| ae94d511d7 | |||
| 436b026696 | |||
| 905fa3a454 | |||
| 011ee80067 | |||
| 3fab743152 | |||
| 52eb9c2277 | |||
| 6794fd8200 | |||
| 9b5c1cdcff | |||
| 3b69e0103b | |||
| c97b1bba5a | |||
| a42253f597 | |||
| e69eafcc9f | |||
| a42b12440c | |||
| a06426c33c | |||
| 28ea0de575 | |||
| 91792e11fc | |||
| 6539efe9fa | |||
| 68397f78d5 | |||
| af901ff1d2 | |||
| dac5f88812 | |||
| 2e663e5968 | |||
| 8c14ed2846 | |||
| 141ae3b04d | |||
| faefea14c4 | |||
| ea6896f2ef | |||
| d7f94cafa2 | |||
| 096f08c866 | |||
| 4eb720029c | |||
| 0db6d31dc2 | |||
| 158a2a84b5 | |||
| 41ebc9cf69 | |||
| 619769bb52 | |||
| 180c953c42 | |||
| 4af919af0c | |||
| 09c37061c1 | |||
| 2be287ea03 | |||
| c3b0406826 | |||
| e76fa1660b | |||
| 9dca376604 | |||
| 91be5079d6 | |||
| 26f198675c | |||
| 72f46d7c0c | |||
| 640f4d5f23 | |||
| 6edeb70bc4 | |||
| bc49d87cb8 | |||
| 14b169c410 | |||
| 2b52324d9a | |||
| b6147a39f6 | |||
| dbfc34cb6d | |||
| a20cb96730 | |||
| cda1a6a917 | |||
| 608e4496be | |||
| fa61dc587c | |||
| 526e3b7459 | |||
| b6b006c651 | |||
| 63eec8a0da | |||
| cbe872b538 | |||
| 7d9a807243 | |||
| 8915e40c7d | |||
| 3f7bc3815e | |||
| 6cb31a10e4 | |||
| 8368a274a0 | |||
| 17127efb88 | |||
| 203f034528 | |||
| 9ee16f5b85 | |||
| 388ef5a3c3 | |||
| be6c45ff78 | |||
| e99cb70a8b | |||
| 987ccef4c7 | |||
| 076cc43f7b | |||
| bad47a8444 | |||
| 955b9ea013 | |||
| 89cb8ff79b | |||
| fa62e9906d | |||
| d7390ccd37 | |||
| aa517ae0ec | |||
| 60496831c2 | |||
| 9a992d0f70 | |||
| 3762aca307 | |||
| 4e27bde2d7 | |||
| b85d9b0e46 | |||
| 856dfc6318 | |||
| 81c1d8e91c | |||
| d345b14e43 | |||
| 9640deeffc | |||
| ad3d81941f | |||
| 8b986a52e0 | |||
| f336bcef39 | |||
| 3ed7bfca67 | |||
| f5c6a0e4fc | |||
| a7aee5b982 | |||
| bf895616a5 | |||
| b874afb0af | |||
| 6eb34a654f | |||
| 7084d99b89 | |||
| ae7845c375 | |||
| 61115f6f61 | |||
| 4b9ebda1b3 | |||
| 1e68d7ee39 | |||
| 6cccd458d4 | |||
| 42820fbe75 | |||
| d91ff886da | |||
| 386e760a5c | |||
| 2e349139e9 | |||
| a639969a9a | |||
| 17c3a69c57 | |||
| 49a5875586 | |||
| 634d33b50b | |||
| 1db79bcaa9 | |||
| 4507bc5a70 | |||
| d7239ed23b | |||
| dfeb9340b4 | |||
| 1513d4f260 | |||
| 4ca7d72303 | |||
| c0545d003d | |||
| 275ac0d251 | |||
| c1e06c9e12 | |||
| 204da67d66 | |||
| b091c51eee | |||
| 84d631b030 | |||
| 3f8c38fc54 | |||
| a4dbc8f8b7 | |||
| ed2fd6646a | |||
| 5441a2b321 | |||
| 384867dfa3 | |||
| f40c19ecf0 | |||
| 034e17bb32 | |||
| 68b428c484 | |||
| e20ccab1eb | |||
| d83821bbce | |||
| ba2f4d16f8 | |||
| db4c98ac60 | |||
| 8f80d267a0 | |||
| 738d34a609 | |||
| a46e4058ac | |||
| f4f5f3106e | |||
| 8d79d229ff | |||
| cb64bc8157 | |||
| ed28b51f12 | |||
| 4f9aba40e7 | |||
| dab697fae0 | |||
| bfac14108f | |||
| 7a3b2bb7ee | |||
| f188947750 | |||
| f606fccf7f | |||
| 735b6af976 | |||
| 8b4320ed24 | |||
| 351ee1ea6d | |||
| d4a51c6274 | |||
| 26f380a00b | |||
| b8182c96c2 | |||
| 3da44eed63 | |||
| 93a9ed3f83 | |||
| 0b032f5a1a | |||
| fad99c4c5e | |||
| 87871eaefb | |||
| a476a14f1c | |||
| 5eb4267a4a | |||
| 7611b69667 | |||
| cc302fe4af | |||
| 4a8a780274 | |||
| 343ce0f4c0 | |||
| cec3e191d4 | |||
| 1850a5f324 | |||
| e4eb3dbed4 | |||
| 57cd96f5e6 | |||
| 202e37e3b3 | |||
| 90253f832d | |||
| f1640f5dcc | |||
| c7903c1efe | |||
| 7d711942fe | |||
| bec87f987c | |||
| 7f8a8829f9 | |||
| af9589783e | |||
| 190436c9cf | |||
| 8ea5c2bb1f | |||
| 8335b12562 | |||
| d25c863118 | |||
| 7c34e8f4f9 | |||
| be07ed2033 | |||
| a6415f3e52 | |||
| bb6fc9e0d7 | |||
| 3366590dbe | |||
| 01adc841fa | |||
| 08771e270b | |||
| b5843ab43f | |||
| d8985719eb | |||
| 5b1e13ca3d | |||
| c89a375e5d | |||
| 37dcefa834 | |||
| 6a7342b1f0 | |||
| a5ad7c6561 | |||
| c71ac231e5 | |||
| 33720c42b3 | |||
| 3833d8e52b | |||
| b37def9238 | |||
| 8f02576df6 | |||
| 525bc1c5f4 | |||
| d59ece6dec | |||
| 32408d1e64 | |||
| 6e23bf8309 | |||
| aa4c0b84c3 | |||
| 40c593cd09 | |||
| 979adf82eb | |||
| 36870836aa | |||
| 136312fb11 | |||
| 4f28da62a3 | |||
| 708f1795ad | |||
| 599419f9e6 | |||
| ac351ee1de | |||
| 274afafde6 | |||
| 8f59019305 | |||
| e966cbadf9 | |||
| b17f37a683 | |||
| c87cc25aa6 | |||
| 3fb331145a | |||
| dcd505286f | |||
| 60fa958da2 | |||
| 9950361bc9 | |||
| 9f74b6619a | |||
| 2a95da2ff4 |
@@ -0,0 +1,23 @@
|
|||||||
|
---
|
||||||
|
name: hunter
|
||||||
|
description: Sweep one assigned scope for defects and report ranked findings without changes.
|
||||||
|
---
|
||||||
|
|
||||||
|
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||||
|
|
||||||
|
You sweep the assigned package or scope for real defects. Read the full assigned scope before you
|
||||||
|
judge it. Report several ranked findings when the evidence supports them. Change nothing: do not
|
||||||
|
edit code, commit, push, or open a pull request.
|
||||||
|
|
||||||
|
You may run the build or tests to check a finding. Read the complete output and report the real
|
||||||
|
result. Do not hide failures with a pipe. State only checks you actually ran. The primary's IDE
|
||||||
|
tools are not yours. A mounted forge tool may use a blocked credential and fail by design.
|
||||||
|
|
||||||
|
Do only the assigned scope. Note anything outside it in one line and do not investigate it further.
|
||||||
|
Use `fleet_ask{question}` only when a decision belongs to the lead, such as an unclear requirement
|
||||||
|
or two defensible fixes. Do not ask about something you can decide by reading more code.
|
||||||
|
|
||||||
|
Your handoff must name the files you read, each ranked finding or `NO FINDINGS`, the checks you ran,
|
||||||
|
and any caveat for review.
|
||||||
|
|
||||||
|
The launcher provides the required bridge reply instructions for every member.
|
||||||
@@ -59,7 +59,7 @@ as `matches HEAD`, `drift`, or `unknown`; do not turn an unclear timestamp into
|
|||||||
Report the process identifier (PID) and uptime too:
|
Report the process identifier (PID) and uptime too:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
PIDS="$(pgrep -f 'target/fleetd.jar' || true)"
|
PIDS="$(pgrep -f 'fleetd.jar' || true)"
|
||||||
if [ -z "$PIDS" ]; then
|
if [ -z "$PIDS" ]; then
|
||||||
printf '%s\n' 'fleetd: not running'
|
printf '%s\n' 'fleetd: not running'
|
||||||
else
|
else
|
||||||
|
|||||||
@@ -141,6 +141,14 @@ refused.**
|
|||||||
own workspace before it hands it to you. The daemon and your pane can run in different
|
own workspace before it hands it to you. The daemon and your pane can run in different
|
||||||
directories, so a path you resolve yourself can point at a different file from the one the daemon
|
directories, so a path you resolve yourself can point at a different file from the one the daemon
|
||||||
will check.
|
will check.
|
||||||
|
|
||||||
|
**Check that the path is ignored by git before you write to it (#491).** A relative
|
||||||
|
`handoverPath` resolves inside YOUR workspace, which is usually a repository — and usually not
|
||||||
|
the `fleetd` one, so an ignore rule added to `fleetd` does not protect it. Run
|
||||||
|
`grep -n handover <your workspace>/.gitignore`. No output means the file you are about to write
|
||||||
|
will show up as untracked content in that repo. The file is a snapshot of live state and must
|
||||||
|
never be committed, so tell the operator rather than committing it or silently editing their
|
||||||
|
`.gitignore`.
|
||||||
2. **Write the handover file at that path**, following sections 1–10 above.
|
2. **Write the handover file at that path**, following sections 1–10 above.
|
||||||
3. **Ask the operator, then `fleet_handover{action: "confirm", token, operatorConfirmed: true}`.**
|
3. **Ask the operator, then `fleet_handover{action: "confirm", token, operatorConfirmed: true}`.**
|
||||||
|
|
||||||
@@ -151,23 +159,102 @@ fails.
|
|||||||
|
|
||||||
`{action: "cancel", token}` drops a pending request without rolling.
|
`{action: "cancel", token}` drops a pending request without rolling.
|
||||||
|
|
||||||
|
**A change is coming: the roll will restart the process instead of sending `/clear` (fleetd #726
|
||||||
|
unit 2, written 2026-10-04).**
|
||||||
|
|
||||||
|
Today a roll types `/clear` into your pane. Your `claude` process keeps running, so a newer CLI on
|
||||||
|
disk is never loaded. Unit 2 replaces that: the daemon ends the old pane, launches a fresh one,
|
||||||
|
waits for the new terminal to be recognised as a lead, and only then sends the bootstrap text.
|
||||||
|
|
||||||
|
**Everything below about `/clear` is accurate while the old jar is running.** Unit 2 was not merged
|
||||||
|
when this note was written, and a merge is not a deployment.
|
||||||
|
|
||||||
|
**How to tell which one is live: read your own tool list.** If `fleet_handover`'s description says
|
||||||
|
it will "clear your pane", the daemon is serving the old behaviour. If it names a restart, the new
|
||||||
|
behaviour is live. The description comes from the running daemon, so it cannot disagree with the
|
||||||
|
code that is actually loaded.
|
||||||
|
|
||||||
|
Two things change for you once it is live. The `status` outcomes are different: three new failures
|
||||||
|
replace the `/clear` ones. And the "never observed as WORKING after 8 consecutive IDLE/DONE polls"
|
||||||
|
warning described below can no longer appear, because that wait is deleted — so if you still see
|
||||||
|
it, the old jar is running. `TURN_NEVER_SETTLED` does not change, and still means nothing was
|
||||||
|
touched.
|
||||||
|
|
||||||
|
**Delete this note and rewrite the `/clear` paragraphs once the new jar is live.**
|
||||||
|
|
||||||
**Things that will surprise you:**
|
**Things that will surprise you:**
|
||||||
|
|
||||||
- **`accepted` does not mean your pane has been cleared.** It means every gate passed and the roll
|
- **`accepted` does not mean your pane has been cleared.** It means every gate passed and the roll
|
||||||
is scheduled to run once your current turn ends. Say your goodbye in the same turn — you will not
|
is scheduled to run once your current turn ends. Say your goodbye in the same turn — you will not
|
||||||
get another one.
|
get another one.
|
||||||
|
- **If you are still running after that turn, the roll did not happen.** A roll that works clears
|
||||||
|
you, so surviving your own goodbye is itself the signal that it refused. Check with
|
||||||
|
`fleet_handover{action: "status", token}`, using the token you confirmed. `TURN_NEVER_SETTLED`
|
||||||
|
means your turn ran past `leadRollover.turnSettleSeconds` and **no `/clear` was ever sent**: your
|
||||||
|
context is intact and nothing was lost. Open a fresh request and retry. Never assume the roll
|
||||||
|
succeeded because `confirm` answered `accepted` — by the time it refuses, there is no caller left
|
||||||
|
to tell, so this check is the only thing that closes that gap.
|
||||||
- **There is no terminal or session parameter, on purpose.** The pane is always your own, resolved
|
- **There is no terminal or session parameter, on purpose.** The pane is always your own, resolved
|
||||||
from your connection, so you can only ever roll yourself.
|
from your connection, so you can only ever roll yourself.
|
||||||
- **`operatorConfirmed` is your report of what a human told you.** Do not pass `true` because you
|
- **`operatorConfirmed` is your report of what a human told you.** Do not pass `true` because you
|
||||||
are confident. Ask, wait for the answer, then pass what they said. `requireOperatorConfirm`
|
are confident. Ask, wait for the answer, then pass what they said.
|
||||||
defaults to `true` and this is the only thing standing between a judgement call and a wiped
|
|
||||||
session.
|
- **Whether you must ask at all depends on `leadRollover.requireOperatorConfirm`. Check it; do not
|
||||||
|
assume.** The default is `true` (`FleetConfig.java:1426`), and then `confirm` refuses unless you
|
||||||
|
also pass `operatorConfirmed: true`. **This host set it to `false` on 2026-09-22**, on the
|
||||||
|
operator's explicit grant, because they do not want to approve routine context rolls. Where it is
|
||||||
|
`false`, the three handover-file checks are the whole gate: the file must exist, be fresher than
|
||||||
|
`maxDocAgeSeconds`, and have been modified after the open request.
|
||||||
|
|
||||||
|
Read the live value rather than trusting this line:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
grep -A1 'requireOperatorConfirm' fleetd/fleetd.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
No match means the key is unset, so the default `true` applies and you must ask. The key is
|
||||||
|
**deferred, not hot** — it is read once at boot, so an edit does nothing until the daemon is
|
||||||
|
redeployed.
|
||||||
|
|
||||||
|
**Until fleetd #621 merges, the nudge text will tell you to ask the operator even where the
|
||||||
|
daemon no longer requires it.** `LeadHeartbeatLoop.contextNotice()` hardcodes "ask the operator"
|
||||||
|
and takes no config, so it cannot know. Trust the config value over the nudge text. Once #621 is
|
||||||
|
merged and deployed, the nudge matches the config and this warning can be deleted.
|
||||||
|
|
||||||
- **The roll can still refuse after `confirm` returns**, and by then there is no caller to tell.
|
- **The roll can still refuse after `confirm` returns**, and by then there is no caller to tell.
|
||||||
Those outcomes are logged only, as `lead-rollover:` lines in the daemon log.
|
Those outcomes are logged only, as `lead-rollover:` lines in the daemon log.
|
||||||
- **If the bootstrap prompt never lands, your context is gone and no fresh session starts.** This
|
- **The bootstrap prompt works end to end. Measured 2026-09-22, re-measured 2026-10-04.** This used
|
||||||
has not yet been proven end-to-end (see fleetd #480). The recovery is the manual path: the file
|
to say the fix was unproven (fleetd #489) and told you to expect a failure. That is no longer
|
||||||
is already written, so the operator starts a session and points it at the file. That is why you
|
true. On 2026-09-22 the daemon log held four `lead-rollover: rolled` lines. On 2026-10-04 it holds
|
||||||
write the file before you confirm, and never the other way round.
|
**20**, against a control of 86 `lead-rollover:` lines. Each roll cleared the old lead and started
|
||||||
|
a fresh session against the handover file, with the configured `bootstrapText` arriving as its
|
||||||
|
first message. No context was lost. The old `Unknown command: /clearFresh` failure from 2026-09-12
|
||||||
|
does not appear in the log at all.
|
||||||
|
|
||||||
|
19 of the 20 carry an `elapsedMs`: median 16507 ms, maximum 48261 ms, and two above 45000 ms. That
|
||||||
|
figure times the **whole** roll, and the wait for your own turn to end dominates it, so do not
|
||||||
|
read it as the cost of the clear. Expect a roll to take tens of seconds, and do not treat a slow
|
||||||
|
one as a failed one. Re-measure all of these with:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
grep -c "lead-rollover: rolled" fleetd/fleetd.out # successful rolls
|
||||||
|
grep -c "lead-rollover:" fleetd/fleetd.out # positive control: must be larger
|
||||||
|
grep -c "Unknown command" fleetd/fleetd.out # the old failure: expect 0
|
||||||
|
```
|
||||||
|
|
||||||
|
Run the control line too. A broken pattern returns a clean `0` that reads exactly like good news.
|
||||||
|
If the first number stops growing across rolls, or `Unknown command` returns anything above 0,
|
||||||
|
the bootstrap has regressed and this paragraph is stale again.
|
||||||
|
|
||||||
|
**You still write the file before you confirm, and never the other way round.** That order is not
|
||||||
|
about the bootstrap being unreliable. It is what the daemon checks: the handover file must have
|
||||||
|
been modified *after* the open request, or `confirm` refuses it as stale.
|
||||||
|
|
||||||
|
- **One warning in the log is normal and is not a failure.** Every one of the three rolls above also
|
||||||
|
logged `/clear on term_… was never observed as WORKING after 8 consecutive IDLE/DONE polls —
|
||||||
|
releasing rather than wedging the roll`. The daemon could not see the pane go WORKING after
|
||||||
|
`/clear`, so it released instead of hanging. The roll then succeeded anyway. That is the safe
|
||||||
|
branch behaving correctly. Do not report it as a broken roll.
|
||||||
|
|
||||||
## Writing style
|
## Writing style
|
||||||
|
|
||||||
|
|||||||
@@ -22,13 +22,22 @@ scripts/redeploy-fleetd.sh --yes # skip the drain prompt (fleet already chec
|
|||||||
scripts/redeploy-fleetd.sh --no-build # restart the jar already on disk
|
scripts/redeploy-fleetd.sh --no-build # restart the jar already on disk
|
||||||
```
|
```
|
||||||
|
|
||||||
`--no-build` skips the build and restarts whatever jar is at `fleetd/target/fleetd.jar`. Use it only
|
`--no-build` skips the build and restarts whatever jar is at `fleetd/run/fleetd.jar` — the runtime
|
||||||
when you just built and nothing changed since. It gives up the protection in the next paragraph: no
|
path, not Maven's output path. Use it only when you just built and nothing changed since. It gives
|
||||||
build runs, so a stale or missing jar is not caught early. The script still checks the file is there
|
up the protection in the next paragraph: no build runs, so a stale or missing jar is not caught
|
||||||
and dies with `no jar at … — run without --no-build` if it is not, but it cannot tell you the jar is
|
early. The script still checks the file is there and dies with `no jar at … — run without
|
||||||
old. A `mvn clean` in the tree deletes that jar while the daemon keeps running on it, and nothing
|
--no-build` if it is not, but it cannot tell you the jar is old.
|
||||||
degrades until the next restart. Run `--check` first: it prints the jar's hash and its modification
|
|
||||||
time, so you can see for yourself whether the jar is missing or older than the code you mean to ship.
|
The daemon runs from `fleetd/run/fleetd.jar`, not from `fleetd/target/fleetd.jar` where Maven
|
||||||
|
writes its output (fleetd #664). That split is what makes a bare `mvn install`/`mvn clean` in the
|
||||||
|
main clone harmless now: neither can reach the file the running daemon holds open, because that
|
||||||
|
file no longer lives under `target/` at all. Verify a merge by building in a throwaway git
|
||||||
|
worktree anyway — a build still produces nothing the fleet runs until this script's own `mv` of
|
||||||
|
`target/fleetd.jar` onto `run/fleetd.jar`, performed only after the old daemon is confirmed gone.
|
||||||
|
Let only `scripts/redeploy-fleetd.sh` touch `fleetd/run/fleetd.jar`. Run `--check` first: it prints
|
||||||
|
the BUILT jar (`target/fleetd.jar`) and the RUNNING jar (`run/fleetd.jar`) as two separately
|
||||||
|
labelled hash-and-mtime facts, so a mismatch between them — a build sitting unswapped, or a stale
|
||||||
|
runtime jar — is visible before you decide anything.
|
||||||
|
|
||||||
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
||||||
old process to exit rather than assuming; it polls `/healthz`; and it anchors its log checks to a
|
old process to exit rather than assuming; it polls `/healthz`; and it anchors its log checks to a
|
||||||
|
|||||||
@@ -39,6 +39,10 @@ jobs:
|
|||||||
# that runs does so against the fake UDS herdr and fake ccs/claude stubs.
|
# that runs does so against the fake UDS herdr and fake ccs/claude stubs.
|
||||||
run: mvn -B clean install
|
run: mvn -B clean install
|
||||||
|
|
||||||
|
- name: javadoc reference lint
|
||||||
|
working-directory: fleetd
|
||||||
|
run: mvn -B -DskipTests javadoc:javadoc -Ddoclint=reference
|
||||||
|
|
||||||
# Deliberately NOT actions/upload-artifact: this Gitea instance presents as GHES, and
|
# Deliberately NOT actions/upload-artifact: this Gitea instance presents as GHES, and
|
||||||
# @actions/artifact v2+ (i.e. upload-artifact@v4) refuses to run there —
|
# @actions/artifact v2+ (i.e. upload-artifact@v4) refuses to run there —
|
||||||
# "GHESNotSupportedError ... not currently supported on GHES", which red-Xes an otherwise
|
# "GHESNotSupportedError ... not currently supported on GHES", which red-Xes an otherwise
|
||||||
@@ -54,6 +58,23 @@ jobs:
|
|||||||
done
|
done
|
||||||
exit 0
|
exit 0
|
||||||
|
|
||||||
|
# fleetd #550 — nothing ran scripts/test-redeploy-fleetd.sh in CI before this, on any platform,
|
||||||
|
# so it had run only on macOS by hand and two Linux-only bugs (this issue's items 1 and 2)
|
||||||
|
# survived undetected: shasum is a macOS-only tool (it ships with Perl; GNU coreutils, i.e. every
|
||||||
|
# mainstream Linux distro including this runner's ubuntu-latest, does not have it and ships
|
||||||
|
# sha256sum instead). The gate here is the step's own exit code, nothing else: a `run:` step in
|
||||||
|
# Gitea/GitHub Actions already fails the job on a non-zero exit with no extra scripting needed,
|
||||||
|
# so this deliberately does NOT grep the output for a `FAIL:` count. That is the #550 item-2
|
||||||
|
# lesson one level up — a suite that dies before it runs a single test prints zero FAIL lines,
|
||||||
|
# which is exactly what a clean pass also prints, so counting FAIL lines can never be the gate.
|
||||||
|
shell-tests:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: redeploy-fleetd.sh shell suite
|
||||||
|
run: bash scripts/test-redeploy-fleetd.sh
|
||||||
|
|
||||||
# CB-521 — actually run the AMQP contract test in CI, against a REAL broker. The broker is a
|
# CB-521 — actually run the AMQP contract test in CI, against a REAL broker. The broker is a
|
||||||
# RabbitMQ SERVICE CONTAINER, not Testcontainers-with-Docker: the runner image has no Docker, so
|
# RabbitMQ SERVICE CONTAINER, not Testcontainers-with-Docker: the runner image has no Docker, so
|
||||||
# AmqpReplyInboxContractTest reads AMQP_URI (set below to the service's network alias) and binds
|
# AmqpReplyInboxContractTest reads AMQP_URI (set below to the service's network alias) and binds
|
||||||
|
|||||||
@@ -20,6 +20,13 @@ fleetd.out
|
|||||||
fleetd/fleetd.out
|
fleetd/fleetd.out
|
||||||
logs/
|
logs/
|
||||||
|
|
||||||
|
# fleetd #635 follow-up — scripts/config-edit.sh's backup directory. No leading slash, so this is
|
||||||
|
# ignored at every depth: the real one lives under fleetd/ (also named in fleetd/.gitignore, next
|
||||||
|
# to the config it backs up), and scripts/test-config-edit.sh's own throwaway fixtures build one
|
||||||
|
# under the repo root while the suite runs. --config can point anywhere, so the directory name is
|
||||||
|
# ignored everywhere rather than only where the live daemon happens to use it.
|
||||||
|
.config-backups/
|
||||||
|
|
||||||
# fleetd #480: the lead rollover handover file. `leadRollover.handoverPath` points here, and the
|
# fleetd #480: the lead rollover handover file. `leadRollover.handoverPath` points here, and the
|
||||||
# outgoing lead rewrites it on every rollover. It is a snapshot of one moment's live state —
|
# outgoing lead rewrites it on every rollover. It is a snapshot of one moment's live state —
|
||||||
# unpushed branches, running builds, open questions — so it is stale the moment it is written and
|
# unpushed branches, running builds, open questions — so it is stale the moment it is written and
|
||||||
|
|||||||
@@ -27,23 +27,34 @@ through its `fleet_*` tools. No session addresses a peer, a broker, or the netwo
|
|||||||
**Every role reads this file.** A member runs in a git worktree of this same repo, so it inherits
|
**Every role reads this file.** A member runs in a git worktree of this same repo, so it inherits
|
||||||
this `CLAUDE.md` verbatim, and every rule below is role-conditional.
|
this `CLAUDE.md` verbatim, and every rule below is role-conditional.
|
||||||
|
|
||||||
**Call `fleet_whoami`.** It returns `primary`, `worker`, or `architect`, resolved by the daemon from
|
**Call `fleet_whoami`.** It returns `primary`, `worker`, `architect`, `collaborator`, or `observer`,
|
||||||
your connection — unforgeable, and the same resolution its authorization gate uses. A worker also
|
resolved by the daemon from your connection — unforgeable, and the same resolution its authorization
|
||||||
carries its `sessionId`, `profile`, `worktree` and `branch`; an architect carries the slot name it
|
gate uses. A worker also carries its `sessionId`, `profile`, `worktree` and `branch`; an architect
|
||||||
was bound to. Don't infer what you can ask.
|
carries the slot name it was bound to; a collaborator carries its registry name and its own
|
||||||
|
`sessionId`, and **no `leader` key** — a collaborator is a named peer, not a primary. An **observer**
|
||||||
|
carries only its own `sessionId`: a pane the daemon could not place as any of the above, authorized
|
||||||
|
to `READ`/`METRICS` and to `REPLY`/`ASK` on its own pane and nothing more — never `SEND`, never a
|
||||||
|
ticket. Don't infer what you can ask.
|
||||||
|
|
||||||
Only if that call is unavailable, fall back to these — each is one-way, so keep reading until one
|
Only if that call is unavailable, fall back to these — each is one-way, so keep reading until one
|
||||||
fires: the reply charter in your system prompt (*"You are a spawned member in the
|
fires: the reply charter in your system prompt (*"You are a spawned member in the
|
||||||
claude-bridge fleet"*) ⇒ **spawned member**; fleet tools prefixed `mcp__fleet__*` ⇒ **spawned
|
claude-bridge fleet"*) ⇒ **spawned member**; fleet tools prefixed `mcp__fleet__*` ⇒ **spawned
|
||||||
member** (the launcher fixes that mount name; a primary's mount is named by whoever wrote its
|
member** (the launcher fixes that mount name; a primary's mount is named by whoever wrote its
|
||||||
`.mcp.json`, so it varies — and a member spawned before CB-632 still says `mcp__bridge__*`); `ANTHROPIC_BASE_URL` set ⇒ **spawned member** (Claude-model members run
|
`.mcp.json`, so it varies — and a member spawned before CB-632 still says `mcp__bridge__*`); `ANTHROPIC_BASE_URL` set ⇒ **spawned member** (Claude-model members run
|
||||||
on a clean env, so its *absence* proves nothing). None of these separate a worker from an architect —
|
on a clean env, so its *absence* proves nothing). None of these separate a worker from an architect,
|
||||||
only `fleet_whoami` does. **Still unsure ⇒ act as a worker**, the most restricted member role. The
|
or a worker from an **observer** — an observer is just as unspawned as a collaborator and carries
|
||||||
two mistakes are not symmetric: a primary acting as a worker is refused by the authorization gate —
|
none of these signals either, so only `fleet_whoami` tells the two apart. **And none of them fires
|
||||||
loud and self-correcting — while a member acting as the primary ends its turn with no `fleet_reply`,
|
for a collaborator at all**: every signal in the ladder detects a *spawned* member, while a
|
||||||
|
collaborator is a tab a person opened by hand, so it has no charter, no fixed mount name and a
|
||||||
|
normal environment. A collaborator — or an observer — that cannot call `fleet_whoami` therefore falls
|
||||||
|
to the line below and acts as a worker. That is the safe direction — it under-privileges, and the
|
||||||
|
refusals are loud — but it means a collaborator or an observer has no way to learn what it is except
|
||||||
|
by asking. **Still unsure ⇒ act as a worker**, the most restricted member role this ladder can name.
|
||||||
|
The two mistakes are not symmetric: a primary acting as a worker is refused by the authorization gate
|
||||||
|
— loud and self-correcting — while a member acting as the primary ends its turn with no `fleet_reply`,
|
||||||
and the sender silently receives nothing. Fail toward the recoverable error.
|
and the sender silently receives nothing. Fail toward the recoverable error.
|
||||||
|
|
||||||
### Invariants — both roles, no exceptions
|
### Invariants — every role, no exceptions
|
||||||
|
|
||||||
1. **Never set, export, or forward `ANTHROPIC_BASE_URL`** (or `ANTHROPIC_AUTH_TOKEN`). The primary
|
1. **Never set, export, or forward `ANTHROPIC_BASE_URL`** (or `ANTHROPIC_AUTH_TOKEN`). The primary
|
||||||
stays on subscription; only the bridge puts a member off it, at spawn. Mounting the bridge must
|
stays on subscription; only the bridge puts a member off it, at spawn. Mounting the bridge must
|
||||||
@@ -51,9 +62,10 @@ and the sender silently receives nothing. Fail toward the recoverable error.
|
|||||||
2. **The bridge is the only channel.** Text you print in your terminal reaches nobody — the other
|
2. **The bridge is the only channel.** Text you print in your terminal reaches nobody — the other
|
||||||
side cannot see your screen. An answer that isn't in a `fleet_*` call is silently discarded.
|
side cannot see your screen. An answer that isn't in a `fleet_*` call is silently discarded.
|
||||||
3. **Identity comes from the connection, never an argument.** Workers never pass a target; you
|
3. **Identity comes from the connection, never an argument.** Workers never pass a target; you
|
||||||
cannot act as another session. Spawn/stop/drain are lead-only; **send is lead or architect**;
|
cannot act as another session. Spawn/stop/drain are lead-only; **send is lead, architect, or
|
||||||
reply/ask are only-as-itself — any peer may answer for its own pane, and for no other. A call
|
collaborator** — and a collaborator may send only to a lead or another collaborator, never to a
|
||||||
outside your role is refused, not queued.
|
spawned member's terminal; reply/ask are only-as-itself — any peer may answer for its own pane,
|
||||||
|
and for no other. A call outside your role is refused, not queued.
|
||||||
4. **Delivery is status-gated: one message per turn.** Don't busy-poll a peer's terminal and don't
|
4. **Delivery is status-gated: one message per turn.** Don't busy-poll a peer's terminal and don't
|
||||||
re-send because a call looks slow — the bridge delivers when the peer is `idle`, `blocked` or
|
re-send because a call looks slow — the bridge delivers when the peer is `idle`, `blocked` or
|
||||||
`done`. A spawned member must **also** have mounted the bridge MCP: until it has, it is not
|
`done`. A spawned member must **also** have mounted the bridge MCP: until it has, it is not
|
||||||
@@ -93,11 +105,27 @@ below are the procedure — run them in order, every task, not only the big ones
|
|||||||
5. **Collect** — `fleet_poll{ticket}` → `fleet_ack{target, msgId}`. Answer a worker's `fleet_ask`
|
5. **Collect** — `fleet_poll{ticket}` → `fleet_ack{target, msgId}`. Answer a worker's `fleet_ask`
|
||||||
with `fleet_send{turnId, content}` — **not** `sessionId`. A worker gone quiet is diagnosed with
|
with `fleet_send{turnId, content}` — **not** `sessionId`. A worker gone quiet is diagnosed with
|
||||||
`fleet_status`, never by reading its terminal; it also reports an open question and the `turnId`
|
`fleet_status`, never by reading its terminal; it also reports an open question and the `turnId`
|
||||||
that answers it. **A worker's ask waits ~55 seconds, and no nudge makes that longer** — so never
|
that answers it — but **only to the caller that created that delegation**, so a question raised
|
||||||
|
under an architect's brief is invisible to you, and seeing none does not mean there is none.
|
||||||
|
**Only that same creator can answer it.** A `turnId` you came by any other way is refused, so an
|
||||||
|
architect's worker waits for that architect and not for you.
|
||||||
|
**A worker's ask waits ~55 seconds, and no nudge makes that longer** — so never
|
||||||
brief a worker to "ask me". Decide before you delegate, or give it an explicit default.
|
brief a worker to "ask me". Decide before you delegate, or give it an explicit default.
|
||||||
|
**A correction cannot reach a busy member.** A `fleet_send` to a working member is *accepted* and
|
||||||
|
returns a ticket, and is then never delivered — measured here three times in one session, and the
|
||||||
|
member was released still executing a brief that had been retracted twice. The receipt is true and
|
||||||
|
it is a fact about the *mailbox*; what you needed was a fact about the *pane*. **A push delivery
|
||||||
|
needs the recipient free at send time; a pull channel needs only that they look before acting.** So
|
||||||
|
put every correction on the **ticket**, which they can read whenever they look, and send as the
|
||||||
|
notification. That obliges you, not them: **all corrections go to the ticket, and the brief is
|
||||||
|
write-once.** The member cannot check which source is newer — it just always prefers the ticket —
|
||||||
|
so the day you revise a brief in place instead of commenting, it obeys your rule and does the wrong
|
||||||
|
thing. A *first* brief for a unit not yet running is not a correction, and may be the whole spec.
|
||||||
6. **Verify yourself.** Re-run the build and the checks. A worker cannot run your IDE tooling, any
|
6. **Verify yourself.** Re-run the build and the checks. A worker cannot run your IDE tooling, any
|
||||||
forge tools it appears to have hold a blocked credential and fail, and a piped command
|
forge MCP server it appears to have holds a blocked credential and fails every call, and a piped
|
||||||
(`… | tail`) hides failures behind a zero exit — never promote a worker's "clean" to a fact.
|
command (`… | tail`) hides failures behind a zero exit — never promote a worker's "clean" to a
|
||||||
|
fact. Its injected repo-scoped `GITEA_TOKEN` is a different credential and does work, so a worker
|
||||||
|
reporting that it opened its own PR is reporting something it really can do.
|
||||||
7. **Review — fan out.** Spawn reviewers against the diff, one per dimension or per file, with
|
7. **Review — fan out.** Spawn reviewers against the diff, one per dimension or per file, with
|
||||||
`wait:false`. Never the implementer of the scope it reviews, and brief them from the diff — not
|
`wait:false`. Never the implementer of the scope it reviews, and brief them from the diff — not
|
||||||
from the implementer's rationale, which carries its own blind spot. Dispatch each PR's reviewers
|
from the implementer's rationale, which carries its own blind spot. Dispatch each PR's reviewers
|
||||||
@@ -127,17 +155,29 @@ prefer `wait:false` + `fleet_poll` for anything non-trivial: a blocking `fleet_s
|
|||||||
**Delegating does not delegate responsibility.** Workers open PRs; you are the gate. Never delegate
|
**Delegating does not delegate responsibility.** Workers open PRs; you are the gate. Never delegate
|
||||||
the merge — and merging on a reviewer's word is delegating it by proxy.
|
the merge — and merging on a reviewer's word is delegating it by proxy.
|
||||||
|
|
||||||
|
**When a decision blocks you, consult architects — not the operator.** Spawn one or more architect
|
||||||
|
members, give them the question and the evidence you have, and act on what they agree. They are
|
||||||
|
authorized to settle it, not only to advise. Architects first form independent positions, then
|
||||||
|
compare them. If they still disagree after that comparison, they return both positions and their
|
||||||
|
checked evidence; the lead decides. Go to the operator only for an action the fleet has no
|
||||||
|
authority to take, such as spending money, granting access, or making a promise to someone else.
|
||||||
|
**Then write the decision on the ticket.** Taking the operator out of the loop also removes the
|
||||||
|
signal they used to get, because that signal was the block itself — work stopped, so they found
|
||||||
|
out. A ticket comment replaces it, and it reaches them whether or not they are at a terminal when
|
||||||
|
you decide.
|
||||||
|
|
||||||
| Intent | Tool |
|
| Intent | Tool |
|
||||||
|---|---|
|
|---|---|
|
||||||
| Confirm your own role | `fleet_whoami` |
|
| Confirm your own role | `fleet_whoami` |
|
||||||
| See backends available | `fleet_profiles` |
|
| See backends available | `fleet_profiles` |
|
||||||
| Start a member | `fleet_spawn{role?, profile?, cwd?, worktree?, ticket?, sessionName?, resumeSessionId?}` → `sessionId` + `paneId` |
|
| Start a member | `fleet_spawn{role?, profile?, cwd?, worktree?, ticket?, sessionName?, resumeSessionId?}` → `sessionId` + `paneId` |
|
||||||
| See the fleet | `fleet_list` → `leads` (your peers) + `members` (each carries `agentSessionId` when its backend knows one) · one peer's state: `fleet_status{sessionId}` |
|
| See the fleet | `fleet_list` → `leads` (your peers) + `members` (each carries `agentSessionId` when its backend knows one) + `loopHealth` (`RUNNING`, `STALLED`, or `STOPPED` for `statusPoller` and `sessionReaper`) · one peer's state: `fleet_status{sessionId}` |
|
||||||
| Delegate (blocking) | `fleet_send{sessionId, content}` |
|
| Delegate (blocking) | `fleet_send{sessionId, content}` |
|
||||||
| Delegate (long task) | `fleet_send{sessionId, content, wait:false}` → ticket → `fleet_poll{ticket}` |
|
| Delegate (long task) | `fleet_send{sessionId, content, wait:false}` → ticket → `fleet_poll{ticket}` |
|
||||||
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId` |
|
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId`, and only the caller that created that delegation |
|
||||||
| Message a **peer lead** on this host | `fleet_send{sessionId: <their terminal>, content}` — `fleet_list` → `leads` reports it. Coordination only, **never** a task |
|
| Message a **peer lead** on this host | `fleet_send{sessionId: <their terminal>, content}` — `fleet_list` → `leads` reports it. Coordination only, **never** a task |
|
||||||
| Message a **peer lead** on another daemon or host | `fleet_send{coordId: <their coord-id>, content}` — needs a `coordinator:` block; your own coord-id is in `fleet_list`. Coordination only, **never** a task |
|
| Message a **peer lead** on another daemon or host | `fleet_send{coordId: <their coord-id>, content}` — needs a `coordinator:` block; your own coord-id is in `fleet_list`. Coordination only, **never** a task |
|
||||||
|
| Message a **collaborator** on this host | `fleet_send{sessionId: <their terminal>, content}` — `fleet_list` reports a `collaborators` array, and each row carries that peer's `name` and the `sessionId` you send to. It is visible to you, to an architect and to another collaborator, never to a worker. Coordination only, **never** a task |
|
||||||
| Answer a peer lead that messaged you | `fleet_send{coordId}` — or `{sessionId}` if they are on this host. **Not** `fleet_reply`: it has no peer route and the publish is refused |
|
| Answer a peer lead that messaged you | `fleet_send{coordId}` — or `{sessionId}` if they are on this host. **Not** `fleet_reply`: it has no peer route and the publish is refused |
|
||||||
| Read your own held lead-to-lead mail (no ack) | `fleet_poll{coordId: <your own coord-id, from fleet_list's coordinator.selfId>}` — primary-only; never acks, so `fleet_list`'s `held[]` still shows it after. `fleet_list`'s `held[]` gives only a truncated preview — this is the only way to read the full body |
|
| Read your own held lead-to-lead mail (no ack) | `fleet_poll{coordId: <your own coord-id, from fleet_list's coordinator.selfId>}` — primary-only; never acks, so `fleet_list`'s `held[]` still shows it after. `fleet_list`'s `held[]` gives only a truncated preview — this is the only way to read the full body |
|
||||||
| Collect a held reply | `fleet_poll{target}` · then `fleet_ack{target, msgId}` |
|
| Collect a held reply | `fleet_poll{target}` · then `fleet_ack{target, msgId}` |
|
||||||
@@ -189,22 +229,49 @@ simply complies has thrown away the reason there are two of you.
|
|||||||
3. **`fleet_ask{question}`** when a decision is genuinely the lead's (ambiguous requirement, two
|
3. **`fleet_ask{question}`** when a decision is genuinely the lead's (ambiguous requirement, two
|
||||||
defensible fixes, "bug or intended?"). It blocks and you resume the *same* turn with the answer.
|
defensible fixes, "bug or intended?"). It blocks and you resume the *same* turn with the answer.
|
||||||
Don't ask what you could decide yourself.
|
Don't ask what you could decide yourself.
|
||||||
4. **End the turn with exactly one `fleet_reply{content}`**, carrying your complete answer. This is
|
4. **Re-read the ticket before you act on anything you were told earlier**, and again before you
|
||||||
|
commit. A message reaches you only while you are free to receive it; the ticket is there whenever
|
||||||
|
you look. **If a ticket comment contradicts your brief, the ticket comment is newer and it wins.**
|
||||||
|
5. **End the turn with exactly one `fleet_reply{content}`**, carrying your complete answer. This is
|
||||||
the whole handoff. No `fleet_reply` ⇒ the sender gets nothing and the exchange stalls.
|
the whole handoff. No `fleet_reply` ⇒ the sender gets nothing and the exchange stalls.
|
||||||
Do **not** lean on the completion fallback to carry your answer for you: when you end a turn
|
Do **not** lean on the completion fallback to carry your answer for you: when you end a turn
|
||||||
without replying, the bridge scrapes your pane, and it can return only the last 4000 characters.
|
without replying, the bridge scrapes your pane, and it can return only the last 4000 characters.
|
||||||
A clipped scrape is marked as partial, but the missing text is gone — your report reaches the
|
A clipped scrape is marked as partial, but the missing text is gone — your report reaches the
|
||||||
lead with its end cut off.
|
lead with its end cut off.
|
||||||
5. **Report honestly.** State only what you actually ran and its real output, including failures,
|
6. **Report honestly.** State only what you actually ran and its real output, including failures,
|
||||||
and never claim the result of a check you had no way to run. **Measure your own tools; do not
|
and never claim the result of a check you had no way to run. **Measure your own tools; do not
|
||||||
assume them.** What you mount depends on your backend: an opencode member gets the bridge and
|
assume them.** What you mount depends on your backend: an opencode member gets the bridge and
|
||||||
nothing else, while a Claude Code member also inherits the operator's user-scope MCP servers,
|
nothing else, while a Claude Code member also inherits the operator's user-scope MCP servers,
|
||||||
which the bridge never chose for you. Two rules follow. The primary's IDE tooling is still not
|
which the bridge never chose for you. Two rules follow. The primary's IDE tooling is still not
|
||||||
yours, whatever you see. And **a mounted tool is not a working tool** — the forge server you may
|
yours, whatever you see. And **a mounted tool is not a working tool** — the forge MCP server you
|
||||||
find there holds a deliberately blocked credential and fails every call, by design.
|
may find there holds a deliberately blocked credential and fails every call, by design. That is
|
||||||
6. **Never merge.** Stage files explicitly — never `git add -A` — and leave alone anything the
|
not your only forge route, and the two must not be confused: the repo-scoped `GITEA_TOKEN` the
|
||||||
|
daemon injects into your environment does work, and using it to open your own PR is part of the
|
||||||
|
job. A blocked MCP tool is never a reason to skip that step.
|
||||||
|
7. **Never merge.** Stage files explicitly — never `git add -A` — and leave alone anything the
|
||||||
project marks as not-yours-to-commit.
|
project marks as not-yours-to-commit.
|
||||||
|
|
||||||
|
### Collaborator — a named peer, not a member
|
||||||
|
|
||||||
|
`fleet_whoami` answered `collaborator`, so your pane's tab matches a `fleet.collaborators.<name>.tab`
|
||||||
|
entry. You are **not** a member: nothing delegates to you, you have no brief, no worktree and no
|
||||||
|
ticket, and **you owe no `fleet_reply`** — the turn contract above is for a session a lead spawned,
|
||||||
|
and it does not apply to you. Read it only to understand what the members around you are doing.
|
||||||
|
|
||||||
|
What you may do: observe the fleet (`fleet_list`, `fleet_profiles`, `fleet_whoami`), and send to a
|
||||||
|
lead or to another collaborator. What you may not: spawn, stop or drain anything, roll a lead's
|
||||||
|
session, answer a member's `fleet_ask`, poll a ticket, or send to a spawned member's terminal. Each
|
||||||
|
of those is refused at the gate, not queued.
|
||||||
|
|
||||||
|
Two limits worth knowing before you hit them. **You cannot reach a worker** — not even to help one —
|
||||||
|
because a worker belongs to the lead that spawned it, and routing around that would make you a
|
||||||
|
second orchestrator with no plan. Send to the lead instead. And **you cannot read a ticket**, so you
|
||||||
|
cannot collect a delegation's reply: `fleet_poll` refuses you at the role gate, and a ticket also
|
||||||
|
records the terminal that created it, so even a leaked id reads nothing.
|
||||||
|
|
||||||
|
Being named buys you a channel, not authority. Your `fleet_send` to a lead is coordination between
|
||||||
|
peers: the lead owes you no obedience, and you owe it none.
|
||||||
|
|
||||||
### Where each rule lives (don't duplicate — extend the right layer)
|
### Where each rule lives (don't duplicate — extend the right layer)
|
||||||
|
|
||||||
| Layer | Scope | Reaches |
|
| Layer | Scope | Reaches |
|
||||||
@@ -257,6 +324,8 @@ must obey belongs in the charter, not here.
|
|||||||
it for a multi-finding sweep hands the worker two contradictory output contracts. That has
|
it for a multi-finding sweep hands the worker two contradictory output contracts. That has
|
||||||
already cost three workers' turns: each wrote a good report to its terminal and ended the turn
|
already cost three workers' turns: each wrote a good report to its terminal and ended the turn
|
||||||
with no `fleet_reply`, and the scrape returned the tail of the brief instead.
|
with no `fleet_reply`, and the scrape returned the tail of the brief instead.
|
||||||
|
Spawn `implementer` with role `dev`, `reviewer` with role `reviewer`, and `hunter` with role
|
||||||
|
`hunter`.
|
||||||
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
||||||
`port-to-opencode` (make an OpenCode session a participant in this workspace),
|
`port-to-opencode` (make an OpenCode session a participant in this workspace),
|
||||||
`fleets-status` (report every fleet that shares one LavinMQ instance),
|
`fleets-status` (report every fleet that shares one LavinMQ instance),
|
||||||
@@ -270,10 +339,22 @@ must obey belongs in the charter, not here.
|
|||||||
(#362). **Read `plugin/` before designing anything about onboarding a project.** Two limits are
|
(#362). **Read `plugin/` before designing anything about onboarding a project.** Two limits are
|
||||||
structural, not bugs: a plugin cannot carry the role agent files, because
|
structural, not bugs: a plugin cannot carry the role agent files, because
|
||||||
`ClaudeCodeLauncher.java:371` requires `<cwd>/.claude/agents/<role>.md` in the member's own
|
`ClaudeCodeLauncher.java:371` requires `<cwd>/.claude/agents/<role>.md` in the member's own
|
||||||
worktree; and a plugin cannot deliver anything to members at all, because
|
worktree; and a plugin reaches a member only through `CLAUDE_CONFIG_DIR`, which
|
||||||
`ClaudeCodeLauncher.java:285` exports `CLAUDE_CONFIG_DIR` and every Claude profile here sets it,
|
`ClaudeCodeLauncher.java:286` exports with `putIfPresent` — so only for a profile that sets
|
||||||
so a member never reads the operator's plugin store. **The plugin is the lead-side surface;
|
`configDir`. Every `claude-code` profile does set one (the four without are `opencode`, which
|
||||||
member-facing assets travel in the worktree.**
|
never reads that variable). **But measured 2026-10-04: two of them point at
|
||||||
|
`~/.ccs/instances/ltms`, which is the operator's own `CLAUDE_CONFIG_DIR` on this host.** So for an
|
||||||
|
`opus` or `sonnet` member, "a member never reads the operator's plugin store" is false — it reads
|
||||||
|
the same store, because that store is the one its `configDir` names. It stays true for `local` and
|
||||||
|
`local-direct`, which point at `~/.ccs/instances/gx10`. `ClaudeCodeLauncher`'s own javadoc names
|
||||||
|
the related hazard: that file is rewritten on every spawn, so for those two profiles fleetd and the
|
||||||
|
operator's live session write the same `.claude.json`, and its compare-and-swap "narrows the
|
||||||
|
lost-update window, it does not close it". Re-measure which profiles share the operator's dir with
|
||||||
|
`awk '/^profiles:/{i=1;next} /^[a-z]/{i=0} i&&/^ [a-z-]+:$/{p=$1} i&&/configDir:/{print p,$2}'
|
||||||
|
fleetd/fleetd.yaml` against `echo $CLAUDE_CONFIG_DIR`; delete this note once no profile names the
|
||||||
|
operator's dir. **Treat the plugin as the lead-side surface and put member-facing assets in the
|
||||||
|
worktree** — that conclusion holds either way, because a worktree asset does not depend on which
|
||||||
|
config dir a member reads.
|
||||||
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
||||||
(a submodule with its own remote).
|
(a submodule with its own remote).
|
||||||
- **A provisioned worktree neutralizes `.mcp.json`, `opencode.json` and `.autoenv`** — the repo's
|
- **A provisioned worktree neutralizes `.mcp.json`, `opencode.json` and `.autoenv`** — the repo's
|
||||||
@@ -292,6 +373,13 @@ must obey belongs in the charter, not here.
|
|||||||
reference**, with the intent→tool table above as the short form. `McpContractDocTest` fails if
|
reference**, with the intent→tool table above as the short form. `McpContractDocTest` fails if
|
||||||
that page names a `fleet_*` tool the server does not register. The flows are kept out of this
|
that page names a `fleet_*` tool the server does not register. The flows are kept out of this
|
||||||
file because this file loads into every session's context.
|
file because this file loads into every session's context.
|
||||||
|
- **The daemon runs from `fleetd/run/fleetd.jar`, not `fleetd/target/fleetd.jar`** (fleetd #664).
|
||||||
|
Maven's own output still lands at `fleetd/target/fleetd.jar` — that part of the build is
|
||||||
|
unchanged — but the running daemon never has that file open, so a bare `mvn install`/`mvn clean`
|
||||||
|
in the main clone no longer corrupts anything a live process is reading. Verify merges in a
|
||||||
|
throwaway git worktree anyway: a build in the main clone still ships nothing until
|
||||||
|
`scripts/redeploy-fleetd.sh` moves it into place with its own atomic `mv`, performed only after
|
||||||
|
the old daemon is confirmed gone. Let only that script touch `fleetd/run/fleetd.jar`.
|
||||||
|
|
||||||
### Redeploying the daemon — the lead may do this (primary only)
|
### Redeploying the daemon — the lead may do this (primary only)
|
||||||
|
|
||||||
@@ -321,7 +409,7 @@ Before you call any work done, check the row that matches what you touched:
|
|||||||
| `ConnectionIdentity` / how a caller is resolved | the `fleet_whoami` paragraph and the fallback ladder |
|
| `ConnectionIdentity` / how a caller is resolved | the `fleet_whoami` paragraph and the fallback ladder |
|
||||||
| `REPLY_CHARTER`, or a launcher's mount/flags | the fallback ladder (`mcp__fleet__*`), and the layering table's top row |
|
| `REPLY_CHARTER`, or a launcher's mount/flags | the fallback ladder (`mcp__fleet__*`), and the layering table's top row |
|
||||||
| the injector / status gating | invariant 4 |
|
| the injector / status gating | invariant 4 |
|
||||||
| worktree provisioning or the parity overlay | the "both roles read this file" premise — it rests on the worker's worktree being a checkout of this repo |
|
| worktree provisioning or the parity overlay | the "every role reads this file" premise — it rests on the worker's worktree being a checkout of this repo |
|
||||||
| `.claude/skills/**` | the addendum's skill list, and the "name the playbook" rule |
|
| `.claude/skills/**` | the addendum's skill list, and the "name the playbook" rule |
|
||||||
| a new peer kind (non-Claude adapter) | what that peer can read — anything it must obey belongs in its charter, not in the block |
|
| a new peer kind (non-Claude adapter) | what that peer can read — anything it must obey belongs in its charter, not in the block |
|
||||||
| **anything an operator can use, configure, or observe** — an MCP tool, a `fleetd.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
| **anything an operator can use, configure, or observe** — an MCP tool, a `fleetd.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
||||||
|
|||||||
@@ -47,7 +47,7 @@
|
|||||||
<string>/Users/dai.ha/LTMS/claude-bridge/scripts/fleetd-launchd-wrapper.sh</string>
|
<string>/Users/dai.ha/LTMS/claude-bridge/scripts/fleetd-launchd-wrapper.sh</string>
|
||||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin/java</string>
|
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin/java</string>
|
||||||
<string>-jar</string>
|
<string>-jar</string>
|
||||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/target/fleetd.jar</string>
|
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/run/fleetd.jar</string>
|
||||||
<string>fleetd.yaml</string>
|
<string>fleetd.yaml</string>
|
||||||
</array>
|
</array>
|
||||||
|
|
||||||
|
|||||||
@@ -50,7 +50,7 @@ WorkingDirectory=%h/LTMS/fleetd/fleetd
|
|||||||
# and looks healthy, and the failure appears hours later as a member that cannot open a pull
|
# and looks healthy, and the failure appears hours later as a member that cannot open a pull
|
||||||
# request. exec keeps it one process, so systemd tracks the right PID.
|
# request. exec keeps it one process, so systemd tracks the right PID.
|
||||||
# This also avoids a SECOND copy of the secrets in a systemd drop-in: one source of truth.
|
# This also avoids a SECOND copy of the secrets in a systemd drop-in: one source of truth.
|
||||||
ExecStart=/bin/zsh -lc "exec java -jar target/fleetd.jar fleetd.yaml"
|
ExecStart=/bin/zsh -lc "exec java -jar run/fleetd.jar fleetd.yaml"
|
||||||
|
|
||||||
# PrivateTmp MUST stay false -- see herdr.service. fleetd creates the member ZDOTDIR scrub dir and
|
# PrivateTmp MUST stay false -- see herdr.service. fleetd creates the member ZDOTDIR scrub dir and
|
||||||
# the opencode config dir under java.io.tmpdir, and the member pane (a herdr child, a different
|
# the opencode config dir under java.io.tmpdir, and the member pane (a herdr child, a different
|
||||||
|
|||||||
@@ -5,6 +5,21 @@
|
|||||||
here took a revert and two upstream fixes — see §7.1, which is the useful part of this document. One
|
here took a revert and two upstream fixes — see §7.1, which is the useful part of this document. One
|
||||||
risk is **accepted rather than solved**: a stream cut by any mid-response timer arrives as HTTP 200
|
risk is **accepted rather than solved**: a stream cut by any mid-response timer arrives as HTTP 200
|
||||||
with no terminator, and our third-party members cannot detect it (§7.2).
|
with no terminator, and our third-party members cannot detect it (§7.2).
|
||||||
|
|
||||||
|
> **Superseded in part — 2026-09-13.** Two claims on this page are no longer true of the live fleet.
|
||||||
|
> I measured both on this host today.
|
||||||
|
>
|
||||||
|
> 1. **The model is named `acoder` now, not `deepseek-v4-flash`.** `acoder` is a stable alias, and
|
||||||
|
> the model behind it changed on 2026-08-28: it is Qwen3.8-27B, not DeepSeek. The old name is
|
||||||
|
> still served, so nothing broke — the gateway answers it and reports `"model": "acoder"` in the
|
||||||
|
> reply, which is how you can see for yourself that it is an alias. `fleetd.yaml` moved to
|
||||||
|
> `acoder` on 2026-09-13. Do not guess behaviour from the name; ask the gateway's own manifest,
|
||||||
|
> `GET https://llm.ltms.dev/v1/deployment`, and read its `generation` field.
|
||||||
|
> 2. **`local` sits at `weight: 0`, not 100.** Only `gx` is auto-selected today.
|
||||||
|
>
|
||||||
|
> §2 and §3 below are the plan as written in August. They are the record of the migration, so they
|
||||||
|
> stay as they are. If this note stops matching `fleetd.yaml`, re-measure and rewrite the note.
|
||||||
|
|
||||||
· **Upstream:** [systems/vms wiki → LLM and MCP Gateway](https://git.ltms.dev/systems/vms/wiki/LLM-and-MCP-Gateway)
|
· **Upstream:** [systems/vms wiki → LLM and MCP Gateway](https://git.ltms.dev/systems/vms/wiki/LLM-and-MCP-Gateway)
|
||||||
· **Upstream issue:** [systems/vms#31](https://git.ltms.dev/systems/vms/issues/31)
|
· **Upstream issue:** [systems/vms#31](https://git.ltms.dev/systems/vms/issues/31)
|
||||||
|
|
||||||
@@ -380,7 +395,10 @@ one turn; this one costs the whole task and is indistinguishable from a slow wor
|
|||||||
|
|
||||||
- Token accepted on both surfaces. **Unauthenticated → 401**, so the Caddy proxy really does gate —
|
- Token accepted on both surfaces. **Unauthenticated → 401**, so the Caddy proxy really does gate —
|
||||||
the wiki's "SecurityPolicy fails open" warning is about the gateway itself, not the edge.
|
the wiki's "SecurityPolicy fails open" warning is about the gateway itself, not the edge.
|
||||||
- `/v1/models` returns exactly `["deepseek-v4-flash"]`, so trap 3 is clear.
|
- `/v1/models` returned exactly `["deepseek-v4-flash"]` **on 2026-08-15**, so trap 3 was clear
|
||||||
|
then. It returns 6 ids now — `acoder`, `qwen3.8-27b-nvfp4`, `deepseek-v4-flash` and three
|
||||||
|
embedding names — measured on this host 2026-09-13. The exact-name rule still holds; the
|
||||||
|
one-item list does not.
|
||||||
- **Reasoning survives both surfaces** — see §3b above.
|
- **Reasoning survives both surfaces** — see §3b above.
|
||||||
- The launcher's generated opencode provider block is correct, carrying a real 48-character `llmk-`
|
- The launcher's generated opencode provider block is correct, carrying a real 48-character `llmk-`
|
||||||
key rather than the `fleetd-local-noauth` placeholder.
|
key rather than the `fleetd-local-noauth` placeholder.
|
||||||
|
|||||||
@@ -2,11 +2,22 @@
|
|||||||
target/
|
target/
|
||||||
dependency-reduced-pom.xml
|
dependency-reduced-pom.xml
|
||||||
|
|
||||||
|
# The daemon's runtime jar (fleetd #664). scripts/redeploy-fleetd.sh moves the built jar here
|
||||||
|
# with a same-filesystem rename; this is never Maven's output path and never belongs in git.
|
||||||
|
run/
|
||||||
|
|
||||||
# Local runtime config (copy from fleetd.example.yaml). Both names are ignored: fleetd.yaml is
|
# Local runtime config (copy from fleetd.example.yaml). Both names are ignored: fleetd.yaml is
|
||||||
# the current name, and bridged.yaml is the legacy name Fleetd still falls back to.
|
# the current name, and bridged.yaml is the legacy name Fleetd still falls back to.
|
||||||
fleetd.yaml
|
fleetd.yaml
|
||||||
bridged.yaml
|
bridged.yaml
|
||||||
|
|
||||||
|
# fleetd #635 follow-up — scripts/config-edit.sh's backups of fleetd.yaml. A backup of a file
|
||||||
|
# that must never be committed inherits that requirement. The directory is the real protection
|
||||||
|
# (it keeps working even if the backup naming changes); the glob is a backstop for a stray
|
||||||
|
# backup written the old way, directly beside fleetd.yaml, or by an older copy of the script.
|
||||||
|
.config-backups/
|
||||||
|
fleetd.yaml.bak.*
|
||||||
|
|
||||||
# CB-505 audit trail + daemon stdout/stderr — runtime records, never source
|
# CB-505 audit trail + daemon stdout/stderr — runtime records, never source
|
||||||
logs/
|
logs/
|
||||||
|
|
||||||
|
|||||||
+75
-37
@@ -62,11 +62,12 @@ bind:
|
|||||||
#
|
#
|
||||||
# Leads are configured under `fleet.leaders:` — see THE FLEET further down.
|
# Leads are configured under `fleet.leaders:` — see THE FLEET further down.
|
||||||
#
|
#
|
||||||
# Two things stop the tab-name convention from becoming a way to claim leadership: the configured
|
# Two things stop the tab-name convention from becoming a way to claim leadership: startup REFUSES
|
||||||
# member spaces are excluded from the scan, so nothing fleetd places can land in a matching tab;
|
# a `tabPrefix` that the fleet tabLabel template, or any per-profile `tabLabel` override, also
|
||||||
# and startup REFUSES a `tabPrefix` that the fleet tabLabel template, or any per-profile `tabLabel`
|
# matches, so the two namespaces cannot overlap by accident; and the CallerResolver asks the live
|
||||||
# override, also matches — so the two namespaces cannot overlap by accident. The label is a NAME,
|
# spawned-member roster BEFORE any tab map, so a live member is never mistaken for a lead no matter
|
||||||
# never a capability: what a pane may do is decided by the role the daemon resolves for it.
|
# what its tab says. The label is a NAME, never a capability: what a pane may do is decided by the
|
||||||
|
# role the daemon resolves for it.
|
||||||
|
|
||||||
# CB-551: IDLE-LEAD HEARTBEAT — nudge the single lead back to work when it has been continuously
|
# CB-551: IDLE-LEAD HEARTBEAT — nudge the single lead back to work when it has been continuously
|
||||||
# idle (no open fleet_send driving it) past the quiet period. The fleet is one lead + architects +
|
# idle (no open fleet_send driving it) past the quiet period. The fleet is one lead + architects +
|
||||||
@@ -77,29 +78,38 @@ bind:
|
|||||||
# a lead turn nobody asked for), so upgrading the daemon must never switch it on for you. Absent
|
# a lead turn nobody asked for), so upgrading the daemon must never switch it on for you. Absent
|
||||||
# block = feature off, exactly as before.
|
# block = feature off, exactly as before.
|
||||||
#
|
#
|
||||||
# Three knobs, each with a default that errs on the side of not burning context:
|
# Four knobs, each with a default that errs on the side of not burning context:
|
||||||
# idleAfterSeconds: 300 # how long the lead must stay idle before the FIRST nudge (default 300 —
|
# idleAfterSeconds: 300 # how long the lead must stay idle before the FIRST nudge (default 300 —
|
||||||
# # absorbs normal post-turn pauses; re-prompting every pause burns context)
|
# # absorbs normal post-turn pauses; re-prompting every pause burns context)
|
||||||
# backoffMs: 60000 # re-check cadence / spacing between nudges past the quiet period (default 60000)
|
# backoffMs: 60000 # re-check cadence / spacing between nudges past the quiet period (default 60000)
|
||||||
# quietNudgeCap: 3 # cap on consecutive nudges that find NOTHING pending, then it stops
|
# quietNudgeCap: 3 # cap on consecutive nudges that find NOTHING pending, then it stops
|
||||||
# # until real state appears (default 3 — never nag an empty fleet forever)
|
# # until real state appears (default 3 — never nag an empty fleet forever)
|
||||||
|
# contextHighNudge: false # fleetd #609 — when true, an idle lead whose OWN Claude Code context
|
||||||
|
# # reads HIGH (see LeadContextGauge; fleet_list's context row) gets a text
|
||||||
|
# # notice appended to its nudge telling it to consider fleet_handover. Text
|
||||||
|
# # only — it never rolls a pane by itself, and only the operator can approve
|
||||||
|
# # a roll. Fires once per HIGH stretch (a later OK reading re-arms it), and
|
||||||
|
# # never spends the quietNudgeCap budget. Default false/absent = off, same
|
||||||
|
# # as every other knob here — an upgraded daemon must not start telling
|
||||||
|
# # leads to hand over on its own.
|
||||||
# leadHeartbeat:
|
# leadHeartbeat:
|
||||||
# idleAfterSeconds: 300
|
# idleAfterSeconds: 300
|
||||||
# backoffMs: 60000
|
# backoffMs: 60000
|
||||||
# quietNudgeCap: 3
|
# quietNudgeCap: 3
|
||||||
|
# contextHighNudge: false
|
||||||
|
|
||||||
# Lead rollover (fleetd #480): replace a lead session that has decided it is ready to be replaced,
|
# Lead rollover (fleetd #480): replace a lead session that has decided it is ready to be replaced,
|
||||||
# without an operator doing it by hand. A lead writes a handover file, then asks fleetd to clear its
|
# without an operator doing it by hand. A lead writes a handover file, then asks fleetd to end its
|
||||||
# own pane and bootstrap a fresh session against that file.
|
# own pane, launch a fresh one, and bootstrap that fresh session against the handover file.
|
||||||
#
|
#
|
||||||
# Opt-in on purpose — it clears the lead's own pane on request, so upgrading the daemon must never
|
# Opt-in on purpose — it tears down the lead's own pane on request, so upgrading the daemon must
|
||||||
# acquire that ability for you. Absent block = feature off, and nothing is constructed at all. Even
|
# never acquire that ability for you. Absent block = feature off, and nothing is constructed at all.
|
||||||
# once present, nothing but an explicit confirm() call — one that passes every check — can ever
|
# Even once present, nothing but an explicit confirm() call — one that passes every check — can ever
|
||||||
# cause a /clear: there is no recurring timer, heartbeat or scheduler anywhere in this feature that
|
# tear a pane down: there is no recurring timer, heartbeat or scheduler anywhere in this feature that
|
||||||
# fires one on its own initiative. confirm() itself is called FROM the calling lead's own turn, so
|
# fires one on its own initiative. confirm() itself is called FROM the calling lead's own turn, so it
|
||||||
# it cannot clear the pane inline (that pane is still WORKING); instead it schedules a one-shot
|
# cannot act on the pane inline (that pane is still WORKING); instead it schedules a one-shot
|
||||||
# continuation that waits for the SAME confirm() call's turn to end, then does the actual work. See
|
# continuation that waits for the SAME confirm() call's turn to end, then does the actual work. See
|
||||||
# dev.ltms.fleet.lead.LeadRollover's class javadoc for the exact order (fleetd #480 correction).
|
# dev.ltms.fleet.lead.LeadRollover's class javadoc for the exact order.
|
||||||
#
|
#
|
||||||
# handoverPath: REQUIRED when this block is present — where the handover file a fresh lead session
|
# handoverPath: REQUIRED when this block is present — where the handover file a fresh lead session
|
||||||
# reads must live. No default (an operator-specific path); a present block with no
|
# reads must live. No default (an operator-specific path); a present block with no
|
||||||
@@ -108,25 +118,30 @@ bind:
|
|||||||
# working directory when that lead has none configured) — never against whatever
|
# working directory when that lead has none configured) — never against whatever
|
||||||
# directory the daemon process happens to have been started in. An absolute path is
|
# directory the daemon process happens to have been started in. An absolute path is
|
||||||
# used unchanged. Prefer an absolute path if the daemon and the lead's pane might not
|
# used unchanged. Prefer an absolute path if the daemon and the lead's pane might not
|
||||||
# share a working directory (fleetd #480 follow-up).
|
# share a working directory.
|
||||||
# requireOperatorConfirm: true # default true — confirm() refuses unless the caller also passes
|
# requireOperatorConfirm: true # default true — confirm() refuses unless the caller also passes
|
||||||
# # operatorConfirmed: true
|
# # operatorConfirmed: true
|
||||||
# maxDocAgeSeconds: 3600 # default 3600 — refuse a handover file older than this
|
# maxDocAgeSeconds: 3600 # default 3600 — refuse a handover file older than this
|
||||||
# turnSettleSeconds: 20 # default 20 — how long the deferred roll waits for the CALLING
|
# turnSettleSeconds: 20 # default 20 — how long the deferred roll waits for the CALLING
|
||||||
# # lead's own turn to end (its pane to report injectable again)
|
# # lead's own turn to end (its pane to report injectable again)
|
||||||
# # before sending /clear at all. If this elapses, /clear is NEVER
|
# # before tearing the old pane down at all. If this elapses, nothing
|
||||||
# # sent — a lead that never goes idle is still doing real work.
|
# # is torn down — a lead that never goes idle is still doing real
|
||||||
# clearSettleSeconds: 20 # default 20 — how long to wait for the pane to become injectable
|
# # work.
|
||||||
# # again AFTER /clear before giving up (never sends bootstrapText
|
# relaunchReadySeconds: 45 # default 45 — bounds two later waits, after the old pane is gone
|
||||||
# # if this elapses). A separate, second wait from turnSettleSeconds.
|
# # and a fresh one has been launched: first, for the fresh pane to
|
||||||
|
# # reach a real turn boundary (never sends bootstrapText if THIS one
|
||||||
|
# # elapses); second, for the new terminal to be recognised as this
|
||||||
|
# # lead (bootstrapText is sent either way once the first wait
|
||||||
|
# # passes). A separate, later pair of waits from turnSettleSeconds.
|
||||||
# bootstrapText: "..." # default names the RESOLVED (absolute) handoverPath — sent to
|
# bootstrapText: "..." # default names the RESOLVED (absolute) handoverPath — sent to
|
||||||
# # the lead once its pane settles after /clear
|
# # the freshly relaunched lead's pane once it reaches a real turn
|
||||||
|
# # boundary
|
||||||
# leadRollover:
|
# leadRollover:
|
||||||
# handoverPath: /path/to/handover.md
|
# handoverPath: /path/to/handover.md
|
||||||
# requireOperatorConfirm: true
|
# requireOperatorConfirm: true
|
||||||
# maxDocAgeSeconds: 3600
|
# maxDocAgeSeconds: 3600
|
||||||
# turnSettleSeconds: 20
|
# turnSettleSeconds: 20
|
||||||
# clearSettleSeconds: 20
|
# relaunchReadySeconds: 45
|
||||||
# bootstrapText: "Fresh lead session: read the handover file and carry on."
|
# bootstrapText: "Fresh lead session: read the handover file and carry on."
|
||||||
|
|
||||||
# Fleet health detection is dormant unless enabled (CB-573). It reads one whole-fleet agent list
|
# Fleet health detection is dormant unless enabled (CB-573). It reads one whole-fleet agent list
|
||||||
@@ -200,13 +215,13 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
|||||||
# e.g. `env DISPLAY=:10.0 idea {dir}`. Best-effort: a failure is logged, never
|
# e.g. `env DISPLAY=:10.0 idea {dir}`. Best-effort: a failure is logged, never
|
||||||
# fails the spawn. Omit to open the member's module by hand. There is no close
|
# fails the spawn. Omit to open the member's module by hand. There is no close
|
||||||
# half yet — an opened module stays open until the operator closes it.
|
# half yet — an opened module stays open until the operator closes it.
|
||||||
# autoCompactWindow → opt-in, default off. A bounded token window that forces a spawned member to
|
# autoCompactWindow → opt-in, default off. A bounded token window that forces a launched Claude Code
|
||||||
# compact its context instead of running on the backend's own default and dying
|
# lead or member to compact its context instead of running on the backend's own
|
||||||
# mid-turn (losing its fleet_reply — the whole point of the turn — with it).
|
# default. A member that runs out of context can die mid-turn and lose its fleet_reply.
|
||||||
# Validated at config load to [100000, 1000000] — the band Claude Code's own
|
# Validated at config load to [100000, 1000000] — the band Claude Code's own
|
||||||
# --autocompact flag accepts.
|
# --autocompact flag accepts.
|
||||||
# CROSS-BACKEND SEMANTICS DIFFER: on claude-code this is a launch-time
|
# CROSS-BACKEND SEMANTICS DIFFER: on claude-code this is a launch-time
|
||||||
# `--autocompact <tokens>` flag — the member compacts AT this window. opencode
|
# `--autocompact <tokens>` flag — the Claude Code session compacts AT this window. opencode
|
||||||
# has no equivalent flag (it only forces `compaction.auto: true`, unconditionally,
|
# has no equivalent flag (it only forces `compaction.auto: true`, unconditionally,
|
||||||
# already), so this is instead applied as the model's `limit.context` in the
|
# already), so this is instead applied as the model's `limit.context` in the
|
||||||
# generated opencode.json — the member compacts WITHIN this window, not exactly
|
# generated opencode.json — the member compacts WITHIN this window, not exactly
|
||||||
@@ -406,7 +421,7 @@ profiles:
|
|||||||
# ideMcpUrl: http://127.0.0.1:29170/index-mcp/streamable-http # opt-in (CB-634): IDE code intelligence, pinned to the worktree
|
# ideMcpUrl: http://127.0.0.1:29170/index-mcp/streamable-http # opt-in (CB-634): IDE code intelligence, pinned to the worktree
|
||||||
# ideProjectDir: fleetd # CB-634: module dir the IDE opens + the overlay pins (this repo's pom is in fleetd/)
|
# ideProjectDir: fleetd # CB-634: module dir the IDE opens + the overlay pins (this repo's pom is in fleetd/)
|
||||||
# ideOpenCommand: env DISPLAY=:10.0 idea {dir} # CB-634 auto-open: opens {dir} in the IDE at spawn; omit to open by hand
|
# ideOpenCommand: env DISPLAY=:10.0 idea {dir} # CB-634 auto-open: opens {dir} in the IDE at spawn; omit to open by hand
|
||||||
# autoCompactWindow: 250000 # opt-in: bound member context; claude-code compacts AT this, opencode within it (model limit.context)
|
# autoCompactWindow: 250000 # opt-in: bound Claude Code lead/member context; claude-code compacts AT this, opencode within it (model limit.context)
|
||||||
gx11: # a second backend, so `placement: weighted` has a choice
|
gx11: # a second backend, so `placement: weighted` has a choice
|
||||||
baseUrl: http://gx01.gw:8000 # self-hosted; ccs handles the model + token
|
baseUrl: http://gx01.gw:8000 # self-hosted; ccs handles the model + token
|
||||||
placement: tab
|
placement: tab
|
||||||
@@ -560,7 +575,7 @@ placement: weighted
|
|||||||
# older keys: `leaders:`, `members:`, `leadScan:` and `defaultProfile:`.
|
# older keys: `leaders:`, `members:`, `leadScan:` and `defaultProfile:`.
|
||||||
#
|
#
|
||||||
# A member is anything a lead spawns, and every member has two INDEPENDENT attributes:
|
# A member is anything a lead spawns, and every member has two INDEPENDENT attributes:
|
||||||
# role — which contract: architect, dev or reviewer. It picks the launch charter, the role
|
# role — which contract: architect, dev, hunter or reviewer. It picks the launch charter, the role
|
||||||
# file, the playbook skill and the authz row.
|
# file, the playbook skill and the authz row.
|
||||||
# profile — which backend: one of the `profiles:` keys above (model, CLI adapter, cost).
|
# profile — which backend: one of the `profiles:` keys above (model, CLI adapter, cost).
|
||||||
# They vary on their own. A reviewer may run on the same profile as the dev whose diff it reads,
|
# They vary on their own. A reviewer may run on the same profile as the dev whose diff it reads,
|
||||||
@@ -572,13 +587,13 @@ placement: weighted
|
|||||||
#
|
#
|
||||||
# Each pool lists the profiles that role MAY run on — these are pools, not identities. That is also
|
# Each pool lists the profiles that role MAY run on — these are pools, not identities. That is also
|
||||||
# what replaced `defaultProfile:`: an unqualified spawn names a role, and that role's pool supplies
|
# what replaced `defaultProfile:`: an unqualified spawn names a role, and that role's pool supplies
|
||||||
# the candidates, in definition order. A dev and a reviewer staying anonymous is exactly compatible
|
# the candidates, in definition order. A dev, hunter and reviewer staying anonymous is exactly
|
||||||
# with being listed here; the entry key just names the entry.
|
# compatible with being listed here; the entry key just names the entry.
|
||||||
fleet:
|
fleet:
|
||||||
# Optional launch-charter text, keyed only by the singular role wire names: architect, dev,
|
# Optional launch-charter text, keyed only by the singular role wire names: architect, dev,
|
||||||
# reviewer. Changes are HOT and reach the next spawn without a daemon restart. Do not put secrets
|
# hunter, reviewer. Changes are HOT and reach the next spawn without a daemon restart. Do not put
|
||||||
# here: a later launch step writes this text to a world-readable temp file, and ${ENV} interpolation
|
# secrets here: a later launch step writes this text to a world-readable temp file, and ${ENV}
|
||||||
# is deliberately not supported.
|
# interpolation is deliberately not supported.
|
||||||
charters:
|
charters:
|
||||||
architect: |-
|
architect: |-
|
||||||
You are an architect in this fleet. You refine work before anyone builds it:
|
You are an architect in this fleet. You refine work before anyone builds it:
|
||||||
@@ -589,6 +604,9 @@ fleet:
|
|||||||
dev: |-
|
dev: |-
|
||||||
You implement the one unit you were given, and nothing else. You test it,
|
You implement the one unit you were given, and nothing else. You test it,
|
||||||
commit it, and open your own pull request. You never merge.
|
commit it, and open your own pull request. You never merge.
|
||||||
|
hunter: |-
|
||||||
|
You sweep the assigned scope for real defects. You may run the build or tests
|
||||||
|
to check a finding. You change nothing, and report several ranked findings.
|
||||||
reviewer: |-
|
reviewer: |-
|
||||||
You review the diff you were given. You report bugs, risks and missing tests.
|
You review the diff you were given. You report bugs, risks and missing tests.
|
||||||
You do not change code.
|
You do not change code.
|
||||||
@@ -647,9 +665,10 @@ fleet:
|
|||||||
# tabPrefix: "lead:" # only used to guard against a worker tabLabel colliding with
|
# tabPrefix: "lead:" # only used to guard against a worker tabLabel colliding with
|
||||||
# # this convention at startup; plays no part in matching a lead
|
# # this convention at startup; plays no part in matching a lead
|
||||||
# scanIntervalSeconds: 10 # rescan cadence, and the worst case before a new tab is seen
|
# scanIntervalSeconds: 10 # rescan cadence, and the worst case before a new tab is seen
|
||||||
# workspace: leads # where a launched lead's tab is created (default "leads").
|
# workspace: leads # where a launched lead's tab is created (default "fleet",
|
||||||
# # MUST NOT be a member workspace — those are excluded from the
|
# # the same shared space the members use). Sharing that space
|
||||||
# # scan, so a lead placed in one is never found again.
|
# # with members is the normal shipped shape: the scanner tells
|
||||||
|
# # a lead from a member by the exact tab label, not by workspace.
|
||||||
# cwd: /path/to/repo # the launched lead's working directory (default: fleetd's own)
|
# cwd: /path/to/repo # the launched lead's working directory (default: fleetd's own)
|
||||||
# kind: claude # descriptive; reported by fleet_whoami
|
# kind: claude # descriptive; reported by fleet_whoami
|
||||||
# gpt-sol-5.6:
|
# gpt-sol-5.6:
|
||||||
@@ -657,6 +676,22 @@ fleet:
|
|||||||
# kind: opencode
|
# kind: opencode
|
||||||
# model: openai/gpt-5.6-terra
|
# model: openai/gpt-5.6-terra
|
||||||
|
|
||||||
|
# A collaborator tab, keyed by name (fleetd #669). A pane whose tab matches resolves as the
|
||||||
|
# COLLABORATOR role: `fleet_whoami` answers `collaborator`, and the session may observe the fleet
|
||||||
|
# (`fleet_list`, `fleet_profiles`, `fleet_whoami`) and `fleet_send` to a lead or to another
|
||||||
|
# collaborator. It may NOT spawn, stop or drain anything, roll a lead's session, answer a
|
||||||
|
# member's `fleet_ask`, poll a ticket, send across hosts, or send to a spawned member's terminal.
|
||||||
|
# Each of those is refused at the gate rather than queued.
|
||||||
|
# Recognise-only, like a profile-less `leaders:` entry above: there is no `profile:`, no
|
||||||
|
# `instances:` and no `kind:`. `tab:` is REQUIRED and is the only field identity depends on,
|
||||||
|
# matched case-insensitively — the same GET-THE-VALUE-RIGHT warning above the `leaders:` block
|
||||||
|
# applies here too.
|
||||||
|
# A lead cannot discover a collaborator yet (#703): `fleet_list` has no `collaborators` key, so
|
||||||
|
# the collaborator must speak first, or pass on the `sessionId` its own `fleet_whoami` reports.
|
||||||
|
# collaborators:
|
||||||
|
# reviewer-alex:
|
||||||
|
# tab: "collab: alex"
|
||||||
|
|
||||||
# architects:
|
# architects:
|
||||||
# architect-1:
|
# architect-1:
|
||||||
# profile: opus # a strong model, on the operator's subscription
|
# profile: opus # a strong model, on the operator's subscription
|
||||||
@@ -666,6 +701,9 @@ fleet:
|
|||||||
developers:
|
developers:
|
||||||
gx10:
|
gx10:
|
||||||
profile: gx10
|
profile: gx10
|
||||||
|
# hunters:
|
||||||
|
# gx10:
|
||||||
|
# profile: gx10 # a hunt may run checks, but never changes code
|
||||||
# reviewers:
|
# reviewers:
|
||||||
# gx10:
|
# gx10:
|
||||||
# profile: gx10 # the same backend may serve two roles; that is the point
|
# profile: gx10 # the same backend may serve two roles; that is the point
|
||||||
|
|||||||
+4
-2
@@ -189,8 +189,10 @@
|
|||||||
|
|
||||||
<build>
|
<build>
|
||||||
<!-- CB-634: the cutover renamed the module dir (bridged/ -> fleetd/), the jar, and the
|
<!-- CB-634: the cutover renamed the module dir (bridged/ -> fleetd/), the jar, and the
|
||||||
launchd plist together. The installed plist names fleetd/target/fleetd.jar and
|
launchd plist together. fleetd #664: the installed plist and the systemd unit now name
|
||||||
KeepAlive is armed, so this name, the plist, and the wrapper must move as one. -->
|
fleetd/run/fleetd.jar, not this plugin's own output path — see
|
||||||
|
scripts/redeploy-fleetd.sh for the mv that gets a build from here to there. KeepAlive is
|
||||||
|
armed, so this name, the plist, and the wrapper must still move as one. -->
|
||||||
<finalName>fleetd</finalName>
|
<finalName>fleetd</finalName>
|
||||||
<plugins>
|
<plugins>
|
||||||
<plugin>
|
<plugin>
|
||||||
|
|||||||
@@ -0,0 +1,22 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 Unit A: everything {@code Fleetd.main} has ready once config is loaded, reported
|
||||||
|
* and validated — the exact point {@code main} used to keep going straight into socket and broker
|
||||||
|
* work. Building this (and handing it to {@link FleetdAssembly#assembleAndStart}) is the seam a
|
||||||
|
* test now has to drive the real boot composition without being {@code main} itself.
|
||||||
|
*
|
||||||
|
* @param cfg the boot-time {@link FleetConfig} snapshot every one-time wiring decision reads —
|
||||||
|
* see {@code Fleetd.main}'s own comment on why this must never be swapped for a live
|
||||||
|
* reference once loaded
|
||||||
|
* @param config the live {@link ConfigRef} the hot-reloadable paths read per use
|
||||||
|
* @param guard the same {@link SubscriptionGuard} {@code Fleetd.main} already used to assert the
|
||||||
|
* launching environment is clean, reused rather than rebuilt so the assembled
|
||||||
|
* launchers see the identical instance {@code main} already validated with
|
||||||
|
*/
|
||||||
|
record AssemblyInputs(FleetConfig cfg, ConfigRef config, SubscriptionGuard guard) {
|
||||||
|
}
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,575 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.auth.CallerResolver;
|
||||||
|
import dev.ltms.fleet.auth.MemberRegistry;
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.ConfigWatcher;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.health.FleetHealthMonitor;
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||||
|
import dev.ltms.fleet.herdr.LeadTabScanner;
|
||||||
|
import dev.ltms.fleet.herdr.PaneLocator;
|
||||||
|
import dev.ltms.fleet.herdr.UnixSocketHerdrClient;
|
||||||
|
import dev.ltms.fleet.inject.BackendErrorPatternLookup;
|
||||||
|
import dev.ltms.fleet.inject.BackendErrorSink;
|
||||||
|
import dev.ltms.fleet.inject.CompletionResolver;
|
||||||
|
import dev.ltms.fleet.inject.ExhaustedPatternLookup;
|
||||||
|
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||||
|
import dev.ltms.fleet.inject.Injector;
|
||||||
|
import dev.ltms.fleet.inject.LiveExhaustedPatterns;
|
||||||
|
import dev.ltms.fleet.inject.MemberPresence;
|
||||||
|
import dev.ltms.fleet.inject.StatusPoller;
|
||||||
|
import dev.ltms.fleet.inject.TurnListener;
|
||||||
|
import dev.ltms.fleet.lead.LeadContextGauge;
|
||||||
|
import dev.ltms.fleet.lead.LeadLauncher;
|
||||||
|
import dev.ltms.fleet.lead.LeadRollover;
|
||||||
|
import dev.ltms.fleet.mcp.ConnectionIdentity;
|
||||||
|
import dev.ltms.fleet.mcp.FleetMcp;
|
||||||
|
import dev.ltms.fleet.mcp.LsofPeerPidLookup;
|
||||||
|
import dev.ltms.fleet.mcp.LsofProcessCwdLookup;
|
||||||
|
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||||
|
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||||
|
import dev.ltms.fleet.member.HerdrPeerLauncher;
|
||||||
|
import dev.ltms.fleet.member.MemberCredentialPolicyView;
|
||||||
|
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||||
|
import dev.ltms.fleet.metrics.Metrics;
|
||||||
|
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||||
|
import dev.ltms.fleet.msg.LeadCoordLoop;
|
||||||
|
import dev.ltms.fleet.msg.LeadHeartbeatLoop;
|
||||||
|
import dev.ltms.fleet.msg.MessageService;
|
||||||
|
import dev.ltms.fleet.msg.Rendezvous;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||||
|
import dev.ltms.fleet.peer.MemberRole;
|
||||||
|
import dev.ltms.fleet.peer.PeerLauncher;
|
||||||
|
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||||
|
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||||
|
import dev.ltms.fleet.power.CaffeinateSleepAssertionMechanism;
|
||||||
|
import dev.ltms.fleet.power.IdleSleepGuard;
|
||||||
|
import dev.ltms.fleet.rest.FleetApp;
|
||||||
|
import dev.ltms.fleet.session.GitWorktrees;
|
||||||
|
import dev.ltms.fleet.session.SessionManager;
|
||||||
|
import dev.ltms.fleet.session.SessionReaper;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.slf4j.Logger;
|
||||||
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.LinkedHashSet;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Set;
|
||||||
|
import java.util.concurrent.ConcurrentHashMap;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.concurrent.atomic.AtomicReference;
|
||||||
|
import java.util.function.Function;
|
||||||
|
import java.util.function.Predicate;
|
||||||
|
import java.util.function.Supplier;
|
||||||
|
import java.util.regex.Pattern;
|
||||||
|
import java.util.stream.Collectors;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 Unit A: the real boot assembly, extracted out of {@code Fleetd.main} so a test can
|
||||||
|
* drive it directly. {@link #assembleAndStart} is <em>the same statements {@code main} used to run
|
||||||
|
* inline</em>, in the same order, against a real {@link ResourcePorts} in production and a fake one
|
||||||
|
* in a test — see {@code FleetdAssemblyLifecycleTest}. {@code Fleetd.main} keeps config loading and
|
||||||
|
* {@code cfg.validateAll()}; everything from immediately after that call onward moved here —
|
||||||
|
* including, since fleetd #612 A-gaps (gap 2), the two post-validation reports ({@code
|
||||||
|
* reportRoleFallbackGaps}, {@code assertChartersNameOnlyRegisteredTools}) that Unit A originally
|
||||||
|
* left behind in {@code main}. Those two calls do no I/O themselves, but leaving them outside this
|
||||||
|
* method meant deleting either one compiled clean and left the whole suite green — nothing drove
|
||||||
|
* {@code main} itself, so nothing could notice. They run first here, in the same relative order,
|
||||||
|
* before the herdr socket or anything else that touches the outside world.
|
||||||
|
*
|
||||||
|
* <p><strong>Construction and start order is preserved exactly, on purpose.</strong> This is not
|
||||||
|
* rebuilt into "construct everything, then start everything" — that would change boot timing. The
|
||||||
|
* order recorded before any code moved (see the ticket and {@code FleetdAssemblyLifecycleTest}):
|
||||||
|
* {@code SessionReaper} starts first (if {@code lifecycle.idleTtlSeconds} is configured), then
|
||||||
|
* {@link StatusPoller}, then the optional {@link LeadHeartbeatLoop} and {@link FleetHealthMonitor},
|
||||||
|
* then the optional {@link LeadCoordLoop} and {@link ConfigWatcher}, and the HTTP server starts
|
||||||
|
* last of all. The close order (see {@link FleetdRuntime#close()}) is the mirror the original
|
||||||
|
* shutdown hook always used.
|
||||||
|
*
|
||||||
|
* <p><strong>One statement could not move without reordering startup.</strong> {@code Fleetd.main}
|
||||||
|
* registered its shutdown hook <em>before</em> building the Javalin {@code FleetApp} — the hook
|
||||||
|
* itself never touched {@code app} (it still doesn't; see {@link FleetdRuntime#close()}), but the
|
||||||
|
* hook needs a {@link FleetdRuntime} to close over, and the ticket asks that runtime to also own
|
||||||
|
* {@code FleetApp}. Building the runtime before the app exists and mutating it afterward (via
|
||||||
|
* {@link FleetdRuntime#attachApp}) preserves the exact original order — hook registered, then app
|
||||||
|
* built, then HTTP started — without moving the app's construction earlier or the hook's
|
||||||
|
* registration later. That is the one seam this ticket did not get to pin any other way.
|
||||||
|
*
|
||||||
|
* <p><strong>No inert variant.</strong> Deliberately, there is no overload of this method that
|
||||||
|
* accepts a smaller/optional {@link ResourcePorts} or defaults one internally. A future edit that
|
||||||
|
* wants to skip {@code FleetdAssembly} entirely and build its own graph is still possible — no
|
||||||
|
* static analysis stops that — but it cannot do so by quietly swapping this call for an inert
|
||||||
|
* substitute that still compiles, because none exists.
|
||||||
|
*/
|
||||||
|
final class FleetdAssembly {
|
||||||
|
|
||||||
|
private static final Logger log = LoggerFactory.getLogger(Fleetd.class);
|
||||||
|
|
||||||
|
/** CB-637: how often the lead coordination loop looks for peer messages — see {@code Fleetd}'s own constant. */
|
||||||
|
private static final long LEAD_COORD_INTERVAL_MS = 3_000L;
|
||||||
|
|
||||||
|
private FleetdAssembly() {
|
||||||
|
}
|
||||||
|
|
||||||
|
static FleetdRuntime assembleAndStart(AssemblyInputs inputs, ResourcePorts ports) {
|
||||||
|
FleetConfig cfg = inputs.cfg();
|
||||||
|
ConfigRef config = inputs.config();
|
||||||
|
SubscriptionGuard guard = inputs.guard();
|
||||||
|
|
||||||
|
// fleetd #612 A-gaps (gap 2): moved in from Fleetd.main, immediately after cfg.validateAll()
|
||||||
|
// there — the exact point main used to call these two, and still the first thing this
|
||||||
|
// method does, before any socket or broker work below. See this class's javadoc and each
|
||||||
|
// method's own for why they run here rather than in FleetConfig#validateAll() itself.
|
||||||
|
Fleetd.reportRoleFallbackGaps(cfg);
|
||||||
|
Fleetd.assertChartersNameOnlyRegisteredTools(cfg);
|
||||||
|
|
||||||
|
Path socket = cfg.herdrSocket() != null && !cfg.herdrSocket().isBlank()
|
||||||
|
? Path.of(cfg.herdrSocket())
|
||||||
|
: UnixSocketHerdrClient.defaultSocketPath();
|
||||||
|
|
||||||
|
HerdrClient herdr = ports.connectHerdr(socket);
|
||||||
|
HerdrClient memberHerdr = cfg.memberHerdrSocket() != null && !cfg.memberHerdrSocket().isBlank()
|
||||||
|
? ports.connectHerdr(Path.of(cfg.memberHerdrSocket()))
|
||||||
|
: herdr;
|
||||||
|
AtomicReference<Supplier<Map<String, String>>> leadsRef = new AtomicReference<>(Map::of);
|
||||||
|
// fleetd #669 Unit E: a collaborator's pane is opened by a person, exactly like a lead's,
|
||||||
|
// so its terminal must also route to the lead herdr daemon rather than the member one.
|
||||||
|
AtomicReference<Supplier<Map<String, String>>> collaboratorTerminalsRef = new AtomicReference<>(Map::of);
|
||||||
|
HerdrRouter router = new HerdrRouter(herdr, memberHerdr,
|
||||||
|
target -> leadsRef.get().get().containsKey(target)
|
||||||
|
|| collaboratorTerminalsRef.get().get().containsKey(target));
|
||||||
|
// CB-402: one adapter per configured peer kind, fronted by a composite router. A profile's
|
||||||
|
// `kind:` selects its adapter — claude-code (the default) and opencode partition the profile
|
||||||
|
// set — and the composite dispatches each SPI call to the adapter that owns the profile/pane.
|
||||||
|
Map<String, FleetConfig.Profile> claudeProfiles = new LinkedHashMap<>();
|
||||||
|
Map<String, FleetConfig.Profile> opencodeProfiles = new LinkedHashMap<>();
|
||||||
|
cfg.profiles().forEach((name, w) -> {
|
||||||
|
if (w.isOpenCode()) {
|
||||||
|
opencodeProfiles.put(name, w);
|
||||||
|
} else {
|
||||||
|
claudeProfiles.put(name, w);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
List<HerdrPeerLauncher> adapters = new ArrayList<>();
|
||||||
|
// fleetd #175: the daemon's real ExhaustionSink can only be built once `sessions` exists
|
||||||
|
// (below), but `sessions` needs `workers`, which needs the adapters built right here — a
|
||||||
|
// genuine cycle. Break it exactly like liveCountRef below: a forwarding sink built now,
|
||||||
|
// pointed at the real one once it exists.
|
||||||
|
AtomicReference<ExhaustionSink> exhaustionSinkRef = new AtomicReference<>(ExhaustionSink.none());
|
||||||
|
ExhaustionSink forwardingExhaustionSink = Fleetd.forwardingExhaustionSink(exhaustionSinkRef);
|
||||||
|
// The claude-code adapter is the always-present default; keep it even with no profiles (so a
|
||||||
|
// bridge configured with no workers, or opencode-only, still has a well-defined base adapter)
|
||||||
|
// unless opencode is the only kind configured.
|
||||||
|
if (!claudeProfiles.isEmpty() || opencodeProfiles.isEmpty()) {
|
||||||
|
adapters.add(Fleetd.claudeCodeLauncher(router.memberAgents(), router.memberSpaces(), guard,
|
||||||
|
claudeProfiles, cfg, config));
|
||||||
|
}
|
||||||
|
if (!opencodeProfiles.isEmpty()) {
|
||||||
|
adapters.add(Fleetd.openCodeLauncher(router.memberAgents(), router.memberSpaces(),
|
||||||
|
opencodeProfiles, cfg, config, forwardingExhaustionSink));
|
||||||
|
}
|
||||||
|
AtomicReference<Function<String, Integer>> liveCountRef = new AtomicReference<>(_ -> 0);
|
||||||
|
// CB-578 stage B: one quarantine tracker for the whole daemon, shared between the launcher
|
||||||
|
// (checked at spawn) and the exhaustion sink wired in below (written on BACKEND_EXHAUSTED).
|
||||||
|
BackendQuarantine quarantine = BackendQuarantine.withEscalation(ports.nanoClock(),
|
||||||
|
TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));
|
||||||
|
// fleetd #201 Unit 5: one outage-cool-off tracker for the whole daemon, shared between the
|
||||||
|
// launcher (checked at spawn, like `quarantine` above) and the backend-error sink wired in
|
||||||
|
// below (written on a classified backend error).
|
||||||
|
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(ports.nanoClock());
|
||||||
|
PeerLauncher workers = new CompositePeerLauncher(
|
||||||
|
adapters,
|
||||||
|
cfg.effectiveDefaultProfile(),
|
||||||
|
config,
|
||||||
|
profileName -> liveCountRef.get().apply(profileName),
|
||||||
|
quarantine,
|
||||||
|
outagePolicy);
|
||||||
|
// fleetd #422 follow-up: say which of the three model-gate states the daemon booted into.
|
||||||
|
log.info("model gate (fleetd #422): {}", Fleetd.modelGateCoverageLine(workers.modelGateState()));
|
||||||
|
// CB-504: under supervision (launchd/systemd) fleetd can start before herdr's socket
|
||||||
|
// exists. Wait, then degrade rather than die: serving with /healthz reporting "degraded" is
|
||||||
|
// strictly more useful than exiting.
|
||||||
|
Fleetd.HerdrAwaitOutcome herdrOutcome = Fleetd.awaitHerdr(herdr, ports.nanoClock(), ports.herdrPollWait());
|
||||||
|
boolean herdrUp = Fleetd.logHerdrWaitOutcomeAndShouldReap(herdrOutcome);
|
||||||
|
if (herdrUp) {
|
||||||
|
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
||||||
|
// with the previous process — reap those leaked orphans now, before we start serving.
|
||||||
|
workers.reapOrphanWorkers();
|
||||||
|
}
|
||||||
|
|
||||||
|
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
||||||
|
// CB-301-ext: worktree provisioning seam, optionally rooted at a configured directory.
|
||||||
|
// CB-303 part 2: context cap is opt-in and disabled (0) when absent/null.
|
||||||
|
int contextCap = 0;
|
||||||
|
if (cfg.lifecycle() != null && cfg.lifecycle().contextCap() != null
|
||||||
|
&& cfg.lifecycle().contextCap() > 0) {
|
||||||
|
contextCap = cfg.lifecycle().contextCap();
|
||||||
|
}
|
||||||
|
boolean clearAfterTurn = cfg.lifecycle() != null && cfg.lifecycle().clearAfterTurn();
|
||||||
|
SessionManager sessions = new SessionManager(workers,
|
||||||
|
new GitWorktrees(cfg.worktreeRoot(), cfg.worktreeGroup(), cfg.memberSkills()),
|
||||||
|
ports.nanoClock(), contextCap, clearAfterTurn);
|
||||||
|
liveCountRef.set(profileName -> Fleetd.liveSessionCount(sessions.roster(), profileName));
|
||||||
|
|
||||||
|
// Idle-sleep guard: hold an OS-level assertion against idle sleep while at least one
|
||||||
|
// member is live. No-op (never constructed) off macOS or when idleSleepGuard.enabled is
|
||||||
|
// explicitly false; the mechanism itself is additionally a no-op if 'caffeinate' cannot be
|
||||||
|
// started, so this can never fail a spawn, a release, or startup.
|
||||||
|
boolean idleSleepGuardEnabled = cfg.idleSleepGuard() == null || cfg.idleSleepGuard().isEnabled();
|
||||||
|
final IdleSleepGuard idleSleepGuard;
|
||||||
|
if (idleSleepGuardEnabled) {
|
||||||
|
idleSleepGuard = new IdleSleepGuard(new CaffeinateSleepAssertionMechanism(), sessions::size);
|
||||||
|
sessions.onAcquire(_ -> idleSleepGuard.recheck());
|
||||||
|
sessions.onRelease(_ -> idleSleepGuard.recheck());
|
||||||
|
} else {
|
||||||
|
idleSleepGuard = null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// CB-303 part 1: idle-ttl reaper — only when configured, defaults to disabled. FIRST of the
|
||||||
|
// recurring background loops to start — see this class's own javadoc for the full order.
|
||||||
|
final SessionReaper reaper;
|
||||||
|
if (cfg.lifecycle() != null
|
||||||
|
&& cfg.lifecycle().idleTtlSeconds() != null
|
||||||
|
&& cfg.lifecycle().idleTtlSeconds() > 0) {
|
||||||
|
reaper = new SessionReaper(sessions, cfg.lifecycle().idleTtlSeconds());
|
||||||
|
reaper.start();
|
||||||
|
} else {
|
||||||
|
reaper = null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// CB-530: every pane the config names as a lead, merged from `leaders:` and the legacy
|
||||||
|
// singular pin. PrimaryRegistry below still tracks ONE terminal — it addresses the push
|
||||||
|
// loop's nudges, which need a single destination — so it keeps the legacy pin.
|
||||||
|
Map<String, String> leadTerminals = cfg.leaderTerminals();
|
||||||
|
if (leadTerminals.size() > 1) {
|
||||||
|
log.info("leads: {} panes recognised {}", leadTerminals.size(), leadTerminals.values());
|
||||||
|
}
|
||||||
|
// CB-531/CB-579: discover leads by the tab labels the operator writes, one scanner per
|
||||||
|
// configured lead's own exact `tab:` label. fleetd #669: the same scan also recognises a
|
||||||
|
// configured collaborator's tab, so one herdr pass answers both.
|
||||||
|
final Supplier<Map<String, String>> leads;
|
||||||
|
final Supplier<Map<String, String>> collaboratorTerminals;
|
||||||
|
var leaders = cfg.fleet().leaders();
|
||||||
|
var collaboratorsConfig = cfg.fleet().collaborators();
|
||||||
|
if (!leaders.isEmpty() || !collaboratorsConfig.isEmpty()) {
|
||||||
|
Map<String, String> tabToName = new LinkedHashMap<>();
|
||||||
|
leaders.forEach((name, leader) -> {
|
||||||
|
if (leader != null && leader.tab() != null && !leader.tab().isBlank()) {
|
||||||
|
tabToName.put(leader.tab(), name);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
Map<String, String> collaboratorTabToName = new LinkedHashMap<>();
|
||||||
|
collaboratorsConfig.forEach((name, collaborator) -> {
|
||||||
|
if (collaborator != null && collaborator.tab() != null && !collaborator.tab().isBlank()) {
|
||||||
|
collaboratorTabToName.put(collaborator.tab(), name);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
// A collaborator-only fleet configures no `leaders:` entry to read a scan interval from
|
||||||
|
// — FleetConfig.Collaborator carries no scanIntervalSeconds of its own. Falling back to
|
||||||
|
// FleetConfig.Leader's own compact-constructor default keeps a collaborator-only
|
||||||
|
// deployment on the same rescan cadence as the default lead cadence, instead of
|
||||||
|
// inventing a second number for the same kind of scan.
|
||||||
|
int scanIntervalSeconds = leaders.isEmpty()
|
||||||
|
? 10
|
||||||
|
: leaders.values().iterator().next().scanIntervalSeconds();
|
||||||
|
// This must use the lead daemon: scanning member tabs would demote the lead to a worker.
|
||||||
|
LeadTabScanner scanner = new LeadTabScanner(herdr, tabToName, collaboratorTabToName, Set.of(),
|
||||||
|
TimeUnit.SECONDS.toNanos(scanIntervalSeconds), ports.nanoClock());
|
||||||
|
leads = scanner;
|
||||||
|
collaboratorTerminals = scanner::collaborators;
|
||||||
|
log.info("lead/collaborator scan: tabs {} host a lead, tabs {} host a collaborator "
|
||||||
|
+ "(rescan every {}s, shared fleet space)",
|
||||||
|
tabToName.keySet(), collaboratorTabToName.keySet(), scanIntervalSeconds);
|
||||||
|
} else {
|
||||||
|
leads = () -> leadTerminals;
|
||||||
|
collaboratorTerminals = Map::of;
|
||||||
|
}
|
||||||
|
leadsRef.set(leads);
|
||||||
|
collaboratorTerminalsRef.set(collaboratorTerminals);
|
||||||
|
|
||||||
|
// Constructed unconditionally — it is cheap and side-effect free — so a LeadRollover built
|
||||||
|
// below can relaunch a lead even on a boot where herdr was down for the ensureLeads() call.
|
||||||
|
LeadLauncher leadLauncher = new LeadLauncher(router.leadAgents(), router.leadSpaces(), cfg);
|
||||||
|
|
||||||
|
// CB-558: start any declared lead that is not already running. After the scanner is built,
|
||||||
|
// and only when herdr answered — the launcher's whole safety property is that it can count
|
||||||
|
// live leads first, and must never guess and risk a second orchestrator.
|
||||||
|
if (herdrUp && !leaders.isEmpty()) {
|
||||||
|
int launched = leadLauncher.ensureLeads();
|
||||||
|
if (launched > 0) {
|
||||||
|
log.info("lead auto-launch: {} lead(s) started", launched);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// CB-548: config-declared architect slots. Nothing here spawns a slot; the terminal → slot
|
||||||
|
// binding is owned by the registry and empty at startup.
|
||||||
|
MemberRegistry members = MemberRegistry.live(() -> config.get().fleet());
|
||||||
|
sessions.setMemberLifecycle(members);
|
||||||
|
if (!members.slots().isEmpty()) {
|
||||||
|
log.info("member slots: {} configured {} — none bound yet (a slot is idle until the "
|
||||||
|
+ "spawn lifecycle binds a live terminal to it)",
|
||||||
|
members.slots().size(), members.slots().keySet());
|
||||||
|
}
|
||||||
|
|
||||||
|
// Status-gated injector (CB-103): the single writer into workers, fed by a poller.
|
||||||
|
// CB-106: a confirmed turn completion resolves a blocked send whose worker never replied.
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
// CB-578 stage A: classify a completion-fallback scrape that matches a profile's configured
|
||||||
|
// usage-limit refusal as BACKEND_EXHAUSTED rather than handing it back as a real answer.
|
||||||
|
LiveExhaustedPatterns liveExhaustedPatterns = Fleetd.liveExhaustedPatterns(config);
|
||||||
|
ExhaustedPatternLookup exhaustedPatterns = Fleetd.exhaustedPatternLookup(sessions::roster, liveExhaustedPatterns);
|
||||||
|
// The startup coverage line still reports the boot-time snapshot only.
|
||||||
|
Set<String> exhaustedConfiguredAtStartup = cfg.profiles().entrySet().stream()
|
||||||
|
.filter(e -> e.getValue().hasExhaustedPattern())
|
||||||
|
.map(Map.Entry::getKey)
|
||||||
|
.collect(Collectors.toCollection(LinkedHashSet::new));
|
||||||
|
log.info("backend-exhausted classification (CB-578 stage A): {}",
|
||||||
|
Fleetd.exhaustedPatternCoverageLine(cfg.profiles().keySet(), exhaustedConfiguredAtStartup));
|
||||||
|
// fleetd #201 Unit 5: classify a completion-fallback scrape that matches a profile's
|
||||||
|
// configured backend-error refusal (a credential outage, a provider 5xx) as a backend error
|
||||||
|
// rather than handing it back as a real answer.
|
||||||
|
Map<String, Pattern> errorPatternsByProfile = new LinkedHashMap<>();
|
||||||
|
cfg.profiles().forEach((name, profile) -> {
|
||||||
|
if (profile.hasErrorPattern()) {
|
||||||
|
errorPatternsByProfile.put(name, Pattern.compile(profile.errorPattern()));
|
||||||
|
}
|
||||||
|
});
|
||||||
|
BackendErrorPatternLookup backendErrorPatterns = Fleetd.backendErrorPatternLookup(sessions::roster,
|
||||||
|
errorPatternsByProfile);
|
||||||
|
log.info("backend-error classification (fleetd #201 Unit 5): {}",
|
||||||
|
Fleetd.errorPatternCoverageLine(cfg.profiles().keySet(), errorPatternsByProfile.keySet()));
|
||||||
|
// CB-578 stage B: on a classification that actually wins, quarantine the exhausted profile's
|
||||||
|
// CREDENTIAL, so a profile sharing that credential is refused too, not just the one that
|
||||||
|
// happened to report it.
|
||||||
|
Map<String, String> quarantineReasonByCredential = new ConcurrentHashMap<>();
|
||||||
|
// fleetd #175: point the forwarding sink handed to OpenCodeLauncher above at the real one,
|
||||||
|
// now that `sessions` exists to resolve target -> session -> profile.
|
||||||
|
ExhaustionSink exhaustionSink = Fleetd.publishExhaustionSink(exhaustionSinkRef, sessions, config,
|
||||||
|
quarantine, quarantineReasonByCredential, cfg);
|
||||||
|
// fleetd #201 Unit 5: the production BackendErrorSink needs `pushLoop` (built further below,
|
||||||
|
// after `sessions`) to tell a lead about an incident or an unmapped target — the same
|
||||||
|
// construction-order cycle `exhaustionSinkRef` breaks above, broken the same way: a mutable
|
||||||
|
// holder set once `pushLoop` exists, read lazily from inside the lambda built here.
|
||||||
|
AtomicReference<ReplyPushLoop> pushLoopRef = new AtomicReference<>();
|
||||||
|
BackendErrorSink backendErrorSink = Fleetd.backendErrorSink(sessions, () -> config.get().profiles(),
|
||||||
|
outagePolicy, pushLoopRef::get);
|
||||||
|
AgentControl agents = router.memberAgents();
|
||||||
|
CompletionResolver completion = new CompletionResolver(agents, rendezvous, exhaustedPatterns,
|
||||||
|
exhaustionSink, backendErrorPatterns, backendErrorSink, ports.nanoClock(),
|
||||||
|
Fleetd.worktreeBranchLookup(sessions::roster));
|
||||||
|
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||||
|
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||||
|
MemberPresence presence = sessions.asPresence();
|
||||||
|
TurnListener turnListener = Fleetd.turnListener(completion, sessions);
|
||||||
|
Predicate<String> deliverable = Fleetd.deliverableTo(presence, leads, collaboratorTerminals);
|
||||||
|
// fleetd #556: registration is wired directly to `completion`, not folded into the
|
||||||
|
// `turnListener` fan-out above — so it survives `sessions.onDelivered` (or any future
|
||||||
|
// listener) throwing, regardless of call order.
|
||||||
|
Injector injector = new Injector(router, turnListener, deliverable,
|
||||||
|
presence::forget, Fleetd.turnRegistrar(completion));
|
||||||
|
StatusPoller poller = new StatusPoller(router, injector, Injector.POLL_INTERVAL_MILLIS);
|
||||||
|
poller.start(); // SECOND of the recurring background loops to start, after the reaper.
|
||||||
|
|
||||||
|
// CB-307: reply inbox. A broker: block selects the AMQP-backed durable adapter; absent (or
|
||||||
|
// unusable), fleetd stays soft-state on the in-memory inbox. The AMQP inbox owns a broker
|
||||||
|
// connection, so keep the reference to close it in the ordered shutdown hook.
|
||||||
|
final ReplyInbox replyInbox = Fleetd.selectReplyInbox(cfg.broker(), ports.environment(),
|
||||||
|
ports.replyInboxOpener());
|
||||||
|
// CB-637: this daemon's lead-to-lead mailbox on the SHARED coordination vhost — a separate
|
||||||
|
// broker from the reply inbox by design. Absent a coordinator: block this is null and every
|
||||||
|
// lead path below is simply not wired, exactly the behaviour before this ticket. It owns a
|
||||||
|
// broker connection, so keep the reference for the ordered shutdown hook.
|
||||||
|
final LeadChannelHandle leadMailbox = Fleetd.openLeadMailbox(cfg.coordinator(), ports.environment(),
|
||||||
|
ports.leadMailboxOpener());
|
||||||
|
// CB-307: learn the primary's terminal from orchestration tool calls (or pin from config).
|
||||||
|
String pinnedPrimaryTerminal = cfg.primary() != null ? cfg.primary().terminal() : null;
|
||||||
|
PrimaryRegistry primaryRegistry = new PrimaryRegistry(pinnedPrimaryTerminal);
|
||||||
|
// CB-532: `primary.terminal` is superseded and no longer needed for either of its jobs.
|
||||||
|
if (pinnedPrimaryTerminal != null && !pinnedPrimaryTerminal.isBlank()) {
|
||||||
|
log.warn("primary.terminal is DEPRECATED (CB-532) and can be deleted: identity now comes "
|
||||||
|
+ "from leaders:/leadScan:, and reply nudges follow the lead that delegated. "
|
||||||
|
+ "It still works, and is still the fallback nudge destination when a restart "
|
||||||
|
+ "has lost the delegation map. Its pushReminders/pushBackoffMs stay valid.");
|
||||||
|
}
|
||||||
|
// CB-307: active push-to-primary loop — nudge the primary when replies land without an
|
||||||
|
// open fleet_send. Uses its own lightweight scheduled executor, separate from the injector.
|
||||||
|
int maxReminders = cfg.primary() != null ? cfg.primary().remindersOrDefault() : 5;
|
||||||
|
long backoffMs = cfg.primary() != null ? cfg.primary().backoffMsOrDefault() : 15_000L;
|
||||||
|
var pushScheduler = ports.newScheduler("bridge-push-");
|
||||||
|
// CB-502: the registry is built before the service and the push loop so send/reply outcomes
|
||||||
|
// are counted at their single funnel rather than at each of the two caller-facing surfaces.
|
||||||
|
Metrics metrics = FleetMetrics.create(sessions, replyInbox);
|
||||||
|
var pushLoop = new ReplyPushLoop(primaryRegistry, router.leadAgents(), replyInbox,
|
||||||
|
pushScheduler, maxReminders, backoffMs, metrics);
|
||||||
|
// fleetd #201 Unit 5: point the forwarding holder captured by the backendErrorSink lambda
|
||||||
|
// above at the real push loop, now that it exists.
|
||||||
|
pushLoopRef.set(pushLoop);
|
||||||
|
// CB-551: idle-lead heartbeat. Opt-in; absent `leadHeartbeat:` this is never constructed, so
|
||||||
|
// an upgraded daemon cannot silently start spending subscription on nudging an idle lead.
|
||||||
|
// THIRD of the recurring background loops to start (optional).
|
||||||
|
final LeadHeartbeatLoop heartbeat;
|
||||||
|
var heartbeatScheduler = ports.newScheduler("bridge-heartbeat-");
|
||||||
|
if (cfg.leadHeartbeat() != null) {
|
||||||
|
var hb = cfg.leadHeartbeat();
|
||||||
|
var leadContextGauge = new LeadContextGauge();
|
||||||
|
// fleetd #621: the context-high notice's own wording must track this same effective
|
||||||
|
// value — LeadRollover.confirm(...) already gates the roll on it (LeadRollover.java:480),
|
||||||
|
// and absent `leadRollover:` entirely the roll is unusable regardless (NOT_CONFIGURED),
|
||||||
|
// so `true` (the FleetConfig.LeadRollover default) is the safe, byte-identical fallback.
|
||||||
|
// Carried in from Fleetd.main when #612 Unit A merged main: #622 added this line to the
|
||||||
|
// block Unit A had already moved here, so the merge would otherwise have silently
|
||||||
|
// dropped it — with a fully green suite, because nothing pins it (see the follow-up issue).
|
||||||
|
boolean requireOperatorConfirm = cfg.leadRollover() == null || cfg.leadRollover().requireOperatorConfirm();
|
||||||
|
heartbeat = new LeadHeartbeatLoop(primaryRegistry, router.leadAgents(), replyInbox, sessions::roster,
|
||||||
|
pushLoop, heartbeatScheduler, ports.nanoClock(),
|
||||||
|
TimeUnit.SECONDS.toNanos(hb.idleAfterSeconds()), hb.backoffMs(), hb.quietNudgeCap(),
|
||||||
|
metrics,
|
||||||
|
Fleetd.leadContextSource(leadContextGauge, router.leadAgents(), leads,
|
||||||
|
Fleetd.leadConfigDirLookup(() -> config.get().profiles(), leaders),
|
||||||
|
Fleetd.leadContextWindowLookup(() -> config.get().profiles(), leaders)),
|
||||||
|
Boolean.TRUE.equals(hb.contextHighNudge()), requireOperatorConfirm);
|
||||||
|
heartbeat.start();
|
||||||
|
} else {
|
||||||
|
heartbeat = null;
|
||||||
|
heartbeatScheduler.shutdownNow();
|
||||||
|
}
|
||||||
|
// fleetd #480: lead rollover. Opt-in; absent `leadRollover:` this is never constructed.
|
||||||
|
LeadRollover leadRollover = Fleetd.leadRollover(cfg, router.leadAgents(), router.leadSpaces(),
|
||||||
|
leadLauncher, config, leads);
|
||||||
|
MessageService messages = new MessageService(router, injector, rendezvous, replyInbox,
|
||||||
|
pushLoop, metrics);
|
||||||
|
|
||||||
|
// Health is a slow whole-fleet observer. Keep it separate from the 250ms delivery poller.
|
||||||
|
// FOURTH of the recurring background loops to start (optional).
|
||||||
|
final FleetHealthMonitor healthMonitor;
|
||||||
|
var healthScheduler = ports.newScheduler("bridge-health-");
|
||||||
|
if (cfg.health() != null && cfg.health().isEnabled()) {
|
||||||
|
// CB-580: a member found GONE/NEVER_READY must fail whatever ticket is waiting on it.
|
||||||
|
healthMonitor = new FleetHealthMonitor(agents, sessions::roster, messages, healthScheduler,
|
||||||
|
ports.nanoClock(), ports.wallClockNanos(),
|
||||||
|
cfg.health().intervalOrDefault(),
|
||||||
|
cfg.health().workingSuspectAfterOrDefault(), Fleetd.healthFailTarget(messages));
|
||||||
|
String coverage = FleetHealthMonitor.coverage(true,
|
||||||
|
cfg.health().notifications() != null && cfg.health().notifications().configured());
|
||||||
|
if ("detection-only".equals(coverage)) {
|
||||||
|
log.warn("fleet health: {} (no notification sink configured)", coverage);
|
||||||
|
} else {
|
||||||
|
log.info("fleet health: {}", coverage);
|
||||||
|
}
|
||||||
|
healthMonitor.start();
|
||||||
|
} else {
|
||||||
|
healthMonitor = null;
|
||||||
|
healthScheduler.shutdownNow();
|
||||||
|
}
|
||||||
|
|
||||||
|
// CB-520: the reply inbox only consumes for agents this gateway owns. own on acquire,
|
||||||
|
// release on teardown. Do this before CB-516 so the inbox is owned before any reply can land.
|
||||||
|
sessions.onAcquire(replyInbox::own);
|
||||||
|
// CB-516: releasing a worker must fail whatever send was waiting on it.
|
||||||
|
sessions.onRelease(Fleetd.releaseCleanup(messages, replyInbox, primaryRegistry));
|
||||||
|
|
||||||
|
// MCP server face (CB-105): fleet_send/fleet_reply/fleet_status, mounted at /mcp.
|
||||||
|
// Caller identity is resolved from the connection (peer PID → herdr pane), not arguments.
|
||||||
|
ConnectionIdentity identity = new ConnectionIdentity(
|
||||||
|
new PaneLocator(herdr, memberHerdr), new LsofPeerPidLookup(), new LsofProcessCwdLookup());
|
||||||
|
|
||||||
|
// fleetd #669 Unit D: a live spawned member resolves as its own role, whatever a tab map
|
||||||
|
// says about the same terminal. fleetd #702: SessionManager.spawnedMemberRole also answers
|
||||||
|
// for a pane mid-teardown, not only one still in the registry — see its javadoc.
|
||||||
|
Function<String, MemberRole> spawnedMemberRole = sessions::spawnedMemberRole;
|
||||||
|
|
||||||
|
// CB-501: one resolver behind both entry paths. Worker identity still comes from the
|
||||||
|
// connection and is never token-gated, so enabling token mode cannot lock the fleet out.
|
||||||
|
final CallerResolver callers;
|
||||||
|
if (cfg.auth().tokenMode()) {
|
||||||
|
String token = ports.environment().get(cfg.auth().tokenEnv());
|
||||||
|
if (token == null || token.isBlank()) {
|
||||||
|
throw new IllegalStateException("auth.mode=token but env var " + cfg.auth().tokenEnv()
|
||||||
|
+ " is unset or empty — export it before starting fleetd");
|
||||||
|
}
|
||||||
|
callers = CallerResolver.withLeadsAndMembers(identity, true, token, leads, members,
|
||||||
|
spawnedMemberRole, collaboratorTerminals);
|
||||||
|
log.info("auth: token mode (bearer required for non-worker callers, env {})",
|
||||||
|
cfg.auth().tokenEnv());
|
||||||
|
} else {
|
||||||
|
callers = CallerResolver.withLeadsAndMembers(identity, false, null, leads, members,
|
||||||
|
spawnedMemberRole, collaboratorTerminals);
|
||||||
|
log.info("auth: loopback-trust (any loopback non-worker caller is the primary)");
|
||||||
|
}
|
||||||
|
|
||||||
|
FleetMcp.QuarantineSource quarantineSource = Fleetd.quarantineSource(config, quarantine,
|
||||||
|
liveExhaustedPatterns, quarantineReasonByCredential);
|
||||||
|
FleetMcp.OutageSource outageSource = new FleetMcp.OutageSource(profile -> {
|
||||||
|
var configured = config.get().profiles().get(profile);
|
||||||
|
return configured == null ? null : configured.effectiveCredentialId();
|
||||||
|
}, outagePolicy);
|
||||||
|
|
||||||
|
FleetMcp.LoopHealthSource loopHealth = Fleetd.loopHealthSource(poller, reaper);
|
||||||
|
FleetMcp mcp = new FleetMcp(messages, workers, sessions, identity, presence,
|
||||||
|
primaryRegistry, callers, FleetMcp.AuthorizationMode.ENFORCED, metrics,
|
||||||
|
Fleetd.capacitySource(config, cfg, profile -> liveCountRef.get().apply(profile)),
|
||||||
|
Fleetd.healthCoverageSource(config),
|
||||||
|
loopHealth,
|
||||||
|
quarantineSource,
|
||||||
|
leadMailbox,
|
||||||
|
outageSource,
|
||||||
|
new FleetMcp.LeadSeatSource(Fleetd.leadSeatLookup(() -> config.get().profiles(), leaders, leads)),
|
||||||
|
Fleetd.leadConfigDirSource(() -> config.get().profiles(), leaders),
|
||||||
|
cfg.coordinator() == null ? List.of() : cfg.coordinator().peers(),
|
||||||
|
leadRollover);
|
||||||
|
|
||||||
|
// CB-637: the receive half. Only constructed when a lead mailbox actually opened.
|
||||||
|
final LeadCoordLoop leadCoordLoop;
|
||||||
|
final ScheduledExecutorService leadCoordSchedulerRef;
|
||||||
|
if (leadMailbox != null) {
|
||||||
|
var leadCoordScheduler = ports.newScheduler("bridge-leadcoord-");
|
||||||
|
leadCoordLoop = new LeadCoordLoop(leadMailbox, router.leadAgents(), leads, leadCoordScheduler,
|
||||||
|
LEAD_COORD_INTERVAL_MS);
|
||||||
|
leadCoordLoop.start(); // FIFTH of the recurring background loops to start (optional).
|
||||||
|
leadCoordSchedulerRef = leadCoordScheduler;
|
||||||
|
} else {
|
||||||
|
leadCoordLoop = null;
|
||||||
|
leadCoordSchedulerRef = null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// CB-559: opt-in config reload. With no `configReload:` block nothing is constructed.
|
||||||
|
// SIXTH of the recurring background loops to start (optional).
|
||||||
|
final ConfigWatcher configWatcher;
|
||||||
|
if (cfg.configReload() != null && cfg.configReload().isEnabled()) {
|
||||||
|
configWatcher = new ConfigWatcher(config, cfg.configReload().intervalSeconds());
|
||||||
|
configWatcher.start();
|
||||||
|
} else {
|
||||||
|
configWatcher = null;
|
||||||
|
}
|
||||||
|
|
||||||
|
// CB-303 part 3: single ordered shutdown hook — see FleetdRuntime#close() for the statements
|
||||||
|
// this used to be. Registered here, at the exact point `main` used to register it: after
|
||||||
|
// configWatcher, before the Javalin app exists (see this class's own javadoc for why).
|
||||||
|
FleetdRuntime runtime = new FleetdRuntime(cfg, sessions, router, poller, messages, pushLoop, heartbeat,
|
||||||
|
leadCoordLoop, leadCoordSchedulerRef, healthMonitor, configWatcher, mcp, reaper, idleSleepGuard,
|
||||||
|
replyInbox, leadMailbox, completion, injector);
|
||||||
|
ports.addShutdownHook(runtime::close);
|
||||||
|
|
||||||
|
// CB-185: give FleetApp both daemons — /healthz must require both to answer and
|
||||||
|
// GET /sessions must merge across both, or a down/unpolled member daemon is invisible.
|
||||||
|
Javalin app = new FleetApp(herdr, memberHerdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||||
|
callers, metrics, deliverable,
|
||||||
|
() -> MemberCredentialPolicyView.of(config.get().memberCredentials()),
|
||||||
|
quarantineSource, outageSource, loopHealth).build();
|
||||||
|
runtime.attachApp(app);
|
||||||
|
ports.startHttp(app, cfg.bind().host(), cfg.bind().port()); // HTTP starts LAST, always.
|
||||||
|
log.info("fleetd listening on {}:{}, herdr socket {}",
|
||||||
|
cfg.bind().host(), cfg.bind().port(), socket);
|
||||||
|
return runtime;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,166 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigWatcher;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.health.FleetHealthMonitor;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||||
|
import dev.ltms.fleet.inject.CompletionResolver;
|
||||||
|
import dev.ltms.fleet.inject.Injector;
|
||||||
|
import dev.ltms.fleet.inject.StatusPoller;
|
||||||
|
import dev.ltms.fleet.mcp.FleetMcp;
|
||||||
|
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||||
|
import dev.ltms.fleet.msg.LeadCoordLoop;
|
||||||
|
import dev.ltms.fleet.msg.LeadHeartbeatLoop;
|
||||||
|
import dev.ltms.fleet.msg.MessageService;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||||
|
import dev.ltms.fleet.power.IdleSleepGuard;
|
||||||
|
import dev.ltms.fleet.session.SessionManager;
|
||||||
|
import dev.ltms.fleet.session.SessionReaper;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.slf4j.Logger;
|
||||||
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 Unit A: the assembled daemon. {@link FleetdAssembly#assembleAndStart} builds exactly
|
||||||
|
* one of these and hands it to {@code ResourcePorts.addShutdownHook}; {@link #close()} is the
|
||||||
|
* single ordered shutdown, moved verbatim out of {@code Fleetd.main}'s old shutdown-hook
|
||||||
|
* {@code Thread} body — same statements, same order, see that method's javadoc.
|
||||||
|
*
|
||||||
|
* <p><strong>Owns the real production objects, never a copy.</strong> Every package-private
|
||||||
|
* accessor below returns the identical instance the running daemon is using. That is the entire
|
||||||
|
* point of this class existing (see fleetd #612's problem statement): a test that inspected a
|
||||||
|
* snapshot built alongside the real objects could pass while production silently received
|
||||||
|
* something else — the exact shape of the #602/#606 defect this ticket exists to stop from
|
||||||
|
* recurring one call site at a time. Nothing here is rebuilt or copied for a test's benefit.
|
||||||
|
*/
|
||||||
|
final class FleetdRuntime implements AutoCloseable {
|
||||||
|
|
||||||
|
private static final Logger log = LoggerFactory.getLogger(FleetdRuntime.class);
|
||||||
|
|
||||||
|
private final FleetConfig cfg;
|
||||||
|
private final SessionManager sessions;
|
||||||
|
private final HerdrRouter router;
|
||||||
|
private final StatusPoller poller;
|
||||||
|
private final MessageService messages;
|
||||||
|
private final ReplyPushLoop pushLoop;
|
||||||
|
private final LeadHeartbeatLoop heartbeat; // nullable — leadHeartbeat: opt-in
|
||||||
|
private final LeadCoordLoop leadCoordLoop; // nullable — coordinator: opt-in
|
||||||
|
private final ScheduledExecutorService leadCoordScheduler; // nullable, paired with leadCoordLoop
|
||||||
|
private final FleetHealthMonitor healthMonitor; // nullable — health.enabled opt-in
|
||||||
|
private final ConfigWatcher configWatcher; // nullable — configReload.enabled opt-in
|
||||||
|
private final FleetMcp mcp;
|
||||||
|
private final SessionReaper reaper; // nullable — lifecycle.idleTtlSeconds opt-in
|
||||||
|
private final IdleSleepGuard idleSleepGuard; // nullable — idleSleepGuard.enabled: false
|
||||||
|
private final ReplyInbox replyInbox;
|
||||||
|
private final LeadChannelHandle leadMailbox; // nullable — coordinator: opt-in
|
||||||
|
private final CompletionResolver completion;
|
||||||
|
private final Injector injector;
|
||||||
|
/**
|
||||||
|
* Not final: {@code Fleetd.main}'s shutdown hook was registered <em>before</em> the Javalin
|
||||||
|
* {@code FleetApp} was built and started — see {@link FleetdAssembly#assembleAndStart}'s javadoc
|
||||||
|
* for why that order could not be preserved AND have this constructor take {@code app}. {@link
|
||||||
|
* #attachApp} is called immediately after the real app is built, still before HTTP starts
|
||||||
|
* listening, so this is set long before any test or caller could observe it unset.
|
||||||
|
*/
|
||||||
|
private Javalin app;
|
||||||
|
|
||||||
|
FleetdRuntime(FleetConfig cfg, SessionManager sessions, HerdrRouter router, StatusPoller poller,
|
||||||
|
MessageService messages, ReplyPushLoop pushLoop, LeadHeartbeatLoop heartbeat,
|
||||||
|
LeadCoordLoop leadCoordLoop, ScheduledExecutorService leadCoordScheduler,
|
||||||
|
FleetHealthMonitor healthMonitor, ConfigWatcher configWatcher, FleetMcp mcp,
|
||||||
|
SessionReaper reaper, IdleSleepGuard idleSleepGuard, ReplyInbox replyInbox,
|
||||||
|
LeadChannelHandle leadMailbox, CompletionResolver completion, Injector injector) {
|
||||||
|
this.cfg = cfg;
|
||||||
|
this.sessions = sessions;
|
||||||
|
this.router = router;
|
||||||
|
this.poller = poller;
|
||||||
|
this.messages = messages;
|
||||||
|
this.pushLoop = pushLoop;
|
||||||
|
this.heartbeat = heartbeat;
|
||||||
|
this.leadCoordLoop = leadCoordLoop;
|
||||||
|
this.leadCoordScheduler = leadCoordScheduler;
|
||||||
|
this.healthMonitor = healthMonitor;
|
||||||
|
this.configWatcher = configWatcher;
|
||||||
|
this.mcp = mcp;
|
||||||
|
this.reaper = reaper;
|
||||||
|
this.idleSleepGuard = idleSleepGuard;
|
||||||
|
this.replyInbox = replyInbox;
|
||||||
|
this.leadMailbox = leadMailbox;
|
||||||
|
this.completion = completion;
|
||||||
|
this.injector = injector;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** See the {@link #app} field doc for why this is a late-bound setter rather than a constructor arg. */
|
||||||
|
void attachApp(Javalin app) {
|
||||||
|
this.app = app;
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- package-private accessors: the SAME instances this runtime owns, never a copy ---------
|
||||||
|
|
||||||
|
SessionManager sessions() { return sessions; }
|
||||||
|
HerdrRouter router() { return router; }
|
||||||
|
StatusPoller poller() { return poller; }
|
||||||
|
MessageService messages() { return messages; }
|
||||||
|
ReplyPushLoop pushLoop() { return pushLoop; }
|
||||||
|
LeadHeartbeatLoop heartbeat() { return heartbeat; }
|
||||||
|
LeadCoordLoop leadCoordLoop() { return leadCoordLoop; }
|
||||||
|
FleetHealthMonitor healthMonitor() { return healthMonitor; }
|
||||||
|
ConfigWatcher configWatcher() { return configWatcher; }
|
||||||
|
FleetMcp mcp() { return mcp; }
|
||||||
|
SessionReaper reaper() { return reaper; }
|
||||||
|
IdleSleepGuard idleSleepGuard() { return idleSleepGuard; }
|
||||||
|
ReplyInbox replyInbox() { return replyInbox; }
|
||||||
|
LeadChannelHandle leadMailbox() { return leadMailbox; }
|
||||||
|
CompletionResolver completion() { return completion; }
|
||||||
|
Injector injector() { return injector; }
|
||||||
|
Javalin app() { return app; }
|
||||||
|
|
||||||
|
/**
|
||||||
|
* CB-303 part 3: the single ordered shutdown. Moved verbatim out of {@code Fleetd.main}'s
|
||||||
|
* shutdown-hook {@code Thread} body (fleetd #612 Unit A) — drain sessions first while herdr is
|
||||||
|
* still open (so releases reach the daemon), then stop poller/message/mcp/reaper, and close
|
||||||
|
* herdr last, exactly as before. {@code Fleetd.main} never calls this directly; it hands the
|
||||||
|
* reference to {@code ResourcePorts.addShutdownHook} the moment this runtime exists, the same
|
||||||
|
* point it used to register the hook {@code Thread} itself.
|
||||||
|
*/
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
sessions.close(cfg.lifecycle() != null ? cfg.lifecycle().drainTimeoutSeconds() : null);
|
||||||
|
poller.stop();
|
||||||
|
messages.close();
|
||||||
|
pushLoop.close();
|
||||||
|
if (heartbeat != null) heartbeat.close(); // CB-551: stop the idle-lead heartbeat scheduler
|
||||||
|
if (leadCoordLoop != null) leadCoordLoop.close(); // CB-637: stop delivering peer-lead messages
|
||||||
|
if (leadCoordScheduler != null) leadCoordScheduler.shutdownNow();
|
||||||
|
if (healthMonitor != null) healthMonitor.stop();
|
||||||
|
if (configWatcher != null) configWatcher.stop(); // CB-559: stop polling the config file
|
||||||
|
mcp.close();
|
||||||
|
if (reaper != null) reaper.stop();
|
||||||
|
// Idle-sleep guard: release unconditionally, even though sessions.close() above already
|
||||||
|
// drained every session (and each release already drove the live count to 0, which
|
||||||
|
// releases the guard's assertion on its own) — this is the backstop for a drain that was
|
||||||
|
// itself interrupted or threw, so no caffeinate child ever outlives the daemon.
|
||||||
|
if (idleSleepGuard != null) idleSleepGuard.close();
|
||||||
|
// Release the broker connection last among message resources (no-op for the in-memory inbox).
|
||||||
|
if (replyInbox instanceof AutoCloseable closeable) {
|
||||||
|
try {
|
||||||
|
closeable.close();
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.debug("reply inbox close: {}", e.toString());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// CB-637: the coordination connection goes with it — after the loop that reads it has
|
||||||
|
// stopped, so no tick can be mid-ack against a closed channel.
|
||||||
|
if (leadMailbox != null) {
|
||||||
|
try {
|
||||||
|
leadMailbox.close();
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.debug("lead mailbox close: {}", e.toString());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
router.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,80 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 Unit A: every boot-time side effect {@link FleetdAssembly#assembleAndStart} performs
|
||||||
|
* that a real daemon must do for real, and a test must not — read the process environment, connect
|
||||||
|
* a herdr client, open a broker (the reply inbox, the lead mailbox), read a clock, start a
|
||||||
|
* background scheduler, register the JVM shutdown hook, and bind the HTTP server.
|
||||||
|
*
|
||||||
|
* <p>{@link #system()} is the one production implementation ({@code SystemResourcePorts}), wired
|
||||||
|
* verbatim from what {@code Fleetd.main} used to call directly at each of these call sites. A test
|
||||||
|
* builds its own implementation instead of receiving an inert default from this interface —
|
||||||
|
* deliberately, there is no {@code ResourcePorts.none()}. fleetd #612's whole problem is a call
|
||||||
|
* site quietly swapped for an inert variant that still compiles; adding one here, even for tests,
|
||||||
|
* would hand a future edit to {@code FleetdAssembly} the exact compiling substitute this ticket
|
||||||
|
* exists to rule out. A test that wants an inert resource writes its own fake and owns that
|
||||||
|
* decision explicitly.
|
||||||
|
*/
|
||||||
|
public interface ResourcePorts {
|
||||||
|
|
||||||
|
/** The process environment. Production: {@link System#getenv()}. */
|
||||||
|
Map<String, String> environment();
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Connect a herdr client bound to {@code socketPath}. Production returns a real
|
||||||
|
* {@code UnixSocketHerdrClient} — connection-per-call, so this itself never touches the socket.
|
||||||
|
*/
|
||||||
|
HerdrClient connectHerdr(Path socketPath);
|
||||||
|
|
||||||
|
/** The reply-inbox AMQP opener (CB-307). Production: {@link Fleetd#replyInboxOpener()}. */
|
||||||
|
Fleetd.AmqpOpener replyInboxOpener();
|
||||||
|
|
||||||
|
/** The lead-mailbox AMQP opener (CB-637). Production: {@link Fleetd#leadMailboxOpener()}. */
|
||||||
|
Fleetd.LeadMailboxOpener leadMailboxOpener();
|
||||||
|
|
||||||
|
/** A monotonic elapsed-time clock. Production: {@link System#nanoTime()}. */
|
||||||
|
LongSupplier nanoClock();
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #629: the per-poll wait {@code FleetdAssembly#assembleAndStart} passes to {@code
|
||||||
|
* Fleetd#awaitHerdr} while polling for herdr's socket. Production: {@link
|
||||||
|
* Fleetd#sleepHerdrPoll()} — a real {@code Thread.sleep}. {@link #nanoClock()} alone is not
|
||||||
|
* enough to make {@code awaitHerdr}'s deadline controllable: the old call site passed {@code
|
||||||
|
* Fleetd::sleepHerdrPoll} directly, hardcoded, so a test that injected a fake clock still had
|
||||||
|
* to wait out the real sleep between each poll to ever reach the deadline — the clock looked
|
||||||
|
* injected and was not actually controllable. A test supplies a no-op that advances its own
|
||||||
|
* injected {@link #nanoClock()} instead, so the deadline becomes reachable without any real
|
||||||
|
* wall-clock time passing.
|
||||||
|
*/
|
||||||
|
Runnable herdrPollWait();
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A wall-clock reading, in nanoseconds. Production: {@code System.currentTimeMillis()}
|
||||||
|
* converted to nanoseconds. Kept separate from {@link #nanoClock()} because {@link
|
||||||
|
* dev.ltms.fleet.health.FleetHealthMonitor} needs both — one monotonic clock for elapsed-time
|
||||||
|
* decisions, one wall clock to detect and correct for a macOS sleep freezing the monotonic one.
|
||||||
|
*/
|
||||||
|
LongSupplier wallClockNanos();
|
||||||
|
|
||||||
|
/** A dedicated single-thread scheduler; production names its (virtual) thread {@code purpose}. */
|
||||||
|
ScheduledExecutorService newScheduler(String purpose);
|
||||||
|
|
||||||
|
/** Register a JVM shutdown hook that runs {@code hook} on JVM exit. */
|
||||||
|
void addShutdownHook(Runnable hook);
|
||||||
|
|
||||||
|
/** Bind and start the HTTP server. */
|
||||||
|
void startHttp(Javalin app, String host, int port);
|
||||||
|
|
||||||
|
/** The real production ports: a live herdr socket, a live broker, real threads, a real bind. */
|
||||||
|
static ResourcePorts system() {
|
||||||
|
return new SystemResourcePorts();
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,71 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.herdr.UnixSocketHerdrClient;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 Unit A: the one production {@link ResourcePorts} — every method here is the exact
|
||||||
|
* call {@code Fleetd.main} used to make directly at each of these sites before this ticket.
|
||||||
|
* Package-private: obtained only through {@link ResourcePorts#system()}.
|
||||||
|
*/
|
||||||
|
final class SystemResourcePorts implements ResourcePorts {
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return System.getenv();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return UnixSocketHerdrClient.connect(socketPath, new ObjectMapper());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return Fleetd.replyInboxOpener();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return Fleetd.leadMailboxOpener();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
return Fleetd::sleepHerdrPoll;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return () -> TimeUnit.MILLISECONDS.toNanos(System.currentTimeMillis());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor(r -> Thread.ofVirtual().name(purpose).unstarted(r));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
Runtime.getRuntime().addShutdownHook(new Thread(hook));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
app.start(host, port);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,5 +1,7 @@
|
|||||||
package dev.ltms.fleet.auth;
|
package dev.ltms.fleet.auth;
|
||||||
|
|
||||||
|
import java.util.function.Predicate;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The authorization table (CB-505), stated once and enforced on both entry paths.
|
* The authorization table (CB-505), stated once and enforced on both entry paths.
|
||||||
*
|
*
|
||||||
@@ -20,16 +22,22 @@ public final class Authz {
|
|||||||
SPAWN,
|
SPAWN,
|
||||||
/** Tear a worker peer down. */
|
/** Tear a worker peer down. */
|
||||||
STOP,
|
STOP,
|
||||||
/** Deliver a turn to a session (or answer a worker's question). */
|
/** Deliver a turn to a local session, addressed by {@code sessionId}. */
|
||||||
SEND,
|
SEND,
|
||||||
|
/** Resolve a worker's blocked question and resume its turn, addressed by {@code turnId}. */
|
||||||
|
ANSWER,
|
||||||
|
/** Address a peer lead on another daemon over the coordination broker, by {@code coordId}. */
|
||||||
|
COORD_SEND,
|
||||||
/** A worker's terminal reply for its own turn. */
|
/** A worker's terminal reply for its own turn. */
|
||||||
REPLY,
|
REPLY,
|
||||||
/** A worker's mid-turn question to the primary. */
|
/** A worker's mid-turn question to the primary. */
|
||||||
ASK,
|
ASK,
|
||||||
/** Collect held replies from a session's inbox. */
|
/** Collect held replies from a session's inbox. */
|
||||||
DRAIN,
|
DRAIN,
|
||||||
/** Read-only observation: status, roster, profiles, task polling. */
|
/** Read-only roster, profile, and identity observation: no ticket, task, or turn state. */
|
||||||
READ,
|
READ,
|
||||||
|
/** Poll a ticket, or read a session's status. */
|
||||||
|
TASK_READ,
|
||||||
/**
|
/**
|
||||||
* Read (never ack) this daemon's own held lead-to-lead coordination mail (fleetd #421).
|
* Read (never ack) this daemon's own held lead-to-lead coordination mail (fleetd #421).
|
||||||
*
|
*
|
||||||
@@ -54,43 +62,97 @@ public final class Authz {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Whether {@code caller} may perform {@code action} against {@code targetSession}.
|
* The fail-closed classifier: answers no for every target, so a collaborator's {@code SEND}
|
||||||
|
* is refused unless a caller supplies a real one. {@code CallerResolver#knownLeadOrCollaborator()}
|
||||||
|
* is the real one, read from the same lead and collaborator maps {@code CallerResolver#resolve}
|
||||||
|
* consults, so a target that classifier calls known is one {@code resolve} would actually
|
||||||
|
* resolve as a lead or collaborator.
|
||||||
|
*/
|
||||||
|
public static final Predicate<String> NO_KNOWN_LEAD_OR_COLLABORATOR = target -> false;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Convenience form for a caller with no classifier to supply. Fails closed: a collaborator's
|
||||||
|
* {@code SEND} is refused, as if no terminal were a configured lead or collaborator — the
|
||||||
|
* same decision {@link #NO_KNOWN_LEAD_OR_COLLABORATOR} gives explicitly. Every other action's
|
||||||
|
* result is identical to the four-argument form's, since none of them consult the classifier.
|
||||||
*
|
*
|
||||||
* @param targetSession the session id in the request path; only consulted for the worker-scoped
|
* <p>Its default classifier denies every collaborator, so a caller enforcing authorization
|
||||||
* actions ({@code REPLY}, {@code ASK}), ignored otherwise, may be
|
* must use the four-argument form instead.
|
||||||
* {@code null}
|
|
||||||
*/
|
*/
|
||||||
public static boolean permits(Principal caller, Action action, String targetSession) {
|
public static boolean permits(Principal caller, Action action, String targetSession) {
|
||||||
|
return permits(caller, action, targetSession, NO_KNOWN_LEAD_OR_COLLABORATOR);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Whether {@code caller} may perform {@code action} against {@code targetSession}.
|
||||||
|
*
|
||||||
|
* @param targetSession the session id in the request path; only consulted for the
|
||||||
|
* worker-scoped actions ({@code REPLY}, {@code ASK}) and for a
|
||||||
|
* collaborator's {@code SEND}, ignored otherwise, may be
|
||||||
|
* {@code null}
|
||||||
|
* @param knownLeadOrCollaborator whether a terminal is a configured lead or collaborator —
|
||||||
|
* consulted only for a collaborator's {@code SEND}, to confine
|
||||||
|
* it to another named peer and never a spawned member's
|
||||||
|
* terminal
|
||||||
|
*/
|
||||||
|
public static boolean permits(Principal caller, Action action, String targetSession,
|
||||||
|
Predicate<String> knownLeadOrCollaborator) {
|
||||||
if (caller == null || caller.isAnonymous()) {
|
if (caller == null || caller.isAnonymous()) {
|
||||||
return false; // authenticated as nothing ⇒ authorized for nothing
|
return false; // authenticated as nothing ⇒ authorized for nothing
|
||||||
}
|
}
|
||||||
return switch (action) {
|
return switch (action) {
|
||||||
// Fleet lifecycle is the primary's alone — spawn, stop, drain. An architect
|
// Fleet lifecycle is the primary's alone — spawn, stop, drain. An architect and a
|
||||||
// deliberately does NOT get these (CB-548), so it cannot tear down or stand up workers
|
// collaborator deliberately do NOT get these, so neither can tear down or stand up
|
||||||
// even though it coordinates them; and a worker driving any of these would be a worker
|
// workers even though one of them coordinates them; and a worker driving any of these
|
||||||
// escalating into the orchestrator role.
|
// would be a worker escalating into the orchestrator role.
|
||||||
case SPAWN, STOP, DRAIN, HANDOVER -> caller.isPrimary();
|
case SPAWN, STOP, DRAIN, HANDOVER -> caller.isPrimary();
|
||||||
|
|
||||||
// Delivering a turn is open to the primary and the architect: an architect delegates
|
// Delivering a turn to a local session is open to the primary, the architect, and a
|
||||||
// to workers (that is the role's point) but still has no lifecycle rights. A worker is
|
// collaborator whose target is itself a configured lead or collaborator: the architect
|
||||||
// excluded — sending would be it escalating.
|
// delegates to workers (that is the role's point); a collaborator may reach only
|
||||||
case SEND -> caller.isPrimary() || caller.isArchitect();
|
// another named peer, never a spawned member's terminal. A worker is excluded —
|
||||||
|
// sending would be it escalating.
|
||||||
|
case SEND -> caller.isPrimary() || caller.isArchitect()
|
||||||
|
|| (caller.isCollaborator() && knownLeadOrCollaborator.test(targetSession));
|
||||||
|
|
||||||
|
// Resolving a worker's blocked question is part of delegating to it, open to the same
|
||||||
|
// two roles that may stand up that delegation in the first place. Not a collaborator:
|
||||||
|
// resuming another session's turn is lifecycle-adjacent, not peer messaging.
|
||||||
|
case ANSWER -> caller.isPrimary() || caller.isArchitect();
|
||||||
|
|
||||||
|
// Leaves the daemon over the coordination broker rather than addressing a local
|
||||||
|
// session, open to the same two roles as ANSWER. Not a collaborator: it is a
|
||||||
|
// local-tab peer with no cross-host route.
|
||||||
|
case COORD_SEND -> caller.isPrimary() || caller.isArchitect();
|
||||||
|
|
||||||
// The load-bearing rule: a caller acts only as the pane it occupies. CB-532 widened who
|
// The load-bearing rule: a caller acts only as the pane it occupies. CB-532 widened who
|
||||||
// that can be — a lead answering another lead is replying for its OWN terminal, which
|
// that can be — a lead answering another lead is replying for its OWN terminal, which
|
||||||
// this already permits — while the rule itself is unchanged, and is what stops anyone
|
// this already permits — while the rule itself is unchanged, and is what stops anyone
|
||||||
// forging a reply for a rendezvous someone else is waiting on. An architect's own pane
|
// forging a reply for a rendezvous someone else is waiting on. An architect's or a
|
||||||
// passes through the same check, so it can answer a funnel that delegated to it. An
|
// collaborator's own pane passes through the same check, so each can answer a funnel
|
||||||
// unnamed primary (token/loopback, no pane) owns nothing and is still excluded.
|
// that delegated to it. An unnamed primary (token/loopback, no pane) owns nothing and
|
||||||
|
// is still excluded.
|
||||||
case REPLY, ASK -> caller.ownsSession(targetSession);
|
case REPLY, ASK -> caller.ownsSession(targetSession);
|
||||||
|
|
||||||
// Observation is open to every authenticated role: a worker legitimately polls its own
|
// READ is roster, profile, and identity observation — fleet_list, fleet_profiles, and
|
||||||
// status, and the roster carries no secrets.
|
// fleet_whoami — and carries no secrets: no ticket reply, no pending question, and no
|
||||||
case READ, METRICS -> caller.isPrimary() || caller.isWorker() || caller.isArchitect();
|
// other session's turn state. Those live under TASK_READ. METRICS is the separate
|
||||||
|
// Prometheus scrape. Both are open to every authenticated role, including a
|
||||||
|
// collaborator and the unconfigured-pane floor.
|
||||||
|
case READ, METRICS -> caller.isPrimary() || caller.isWorker() || caller.isArchitect()
|
||||||
|
|| caller.isCollaborator() || caller.isObserver();
|
||||||
|
|
||||||
|
// Ticket polling and session status, open to every role READ is open to except a
|
||||||
|
// collaborator or an observer. MessageService compares a ticket's creator to the
|
||||||
|
// caller on every read as well, so dropping this gate would not expose another
|
||||||
|
// session's reply — it would move the refusal later and widen what a caller that
|
||||||
|
// never orchestrates can probe.
|
||||||
|
case TASK_READ -> caller.isPrimary() || caller.isWorker() || caller.isArchitect();
|
||||||
|
|
||||||
// fleetd #421: reading held lead-to-lead mail is the primary's alone. An architect
|
// fleetd #421: reading held lead-to-lead mail is the primary's alone. An architect
|
||||||
// holds READ today (CB-548), so "not primary" must mean not-architect here too — this
|
// holds READ today (CB-548), so "not primary" must mean not-architect here too — this
|
||||||
// is coordination between leads, not observation of the roster.
|
// is coordination between leads, not observation of the roster. The same reasoning
|
||||||
|
// excludes a collaborator.
|
||||||
case COORD_READ -> caller.isPrimary();
|
case COORD_READ -> caller.isPrimary();
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ import java.nio.charset.StandardCharsets;
|
|||||||
import java.security.MessageDigest;
|
import java.security.MessageDigest;
|
||||||
import java.util.Map;
|
import java.util.Map;
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
|
import java.util.function.Predicate;
|
||||||
import java.util.function.Supplier;
|
import java.util.function.Supplier;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -19,6 +20,13 @@ import java.util.function.Supplier;
|
|||||||
*
|
*
|
||||||
* <p><strong>Resolution order</strong> — connection identity first, token second, nothing third:
|
* <p><strong>Resolution order</strong> — connection identity first, token second, nothing third:
|
||||||
* <ol>
|
* <ol>
|
||||||
|
* <li>A loopback peer PID that maps to a pane this gateway itself spawned ⇒ that member's own
|
||||||
|
* role: {@link Role#WORKER} for a dev, hunter, or reviewer; {@link Role#ARCHITECT} for an
|
||||||
|
* architect, but only while the live slot role still confirms it (fleetd #424 — a slot
|
||||||
|
* revoked from config demotes an already-bound session on its very next request, so the
|
||||||
|
* roster's own role is never granted on its word alone). No tab map is consulted — a live
|
||||||
|
* spawned member's identity comes from the registry that spawned it, never from a label a
|
||||||
|
* pane could also carry.</li>
|
||||||
* <li>A loopback peer PID that maps to a pane named by {@code leaders:}, by the legacy
|
* <li>A loopback peer PID that maps to a pane named by {@code leaders:}, by the legacy
|
||||||
* {@code primary.terminal} pin, or by an operator-labelled lead tab (CB-307, CB-530, CB-531)
|
* {@code primary.terminal} pin, or by an operator-labelled lead tab (CB-307, CB-530, CB-531)
|
||||||
* ⇒ {@link Role#PRIMARY}, carrying that lead's
|
* ⇒ {@link Role#PRIMARY}, carrying that lead's
|
||||||
@@ -28,9 +36,11 @@ import java.util.function.Supplier;
|
|||||||
* so two leads can work as peers rather than one being demoted.</li>
|
* so two leads can work as peers rather than one being demoted.</li>
|
||||||
* <li>A loopback peer PID that maps to a pane bound to a CB-548 architect slot ⇒
|
* <li>A loopback peer PID that maps to a pane bound to a CB-548 architect slot ⇒
|
||||||
* {@link Role#ARCHITECT}, carrying the slot name. Just unforgeable as a worker's, and
|
* {@link Role#ARCHITECT}, carrying the slot name. Just unforgeable as a worker's, and
|
||||||
* resolved from the <em>live</em> terminal→slot binding (never a request argument), before
|
* resolved from the <em>live</em> terminal→slot binding (never a request argument). This is
|
||||||
* the generic worker fallback.</li>
|
* the case the previous step does not catch: a binding with no live spawned-member session.</li>
|
||||||
* <li>A loopback peer PID that maps to any other herdr pane ⇒ {@link Role#WORKER}. This is
|
* <li>A loopback peer PID that maps to an operator-labelled collaborator tab ⇒
|
||||||
|
* {@link Role#COLLABORATOR}, carrying that collaborator's name.</li>
|
||||||
|
* <li>A loopback peer PID that maps to any other herdr pane ⇒ {@link Role#OBSERVER}. This is
|
||||||
* unforgeable (the OS reports the PID, herdr owns the PID→pane map) and is honoured
|
* unforgeable (the OS reports the PID, herdr owns the PID→pane map) and is honoured
|
||||||
* regardless of auth mode, so enabling auth never breaks the fleet.</li>
|
* regardless of auth mode, so enabling auth never breaks the fleet.</li>
|
||||||
* <li>Otherwise, under {@code token} mode, a valid bearer token ⇒ {@link Role#PRIMARY}.</li>
|
* <li>Otherwise, under {@code token} mode, a valid bearer token ⇒ {@link Role#PRIMARY}.</li>
|
||||||
@@ -64,6 +74,20 @@ public final class CallerResolver {
|
|||||||
private final Supplier<Map<String, String>> architectTerminals;
|
private final Supplier<Map<String, String>> architectTerminals;
|
||||||
private final Function<String, MemberRole> memberSlotRoles;
|
private final Function<String, MemberRole> memberSlotRoles;
|
||||||
private final Function<String, String> memberSlotNames;
|
private final Function<String, String> memberSlotNames;
|
||||||
|
/**
|
||||||
|
* terminal_id → the role of the live spawned member occupying it, or {@code null} for a
|
||||||
|
* terminal no spawned member occupies. Consulted first, ahead of every tab map: a live
|
||||||
|
* spawned member's identity is its own, whatever a tab map says about the same terminal.
|
||||||
|
* A function rather than the roster itself, so a resolve on the hot path never scans a list —
|
||||||
|
* the lookup strategy is the caller's to choose.
|
||||||
|
*/
|
||||||
|
private final Function<String, MemberRole> spawnedMemberRole;
|
||||||
|
/**
|
||||||
|
* terminal_id → collaborator name; empty when none are configured. A supplier for the same
|
||||||
|
* reason as {@link #leadTerminals}: a collaborator tab recognised after construction (the tab
|
||||||
|
* scan discovering a newly-labelled tab) takes effect without a restart.
|
||||||
|
*/
|
||||||
|
private final Supplier<Map<String, String>> collaboratorTerminals;
|
||||||
|
|
||||||
/** Loopback-trust resolver: no token required, historical behaviour. Test-only. */
|
/** Loopback-trust resolver: no token required, historical behaviour. Test-only. */
|
||||||
CallerResolver(ConnectionIdentity identity) {
|
CallerResolver(ConnectionIdentity identity) {
|
||||||
@@ -124,17 +148,42 @@ public final class CallerResolver {
|
|||||||
/**
|
/**
|
||||||
* Live registry form that can confirm a bound slot is an architect slot.
|
* Live registry form that can confirm a bound slot is an architect slot.
|
||||||
*
|
*
|
||||||
* <p>This is the only public construction path. It keeps terminal bindings and slot roles in
|
* <p>It keeps terminal bindings and slot roles in the same {@link MemberRegistry}, so a
|
||||||
* the same {@link MemberRegistry}, so a configured architect can resolve as an architect.
|
* configured architect can resolve as an architect. No spawned-member roster or collaborator
|
||||||
|
* registry is consulted — equivalent to {@link #withLeadsAndMembers(ConnectionIdentity,
|
||||||
|
* boolean, String, Supplier, MemberRegistry, Function, Supplier)} with both absent. Kept for
|
||||||
|
* every caller that has neither to offer, so adding them did not churn every construction site.
|
||||||
*/
|
*/
|
||||||
public static CallerResolver withLeadsAndMembers(ConnectionIdentity identity,
|
public static CallerResolver withLeadsAndMembers(ConnectionIdentity identity,
|
||||||
boolean tokenMode, String token,
|
boolean tokenMode, String token,
|
||||||
Supplier<Map<String, String>> leadTerminals,
|
Supplier<Map<String, String>> leadTerminals,
|
||||||
MemberRegistry members) {
|
MemberRegistry members) {
|
||||||
|
return withLeadsAndMembers(identity, tokenMode, token, leadTerminals, members, null, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Live registry form that also resolves a live spawned member to its own role, and a
|
||||||
|
* configured collaborator tab to {@link Role#COLLABORATOR}.
|
||||||
|
*
|
||||||
|
* <p>This is the only public construction path that exercises the full resolution order.
|
||||||
|
*
|
||||||
|
* @param spawnedMemberRole terminal_id → the role of the live spawned member occupying
|
||||||
|
* it, or {@code null} for a terminal no spawned member occupies.
|
||||||
|
* {@code null} here means no roster is consulted at all (every
|
||||||
|
* terminal falls through to the tab maps), not that none matches.
|
||||||
|
* @param collaboratorTerminals terminal_id → collaborator name, live like {@code leadTerminals}
|
||||||
|
*/
|
||||||
|
public static CallerResolver withLeadsAndMembers(ConnectionIdentity identity,
|
||||||
|
boolean tokenMode, String token,
|
||||||
|
Supplier<Map<String, String>> leadTerminals,
|
||||||
|
MemberRegistry members,
|
||||||
|
Function<String, MemberRole> spawnedMemberRole,
|
||||||
|
Supplier<Map<String, String>> collaboratorTerminals) {
|
||||||
return new CallerResolver(identity, tokenMode, token, leadTerminals,
|
return new CallerResolver(identity, tokenMode, token, leadTerminals,
|
||||||
members == null ? null : members::snapshot,
|
members == null ? null : members::snapshot,
|
||||||
members == null ? null : members::roleForSlot,
|
members == null ? null : members::roleForSlot,
|
||||||
members == null ? null : members::nameForSlot);
|
members == null ? null : members::nameForSlot,
|
||||||
|
spawnedMemberRole, collaboratorTerminals);
|
||||||
}
|
}
|
||||||
|
|
||||||
private static Supplier<Map<String, String>> fixed(Map<String, String> leadTerminals) {
|
private static Supplier<Map<String, String>> fixed(Map<String, String> leadTerminals) {
|
||||||
@@ -160,6 +209,17 @@ public final class CallerResolver {
|
|||||||
Supplier<Map<String, String>> architectTerminals,
|
Supplier<Map<String, String>> architectTerminals,
|
||||||
Function<String, MemberRole> memberSlotRoles,
|
Function<String, MemberRole> memberSlotRoles,
|
||||||
Function<String, String> memberSlotNames) {
|
Function<String, String> memberSlotNames) {
|
||||||
|
this(identity, tokenMode, token, leadTerminals, architectTerminals, memberSlotRoles,
|
||||||
|
memberSlotNames, null, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||||
|
Supplier<Map<String, String>> leadTerminals,
|
||||||
|
Supplier<Map<String, String>> architectTerminals,
|
||||||
|
Function<String, MemberRole> memberSlotRoles,
|
||||||
|
Function<String, String> memberSlotNames,
|
||||||
|
Function<String, MemberRole> spawnedMemberRole,
|
||||||
|
Supplier<Map<String, String>> collaboratorTerminals) {
|
||||||
if (tokenMode && (token == null || token.isBlank())) {
|
if (tokenMode && (token == null || token.isBlank())) {
|
||||||
throw new IllegalArgumentException(
|
throw new IllegalArgumentException(
|
||||||
"auth.mode=token requires a non-empty token; check that the env var named by "
|
"auth.mode=token requires a non-empty token; check that the env var named by "
|
||||||
@@ -172,6 +232,8 @@ public final class CallerResolver {
|
|||||||
this.architectTerminals = architectTerminals == null ? Map::of : architectTerminals;
|
this.architectTerminals = architectTerminals == null ? Map::of : architectTerminals;
|
||||||
this.memberSlotRoles = memberSlotRoles == null ? _ -> null : memberSlotRoles;
|
this.memberSlotRoles = memberSlotRoles == null ? _ -> null : memberSlotRoles;
|
||||||
this.memberSlotNames = memberSlotNames == null ? Function.identity() : memberSlotNames;
|
this.memberSlotNames = memberSlotNames == null ? Function.identity() : memberSlotNames;
|
||||||
|
this.spawnedMemberRole = spawnedMemberRole == null ? _ -> null : spawnedMemberRole;
|
||||||
|
this.collaboratorTerminals = collaboratorTerminals == null ? Map::of : collaboratorTerminals;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -188,14 +250,24 @@ public final class CallerResolver {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The currently-recognised architect slots, {@code terminal_id → slot name} (CB-548).
|
* The currently-recognised collaborator tabs, {@code terminal_id → name}.
|
||||||
*
|
*
|
||||||
* <p>Read from the same supplier {@link #resolve} consults, so a slot that is <em>listed</em>
|
* <p>Read from the same supplier {@link #resolve} consults, for the reason given in
|
||||||
* here but would not <em>resolve</em> (or the reverse) cannot drift apart. Live for the same
|
* {@link #leads()}. Live for the same reason as {@link #leads()}.
|
||||||
* reason as {@link #leads()}.
|
|
||||||
*/
|
*/
|
||||||
public Map<String, String> members() {
|
public Map<String, String> collaborators() {
|
||||||
return architectTerminals.get();
|
return collaboratorTerminals.get();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Whether {@code target} names a terminal this resolver would resolve as a lead or a
|
||||||
|
* collaborator — the classifier a collaborator's {@code SEND} is checked against, read from the
|
||||||
|
* exact maps {@link #resolve} consults so a target that would resolve as a lead or collaborator
|
||||||
|
* is never the one a collaborator is refused to reach, or the reverse.
|
||||||
|
*/
|
||||||
|
public Predicate<String> knownLeadOrCollaborator() {
|
||||||
|
return target -> leadTerminals.get().containsKey(target)
|
||||||
|
|| collaboratorTerminals.get().containsKey(target);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -208,6 +280,25 @@ public final class CallerResolver {
|
|||||||
public Principal resolve(String remoteAddr, int remotePort, String authorizationHeader) {
|
public Principal resolve(String remoteAddr, int remotePort, String authorizationHeader) {
|
||||||
ConnectionIdentity.Caller c = identity.resolve(remoteAddr, remotePort);
|
ConnectionIdentity.Caller c = identity.resolve(remoteAddr, remotePort);
|
||||||
if (c.terminal() != null) {
|
if (c.terminal() != null) {
|
||||||
|
MemberRole spawnedRole = spawnedMemberRole.apply(c.terminal());
|
||||||
|
if (spawnedRole != null) {
|
||||||
|
// A live spawned member occupies this pane. Its identity is its own, whatever a tab
|
||||||
|
// map says about the same terminal — checked before every tab map, consulting none
|
||||||
|
// of them, so a tab label can never override a roster entry for the same terminal.
|
||||||
|
if (spawnedRole == MemberRole.ARCHITECT) {
|
||||||
|
// The roster only answers THAT this pane is a live spawned member; config still
|
||||||
|
// decides WHAT that member's slot grants (fleetd #424). A slot revoked after the
|
||||||
|
// bind must still demote this session on its very next request, so the roster's
|
||||||
|
// own ARCHITECT role is confirmed against the live slot role, exactly as the
|
||||||
|
// architect-slot step below confirms a binding with no live member session.
|
||||||
|
String slot = architectTerminals.get().get(c.terminal());
|
||||||
|
if (slot != null && memberSlotRoles.apply(slot) == MemberRole.ARCHITECT) {
|
||||||
|
return Principal.architect(memberSlotNames.apply(slot), c.terminal(), c.pid());
|
||||||
|
}
|
||||||
|
return Principal.worker(c.terminal(), c.pid());
|
||||||
|
}
|
||||||
|
return Principal.worker(c.terminal(), c.pid());
|
||||||
|
}
|
||||||
String lead = leadTerminals.get().get(c.terminal());
|
String lead = leadTerminals.get().get(c.terminal());
|
||||||
if (lead != null) {
|
if (lead != null) {
|
||||||
// The config names this pane as a lead's own. The pane mapping is exactly as
|
// The config names this pane as a lead's own. The pane mapping is exactly as
|
||||||
@@ -221,10 +312,18 @@ public final class CallerResolver {
|
|||||||
// The config/live binding names this pane as an architect slot's own. Same
|
// The config/live binding names this pane as an architect slot's own. Same
|
||||||
// unforgeable pane mapping; the live binding, never a request argument, decides.
|
// unforgeable pane mapping; the live binding, never a request argument, decides.
|
||||||
// Check the slot role too: this defence in depth prevents a bad lifecycle bind from
|
// Check the slot role too: this defence in depth prevents a bad lifecycle bind from
|
||||||
// escalating a dev or reviewer into an architect. Checked before the worker fallback.
|
// escalating a dev, hunter or reviewer into an architect. This is the case the
|
||||||
|
// spawned-member step above does not catch: a binding with no live member session.
|
||||||
return Principal.architect(memberSlotNames.apply(slot), c.terminal(), c.pid());
|
return Principal.architect(memberSlotNames.apply(slot), c.terminal(), c.pid());
|
||||||
}
|
}
|
||||||
return Principal.worker(c.terminal(), c.pid()); // unforgeable; never token-gated
|
String collaborator = collaboratorTerminals.get().get(c.terminal());
|
||||||
|
if (collaborator != null) {
|
||||||
|
// An operator-labelled collaborator tab, confirmed live by the same scan that
|
||||||
|
// confirms a lead tab. Checked last among the tab maps so a pane also matching one
|
||||||
|
// of the above keeps that stronger role.
|
||||||
|
return Principal.collaborator(collaborator, c.terminal(), c.pid());
|
||||||
|
}
|
||||||
|
return Principal.observer(c.terminal(), c.pid()); // unforgeable; never token-gated
|
||||||
}
|
}
|
||||||
|
|
||||||
if (tokenMode) {
|
if (tokenMode) {
|
||||||
@@ -244,7 +343,15 @@ public final class CallerResolver {
|
|||||||
// already names what happens if that case is handed the primary role: a worker→primary
|
// already names what happens if that case is handed the primary role: a worker→primary
|
||||||
// escalation. So an unresolved caller is refused (ANONYMOUS — the same clean, already-tested
|
// escalation. So an unresolved caller is refused (ANONYMOUS — the same clean, already-tested
|
||||||
// "authenticated as nothing" outcome used everywhere else in this method), never promoted.
|
// "authenticated as nothing" outcome used everywhere else in this method), never promoted.
|
||||||
return isLoopback(remoteAddr) && c.resolved() ? Principal.primary(c.pid()) : Principal.anonymous();
|
//
|
||||||
|
// fleetd #505: the OTHER way a real pid can wrongly reach here with a null terminal — not a
|
||||||
|
// failed lsof lookup, but a herdr error partway through PaneLocator's pane scan. c.resolved()
|
||||||
|
// says nothing about that; it only tests the lsof sentinel (by design — see
|
||||||
|
// ConnectionIdentity.Caller#resolved). c.scanComplete() is the separate signal: a scan that
|
||||||
|
// could not check every pane must not be read as "checked everywhere, no match" — the pane it
|
||||||
|
// could not check might have been the caller's own. So both must hold before this promotes.
|
||||||
|
return isLoopback(remoteAddr) && c.resolved() && c.scanComplete()
|
||||||
|
? Principal.primary(c.pid()) : Principal.anonymous();
|
||||||
}
|
}
|
||||||
|
|
||||||
private boolean presentedTokenMatches(String authorizationHeader) {
|
private boolean presentedTokenMatches(String authorizationHeader) {
|
||||||
|
|||||||
@@ -42,7 +42,8 @@ public interface MemberLifecycle {
|
|||||||
* Try to bind a newly spawned {@code terminal} into the role it was granted.
|
* Try to bind a newly spawned {@code terminal} into the role it was granted.
|
||||||
*
|
*
|
||||||
* @return the role this session actually holds: {@code role} unchanged for a role with no
|
* @return the role this session actually holds: {@code role} unchanged for a role with no
|
||||||
* live slot-binding semantics (dev, reviewer), or when the bind succeeded; a fallback
|
* live slot-binding semantics (dev, hunter, reviewer), or when the bind
|
||||||
|
* succeeded; a fallback
|
||||||
* role — never {@code role} — when a slot-bound role (architect) could not be bound.
|
* role — never {@code role} — when a slot-bound role (architect) could not be bound.
|
||||||
* Callers must record THIS value on the session, never the requested {@code role}, so
|
* Callers must record THIS value on the session, never the requested {@code role}, so
|
||||||
* a later roster read never reports a role the session does not hold (CB-619). In
|
* a later roster read never reports a role the session does not hold (CB-619). In
|
||||||
|
|||||||
@@ -20,7 +20,8 @@ import java.util.function.Supplier;
|
|||||||
*
|
*
|
||||||
* <p>Two halves, split by who owns each:
|
* <p>Two halves, split by who owns each:
|
||||||
* <ul>
|
* <ul>
|
||||||
* <li><b>slots</b> — read from {@code fleet.architects}/{@code developers}/{@code reviewers}
|
* <li><b>slots</b> — read from {@code fleet.architects}/{@code developers}/
|
||||||
|
* {@code hunters}/{@code reviewers}
|
||||||
* (see {@link #slots()}), each carrying the {@code profile} reference the spawn lifecycle
|
* (see {@link #slots()}), each carrying the {@code profile} reference the spawn lifecycle
|
||||||
* reads when it stands the slot up. <strong>Live, since fleetd #424</strong>: {@link #live}
|
* reads when it stands the slot up. <strong>Live, since fleetd #424</strong>: {@link #live}
|
||||||
* re-reads {@code fleet:} on every call, through a supplier the same shape as
|
* re-reads {@code fleet:} on every call, through a supplier the same shape as
|
||||||
@@ -322,7 +323,7 @@ public final class MemberRegistry implements MemberLifecycle {
|
|||||||
* CB-619 / fleetd #123: refuse an architect acquire before anything spawns when no configured
|
* CB-619 / fleetd #123: refuse an architect acquire before anything spawns when no configured
|
||||||
* slot carries {@code profile} — the config-gap case from the original defect report (a spawn
|
* slot carries {@code profile} — the config-gap case from the original defect report (a spawn
|
||||||
* asked for {@code role=architect, profile=sonnet}, and {@code fleet.architects} carried only
|
* asked for {@code role=architect, profile=sonnet}, and {@code fleet.architects} carried only
|
||||||
* {@code opus} and {@code sol}). A dev/reviewer acquire is always a no-op: those pools are
|
* {@code opus} and {@code sol}). A dev/hunter/reviewer acquire is always a no-op: those pools are
|
||||||
* placement candidates only (see {@code CompositePeerLauncher}), never a live identity binding,
|
* placement candidates only (see {@code CompositePeerLauncher}), never a live identity binding,
|
||||||
* so there is nothing here to refuse — an explicit profile outside the pool for those roles is a
|
* so there is nothing here to refuse — an explicit profile outside the pool for those roles is a
|
||||||
* documented operator override, not a defect.
|
* documented operator override, not a defect.
|
||||||
|
|||||||
@@ -11,7 +11,9 @@ package dev.ltms.fleet.auth;
|
|||||||
* @param pid the connecting process id, or {@code -1} when not resolvable (audit context)
|
* @param pid the connecting process id, or {@code -1} when not resolvable (audit context)
|
||||||
* @param name for a lead resolved from the CB-530 {@code leaders:} registry, which lead it is;
|
* @param name for a lead resolved from the CB-530 {@code leaders:} registry, which lead it is;
|
||||||
* for an architect resolved from the CB-548 {@code architects:} registry, which
|
* for an architect resolved from the CB-548 {@code architects:} registry, which
|
||||||
* slot it occupies; {@code null} for every other caller, including an unnamed primary
|
* slot it occupies; for a collaborator resolved from the {@code collaborators:}
|
||||||
|
* registry, which collaborator it is; {@code null} for every other caller,
|
||||||
|
* including an unnamed primary
|
||||||
*/
|
*/
|
||||||
public record Principal(Role role, String terminal, long pid, String name) {
|
public record Principal(Role role, String terminal, long pid, String name) {
|
||||||
|
|
||||||
@@ -73,6 +75,26 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
|||||||
return new Principal(Role.ARCHITECT, terminal, pid, slotName);
|
return new Principal(Role.ARCHITECT, terminal, pid, slotName);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A collaborator: a human-opened tab recognised by its exact label in the
|
||||||
|
* {@code collaborators:} registry.
|
||||||
|
*
|
||||||
|
* <p>Carries {@link Role#COLLABORATOR}. {@code name} is reporting only — it lets
|
||||||
|
* {@code fleet_whoami} say which collaborator is asking. Identity is the {@code terminal}:
|
||||||
|
* like a worker's it comes from the connection, so {@code ownsSession} works exactly as it
|
||||||
|
* does for a worker — a collaborator acts as its own pane and no other.
|
||||||
|
*/
|
||||||
|
public static Principal collaborator(String name, String terminal, long pid) {
|
||||||
|
return new Principal(Role.COLLABORATOR, terminal, pid, name);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The unconfigured-pane floor: a loopback caller whose pane matched no other role.
|
||||||
|
*/
|
||||||
|
public static Principal observer(String terminal, long pid) {
|
||||||
|
return new Principal(Role.OBSERVER, terminal, pid);
|
||||||
|
}
|
||||||
|
|
||||||
public boolean isPrimary() {
|
public boolean isPrimary() {
|
||||||
return role == Role.PRIMARY;
|
return role == Role.PRIMARY;
|
||||||
}
|
}
|
||||||
@@ -81,15 +103,23 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
|||||||
return role == Role.ARCHITECT;
|
return role == Role.ARCHITECT;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public boolean isCollaborator() {
|
||||||
|
return role == Role.COLLABORATOR;
|
||||||
|
}
|
||||||
|
|
||||||
public boolean isWorker() {
|
public boolean isWorker() {
|
||||||
return role == Role.WORKER;
|
return role == Role.WORKER;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
public boolean isObserver() {
|
||||||
|
return role == Role.OBSERVER;
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Whether this caller is a spawned member with its own pane.
|
* Whether this caller is a spawned member with its own pane.
|
||||||
*
|
*
|
||||||
* <p>Both workers and architects are spawned members. A lead is excluded because recording it
|
* <p>Both workers and architects are spawned members. A lead is not: it is a peer the
|
||||||
* as present would count it as an available member in the roster.
|
* operator started and named, never a pane this daemon spawned.
|
||||||
*/
|
*/
|
||||||
public boolean isSpawnedMember() {
|
public boolean isSpawnedMember() {
|
||||||
return role == Role.WORKER || role == Role.ARCHITECT;
|
return role == Role.WORKER || role == Role.ARCHITECT;
|
||||||
@@ -115,11 +145,32 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
|||||||
return terminal != null && terminal.equals(sessionId);
|
return terminal != null && terminal.equals(sessionId);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Stable identity used to own tickets and open turns. The unnamed primary has no owner key so
|
||||||
|
* it can use the message layer's primary-wide ticket access rule.
|
||||||
|
*/
|
||||||
|
public String ownerKey() {
|
||||||
|
return switch (role) {
|
||||||
|
case PRIMARY -> name == null ? null : prefixed("leader", name);
|
||||||
|
case WORKER -> prefixed("worker", terminal);
|
||||||
|
case ARCHITECT -> prefixed("architect", terminal);
|
||||||
|
case COLLABORATOR -> prefixed("collaborator", name);
|
||||||
|
case OBSERVER -> prefixed("observer", terminal);
|
||||||
|
case ANONYMOUS -> "anonymous";
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
private static String prefixed(String role, String identity) {
|
||||||
|
return role + ":" + identity;
|
||||||
|
}
|
||||||
|
|
||||||
/** Short, non-sensitive description for audit lines and error details. */
|
/** Short, non-sensitive description for audit lines and error details. */
|
||||||
public String describe() {
|
public String describe() {
|
||||||
return switch (role) {
|
return switch (role) {
|
||||||
case WORKER -> "worker:" + terminal;
|
case WORKER -> "worker:" + terminal;
|
||||||
case ARCHITECT -> "architect:" + name;
|
case ARCHITECT -> "architect:" + name;
|
||||||
|
case COLLABORATOR -> "collaborator:" + name;
|
||||||
|
case OBSERVER -> "observer:" + terminal;
|
||||||
case PRIMARY -> name == null ? "primary" : "leader:" + name;
|
case PRIMARY -> name == null ? "primary" : "leader:" + name;
|
||||||
case ANONYMOUS -> "anonymous";
|
case ANONYMOUS -> "anonymous";
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -34,6 +34,28 @@ public enum Role {
|
|||||||
*/
|
*/
|
||||||
ARCHITECT,
|
ARCHITECT,
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A config-declared, human-opened tab recognised by its exact label (the {@code
|
||||||
|
* fleet.collaborators.<name>.tab} registry). Never spawned — identity comes from the
|
||||||
|
* connection, never a request argument, exactly like {@link #WORKER} and {@link #ARCHITECT}.
|
||||||
|
* May {@code SEND} only to a configured lead or collaborator, {@code REPLY}/{@code ASK} only
|
||||||
|
* as its own pane, and {@code READ}/{@code METRICS}; may not {@code SPAWN}/{@code STOP}/
|
||||||
|
* {@code DRAIN}/{@code HANDOVER}, poll a ticket ({@code TASK_READ}), or reach the
|
||||||
|
* coordination broker ({@code COORD_SEND}/{@code COORD_READ}).
|
||||||
|
*/
|
||||||
|
COLLABORATOR,
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A loopback pane that resolved to none of the roles above: not a live spawned member, not a
|
||||||
|
* configured lead, not a bound architect slot, not a configured collaborator tab. Unforgeable
|
||||||
|
* like a worker's — derived from the connection's pane, never from a request argument, and
|
||||||
|
* honoured regardless of auth mode. May {@code READ} and {@code METRICS}, and {@code REPLY}/
|
||||||
|
* {@code ASK} only as its own pane; may not {@code SPAWN}/{@code STOP}/{@code DRAIN}/
|
||||||
|
* {@code HANDOVER}, {@code SEND}, poll a ticket ({@code TASK_READ}), or reach the coordination
|
||||||
|
* broker ({@code COORD_SEND}/{@code COORD_READ}).
|
||||||
|
*/
|
||||||
|
OBSERVER,
|
||||||
|
|
||||||
/** Authenticated as nothing. Authorized for nothing but {@code /healthz}. */
|
/** Authenticated as nothing. Authorized for nothing but {@code /healthz}. */
|
||||||
ANONYMOUS
|
ANONYMOUS
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -31,7 +31,7 @@ import java.util.function.Supplier;
|
|||||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Both are
|
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Both are
|
||||||
* read through a supplier on {@code CompositePeerLauncher}, which is what makes them hot —
|
* read through a supplier on {@code CompositePeerLauncher}, which is what makes them hot —
|
||||||
* not the fact that they are config. Most of {@code fleet:} — every role pool
|
* not the fact that they are config. Most of {@code fleet:} — every role pool
|
||||||
* ({@code architects}/{@code developers}/{@code reviewers}), {@code charters}, and
|
* ({@code architects}/{@code developers}/{@code hunters}/{@code reviewers}), {@code charters}, and
|
||||||
* {@code tabLabel} — is read the same live way, through the same supplier
|
* {@code tabLabel} — is read the same live way, through the same supplier
|
||||||
* ({@code () -> config.get().fleet()}). {@code architects} in particular is hot for
|
* ({@code () -> config.get().fleet()}). {@code architects} in particular is hot for
|
||||||
* <strong>two independent consumers</strong> (fleetd #424): {@code CompositePeerLauncher}
|
* <strong>two independent consumers</strong> (fleetd #424): {@code CompositePeerLauncher}
|
||||||
@@ -67,7 +67,7 @@ import java.util.function.Supplier;
|
|||||||
* {@code models:} above: {@code dev.ltms.fleet.lead.LeadRollover} holds a
|
* {@code models:} above: {@code dev.ltms.fleet.lead.LeadRollover} holds a
|
||||||
* {@code Supplier<FleetConfig.LeadRollover>} (the same {@code () -> config.get().x()} shape)
|
* {@code Supplier<FleetConfig.LeadRollover>} (the same {@code () -> config.get().x()} shape)
|
||||||
* and reads {@code handoverPath}/{@code requireOperatorConfirm}/{@code maxDocAgeSeconds}/
|
* and reads {@code handoverPath}/{@code requireOperatorConfirm}/{@code maxDocAgeSeconds}/
|
||||||
* {@code turnSettleSeconds}/{@code clearSettleSeconds}/{@code bootstrapText} fresh on every
|
* {@code turnSettleSeconds}/{@code relaunchReadySeconds}/{@code bootstrapText} fresh on every
|
||||||
* {@code open()}/{@code confirm()} call (and on the deferred post-{@code confirm()}
|
* {@code open()}/{@code confirm()} call (and on the deferred post-{@code confirm()}
|
||||||
* continuation fleetd #480's correction added — see {@code LeadRollover}'s class doc) rather
|
* continuation fleetd #480's correction added — see {@code LeadRollover}'s class doc) rather
|
||||||
* than capturing them into fields at construction — unlike its closest
|
* than capturing them into fields at construction — unlike its closest
|
||||||
@@ -208,7 +208,7 @@ import java.util.function.Supplier;
|
|||||||
* that is correctly hot and for a key nobody triaged. Three times now — {@code worktreeGroup} (#323),
|
* that is correctly hot and for a key nobody triaged. Three times now — {@code worktreeGroup} (#323),
|
||||||
* {@code primary}/{@code configReload} (#326), and {@code fleet.leaders} sitting in the escape hatch
|
* {@code primary}/{@code configReload} (#326), and {@code fleet.leaders} sitting in the escape hatch
|
||||||
* (#333) — the second kind hid among the first. A top-level coverage checker in the
|
* (#333) — the second kind hid among the first. A top-level coverage checker in the
|
||||||
* {@link ConfigRefProfileCoverageTest} shape (one level up, over {@code FleetConfig} itself rather
|
* {@code ConfigRefProfileCoverageTest} shape (one level up, over {@code FleetConfig} itself rather
|
||||||
* than {@code FleetConfig.Profile}) proves this file's four classes exhaust the record's components
|
* than {@code FleetConfig.Profile}) proves this file's four classes exhaust the record's components
|
||||||
* — see {@code ConfigRefTopLevelCoverageTest}. That test proves the record's <em>shape</em> is fully
|
* — see {@code ConfigRefTopLevelCoverageTest}. That test proves the record's <em>shape</em> is fully
|
||||||
* triaged; it does NOT prove a {@code SPLIT_KEYS}/{@code COLD_KEYS}/{@code DEFERRED_KEYS} member has
|
* triaged; it does NOT prove a {@code SPLIT_KEYS}/{@code COLD_KEYS}/{@code DEFERRED_KEYS} member has
|
||||||
@@ -605,7 +605,7 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
|||||||
+ "opened once and needs a restart; the broker URI env-var name kept out of a "
|
+ "opened once and needs a restart; the broker URI env-var name kept out of a "
|
||||||
+ "member's environment is read live on every spawn and already applied");
|
+ "member's environment is read live on every spawn and already applied");
|
||||||
}
|
}
|
||||||
// fleetd #333: unlike health/coordinator above, most of `fleet:` (developers, reviewers,
|
// fleetd #333: unlike health/coordinator above, most of `fleet:` (developers, hunters, reviewers,
|
||||||
// charters, tabLabel) is genuinely hot — ConfigRefTest.aHotChangeIsAppliedAndRead-
|
// charters, tabLabel) is genuinely hot — ConfigRefTest.aHotChangeIsAppliedAndRead-
|
||||||
// ThroughGet and aCharterChangeIsHotAndReachesTheLiveConfig prove it reaches the live config
|
// ThroughGet and aCharterChangeIsHotAndReachesTheLiveConfig prove it reaches the live config
|
||||||
// with no restart note. `architects` is hot too, and — since fleetd #424 — hot for BOTH of
|
// with no restart note. `architects` is hot too, and — since fleetd #424 — hot for BOTH of
|
||||||
@@ -615,7 +615,7 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
|||||||
// may bind to, AND what a slot already bound still grants) through its own instance of that
|
// may bind to, AND what a slot already bound still grants) through its own instance of that
|
||||||
// same supplier shape — see MemberRegistry.live and its class doc for the binding rule:
|
// same supplier shape — see MemberRegistry.live and its class doc for the binding rule:
|
||||||
// removing a slot revokes ARCHITECT on the bound pane's very next request, and only the slot
|
// removing a slot revokes ARCHITECT on the bound pane's very next request, and only the slot
|
||||||
// OCCUPANCY survives, so the demoted session keeps its slot key until it unbinds. Only
|
// OCCUPANCY survives, so the demoted session keeps its slot key until it unbinds.
|
||||||
// fleet.leaders is frozen (Fleetd.java:281 reads cfg.fleet().leaders() off the startup
|
// fleet.leaders is frozen (Fleetd.java:281 reads cfg.fleet().leaders() off the startup
|
||||||
// snapshot to build both the LeadTabScanner's tab-label-to-name map, wired into
|
// snapshot to build both the LeadTabScanner's tab-label-to-name map, wired into
|
||||||
// CallerResolver.withLeadsAndMembers at Fleetd.java:620/624, and — when herdr answered —
|
// CallerResolver.withLeadsAndMembers at Fleetd.java:620/624, and — when herdr answered —
|
||||||
@@ -630,11 +630,16 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
|||||||
+ "identity map and to auto-launch leads, and neither is rebuilt on reload, so a "
|
+ "identity map and to auto-launch leads, and neither is rebuilt on reload, so a "
|
||||||
+ "lead added, removed, or given a new tab: label needs a restart — until then it "
|
+ "lead added, removed, or given a new tab: label needs a restart — until then it "
|
||||||
+ "stays unrecognised, and a caller from its new tab resolves as a worker, not a "
|
+ "stays unrecognised, and a caller from its new tab resolves as a worker, not a "
|
||||||
+ "lead; the rest of fleet: (developers, reviewers, charters, tabLabel) is read "
|
+ "lead; the rest of fleet: (developers, hunters, reviewers, charters, tabLabel) is read "
|
||||||
+ "live through the supplier on CompositePeerLauncher, and architects is read "
|
+ "live through the supplier on CompositePeerLauncher, and architects is read "
|
||||||
+ "live through that same supplier for placement AND through a separate supplier "
|
+ "live through that same supplier for placement AND through a separate supplier "
|
||||||
+ "on MemberRegistry for spawn-time identity — both already applied");
|
+ "on MemberRegistry for spawn-time identity — both already applied");
|
||||||
}
|
}
|
||||||
|
if (!Objects.equals(collaboratorsOf(old), collaboratorsOf(fresh))) {
|
||||||
|
changed.add("fleet: fleet.collaborators (each collaborator's tab) is read once at "
|
||||||
|
+ "startup to build the LeadTabScanner's identity map, which is not rebuilt on "
|
||||||
|
+ "reload, so a collaborator added, removed, or given a new tab: needs a restart");
|
||||||
|
}
|
||||||
// Kept in step with SPLIT_KEYS the same way changedColdKeys is kept in step with COLD_KEYS —
|
// Kept in step with SPLIT_KEYS the same way changedColdKeys is kept in step with COLD_KEYS —
|
||||||
// every message here must be traceable to one of the split keys the class doc documents.
|
// every message here must be traceable to one of the split keys the class doc documents.
|
||||||
// NOTE what this does NOT prove, per the javadoc above: it does not catch a SPLIT_KEYS
|
// NOTE what this does NOT prove, per the javadoc above: it does not catch a SPLIT_KEYS
|
||||||
@@ -650,6 +655,11 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
|||||||
return cfg.fleet() == null ? Map.of() : cfg.fleet().leaders();
|
return cfg.fleet() == null ? Map.of() : cfg.fleet().leaders();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** {@code cfg.fleet().collaborators()}, defensively, in case a caller hands in a non-defaulted config. */
|
||||||
|
private static Map<String, FleetConfig.Collaborator> collaboratorsOf(FleetConfig cfg) {
|
||||||
|
return cfg.fleet() == null ? Map.of() : cfg.fleet().collaborators();
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* {@link FleetConfig.Profile} record components deliberately left out of
|
* {@link FleetConfig.Profile} record components deliberately left out of
|
||||||
* {@link #sameLaunchSettings} because they are read <em>live</em>, not baked in at spawn — see
|
* {@link #sameLaunchSettings} because they are read <em>live</em>, not baked in at spawn — see
|
||||||
|
|||||||
@@ -62,7 +62,7 @@ import java.util.regex.PatternSyntaxException;
|
|||||||
* @param fleet who the daemon may run and under which role (CB-557). One block replacing
|
* @param fleet who the daemon may run and under which role (CB-557). One block replacing
|
||||||
* the former {@code leaders:}, {@code members:}, {@code leadScan:} and
|
* the former {@code leaders:}, {@code members:}, {@code leadScan:} and
|
||||||
* {@code defaultProfile:}. Role is the containing key — {@code leaders},
|
* {@code defaultProfile:}. Role is the containing key — {@code leaders},
|
||||||
* {@code architects}, {@code developers}, {@code reviewers} — and each entry
|
* {@code architects}, {@code developers}, {@code hunters}, {@code reviewers} — and each entry
|
||||||
* names the {@code profiles:} backend it runs on. See {@link Fleet}
|
* names the {@code profiles:} backend it runs on. See {@link Fleet}
|
||||||
* @param leadHeartbeat opt-in idle-lead heartbeat (CB-551); {@code null} ⇒ off, and an upgraded
|
* @param leadHeartbeat opt-in idle-lead heartbeat (CB-551); {@code null} ⇒ off, and an upgraded
|
||||||
* daemon never nudges an idle lead on its own initiative
|
* daemon never nudges an idle lead on its own initiative
|
||||||
@@ -462,8 +462,8 @@ public record FleetConfig(
|
|||||||
* profile that does not opt in. Read live off the current config, so it is
|
* profile that does not opt in. Read live off the current config, so it is
|
||||||
* HOT: a change takes effect on the next exhaustion classification / spawn,
|
* HOT: a change takes effect on the next exhaustion classification / spawn,
|
||||||
* no restart needed.
|
* no restart needed.
|
||||||
* @param autoCompactWindow opt-in per-profile token window that forces a spawned member to
|
* @param autoCompactWindow opt-in per-profile token window that forces a launched Claude Code
|
||||||
* auto-compact its context at (Claude Code) or within (opencode) a bound the
|
* session to auto-compact its context at, or an opencode session within, a bound the
|
||||||
* operator chooses, instead of the backend's own default. {@code null} (the
|
* operator chooses, instead of the backend's own default. {@code null} (the
|
||||||
* default) leaves today's behaviour exactly — opencode already forces
|
* default) leaves today's behaviour exactly — opencode already forces
|
||||||
* {@code compaction.auto: true} unconditionally (CB-523) but has no absolute
|
* {@code compaction.auto: true} unconditionally (CB-523) but has no absolute
|
||||||
@@ -798,6 +798,25 @@ public record FleetConfig(
|
|||||||
return isSubscription() ? SUBSCRIPTION_CREDENTIAL_ID : profile;
|
return isSubscription() ? SUBSCRIPTION_CREDENTIAL_ID : profile;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The auto-compaction window a launched Claude Code session actually runs on: {@code env:
|
||||||
|
* CLAUDE_CODE_AUTO_COMPACT_WINDOW} when it parses as an integer, since that environment
|
||||||
|
* variable wins over the {@code --autocompact} flag {@link #autoCompactWindow} produces (see
|
||||||
|
* {@code ClaudeCodeArguments}); {@link #autoCompactWindow} otherwise. {@code null} when
|
||||||
|
* neither resolves to a usable number.
|
||||||
|
*/
|
||||||
|
public Integer effectiveAutoCompactWindow() {
|
||||||
|
String envValue = env.get(CLAUDE_CODE_AUTO_COMPACT_WINDOW_ENV);
|
||||||
|
if (envValue == null) {
|
||||||
|
return autoCompactWindow;
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
return Integer.valueOf(envValue.trim());
|
||||||
|
} catch (NumberFormatException e) {
|
||||||
|
return autoCompactWindow;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/** True when this profile's workers are granted a forge token to open their own PR (CB-302). */
|
/** True when this profile's workers are granted a forge token to open their own PR (CB-302). */
|
||||||
public boolean hasGitToken() {
|
public boolean hasGitToken() {
|
||||||
return gitTokenEnv != null && !gitTokenEnv.isBlank();
|
return gitTokenEnv != null && !gitTokenEnv.isBlank();
|
||||||
@@ -1110,11 +1129,8 @@ public record FleetConfig(
|
|||||||
* {@code tab} can never be discovered, launched or not
|
* {@code tab} can never be discovered, launched or not
|
||||||
* @param instances how many of this lead should be live (default 1). The daemon
|
* @param instances how many of this lead should be live (default 1). The daemon
|
||||||
* launches only the shortfall, so a restart adopts rather than doubles
|
* launches only the shortfall, so a restart adopts rather than doubles
|
||||||
* @param tabPrefix no longer used to find a lead's tab — {@code tab} is matched
|
* @param tabPrefix lead-tab naming convention checked against member labels. Lead
|
||||||
* exactly. Its only remaining job is the startup collision guard
|
* identity uses {@code tab}. Default {@code "lead:"}
|
||||||
* ({@link #validateLeadTabPrefixes()}), which still uses it to refuse
|
|
||||||
* a worker {@code tabLabel} template that could be misread as a lead.
|
|
||||||
* Default {@code "lead:"}
|
|
||||||
* @param scanIntervalSeconds how long a tab scan is cached before herdr is asked again; also the
|
* @param scanIntervalSeconds how long a tab scan is cached before herdr is asked again; also the
|
||||||
* worst case before a newly-labelled tab is recognised. Default 10
|
* worst case before a newly-labelled tab is recognised. Default 10
|
||||||
* @param kind which agent runs there ({@code claude}, {@code opencode}, …)
|
* @param kind which agent runs there ({@code claude}, {@code opencode}, …)
|
||||||
@@ -1160,6 +1176,25 @@ public record FleetConfig(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A tab fleetd recognises as a collaborator, keyed by name (fleetd #669).
|
||||||
|
*
|
||||||
|
* <p>Recognise-only: there is no {@code profile}, no {@code instances} and no {@code kind}.
|
||||||
|
* Nothing here ever launches a pane.
|
||||||
|
*
|
||||||
|
* <p>{@code tabPrefix} is absent. Identity is matched on the exact {@code tab} alone.
|
||||||
|
*
|
||||||
|
* @param tab the exact tab label hosting this collaborator, matched case-insensitively; the
|
||||||
|
* only field identity depends on. Required — an entry with no {@code tab} can
|
||||||
|
* never be discovered.
|
||||||
|
*/
|
||||||
|
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||||
|
public record Collaborator(String tab) {
|
||||||
|
public Collaborator {
|
||||||
|
tab = (tab == null || tab.isBlank()) ? null : tab.strip();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* One entry of a {@code fleet:} role pool — a role paired with the backend it runs on.
|
* One entry of a {@code fleet:} role pool — a role paired with the backend it runs on.
|
||||||
*
|
*
|
||||||
@@ -1197,28 +1232,31 @@ public record FleetConfig(
|
|||||||
* is exactly compatible with that. The pool is also what replaced {@code defaultProfile:} — an
|
* is exactly compatible with that. The pool is also what replaced {@code defaultProfile:} — an
|
||||||
* unqualified spawn names a role, and the role's pool supplies the candidates.
|
* unqualified spawn names a role, and the role's pool supplies the candidates.
|
||||||
*
|
*
|
||||||
* @param leaders panes that orchestrate rather than are orchestrated, keyed by lead name
|
* @param leaders panes that orchestrate rather than are orchestrated, keyed by lead name
|
||||||
* @param architects profiles the {@code architect} role may run on
|
* @param architects profiles the {@code architect} role may run on
|
||||||
* @param developers profiles the {@code dev} role may run on
|
* @param developers profiles the {@code dev} role may run on
|
||||||
* @param reviewers profiles the {@code reviewer} role may run on
|
* @param hunters profiles the {@code hunter} role may run on
|
||||||
* @param charters optional launch-charter text keyed by singular role wire name
|
* @param reviewers profiles the {@code reviewer} role may run on
|
||||||
* @param tabLabel template for a member tab's label; {@code {role}}, {@code {profile}},
|
* @param charters optional launch-charter text keyed by singular role wire name
|
||||||
* {@code {model}} and {@code {n}} (a per role+profile counter) are
|
* @param tabLabel template for a member tab's label; {@code {role}}, {@code {profile}},
|
||||||
* substituted. Default {@link #DEFAULT_TAB_LABEL}
|
* {@code {model}} and {@code {n}} (a per role+profile counter) are
|
||||||
|
* substituted. Default {@link #DEFAULT_TAB_LABEL}
|
||||||
|
* @param collaborators tabs fleetd recognises as collaborators (fleetd #669), keyed by name.
|
||||||
|
* Recognise-only, exactly like a {@code profile}-less {@link Leader}:
|
||||||
|
* nothing here is ever auto-launched.
|
||||||
*/
|
*/
|
||||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||||
public record Fleet(Map<String, Leader> leaders,
|
public record Fleet(Map<String, Leader> leaders,
|
||||||
Map<String, Slot> architects,
|
Map<String, Slot> architects,
|
||||||
Map<String, Slot> developers,
|
Map<String, Slot> developers,
|
||||||
|
Map<String, Slot> hunters,
|
||||||
Map<String, Slot> reviewers,
|
Map<String, Slot> reviewers,
|
||||||
Map<String, String> charters,
|
Map<String, String> charters,
|
||||||
String tabLabel) {
|
String tabLabel,
|
||||||
|
Map<String, Collaborator> collaborators) {
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Role first, so the tab bar reads as the fleet and so the label shares a namespace with a
|
* Role first, so the tab bar identifies the member's fleet role.
|
||||||
* lead's {@code tabPrefix}. Because {@code {role}} comes from a closed enum, a generated
|
|
||||||
* member label can never begin with {@code "lead:"} — the clash that
|
|
||||||
* {@link #validateLeadTabPrefixes()} used to have to check for is unrepresentable here.
|
|
||||||
*/
|
*/
|
||||||
public static final String DEFAULT_TAB_LABEL = "{role}: {profile} #{n}";
|
public static final String DEFAULT_TAB_LABEL = "{role}: {profile} #{n}";
|
||||||
|
|
||||||
@@ -1226,23 +1264,34 @@ public record FleetConfig(
|
|||||||
leaders = unmodifiableOrEmpty(leaders);
|
leaders = unmodifiableOrEmpty(leaders);
|
||||||
architects = unmodifiableOrEmpty(architects);
|
architects = unmodifiableOrEmpty(architects);
|
||||||
developers = unmodifiableOrEmpty(developers);
|
developers = unmodifiableOrEmpty(developers);
|
||||||
|
hunters = unmodifiableOrEmpty(hunters);
|
||||||
reviewers = unmodifiableOrEmpty(reviewers);
|
reviewers = unmodifiableOrEmpty(reviewers);
|
||||||
charters = unmodifiableOrEmpty(charters);
|
charters = unmodifiableOrEmpty(charters);
|
||||||
tabLabel = (tabLabel == null || tabLabel.isBlank()) ? DEFAULT_TAB_LABEL : tabLabel;
|
tabLabel = (tabLabel == null || tabLabel.isBlank()) ? DEFAULT_TAB_LABEL : tabLabel;
|
||||||
|
collaborators = unmodifiableOrEmpty(collaborators);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* A fleet with no configured launch charters — the shape every deployment had before
|
* A fleet with no configured launch charters and no collaborators — the shape every
|
||||||
* CB-566, and what most tests want.
|
* deployment had before CB-566, and what most tests want.
|
||||||
*
|
*
|
||||||
* <p>Kept deliberately, even though an overload that drops a new field is normally the
|
* <p>Kept deliberately, even though an overload that drops a new field is normally the
|
||||||
* shape to avoid. It is safe here because nothing <em>reads</em> a charter through a
|
* shape to avoid. It is safe here because nothing <em>reads</em> a charter through a
|
||||||
* constructor: the launcher reads {@code fleet.charters()} from the live config. Jackson
|
* constructor: the launcher reads {@code fleet.charters()} from the live config. Jackson
|
||||||
* binds the canonical constructor, so this one cannot swallow an operator's YAML.
|
* binds the canonical constructor, so this one cannot swallow an operator's YAML.
|
||||||
|
* {@code collaborators} is dropped the same way and for the same reason: no caller of
|
||||||
|
* this overload has ever needed to set it, so it defaults to empty here exactly as the
|
||||||
|
* canonical constructor would default an absent YAML key.
|
||||||
*/
|
*/
|
||||||
|
public Fleet(Map<String, Leader> leaders, Map<String, Slot> architects,
|
||||||
|
Map<String, Slot> developers, Map<String, Slot> reviewers,
|
||||||
|
Map<String, String> charters, String tabLabel) {
|
||||||
|
this(leaders, architects, developers, null, reviewers, charters, tabLabel, null);
|
||||||
|
}
|
||||||
|
|
||||||
public Fleet(Map<String, Leader> leaders, Map<String, Slot> architects,
|
public Fleet(Map<String, Leader> leaders, Map<String, Slot> architects,
|
||||||
Map<String, Slot> developers, Map<String, Slot> reviewers, String tabLabel) {
|
Map<String, Slot> developers, Map<String, Slot> reviewers, String tabLabel) {
|
||||||
this(leaders, architects, developers, reviewers, null, tabLabel);
|
this(leaders, architects, developers, null, reviewers, null, tabLabel, null);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -1264,6 +1313,7 @@ public record FleetConfig(
|
|||||||
return switch (role) {
|
return switch (role) {
|
||||||
case ARCHITECT -> architects;
|
case ARCHITECT -> architects;
|
||||||
case DEV -> developers;
|
case DEV -> developers;
|
||||||
|
case HUNTER -> hunters;
|
||||||
case REVIEWER -> reviewers;
|
case REVIEWER -> reviewers;
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
@@ -1319,9 +1369,20 @@ public record FleetConfig(
|
|||||||
* each such nudge costs the lead a turn just to read "nothing pending";
|
* each such nudge costs the lead a turn just to read "nothing pending";
|
||||||
* three is enough to tell it it may stand down without nagging forever, and
|
* three is enough to tell it it may stand down without nagging forever, and
|
||||||
* it is the bound that stops an idle fleet from being a subscription burner.
|
* it is the bound that stops an idle fleet from being a subscription burner.
|
||||||
|
* @param contextHighNudge fleetd #609: when {@code true}, an idle lead whose own {@code
|
||||||
|
* LeadContextGauge} reading is {@code HIGH} gets a text notice telling it
|
||||||
|
* to consider a handover, appended to whatever heartbeat nudge the loop
|
||||||
|
* already sends. Default {@code false} ({@code null} also means off) — an
|
||||||
|
* upgraded daemon must not silently start telling leads to hand over.
|
||||||
*/
|
*/
|
||||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||||
public record LeadHeartbeat(Integer idleAfterSeconds, Long backoffMs, Integer quietNudgeCap) {
|
public record LeadHeartbeat(Integer idleAfterSeconds, Long backoffMs, Integer quietNudgeCap,
|
||||||
|
Boolean contextHighNudge) {
|
||||||
|
/** Convenience constructor for every call site that predates fleetd #609: no context notice. */
|
||||||
|
public LeadHeartbeat(Integer idleAfterSeconds, Long backoffMs, Integer quietNudgeCap) {
|
||||||
|
this(idleAfterSeconds, backoffMs, quietNudgeCap, null);
|
||||||
|
}
|
||||||
|
|
||||||
public LeadHeartbeat {
|
public LeadHeartbeat {
|
||||||
idleAfterSeconds = (idleAfterSeconds == null || idleAfterSeconds <= 0) ? 300 : idleAfterSeconds;
|
idleAfterSeconds = (idleAfterSeconds == null || idleAfterSeconds <= 0) ? 300 : idleAfterSeconds;
|
||||||
backoffMs = (backoffMs == null || backoffMs <= 0) ? 60_000L : backoffMs;
|
backoffMs = (backoffMs == null || backoffMs <= 0) ? 60_000L : backoffMs;
|
||||||
@@ -1355,15 +1416,16 @@ public record FleetConfig(
|
|||||||
* before anything exists to call — the same fact already true of adding a brand-new
|
* before anything exists to call — the same fact already true of adding a brand-new
|
||||||
* {@code profiles:} entry.
|
* {@code profiles:} entry.
|
||||||
*
|
*
|
||||||
* <p><strong>{@code turnSettleSeconds} (fleetd #480 correction):</strong> {@code confirm()} is
|
* <p><strong>{@code turnSettleSeconds}:</strong> {@code confirm()} is called FROM the calling
|
||||||
* called FROM the calling lead's own turn, so its pane is still {@code WORKING} the instant
|
* lead's own turn, so its pane is still {@code WORKING} the instant {@code confirm()} validates
|
||||||
* {@code confirm()} validates every gate and schedules the roll. {@code
|
* every gate and schedules the roll. {@code dev.ltms.fleet.lead.LeadRollover}'s deferred
|
||||||
* dev.ltms.fleet.lead.LeadRollover}'s deferred continuation waits up to this many seconds for
|
* continuation waits up to this many seconds for that SAME pane to report {@code IDLE} or
|
||||||
* that SAME pane to report an injectable state again — i.e. for the calling turn to actually
|
* {@code DONE} — i.e. for the calling turn to actually end — before it ends the old pane's
|
||||||
* end — before it sends {@code /clear} at all. If that wait times out, no {@code /clear} is
|
* process at all. {@code BLOCKED} does not count: that is a live turn merely paused, not one
|
||||||
* ever sent: a lead that never goes idle is still doing real work, and clearing it would
|
* that has finished. If that wait times out, the old pane is never touched: a lead that never
|
||||||
* destroy live context. This is a separate wait from {@code clearSettleSeconds} below, which
|
* goes idle is still doing real work, and the roll ends that pane's whole process — there is no
|
||||||
* bounds the SECOND wait, for the pane to re-settle AFTER {@code /clear} has already gone out.
|
* way back from this once it runs, so this wait is the only thing standing between "still
|
||||||
|
* working" and "gone".
|
||||||
*
|
*
|
||||||
* @param handoverPath required when this block is present — where the handover file a fresh
|
* @param handoverPath required when this block is present — where the handover file a fresh
|
||||||
* lead session reads must live. There is no sane non-null default for an
|
* lead session reads must live. There is no sane non-null default for an
|
||||||
@@ -1382,37 +1444,51 @@ public record FleetConfig(
|
|||||||
* @param maxDocAgeSeconds default 3600 — refuse a handover file whose modified time is older
|
* @param maxDocAgeSeconds default 3600 — refuse a handover file whose modified time is older
|
||||||
* than this many seconds, so a stale leftover from an earlier rollover
|
* than this many seconds, so a stale leftover from an earlier rollover
|
||||||
* attempt can never be mistaken for a fresh one.
|
* attempt can never be mistaken for a fresh one.
|
||||||
* @param turnSettleSeconds default 20 — bound on how long the deferred roll waits for the
|
* @param turnSettleSeconds default 300 — bound on how long the deferred roll waits for the
|
||||||
* CALLING lead's own turn to end (its pane to report injectable again)
|
* CALLING lead's own turn to end (its pane to report {@code IDLE} or
|
||||||
* before sending {@code /clear} at all. See the paragraph above.
|
* {@code DONE}) before ending that pane's process at all. See the
|
||||||
* @param clearSettleSeconds default 20 — bound on how long to wait for the lead's pane to
|
* paragraph above.
|
||||||
* report an injectable state again after {@code /clear} before giving up. A
|
* @param relaunchReadySeconds default 45 — bound on EACH of two separate waits that run after
|
||||||
* roll that times out here never sends {@code bootstrapText}.
|
* the old lead's pane has been torn down and a fresh one launched: first,
|
||||||
|
* for the fresh pane itself to reach a real turn boundary ({@code IDLE} or
|
||||||
|
* {@code DONE}, never merely {@code BLOCKED}) — the safety gate, since
|
||||||
|
* typing into a pane that has not finished booting loses the keystrokes;
|
||||||
|
* second, for the fresh terminal to show up as a recognised lead, which is
|
||||||
|
* bookkeeping rather than a safety gate, so a timeout on this second wait
|
||||||
|
* does not withhold {@code bootstrapText} — it is sent once the pane is
|
||||||
|
* ready regardless. Recognition comes from the same periodically-refreshed
|
||||||
|
* scan {@code LeadTabScanner} already keeps ({@code scanIntervalSeconds},
|
||||||
|
* 10s live), so a budget has to clear more than one scan interval to leave
|
||||||
|
* any real margin for the CLI's own boot time; 20 was rejected for exactly
|
||||||
|
* that reason — at a 10s scan interval it only buys two scans. 45 buys
|
||||||
|
* roughly four. Only a timeout on the FIRST wait (the pane never becomes
|
||||||
|
* ready) withholds {@code bootstrapText}.
|
||||||
* @param bootstrapText default a sentence naming the RESOLVED handover path — sent to the
|
* @param bootstrapText default a sentence naming the RESOLVED handover path — sent to the
|
||||||
* lead's pane once it settles after {@code /clear}, telling the fresh
|
* fresh lead's pane once it reaches a real turn boundary after relaunch,
|
||||||
* session where to read the handover and carry on. Left {@code null} here
|
* telling the fresh session where to read the handover and carry on. Left
|
||||||
* when the operator configures none: the default sentence cannot be built
|
* {@code null} here when the operator configures none: the default sentence
|
||||||
* at construction time because it must name the path AFTER {@code
|
* cannot be built at construction time because it must name the path AFTER
|
||||||
* dev.ltms.fleet.lead.LeadRollover#open} has resolved a relative {@code
|
* {@code dev.ltms.fleet.lead.LeadRollover#open} has resolved a relative
|
||||||
* handoverPath} against the calling lead's workspace, which this record has
|
* {@code handoverPath} against the calling lead's workspace, which this
|
||||||
* no way to know — see {@link #bootstrapTextFor(String)}.
|
* record has no way to know — see {@link #bootstrapTextFor(String)}.
|
||||||
*/
|
*/
|
||||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||||
public record LeadRollover(String handoverPath, Boolean requireOperatorConfirm,
|
public record LeadRollover(String handoverPath, Boolean requireOperatorConfirm,
|
||||||
Integer maxDocAgeSeconds, Integer turnSettleSeconds,
|
Integer maxDocAgeSeconds, Integer turnSettleSeconds,
|
||||||
Integer clearSettleSeconds, String bootstrapText) {
|
Integer relaunchReadySeconds, String bootstrapText) {
|
||||||
public LeadRollover {
|
public LeadRollover {
|
||||||
requireOperatorConfirm = requireOperatorConfirm == null || requireOperatorConfirm;
|
requireOperatorConfirm = requireOperatorConfirm == null || requireOperatorConfirm;
|
||||||
maxDocAgeSeconds = (maxDocAgeSeconds == null || maxDocAgeSeconds <= 0) ? 3600 : maxDocAgeSeconds;
|
maxDocAgeSeconds = (maxDocAgeSeconds == null || maxDocAgeSeconds <= 0) ? 3600 : maxDocAgeSeconds;
|
||||||
turnSettleSeconds = (turnSettleSeconds == null || turnSettleSeconds <= 0) ? 20 : turnSettleSeconds;
|
turnSettleSeconds = (turnSettleSeconds == null || turnSettleSeconds <= 0) ? 300 : turnSettleSeconds;
|
||||||
clearSettleSeconds = (clearSettleSeconds == null || clearSettleSeconds <= 0) ? 20 : clearSettleSeconds;
|
relaunchReadySeconds = (relaunchReadySeconds == null || relaunchReadySeconds <= 0)
|
||||||
|
? 45 : relaunchReadySeconds;
|
||||||
bootstrapText = (bootstrapText == null || bootstrapText.isBlank()) ? null : bootstrapText;
|
bootstrapText = (bootstrapText == null || bootstrapText.isBlank()) ? null : bootstrapText;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The text actually sent to the lead's pane once it settles after {@code /clear}: the
|
* The text actually sent to the fresh lead's pane once it reaches a real turn boundary
|
||||||
* operator's configured {@link #bootstrapText} when one is set, otherwise the default
|
* after relaunch: the operator's configured {@link #bootstrapText} when one is set,
|
||||||
* sentence built from {@code resolvedHandoverPath}.
|
* otherwise the default sentence built from {@code resolvedHandoverPath}.
|
||||||
*
|
*
|
||||||
* @param resolvedHandoverPath the ABSOLUTE path {@code dev.ltms.fleet.lead.LeadRollover
|
* @param resolvedHandoverPath the ABSOLUTE path {@code dev.ltms.fleet.lead.LeadRollover
|
||||||
* #open} already resolved — never the raw configured {@link
|
* #open} already resolved — never the raw configured {@link
|
||||||
@@ -1670,7 +1746,7 @@ public record FleetConfig(
|
|||||||
* <p>A herdr pane runs a login shell that re-sources the operator's own secret store, so a
|
* <p>A herdr pane runs a login shell that re-sources the operator's own secret store, so a
|
||||||
* member inherits every credential the operator's shell holds — measured at 31 names on this
|
* member inherits every credential the operator's shell holds — measured at 31 names on this
|
||||||
* host, of which only one ({@code GITEA_ACCESS_TOKEN}) used to be blocked, and that block was a
|
* host, of which only one ({@code GITEA_ACCESS_TOKEN}) used to be blocked, and that block was a
|
||||||
* single name hardcoded in {@link HerdrPeerLauncher} rather than driven by config (gitea issue
|
* single name hardcoded in {@link dev.ltms.fleet.member.HerdrPeerLauncher} rather than driven by config (gitea issue
|
||||||
* #82). This record replaces that hardcoded shadow with a config-driven one.
|
* #82). This record replaces that hardcoded shadow with a config-driven one.
|
||||||
*
|
*
|
||||||
* <p><b>deny-by-default, not a deny-list.</b> A deny-list (block these specific names, let
|
* <p><b>deny-by-default, not a deny-list.</b> A deny-list (block these specific names, let
|
||||||
@@ -1854,6 +1930,8 @@ public record FleetConfig(
|
|||||||
rejectDuplicateMemberSlots(yaml);
|
rejectDuplicateMemberSlots(yaml);
|
||||||
rejectNegativeMaxLoad(yaml);
|
rejectNegativeMaxLoad(yaml);
|
||||||
rejectAutoCompactWindowOutOfRange(yaml);
|
rejectAutoCompactWindowOutOfRange(yaml);
|
||||||
|
warnConflictingAutoCompactWindows(yaml);
|
||||||
|
warnRetiredClearSettleSecondsKey(yaml);
|
||||||
rejectMalformedProfilePatterns(yaml);
|
rejectMalformedProfilePatterns(yaml);
|
||||||
rejectUnknownKind(yaml);
|
rejectUnknownKind(yaml);
|
||||||
rejectUnknownAuthMode(yaml);
|
rejectUnknownAuthMode(yaml);
|
||||||
@@ -1872,7 +1950,7 @@ public record FleetConfig(
|
|||||||
|
|
||||||
/** The {@code fleet:} child blocks whose direct children are slot names. */
|
/** The {@code fleet:} child blocks whose direct children are slot names. */
|
||||||
private static final Set<String> FLEET_POOL_KEYS =
|
private static final Set<String> FLEET_POOL_KEYS =
|
||||||
Set.of("leaders", "architects", "developers", "reviewers");
|
Set.of("leaders", "architects", "developers", "hunters", "reviewers", "collaborators");
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Reject a {@code fleet:} role pool whose slot names repeat (CB-548, re-homed by CB-557).
|
* Reject a {@code fleet:} role pool whose slot names repeat (CB-548, re-homed by CB-557).
|
||||||
@@ -1882,7 +1960,7 @@ public record FleetConfig(
|
|||||||
* daemon would never know. Jackson's YAML parser does not fail on duplicate mapping keys by
|
* daemon would never know. Jackson's YAML parser does not fail on duplicate mapping keys by
|
||||||
* default, so duplicates are caught here, at parse time, before the map is built.
|
* default, so duplicates are caught here, at parse time, before the map is built.
|
||||||
*
|
*
|
||||||
* <p>Only the four pools <em>directly under the top-level {@code fleet:}</em> are considered,
|
* <p>Only the six pools <em>directly under the top-level {@code fleet:}</em> are considered,
|
||||||
* and only their direct child keys (the slot names). A nested field elsewhere, even one also
|
* and only their direct child keys (the slot names). A nested field elsewhere, even one also
|
||||||
* named {@code developers:}, is ignored, so parsing of the rest of the config is unaffected.
|
* named {@code developers:}, is ignored, so parsing of the rest of the config is unaffected.
|
||||||
*
|
*
|
||||||
@@ -2049,8 +2127,9 @@ public record FleetConfig(
|
|||||||
"defaultProfile", "a role pool under 'fleet:' — an unqualified spawn now names a role,"
|
"defaultProfile", "a role pool under 'fleet:' — an unqualified spawn now names a role,"
|
||||||
+ " and that role's pool supplies the candidate profiles",
|
+ " and that role's pool supplies the candidate profiles",
|
||||||
"architects", "'fleet.architects'",
|
"architects", "'fleet.architects'",
|
||||||
"members", "a role pool under 'fleet:' — 'fleet.architects', 'fleet.developers' or"
|
"members", "a role pool under 'fleet:' — 'fleet.architects', 'fleet.developers',"
|
||||||
+ " 'fleet.reviewers'; the role is the containing key, not a 'role:' field",
|
+ " 'fleet.hunters' or 'fleet.reviewers'; the role is the containing key, not"
|
||||||
|
+ " a 'role:' field",
|
||||||
"leaders", "'fleet.leaders'",
|
"leaders", "'fleet.leaders'",
|
||||||
"leadScan", "'fleet.leaders.<name>.tabPrefix' and '.scanIntervalSeconds' — lead"
|
"leadScan", "'fleet.leaders.<name>.tabPrefix' and '.scanIntervalSeconds' — lead"
|
||||||
+ " discovery is now configured on the lead it discovers");
|
+ " discovery is now configured on the lead it discovers");
|
||||||
@@ -2161,6 +2240,12 @@ public record FleetConfig(
|
|||||||
static final int AUTO_COMPACT_WINDOW_MIN = 100_000;
|
static final int AUTO_COMPACT_WINDOW_MIN = 100_000;
|
||||||
/** Highest {@code autoCompactWindow} Claude Code's {@code --autocompact <tokens>} flag accepts. */
|
/** Highest {@code autoCompactWindow} Claude Code's {@code --autocompact <tokens>} flag accepts. */
|
||||||
static final int AUTO_COMPACT_WINDOW_MAX = 1_000_000;
|
static final int AUTO_COMPACT_WINDOW_MAX = 1_000_000;
|
||||||
|
/**
|
||||||
|
* The {@code env:} key a launched Claude Code session reads for its auto-compaction window,
|
||||||
|
* ahead of the {@code --autocompact} launch flag {@code autoCompactWindow} produces (see
|
||||||
|
* {@link Profile#effectiveAutoCompactWindow()}).
|
||||||
|
*/
|
||||||
|
static final String CLAUDE_CODE_AUTO_COMPACT_WINDOW_ENV = "CLAUDE_CODE_AUTO_COMPACT_WINDOW";
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Reject a profile whose {@code autoCompactWindow:} is set but outside the token band Claude
|
* Reject a profile whose {@code autoCompactWindow:} is set but outside the token band Claude
|
||||||
@@ -2203,6 +2288,100 @@ public record FleetConfig(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Warn (never refuse to start) about a Claude Code profile whose auto-compaction flag and
|
||||||
|
* environment setting disagree.
|
||||||
|
*
|
||||||
|
* <p>Renamed from {@code rejectConflictingAutoCompactWindows} (fleetd #601 review, measured
|
||||||
|
* 2026-09-22): that method threw {@link IllegalStateException}, so {@link #load(Path)} refused
|
||||||
|
* to start on a config carrying this conflict. On this host, four profiles trip it, including
|
||||||
|
* the lead's own profile and the one every worker spawns on — so the throw is not a rare edge
|
||||||
|
* case. Under launchd, a throw inside {@code load()} is a restart loop, not an error an operator
|
||||||
|
* reads once, and the config that would fix it ({@code fleetd/fleetd.yaml}) is gitignored, so
|
||||||
|
* the cause is invisible on the host where it bites. A WARN gives the operator the same
|
||||||
|
* information — which profiles, and now both values, so they can fix it without reading the
|
||||||
|
* source — without ever taking the fleet down.
|
||||||
|
*
|
||||||
|
* <p>fleetd #618 measured which of the two inputs Claude Code actually follows when they
|
||||||
|
* disagree: the environment variable wins, so {@code autoCompactWindow} is inert on a profile
|
||||||
|
* that also sets the env var. This method only detects and reports the disagreement — it does
|
||||||
|
* not correct it — see {@link dev.ltms.fleet.launch.ClaudeCodeArguments} for the full measured
|
||||||
|
* precedence.
|
||||||
|
*
|
||||||
|
* <p>Equal values never warn: either input then produces the same session window, so there is
|
||||||
|
* nothing to reconcile.
|
||||||
|
*/
|
||||||
|
static void warnConflictingAutoCompactWindows(String yaml) {
|
||||||
|
Map<?, ?> raw;
|
||||||
|
try {
|
||||||
|
raw = YAML.readValue(yaml, Map.class);
|
||||||
|
} catch (IOException | IllegalArgumentException e) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (raw == null || !(raw.get("profiles") instanceof Map<?, ?> profiles)) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
List<String> names = new ArrayList<>();
|
||||||
|
List<String> detail = new ArrayList<>();
|
||||||
|
for (Map.Entry<?, ?> entry : profiles.entrySet()) {
|
||||||
|
if (!(entry.getValue() instanceof Map<?, ?> profile)
|
||||||
|
|| !(profile.get("autoCompactWindow") instanceof Number window)
|
||||||
|
|| !(profile.get("env") instanceof Map<?, ?> env)
|
||||||
|
|| !env.containsKey(CLAUDE_CODE_AUTO_COMPACT_WINDOW_ENV)) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
Object kind = profile.get("kind");
|
||||||
|
boolean claudeCode = kind == null || String.valueOf(kind).isBlank()
|
||||||
|
|| Profile.KIND_CLAUDE_CODE.equalsIgnoreCase(String.valueOf(kind));
|
||||||
|
Object envValue = env.get(CLAUDE_CODE_AUTO_COMPACT_WINDOW_ENV);
|
||||||
|
if (claudeCode && !String.valueOf(window).equals(String.valueOf(envValue))) {
|
||||||
|
String name = String.valueOf(entry.getKey());
|
||||||
|
names.add(name);
|
||||||
|
detail.add(name + " (autoCompactWindow=" + window
|
||||||
|
+ ", env.CLAUDE_CODE_AUTO_COMPACT_WINDOW=" + envValue + ")");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (names.isEmpty()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
names.sort(String::compareTo);
|
||||||
|
detail.sort(String::compareTo);
|
||||||
|
log.warn("Claude Code profile(s) {} set disagreeing autoCompactWindow and env."
|
||||||
|
+ "CLAUDE_CODE_AUTO_COMPACT_WINDOW — the daemon starts anyway: {}. fleetd "
|
||||||
|
+ "#618 measured that CLAUDE_CODE_AUTO_COMPACT_WINDOW wins, so "
|
||||||
|
+ "autoCompactWindow is inert on these profiles. Set equal values on each "
|
||||||
|
+ "to resolve this — do not just delete the env var, since that LOWERS the "
|
||||||
|
+ "live window to autoCompactWindow's value rather than fixing anything.",
|
||||||
|
names, String.join(", ", detail));
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Warn when a {@code leadRollover:} block still sets the retired {@code clearSettleSeconds}
|
||||||
|
* key. {@link LeadRollover} carries {@code @JsonIgnoreProperties(ignoreUnknown = true)} and no
|
||||||
|
* longer declares that component, so Jackson drops it with no signal of its own — this raw-YAML
|
||||||
|
* check is the only place an operator's now-inert setting is reported at all; by the time a
|
||||||
|
* {@link LeadRollover} instance exists to run a validator against, the key is already gone.
|
||||||
|
*
|
||||||
|
* @param yaml the raw config text
|
||||||
|
*/
|
||||||
|
static void warnRetiredClearSettleSecondsKey(String yaml) {
|
||||||
|
Map<?, ?> raw;
|
||||||
|
try {
|
||||||
|
raw = YAML.readValue(yaml, Map.class);
|
||||||
|
} catch (IOException | IllegalArgumentException e) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (raw == null || !(raw.get("leadRollover") instanceof Map<?, ?> leadRollover)) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (leadRollover.containsKey("clearSettleSeconds")) {
|
||||||
|
log.warn("leadRollover.clearSettleSeconds is retired and no longer read. Set "
|
||||||
|
+ "leadRollover.relaunchReadySeconds instead: it bounds how long to wait, after "
|
||||||
|
+ "a lead is relaunched, for its pane to become ready and then for it to be "
|
||||||
|
+ "recognised as a lead. Remove clearSettleSeconds from fleetd.yaml.");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Reject a profile whose {@code errorPattern} (fleetd #201 Unit 5) or {@code exhaustedPattern}
|
* Reject a profile whose {@code errorPattern} (fleetd #201 Unit 5) or {@code exhaustedPattern}
|
||||||
* (CB-578 stage A) is not a valid Java regex, naming the profile, the key, and the parser's own
|
* (CB-578 stage A) is not a valid Java regex, naming the profile, the key, and the parser's own
|
||||||
@@ -2567,30 +2746,21 @@ public record FleetConfig(
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Reject a lead-scan convention that a worker tab would also satisfy (CB-531).
|
* Reject a member tab-label template that could render as a configured lead or collaborator
|
||||||
|
* tab or match a lead-tab naming convention, and reject two {@code fleet.leaders} or
|
||||||
|
* {@code fleet.collaborators} entries — across either registry — that share one exact tab.
|
||||||
*
|
*
|
||||||
* <p>The scan reads a tab label and concludes "a lead lives here". fleetd also <em>writes</em>
|
* <p>{@code fleet.collaborators} has no {@code tabPrefix}: identity is matched on the exact
|
||||||
* tab labels — every member gets one rendered into its tab. Choose a lead {@code tabPrefix} that
|
* {@code tab} alone, so only the exact-render check applies there, not the prefix check.
|
||||||
* a member template matches and the daemon starts labelling its own members as leads, promoting
|
|
||||||
* the entire fleet to {@link dev.ltms.fleet.auth.Role#PRIMARY} with no message and no diff.
|
|
||||||
* The member-space exclusion in {@link dev.ltms.fleet.herdr.LeadTabScanner} already blocks the
|
|
||||||
* realistic path, but defence that depends on one workspace label holding is not defence enough
|
|
||||||
* for a privilege boundary.
|
|
||||||
*
|
*
|
||||||
* <p>CB-557 shrank this check rather than removing it. The default template is
|
* @throws IllegalStateException when the fleet template or a profile {@code tabLabel} override
|
||||||
* {@code "{role}: {profile} #{n}"} and {@code {role}} comes from a closed enum, so a
|
* can render as a configured lead or collaborator tab or match a
|
||||||
* <em>generated</em> label can no longer collide by construction. What remains checkable is what
|
* lead-tab prefix, or when two entries — of either registry, or
|
||||||
* an operator still writes by hand: the {@code fleet.tabLabel} template and any per-profile
|
* one of each — carry the same exact {@code tab}
|
||||||
* {@code tabLabel} override.
|
* (case-insensitively)
|
||||||
*
|
|
||||||
* <p>Fatal rather than a warning, unlike {@link #warnUnknownTopLevelKeys}: an unknown key means
|
|
||||||
* a feature does nothing, while this means a feature does the opposite of what it says.
|
|
||||||
*
|
|
||||||
* @throws IllegalStateException when the fleet template or any profile's {@code tabLabel}
|
|
||||||
* override starts with a configured lead prefix
|
|
||||||
*/
|
*/
|
||||||
public void validateLeadTabPrefixes() {
|
public void validateLeadTabPrefixes() {
|
||||||
if (fleet == null || fleet.leaders().isEmpty()) {
|
if (fleet == null) {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
List<String> bad = new ArrayList<>();
|
List<String> bad = new ArrayList<>();
|
||||||
@@ -2598,28 +2768,193 @@ public record FleetConfig(
|
|||||||
if (leader == null) {
|
if (leader == null) {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
String tab = leader.tab();
|
||||||
String prefix = leader.tabPrefix();
|
String prefix = leader.tabPrefix();
|
||||||
// The fleet-wide template is checked once per prefix: it labels every member that has no
|
if (templateCanRenderAs(fleet.tabLabel(), tab)) {
|
||||||
// override, so one bad template promotes the entire fleet, not one profile.
|
bad.add("fleet.tabLabel=\"" + fleet.tabLabel() + "\" can render as the tab of "
|
||||||
if (startsWithIgnoreCase(fleet.tabLabel(), prefix)) {
|
+ "lead '" + leadName + "' (\"" + tab + "\")");
|
||||||
|
} else if (startsWithIgnoreCase(fleet.tabLabel(), prefix)) {
|
||||||
bad.add("fleet.tabLabel=\"" + fleet.tabLabel() + "\" starts with the tabPrefix of "
|
bad.add("fleet.tabLabel=\"" + fleet.tabLabel() + "\" starts with the tabPrefix of "
|
||||||
+ "lead '" + leadName + "' (\"" + prefix + "\")");
|
+ "lead '" + leadName + "' (\"" + prefix + "\")");
|
||||||
}
|
}
|
||||||
profiles().entrySet().stream()
|
profiles().entrySet().stream()
|
||||||
.filter(e -> startsWithIgnoreCase(e.getValue().tabLabel(), prefix))
|
|
||||||
.map(Map.Entry::getKey)
|
.map(Map.Entry::getKey)
|
||||||
.sorted()
|
.sorted()
|
||||||
.forEach(p -> bad.add("profile '" + p + "' overrides tabLabel with \""
|
.forEach(p -> {
|
||||||
+ profiles().get(p).tabLabel() + "\", which starts with the tabPrefix of "
|
String label = profiles().get(p).tabLabel();
|
||||||
+ "lead '" + leadName + "' (\"" + prefix + "\")"));
|
if (templateCanRenderAs(label, tab)) {
|
||||||
|
bad.add("profile '" + p + "' overrides tabLabel with \"" + label
|
||||||
|
+ "\", which can render as the tab of lead '" + leadName
|
||||||
|
+ "' (\"" + tab + "\")");
|
||||||
|
} else if (startsWithIgnoreCase(label, prefix)) {
|
||||||
|
bad.add("profile '" + p + "' overrides tabLabel with \"" + label
|
||||||
|
+ "\", which starts with the tabPrefix of lead '" + leadName
|
||||||
|
+ "' (\"" + prefix + "\")");
|
||||||
|
}
|
||||||
|
});
|
||||||
});
|
});
|
||||||
|
fleet.collaborators().forEach((collabName, collaborator) -> {
|
||||||
|
if (collaborator == null) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
String tab = collaborator.tab();
|
||||||
|
if (templateCanRenderAs(fleet.tabLabel(), tab)) {
|
||||||
|
bad.add("fleet.tabLabel=\"" + fleet.tabLabel() + "\" can render as the tab of "
|
||||||
|
+ "collaborator '" + collabName + "' (\"" + tab + "\")");
|
||||||
|
}
|
||||||
|
profiles().entrySet().stream()
|
||||||
|
.map(Map.Entry::getKey)
|
||||||
|
.sorted()
|
||||||
|
.forEach(p -> {
|
||||||
|
String label = profiles().get(p).tabLabel();
|
||||||
|
if (templateCanRenderAs(label, tab)) {
|
||||||
|
bad.add("profile '" + p + "' overrides tabLabel with \"" + label
|
||||||
|
+ "\", which can render as the tab of collaborator '"
|
||||||
|
+ collabName + "' (\"" + tab + "\")");
|
||||||
|
}
|
||||||
|
});
|
||||||
|
});
|
||||||
|
if (!bad.isEmpty()) {
|
||||||
|
throw new IllegalStateException("refusing to start: " + String.join("; ", bad)
|
||||||
|
+ ". A member labelled that way, while its pane carries no entry in the "
|
||||||
|
+ "spawned-member roster, is read back as a lead or collaborator and granted "
|
||||||
|
+ "that identity's authority. Change one of the two so member tabs cannot be "
|
||||||
|
+ "confused with a lead's or collaborator's tab.");
|
||||||
|
}
|
||||||
|
|
||||||
|
List<String> collisions = new ArrayList<>();
|
||||||
|
List<String> leadNames = fleet.leaders().keySet().stream().sorted().toList();
|
||||||
|
for (int i = 0; i < leadNames.size(); i++) {
|
||||||
|
String nameA = leadNames.get(i);
|
||||||
|
Leader a = fleet.leaders().get(nameA);
|
||||||
|
if (a == null || a.tab() == null || a.tab().isBlank()) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
for (int j = i + 1; j < leadNames.size(); j++) {
|
||||||
|
String nameB = leadNames.get(j);
|
||||||
|
Leader b = fleet.leaders().get(nameB);
|
||||||
|
if (b == null || b.tab() == null || b.tab().isBlank()) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (a.tab().equalsIgnoreCase(b.tab())) {
|
||||||
|
collisions.add("lead '" + nameA + "' and lead '" + nameB + "' both use tab \""
|
||||||
|
+ a.tab() + "\"");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
List<String> collabNames = fleet.collaborators().keySet().stream().sorted().toList();
|
||||||
|
for (int i = 0; i < collabNames.size(); i++) {
|
||||||
|
String nameA = collabNames.get(i);
|
||||||
|
Collaborator a = fleet.collaborators().get(nameA);
|
||||||
|
if (a == null || a.tab() == null || a.tab().isBlank()) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
for (int j = i + 1; j < collabNames.size(); j++) {
|
||||||
|
String nameB = collabNames.get(j);
|
||||||
|
Collaborator b = fleet.collaborators().get(nameB);
|
||||||
|
if (b == null || b.tab() == null || b.tab().isBlank()) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (a.tab().equalsIgnoreCase(b.tab())) {
|
||||||
|
collisions.add("collaborator '" + nameA + "' and collaborator '" + nameB
|
||||||
|
+ "' both use tab \"" + a.tab() + "\"");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (String leadName : leadNames) {
|
||||||
|
Leader lead = fleet.leaders().get(leadName);
|
||||||
|
if (lead == null || lead.tab() == null || lead.tab().isBlank()) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
for (String collabName : collabNames) {
|
||||||
|
Collaborator collaborator = fleet.collaborators().get(collabName);
|
||||||
|
if (collaborator == null || collaborator.tab() == null
|
||||||
|
|| collaborator.tab().isBlank()) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (lead.tab().equalsIgnoreCase(collaborator.tab())) {
|
||||||
|
collisions.add("lead '" + leadName + "' and collaborator '" + collabName
|
||||||
|
+ "' both use tab \"" + lead.tab() + "\"");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (collisions.isEmpty()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
throw new IllegalStateException("refusing to start: " + String.join("; ", collisions)
|
||||||
|
+ ". Tab identity is matched exactly, so only one of two entries sharing a tab can "
|
||||||
|
+ "ever be found — the other is silently unreachable. Give each lead and "
|
||||||
|
+ "collaborator its own exact tab.");
|
||||||
|
}
|
||||||
|
|
||||||
|
private static boolean templateCanRenderAs(String template, String tab) {
|
||||||
|
if (template == null || template.isBlank() || tab == null || tab.isBlank()) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
var placeholders = Pattern.compile("\\{(?:role|profile|model|n)}").matcher(template);
|
||||||
|
StringBuilder expression = new StringBuilder("^");
|
||||||
|
int literalStart = 0;
|
||||||
|
while (placeholders.find()) {
|
||||||
|
expression.append(Pattern.quote(template.substring(literalStart, placeholders.start())));
|
||||||
|
expression.append(".*");
|
||||||
|
literalStart = placeholders.end();
|
||||||
|
}
|
||||||
|
expression.append(Pattern.quote(template.substring(literalStart))).append("$");
|
||||||
|
return Pattern.compile(expression.toString(), Pattern.CASE_INSENSITIVE).matcher(tab).matches();
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Case-insensitive prefix test that tolerates a null or blank label. */
|
||||||
|
private static boolean startsWithIgnoreCase(String label, String prefix) {
|
||||||
|
if (label == null || prefix == null || prefix.isBlank()) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
String stripped = label.strip();
|
||||||
|
return stripped.regionMatches(true, 0, prefix, 0, prefix.length());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reject a profile that places its members by {@code "pane"} while any {@code fleet.leaders}
|
||||||
|
* or {@code fleet.collaborators} entry names a {@code tab}. A pane-placed member lands inside
|
||||||
|
* the focused tab rather than its own, so it can land inside a lead's or collaborator's own
|
||||||
|
* labelled tab. {@link dev.ltms.fleet.herdr.LeadTabScanner} identifies a lead or collaborator
|
||||||
|
* purely by that tab's label — it does not exclude the member space — so a member that ends up
|
||||||
|
* there, while its pane carries no entry in the spawned-member roster, is read back as that
|
||||||
|
* lead or collaborator and granted that identity's authority.
|
||||||
|
*
|
||||||
|
* <p>Only an entry with a non-blank {@code tab} is in scope: one with no {@code tab} feeds
|
||||||
|
* nothing into {@link dev.ltms.fleet.herdr.LeadTabScanner}, so it creates no hazard here.
|
||||||
|
*
|
||||||
|
* @throws IllegalStateException when any {@code profiles:} entry is pane-placed while any
|
||||||
|
* {@code fleet.leaders} or {@code fleet.collaborators} entry
|
||||||
|
* names a non-blank {@code tab}
|
||||||
|
*/
|
||||||
|
public void validatePanePlacementAgainstLeadTabs() {
|
||||||
|
if (fleet == null) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
boolean anyLeaderHasTab = fleet.leaders().values().stream()
|
||||||
|
.anyMatch(leader -> leader != null && leader.tab() != null && !leader.tab().isBlank());
|
||||||
|
boolean anyCollaboratorHasTab = fleet.collaborators().values().stream()
|
||||||
|
.anyMatch(c -> c != null && c.tab() != null && !c.tab().isBlank());
|
||||||
|
if (!anyLeaderHasTab && !anyCollaboratorHasTab) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
List<String> bad = new ArrayList<>();
|
||||||
|
profiles().entrySet().stream()
|
||||||
|
.filter(e -> !e.getValue().tabPlacement())
|
||||||
|
.map(Map.Entry::getKey)
|
||||||
|
.sorted()
|
||||||
|
.forEach(bad::add);
|
||||||
if (bad.isEmpty()) {
|
if (bad.isEmpty()) {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
throw new IllegalStateException("refusing to start: " + String.join("; ", bad)
|
throw new IllegalStateException("refusing to start: profile(s) " + bad
|
||||||
+ ". Every member labelled that way would be read back as a lead and granted "
|
+ " use placement: pane while fleet.leaders or fleet.collaborators names a tab. A "
|
||||||
+ "spawn/stop/send on the whole fleet. Change one of the two so member tabs and "
|
+ "pane-placed member can land inside that labelled tab, and while its pane "
|
||||||
+ "lead tabs cannot be confused.");
|
+ "carries no entry in the spawned-member roster, it is read back as the lead or "
|
||||||
|
+ "collaborator and granted that identity's authority. Set placement: tab for "
|
||||||
|
+ "each named profile, or remove the tab from every fleet.leaders and "
|
||||||
|
+ "fleet.collaborators entry.");
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -2642,15 +2977,6 @@ public record FleetConfig(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Case-insensitive prefix test that tolerates a null/blank label. */
|
|
||||||
private static boolean startsWithIgnoreCase(String label, String prefix) {
|
|
||||||
if (label == null || prefix == null || prefix.isBlank()) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
String stripped = label.strip();
|
|
||||||
return stripped.regionMatches(true, 0, prefix, 0, prefix.length());
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Reject a subscription profile whose {@code env:} block tries to reseat the Anthropic binding
|
* Reject a subscription profile whose {@code env:} block tries to reseat the Anthropic binding
|
||||||
* (CB-542).
|
* (CB-542).
|
||||||
@@ -2735,8 +3061,14 @@ public record FleetConfig(
|
|||||||
* so duplicates are unrepresentable by construction once loaded — and {@link #load(Path)}
|
* so duplicates are unrepresentable by construction once loaded — and {@link #load(Path)}
|
||||||
* already rejects a duplicated slot name at parse time, before the map collapses.
|
* already rejects a duplicated slot name at parse time, before the map collapses.
|
||||||
*
|
*
|
||||||
* @throws IllegalStateException when a slot names no profile or an unknown one, or when a lead
|
* <p>Also rejects a {@code fleet.collaborators} entry with no (or a blank) {@code tab}. A
|
||||||
* can be neither found nor created, naming the offending entry
|
* {@code profile}-less lead is still useful recognise-only — {@code tab} is the only field
|
||||||
|
* that matters to it either way. A collaborator carries no other field at all, so a blank
|
||||||
|
* {@code tab} leaves nothing for the entry to mean.
|
||||||
|
*
|
||||||
|
* @throws IllegalStateException when a slot names no profile or an unknown one, when a lead
|
||||||
|
* can be neither found nor created, or when a collaborator names
|
||||||
|
* no tab, naming the offending entry
|
||||||
*/
|
*/
|
||||||
public void validateMembers() {
|
public void validateMembers() {
|
||||||
if (fleet == null) {
|
if (fleet == null) {
|
||||||
@@ -2774,6 +3106,16 @@ public record FleetConfig(
|
|||||||
+ "auto-launched, labelled) purely by its tab, so every entry must name one.");
|
+ "auto-launched, labelled) purely by its tab, so every entry must name one.");
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
fleet.collaborators().forEach((name, collaborator) -> {
|
||||||
|
if (collaborator == null) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (collaborator.tab() == null || collaborator.tab().isBlank()) {
|
||||||
|
bad.add("fleet.collaborators." + name + " has no tab: — a collaborator is "
|
||||||
|
+ "recognised purely by its tab, and carries no other field, so every "
|
||||||
|
+ "entry must name one.");
|
||||||
|
}
|
||||||
|
});
|
||||||
if (!bad.isEmpty()) {
|
if (!bad.isEmpty()) {
|
||||||
throw new IllegalStateException("refusing to start: " + String.join(" ", bad));
|
throw new IllegalStateException("refusing to start: " + String.join(" ", bad));
|
||||||
}
|
}
|
||||||
@@ -2822,11 +3164,11 @@ public record FleetConfig(
|
|||||||
* Runs every validator this class declares — found by reflection, not by name.
|
* Runs every validator this class declares — found by reflection, not by name.
|
||||||
*
|
*
|
||||||
* <p>fleetd ticket "central allow-list of usable models", follow-up: mutation testing found
|
* <p>fleetd ticket "central allow-list of usable models", follow-up: mutation testing found
|
||||||
* that although each of the six validators above was well pinned on its own, nothing proved
|
* that although each validator above was well pinned on its own, nothing proved
|
||||||
* either real caller ({@code Fleetd.main} and {@link ConfigRef#reload()}) still
|
* either real caller ({@code Fleetd.main} and {@link ConfigRef#reload()}) still
|
||||||
* invoked it — deleting a call site left the full suite green. The fix is not a seventh test
|
* invoked it — deleting a call site left the full suite green. The fix is not one more test
|
||||||
* per caller; a hand-maintained list of six names here would have the exact same defect its
|
* per caller; a hand-maintained list of names here would have the exact same defect its
|
||||||
* own javadoc would warn against: the seventh validator someone adds next month has no reason
|
* own javadoc would warn against: the next validator someone adds has no reason
|
||||||
* to be added to it. So this method does not name any validator. It sweeps {@link
|
* to be added to it. So this method does not name any validator. It sweeps {@link
|
||||||
* #getClass()}'s own public, no-argument, {@code void} methods whose name starts with {@code
|
* #getClass()}'s own public, no-argument, {@code void} methods whose name starts with {@code
|
||||||
* "validate"} (excluding itself) and invokes every one it finds, via {@link
|
* "validate"} (excluding itself) and invokes every one it finds, via {@link
|
||||||
@@ -2835,7 +3177,7 @@ public record FleetConfig(
|
|||||||
* which it silently never runs.
|
* which it silently never runs.
|
||||||
*
|
*
|
||||||
* <p>{@code Fleetd.main} and {@link ConfigRef#reload()} each call this one method instead of
|
* <p>{@code Fleetd.main} and {@link ConfigRef#reload()} each call this one method instead of
|
||||||
* the six individually — see the comments at those two call sites for why
|
* each validator individually — see the comments at those two call sites for why
|
||||||
* each must run it.
|
* each must run it.
|
||||||
*
|
*
|
||||||
* <p>Methods run in a fixed (alphabetical) order, so a config with more than one violation
|
* <p>Methods run in a fixed (alphabetical) order, so a config with more than one violation
|
||||||
@@ -2852,9 +3194,9 @@ public record FleetConfig(
|
|||||||
/**
|
/**
|
||||||
* The reflective sweep behind {@link #validateAll()}, kept as its own method — taking any
|
* The reflective sweep behind {@link #validateAll()}, kept as its own method — taking any
|
||||||
* {@code target}, not just {@code this} — so a test can prove the MECHANISM is generic (it
|
* {@code target}, not just {@code this} — so a test can prove the MECHANISM is generic (it
|
||||||
* would sweep a seventh {@code validateXxx()} method added to any class, not just something
|
* would sweep any new {@code validateXxx()} method added to any class, not just something
|
||||||
* special-cased to today's six on {@link FleetConfig}) without needing to add a real, unwanted
|
* special-cased to the set {@link FleetConfig} declares today) without needing to add a real,
|
||||||
* seventh validator to this class just to exercise that claim. See {@code
|
* unwanted extra validator to this class just to exercise that claim. See {@code
|
||||||
* FleetConfigValidateAllTest} for that proof.
|
* FleetConfigValidateAllTest} for that proof.
|
||||||
*
|
*
|
||||||
* @param target an object whose public, no-argument, {@code void} methods named {@code
|
* @param target an object whose public, no-argument, {@code void} methods named {@code
|
||||||
|
|||||||
@@ -11,12 +11,18 @@ public final class HerdrRouter implements AutoCloseable {
|
|||||||
private final AgentControl memberAgents;
|
private final AgentControl memberAgents;
|
||||||
private final WorkspaceControl leadSpaces;
|
private final WorkspaceControl leadSpaces;
|
||||||
private final WorkspaceControl memberSpaces;
|
private final WorkspaceControl memberSpaces;
|
||||||
private final Predicate<String> isLead;
|
private final Predicate<String> routeToLead;
|
||||||
|
|
||||||
public HerdrRouter(HerdrClient lead, HerdrClient member, Predicate<String> isLead) {
|
/**
|
||||||
|
* @param routeToLead true for a terminal whose pane lives in the lead herdr daemon — a lead's
|
||||||
|
* own pane or a configured collaborator's, both opened by a person at a
|
||||||
|
* terminal rather than spawned, so both are found in the lead daemon rather
|
||||||
|
* than the member one
|
||||||
|
*/
|
||||||
|
public HerdrRouter(HerdrClient lead, HerdrClient member, Predicate<String> routeToLead) {
|
||||||
this.lead = Objects.requireNonNull(lead, "lead");
|
this.lead = Objects.requireNonNull(lead, "lead");
|
||||||
this.member = member != null ? member : lead;
|
this.member = member != null ? member : lead;
|
||||||
this.isLead = Objects.requireNonNull(isLead, "isLead");
|
this.routeToLead = Objects.requireNonNull(routeToLead, "routeToLead");
|
||||||
leadAgents = new AgentControl(this.lead);
|
leadAgents = new AgentControl(this.lead);
|
||||||
memberAgents = this.member == this.lead ? leadAgents : new AgentControl(this.member);
|
memberAgents = this.member == this.lead ? leadAgents : new AgentControl(this.member);
|
||||||
leadSpaces = new WorkspaceControl(this.lead);
|
leadSpaces = new WorkspaceControl(this.lead);
|
||||||
@@ -27,7 +33,7 @@ public final class HerdrRouter implements AutoCloseable {
|
|||||||
public WorkspaceControl leadSpaces() { return leadSpaces; }
|
public WorkspaceControl leadSpaces() { return leadSpaces; }
|
||||||
public AgentControl memberAgents() { return memberAgents; }
|
public AgentControl memberAgents() { return memberAgents; }
|
||||||
public WorkspaceControl memberSpaces() { return memberSpaces; }
|
public WorkspaceControl memberSpaces() { return memberSpaces; }
|
||||||
public AgentControl agentsFor(String targetId) { return isLead.test(targetId) ? leadAgents : memberAgents; }
|
public AgentControl agentsFor(String targetId) { return routeToLead.test(targetId) ? leadAgents : memberAgents; }
|
||||||
|
|
||||||
HerdrClient leadClient() { return lead; }
|
HerdrClient leadClient() { return lead; }
|
||||||
HerdrClient memberClient() { return member; }
|
HerdrClient memberClient() { return member; }
|
||||||
|
|||||||
@@ -37,8 +37,11 @@ import java.util.function.Supplier;
|
|||||||
* <p><strong>Direction of trust.</strong> The label names the lead; it never <em>grants</em>
|
* <p><strong>Direction of trust.</strong> The label names the lead; it never <em>grants</em>
|
||||||
* anything a pane could take for itself. Three properties keep that honest:
|
* anything a pane could take for itself. Three properties keep that honest:
|
||||||
* <ol>
|
* <ol>
|
||||||
* <li>Worker spaces are excluded wholesale ({@code excludedWorkspaceLabels}), so a worker cannot
|
* <li>{@code excludedWorkspaceLabels} can filter a workspace out of the scan, but this class does
|
||||||
* become a lead by being placed — as a split, say — inside a matching tab.</li>
|
* not by itself stop a worker from landing inside a matching tab — a caller may pass an empty
|
||||||
|
* set, and the daemon does. The guard against that is {@code
|
||||||
|
* FleetConfig.validatePanePlacementAgainstLeadTabs}: it refuses, at startup, any profile that
|
||||||
|
* places members by pane while a lead names a tab.</li>
|
||||||
* <li>A worker cannot rename a tab: {@code tab.rename} is reachable only through
|
* <li>A worker cannot rename a tab: {@code tab.rename} is reachable only through
|
||||||
* {@link WorkspaceControl}, which no {@code fleet_*} tool exposes. The label is writable by
|
* {@link WorkspaceControl}, which no {@code fleet_*} tool exposes. The label is writable by
|
||||||
* the human at the terminal and by nobody the bridge is defending against.</li>
|
* the human at the terminal and by nobody the bridge is defending against.</li>
|
||||||
@@ -93,13 +96,19 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
|||||||
|
|
||||||
private static final Logger log = LoggerFactory.getLogger(LeadTabScanner.class);
|
private static final Logger log = LoggerFactory.getLogger(LeadTabScanner.class);
|
||||||
|
|
||||||
|
/** What a matched tab names: a lead or a collaborator. */
|
||||||
|
private enum Kind { LEAD, COLLABORATOR }
|
||||||
|
|
||||||
|
/** One matched tab's name and what it names. */
|
||||||
|
private record Entry(String name, Kind kind) {}
|
||||||
|
|
||||||
private final HerdrClient herdr;
|
private final HerdrClient herdr;
|
||||||
private final Map<String, String> tabToName;
|
private final Map<String, Entry> tabToEntry;
|
||||||
private final Set<String> excludedWorkspaceLabels;
|
private final Set<String> excludedWorkspaceLabels;
|
||||||
private final long ttlNanos;
|
private final long ttlNanos;
|
||||||
private final LongSupplier clock;
|
private final LongSupplier clock;
|
||||||
|
|
||||||
private Map<String, String> cached = Map.of();
|
private Map<String, Entry> cached = Map.of();
|
||||||
private long scannedAtNanos;
|
private long scannedAtNanos;
|
||||||
private boolean everScanned;
|
private boolean everScanned;
|
||||||
|
|
||||||
@@ -123,26 +132,53 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
|||||||
*/
|
*/
|
||||||
public LeadTabScanner(HerdrClient herdr, Map<String, String> tabToName,
|
public LeadTabScanner(HerdrClient herdr, Map<String, String> tabToName,
|
||||||
Set<String> excludedWorkspaceLabels, long ttlNanos, LongSupplier clock) {
|
Set<String> excludedWorkspaceLabels, long ttlNanos, LongSupplier clock) {
|
||||||
|
this(herdr, tabToName, Map.of(), excludedWorkspaceLabels, ttlNanos, clock);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* As {@link #LeadTabScanner(HerdrClient, Map, Set, long, LongSupplier)}, additionally scanning
|
||||||
|
* for configured collaborator tabs in the same pass.
|
||||||
|
*
|
||||||
|
* @param collaboratorTabToName every configured collaborator's exact tab label → its name
|
||||||
|
* ({@code fleet.collaborators.<name>.tab}), matched the same way as
|
||||||
|
* {@code tabToName}
|
||||||
|
*/
|
||||||
|
public LeadTabScanner(HerdrClient herdr, Map<String, String> tabToName,
|
||||||
|
Map<String, String> collaboratorTabToName,
|
||||||
|
Set<String> excludedWorkspaceLabels, long ttlNanos, LongSupplier clock) {
|
||||||
this.herdr = herdr;
|
this.herdr = herdr;
|
||||||
this.tabToName = normalize(tabToName);
|
this.tabToEntry = buildTabIndex(tabToName, collaboratorTabToName);
|
||||||
this.excludedWorkspaceLabels = excludedWorkspaceLabels == null
|
this.excludedWorkspaceLabels = excludedWorkspaceLabels == null
|
||||||
? Set.of() : Set.copyOf(excludedWorkspaceLabels);
|
? Set.of() : Set.copyOf(excludedWorkspaceLabels);
|
||||||
this.ttlNanos = ttlNanos;
|
this.ttlNanos = ttlNanos;
|
||||||
this.clock = clock;
|
this.clock = clock;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Keys stripped and lower-cased once, so every lookup is a plain map hit. */
|
/**
|
||||||
private static Map<String, String> normalize(Map<String, String> tabToName) {
|
* Keys stripped and lower-cased once, so every lookup is a plain map hit. Leads and
|
||||||
if (tabToName == null || tabToName.isEmpty()) {
|
* collaborators merge into a single index, so {@link #scan()} matches both kinds in one pass
|
||||||
return Map.of();
|
* over the tab list; a label naming both a lead and a collaborator takes the lead entry —
|
||||||
|
* leads are put last, so a colliding key's lead entry is the one that overwrites — since a lead
|
||||||
|
* can already do everything a collaborator can. Config validation already refuses a lead and a
|
||||||
|
* collaborator sharing one exact tab, so this ordering is defence in depth, not the control.
|
||||||
|
*/
|
||||||
|
private static Map<String, Entry> buildTabIndex(Map<String, String> tabToName,
|
||||||
|
Map<String, String> collaboratorTabToName) {
|
||||||
|
Map<String, Entry> out = new LinkedHashMap<>();
|
||||||
|
putNormalized(out, collaboratorTabToName, Kind.COLLABORATOR);
|
||||||
|
putNormalized(out, tabToName, Kind.LEAD);
|
||||||
|
return Collections.unmodifiableMap(out);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void putNormalized(Map<String, Entry> out, Map<String, String> tabToName, Kind kind) {
|
||||||
|
if (tabToName == null) {
|
||||||
|
return;
|
||||||
}
|
}
|
||||||
Map<String, String> out = new LinkedHashMap<>();
|
|
||||||
tabToName.forEach((tab, name) -> {
|
tabToName.forEach((tab, name) -> {
|
||||||
if (tab != null && !tab.isBlank() && name != null && !name.isBlank()) {
|
if (tab != null && !tab.isBlank() && name != null && !name.isBlank()) {
|
||||||
out.put(tab.strip().toLowerCase(Locale.ROOT), name);
|
out.put(tab.strip().toLowerCase(Locale.ROOT), new Entry(name, kind));
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
return Collections.unmodifiableMap(out);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -153,6 +189,29 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
|||||||
*/
|
*/
|
||||||
@Override
|
@Override
|
||||||
public synchronized Map<String, String> get() {
|
public synchronized Map<String, String> get() {
|
||||||
|
return byKind(refresh(), Kind.LEAD);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The current {@code terminal_id → collaborator name} map, sharing the same scan and cache as
|
||||||
|
* {@link #get()} — both kinds are matched in one pass, so this never costs a second herdr call.
|
||||||
|
*/
|
||||||
|
public synchronized Map<String, String> collaborators() {
|
||||||
|
return byKind(refresh(), Kind.COLLABORATOR);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Map<String, String> byKind(Map<String, Entry> entries, Kind kind) {
|
||||||
|
Map<String, String> out = new LinkedHashMap<>();
|
||||||
|
entries.forEach((terminal, entry) -> {
|
||||||
|
if (entry.kind() == kind) {
|
||||||
|
out.put(terminal, entry.name());
|
||||||
|
}
|
||||||
|
});
|
||||||
|
return Collections.unmodifiableMap(out);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Rescans if the cache has expired, otherwise returns the cached answer. */
|
||||||
|
private Map<String, Entry> refresh() {
|
||||||
long now = clock.getAsLong();
|
long now = clock.getAsLong();
|
||||||
if (everScanned && now - scannedAtNanos < ttlNanos) {
|
if (everScanned && now - scannedAtNanos < ttlNanos) {
|
||||||
return cached;
|
return cached;
|
||||||
@@ -162,21 +221,21 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
|||||||
scannedAtNanos = now;
|
scannedAtNanos = now;
|
||||||
everScanned = true;
|
everScanned = true;
|
||||||
try {
|
try {
|
||||||
Map<String, String> fresh = scan();
|
Map<String, Entry> fresh = scan();
|
||||||
if (!fresh.equals(cached)) {
|
if (!fresh.equals(cached)) {
|
||||||
log.info("lead panes: {}", fresh);
|
log.info("lead/collaborator panes: {}", fresh);
|
||||||
}
|
}
|
||||||
cached = fresh;
|
cached = fresh;
|
||||||
} catch (HerdrException e) {
|
} catch (HerdrException e) {
|
||||||
log.warn("lead-tab scan failed, keeping the {} lead(s) already known: {}",
|
log.warn("lead-tab scan failed, keeping the {} entr(y/ies) already known: {}",
|
||||||
cached.size(), e.getMessage());
|
cached.size(), e.getMessage());
|
||||||
}
|
}
|
||||||
return cached;
|
return cached;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** One full pass: labelled tabs → live agents in them → those panes' terminals. */
|
/** One full pass: labelled tabs → live agents in them → those panes' terminals. */
|
||||||
private Map<String, String> scan() {
|
private Map<String, Entry> scan() {
|
||||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
Map<String, Entry> entryByTab = new LinkedHashMap<>();
|
||||||
for (JsonNode w : herdr.call("workspace.list").path("workspaces")) {
|
for (JsonNode w : herdr.call("workspace.list").path("workspaces")) {
|
||||||
Workspace ws = Workspace.from(w);
|
Workspace ws = Workspace.from(w);
|
||||||
if (ws.workspaceId() == null || excludedWorkspaceLabels.contains(ws.label())) {
|
if (ws.workspaceId() == null || excludedWorkspaceLabels.contains(ws.label())) {
|
||||||
@@ -184,21 +243,22 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
|||||||
}
|
}
|
||||||
for (JsonNode t : herdr.call("tab.list", Map.of("workspace_id", ws.workspaceId())).path("tabs")) {
|
for (JsonNode t : herdr.call("tab.list", Map.of("workspace_id", ws.workspaceId())).path("tabs")) {
|
||||||
Tab tab = Tab.from(t);
|
Tab tab = Tab.from(t);
|
||||||
String name = leadNameOf(tab.label());
|
Entry entry = entryOf(tab.label());
|
||||||
if (name != null && tab.tabId() != null) {
|
if (entry != null && tab.tabId() != null) {
|
||||||
nameByTab.put(tab.tabId(), name);
|
entryByTab.put(tab.tabId(), entry);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if (nameByTab.isEmpty()) {
|
if (entryByTab.isEmpty()) {
|
||||||
gracedTerminals = Set.of();
|
gracedTerminals = Set.of();
|
||||||
return Map.of();
|
return Map.of();
|
||||||
}
|
}
|
||||||
|
|
||||||
// fleetd #359: a labelled tab is only a lead when herdr also reports a running agent in
|
// fleetd #359: a labelled tab is only a lead (or collaborator) when herdr also reports a
|
||||||
// it — the same liveness signal LeadLauncher.countLeads trusts for the identical purpose.
|
// running agent in it — the same liveness signal LeadLauncher.countLeads trusts for the
|
||||||
// Without this, a tab left behind by a session that has since died reads as live forever.
|
// identical purpose. Without this, a tab left behind by a session that has since died reads
|
||||||
|
// as live forever.
|
||||||
Set<String> tabsWithAgent = new HashSet<>();
|
Set<String> tabsWithAgent = new HashSet<>();
|
||||||
for (JsonNode a : herdr.call("agent.list").path("agents")) {
|
for (JsonNode a : herdr.call("agent.list").path("agents")) {
|
||||||
String tabId = a.path("tab_id").asText(null);
|
String tabId = a.path("tab_id").asText(null);
|
||||||
@@ -207,18 +267,18 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Map<String, String> byTerminal = new LinkedHashMap<>();
|
Map<String, Entry> byTerminal = new LinkedHashMap<>();
|
||||||
Set<String> stillGraced = new HashSet<>();
|
Set<String> stillGraced = new HashSet<>();
|
||||||
// One pane.list for every tab: panes carry tab_id, so the join is local.
|
// One pane.list for every tab: panes carry tab_id, so the join is local.
|
||||||
for (JsonNode p : herdr.call("pane.list", Map.of()).path("panes")) {
|
for (JsonNode p : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||||
String tabId = p.path("tab_id").asText(null);
|
String tabId = p.path("tab_id").asText(null);
|
||||||
String name = nameByTab.get(tabId);
|
Entry entry = entryByTab.get(tabId);
|
||||||
String terminal = p.path("terminal_id").asText(null);
|
String terminal = p.path("terminal_id").asText(null);
|
||||||
if (name == null || terminal == null || terminal.isBlank()) {
|
if (entry == null || terminal == null || terminal.isBlank()) {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if (tabsWithAgent.contains(tabId)) {
|
if (tabsWithAgent.contains(tabId)) {
|
||||||
byTerminal.put(terminal, name);
|
byTerminal.put(terminal, entry);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
// No agent reported for this tab, but its tab/pane are still here — this is the
|
// No agent reported for this tab, but its tab/pane are still here — this is the
|
||||||
@@ -227,7 +287,7 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
|||||||
// reported as live; a terminal we never reported live gets none, so the original #359
|
// reported as live; a terminal we never reported live gets none, so the original #359
|
||||||
// fix (a genuinely dead tab is never reported) is unaffected for the common case.
|
// fix (a genuinely dead tab is never reported) is unaffected for the common case.
|
||||||
if (cached.containsKey(terminal) && !gracedTerminals.contains(terminal)) {
|
if (cached.containsKey(terminal) && !gracedTerminals.contains(terminal)) {
|
||||||
byTerminal.put(terminal, name);
|
byTerminal.put(terminal, entry);
|
||||||
stillGraced.add(terminal);
|
stillGraced.add(terminal);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -236,18 +296,19 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The lead name a tab label declares, or {@code null} if it names none of the configured leads.
|
* The entry a tab label declares, or {@code null} if it names neither a configured lead nor a
|
||||||
|
* configured collaborator.
|
||||||
*
|
*
|
||||||
* <p>Exact match (case-insensitive, ends stripped) against {@link #tabToName} — no prefix
|
* <p>Exact match (case-insensitive, ends stripped) against {@link #tabToEntry} — no prefix
|
||||||
* stripping, so an operator's {@code "lead: something-else"} tab is never mistaken for a
|
* stripping, so an operator's {@code "lead: something-else"} tab is never mistaken for a
|
||||||
* configured lead just because it shares a prefix. The match strips a trailing
|
* configured lead just because it shares a prefix. The match strips a trailing
|
||||||
* {@link PendingCloseMarker} first, so a tab {@code LeadLauncher} has flagged as maybe-dead but
|
* {@link PendingCloseMarker} first, so a tab {@code LeadLauncher} has flagged as maybe-dead but
|
||||||
* not yet closed keeps resolving normally while that reconcile is pending.
|
* not yet closed keeps resolving normally while that reconcile is pending.
|
||||||
*/
|
*/
|
||||||
private String leadNameOf(String label) {
|
private Entry entryOf(String label) {
|
||||||
if (label == null) {
|
if (label == null) {
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
return tabToName.get(PendingCloseMarker.strip(label).toLowerCase(Locale.ROOT));
|
return tabToEntry.get(PendingCloseMarker.strip(label).toLowerCase(Locale.ROOT));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,6 +1,8 @@
|
|||||||
package dev.ltms.fleet.herdr;
|
package dev.ltms.fleet.herdr;
|
||||||
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
import org.slf4j.Logger;
|
||||||
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import java.util.LinkedHashSet;
|
import java.util.LinkedHashSet;
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
@@ -36,6 +38,8 @@ import java.util.Set;
|
|||||||
*/
|
*/
|
||||||
public final class PaneLocator {
|
public final class PaneLocator {
|
||||||
|
|
||||||
|
private static final Logger log = LoggerFactory.getLogger(PaneLocator.class);
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Bound on how many ancestor generations {@link #ancestorsOf} walks. This runs on every MCP
|
* Bound on how many ancestor generations {@link #ancestorsOf} walks. This runs on every MCP
|
||||||
* call, so a cycle or a pathologically deep process tree must not hang identity resolution;
|
* call, so a cycle or a pathologically deep process tree must not hang identity resolution;
|
||||||
@@ -73,22 +77,46 @@ public final class PaneLocator {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The {@code terminal_id} of the agent pane whose process tree contains {@code pid}, or
|
* The outcome of a {@link #terminalForPid} scan: the {@code terminal_id} of the agent pane
|
||||||
* {@code null} if no agent pane on any searched daemon owns it (e.g. the caller is the
|
* whose process tree contains the pid ({@link #terminal} is {@code null} if none matched),
|
||||||
* primary, or off-host).
|
* and whether the scan that produced that answer ran to completion on every daemon searched.
|
||||||
|
*
|
||||||
|
* <p>{@link #complete} is {@code false} exactly when some {@code pane.process_info} call
|
||||||
|
* failed and, despite that, no pane was ever found to own the pid. In that case a {@code null}
|
||||||
|
* {@link #terminal} means "could not tell", not "definitely not a worker" — fleetd #505: a
|
||||||
|
* transient herdr error on the very pane that <em>does</em> own the caller's pid must not read
|
||||||
|
* as a clean negative and fall through to {@code Principal.primary}, the same way #317's
|
||||||
|
* {@code Caller.resolved()} already guards a failed lsof lookup. Callers ({@code
|
||||||
|
* ConnectionIdentity}, {@code CallerResolver}) must refuse rather than promote on an incomplete
|
||||||
|
* scan.
|
||||||
|
*
|
||||||
|
* <p>When a pane genuinely owns the pid, {@link #complete} is {@code true} regardless of
|
||||||
|
* whether some other, unrelated pane failed to answer earlier in the same scan — a positive
|
||||||
|
* match is definitive and does not need every pane to have been checked (a pane that "vanished
|
||||||
|
* mid-scan" but was never the match is still a clean, complete result).
|
||||||
*/
|
*/
|
||||||
public String terminalForPid(long pid) {
|
public record Lookup(String terminal, boolean complete) {
|
||||||
|
private static final Lookup NOT_FOUND = new Lookup(null, true);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Resolve {@code pid} to the agent pane whose process tree contains it, across every searched
|
||||||
|
* herdr daemon. See {@link Lookup} for how to read a {@code null} terminal.
|
||||||
|
*/
|
||||||
|
public Lookup terminalForPid(long pid) {
|
||||||
if (pid <= 0) {
|
if (pid <= 0) {
|
||||||
return null;
|
return Lookup.NOT_FOUND;
|
||||||
}
|
}
|
||||||
Set<Long> ancestry = ancestorsOf(pid);
|
Set<Long> ancestry = ancestorsOf(pid);
|
||||||
for (HerdrClient herdr : herdrs) {
|
boolean complete = true;
|
||||||
String terminal = terminalForPid(herdr, ancestry);
|
for (int i = 0; i < herdrs.size(); i++) {
|
||||||
if (terminal != null) {
|
Lookup outcome = scan(herdrs.get(i), i, herdrs.size(), ancestry);
|
||||||
return terminal;
|
if (outcome.terminal() != null) {
|
||||||
|
return outcome; // a definite match — no need to finish checking other clients
|
||||||
}
|
}
|
||||||
|
complete = complete && outcome.complete();
|
||||||
}
|
}
|
||||||
return null;
|
return new Lookup(null, complete);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -117,31 +145,51 @@ public final class PaneLocator {
|
|||||||
return ancestry;
|
return ancestry;
|
||||||
}
|
}
|
||||||
|
|
||||||
private static String terminalForPid(HerdrClient herdr, Set<Long> ancestry) {
|
/** Whether a pane owns one of the scanned pid's ancestors, or the check of it failed outright. */
|
||||||
|
private enum Ownership { OWNS, DOES_NOT_OWN, UNKNOWN }
|
||||||
|
|
||||||
|
private static Lookup scan(HerdrClient herdr, int clientIndex, int clientCount, Set<Long> ancestry) {
|
||||||
|
boolean complete = true;
|
||||||
for (JsonNode pane : herdr.call("pane.list", Map.of()).path("panes")) {
|
for (JsonNode pane : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||||
String paneId = pane.path("pane_id").asText(null);
|
String paneId = pane.path("pane_id").asText(null);
|
||||||
if (paneId != null && paneOwnsAnyOf(herdr, paneId, ancestry)) {
|
if (paneId == null) {
|
||||||
return pane.path("terminal_id").asText(null);
|
continue;
|
||||||
|
}
|
||||||
|
Ownership owns = paneOwnsAnyOf(herdr, clientIndex, clientCount, paneId, ancestry);
|
||||||
|
if (owns == Ownership.OWNS) {
|
||||||
|
return new Lookup(pane.path("terminal_id").asText(null), true);
|
||||||
|
}
|
||||||
|
if (owns == Ownership.UNKNOWN) {
|
||||||
|
complete = false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return null;
|
return new Lookup(null, complete);
|
||||||
}
|
}
|
||||||
|
|
||||||
private static boolean paneOwnsAnyOf(HerdrClient herdr, String paneId, Set<Long> ancestry) {
|
private static Ownership paneOwnsAnyOf(HerdrClient herdr, int clientIndex, int clientCount,
|
||||||
|
String paneId, Set<Long> ancestry) {
|
||||||
JsonNode info;
|
JsonNode info;
|
||||||
try {
|
try {
|
||||||
info = herdr.call("pane.process_info", Map.of("pane_id", paneId)).path("process_info");
|
info = herdr.call("pane.process_info", Map.of("pane_id", paneId)).path("process_info");
|
||||||
} catch (HerdrException e) {
|
} catch (HerdrException e) {
|
||||||
return false; // pane vanished mid-scan — just skip it
|
// fleetd #505: this used to be read as a clean "does not own it" (the pane vanished
|
||||||
|
// mid-scan, just skip it) — one boolean carrying two different facts. It is UNKNOWN
|
||||||
|
// now: if THIS pane is the one that owns the pid, the caller must not be told "no pane
|
||||||
|
// owns it", because that reads as a real primary and is promoted under loopback-trust.
|
||||||
|
log.warn("pane.process_info failed for pane {} on herdr client {} of {} during a "
|
||||||
|
+ "pid-owner scan — treating it as \"could not tell\", not a clean "
|
||||||
|
+ "negative (fleetd #505): {}",
|
||||||
|
paneId, clientIndex + 1, clientCount, e.getMessage());
|
||||||
|
return Ownership.UNKNOWN;
|
||||||
}
|
}
|
||||||
if (ancestry.contains(info.path("shell_pid").asLong(-1))) {
|
if (ancestry.contains(info.path("shell_pid").asLong(-1))) {
|
||||||
return true;
|
return Ownership.OWNS;
|
||||||
}
|
}
|
||||||
for (JsonNode p : info.path("foreground_processes")) {
|
for (JsonNode p : info.path("foreground_processes")) {
|
||||||
if (ancestry.contains(p.path("pid").asLong(-1))) {
|
if (ancestry.contains(p.path("pid").asLong(-1))) {
|
||||||
return true;
|
return Ownership.OWNS;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return false;
|
return Ownership.DOES_NOT_OWN;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,146 @@
|
|||||||
|
package dev.ltms.fleet.herdr;
|
||||||
|
|
||||||
|
import org.slf4j.Logger;
|
||||||
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
import java.nio.charset.StandardCharsets;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.function.IntFunction;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The three protections every {@code agent.start} caller needs against herdr's pane-typed launch
|
||||||
|
* surface (fleetd #220, #727): a byte-limit check on the assembled command line, a bounded retry
|
||||||
|
* on {@code agent_pane_busy} (the target pane's shell has not reached its prompt yet), and a
|
||||||
|
* bounded retry on {@code agent_name_taken} with a fresh name each attempt. One implementation —
|
||||||
|
* every caller of {@code agent.start}, lead or member, goes through this seam rather than carrying
|
||||||
|
* its own copy.
|
||||||
|
*/
|
||||||
|
public final class ResilientAgentLaunch {
|
||||||
|
|
||||||
|
private static final Logger log = LoggerFactory.getLogger(ResilientAgentLaunch.class);
|
||||||
|
|
||||||
|
private ResilientAgentLaunch() {
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The pty line buffer herdr types a launch command into: BSD/macOS {@code MAX_CANON}. Not a
|
||||||
|
* fleetd choice and not configurable — see {@link #checkFits}.
|
||||||
|
*/
|
||||||
|
public static final int PANE_COMMAND_BYTE_LIMIT = 1024;
|
||||||
|
|
||||||
|
/** Per-argument allowance for the separating space and a shell quote pair fleetd cannot see. */
|
||||||
|
private static final int QUOTING_OVERHEAD_PER_ARG = 3;
|
||||||
|
|
||||||
|
/** herdr rejects a duplicate agent {@code name}; a caller retries a bumped name this many times. */
|
||||||
|
public static final int NAME_RETRIES = 8;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Retries for {@code agent.start} against a seed pane whose shell has not reached its prompt
|
||||||
|
* yet — {@code tab.create}/{@code pane.split} return as soon as the pane exists, and herdr
|
||||||
|
* refuses to start an agent in a pane that is not "an available shell" ({@code agent_pane_busy}).
|
||||||
|
*/
|
||||||
|
public static final int SHELL_READY_RETRIES = 20;
|
||||||
|
|
||||||
|
/** Raised by {@link #checkFits} when the assembled command cannot fit the pane line. */
|
||||||
|
public static final class TooLargeException extends RuntimeException {
|
||||||
|
public TooLargeException(String message) {
|
||||||
|
super(message);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Verify the assembled launch line fits the pane herdr types it into. herdr does not exec the
|
||||||
|
* launch command — it TYPES it into the pane as one line, and a pty line buffer holds only
|
||||||
|
* {@value #PANE_COMMAND_BYTE_LIMIT} bytes. Everything past that byte is dropped with no error
|
||||||
|
* anywhere: herdr answers "agent started", the backend exits on the mangled argument it was
|
||||||
|
* handed, the pane closes, and the only symptom is a readiness timeout with no reason. That is
|
||||||
|
* how fleetd #214 broke every claude-code spawn — one 50-byte flag pushed a 978-byte command to
|
||||||
|
* 1028, and the tail that got cut was {@code --autocompact 250000}.
|
||||||
|
*
|
||||||
|
* <p>So measure it here and refuse, loudly and immediately, rather than start something that
|
||||||
|
* cannot work. The estimate is deliberately conservative: fleetd cannot see herdr's quoting, so
|
||||||
|
* every argument is charged its own bytes plus a separator and a quote pair. An over-estimate
|
||||||
|
* costs a clear error at a length that was already unsafe; an under-estimate would let the
|
||||||
|
* silent truncation back in.
|
||||||
|
*
|
||||||
|
* @param label names the launch in the refusal message (a profile name)
|
||||||
|
* @param argv the full argv, including the executable at index 0
|
||||||
|
* @throws TooLargeException naming the limit, the estimate, and the longest argument
|
||||||
|
*/
|
||||||
|
public static void checkFits(String label, List<String> argv) {
|
||||||
|
int bytes = 0;
|
||||||
|
String longest = null;
|
||||||
|
int longestBytes = 0;
|
||||||
|
for (String arg : argv) {
|
||||||
|
int argBytes = arg == null ? 0 : arg.getBytes(StandardCharsets.UTF_8).length;
|
||||||
|
bytes += argBytes + QUOTING_OVERHEAD_PER_ARG;
|
||||||
|
if (argBytes > longestBytes) {
|
||||||
|
longestBytes = argBytes;
|
||||||
|
longest = arg;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (bytes <= PANE_COMMAND_BYTE_LIMIT) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
String culprit = longest == null ? "<none>"
|
||||||
|
: longest.substring(0, Math.min(longest.length(), 60)) + (longest.length() > 60 ? "…" : "");
|
||||||
|
throw new TooLargeException(
|
||||||
|
"launch command for " + label + " is about " + bytes + " bytes, over the "
|
||||||
|
+ PANE_COMMAND_BYTE_LIMIT + "-byte limit of the pane line herdr types it into. "
|
||||||
|
+ "The pty would drop the tail silently and the backend would exit on a mangled "
|
||||||
|
+ "argument. Longest argument is " + longestBytes + " bytes: " + culprit
|
||||||
|
+ " — move it off the command line (a file flag) or shorten it.");
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Start an agent into {@code paneId}, retrying {@code agent_pane_busy} up to {@code retries}
|
||||||
|
* times with {@code sleeper} run between attempts.
|
||||||
|
*
|
||||||
|
* @throws HerdrException the last {@code agent_pane_busy} failure once {@code retries} is
|
||||||
|
* spent, or immediately for any other herdr failure
|
||||||
|
*/
|
||||||
|
public static Agent startAwaitingShellPrompt(AgentControl agents, String name, String kind,
|
||||||
|
List<String> args, String paneId,
|
||||||
|
int retries, Runnable sleeper) {
|
||||||
|
HerdrException busy = null;
|
||||||
|
for (int attempt = 0; attempt < retries; attempt++) {
|
||||||
|
try {
|
||||||
|
return agents.start(name, kind, args, paneId);
|
||||||
|
} catch (HerdrException e) {
|
||||||
|
if (!"agent_pane_busy".equals(e.code())) throw e;
|
||||||
|
log.debug("pane {} not at its shell prompt yet, retrying agent.start", paneId);
|
||||||
|
busy = e;
|
||||||
|
sleeper.run();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
throw busy;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Start an agent under a freshly generated name each attempt, retrying {@code agent_name_taken}
|
||||||
|
* up to {@code nameRetries} times — herdr refuses a duplicate {@code name} outright, so a stale
|
||||||
|
* registry entry (a crashed session, a name the registry has not yet released) must not block a
|
||||||
|
* legitimate relaunch. Each attempt also carries its own {@link #startAwaitingShellPrompt} retry.
|
||||||
|
*
|
||||||
|
* @param nameForAttempt called once per attempt (0-based) to produce that attempt's name
|
||||||
|
* @throws HerdrException the last {@code agent_name_taken} failure once {@code nameRetries} is
|
||||||
|
* spent, or immediately for any other herdr failure
|
||||||
|
*/
|
||||||
|
public static Agent startUniquelyNamed(AgentControl agents, String kind, List<String> args,
|
||||||
|
String paneId, IntFunction<String> nameForAttempt,
|
||||||
|
int nameRetries, int shellReadyRetries, Runnable sleeper) {
|
||||||
|
HerdrException last = null;
|
||||||
|
for (int attempt = 0; attempt < nameRetries; attempt++) {
|
||||||
|
String name = nameForAttempt.apply(attempt);
|
||||||
|
try {
|
||||||
|
return startAwaitingShellPrompt(agents, name, kind, args, paneId,
|
||||||
|
shellReadyRetries, sleeper);
|
||||||
|
} catch (HerdrException e) {
|
||||||
|
if (!"agent_name_taken".equals(e.code())) throw e;
|
||||||
|
log.debug("agent name '{}' taken, retrying", name);
|
||||||
|
last = e;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
throw last;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -50,7 +50,7 @@ import java.util.regex.Pattern;
|
|||||||
* so the scrape's herdr round-trip never stalls the status poller. The captured waiter is read on the
|
* so the scrape's herdr round-trip never stalls the status poller. The captured waiter is read on the
|
||||||
* poller thread (before any next-turn delivery can overwrite it) and passed into the virtual thread.
|
* poller thread (before any next-turn delivery can overwrite it) and passed into the virtual thread.
|
||||||
*/
|
*/
|
||||||
public final class CompletionResolver implements TurnListener {
|
public final class CompletionResolver implements TurnListener, TurnRegistrar {
|
||||||
|
|
||||||
private static final Logger log = LoggerFactory.getLogger(CompletionResolver.class);
|
private static final Logger log = LoggerFactory.getLogger(CompletionResolver.class);
|
||||||
|
|
||||||
@@ -235,6 +235,26 @@ public final class CompletionResolver implements TurnListener {
|
|||||||
this.worktreeBranches = Objects.requireNonNull(worktreeBranches, "worktreeBranches");
|
this.worktreeBranches = Objects.requireNonNull(worktreeBranches, "worktreeBranches");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #556: {@link TurnRegistrar}'s structural half of what {@link #captureBaseline} used to
|
||||||
|
* do alone — record the waiter, no I/O. Called directly and unconditionally by the {@link
|
||||||
|
* Injector} as part of delivering a turn, before {@link #onDelivered} ever runs, so this
|
||||||
|
* registration cannot be skipped by a {@link TurnListener} throwing (from {@code onDelivered} or
|
||||||
|
* any other callback). {@link #captureBaseline} still performs this exact check-and-put itself
|
||||||
|
* as well — harmless and idempotent when it runs right after this — so a caller that only wires
|
||||||
|
* the {@link TurnListener} path (e.g. an existing test that never mentions {@link TurnRegistrar})
|
||||||
|
* keeps working unchanged.
|
||||||
|
*/
|
||||||
|
@Override
|
||||||
|
public void register(String target, TurnToken token) {
|
||||||
|
CompletableFuture<Rendezvous.Resolution> waiter = token.waiter();
|
||||||
|
if (waiter == null) {
|
||||||
|
inFlight.remove(target); // no send is waiting on this delivery — nothing to resolve later
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
inFlight.put(target, new InFlight(waiter, null, nowNanos.getAsLong(), token.injectedText()));
|
||||||
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public void onDelivered(String target, TurnToken token) {
|
public void onDelivered(String target, TurnToken token) {
|
||||||
// Capture the exact waiter this turn belongs to (CB-116) and snapshot the pane's pre-turn
|
// Capture the exact waiter this turn belongs to (CB-116) and snapshot the pane's pre-turn
|
||||||
@@ -244,7 +264,14 @@ public final class CompletionResolver implements TurnListener {
|
|||||||
captureBaseline(target, token);
|
captureBaseline(target, token);
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Capture the in-flight turn: its waiter and pre-turn baseline (the testable core of {@link #onDelivered}). */
|
/**
|
||||||
|
* Capture the in-flight turn: its waiter and pre-turn baseline (the testable core of {@link
|
||||||
|
* #onDelivered}). fleetd #556: this still does its own full waiter check and {@code inFlight.put}
|
||||||
|
* — the same registration {@link #register} performs — so it keeps working standalone (as every
|
||||||
|
* test calling it directly already does) even though the {@link Injector} now also calls {@link
|
||||||
|
* #register} on its own, earlier and unconditionally. The two writes are idempotent with each
|
||||||
|
* other; only the second (this one) carries a real baseline, since only this one pays for a scrape.
|
||||||
|
*/
|
||||||
void captureBaseline(String target, TurnToken token) {
|
void captureBaseline(String target, TurnToken token) {
|
||||||
CompletableFuture<Rendezvous.Resolution> waiter = token.waiter();
|
CompletableFuture<Rendezvous.Resolution> waiter = token.waiter();
|
||||||
if (waiter == null) {
|
if (waiter == null) {
|
||||||
|
|||||||
@@ -2,6 +2,7 @@ package dev.ltms.fleet.inject;
|
|||||||
|
|
||||||
import dev.ltms.fleet.herdr.AgentControl;
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
import dev.ltms.fleet.herdr.AgentStatus;
|
import dev.ltms.fleet.herdr.AgentStatus;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrException;
|
||||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||||
import dev.ltms.fleet.msg.TurnToken;
|
import dev.ltms.fleet.msg.TurnToken;
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
@@ -11,10 +12,12 @@ import java.util.ArrayDeque;
|
|||||||
import java.util.ArrayList;
|
import java.util.ArrayList;
|
||||||
import java.util.Deque;
|
import java.util.Deque;
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
|
import java.util.Objects;
|
||||||
import java.util.Set;
|
import java.util.Set;
|
||||||
import java.util.concurrent.CompletableFuture;
|
import java.util.concurrent.CompletableFuture;
|
||||||
import java.util.concurrent.ConcurrentHashMap;
|
import java.util.concurrent.ConcurrentHashMap;
|
||||||
import java.util.function.Consumer;
|
import java.util.function.Consumer;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
import java.util.function.Predicate;
|
import java.util.function.Predicate;
|
||||||
import java.util.stream.Collectors;
|
import java.util.stream.Collectors;
|
||||||
|
|
||||||
@@ -44,6 +47,17 @@ import java.util.stream.Collectors;
|
|||||||
* turn — the pickup-grace path (a turn too fast to sample) unwedges the queue but does not fire
|
* turn — the pickup-grace path (a turn too fast to sample) unwedges the queue but does not fire
|
||||||
* completion, since without a sampled {@code working} there is no trustworthy "the worker just
|
* completion, since without a sampled {@code working} there is no trustworthy "the worker just
|
||||||
* finished the task" signal to act on.
|
* finished the task" signal to act on.
|
||||||
|
*
|
||||||
|
* <p><strong>Delivery honesty (fleetd #551).</strong> The one external send call
|
||||||
|
* ({@link AgentControl#send}) is irreversible, and its own response can still fail after the text
|
||||||
|
* has already reached the worker's pane — herdr replied (or may have), so the request was already
|
||||||
|
* processed, but the reply itself then failed to parse or carried an error. A queued entry is
|
||||||
|
* polled off the queue and marked {@code ATTEMPTED} <em>before</em> that call is made, not after,
|
||||||
|
* so a failure from the call itself can never be recorded as a confident {@code NOT_DELIVERED} for
|
||||||
|
* text that may already be sitting in the pane. {@code NOT_DELIVERED} stays reserved for the cases
|
||||||
|
* where nothing was ever attempted — the readiness grace expiring, {@link #drop}, or a herdr
|
||||||
|
* {@code *_not_found} error, which the rest of this codebase already treats as a confirmed absence
|
||||||
|
* rather than a merely inconclusive failure (see {@code StatusPoller}, {@code AgentControl}).
|
||||||
*/
|
*/
|
||||||
public final class Injector {
|
public final class Injector {
|
||||||
|
|
||||||
@@ -92,8 +106,21 @@ public final class Injector {
|
|||||||
private final AgentControl agents;
|
private final AgentControl agents;
|
||||||
private final HerdrRouter router;
|
private final HerdrRouter router;
|
||||||
private final TurnListener turnListener;
|
private final TurnListener turnListener;
|
||||||
|
/**
|
||||||
|
* fleetd #556: the Injector's own registration invariant, called directly and unconditionally —
|
||||||
|
* never folded into {@link #turnListener}'s fan-out. See {@link TurnRegistrar}.
|
||||||
|
*/
|
||||||
|
private final TurnRegistrar registrar;
|
||||||
private final Predicate<String> ready; // CB-113: a target is deliverable only when available
|
private final Predicate<String> ready; // CB-113: a target is deliverable only when available
|
||||||
private final Consumer<String> forget; // CB-114: clear a gone worker's readiness/presence
|
private final Consumer<String> forget; // CB-114: clear a gone worker's readiness/presence
|
||||||
|
/**
|
||||||
|
* Wall-clock source for the readiness-grace elapsed time logged in {@link #onStatus} (fleetd
|
||||||
|
* #501). Production constructors default this to {@code System::currentTimeMillis}; the
|
||||||
|
* package-private constructors below take it explicitly so a test can supply a stub whose
|
||||||
|
* advance does not track {@link #POLL_INTERVAL_MILLIS} — copying the shape {@code LeadRollover}
|
||||||
|
* already uses for the same purpose.
|
||||||
|
*/
|
||||||
|
private final LongSupplier nowMillis;
|
||||||
private final ConcurrentHashMap<String, Target> targets = new ConcurrentHashMap<>();
|
private final ConcurrentHashMap<String, Target> targets = new ConcurrentHashMap<>();
|
||||||
|
|
||||||
/** Delivery only; completion signalling is a no-op and every target is treated as available. */
|
/** Delivery only; completion signalling is a no-op and every target is treated as available. */
|
||||||
@@ -125,20 +152,75 @@ public final class Injector {
|
|||||||
*/
|
*/
|
||||||
public Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
public Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||||
Consumer<String> forget) {
|
Consumer<String> forget) {
|
||||||
|
// fleetd #556: no explicit registrar named here, so fall back to turnListener itself when it
|
||||||
|
// happens to also implement TurnRegistrar (true for CompletionResolver, the only production
|
||||||
|
// TurnListener that needs CB-106 registration) — every existing caller of this overload keeps
|
||||||
|
// working unchanged. A turnListener that does NOT implement TurnRegistrar (a fan-out object,
|
||||||
|
// or a bare test lambda) gets TurnRegistrar.NOOP here, same as before this ticket.
|
||||||
|
this(agents, turnListener, ready, forget,
|
||||||
|
turnListener instanceof TurnRegistrar r ? r : TurnRegistrar.NOOP);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #556: explicit-registrar overload. Use this whenever {@code turnListener} does not
|
||||||
|
* itself implement {@link TurnRegistrar} — e.g. a fan-out object that also notifies unrelated
|
||||||
|
* observers — so registration is wired directly to the one component that owns it, rather than
|
||||||
|
* relying on {@code turnListener} happening to implement both interfaces.
|
||||||
|
*/
|
||||||
|
public Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||||
|
Consumer<String> forget, TurnRegistrar registrar) {
|
||||||
|
this(agents, turnListener, ready, forget, registrar, System::currentTimeMillis);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Package-private constructor — for tests: an injectable wall-clock supplier (fleetd #501),
|
||||||
|
* with no explicit registrar (fleetd #556 fallback, same as the public 4-arg overload above).
|
||||||
|
*/
|
||||||
|
Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||||
|
Consumer<String> forget, LongSupplier nowMillis) {
|
||||||
|
this(agents, turnListener, ready, forget,
|
||||||
|
turnListener instanceof TurnRegistrar r ? r : TurnRegistrar.NOOP, nowMillis);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Full constructor — for tests: an explicit registrar and an injectable wall-clock supplier. */
|
||||||
|
Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||||
|
Consumer<String> forget, TurnRegistrar registrar, LongSupplier nowMillis) {
|
||||||
this.agents = agents;
|
this.agents = agents;
|
||||||
this.router = null;
|
this.router = null;
|
||||||
this.turnListener = turnListener;
|
this.turnListener = turnListener;
|
||||||
|
this.registrar = Objects.requireNonNull(registrar, "registrar");
|
||||||
this.ready = ready;
|
this.ready = ready;
|
||||||
this.forget = forget;
|
this.forget = forget;
|
||||||
|
this.nowMillis = nowMillis;
|
||||||
}
|
}
|
||||||
|
|
||||||
public Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
public Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
||||||
Consumer<String> forget) {
|
Consumer<String> forget) {
|
||||||
|
// fleetd #556: see the AgentControl overload above for the fallback rationale.
|
||||||
|
this(router, turnListener, ready, forget,
|
||||||
|
turnListener instanceof TurnRegistrar r ? r : TurnRegistrar.NOOP);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #556: explicit-registrar overload — production wiring ({@code Fleetd}) uses this, since
|
||||||
|
* its {@code turnListener} is a fan-out object that also notifies {@code SessionManager} and does
|
||||||
|
* not itself implement {@link TurnRegistrar}.
|
||||||
|
*/
|
||||||
|
public Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
||||||
|
Consumer<String> forget, TurnRegistrar registrar) {
|
||||||
|
this(router, turnListener, ready, forget, registrar, System::currentTimeMillis);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Full constructor — for tests: an injectable wall-clock supplier (fleetd #501). */
|
||||||
|
Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
||||||
|
Consumer<String> forget, TurnRegistrar registrar, LongSupplier nowMillis) {
|
||||||
this.agents = null;
|
this.agents = null;
|
||||||
this.router = router;
|
this.router = router;
|
||||||
this.turnListener = turnListener;
|
this.turnListener = turnListener;
|
||||||
|
this.registrar = Objects.requireNonNull(registrar, "registrar");
|
||||||
this.ready = ready;
|
this.ready = ready;
|
||||||
this.forget = forget;
|
this.forget = forget;
|
||||||
|
this.nowMillis = nowMillis;
|
||||||
}
|
}
|
||||||
|
|
||||||
private AgentControl agentsFor(String target) {
|
private AgentControl agentsFor(String target) {
|
||||||
@@ -147,8 +229,23 @@ public final class Injector {
|
|||||||
|
|
||||||
/** The result of trying to remove an undelivered message from the injector. */
|
/** The result of trying to remove an undelivered message from the injector. */
|
||||||
public enum Cancellation {
|
public enum Cancellation {
|
||||||
|
/** The message was still queued and this call removed it; the target saw nothing. */
|
||||||
CANCELLED,
|
CANCELLED,
|
||||||
|
/** The message reached the worker's pane and is confirmed delivered. */
|
||||||
DELIVERED,
|
DELIVERED,
|
||||||
|
/**
|
||||||
|
* fleetd #551: the injector called {@link AgentControl#send} for this message and that call
|
||||||
|
* threw before its outcome was known — the text may or may not have reached the pane. A
|
||||||
|
* caller reporting this to an operator must say "uncertain", not "definitely not
|
||||||
|
* delivered": reading it as a confident negative invites a resend of text that may already
|
||||||
|
* be sitting in the pane (a double delivery), which is worse than the ambiguity itself.
|
||||||
|
*/
|
||||||
|
ATTEMPTED,
|
||||||
|
/**
|
||||||
|
* The message never reached the worker's pane — nothing was ever attempted for it (the
|
||||||
|
* readiness grace expired, the target was dropped, or send failed with a herdr
|
||||||
|
* {@code *_not_found} error, which is a confirmed absence, not merely inconclusive).
|
||||||
|
*/
|
||||||
NOT_DELIVERED
|
NOT_DELIVERED
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -171,7 +268,29 @@ public final class Injector {
|
|||||||
|
|
||||||
/** A pending message and the future that completes when it has been delivered. */
|
/** A pending message and the future that completes when it has been delivered. */
|
||||||
private static final class Pending {
|
private static final class Pending {
|
||||||
enum State { QUEUED, DELIVERED, NOT_DELIVERED, CANCELLED }
|
enum State {
|
||||||
|
/** Still in the target's queue, not yet attempted. */
|
||||||
|
QUEUED,
|
||||||
|
/**
|
||||||
|
* fleetd #551: {@link AgentControl#send} has been called for this entry and its outcome
|
||||||
|
* is not yet known — recorded BEFORE the call (see {@link #onStatus}), so a Throwable
|
||||||
|
* from send() can never leave the entry either still QUEUED or falsely marked
|
||||||
|
* NOT_DELIVERED. Upgraded to DELIVERED on success; left as ATTEMPTED on an ordinary
|
||||||
|
* failure, since reaching the catch does not prove the text never reached the pane.
|
||||||
|
*/
|
||||||
|
ATTEMPTED,
|
||||||
|
/** {@code send()} returned normally: the text is confirmed to have reached the pane. */
|
||||||
|
DELIVERED,
|
||||||
|
/**
|
||||||
|
* Confirmed — not merely inconclusive — that nothing was ever sent for this entry: the
|
||||||
|
* worker never became ready ({@link #onStatus}'s readiness-grace expiry), the target was
|
||||||
|
* dropped ({@link #drop}), or send failed with a herdr {@code *_not_found} error. Never
|
||||||
|
* written for an entry whose send() outcome is unknown; see ATTEMPTED.
|
||||||
|
*/
|
||||||
|
NOT_DELIVERED,
|
||||||
|
/** Removed from the queue by {@link #cancel} before it was ever attempted. */
|
||||||
|
CANCELLED
|
||||||
|
}
|
||||||
|
|
||||||
final String target;
|
final String target;
|
||||||
final String text;
|
final String text;
|
||||||
@@ -209,6 +328,7 @@ public final class Injector {
|
|||||||
int unknownSinceTurn; // consecutive `unknown` samples while a delegation is outstanding (CB-109)
|
int unknownSinceTurn; // consecutive `unknown` samples while a delegation is outstanding (CB-109)
|
||||||
int unknownSincePostTurn; // the same, for the post-turn housekeeping phase (fleetd #306)
|
int unknownSincePostTurn; // the same, for the post-turn housekeeping phase (fleetd #306)
|
||||||
int notReadySincePoll; // consecutive injectable samples a queued message waited on the readiness gate (CB-114)
|
int notReadySincePoll; // consecutive injectable samples a queued message waited on the readiness gate (CB-114)
|
||||||
|
long notReadySinceMillis; // wall-clock time of the FIRST non-ready sample in the current notReadySincePoll streak (fleetd #501); reset alongside it
|
||||||
boolean postTurnPending; // completion observed; adapter housekeeping has not started yet
|
boolean postTurnPending; // completion observed; adapter housekeeping has not started yet
|
||||||
boolean awaitingPostTurnPickup;
|
boolean awaitingPostTurnPickup;
|
||||||
boolean postTurnObserved;
|
boolean postTurnObserved;
|
||||||
@@ -262,7 +382,14 @@ public final class Injector {
|
|||||||
}
|
}
|
||||||
|
|
||||||
private static Cancellation cancellationOf(Pending p) {
|
private static Cancellation cancellationOf(Pending p) {
|
||||||
return p.state == Pending.State.DELIVERED ? Cancellation.DELIVERED : Cancellation.NOT_DELIVERED;
|
// fleetd #551 (comment 17037): a two-way split on a now-three-way question folded ATTEMPTED
|
||||||
|
// into NOT_DELIVERED with no compiler error and no failing test — the exact defect this
|
||||||
|
// ticket exists to fix, one layer up. ATTEMPTED gets its own answer instead.
|
||||||
|
return switch (p.state) {
|
||||||
|
case DELIVERED -> Cancellation.DELIVERED;
|
||||||
|
case ATTEMPTED -> Cancellation.ATTEMPTED;
|
||||||
|
case QUEUED, NOT_DELIVERED, CANCELLED -> Cancellation.NOT_DELIVERED;
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
private static boolean isQuiescent(Target t) {
|
private static boolean isQuiescent(Target t) {
|
||||||
@@ -282,7 +409,7 @@ public final class Injector {
|
|||||||
if (t == null) return;
|
if (t == null) return;
|
||||||
|
|
||||||
Pending sent = null;
|
Pending sent = null;
|
||||||
RuntimeException sendError = null;
|
Throwable sendError = null;
|
||||||
boolean turnCompleted = false;
|
boolean turnCompleted = false;
|
||||||
boolean turnFailed = false;
|
boolean turnFailed = false;
|
||||||
boolean resubmit = false;
|
boolean resubmit = false;
|
||||||
@@ -302,6 +429,7 @@ public final class Injector {
|
|||||||
t.unknownSinceTurn = 0;
|
t.unknownSinceTurn = 0;
|
||||||
t.unknownSincePostTurn = 0;
|
t.unknownSincePostTurn = 0;
|
||||||
t.notReadySincePoll = 0;
|
t.notReadySincePoll = 0;
|
||||||
|
t.notReadySinceMillis = 0;
|
||||||
if (t.awaitingCompletion) t.turnObserved = true;
|
if (t.awaitingCompletion) t.turnObserved = true;
|
||||||
} else if (status.injectable()) { // IDLE or BLOCKED
|
} else if (status.injectable()) { // IDLE or BLOCKED
|
||||||
t.unknownSinceTurn = 0;
|
t.unknownSinceTurn = 0;
|
||||||
@@ -353,40 +481,93 @@ public final class Injector {
|
|||||||
Pending p = t.queue.peek();
|
Pending p = t.queue.peek();
|
||||||
if (p != null && ready.test(target)) {
|
if (p != null && ready.test(target)) {
|
||||||
t.notReadySincePoll = 0;
|
t.notReadySincePoll = 0;
|
||||||
|
t.notReadySinceMillis = 0;
|
||||||
|
// fleetd #551: poll and record BEFORE the irreversible send, not after.
|
||||||
|
// The entry comes off the queue and its state is set to ATTEMPTED here,
|
||||||
|
// unconditionally — so a Throwable escaping the send call below (caught or
|
||||||
|
// not) can never leave the entry QUEUED at the head of t.queue (the fleetd
|
||||||
|
// #546 hazard, since peek() alone would let the next onStatus round re-enter
|
||||||
|
// this block and send the same text again), and no path can write a
|
||||||
|
// confident DELIVERED or NOT_DELIVERED before we actually know which one
|
||||||
|
// happened.
|
||||||
|
t.queue.poll();
|
||||||
|
p.state = Pending.State.ATTEMPTED;
|
||||||
try {
|
try {
|
||||||
agentsFor(target).send(target, p.text());
|
agentsFor(target).send(target, p.text());
|
||||||
t.queue.poll();
|
|
||||||
p.state = Pending.State.DELIVERED;
|
p.state = Pending.State.DELIVERED;
|
||||||
t.awaitingPickup = true;
|
t.awaitingPickup = true;
|
||||||
t.awaitingCompletion = true;
|
t.awaitingCompletion = true;
|
||||||
t.turnObserved = false;
|
t.turnObserved = false;
|
||||||
t.injectableSincePickup = 0;
|
t.injectableSincePickup = 0;
|
||||||
sent = p;
|
sent = p;
|
||||||
} catch (RuntimeException e) {
|
} catch (Throwable e) {
|
||||||
// Delivery failed at herdr; drop the poisoned message and surface it
|
// fleetd #551: leave p.state == ATTEMPTED (recorded above, before the
|
||||||
// rather than blocking the queue behind it.
|
// call) rather than downgrading it to NOT_DELIVERED here — reaching this
|
||||||
t.queue.poll();
|
// catch does not prove the text never reached the pane. Three of the
|
||||||
p.state = Pending.State.NOT_DELIVERED;
|
// four HerdrException throw sites in HerdrCodec fire only after herdr
|
||||||
|
// has already replied (so it processed the request), and the fourth (a
|
||||||
|
// transport IOException) leaves it genuinely unknown whether herdr even
|
||||||
|
// received the bytes — see #551 comment 16867. The one exception is a
|
||||||
|
// herdr `*_not_found` error: that family is already read as "definitely
|
||||||
|
// absent, not merely inconclusive" everywhere else in this codebase
|
||||||
|
// (StatusPoller, AgentControl's own retry, WorkspaceControl,
|
||||||
|
// HerdrPeerLauncher, FleetApp, ReplyPushLoop) because it means the
|
||||||
|
// target pane/agent does not exist at all, so nothing could have been
|
||||||
|
// pasted anywhere — #551 keeps the new state consistent with that
|
||||||
|
// existing vocabulary rather than inventing a second one.
|
||||||
|
if (e instanceof HerdrException he && he.code() != null
|
||||||
|
&& he.code().endsWith("_not_found")) {
|
||||||
|
p.state = Pending.State.NOT_DELIVERED;
|
||||||
|
}
|
||||||
sent = p;
|
sent = p;
|
||||||
sendError = e;
|
sendError = e;
|
||||||
}
|
}
|
||||||
} else if (p != null && ++t.notReadySincePoll >= READINESS_GRACE_POLLS) {
|
} else if (p != null) {
|
||||||
// The worker has been idle-but-not-ready for the whole grace: its Claude
|
// fleetd #501: stamp the wall-clock time of the FIRST non-ready sample in
|
||||||
// never connected the bridge MCP (crashed during boot, or wedged on a
|
// this streak, so the expiry log below can print how long the target
|
||||||
// startup prompt). The readiness gate would hold this message forever, so
|
// actually sat non-ready — not just how many polls that took.
|
||||||
// fail every queued message and release the target (CB-114) instead of
|
if (t.notReadySincePoll == 0) {
|
||||||
// polling it indefinitely with the caller's future never completing.
|
t.notReadySinceMillis = nowMillis.getAsLong();
|
||||||
notReady = new ArrayList<>(t.queue);
|
}
|
||||||
for (Pending pending : notReady) {
|
if (++t.notReadySincePoll >= READINESS_GRACE_POLLS) {
|
||||||
pending.state = Pending.State.NOT_DELIVERED;
|
// The worker has been idle-but-not-ready for the whole grace: its Claude
|
||||||
|
// never connected the bridge MCP (crashed during boot, or wedged on a
|
||||||
|
// startup prompt). The readiness gate would hold this message forever, so
|
||||||
|
// fail every queued message and release the target (CB-114) instead of
|
||||||
|
// polling it indefinitely with the caller's future never completing.
|
||||||
|
notReady = new ArrayList<>(t.queue);
|
||||||
|
for (Pending pending : notReady) {
|
||||||
|
pending.state = Pending.State.NOT_DELIVERED;
|
||||||
|
}
|
||||||
|
// fleetd #501: t.notReadySincePoll — the loop's own counter, already in
|
||||||
|
// scope — is printed here instead of the READINESS_GRACE_POLLS constant.
|
||||||
|
// On this branch the counter has JUST reached the threshold, so the two
|
||||||
|
// agree by construction and no test can tell them apart. Printed anyway:
|
||||||
|
// it gives this line one source of truth instead of two, so a later
|
||||||
|
// change to the loop above cannot leave this message reporting a number
|
||||||
|
// the loop no longer produces.
|
||||||
|
//
|
||||||
|
// elapsedMillis is a different case: it is NOT equal-by-construction to
|
||||||
|
// the truth. notReadySincePoll only increments on a sample that reaches
|
||||||
|
// this branch (p != null, not ready) — a poll that misses that condition
|
||||||
|
// advances real time without advancing the counter — and this loop's real
|
||||||
|
// period is not guaranteed to equal POLL_INTERVAL_MILLIS (load, or a host
|
||||||
|
// sleep, can widen the real gap far past it). READINESS_GRACE_POLLS *
|
||||||
|
// POLL_INTERVAL_MILLIS / 1000 is arithmetic on two constants, not a
|
||||||
|
// measurement, so it stays here only as the labelled CONFIGURED budget,
|
||||||
|
// never presented as elapsed time.
|
||||||
|
long elapsedMillis = nowMillis.getAsLong() - t.notReadySinceMillis;
|
||||||
|
log.warn("readiness grace for {} expired after {} polls (configured={} "
|
||||||
|
+ "polls/{}s elapsed={}ms): target never became "
|
||||||
|
+ "deliverable, so failing {} queued message(s) that "
|
||||||
|
+ "never reached its pane",
|
||||||
|
target, t.notReadySincePoll, READINESS_GRACE_POLLS,
|
||||||
|
READINESS_GRACE_POLLS * POLL_INTERVAL_MILLIS / 1000, elapsedMillis,
|
||||||
|
notReady.size());
|
||||||
|
t.queue.clear();
|
||||||
|
t.notReadySincePoll = 0;
|
||||||
|
t.notReadySinceMillis = 0;
|
||||||
}
|
}
|
||||||
log.warn("readiness grace for {} expired after {} polls ({}s): target never "
|
|
||||||
+ "became deliverable, so failing {} queued message(s) that never "
|
|
||||||
+ "reached its pane",
|
|
||||||
target, READINESS_GRACE_POLLS,
|
|
||||||
READINESS_GRACE_POLLS * POLL_INTERVAL_MILLIS / 1000, notReady.size());
|
|
||||||
t.queue.clear();
|
|
||||||
t.notReadySincePoll = 0;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -434,54 +615,182 @@ public final class Injector {
|
|||||||
|
|
||||||
// Fire listeners / herdr calls after releasing the monitor so nothing runs on the poller
|
// Fire listeners / herdr calls after releasing the monitor so nothing runs on the poller
|
||||||
// thread while it holds the target lock.
|
// thread while it holds the target lock.
|
||||||
if (resubmit) {
|
//
|
||||||
try {
|
// fleetd #553: wrapped in try/finally. By this point, if `sent != null`, the delivery has
|
||||||
agentsFor(target).submit(target); // nudge a raced Enter so the pending paste submits
|
// already happened inside the monitor above — off the queue, p.state == DELIVERED, text
|
||||||
} catch (RuntimeException e) {
|
// typed into the target's pane — so `sent`'s future MUST be completed one way or another,
|
||||||
log.debug("resubmit to {} failed (will retry next poll): {}", target, e.getMessage());
|
// in every path out of this region, or the caller (a blocking fleet_send, or an async
|
||||||
|
// ticket) waits forever on a message it actually received. But the finally must not
|
||||||
|
// swallow whatever escaped: a listener that throws is a defect in THAT listener, and it
|
||||||
|
// must still reach StatusPoller's catch (Throwable) so log.error fires — converting a
|
||||||
|
// loud listener bug into a silently orphaned future would be worse than the bug itself.
|
||||||
|
//
|
||||||
|
// There are TWO futures at stake here, not one: `sent.delivered()` (the delivery future)
|
||||||
|
// and `sent.token().waiter()` (the rendezvous waiter a blocking fleet_send actually waits
|
||||||
|
// on for the worker's ANSWER). fleetd #556: the waiter is registered by `registrar.register`
|
||||||
|
// now — called directly and unconditionally, structural rather than delegated to a listener
|
||||||
|
// callback — never by turnListener.onDelivered() alone (before this ticket, that WAS the
|
||||||
|
// only path, and any TurnListener wired ahead of the one that mattered could throw and skip
|
||||||
|
// it; #553 could only make the reachable listener behave, not remove that dependency).
|
||||||
|
// Completing `sent.delivered()` without also calling `registrar.register` leaves the waiter
|
||||||
|
// unregistered forever — a hang with a success receipt, which is worse than the plain hang
|
||||||
|
// #553 was about. So the `finally` below does not just complete the future: on the path
|
||||||
|
// where the `if (sent != null)` block never ran, it does that block's whole job —
|
||||||
|
// register, then onDelivered (notification only, may throw), then complete.
|
||||||
|
//
|
||||||
|
// `sentHandled` is NOT allowed to move earlier than this: an earlier comment on #553
|
||||||
|
// proposed running `if (sent != null)` first, before turnCompleted/turnFailed, and that
|
||||||
|
// was withdrawn — `registrar.register` WRITES CompletionResolver's inFlight record for this
|
||||||
|
// turn, onTurnComplete READS it for the PREVIOUS turn, and running register first makes
|
||||||
|
// onTurnComplete resolve the NEW turn's waiter with the PREVIOUS turn's stale output
|
||||||
|
// (CB-116). The `if (sent != null)` block stays last; the `finally` is a backstop for it,
|
||||||
|
// not a replacement. fleetd #556 keeps this ordering unchanged — it only splits what used
|
||||||
|
// to be one listener call (register + notify) into two calls in the same position.
|
||||||
|
boolean sentHandled = false;
|
||||||
|
try {
|
||||||
|
if (resubmit) {
|
||||||
|
try {
|
||||||
|
agentsFor(target).submit(target); // nudge a raced Enter so the pending paste submits
|
||||||
|
} catch (Throwable e) {
|
||||||
|
// fleetd #553: widened from RuntimeException (same reasoning as #549 at :382) —
|
||||||
|
// this is the FIRST block after the monitor, so an Error escaping it used to skip
|
||||||
|
// every block below it, including the `sent` completion.
|
||||||
|
log.debug("resubmit to {} failed (will retry next poll): {}", target, e.getMessage());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
if (notReady != null) {
|
||||||
if (notReady != null) {
|
// Worker never became available: unblock every queued caller FIRST — these messages'
|
||||||
// Worker never became available: forget its (never-set) readiness, unblock every queued
|
// fate (NOT_DELIVERED, off the queue) was already decided inside the monitor above, so
|
||||||
// caller, and route the awaiting send through the same failure path as a stalled turn so
|
// fleetd #553 completes every one of them before calling forget.accept or
|
||||||
// a blocking or async waiter resolves WORKER_FAILED rather than riding out the timeout.
|
// turnListener.onTurnFailed below. Either of those is a listener/consumer callback and
|
||||||
forget.accept(target);
|
// can throw (an ordinary RuntimeException is enough — the same reasoning as the rest of
|
||||||
RuntimeException cause = new IllegalStateException(
|
// this ticket): completing the futures first means such a throw can no longer leave any
|
||||||
target + " never became available (no bridge MCP connection within the boot window)");
|
// of them permanently pending, regardless of which one throws or in which order.
|
||||||
for (Pending p : notReady) {
|
RuntimeException cause = new IllegalStateException(
|
||||||
p.delivered().completeExceptionally(cause);
|
target + " never became available (no bridge MCP connection within the boot window)");
|
||||||
|
for (Pending p : notReady) {
|
||||||
|
p.delivered().completeExceptionally(cause);
|
||||||
|
}
|
||||||
|
// route the awaiting send through the same failure path as a stalled turn so a blocking
|
||||||
|
// or async waiter resolves WORKER_FAILED rather than riding out the timeout.
|
||||||
|
forget.accept(target);
|
||||||
|
turnListener.onTurnFailed(target);
|
||||||
}
|
}
|
||||||
turnListener.onTurnFailed(target);
|
if (turnCompleted) {
|
||||||
}
|
if (startPostTurn) {
|
||||||
if (turnCompleted) {
|
// fleetd #553: the listener call is wrapped so `t.postTurnPending` (set true inside
|
||||||
if (startPostTurn) {
|
// the monitor above, before this call) is always reset. Before this wrapping, a
|
||||||
boolean started = turnListener.onTurnCompleteWithPostAction(target);
|
// RuntimeException from onTurnCompleteWithPostAction skipped the reset below,
|
||||||
synchronized (t) {
|
// permanently wedging the target: postTurnPending stayed true forever, so this
|
||||||
t.postTurnPending = false;
|
// target's delivery guard (:376) would never again pass and no further message to it
|
||||||
if (started) {
|
// would ever be delivered — a target-level lockup, not just one skipped future.
|
||||||
t.awaitingPostTurnPickup = true;
|
// `started` defaults to false so an exception is treated as "the post-turn action did
|
||||||
t.injectableSincePostTurnPickup = 0;
|
// not start" rather than falsely arming the post-turn pickup latch for an action that
|
||||||
|
// never ran.
|
||||||
|
boolean started = false;
|
||||||
|
try {
|
||||||
|
started = turnListener.onTurnCompleteWithPostAction(target);
|
||||||
|
} finally {
|
||||||
|
synchronized (t) {
|
||||||
|
t.postTurnPending = false;
|
||||||
|
if (started) {
|
||||||
|
t.awaitingPostTurnPickup = true;
|
||||||
|
t.injectableSincePostTurnPickup = 0;
|
||||||
|
}
|
||||||
|
if (t.queue.isEmpty() && !t.awaitingPostTurnPickup) {
|
||||||
|
targets.remove(target, t);
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if (t.queue.isEmpty() && !t.awaitingPostTurnPickup) {
|
} else {
|
||||||
targets.remove(target, t);
|
turnListener.onTurnComplete(target);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (turnFailed) {
|
||||||
|
turnListener.onTurnFailed(target);
|
||||||
|
}
|
||||||
|
if (sent != null) {
|
||||||
|
// fleetd #553: record the INTENT to handle before any side effect, not the fact of
|
||||||
|
// having completed it. If register()/onDelivered() throws part way through — e.g.
|
||||||
|
// after CompletionResolver's captureBaseline has already done its inFlight.put — a
|
||||||
|
// flag set only after this block would still read false, and the `finally` below
|
||||||
|
// would re-run this block's job a SECOND time. A second captureBaseline runs later,
|
||||||
|
// once onTurnComplete has already thrown and possibly scraped the pane, and can
|
||||||
|
// snapshot a pane that already absorbed this turn's output — which means the pane's
|
||||||
|
// tail never differs from that baseline again and CompletionResolver.resolve's own
|
||||||
|
// suppression at :364-369 (`return` keeping the in-flight record) drops every future
|
||||||
|
// completion for this turn, permanently. So this flag must go true FIRST.
|
||||||
|
sentHandled = true;
|
||||||
|
if (sendError != null) {
|
||||||
|
log.warn("inject to {} failed, dropped message: {}", target, sendError.getMessage());
|
||||||
|
sent.delivered().completeExceptionally(sendError);
|
||||||
|
} else {
|
||||||
|
// fleetd #556: register the waiter FIRST and unconditionally — this is the
|
||||||
|
// Injector's own invariant, kept structural rather than delegated. Even if
|
||||||
|
// turnListener.onDelivered below throws (a bug in an unrelated observer, e.g.
|
||||||
|
// SessionManager's bookkeeping, or a fan-out wired in a different order), the
|
||||||
|
// waiter is already registered and resolvable — a throwing TurnListener can no
|
||||||
|
// longer take this invariant down with it.
|
||||||
|
registrar.register(target, sent.token());
|
||||||
|
// Notification only, from here down: baseline the pane's pre-turn content so a
|
||||||
|
// misattributed completion (no new output) can't resolve this send with the
|
||||||
|
// previous turn's stale answer (CB-115). Allowed to throw and to fail (the herdr
|
||||||
|
// scrape inside already fails open with baseline == null on its own).
|
||||||
|
turnListener.onDelivered(target, sent.token());
|
||||||
|
sent.delivered().complete(null);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} finally {
|
||||||
|
// Backstop (fleetd #553, extended #556): if an earlier block in this try threw before the
|
||||||
|
// `if (sent != null)` block above ran, `sentHandled` is still false here, and this does
|
||||||
|
// that block's WHOLE job — not just the future completion. Skipping registrar.register()
|
||||||
|
// here would leave sent.token().waiter() never registered with CompletionResolver, so a
|
||||||
|
// blocking fleet_send would be told its message was delivered and then wait out its full
|
||||||
|
// timeout for an answer that can never resolve — worse than the plain hang, because it now
|
||||||
|
// looks like success. This block runs only when sendError == null: nothing was delivered
|
||||||
|
// on the error path, so there is nothing to register.
|
||||||
|
//
|
||||||
|
// fleetd #553 (lead review, comment 16916): `sentHandled` guards ONLY the register/notify
|
||||||
|
// re-call below, never the future completion outside this inner try. One boolean cannot
|
||||||
|
// carry both meanings — "the turn was handled" and "the future has been dealt with" —
|
||||||
|
// because they come apart exactly when onDelivered throws PART WAY THROUGH: sentHandled
|
||||||
|
// is already true (set before the call, correctly — see the `if (sent != null)` block
|
||||||
|
// above), so a guard on the outer `if` here would skip this whole recovery, including the
|
||||||
|
// completion, and leave sent.delivered() pending forever even though the message really
|
||||||
|
// was typed into the pane and taken off the queue. So `sent != null` alone gates whether
|
||||||
|
// this target has anything to finish; `!sentHandled` gates only the register/notify
|
||||||
|
// re-call inside. CompletableFuture.complete/completeExceptionally are idempotent — on the
|
||||||
|
// ordinary path (sentHandled == true, no throw) the `if (sent != null)` block above has
|
||||||
|
// already completed this future, so the calls below are a no-op returning false.
|
||||||
|
//
|
||||||
|
// This recovery is wrapped in its own try/catch(Throwable) that swallows only ITS OWN
|
||||||
|
// throwable and logs at WARN — the future is still completed either way — while the
|
||||||
|
// ORIGINAL throwable from the try block above is left alone to keep unwinding out of
|
||||||
|
// this method to StatusPoller's catch (Throwable), so a listener bug stays loud.
|
||||||
|
if (sent != null) {
|
||||||
|
try {
|
||||||
|
if (!sentHandled && sendError == null) {
|
||||||
|
// fleetd #556: register first here too, same as the ordinary path above — this
|
||||||
|
// recovery path is reached only when an EARLIER block (resubmit/turnCompleted/
|
||||||
|
// turnFailed) threw before the ordinary path ever ran, so nothing has
|
||||||
|
// registered this turn yet.
|
||||||
|
registrar.register(target, sent.token());
|
||||||
|
turnListener.onDelivered(target, sent.token());
|
||||||
|
}
|
||||||
|
} catch (Throwable recoveryError) {
|
||||||
|
log.warn("fleetd #553 backstop: onDelivered failed for {} while recovering from "
|
||||||
|
+ "an earlier listener failure; completing its delivery future "
|
||||||
|
+ "anyway: {}",
|
||||||
|
target, recoveryError.getMessage());
|
||||||
|
} finally {
|
||||||
|
// No-op (returns false) on the ordinary path, where the `if (sent != null)` block
|
||||||
|
// above already completed this future — see the comment above this block.
|
||||||
|
if (sendError != null) {
|
||||||
|
sent.delivered().completeExceptionally(sendError);
|
||||||
|
} else {
|
||||||
|
sent.delivered().complete(null);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
} else {
|
|
||||||
turnListener.onTurnComplete(target);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (turnFailed) {
|
|
||||||
turnListener.onTurnFailed(target);
|
|
||||||
}
|
|
||||||
if (sent != null) {
|
|
||||||
if (sendError != null) {
|
|
||||||
log.warn("inject to {} failed, dropped message: {}", target, sendError.getMessage());
|
|
||||||
sent.delivered().completeExceptionally(sendError);
|
|
||||||
} else {
|
|
||||||
// Baseline the pane's pre-turn content so a misattributed completion (no new output)
|
|
||||||
// can't resolve this send with the previous turn's stale answer (CB-115).
|
|
||||||
turnListener.onDelivered(target, sent.token());
|
|
||||||
sent.delivered().complete(null);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,91 @@
|
|||||||
|
package dev.ltms.fleet.inject;
|
||||||
|
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A progress watchdog for a singleton background loop (fleetd #544): {@link StatusPoller} and
|
||||||
|
* {@code dev.ltms.fleet.session.SessionReaper} each hold one. It tracks the monotonic timestamp
|
||||||
|
* of the loop's last <em>completed</em> round and classifies health from that timestamp alone —
|
||||||
|
* never from thread liveness. That choice is deliberate: measured on {@code main} at
|
||||||
|
* {@code 7611b69}, {@code UnixSocketHerdrClient.call()} reads with no deadline anywhere in
|
||||||
|
* {@code herdr/}, so a herdrd that accepts a connection and never answers parks the calling
|
||||||
|
* loop's thread forever — the thread stays alive and {@code Thread.isAlive()} keeps reporting
|
||||||
|
* {@code true} the whole time. A last-round timestamp that stops advancing catches that parked
|
||||||
|
* case exactly the same way it catches a thread that died outright: either way, nothing marks a
|
||||||
|
* new round complete.
|
||||||
|
*
|
||||||
|
* <p>Lives in {@code inject} (rather than {@code health}, which would be the more obvious home
|
||||||
|
* for a health-reporting fact) because {@code health} already depends on {@code session}
|
||||||
|
* ({@code HealthSnapshot}/{@code FleetHealthMonitor} read {@code MemberSession}), and
|
||||||
|
* {@code session} already depends on {@code inject} ({@code SessionManager} implements
|
||||||
|
* {@code TurnListener} and extends {@code MemberPresence}). Putting this class in {@code health}
|
||||||
|
* would have made {@code SessionReaper} (in {@code session}) depend back on {@code health},
|
||||||
|
* closing a cycle through {@code health -> session -> health} — {@code PackageCyclesTest} catches
|
||||||
|
* exactly this. {@code inject} has no dependency on {@code health}, so {@code session} depending
|
||||||
|
* on {@code inject} for this stays a one-way edge, same direction it already depends in.
|
||||||
|
*
|
||||||
|
* <p>Three states, not two (the fleetd #512 shape — one flag carrying two conditions that need
|
||||||
|
* opposite handling). {@code stop()} also makes the loop go quiet, exactly like a crash or a
|
||||||
|
* hang does, so a caller must say which one happened: {@link #markStoppedByCaller()} records
|
||||||
|
* "this halt was on purpose," and {@link #state()} reports {@link State#STOPPED} for it
|
||||||
|
* regardless of how stale the last round looks — an intentional stop is never an alarm.
|
||||||
|
*/
|
||||||
|
public final class LoopWatchdog {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* {@code RUNNING} — the loop is going and its last round finished recently.
|
||||||
|
* {@code STALLED} — the loop was never stopped on purpose, and its last-completed-round
|
||||||
|
* timestamp is older than the threshold. Covers both a dead loop and a parked one.
|
||||||
|
* {@code STOPPED} — {@link #markStoppedByCaller()} was called. Not an alarm.
|
||||||
|
*/
|
||||||
|
public enum State { RUNNING, STALLED, STOPPED }
|
||||||
|
|
||||||
|
private final LongSupplier nowNanos;
|
||||||
|
private final long staleAfterNanos;
|
||||||
|
private volatile long lastRoundNanos;
|
||||||
|
private volatile boolean stoppedByCaller;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @param nowNanos a monotonic elapsed-time clock, e.g. {@code System::nanoTime} live, a
|
||||||
|
* controllable stub in tests.
|
||||||
|
* @param staleAfterNanos how long a last-completed-round timestamp may age before {@link #state()}
|
||||||
|
* reports {@link State#STALLED}. Derive this from the loop's own poll
|
||||||
|
* interval with generous headroom — see the caller's own derivation comment.
|
||||||
|
*/
|
||||||
|
public LoopWatchdog(LongSupplier nowNanos, long staleAfterNanos) {
|
||||||
|
this.nowNanos = nowNanos;
|
||||||
|
this.staleAfterNanos = staleAfterNanos;
|
||||||
|
this.lastRoundNanos = nowNanos.getAsLong();
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Call once per completed round, from inside the loop. */
|
||||||
|
public void recordRoundComplete() {
|
||||||
|
lastRoundNanos = nowNanos.getAsLong();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Call from {@code start()} (and any future restart): clears a previous stop's mark and
|
||||||
|
* resets the clock, so the loop gets a fresh grace period before its first round completes
|
||||||
|
* rather than being judged against a timestamp left over from before it was (re)started.
|
||||||
|
*/
|
||||||
|
public void reset() {
|
||||||
|
stoppedByCaller = false;
|
||||||
|
lastRoundNanos = nowNanos.getAsLong();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Call from {@code stop()}: marks the halt as intentional, so {@link #state()} reports
|
||||||
|
* {@link State#STOPPED} rather than {@link State#STALLED} no matter how stale the last round is.
|
||||||
|
*/
|
||||||
|
public void markStoppedByCaller() {
|
||||||
|
stoppedByCaller = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The current fact about this loop, per the three-state contract above. */
|
||||||
|
public State state() {
|
||||||
|
if (stoppedByCaller) {
|
||||||
|
return State.STOPPED;
|
||||||
|
}
|
||||||
|
return (nowNanos.getAsLong() - lastRoundNanos >= staleAfterNanos) ? State.STALLED : State.RUNNING;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -8,6 +8,8 @@ import org.slf4j.Logger;
|
|||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import java.util.Set;
|
import java.util.Set;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Drives the {@link Injector} by sampling each active worker's {@code agent_status} and
|
* Drives the {@link Injector} by sampling each active worker's {@code agent_status} and
|
||||||
@@ -22,11 +24,25 @@ public final class StatusPoller {
|
|||||||
|
|
||||||
private static final Logger log = LoggerFactory.getLogger(StatusPoller.class);
|
private static final Logger log = LoggerFactory.getLogger(StatusPoller.class);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* How many multiples of {@code intervalMillis} the last-completed round may age before
|
||||||
|
* {@link #health()} reports {@link LoopWatchdog.State#STALLED} (fleetd #544). A round this
|
||||||
|
* loop runs is normally a small fraction of one interval — it only takes longer when a herdr
|
||||||
|
* call is genuinely wedged (herdr's {@code UnixSocketHerdrClient} has no read deadline, so a
|
||||||
|
* herdrd that accepts and never answers parks the thread forever) — so a wide multiplier
|
||||||
|
* avoids false alarms from an ordinarily slow round. At the production interval
|
||||||
|
* ({@link Injector#POLL_INTERVAL_MILLIS} = 250ms) this puts the threshold at 10s: long enough
|
||||||
|
* that a transient slow poll never trips it, short enough that a stuck poller is visible well
|
||||||
|
* before anything else downstream (delivery, `/healthz` consumers) would show a symptom.
|
||||||
|
*/
|
||||||
|
private static final long STALE_AFTER_INTERVAL_MULTIPLIER = 40;
|
||||||
|
|
||||||
private final AgentControl agents;
|
private final AgentControl agents;
|
||||||
private final HerdrRouter router;
|
private final HerdrRouter router;
|
||||||
private final Injector injector;
|
private final Injector injector;
|
||||||
private final StatusRefiner refiner;
|
private final StatusRefiner refiner;
|
||||||
private final long intervalMillis;
|
private final long intervalMillis;
|
||||||
|
private final LoopWatchdog watchdog;
|
||||||
private volatile boolean running;
|
private volatile boolean running;
|
||||||
private Thread thread;
|
private Thread thread;
|
||||||
|
|
||||||
@@ -36,14 +52,26 @@ public final class StatusPoller {
|
|||||||
|
|
||||||
public StatusPoller(AgentControl agents, Injector injector, StatusRefiner refiner,
|
public StatusPoller(AgentControl agents, Injector injector, StatusRefiner refiner,
|
||||||
long intervalMillis) {
|
long intervalMillis) {
|
||||||
|
this(agents, injector, refiner, intervalMillis, System::nanoTime);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Full constructor — for tests: an injectable monotonic clock (fleetd #544). */
|
||||||
|
StatusPoller(AgentControl agents, Injector injector, StatusRefiner refiner,
|
||||||
|
long intervalMillis, LongSupplier nowNanos) {
|
||||||
this.agents = agents;
|
this.agents = agents;
|
||||||
this.router = null;
|
this.router = null;
|
||||||
this.injector = injector;
|
this.injector = injector;
|
||||||
this.refiner = refiner;
|
this.refiner = refiner;
|
||||||
this.intervalMillis = intervalMillis;
|
this.intervalMillis = intervalMillis;
|
||||||
|
this.watchdog = new LoopWatchdog(nowNanos, staleAfterNanos(intervalMillis));
|
||||||
}
|
}
|
||||||
|
|
||||||
public StatusPoller(HerdrRouter router, Injector injector, long intervalMillis) {
|
public StatusPoller(HerdrRouter router, Injector injector, long intervalMillis) {
|
||||||
|
this(router, injector, intervalMillis, System::nanoTime);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Full constructor — for tests: an injectable monotonic clock (fleetd #544). */
|
||||||
|
StatusPoller(HerdrRouter router, Injector injector, long intervalMillis, LongSupplier nowNanos) {
|
||||||
this.agents = null;
|
this.agents = null;
|
||||||
this.router = router;
|
this.router = router;
|
||||||
this.injector = injector;
|
this.injector = injector;
|
||||||
@@ -53,45 +81,70 @@ public final class StatusPoller {
|
|||||||
// against the LEAD daemon even though this field points at the member one.
|
// against the LEAD daemon even though this field points at the member one.
|
||||||
this.refiner = new StatusRefiner(router.memberAgents());
|
this.refiner = new StatusRefiner(router.memberAgents());
|
||||||
this.intervalMillis = intervalMillis;
|
this.intervalMillis = intervalMillis;
|
||||||
|
this.watchdog = new LoopWatchdog(nowNanos, staleAfterNanos(intervalMillis));
|
||||||
|
}
|
||||||
|
|
||||||
|
private static long staleAfterNanos(long intervalMillis) {
|
||||||
|
return TimeUnit.MILLISECONDS.toNanos(Math.max(intervalMillis, 1)) * STALE_AFTER_INTERVAL_MULTIPLIER;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* This loop's progress fact (fleetd #544): {@code RUNNING}, {@code STALLED} (dead or parked —
|
||||||
|
* indistinguishable from outside, and never observed on purpose), or {@code STOPPED}
|
||||||
|
* ({@link #stop()} was called). See {@link LoopWatchdog} for why staleness — not thread
|
||||||
|
* liveness — is the signal.
|
||||||
|
*/
|
||||||
|
public LoopWatchdog.State health() {
|
||||||
|
return watchdog.state();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Start the polling loop on a virtual thread. Idempotent. */
|
/** Start the polling loop on a virtual thread. Idempotent. */
|
||||||
public synchronized void start() {
|
public synchronized void start() {
|
||||||
if (running) return;
|
if (running) return;
|
||||||
running = true;
|
running = true;
|
||||||
|
watchdog.reset(); // fleetd #544: a fresh grace period, not last run's stale mark/timestamp.
|
||||||
thread = Thread.ofVirtual().name("status-poller").start(this::loop);
|
thread = Thread.ofVirtual().name("status-poller").start(this::loop);
|
||||||
log.info("status poller started (interval {}ms)", intervalMillis);
|
log.info("status poller started (interval {}ms)", intervalMillis);
|
||||||
}
|
}
|
||||||
|
|
||||||
private void loop() {
|
private void loop() {
|
||||||
while (running) {
|
try {
|
||||||
Set<String> active = injector.activeTargets();
|
while (running) {
|
||||||
for (String target : active) {
|
Set<String> active = injector.activeTargets();
|
||||||
if (!running) return;
|
for (String target : active) {
|
||||||
try {
|
if (!running) return;
|
||||||
// herdr's agent_status can misreport a settled worker as `unknown`; refine it
|
try {
|
||||||
// against the pane content before it drives delivery/completion (CB-115).
|
// herdr's agent_status can misreport a settled worker as `unknown`; refine it
|
||||||
// CB-185: refine THROUGH the same control the raw status came from — a router
|
// against the pane content before it drives delivery/completion (CB-115).
|
||||||
// splits lead/member targets across two herdr daemons, and reading a lead's pane
|
// CB-185: refine THROUGH the same control the raw status came from — a router
|
||||||
// through the (fixed) member refiner never finds it, wedging that lead at UNKNOWN.
|
// splits lead/member targets across two herdr daemons, and reading a lead's pane
|
||||||
AgentControl control = router != null ? router.agentsFor(target) : agents;
|
// through the (fixed) member refiner never finds it, wedging that lead at UNKNOWN.
|
||||||
AgentStatus status = refiner.refine(target, control.status(target), control);
|
AgentControl control = router != null ? router.agentsFor(target) : agents;
|
||||||
injector.onStatus(target, status);
|
AgentStatus status = refiner.refine(target, control.status(target), control);
|
||||||
} catch (HerdrException e) {
|
injector.onStatus(target, status);
|
||||||
// The worker's agent is gone — stop trying and unblock its waiters.
|
} catch (HerdrException e) {
|
||||||
if (e.code() != null && e.code().endsWith("_not_found")) {
|
// The worker's agent is gone — stop trying and unblock its waiters.
|
||||||
log.debug("target {} gone; dropping its queue", target);
|
if (e.code() != null && e.code().endsWith("_not_found")) {
|
||||||
injector.drop(target, e);
|
log.debug("target {} gone; dropping its queue", target);
|
||||||
} else {
|
injector.drop(target, e);
|
||||||
log.debug("status poll for {} failed (will retry): {}", target, e.getMessage());
|
} else {
|
||||||
|
log.debug("status poll for {} failed (will retry): {}", target, e.getMessage());
|
||||||
|
}
|
||||||
|
} catch (Throwable e) {
|
||||||
|
log.error("unexpected failure polling {}; skipping this round", target, e);
|
||||||
}
|
}
|
||||||
} catch (RuntimeException e) {
|
|
||||||
// Never let one target's unexpected error (e.g. an odd agent.get shape) kill
|
|
||||||
// the single poller thread and stall injection for every worker.
|
|
||||||
log.warn("unexpected error polling {}; skipping this round", target, e);
|
|
||||||
}
|
}
|
||||||
|
// fleetd #544: a round is "complete" only once every active target has been polled —
|
||||||
|
// a target parked mid-for-loop (control.status(target) never returning) means this
|
||||||
|
// line is never reached, so the watchdog goes stale exactly like a dead loop would.
|
||||||
|
watchdog.recordRoundComplete();
|
||||||
|
sleep();
|
||||||
}
|
}
|
||||||
sleep();
|
} finally {
|
||||||
|
if (running) {
|
||||||
|
log.error("status poller loop exited unexpectedly; it can be restarted");
|
||||||
|
}
|
||||||
|
running = false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -107,6 +160,7 @@ public final class StatusPoller {
|
|||||||
/** Stop the polling loop. Idempotent. */
|
/** Stop the polling loop. Idempotent. */
|
||||||
public synchronized void stop() {
|
public synchronized void stop() {
|
||||||
running = false;
|
running = false;
|
||||||
|
watchdog.markStoppedByCaller(); // fleetd #544: this halt is on purpose — health() must say STOPPED, not STALLED.
|
||||||
if (thread != null) thread.interrupt();
|
if (thread != null) thread.interrupt();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,29 @@
|
|||||||
|
package dev.ltms.fleet.inject;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.msg.TurnToken;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #556: the {@link Injector}'s own invariant — every delivered turn has a registered
|
||||||
|
* waiter — kept structural rather than delegated to a {@link TurnListener} callback that can
|
||||||
|
* throw. The {@link Injector} calls {@link #register} directly and unconditionally as part of
|
||||||
|
* delivering a turn, before it ever calls {@code turnListener.onDelivered}. A {@link TurnListener}
|
||||||
|
* that throws from every callback (a bug in an unrelated observer, e.g. session bookkeeping, or a
|
||||||
|
* fan-out object wired in the wrong order) can therefore never leave a delivered turn unregistered
|
||||||
|
* — the shape #553 could not close, since that fix could only make the reachable listener behave,
|
||||||
|
* never remove the Injector's dependency on a listener behaving at all.
|
||||||
|
*
|
||||||
|
* <p>Deliberately narrow: registration only, no I/O, nothing that scrapes a pane or can reasonably
|
||||||
|
* fail. The CB-115 staleness baseline (a herdr round-trip, allowed to fail) stays a
|
||||||
|
* {@link TurnListener#onDelivered} concern — fired by the {@link Injector} only after this call has
|
||||||
|
* already run, so its own failure cannot un-register anything.
|
||||||
|
*/
|
||||||
|
@FunctionalInterface
|
||||||
|
public interface TurnRegistrar {
|
||||||
|
|
||||||
|
/** Record {@code token}'s waiter as the turn currently in flight for {@code target}. */
|
||||||
|
void register(String target, TurnToken token);
|
||||||
|
|
||||||
|
/** No-op registrar for callers that don't need CB-106 rendezvous tracking. */
|
||||||
|
TurnRegistrar NOOP = (_, _) -> {
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
package dev.ltms.fleet.launch;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.List;
|
||||||
|
|
||||||
|
/** Arguments shared by every fleetd path that starts Claude Code. */
|
||||||
|
public final class ClaudeCodeArguments {
|
||||||
|
|
||||||
|
private ClaudeCodeArguments() {
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Append the configured Claude Code auto-compaction window when the profile opts in.
|
||||||
|
*
|
||||||
|
* <p>This flag and the environment variable {@code CLAUDE_CODE_AUTO_COMPACT_WINDOW} can
|
||||||
|
* disagree, and fleetd #618 measured which one Claude Code actually follows: the environment
|
||||||
|
* variable wins, ahead of this {@code --autocompact} flag, ahead of the settings file, ahead of
|
||||||
|
* clientdata, the experiment, and the model default. So when a profile sets both, the flag this
|
||||||
|
* method appends has NO effect — Claude Code reads {@code CLAUDE_CODE_AUTO_COMPACT_WINDOW}
|
||||||
|
* first and never consults the flag. {@link FleetConfig#load(java.nio.file.Path)} only WARNS
|
||||||
|
* when a Claude Code profile sets both to different values (see {@code
|
||||||
|
* FleetConfig.warnConflictingAutoCompactWindows}) — it does not stop the daemon from starting,
|
||||||
|
* and the launched session honours the env var, not this flag. Measured against Claude Code
|
||||||
|
* 2.1.278 (fleetd #618) — a later version could reorder this precedence.
|
||||||
|
*/
|
||||||
|
public static List<String> withAutoCompactWindow(List<String> argv, FleetConfig.Profile profile) {
|
||||||
|
if (profile.autoCompactWindow() == null) {
|
||||||
|
return argv;
|
||||||
|
}
|
||||||
|
List<String> withAutoCompact = new ArrayList<>(argv);
|
||||||
|
withAutoCompact.add("--autocompact");
|
||||||
|
withAutoCompact.add(String.valueOf(profile.autoCompactWindow()));
|
||||||
|
return withAutoCompact;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,334 @@
|
|||||||
|
package dev.ltms.fleet.lead;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
|
||||||
|
import java.io.IOException;
|
||||||
|
import java.io.RandomAccessFile;
|
||||||
|
import java.nio.charset.StandardCharsets;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.ConcurrentHashMap;
|
||||||
|
import java.util.concurrent.atomic.AtomicInteger;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reads how full a lead's own Claude Code context window is, from the transcript Claude Code
|
||||||
|
* itself writes — never from the lead's pane (fleetd has {@code AgentControl.read} for that, and
|
||||||
|
* this must not use it: a pane holds terminal text, not the structured usage numbers a transcript
|
||||||
|
* carries, and scraping it would also race the lead's own rendering).
|
||||||
|
*
|
||||||
|
* <p><strong>Why this exists.</strong> A lead auto-compacts when its context fills — on the host
|
||||||
|
* this was built for, that happened 30 times in one session, discarding roughly 250,000 tokens
|
||||||
|
* and costing 46 seconds to 3 minutes each time, and fleetd had no way to see it coming. This
|
||||||
|
* class is the first thing that looks.
|
||||||
|
*
|
||||||
|
* <p><strong>The route.</strong> Claude Code appends one JSON object per line to
|
||||||
|
* {@code <configDir>/projects/<slug>/<sessionId>.jsonl}. {@code <slug>} is an undocumented,
|
||||||
|
* internal encoding of the working directory — this class never derives it. Instead it lists the
|
||||||
|
* one-level-deep subdirectories of {@code <configDir>/projects/} and looks for
|
||||||
|
* {@code <sessionId>.jsonl} by name, so the slug rule can change without breaking this reader.
|
||||||
|
*
|
||||||
|
* <ul>
|
||||||
|
* <li><strong>Live context</strong> is read off the last record in the read window that carries
|
||||||
|
* a {@code message.usage} object: {@code input_tokens + cache_read_input_tokens +
|
||||||
|
* cache_creation_input_tokens}. This is what actually fills the window — a plain
|
||||||
|
* {@code input_tokens} count alone understates it once the conversation has any cached
|
||||||
|
* prefix, which on a long-lived lead is always.</li>
|
||||||
|
* <li><strong>Compaction history</strong> is a count of {@code subtype: "compact_boundary"}
|
||||||
|
* records seen in the same read window — see {@link Reading#compactions()}. It is a count
|
||||||
|
* within the window this reader actually looked at, not a lifetime total: a session with
|
||||||
|
* more compactions than fit in {@link #TAIL_BYTES} of transcript will undercount. That
|
||||||
|
* trade-off is deliberate — see {@link #TAIL_BYTES}.</li>
|
||||||
|
* </ul>
|
||||||
|
*
|
||||||
|
* <p><strong>Three states, not two (OK / HIGH / UNKNOWN).</strong> Every path that cannot
|
||||||
|
* positively establish the live token count — a missing file, an unreadable one, a peer that
|
||||||
|
* is not a Claude backend, or every line in the read window failing to parse as JSON — returns
|
||||||
|
* {@link State#UNKNOWN} with no token number, never a default "0" or "ok" that would read as
|
||||||
|
* "this lead is fine" when the honest answer is "I could not look".
|
||||||
|
*
|
||||||
|
* <p><strong>A torn final line does not mean UNKNOWN.</strong> {@code fleet_list} reads this
|
||||||
|
* transcript while Claude Code may be mid-write on it, so the last line in the window can be cut
|
||||||
|
* off mid-flush — that is an ordinary, expected race, not a sign the format has changed. Earlier
|
||||||
|
* this class treated ANY unparseable last line as UNKNOWN, on the theory that "if the format
|
||||||
|
* changes, we should see UNKNOWN". That reasoning does not hold: a real format change makes
|
||||||
|
* <em>every</em> line in the window unparseable, not only the last one written. So a single
|
||||||
|
* malformed line (most often the final, torn one, but the check is not position-specific) is
|
||||||
|
* simply skipped rather than treated as fatal, and the reading is built from whatever lines in the
|
||||||
|
* window did parse. Only when <em>none</em> of them parse — the real format-change signal — does
|
||||||
|
* this return {@link State#UNKNOWN}, still with no stale number standing in for "I could not
|
||||||
|
* tell".
|
||||||
|
*
|
||||||
|
* <p><strong>Bounded cost.</strong> {@code fleet_list} is polled constantly, so every read is
|
||||||
|
* capped two ways: {@link #TAIL_BYTES} bounds how much of the transcript is ever read from disk
|
||||||
|
* (never the whole 52 MB a long-lived transcript reaches on the host this was measured on), and
|
||||||
|
* {@link #DEFAULT_CACHE_TTL_MILLIS} bounds how often that bounded read actually happens — a burst
|
||||||
|
* of {@code fleet_list} calls inside one TTL window reads the file once. One instance's cache is
|
||||||
|
* keyed by {@code (configDir, sessionId, highThreshold)}, so it is safe to share across every lead
|
||||||
|
* a single {@code fleet_list} call reports on, and a call that resolves a different effective
|
||||||
|
* window for the same lead never reads back a state computed against the other window's
|
||||||
|
* threshold.
|
||||||
|
*/
|
||||||
|
public final class LeadContextGauge {
|
||||||
|
|
||||||
|
private static final String PROJECTS_DIR = "projects";
|
||||||
|
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||||
|
|
||||||
|
/**
|
||||||
|
* How many trailing bytes of a transcript a single read ever pulls off disk. Chosen so one
|
||||||
|
* read comfortably spans many recent turns — each usage or compact_boundary record is at most
|
||||||
|
* a few KB — while staying nowhere near the 52 MB a long session's real transcript reaches on
|
||||||
|
* the host this was built for; reading that whole file on every {@code fleet_list} call is
|
||||||
|
* exactly the cost this bound exists to avoid. 2 MiB holds on the order of hundreds of recent
|
||||||
|
* lines even when a turn's tool output is unusually large, which is far more than needed to
|
||||||
|
* find the most recent usage record and any recent compaction.
|
||||||
|
*/
|
||||||
|
static final int TAIL_BYTES = 2 * 1024 * 1024;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* How long a {@link Reading} is served from cache before the file is read again.
|
||||||
|
* {@code fleet_list} is called constantly (by design — it is the fleet's own status probe), so
|
||||||
|
* without a TTL a burst of calls would re-read the transcript tail once per call. 5 seconds is
|
||||||
|
* short enough that a caller watching for a state change never waits long, and long enough that
|
||||||
|
* a poll loop calling every second or two only touches disk once per window.
|
||||||
|
*/
|
||||||
|
static final long DEFAULT_CACHE_TTL_MILLIS = 5_000;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Fallback HIGH threshold used when a caller resolves no effective auto-compact window for the
|
||||||
|
* lead being read (see {@link #read(String, String, String, Long)}) — the built-in default so
|
||||||
|
* no config key is required to get a warning at all.
|
||||||
|
*/
|
||||||
|
static final long HIGH_THRESHOLD_TOKENS = 200_000;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The fraction of a resolved effective auto-compact window that HIGH warns at, so the warning
|
||||||
|
* margin scales with the window instead of only ever meaning something against the fixed
|
||||||
|
* {@link #HIGH_THRESHOLD_TOKENS} fallback.
|
||||||
|
*/
|
||||||
|
static final double HIGH_THRESHOLD_FRACTION = 2.0 / 3.0;
|
||||||
|
|
||||||
|
/** The only peer kind this reader understands ({@code Agent.agentType()}'s wire value). */
|
||||||
|
private static final String CLAUDE_AGENT_TYPE = "claude";
|
||||||
|
|
||||||
|
public enum State { OK, HIGH, UNKNOWN }
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @param state {@link State#UNKNOWN} whenever {@code tokens} could not be established
|
||||||
|
* @param tokens live context tokens, or {@code null} exactly when {@code state} is
|
||||||
|
* {@link State#UNKNOWN}
|
||||||
|
* @param compactions {@code compact_boundary} records seen in the read window (see class
|
||||||
|
* javadoc) — {@code 0} both for "genuinely none seen" and for "unknown",
|
||||||
|
* since a caller that already sees {@code state: UNKNOWN} has no reason to
|
||||||
|
* trust this number either way
|
||||||
|
*/
|
||||||
|
public record Reading(State state, Long tokens, int compactions) {
|
||||||
|
/**
|
||||||
|
* fleetd #609: widened from package-private to public so {@code
|
||||||
|
* dev.ltms.fleet.msg.LeadHeartbeatLoop.LeadContextSource.none()} (a different package) can
|
||||||
|
* return the same inert "I could not look" reading the gauge itself uses, without inventing
|
||||||
|
* a parallel unknown-reading constant. Behaviour of this class is otherwise unchanged.
|
||||||
|
*/
|
||||||
|
public static Reading unknown() {
|
||||||
|
return new Reading(State.UNKNOWN, null, 0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private record CacheEntry(Reading reading, long readAtMillis) {
|
||||||
|
}
|
||||||
|
|
||||||
|
private final LongSupplier clock;
|
||||||
|
private final long ttlMillis;
|
||||||
|
private final Map<String, CacheEntry> cache = new ConcurrentHashMap<>();
|
||||||
|
/** Test seam only (package-private) — counts real disk reads, i.e. cache misses. */
|
||||||
|
private final AtomicInteger diskReads = new AtomicInteger();
|
||||||
|
|
||||||
|
public LeadContextGauge() {
|
||||||
|
this(System::currentTimeMillis, DEFAULT_CACHE_TTL_MILLIS);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Test seam: an injectable clock and TTL so cache expiry is provable without sleeping. */
|
||||||
|
LeadContextGauge(LongSupplier clock, long ttlMillis) {
|
||||||
|
this.clock = clock;
|
||||||
|
this.ttlMillis = ttlMillis;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** How many times this instance has actually read a transcript off disk — test seam only. */
|
||||||
|
int diskReadCount() {
|
||||||
|
return diskReads.get();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @param configDir the lead's {@code CLAUDE_CONFIG_DIR}, or {@code null}/blank to use the
|
||||||
|
* default {@code <user.home>/.claude} — the right answer for the common case
|
||||||
|
* where the lead's profile sets no {@code configDir} override
|
||||||
|
* @param sessionId the lead's own Claude session id ({@code Agent.sessionId()}), or
|
||||||
|
* {@code null} when herdr has not resolved one yet
|
||||||
|
* @param agentType the detected peer kind ({@code Agent.agentType()}); anything other than
|
||||||
|
* {@code "claude"} (including {@code null}, meaning undetected) reports
|
||||||
|
* {@link State#UNKNOWN} — this reader only understands Claude Code's own
|
||||||
|
* transcript format
|
||||||
|
* @param effectiveWindowTokens the caller's resolved effective auto-compact window for this
|
||||||
|
* lead's own profile, or {@code null} when it cannot be resolved.
|
||||||
|
* HIGH fires at {@link #HIGH_THRESHOLD_FRACTION} of this value;
|
||||||
|
* {@code null} (or a non-positive value) falls back to the fixed
|
||||||
|
* {@link #HIGH_THRESHOLD_TOKENS}
|
||||||
|
*/
|
||||||
|
public Reading read(String configDir, String sessionId, String agentType, Long effectiveWindowTokens) {
|
||||||
|
if (sessionId == null || sessionId.isBlank()) {
|
||||||
|
return Reading.unknown();
|
||||||
|
}
|
||||||
|
if (!CLAUDE_AGENT_TYPE.equalsIgnoreCase(agentType)) {
|
||||||
|
return Reading.unknown();
|
||||||
|
}
|
||||||
|
String base = (configDir == null || configDir.isBlank())
|
||||||
|
? System.getProperty("user.home") + "/.claude"
|
||||||
|
: configDir;
|
||||||
|
long highThreshold = highThreshold(effectiveWindowTokens);
|
||||||
|
String cacheKey = base + '\u0000' + sessionId + '\u0000' + highThreshold;
|
||||||
|
long now = clock.getAsLong();
|
||||||
|
CacheEntry cached = cache.get(cacheKey);
|
||||||
|
if (cached != null && now - cached.readAtMillis() < ttlMillis) {
|
||||||
|
return cached.reading();
|
||||||
|
}
|
||||||
|
Reading fresh = readUncached(base, sessionId, highThreshold);
|
||||||
|
cache.put(cacheKey, new CacheEntry(fresh, now));
|
||||||
|
return fresh;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** {@link #HIGH_THRESHOLD_FRACTION} of {@code effectiveWindowTokens}, or the fixed fallback. */
|
||||||
|
private static long highThreshold(Long effectiveWindowTokens) {
|
||||||
|
if (effectiveWindowTokens == null || effectiveWindowTokens <= 0) {
|
||||||
|
return HIGH_THRESHOLD_TOKENS;
|
||||||
|
}
|
||||||
|
return (long) (effectiveWindowTokens * HIGH_THRESHOLD_FRACTION);
|
||||||
|
}
|
||||||
|
|
||||||
|
private Reading readUncached(String base, String sessionId, long highThreshold) {
|
||||||
|
diskReads.incrementAndGet();
|
||||||
|
Path file = findTranscript(base, sessionId);
|
||||||
|
if (file == null) {
|
||||||
|
return Reading.unknown();
|
||||||
|
}
|
||||||
|
TailRead tail;
|
||||||
|
try {
|
||||||
|
tail = tailBytes(file, TAIL_BYTES);
|
||||||
|
} catch (IOException e) {
|
||||||
|
return Reading.unknown();
|
||||||
|
}
|
||||||
|
return parse(tail, highThreshold);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Finds {@code <sessionId>.jsonl} under {@code <base>/projects/}, one level deep — never by
|
||||||
|
* deriving the slug directory from a working directory (see class javadoc). Bounded to a
|
||||||
|
* single {@code list()} of {@code projects/} itself: it never recurses further, so the cost is
|
||||||
|
* the number of project directories, not the size of any transcript inside them.
|
||||||
|
*/
|
||||||
|
private Path findTranscript(String base, String sessionId) {
|
||||||
|
Path projectsDir = Path.of(base, PROJECTS_DIR);
|
||||||
|
if (!Files.isDirectory(projectsDir)) {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
String filename = sessionId + ".jsonl";
|
||||||
|
Path direct = projectsDir.resolve(filename);
|
||||||
|
if (Files.isRegularFile(direct)) {
|
||||||
|
return direct;
|
||||||
|
}
|
||||||
|
try (var children = Files.list(projectsDir)) {
|
||||||
|
return children.filter(Files::isDirectory)
|
||||||
|
.map(dir -> dir.resolve(filename))
|
||||||
|
.filter(Files::isRegularFile)
|
||||||
|
.findFirst()
|
||||||
|
.orElse(null);
|
||||||
|
} catch (IOException e) {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Package-private (not {@code private}): {@link #tailBytes} is a test seam, see its javadoc. */
|
||||||
|
record TailRead(byte[] bytes, boolean fromStart) {
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reads at most {@code maxBytes} trailing bytes of {@code file}. Package-private (not
|
||||||
|
* {@code private}) so a test can assert directly on the returned array's length — "bytes
|
||||||
|
* actually read", not on any parsed answer — without needing a file anywhere near
|
||||||
|
* {@link #TAIL_BYTES} in size to prove the cap holds.
|
||||||
|
*/
|
||||||
|
static TailRead tailBytes(Path file, int maxBytes) throws IOException {
|
||||||
|
try (RandomAccessFile raf = new RandomAccessFile(file.toFile(), "r")) {
|
||||||
|
long length = raf.length();
|
||||||
|
long start = Math.max(0, length - maxBytes);
|
||||||
|
raf.seek(start);
|
||||||
|
byte[] buf = new byte[(int) (length - start)];
|
||||||
|
raf.readFully(buf);
|
||||||
|
return new TailRead(buf, start == 0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Parses the tail into a {@link Reading}. The first line is dropped unconditionally whenever
|
||||||
|
* the tail is not the whole file (it starts mid-line, cut by {@link #TAIL_BYTES} — an expected
|
||||||
|
* artefact of the bound, not a data problem). Every remaining line is then parsed on a
|
||||||
|
* best-effort basis: a line that fails to parse (most often the last one, torn by a write this
|
||||||
|
* read raced — see the "torn final line" section of the class javadoc) is skipped, not fatal.
|
||||||
|
* Only when none of the remaining lines parse does this report {@link State#UNKNOWN}.
|
||||||
|
*/
|
||||||
|
private Reading parse(TailRead tail, long highThreshold) {
|
||||||
|
String text = new String(tail.bytes(), StandardCharsets.UTF_8);
|
||||||
|
List<String> lines = new ArrayList<>(List.of(text.split("\n", -1)));
|
||||||
|
if (!lines.isEmpty() && lines.get(lines.size() - 1).isEmpty()) {
|
||||||
|
lines.remove(lines.size() - 1); // trailing newline leaves a phantom empty last element
|
||||||
|
}
|
||||||
|
if (!tail.fromStart() && !lines.isEmpty()) {
|
||||||
|
lines.remove(0); // first line is a fragment cut by our own tail bound, not real data
|
||||||
|
}
|
||||||
|
if (lines.isEmpty()) {
|
||||||
|
return Reading.unknown();
|
||||||
|
}
|
||||||
|
Long tokens = null;
|
||||||
|
int compactions = 0;
|
||||||
|
boolean anyLineParsed = false;
|
||||||
|
for (String line : lines) {
|
||||||
|
JsonNode node = tryParse(line);
|
||||||
|
if (node == null) {
|
||||||
|
// A malformed line — typically the last one, cut mid-flush by a write this read
|
||||||
|
// raced — is skipped rather than treated as fatal. See the class javadoc's "torn
|
||||||
|
// final line" section for why: a real format change makes EVERY line unparseable,
|
||||||
|
// not only this one, and that case is still caught below by anyLineParsed.
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
anyLineParsed = true;
|
||||||
|
JsonNode usage = node.path("message").path("usage");
|
||||||
|
if (usage.isObject()) {
|
||||||
|
tokens = usage.path("input_tokens").asLong(0)
|
||||||
|
+ usage.path("cache_read_input_tokens").asLong(0)
|
||||||
|
+ usage.path("cache_creation_input_tokens").asLong(0);
|
||||||
|
}
|
||||||
|
if ("compact_boundary".equals(node.path("subtype").asText(null))) {
|
||||||
|
compactions++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!anyLineParsed) {
|
||||||
|
return Reading.unknown();
|
||||||
|
}
|
||||||
|
if (tokens == null) {
|
||||||
|
return new Reading(State.UNKNOWN, null, compactions);
|
||||||
|
}
|
||||||
|
State state = tokens >= highThreshold ? State.HIGH : State.OK;
|
||||||
|
return new Reading(state, tokens, compactions);
|
||||||
|
}
|
||||||
|
|
||||||
|
private JsonNode tryParse(String line) {
|
||||||
|
try {
|
||||||
|
return MAPPER.readTree(line);
|
||||||
|
} catch (IOException e) {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -5,12 +5,15 @@ import dev.ltms.fleet.herdr.Agent;
|
|||||||
import dev.ltms.fleet.herdr.AgentControl;
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
import dev.ltms.fleet.herdr.HerdrException;
|
import dev.ltms.fleet.herdr.HerdrException;
|
||||||
import dev.ltms.fleet.herdr.PendingCloseMarker;
|
import dev.ltms.fleet.herdr.PendingCloseMarker;
|
||||||
|
import dev.ltms.fleet.herdr.ResilientAgentLaunch;
|
||||||
import dev.ltms.fleet.herdr.Tab;
|
import dev.ltms.fleet.herdr.Tab;
|
||||||
import dev.ltms.fleet.herdr.Workspace;
|
import dev.ltms.fleet.herdr.Workspace;
|
||||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
|
import dev.ltms.fleet.launch.ClaudeCodeArguments;
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
import java.security.SecureRandom;
|
||||||
import java.util.ArrayList;
|
import java.util.ArrayList;
|
||||||
import java.util.LinkedHashMap;
|
import java.util.LinkedHashMap;
|
||||||
import java.util.LinkedHashSet;
|
import java.util.LinkedHashSet;
|
||||||
@@ -18,6 +21,7 @@ import java.util.List;
|
|||||||
import java.util.Map;
|
import java.util.Map;
|
||||||
import java.util.Objects;
|
import java.util.Objects;
|
||||||
import java.util.Set;
|
import java.util.Set;
|
||||||
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
import dev.ltms.fleet.peer.PeerLauncher;
|
import dev.ltms.fleet.peer.PeerLauncher;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -79,9 +83,19 @@ public final class LeadLauncher {
|
|||||||
|
|
||||||
private static final Logger log = LoggerFactory.getLogger(LeadLauncher.class);
|
private static final Logger log = LoggerFactory.getLogger(LeadLauncher.class);
|
||||||
|
|
||||||
|
/** Attempts {@link #relaunch(String)} makes before giving up and returning {@code null}. */
|
||||||
|
static final int RELAUNCH_ATTEMPTS = 3;
|
||||||
|
|
||||||
private final AgentControl agents;
|
private final AgentControl agents;
|
||||||
private final WorkspaceControl spaces;
|
private final WorkspaceControl spaces;
|
||||||
private final FleetConfig cfg;
|
private final FleetConfig cfg;
|
||||||
|
private final Runnable sleeper;
|
||||||
|
|
||||||
|
// Per-process token mixed into each lead agent name so a fresh daemon process (seq back at 0)
|
||||||
|
// cannot collide with a same-name lead that outlived a restart — the same scheme
|
||||||
|
// HerdrPeerLauncher uses for members (fleetd #727).
|
||||||
|
private final String nameNonce = String.format("%06x", new SecureRandom().nextInt(1 << 24));
|
||||||
|
private final AtomicLong nameSeq = new AtomicLong();
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @param agents herdr agent control (start, list)
|
* @param agents herdr agent control (start, list)
|
||||||
@@ -89,9 +103,27 @@ public final class LeadLauncher {
|
|||||||
* @param cfg the loaded config — {@code fleet.leaders}, {@code profiles} and each lead's tab
|
* @param cfg the loaded config — {@code fleet.leaders}, {@code profiles} and each lead's tab
|
||||||
*/
|
*/
|
||||||
public LeadLauncher(AgentControl agents, WorkspaceControl spaces, FleetConfig cfg) {
|
public LeadLauncher(AgentControl agents, WorkspaceControl spaces, FleetConfig cfg) {
|
||||||
|
this(agents, spaces, cfg, () -> sleepUninterruptibly(300));
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Test seam: as above, plus an injectable {@code sleeper} for the {@code agent_pane_busy}
|
||||||
|
* retry (fleetd #727), so a test can prove the retry budget without a real sleep.
|
||||||
|
*/
|
||||||
|
LeadLauncher(AgentControl agents, WorkspaceControl spaces, FleetConfig cfg, Runnable sleeper) {
|
||||||
this.agents = agents;
|
this.agents = agents;
|
||||||
this.spaces = spaces;
|
this.spaces = spaces;
|
||||||
this.cfg = cfg;
|
this.cfg = cfg;
|
||||||
|
this.sleeper = sleeper;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Uninterruptible sleep — the production {@link #sleeper} between {@code agent_pane_busy} retries. */
|
||||||
|
private static void sleepUninterruptibly(long ms) {
|
||||||
|
try {
|
||||||
|
Thread.sleep(ms);
|
||||||
|
} catch (InterruptedException e) {
|
||||||
|
Thread.currentThread().interrupt();
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -172,24 +204,14 @@ public final class LeadLauncher {
|
|||||||
log.info("lead '{}': {} live, {} wanted — nothing to start", name, running, wanted);
|
log.info("lead '{}': {} live, {} wanted — nothing to start", name, running, wanted);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if (!lead.isCreatable()) {
|
|
||||||
// A lead with a `tab:` but no `profile:` is recognise-only by design: the operator
|
|
||||||
// opens it by hand. Say so once rather than looking like a silent failure.
|
|
||||||
log.info("lead '{}' is not live, and names no profile — it can be recognised but not "
|
|
||||||
+ "launched. Add `profile:` under fleet.leaders.{} to have fleetd start it.",
|
|
||||||
name, name);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
FleetConfig.Profile profile = cfg.profiles().get(lead.profile());
|
ResolvedLead resolved = resolveLaunchable(name);
|
||||||
if (profile == null) {
|
if (resolved == null) {
|
||||||
log.warn("lead '{}' names profile '{}', which is not configured — not launching",
|
|
||||||
name, lead.profile());
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
for (int i = running; i < wanted; i++) {
|
for (int i = running; i < wanted; i++) {
|
||||||
if (launch(name, lead, profile)) {
|
if (launch(name, resolved.lead(), resolved.profile()) != null) {
|
||||||
started++;
|
started++;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -197,11 +219,87 @@ public final class LeadLauncher {
|
|||||||
return started;
|
return started;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** A declared lead paired with the profile it launches on — {@link #resolveLaunchable}'s result. */
|
||||||
|
private record ResolvedLead(FleetConfig.Leader lead, FleetConfig.Profile profile) {
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The declared {@code Leader} and its {@code Profile} for {@code name}, read from the config
|
||||||
|
* snapshot this launcher was constructed with.
|
||||||
|
*
|
||||||
|
* @return the resolved pair, or {@code null} (having logged) if {@code name} is not declared
|
||||||
|
* under {@code fleet.leaders}, that lead names no {@code profile:} (a {@code tab:}-only,
|
||||||
|
* recognise-only lead), or its {@code profile:} is not configured. Shared by
|
||||||
|
* {@link #ensureLeads()} and {@link #relaunch(String)} so the three refusals and their
|
||||||
|
* wording live in one place.
|
||||||
|
*/
|
||||||
|
private ResolvedLead resolveLaunchable(String name) {
|
||||||
|
FleetConfig.Leader lead = cfg.fleet().leaders().get(name);
|
||||||
|
if (lead == null) {
|
||||||
|
log.warn("lead '{}' is not declared under fleet.leaders — not launching", name);
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
if (!lead.isCreatable()) {
|
||||||
|
// A lead with a `tab:` but no `profile:` is recognise-only by design: the operator
|
||||||
|
// opens it by hand. Say so once rather than looking like a silent failure.
|
||||||
|
log.info("lead '{}' names no profile — it can be recognised but not launched. Add "
|
||||||
|
+ "`profile:` under fleet.leaders.{} to have fleetd start it.", name, name);
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
FleetConfig.Profile profile = cfg.profiles().get(lead.profile());
|
||||||
|
if (profile == null) {
|
||||||
|
log.warn("lead '{}' names profile '{}', which is not configured — not launching",
|
||||||
|
name, lead.profile());
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
return new ResolvedLead(lead, profile);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Start the named lead from the config snapshot this launcher was constructed with — not a
|
||||||
|
* live read, so a lead's {@code profile:} or {@code tab:} edited in config needs a daemon
|
||||||
|
* restart to take effect here — outside of {@link #ensureLeads()}'s {@code instances}
|
||||||
|
* bookkeeping.
|
||||||
|
*
|
||||||
|
* @return the started {@link Agent}, or {@code null} if {@code name} is not declared under
|
||||||
|
* {@code fleet.leaders}, that lead names no {@code profile:} (a {@code tab:}-only,
|
||||||
|
* recognise-only lead), its {@code profile:} is not configured, or every attempt up to
|
||||||
|
* {@link #RELAUNCH_ATTEMPTS} failed to start it. Never throws.
|
||||||
|
*
|
||||||
|
* <p>Does not count how many instances of this lead are already live. {@link #ensureLeads()}'s
|
||||||
|
* count exists to avoid starting a second orchestrator; the caller of this method has already
|
||||||
|
* decided to replace the lead and owns that decision.
|
||||||
|
*
|
||||||
|
* <p>Retries the whole launch attempt — not only the {@code agent_name_taken}/
|
||||||
|
* {@code agent_pane_busy} cases {@link ResilientAgentLaunch} already retries inside one
|
||||||
|
* {@code agents.start} call — up to {@link #RELAUNCH_ATTEMPTS} times, sleeping via the
|
||||||
|
* injected sleeper between attempts, and returns the agent from the first attempt that
|
||||||
|
* succeeds.
|
||||||
|
*/
|
||||||
|
public Agent relaunch(String name) {
|
||||||
|
ResolvedLead resolved = resolveLaunchable(name);
|
||||||
|
if (resolved == null) {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (int attempt = 1; attempt <= RELAUNCH_ATTEMPTS; attempt++) {
|
||||||
|
Agent started = launch(name, resolved.lead(), resolved.profile());
|
||||||
|
if (started != null) {
|
||||||
|
return started;
|
||||||
|
}
|
||||||
|
if (attempt < RELAUNCH_ATTEMPTS) {
|
||||||
|
sleeper.run();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* How many live leads exist per configured name, and which of that name's labelled tabs are
|
* How many live leads exist per configured name, and which of that name's labelled tabs are
|
||||||
* <em>not</em> live: a running agent in a tab labelled with that lead's exact {@code tab}
|
* <em>not</em> live: a running agent in a tab labelled with that lead's exact {@code tab}
|
||||||
* (CB-579). Member workspaces are excluded, exactly as the scanner excludes them: a member must
|
* (CB-579). A member sitting in the same shared workspace is not counted as a lead because its
|
||||||
* not be counted as a lead because it happens to sit in a matching tab.
|
* tab carries a different label, not because any workspace is excluded from this count.
|
||||||
*
|
*
|
||||||
* <p>There used to be a second path here — a running agent on the terminal a
|
* <p>There used to be a second path here — a running agent on the terminal a
|
||||||
* {@code fleet.leaders.<name>.terminal} pin named, for a lead opened and pinned by hand. That
|
* {@code fleet.leaders.<name>.terminal} pin named, for a lead opened and pinned by hand. That
|
||||||
@@ -305,8 +403,17 @@ public final class LeadLauncher {
|
|||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Start one lead. Returns false (having logged) rather than throwing on any failure. */
|
/**
|
||||||
private boolean launch(String name, FleetConfig.Leader lead, FleetConfig.Profile profile) {
|
* Start one lead. Returns null (having logged) rather than throwing on any failure.
|
||||||
|
*
|
||||||
|
* <p>Goes through the same {@link ResilientAgentLaunch} seam every member spawn uses
|
||||||
|
* (fleetd #727): the assembled argv is refused outright if it cannot fit the pane line herdr
|
||||||
|
* types it into, a stale {@code agent_name_taken} (a crashed session's name the registry has
|
||||||
|
* not yet released) is retried under a fresh per-attempt name rather than refusing the whole
|
||||||
|
* relaunch, and a seed pane whose shell has not reached its prompt yet ({@code
|
||||||
|
* agent_pane_busy}) is retried rather than failing on the first miss.
|
||||||
|
*/
|
||||||
|
private Agent launch(String name, FleetConfig.Leader lead, FleetConfig.Profile profile) {
|
||||||
String label = lead.tabLabel();
|
String label = lead.tabLabel();
|
||||||
String cwd = (lead.cwd() == null || lead.cwd().isBlank())
|
String cwd = (lead.cwd() == null || lead.cwd().isBlank())
|
||||||
? System.getProperty("user.dir") : lead.cwd();
|
? System.getProperty("user.dir") : lead.cwd();
|
||||||
@@ -322,8 +429,12 @@ public final class LeadLauncher {
|
|||||||
// Same shape as the member launchers: herdr resolves the executable from `kind`, so
|
// Same shape as the member launchers: herdr resolves the executable from `kind`, so
|
||||||
// argv[0] (the configured launcher, e.g. `ccs`) is dropped and only the rest is passed.
|
// argv[0] (the configured launcher, e.g. `ccs`) is dropped and only the rest is passed.
|
||||||
List<String> argv = leadArgv(profile);
|
List<String> argv = leadArgv(profile);
|
||||||
Agent started = agents.start("lead-" + name, herdrKind(profile),
|
ResilientAgentLaunch.checkFits(profile.profile(), argv);
|
||||||
argv.isEmpty() ? argv : argv.subList(1, argv.size()), tab.rootPaneId());
|
List<String> args = argv.isEmpty() ? argv : argv.subList(1, argv.size());
|
||||||
|
Agent started = ResilientAgentLaunch.startUniquelyNamed(agents, herdrKind(profile), args,
|
||||||
|
tab.rootPaneId(),
|
||||||
|
attempt -> "lead-" + name + "-" + nameNonce + "-" + nameSeq.incrementAndGet(),
|
||||||
|
ResilientAgentLaunch.NAME_RETRIES, ResilientAgentLaunch.SHELL_READY_RETRIES, sleeper);
|
||||||
|
|
||||||
// Label AFTER the start succeeds. A label written before would survive a failed start
|
// Label AFTER the start succeeds. A label written before would survive a failed start
|
||||||
// and then read back as a live lead on the next boot, which is the exact staleness the
|
// and then read back as a live lead on the next boot, which is the exact staleness the
|
||||||
@@ -333,7 +444,7 @@ public final class LeadLauncher {
|
|||||||
log.info("lead '{}' launched: profile={} tab={} pane={} terminal={} label='{}' cwd={}",
|
log.info("lead '{}' launched: profile={} tab={} pane={} terminal={} label='{}' cwd={}",
|
||||||
name, profile.profile(), tab.tab().tabId(), started.paneId(),
|
name, profile.profile(), tab.tab().tabId(), started.paneId(),
|
||||||
started.terminalId(), label, cwd);
|
started.terminalId(), label, cwd);
|
||||||
return true;
|
return started;
|
||||||
} catch (RuntimeException e) {
|
} catch (RuntimeException e) {
|
||||||
log.warn("lead '{}' failed to launch on profile '{}': {}",
|
log.warn("lead '{}' failed to launch on profile '{}': {}",
|
||||||
name, profile.profile(), e.getMessage());
|
name, profile.profile(), e.getMessage());
|
||||||
@@ -345,7 +456,7 @@ public final class LeadLauncher {
|
|||||||
tab.tab().tabId(), cleanup.getMessage());
|
tab.tab().tabId(), cleanup.getMessage());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return false;
|
return null;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -358,7 +469,8 @@ public final class LeadLauncher {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The lead's argv: the profile's own command, the model pin, and the bridge MCP mount.
|
* The lead's argv: the profile's own command, the model and auto-compaction pins, and the bridge
|
||||||
|
* MCP mount.
|
||||||
*
|
*
|
||||||
* <p>No {@code --append-system-prompt}. That flag carries the worker reply charter, and a lead
|
* <p>No {@code --append-system-prompt}. That flag carries the worker reply charter, and a lead
|
||||||
* is not a worker — it reads its orchestration rules from the project's {@code CLAUDE.md} like
|
* is not a worker — it reads its orchestration rules from the project's {@code CLAUDE.md} like
|
||||||
@@ -380,7 +492,7 @@ public final class LeadLauncher {
|
|||||||
argv.add("--model");
|
argv.add("--model");
|
||||||
argv.add(profile.model());
|
argv.add(profile.model());
|
||||||
}
|
}
|
||||||
return argv;
|
return profile.isOpenCode() ? argv : ClaudeCodeArguments.withAutoCompactWindow(argv, profile);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
@@ -1,8 +1,11 @@
|
|||||||
package dev.ltms.fleet.lead;
|
package dev.ltms.fleet.lead;
|
||||||
|
|
||||||
import dev.ltms.fleet.config.FleetConfig;
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.herdr.Agent;
|
||||||
import dev.ltms.fleet.herdr.AgentControl;
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
import dev.ltms.fleet.herdr.AgentStatus;
|
import dev.ltms.fleet.herdr.AgentStatus;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrException;
|
||||||
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
@@ -10,6 +13,8 @@ import java.io.IOException;
|
|||||||
import java.io.UncheckedIOException;
|
import java.io.UncheckedIOException;
|
||||||
import java.nio.file.Files;
|
import java.nio.file.Files;
|
||||||
import java.nio.file.Path;
|
import java.nio.file.Path;
|
||||||
|
import java.util.Collections;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
import java.util.Map;
|
import java.util.Map;
|
||||||
import java.util.UUID;
|
import java.util.UUID;
|
||||||
import java.util.concurrent.ConcurrentHashMap;
|
import java.util.concurrent.ConcurrentHashMap;
|
||||||
@@ -22,58 +27,69 @@ import java.util.function.Supplier;
|
|||||||
/**
|
/**
|
||||||
* fleetd #480: replace a lead session that has decided it is ready to be rolled over, without an
|
* fleetd #480: replace a lead session that has decided it is ready to be rolled over, without an
|
||||||
* operator doing it by hand. A lead writes a handover file, calls {@link #open}, and then — once
|
* operator doing it by hand. A lead writes a handover file, calls {@link #open}, and then — once
|
||||||
* every gate ({@link #confirm}'s own checks) has passed — a deferred, single-shot continuation
|
* every gate ({@link #confirm}'s own checks) has passed — a deferred, single-shot continuation ends
|
||||||
* clears the lead's own pane and bootstraps a fresh session against that file.
|
* the lead's own pane, launches a fresh one, and bootstraps that fresh session against the file.
|
||||||
*
|
*
|
||||||
* <p>This is the executor only. Nothing in this ticket wires an MCP tool onto {@link #open}/
|
* <p>This is the executor behind the {@code fleet_handover} MCP tool ({@code
|
||||||
* {@link #confirm}/{@link #cancel} — that is a separate, later unit; until it lands, nothing calls
|
* dev.ltms.fleet.mcp.FleetMcp#handover}), which drives {@link #open}, {@link #confirm}, {@link
|
||||||
* this class at all.
|
* #cancel}, and {@link #status} from a tool call.
|
||||||
*
|
*
|
||||||
* <p><strong>{@code confirm()} cannot roll inline — a fleetd #480 correction.</strong> The first
|
* <p><strong>{@code confirm()} cannot roll inline.</strong> {@code confirm()} is called BY the
|
||||||
* version of this class called {@code agents.send(lead, "/clear")} directly from inside {@code
|
* lead, FROM the lead's own turn: the lead's pane is {@code WORKING} for the whole duration of that
|
||||||
* confirm()}, then polled for the pane to become injectable again. That is wrong, because {@code
|
* call and cannot possibly report a real turn boundary until {@code confirm()} itself returns. So
|
||||||
* confirm()} is called BY the lead, FROM the lead's own turn: the lead's pane is {@code WORKING}
|
* {@link #confirm} validates every gate, then does no I/O against the lead's own pane at all — it
|
||||||
* for the whole duration of that call and cannot possibly report injectable until {@code confirm()}
|
* only records that the request is approved and hands a one-shot continuation to {@code
|
||||||
* itself returns. The poll always timed out — but only after the {@code /clear} had already been
|
* continuationRunner} before returning. That continuation is what actually touches the pane, once
|
||||||
* sent and queued in the pane, where it fired the instant the turn ended anyway. The result was the
|
* the calling turn has ended, in this order:
|
||||||
* worst outcome this feature can produce: a silently destroyed lead context with no fresh session
|
|
||||||
* ever started, and a refusal return value that claimed nothing had happened.
|
|
||||||
*
|
|
||||||
* <p>The fix: {@link #confirm} validates every gate, then does no I/O against the lead's own pane
|
|
||||||
* at all — it only records that the request is approved and hands a one-shot continuation to
|
|
||||||
* {@code continuationRunner} before returning. That continuation is what actually touches the pane,
|
|
||||||
* once the calling turn has ended, in this order:
|
|
||||||
* <ol>
|
* <ol>
|
||||||
* <li>wait for the lead's own pane to report a real turn boundary — {@code IDLE} or {@code
|
* <li>wait for the lead's own pane to report a real turn boundary — {@code IDLE} or {@code
|
||||||
* DONE}, never merely {@code BLOCKED} — i.e. wait for the very {@code confirm()} call that
|
* DONE}, never merely {@code BLOCKED} — i.e. wait for the very {@code confirm()} call that
|
||||||
* approved this roll to finish its turn — bounded by {@code turnSettleSeconds}. <strong>If
|
* approved this roll to finish its turn — bounded by {@code turnSettleSeconds}. <strong>If
|
||||||
* this never happens, nothing else in this list runs: no {@code /clear} is ever sent.</strong>
|
* this never happens, nothing else in this list runs: the old pane is never touched.</strong>
|
||||||
* A lead that never goes idle is a lead still doing real work, and clearing it would throw
|
* A lead that never goes idle is a lead still doing real work, and tearing it down would throw
|
||||||
* away live context — exactly the failure this correction exists to prevent.</li>
|
* away live context.</li>
|
||||||
* <li>{@code agents.send(lead, "/clear")}</li>
|
* <li>capture the old pane id (and, through it, the old tab) from {@link AgentControl#get}, with
|
||||||
* <li>wait for {@code /clear} to be picked up and settle, bounded by {@code clearSettleSeconds}
|
* a bounded retry — the terminal-to-pane lookup it goes through can itself report a genuinely
|
||||||
* (fleetd #489: no longer a plain re-check of the same boundary — {@code /clear} starts no
|
* live agent as not found (see {@code AgentControl#agentCall}'s own re-resolve-once
|
||||||
* turn of its own, so this instead nudges the submit keystroke while no pickup has been seen,
|
* behaviour), and one false negative here must not abort an otherwise-healthy roll. Neither id
|
||||||
* then waits for a real {@code WORKING} → {@code IDLE}/{@code DONE} boundary once one has;
|
* is ever re-resolved from the terminal again after this — once the pane below is closed there
|
||||||
* see {@link #waitForClearPickupAndSettle})</li>
|
* is nothing left to resolve it from.</li>
|
||||||
* <li>{@code agents.send(lead, cfg.bootstrapTextFor(p.handoverPath()))}</li>
|
* <li>resolve the lead's configured name from its terminal, for the relaunch step below.</li>
|
||||||
|
* <li>end the old session: close the pane (an already-gone pane counts as success; any other
|
||||||
|
* failure propagates), then close its tab only when the pane was that tab's sole occupant —
|
||||||
|
* the same pane-then-tab teardown {@code HerdrPeerLauncher#stop} uses for a member.</li>
|
||||||
|
* <li>confirm the old pane is actually gone by polling {@link
|
||||||
|
* dev.ltms.fleet.herdr.WorkspaceControl#locatePane} for a {@code null} result — never {@link
|
||||||
|
* AgentControl#status}, and never the live-lead terminal map, each of which answers a
|
||||||
|
* different question. <strong>If the old pane is never confirmed gone, no relaunch is
|
||||||
|
* attempted</strong> — see {@link RollState#OLD_PANE_NEVER_DIED}.</li>
|
||||||
|
* <li>launch a fresh lead with {@code LeadLauncher#relaunch}. <strong>If every attempt fails,
|
||||||
|
* {@code bootstrapText} is never sent</strong> — see {@link RollState#RELAUNCH_FAILED}.</li>
|
||||||
|
* <li>wait for the fresh pane to reach a real turn boundary ({@code IDLE} or {@code DONE},
|
||||||
|
* never merely {@code BLOCKED}), bounded by {@code relaunchReadySeconds}. This is the
|
||||||
|
* safety gate: typing into a pane that has not actually finished booting loses the
|
||||||
|
* keystrokes. <strong>If the pane never becomes ready, {@code bootstrapText} is never
|
||||||
|
* sent</strong> — see {@link RollState#RELAUNCH_NEVER_READY}.</li>
|
||||||
|
* <li>wait for the fresh terminal to be recognised as a live lead — present in the live-lead
|
||||||
|
* terminal map — bounded by {@code relaunchReadySeconds}. This is bookkeeping, not a
|
||||||
|
* safety gate: {@code bootstrapText} is sent either way once the pane is ready, whether or
|
||||||
|
* not this wait itself times out — see {@link RollState#RELAUNCH_NOT_RECOGNISED}.</li>
|
||||||
|
* <li>{@code agents.send(newTerminal, cfg.bootstrapTextFor(p.handoverPath()))} — sent to the
|
||||||
|
* FRESH terminal, never the one that was just torn down.</li>
|
||||||
* </ol>
|
* </ol>
|
||||||
* A {@link #confirm} that returns {@link RollDecision#approved()} therefore means <em>"every gate
|
* A {@link #confirm} that returns {@link RollDecision#approved()} therefore means <em>"every gate
|
||||||
* passed and the roll is scheduled"</em>, never <em>"the pane has been cleared"</em> — the pane may
|
* passed and the roll is scheduled"</em>, never <em>"the lead has already been replaced"</em> — the
|
||||||
* still be mid-turn, possibly for a long time, when the caller gets that answer back.
|
* old pane may still be mid-turn, possibly for a long time, when the caller gets that answer back.
|
||||||
*
|
*
|
||||||
* <p><strong>The safety invariant survives this change, restated precisely.</strong> The ticket
|
* <p><strong>The safety invariant.</strong> "No timer, no scheduler, no background thread" means
|
||||||
* that first defined this class required "no timer, no scheduler, no background thread" so that
|
* that nothing but an explicit {@link #confirm} call can ever tear a lead's pane down.
|
||||||
* nothing but an explicit {@link #confirm} call could ever cause a {@code /clear}. That invariant
|
* {@code continuationRunner} launches a single-shot task that exists only because one specific,
|
||||||
* is about INITIATIVE, not about synchronicity, and this correction keeps it: {@code
|
|
||||||
* continuationRunner} launches a single-shot task that exists only because one specific,
|
|
||||||
* already-approved {@link #confirm} call created it — it is not recurring, it is not started at
|
* already-approved {@link #confirm} call created it — it is not recurring, it is not started at
|
||||||
* construction time or on any schedule, and no two invocations of it ever share state. A recurring
|
* construction time or on any schedule, and no two invocations of it ever share state. A recurring
|
||||||
* heartbeat or timer that could decide on its own initiative to roll a pane is still, and will
|
* heartbeat or timer that could decide on its own initiative to roll a pane is absent from this
|
||||||
* always be, absent from this class. <strong>Nothing but an explicit {@link #confirm} call that
|
* class. <strong>Nothing but an explicit {@link #confirm} call that passes every gate can ever tear
|
||||||
* passes every gate can ever cause a {@code /clear} — that call may simply finish its own work
|
* a pane down — that call may simply finish its own work slightly later than the method return, as
|
||||||
* slightly later than the method return, as a continuation of the same approved request, rather
|
* a continuation of the same approved request, rather than entirely inside the method body.</strong>
|
||||||
* than entirely inside the method body.</strong>
|
|
||||||
*
|
*
|
||||||
* <p><strong>Identity is resolved by the caller, never looked up here — a second fleetd #480
|
* <p><strong>Identity is resolved by the caller, never looked up here — a second fleetd #480
|
||||||
* correction.</strong> The first version resolved the pane to clear via {@code
|
* correction.</strong> The first version resolved the pane to clear via {@code
|
||||||
@@ -111,19 +127,17 @@ public final class LeadRollover {
|
|||||||
|
|
||||||
private static final Logger log = LoggerFactory.getLogger(LeadRollover.class);
|
private static final Logger log = LoggerFactory.getLogger(LeadRollover.class);
|
||||||
|
|
||||||
/** Poll interval while waiting for the lead's pane to settle after {@code /clear}. */
|
/** Poll interval shared by every bounded wait in this class. */
|
||||||
static final long SETTLE_POLL_MS = 250;
|
static final long POLL_INTERVAL_MS = 250;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* How many consecutive not-yet-picked-up polls {@link #waitForClearPickupAndSettle} allows
|
* How long {@link #waitUntilPaneGone} polls {@link WorkspaceControl#locatePane} before giving
|
||||||
* before releasing rather than wedging the roll — the same constant and the same
|
* up on ever seeing the old pane disappear. Not configurable: once {@link #endOldSession} has
|
||||||
* release-not-wedge choice {@link dev.ltms.fleet.inject.Injector} already makes for its own
|
* closed the pane (and, usually, its tab), herdr dropping the pane from its own bookkeeping is
|
||||||
* post-turn {@code /clear} housekeeping (fleetd #306). <strong>This bounds the number of
|
* expected to show up within one or two polls, not on an operator-tunable timescale the way a
|
||||||
* consecutive polls, not the number of nudges:</strong> the first {@code PICKUP_GRACE_POLLS - 1}
|
* CLI boot is.
|
||||||
* of those polls each send a nudge, and the {@code PICKUP_GRACE_POLLS}th releases instead of
|
|
||||||
* nudging again — so 8 polls produce 7 nudges, not 8.
|
|
||||||
*/
|
*/
|
||||||
static final int PICKUP_GRACE_POLLS = 8;
|
static final int PANE_DEATH_TIMEOUT_SECONDS = 10;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* One request opened by {@link #open}, pending its {@link #confirm} (or {@link #cancel}).
|
* One request opened by {@link #open}, pending its {@link #confirm} (or {@link #cancel}).
|
||||||
@@ -160,15 +174,21 @@ public final class LeadRollover {
|
|||||||
* The handover file's modified time is not after {@link #open}'s request timestamp, or is
|
* The handover file's modified time is not after {@link #open}'s request timestamp, or is
|
||||||
* older than {@code maxDocAgeSeconds}.
|
* older than {@code maxDocAgeSeconds}.
|
||||||
*/
|
*/
|
||||||
HANDOVER_STALE
|
HANDOVER_STALE,
|
||||||
|
/**
|
||||||
|
* This lead terminal already has a roll running: an earlier {@link #confirm} call claimed
|
||||||
|
* it and that roll's continuation has not released it yet. {@code detail} names the lead
|
||||||
|
* terminal and the token that holds the claim.
|
||||||
|
*/
|
||||||
|
ROLL_ALREADY_RUNNING
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The outcome of a {@link #confirm} call. {@link #approved()} means every gate passed and the
|
* The outcome of a {@link #confirm} call. {@link #approved()} means every gate passed and the
|
||||||
* roll has been handed to a one-shot continuation — <strong>not</strong> that the pane has been
|
* roll has been handed to a one-shot continuation — <strong>not</strong> that the lead has
|
||||||
* cleared; the continuation may still be waiting for the calling turn to end when this returns.
|
* already been replaced; the continuation may still be waiting for the calling turn to end when
|
||||||
* Whether the deferred roll itself later goes on to clear the pane, refuse for never going
|
* this returns. Whether the deferred roll itself later goes on to tear the old pane down and
|
||||||
* idle, or refuse for never re-settling after {@code /clear} is logged only (see this class's
|
* relaunch the lead, or refuses at any of its own steps, is logged only (see this class's
|
||||||
* javadoc) — there is deliberately no synchronous caller left by that point to hand a result to.
|
* javadoc) — there is deliberately no synchronous caller left by that point to hand a result to.
|
||||||
*/
|
*/
|
||||||
public record RollDecision(boolean accepted, RefusalReason reason, String detail) {
|
public record RollDecision(boolean accepted, RefusalReason reason, String detail) {
|
||||||
@@ -181,7 +201,134 @@ public final class LeadRollover {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* How many tokens {@link #outcomes} remembers before it starts evicting the oldest — bounded
|
||||||
|
* so a long-running daemon never grows this map without limit. Chosen generously rather than
|
||||||
|
* tightly: production rolls are rare (this class's own ticket found exactly ONE completed roll
|
||||||
|
* ever logged on this host), and each entry is a handful of short strings, so even a full cap
|
||||||
|
* costs a few tens of kilobytes — nowhere near a reason to make it configurable. 200 entries
|
||||||
|
* comfortably outlasts any operator's own memory of "did that roll I asked for actually
|
||||||
|
* happen", which is the whole reason {@link #status} exists.
|
||||||
|
*
|
||||||
|
* <p><strong>This cap counts {@link RollState#IN_PROGRESS} entries exactly the same as
|
||||||
|
* finished ones.</strong> There is only the one bounded map: {@link #confirm} writes an {@link
|
||||||
|
* RollState#IN_PROGRESS} entry into {@link #outcomes} at hand-off, and the deferred
|
||||||
|
* continuation later overwrites that SAME key with a terminal state — it never inserts a
|
||||||
|
* second entry. An approved roll therefore occupies one slot in this map for its entire
|
||||||
|
* lifetime, from the moment {@link #confirm} hands off, not only once it finishes; a
|
||||||
|
* confirmed-but-not-yet-finished roll counts against the cap exactly like a finished one. The
|
||||||
|
* alternative (a separate, uncapped in-flight map) would let a burst of confirmed-but-stuck
|
||||||
|
* rolls grow without bound — the exact failure this cap exists to prevent — so it was rejected.
|
||||||
|
*/
|
||||||
|
static final int OUTCOME_HISTORY_CAP = 200;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* What is known about one token, right now — the answer {@link #status} gives. Distinguishes
|
||||||
|
* five terminal outcomes an approved roll can finish with, one in-flight outcome for a roll
|
||||||
|
* that has been approved but has not finished yet, and two answers for a token that names no
|
||||||
|
* active work at all: still pending confirmation, or nothing known about this token at all.
|
||||||
|
*/
|
||||||
|
public enum RollState {
|
||||||
|
/**
|
||||||
|
* {@code token} is still open: either {@link #open} was called and {@link #confirm} has not
|
||||||
|
* been (or not successfully) yet, or a {@link #confirm} call failed one of its gate checks
|
||||||
|
* and left the token pending for a retry — see {@link #confirm}'s javadoc ("token stays
|
||||||
|
* pending"). Indistinguishable from a genuinely fresh request; a caller wanting to know
|
||||||
|
* WHICH gate most recently refused should read the {@link RollDecision} that {@link
|
||||||
|
* #confirm} itself returned, not this status. <strong>Never the state of an APPROVED
|
||||||
|
* roll</strong> — see {@link #IN_PROGRESS}, which {@link #confirm} records at the moment it
|
||||||
|
* hands off, before this token is even removed from the pending set.
|
||||||
|
*/
|
||||||
|
PENDING,
|
||||||
|
/**
|
||||||
|
* {@link #confirm} approved this roll and handed it to the deferred continuation, which has
|
||||||
|
* not finished yet. Recorded by {@link #confirm} itself, at hand-off — <strong>before</strong>
|
||||||
|
* {@code token} is removed from the pending set — so there is never a gap in which {@link
|
||||||
|
* #status} could wrongly answer {@link #UNKNOWN} ("nothing was ever requested") for a roll
|
||||||
|
* that is, in fact, actively running. This is not sticky: the deferred continuation
|
||||||
|
* overwrites this same entry with a terminal state ({@link #ROLLED}, {@link
|
||||||
|
* #TURN_NEVER_SETTLED}, {@link #OLD_PANE_NEVER_DIED}, {@link #RELAUNCH_FAILED}, {@link
|
||||||
|
* #RELAUNCH_NEVER_READY}, {@link #RELAUNCH_NOT_RECOGNISED}, or {@link #FAILED}) once it
|
||||||
|
* finishes — including by throwing, which {@link #runRollover}'s catch turns into {@link
|
||||||
|
* #FAILED} instead of leaving this entry stuck forever.
|
||||||
|
*/
|
||||||
|
IN_PROGRESS,
|
||||||
|
/**
|
||||||
|
* {@link #confirm} was approved and the deferred continuation completed the entire roll: the
|
||||||
|
* calling lead's turn settled, the old pane was torn down and confirmed gone, a fresh lead
|
||||||
|
* was launched and recognised, and {@code bootstrapText} was sent to it.
|
||||||
|
*/
|
||||||
|
ROLLED,
|
||||||
|
/**
|
||||||
|
* {@link #confirm} was approved, but the calling lead's own turn never reached a boundary
|
||||||
|
* (IDLE or DONE) within {@code turnSettleSeconds} — the old pane was never touched at all.
|
||||||
|
* This is the state that makes a lead's own stuck turn VISIBLE: without it, a lead that hit
|
||||||
|
* this case would have no way to find out, and would carry on believing it was about to be
|
||||||
|
* replaced. See this class's javadoc.
|
||||||
|
*/
|
||||||
|
TURN_NEVER_SETTLED,
|
||||||
|
/**
|
||||||
|
* {@link #confirm} was approved and the calling lead's turn settled, the old pane was closed
|
||||||
|
* (and its tab, if it was the sole occupant), but {@link
|
||||||
|
* dev.ltms.fleet.herdr.WorkspaceControl#locatePane} kept reporting it as still present for
|
||||||
|
* the whole pane-death timeout. No relaunch was ever attempted, and {@code bootstrapText}
|
||||||
|
* was never sent.
|
||||||
|
*/
|
||||||
|
OLD_PANE_NEVER_DIED,
|
||||||
|
/**
|
||||||
|
* The old pane was confirmed gone, but {@code LeadLauncher#relaunch} returned {@code null}
|
||||||
|
* — every launch attempt failed. {@code bootstrapText} was never sent, and no fresh terminal
|
||||||
|
* exists for this roll to have recognised.
|
||||||
|
*/
|
||||||
|
RELAUNCH_FAILED,
|
||||||
|
/**
|
||||||
|
* A fresh lead was launched, but its pane never reached a real turn boundary ({@code IDLE}
|
||||||
|
* or {@code DONE}, never merely {@code BLOCKED}) within {@code relaunchReadySeconds} — the
|
||||||
|
* CLI never finished booting, or it stayed paused on a startup prompt. {@code bootstrapText}
|
||||||
|
* was never sent: typing into a pane that is not actually ready to accept input loses the
|
||||||
|
* keystrokes.
|
||||||
|
*/
|
||||||
|
RELAUNCH_NEVER_READY,
|
||||||
|
/**
|
||||||
|
* A fresh lead was launched and its pane reached a real turn boundary, so {@code
|
||||||
|
* bootstrapText} WAS sent to it, but the terminal was never recognised as a live lead —
|
||||||
|
* present in the live-lead terminal map — within {@code relaunchReadySeconds}. The session
|
||||||
|
* itself is alive and bootstrapped; only the daemon's own bookkeeping has not caught up, and
|
||||||
|
* an operator should check why the tab was not recognised.
|
||||||
|
*/
|
||||||
|
RELAUNCH_NOT_RECOGNISED,
|
||||||
|
/**
|
||||||
|
* The deferred continuation threw a {@link RuntimeException} and the continuation thread
|
||||||
|
* died with it. Without this state, that throw would leave {@link #outcomes} holding {@link
|
||||||
|
* #IN_PROGRESS} forever, because the production {@code continuationRunner} is a bare virtual
|
||||||
|
* thread with no uncaught-exception handler and nothing downstream of the throw ever runs to
|
||||||
|
* write a terminal outcome. {@code detail} names the exception, so a reader has something to
|
||||||
|
* act on. The roll is dead at this point and does not retry itself; a stuck lead must
|
||||||
|
* {@link #open} a fresh request.
|
||||||
|
*/
|
||||||
|
FAILED,
|
||||||
|
/**
|
||||||
|
* {@code token} names nothing this instance currently knows about: never issued by {@link
|
||||||
|
* #open}, dropped by {@link #cancel}, or aged out of {@link #outcomes}'s bounded history.
|
||||||
|
* These three causes are not distinguished — all of them mean "there is nothing to tell
|
||||||
|
* you", which is the entire content of a clean answer here.
|
||||||
|
*/
|
||||||
|
UNKNOWN
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The answer {@link #status} gives for one token: a {@link RollState} and a human-readable
|
||||||
|
* {@code detail}. For {@link RollState#TURN_NEVER_SETTLED}, {@code detail} names {@code
|
||||||
|
* turnSettleSeconds} and its configured value explicitly, so a reader who sees this knows what
|
||||||
|
* to raise.
|
||||||
|
*/
|
||||||
|
public record RollStatus(RollState state, String detail) {}
|
||||||
|
|
||||||
private final AgentControl agents;
|
private final AgentControl agents;
|
||||||
|
/** Workspace/tab/pane control — used to tear down the old pane and confirm it is gone. */
|
||||||
|
private final WorkspaceControl spaces;
|
||||||
|
/** Starts the fresh lead that replaces the one this roll tears down. */
|
||||||
|
private final LeadLauncher launcher;
|
||||||
private final Supplier<FleetConfig.LeadRollover> configSupplier;
|
private final Supplier<FleetConfig.LeadRollover> configSupplier;
|
||||||
/**
|
/**
|
||||||
* Terminal id → that lead's configured workspace directory (their {@code
|
* Terminal id → that lead's configured workspace directory (their {@code
|
||||||
@@ -191,8 +338,21 @@ public final class LeadRollover {
|
|||||||
* daemon-cwd bug this parameter exists to fix.
|
* daemon-cwd bug this parameter exists to fix.
|
||||||
*/
|
*/
|
||||||
private final Function<String, String> leadWorkspace;
|
private final Function<String, String> leadWorkspace;
|
||||||
|
/**
|
||||||
|
* Terminal id → that lead's configured name under {@code fleet.leaders}, or {@code null} when
|
||||||
|
* the terminal names no currently-recognised lead. The deferred continuation calls this, on the
|
||||||
|
* OLD terminal, before tearing it down, so it knows which lead to pass to {@link
|
||||||
|
* LeadLauncher#relaunch}.
|
||||||
|
*/
|
||||||
|
private final Function<String, String> leadNameForTerminal;
|
||||||
|
/**
|
||||||
|
* The daemon's current terminal id → lead name map, read fresh on every poll. The deferred
|
||||||
|
* continuation polls this for the FRESH terminal {@link LeadLauncher#relaunch} returns, to
|
||||||
|
* learn when that terminal has been recognised as a live lead — see this class's javadoc.
|
||||||
|
*/
|
||||||
|
private final Supplier<Map<String, String>> liveLeadTerminals;
|
||||||
private final LongSupplier nowMillis;
|
private final LongSupplier nowMillis;
|
||||||
private final Runnable settleSleeper;
|
private final Runnable pollSleeper;
|
||||||
/**
|
/**
|
||||||
* Launches the post-{@code confirm()} continuation. Production uses a single unstarted virtual
|
* Launches the post-{@code confirm()} continuation. Production uses a single unstarted virtual
|
||||||
* thread per confirmed request — see this class's javadoc for why that is a single-shot task,
|
* thread per confirmed request — see this class's javadoc for why that is a single-shot task,
|
||||||
@@ -201,30 +361,67 @@ public final class LeadRollover {
|
|||||||
*/
|
*/
|
||||||
private final Consumer<Runnable> continuationRunner;
|
private final Consumer<Runnable> continuationRunner;
|
||||||
private final Map<String, PendingRollover> pending = new ConcurrentHashMap<>();
|
private final Map<String, PendingRollover> pending = new ConcurrentHashMap<>();
|
||||||
|
/**
|
||||||
|
* Lead terminal → the token of the roll currently holding that terminal exclusive, for
|
||||||
|
* {@link #confirm}'s single-flight claim. {@link #confirm} claims an entry here with an
|
||||||
|
* atomic put-if-absent once every other gate has passed, refusing with {@link
|
||||||
|
* RefusalReason#ROLL_ALREADY_RUNNING} when a claim is already held; {@link #runRollover}
|
||||||
|
* releases it in a {@code finally}, on both the success and the thrown-exception path. A
|
||||||
|
* terminal absent from this map has no roll currently in flight for it.
|
||||||
|
*/
|
||||||
|
private final Map<String, String> rollingByTerminal = new ConcurrentHashMap<>();
|
||||||
|
/**
|
||||||
|
* Finished tokens → what actually happened, for {@link #status}. Bounded by {@link
|
||||||
|
* #OUTCOME_HISTORY_CAP}, oldest evicted first ({@code removeEldestEntry} on an insertion-order
|
||||||
|
* {@link LinkedHashMap}). Wrapped in {@link Collections#synchronizedMap} because entries are
|
||||||
|
* written from whatever thread {@code continuationRunner} runs the roll on (a fresh virtual
|
||||||
|
* thread in production, the calling test thread under {@code Runnable::run}) and read from
|
||||||
|
* whatever thread calls {@link #status} (the MCP handler thread) — a plain {@code
|
||||||
|
* LinkedHashMap} is not safe for that, and {@code removeEldestEntry} additionally requires
|
||||||
|
* external synchronization even for a thread-safe map that merely wraps it.
|
||||||
|
*/
|
||||||
|
private final Map<String, RollStatus> outcomes = Collections.synchronizedMap(
|
||||||
|
new LinkedHashMap<>(16, 0.75f, false) {
|
||||||
|
@Override
|
||||||
|
protected boolean removeEldestEntry(Map.Entry<String, RollStatus> eldest) {
|
||||||
|
return size() > OUTCOME_HISTORY_CAP;
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
/** Production constructor — wall clock, real sleep between settle polls, a real virtual thread. */
|
/** Production constructor — wall clock, real sleep between polls, a real virtual thread. */
|
||||||
public LeadRollover(AgentControl agents, Supplier<FleetConfig.LeadRollover> configSupplier,
|
public LeadRollover(AgentControl agents, WorkspaceControl spaces, LeadLauncher launcher,
|
||||||
Function<String, String> leadWorkspace) {
|
Supplier<FleetConfig.LeadRollover> configSupplier,
|
||||||
this(agents, configSupplier, leadWorkspace, System::currentTimeMillis,
|
Function<String, String> leadWorkspace,
|
||||||
() -> sleepUninterruptibly(SETTLE_POLL_MS),
|
Function<String, String> leadNameForTerminal,
|
||||||
|
Supplier<Map<String, String>> liveLeadTerminals) {
|
||||||
|
this(agents, spaces, launcher, configSupplier, leadWorkspace, leadNameForTerminal,
|
||||||
|
liveLeadTerminals, System::currentTimeMillis,
|
||||||
|
() -> sleepUninterruptibly(POLL_INTERVAL_MS),
|
||||||
r -> Thread.ofVirtual().name("lead-rollover-continuation-").start(r));
|
r -> Thread.ofVirtual().name("lead-rollover-continuation-").start(r));
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Full constructor — an injectable wall-clock supplier, settle-poll sleeper, and continuation
|
* Full constructor — an injectable wall-clock supplier, poll sleeper, and continuation runner,
|
||||||
* runner, for tests. {@code nowMillis} MUST be a wall-clock source (e.g. {@code
|
* for tests. {@code nowMillis} MUST be a wall-clock source (e.g. {@code
|
||||||
* System.currentTimeMillis()}), never {@code System.nanoTime()}: the freshness check compares
|
* System.currentTimeMillis()}), never {@code System.nanoTime()}: the freshness check compares
|
||||||
* against a file's modified time, which only a wall clock is comparable to, and {@code
|
* against a file's modified time, which only a wall clock is comparable to, and {@code
|
||||||
* nanoTime} freezes while the host sleeps (fleetd #386).
|
* nanoTime} freezes while the host sleeps.
|
||||||
*/
|
*/
|
||||||
LeadRollover(AgentControl agents, Supplier<FleetConfig.LeadRollover> configSupplier,
|
LeadRollover(AgentControl agents, WorkspaceControl spaces, LeadLauncher launcher,
|
||||||
Function<String, String> leadWorkspace, LongSupplier nowMillis,
|
Supplier<FleetConfig.LeadRollover> configSupplier,
|
||||||
Runnable settleSleeper, Consumer<Runnable> continuationRunner) {
|
Function<String, String> leadWorkspace,
|
||||||
|
Function<String, String> leadNameForTerminal,
|
||||||
|
Supplier<Map<String, String>> liveLeadTerminals,
|
||||||
|
LongSupplier nowMillis, Runnable pollSleeper, Consumer<Runnable> continuationRunner) {
|
||||||
this.agents = agents;
|
this.agents = agents;
|
||||||
|
this.spaces = spaces;
|
||||||
|
this.launcher = launcher;
|
||||||
this.configSupplier = configSupplier;
|
this.configSupplier = configSupplier;
|
||||||
this.leadWorkspace = leadWorkspace;
|
this.leadWorkspace = leadWorkspace;
|
||||||
|
this.leadNameForTerminal = leadNameForTerminal;
|
||||||
|
this.liveLeadTerminals = liveLeadTerminals;
|
||||||
this.nowMillis = nowMillis;
|
this.nowMillis = nowMillis;
|
||||||
this.settleSleeper = settleSleeper;
|
this.pollSleeper = pollSleeper;
|
||||||
this.continuationRunner = continuationRunner;
|
this.continuationRunner = continuationRunner;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -366,51 +563,372 @@ public final class LeadRollover {
|
|||||||
return docCheck;
|
return docCheck;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Single-flight claim: atomic put-if-absent, taken only after every other gate has
|
||||||
|
// passed, so a refused confirm() never takes it. A non-null previous value means a
|
||||||
|
// different, still-running roll already holds this lead terminal.
|
||||||
|
String holder = rollingByTerminal.putIfAbsent(p.leadTerminal(), token);
|
||||||
|
if (holder != null) {
|
||||||
|
return RollDecision.refused(RefusalReason.ROLL_ALREADY_RUNNING,
|
||||||
|
"lead terminal " + p.leadTerminal() + " already has a roll running under token "
|
||||||
|
+ holder);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Record IN_PROGRESS BEFORE removing from `pending` — see RollState#IN_PROGRESS and
|
||||||
|
// OUTCOME_HISTORY_CAP's javadoc. This ordering means `token` is written into `outcomes`
|
||||||
|
// while it is STILL present in `pending`; status() checks `outcomes` first (see that
|
||||||
|
// method), so it reports IN_PROGRESS immediately, not the brief-but-real gap a
|
||||||
|
// remove-then-put ordering would leave in which the token is in neither map.
|
||||||
|
outcomes.put(token, new RollStatus(RollState.IN_PROGRESS,
|
||||||
|
"confirm() approved this roll and handed it to the deferred continuation; it has "
|
||||||
|
+ "not finished yet — still waiting for the calling turn to settle, for the "
|
||||||
|
+ "old pane to be torn down and confirmed gone, for the fresh lead to be "
|
||||||
|
+ "recognised, or for bootstrapText to be sent"));
|
||||||
pending.remove(token);
|
pending.remove(token);
|
||||||
log.info("lead-rollover: confirmed token={} lead={} — roll scheduled once the calling turn ends",
|
log.info("lead-rollover: confirmed token={} lead={} — roll scheduled once the calling turn ends",
|
||||||
token, callerTerminal);
|
token, callerTerminal);
|
||||||
continuationRunner.accept(() -> runRollover(p, cfg));
|
try {
|
||||||
|
continuationRunner.accept(() -> runRollover(p, cfg));
|
||||||
|
} catch (RuntimeException e) {
|
||||||
|
// continuationRunner can reject the hand-off itself (e.g. a bounded executor's
|
||||||
|
// RejectedExecutionException) before runRollover ever starts, so runRollover's own
|
||||||
|
// finally — the only other place that releases rollingByTerminal — never runs either.
|
||||||
|
// Release the claim here and overwrite the IN_PROGRESS entry with a terminal outcome,
|
||||||
|
// or this lead terminal could never be rolled again and status() would report
|
||||||
|
// IN_PROGRESS forever for a roll that in fact never started.
|
||||||
|
log.warn("lead-rollover: continuationRunner rejected token={} lead={}: {} — the roll "
|
||||||
|
+ "never started; releasing its claim and reporting it as FAILED",
|
||||||
|
token, callerTerminal, e.toString(), e);
|
||||||
|
rollingByTerminal.remove(p.leadTerminal(), token);
|
||||||
|
outcomes.put(token, new RollStatus(RollState.FAILED,
|
||||||
|
"continuationRunner rejected this roll before it ever started: " + e.toString()
|
||||||
|
+ " — the roll never ran; open() a fresh rollover request"));
|
||||||
|
}
|
||||||
return RollDecision.approved();
|
return RollDecision.approved();
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The single-shot continuation {@link #confirm} hands to {@code continuationRunner}. Runs
|
* The single-shot continuation {@link #confirm} hands to {@code continuationRunner}. Runs
|
||||||
* entirely after {@link #confirm} has returned to its caller — see this class's javadoc for the
|
* entirely after {@link #confirm} has returned to its caller — see this class's javadoc for the
|
||||||
* four-step order. There is no result to return to by this point, so every outcome is logged
|
* full order. There is no result to return to by this point, so every outcome is logged only.
|
||||||
* only.
|
*
|
||||||
|
* <p><strong>The whole body is wrapped in one {@code try}.</strong> Several calls below —
|
||||||
|
* {@code agents.get}, {@code agents.close}, {@code agents.send} — can throw an unchecked {@link
|
||||||
|
* dev.ltms.fleet.herdr.HerdrException} (see {@code AgentControl.java}), and the production
|
||||||
|
* {@code continuationRunner} is a bare virtual thread with no uncaught-exception handler (see
|
||||||
|
* this class's public constructor). An uncaught throw would kill the continuation thread
|
||||||
|
* silently, leaving the {@link RollState#IN_PROGRESS} entry {@link #confirm} wrote at hand-off
|
||||||
|
* stuck forever — {@link #status} would have no way to tell a dead roll from one still
|
||||||
|
* genuinely running. The {@code catch} below is scoped to the method body rather than to each
|
||||||
|
* call individually, so it also covers every call in this continuation, not a fixed list of
|
||||||
|
* call sites — the same reasoning that put the write-a-terminal-outcome step at each of this
|
||||||
|
* method's other exits rather than inside the helpers that detect them.</p>
|
||||||
|
*
|
||||||
|
* <p>Only {@link RuntimeException} is caught, matching the local convention {@link
|
||||||
|
* #waitUntilAtTurnBoundary} already set around its own {@code agents.status} call — not the
|
||||||
|
* broader {@link Exception} or {@link Throwable}, which would also swallow something like an
|
||||||
|
* {@link OutOfMemoryError} this continuation has no business handling.</p>
|
||||||
*/
|
*/
|
||||||
private void runRollover(PendingRollover p, FleetConfig.LeadRollover cfg) {
|
private void runRollover(PendingRollover p, FleetConfig.LeadRollover cfg) {
|
||||||
|
try {
|
||||||
|
runRolloverUnguarded(p, cfg);
|
||||||
|
} catch (RuntimeException e) {
|
||||||
|
log.warn("lead-rollover: continuation for token={} lead={} threw {} — the roll is dead; "
|
||||||
|
+ "no further step in this continuation will run",
|
||||||
|
p.token(), p.leadTerminal(), e.toString(), e);
|
||||||
|
outcomes.put(p.token(), new RollStatus(RollState.FAILED,
|
||||||
|
"the roll's continuation threw " + e.toString() + " — the roll is dead and will "
|
||||||
|
+ "not retry itself; check the daemon log for the stack trace, then open() "
|
||||||
|
+ "a fresh rollover request"));
|
||||||
|
} finally {
|
||||||
|
// Release the single-flight claim on both the normal return and the thrown-exception
|
||||||
|
// path above — a release only on success would leave this lead terminal unrollable
|
||||||
|
// forever after one failure. The conditional two-argument remove only clears the
|
||||||
|
// entry this roll itself holds, never a different roll's claim on the same terminal.
|
||||||
|
rollingByTerminal.remove(p.leadTerminal(), p.token());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The actual body of {@link #runRollover}, unwrapped — see that method's javadoc for the catch. */
|
||||||
|
private void runRolloverUnguarded(PendingRollover p, FleetConfig.LeadRollover cfg) {
|
||||||
String lead = p.leadTerminal();
|
String lead = p.leadTerminal();
|
||||||
boolean turnSettled = waitUntilAtTurnBoundary(lead, cfg.turnSettleSeconds());
|
long rollStartMillis = nowMillis.getAsLong();
|
||||||
if (!turnSettled) {
|
|
||||||
log.warn("lead-rollover: pane {} did not reach a turn boundary (IDLE or DONE) within {}s "
|
TurnSettleResult turnResult = waitUntilAtTurnBoundary(lead, cfg.turnSettleSeconds());
|
||||||
+ "after confirm() — refusing to send /clear at all; the calling lead's "
|
if (!turnResult.settled()) {
|
||||||
+ "own turn is still live and clearing it now would destroy live context "
|
log.warn("lead-rollover: pane {} did not reach a turn boundary (IDLE or DONE) after "
|
||||||
+ "(token={})",
|
+ "confirm() — the old pane is never touched; the calling lead's own "
|
||||||
lead, cfg.turnSettleSeconds(), p.token());
|
+ "turn is still live and tearing it down now would destroy live context "
|
||||||
|
+ "(token={}, configured={}s elapsed={}ms)",
|
||||||
|
lead, p.token(), cfg.turnSettleSeconds(), turnResult.elapsedMillis());
|
||||||
|
outcomes.put(p.token(), new RollStatus(RollState.TURN_NEVER_SETTLED,
|
||||||
|
"the calling lead's own turn never reached a boundary (IDLE or DONE) within "
|
||||||
|
+ "turnSettleSeconds=" + cfg.turnSettleSeconds() + "s (measured elapsed="
|
||||||
|
+ turnResult.elapsedMillis() + "ms) — the old pane was never touched. If "
|
||||||
|
+ "this keeps happening, raise turnSettleSeconds in fleetd.yaml"));
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
// This deliberately bypasses Injector, exactly like ClaudeCodeLauncher#clearContext:
|
// Captured once, here, and never re-resolved from `lead` again below: once the pane is
|
||||||
// /clear is housekeeping, not a delegated turn, and routing it through Injector wedges the
|
// closed there is nothing left for a terminal lookup to find.
|
||||||
// pane forever (see this class's javadoc).
|
Agent oldAgent = captureAgentWithRetry(lead);
|
||||||
agents.send(lead, "/clear");
|
String oldPaneId = oldAgent.paneId();
|
||||||
boolean clearSettled = waitForClearPickupAndSettle(lead, cfg.clearSettleSeconds());
|
String leadName = leadNameForTerminal.apply(lead);
|
||||||
if (!clearSettled) {
|
|
||||||
log.warn("lead-rollover: pane {} did not reach a turn boundary (IDLE or DONE) within {}s "
|
endOldSession(oldPaneId);
|
||||||
+ "after /clear — NOT sending bootstrapText (token={})",
|
DeathResult deathResult = waitUntilPaneGone(oldPaneId);
|
||||||
lead, cfg.clearSettleSeconds(), p.token());
|
if (!deathResult.gone()) {
|
||||||
|
log.warn("lead-rollover: old pane {} for lead {} was never confirmed gone after being "
|
||||||
|
+ "closed — not attempting a relaunch (token={}, timeout={}s "
|
||||||
|
+ "elapsed={}ms)",
|
||||||
|
oldPaneId, lead, p.token(), PANE_DEATH_TIMEOUT_SECONDS, deathResult.elapsedMillis());
|
||||||
|
outcomes.put(p.token(), new RollStatus(RollState.OLD_PANE_NEVER_DIED,
|
||||||
|
"the old pane was closed, but locatePane kept reporting it as still present "
|
||||||
|
+ "after a pane-death timeout=" + PANE_DEATH_TIMEOUT_SECONDS
|
||||||
|
+ "s (measured elapsed=" + deathResult.elapsedMillis() + "ms) — no "
|
||||||
|
+ "relaunch was attempted"));
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
agents.send(lead, cfg.bootstrapTextFor(p.handoverPath()));
|
|
||||||
log.info("lead-rollover: rolled token={} lead={}", p.token(), lead);
|
Agent newAgent = launcher.relaunch(leadName);
|
||||||
|
if (newAgent == null) {
|
||||||
|
log.warn("lead-rollover: relaunch of lead '{}' (old terminal {}) failed every attempt "
|
||||||
|
+ "— bootstrapText was never sent (token={})", leadName, lead, p.token());
|
||||||
|
outcomes.put(p.token(), new RollStatus(RollState.RELAUNCH_FAILED,
|
||||||
|
"lead '" + leadName + "' could not be relaunched — every attempt failed; "
|
||||||
|
+ "bootstrapText was never sent"));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
ReadinessResult readinessResult = waitUntilPaneReady(newAgent.terminalId(),
|
||||||
|
cfg.relaunchReadySeconds());
|
||||||
|
if (!readinessResult.ready()) {
|
||||||
|
log.warn("lead-rollover: fresh pane for lead '{}' (terminal {}) never reached a real "
|
||||||
|
+ "turn boundary — bootstrapText was never sent (token={}, configured={}s "
|
||||||
|
+ "elapsed={}ms)",
|
||||||
|
leadName, newAgent.terminalId(), p.token(), cfg.relaunchReadySeconds(),
|
||||||
|
readinessResult.elapsedMillis());
|
||||||
|
outcomes.put(p.token(), new RollStatus(RollState.RELAUNCH_NEVER_READY,
|
||||||
|
"fresh terminal " + newAgent.terminalId() + " never reached a real turn "
|
||||||
|
+ "boundary (IDLE or DONE) within relaunchReadySeconds="
|
||||||
|
+ cfg.relaunchReadySeconds() + "s (measured elapsed="
|
||||||
|
+ readinessResult.elapsedMillis() + "ms) — bootstrapText was never "
|
||||||
|
+ "sent"));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
IdentityResult identityResult = waitUntilRecognisedAsLead(newAgent.terminalId(),
|
||||||
|
cfg.relaunchReadySeconds());
|
||||||
|
agents.send(newAgent.terminalId(), cfg.bootstrapTextFor(p.handoverPath()));
|
||||||
|
if (!identityResult.ready()) {
|
||||||
|
log.warn("lead-rollover: fresh terminal {} for lead '{}' is alive and bootstrapped, but "
|
||||||
|
+ "was never recognised as a live lead — an operator should check why "
|
||||||
|
+ "the tab was not recognised (token={}, configured={}s elapsed={}ms)",
|
||||||
|
newAgent.terminalId(), leadName, p.token(), cfg.relaunchReadySeconds(),
|
||||||
|
identityResult.elapsedMillis());
|
||||||
|
outcomes.put(p.token(), new RollStatus(RollState.RELAUNCH_NOT_RECOGNISED,
|
||||||
|
"bootstrapText was sent to fresh terminal " + newAgent.terminalId() + ", but "
|
||||||
|
+ "it was never recognised as a live lead within relaunchReadySeconds="
|
||||||
|
+ cfg.relaunchReadySeconds() + "s (measured elapsed="
|
||||||
|
+ identityResult.elapsedMillis() + "ms) — check why the tab was not "
|
||||||
|
+ "recognised"));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
long rollElapsedMillis = nowMillis.getAsLong() - rollStartMillis;
|
||||||
|
log.info("lead-rollover: rolled token={} oldLead={} newTerminal={} elapsedMs={}",
|
||||||
|
p.token(), lead, newAgent.terminalId(), rollElapsedMillis);
|
||||||
|
outcomes.put(p.token(), new RollStatus(RollState.ROLLED,
|
||||||
|
"rolled successfully in " + rollElapsedMillis + "ms; new terminal="
|
||||||
|
+ newAgent.terminalId()));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** Attempts {@link #captureAgentWithRetry} makes before letting the failure propagate. */
|
||||||
|
static final int CAPTURE_RETRIES = 3;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* {@link AgentControl#get} for {@code lead}, retried up to {@link #CAPTURE_RETRIES} times. The
|
||||||
|
* terminal-to-pane lookup it goes through can report a genuinely live agent as not found (see
|
||||||
|
* {@code AgentControl#agentCall}'s own re-resolve-once behaviour), and one such false negative
|
||||||
|
* must not abort an otherwise-healthy roll. The result is captured once by the caller and never
|
||||||
|
* looked up again — see this class's javadoc.
|
||||||
|
*
|
||||||
|
* @throws RuntimeException the last failure, if every attempt fails — {@link #runRollover}'s
|
||||||
|
* catch turns that into {@link RollState#FAILED}
|
||||||
|
*/
|
||||||
|
private Agent captureAgentWithRetry(String lead) {
|
||||||
|
RuntimeException last = null;
|
||||||
|
for (int attempt = 1; attempt <= CAPTURE_RETRIES; attempt++) {
|
||||||
|
try {
|
||||||
|
return agents.get(lead);
|
||||||
|
} catch (RuntimeException e) {
|
||||||
|
last = e;
|
||||||
|
log.debug("lead-rollover: agents.get({}) failed on attempt {}/{}: {}",
|
||||||
|
lead, attempt, CAPTURE_RETRIES, e.toString());
|
||||||
|
if (attempt < CAPTURE_RETRIES) {
|
||||||
|
pollSleeper.run();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
throw last;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* End the old lead's session: close its pane, then close its tab only when the pane was that
|
||||||
|
* tab's sole occupant — the same pane-then-tab teardown {@code HerdrPeerLauncher#stop} uses for
|
||||||
|
* a member. An already-gone pane counts as success; any other {@code agents.close} failure
|
||||||
|
* propagates, so a genuinely failed teardown is never reported as done. A failing
|
||||||
|
* {@code spaces.closeTab} never propagates — by the time it runs the pane is already closed, so
|
||||||
|
* it is cosmetic tidying, not a real teardown failure.
|
||||||
|
*/
|
||||||
|
private void endOldSession(String paneId) {
|
||||||
|
WorkspaceControl.PaneLocation loc = spaces.locatePane(paneId);
|
||||||
|
try {
|
||||||
|
agents.close(paneId);
|
||||||
|
} catch (HerdrException e) {
|
||||||
|
if (!isAlreadyGone(e)) {
|
||||||
|
throw e;
|
||||||
|
}
|
||||||
|
log.debug("lead-rollover: pane.close({}) ignored — already gone: {}", paneId, e.getMessage());
|
||||||
|
}
|
||||||
|
if (loc != null && loc.tabPaneCount() == 1) {
|
||||||
|
try {
|
||||||
|
spaces.closeTab(loc.tabId());
|
||||||
|
} catch (RuntimeException e) {
|
||||||
|
log.warn("lead-rollover: tab.close({}) failed — the pane is already torn down, so "
|
||||||
|
+ "continuing; the tab may need manual cleanup: {}", loc.tabId(), e.getMessage());
|
||||||
|
}
|
||||||
|
} else if (loc != null) {
|
||||||
|
log.debug("lead-rollover: not closing tab {} — it holds {} panes (not a dedicated lead "
|
||||||
|
+ "tab)", loc.tabId(), loc.tabPaneCount());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** True when a herdr error means the target is already gone (safe to treat as done). */
|
||||||
|
private static boolean isAlreadyGone(HerdrException e) {
|
||||||
|
return e.code() != null && e.code().endsWith("_not_found");
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Poll {@link WorkspaceControl#locatePane} for {@code paneId} until it reports {@code null}
|
||||||
|
* (the pane is gone) or {@link #PANE_DEATH_TIMEOUT_SECONDS} elapses. Deliberately never calls
|
||||||
|
* {@link AgentControl#status} and never reads the live-lead terminal map — both answer a
|
||||||
|
* different question (whether an AGENT is live, not whether this PANE still exists) and
|
||||||
|
* {@code locatePane} alone catches a {@link HerdrException} from the underlying {@code
|
||||||
|
* pane.get} and turns it into {@code null} — see this class's javadoc.
|
||||||
|
*/
|
||||||
|
private DeathResult waitUntilPaneGone(String paneId) {
|
||||||
|
long startMillis = nowMillis.getAsLong();
|
||||||
|
long deadline = startMillis + TimeUnit.SECONDS.toMillis(PANE_DEATH_TIMEOUT_SECONDS);
|
||||||
|
while (nowMillis.getAsLong() < deadline) {
|
||||||
|
if (spaces.locatePane(paneId) == null) {
|
||||||
|
return new DeathResult(true, nowMillis.getAsLong() - startMillis);
|
||||||
|
}
|
||||||
|
pollSleeper.run();
|
||||||
|
}
|
||||||
|
return new DeathResult(false, nowMillis.getAsLong() - startMillis);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The measured outcome of {@link #waitUntilPaneGone}. */
|
||||||
|
private record DeathResult(boolean gone, long elapsedMillis) {}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Poll until {@code newTerminal}'s own pane reaches a real turn boundary ({@link
|
||||||
|
* AgentStatus#IDLE} or {@link AgentStatus#DONE}, never merely {@link AgentStatus#BLOCKED}) —
|
||||||
|
* the same exclusion {@link #waitUntilAtTurnBoundary} applies to the calling lead's own turn,
|
||||||
|
* applied here to the fresh one, so {@code bootstrapText} is never typed into a pane that has
|
||||||
|
* not actually finished booting — or {@code readySeconds} elapses. A failed status read
|
||||||
|
* degrades to "not yet ready" and is retried on the next poll.
|
||||||
|
*/
|
||||||
|
private ReadinessResult waitUntilPaneReady(String newTerminal, int readySeconds) {
|
||||||
|
long startMillis = nowMillis.getAsLong();
|
||||||
|
long deadline = startMillis + TimeUnit.SECONDS.toMillis(readySeconds);
|
||||||
|
while (nowMillis.getAsLong() < deadline) {
|
||||||
|
AgentStatus status;
|
||||||
|
try {
|
||||||
|
status = agents.status(newTerminal);
|
||||||
|
} catch (RuntimeException e) {
|
||||||
|
log.debug("lead-rollover: status check failed while waiting for {} to be ready: {}",
|
||||||
|
newTerminal, e.toString());
|
||||||
|
status = null;
|
||||||
|
}
|
||||||
|
if (status == AgentStatus.IDLE || status == AgentStatus.DONE) {
|
||||||
|
return new ReadinessResult(true, nowMillis.getAsLong() - startMillis);
|
||||||
|
}
|
||||||
|
pollSleeper.run();
|
||||||
|
}
|
||||||
|
return new ReadinessResult(false, nowMillis.getAsLong() - startMillis);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The measured outcome of {@link #waitUntilPaneReady}. */
|
||||||
|
private record ReadinessResult(boolean ready, long elapsedMillis) {}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Poll until {@code newTerminal} is present in {@link #liveLeadTerminals} or {@code
|
||||||
|
* readySeconds} elapses. This is bookkeeping, not a safety gate: the pane's own readiness (see
|
||||||
|
* {@link #waitUntilPaneReady}) is what decides whether {@code bootstrapText} is safe to send —
|
||||||
|
* a timeout here only means the daemon's own lead-discovery scan has not caught up yet.
|
||||||
|
*/
|
||||||
|
private IdentityResult waitUntilRecognisedAsLead(String newTerminal, int readySeconds) {
|
||||||
|
long startMillis = nowMillis.getAsLong();
|
||||||
|
long deadline = startMillis + TimeUnit.SECONDS.toMillis(readySeconds);
|
||||||
|
while (nowMillis.getAsLong() < deadline) {
|
||||||
|
if (liveLeadTerminals.get().containsKey(newTerminal)) {
|
||||||
|
return new IdentityResult(true, nowMillis.getAsLong() - startMillis);
|
||||||
|
}
|
||||||
|
pollSleeper.run();
|
||||||
|
}
|
||||||
|
return new IdentityResult(false, nowMillis.getAsLong() - startMillis);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The measured outcome of {@link #waitUntilRecognisedAsLead}. */
|
||||||
|
private record IdentityResult(boolean ready, long elapsedMillis) {}
|
||||||
|
|
||||||
/** Drop a pending request without rolling. @return whether a pending request existed for {@code token} */
|
/** Drop a pending request without rolling. @return whether a pending request existed for {@code token} */
|
||||||
public boolean cancel(String token) {
|
public boolean cancel(String token) {
|
||||||
return pending.remove(token) != null;
|
return pending.remove(token) != null;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Read-only: what is currently known about {@code token}. <strong>Never sends anything, never
|
||||||
|
* schedules, cancels, or retries a roll</strong> — a caller may poll this as often as it likes
|
||||||
|
* with no side effect at all, which is exactly why it exists: every failure past {@link
|
||||||
|
* #confirm} used to be a {@code log.warn} a lead can never read (see this class's javadoc), and
|
||||||
|
* this is the only route back.
|
||||||
|
*
|
||||||
|
* @param token the token {@link #open} returned; {@code null} or blank is a clean {@link
|
||||||
|
* RollState#UNKNOWN}, never a {@link NullPointerException} — {@link #pending} is a
|
||||||
|
* {@link ConcurrentHashMap}, which throws on a {@code null} key lookup, so this
|
||||||
|
* short-circuits before ever reaching it
|
||||||
|
* @return {@link RollState#IN_PROGRESS} for an approved roll whose continuation has not
|
||||||
|
* finished yet, or a terminal state once it has (both read from {@link #outcomes} —
|
||||||
|
* checked FIRST, see below); {@link RollState#PENDING} while {@code token} is still
|
||||||
|
* open and has not yet been approved (including one left pending by a {@link #confirm}
|
||||||
|
* gate refusal — see that method's javadoc); or {@link RollState#UNKNOWN} for a token
|
||||||
|
* never issued, cancelled, or aged out of the bounded history
|
||||||
|
*/
|
||||||
|
public RollStatus status(String token) {
|
||||||
|
if (token == null || token.isBlank()) {
|
||||||
|
return new RollStatus(RollState.UNKNOWN, "no token given");
|
||||||
|
}
|
||||||
|
// `outcomes` is checked BEFORE `pending`, deliberately: `confirm` writes an IN_PROGRESS
|
||||||
|
// entry into `outcomes` before it removes `token` from `pending` (see `confirm`'s own
|
||||||
|
// comment at that call site), so for the brief window where a token is present in BOTH
|
||||||
|
// maps, this order reports the more accurate answer (IN_PROGRESS, already approved) rather
|
||||||
|
// than the stale one (PENDING, not yet approved) a pending-first check would give.
|
||||||
|
RollStatus recorded = outcomes.get(token);
|
||||||
|
if (recorded != null) {
|
||||||
|
return recorded;
|
||||||
|
}
|
||||||
|
if (pending.containsKey(token)) {
|
||||||
|
return new RollStatus(RollState.PENDING, "open() has been called for this token and "
|
||||||
|
+ "it has not yet been confirmed — or a confirm() gate check failed and left it "
|
||||||
|
+ "pending, so the same token may be retried once the problem is fixed");
|
||||||
|
}
|
||||||
|
return new RollStatus(RollState.UNKNOWN, "token names no pending or finished rollover "
|
||||||
|
+ "request known to this instance — never issued, cancelled, or aged out of the "
|
||||||
|
+ "bounded history (cap=" + OUTCOME_HISTORY_CAP + ")");
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The three handover-file checks, in order: exists, not empty, fresh (modified after
|
* The three handover-file checks, in order: exists, not empty, fresh (modified after
|
||||||
* {@link #open}'s timestamp and not older than {@code maxDocAgeSeconds}). Stats {@code
|
* {@link #open}'s timestamp and not older than {@code maxDocAgeSeconds}). Stats {@code
|
||||||
@@ -454,15 +972,12 @@ public final class LeadRollover {
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* Poll {@link AgentControl#status} until {@code target} reports a real turn boundary — {@link
|
* Poll {@link AgentControl#status} until {@code target} reports a real turn boundary — {@link
|
||||||
* AgentStatus#IDLE} or {@link AgentStatus#DONE} — bounded by {@code settleSeconds}. Used once by
|
* AgentStatus#IDLE} or {@link AgentStatus#DONE} — bounded by {@code settleSeconds}. Used by
|
||||||
* {@link #runRollover}, to wait for the CALLING turn's own pane to settle before {@code /clear}
|
* {@link #runRollover} to wait for the CALLING turn's own pane to settle before the old pane is
|
||||||
* is ever sent at all — the {@code turnSettleSeconds} gate that makes this correction safe. The
|
* touched at all — the {@code turnSettleSeconds} gate that makes tearing it down safe. A failed
|
||||||
* SECOND wait, after {@code /clear}, is {@link #waitForClearPickupAndSettle} instead (fleetd
|
* status read degrades to "not yet settled" and is retried on the next poll, the same posture
|
||||||
* #489) — a plain boundary check is not enough there, because {@code /clear} starts no turn of
|
* {@code LeadHeartbeatLoop} and {@code HerdrPeerLauncher}'s readiness gate already take toward
|
||||||
* its own, so this method would (wrongly) report "settled" on its very first poll whether or not
|
* an unreadable status.
|
||||||
* {@code /clear} was actually picked up. A failed status read degrades to "not yet settled" and
|
|
||||||
* is retried on the next poll, the same posture {@code LeadHeartbeatLoop} and {@code
|
|
||||||
* HerdrPeerLauncher}'s readiness gate already take toward an unreadable status.
|
|
||||||
*
|
*
|
||||||
* <p><strong>Deliberately not {@link AgentStatus#injectable()}.</strong> {@code injectable()}
|
* <p><strong>Deliberately not {@link AgentStatus#injectable()}.</strong> {@code injectable()}
|
||||||
* answers the {@code Injector}'s question — "may I deliver a message without stepping on a live
|
* answers the {@code Injector}'s question — "may I deliver a message without stepping on a live
|
||||||
@@ -470,14 +985,20 @@ public final class LeadRollover {
|
|||||||
* an approval prompt is safe to queue a message behind. This class asks a stricter question —
|
* an approval prompt is safe to queue a message behind. This class asks a stricter question —
|
||||||
* "has the turn actually ended" — and {@code BLOCKED} answers no: it is a live turn that is
|
* "has the turn actually ended" — and {@code BLOCKED} answers no: it is a live turn that is
|
||||||
* merely paused, not one that has finished. Reusing {@code injectable()} here would let this
|
* merely paused, not one that has finished. Reusing {@code injectable()} here would let this
|
||||||
* wait fire {@code /clear} while the lead's own {@code confirm()}-calling turn is still live and
|
* wait tear the old pane down while the lead's own {@code confirm()}-calling turn is still live
|
||||||
* paused on a prompt — exactly the live-context-destroying failure the {@code turnSettleSeconds}
|
* and paused on a prompt — exactly the live-context-destroying failure {@code turnSettleSeconds}
|
||||||
* gate exists to prevent. Do not "simplify" this back to {@code injectable()}. ({@link
|
* exists to prevent. Do not "simplify" this back to {@code injectable()}. ({@link
|
||||||
* #waitForClearPickupAndSettle} keeps the same exclusion of {@code BLOCKED}, for the same
|
* #waitUntilPaneReady} applies the same exclusion of {@code BLOCKED} to the fresh lead's own
|
||||||
* reason, on the second wait.)
|
* turn.)
|
||||||
|
*
|
||||||
|
* @return a {@link TurnSettleResult} whose {@code settled()} is {@code true} once a real
|
||||||
|
* boundary was observed, {@code false} if {@code settleSeconds} elapses first.
|
||||||
|
* {@code elapsedMillis()} is a MEASURED value from the injected {@link #nowMillis}
|
||||||
|
* clock, never the configured {@code settleSeconds} budget.
|
||||||
*/
|
*/
|
||||||
private boolean waitUntilAtTurnBoundary(String target, int settleSeconds) {
|
private TurnSettleResult waitUntilAtTurnBoundary(String target, int settleSeconds) {
|
||||||
long deadline = nowMillis.getAsLong() + TimeUnit.SECONDS.toMillis(settleSeconds);
|
long startMillis = nowMillis.getAsLong();
|
||||||
|
long deadline = startMillis + TimeUnit.SECONDS.toMillis(settleSeconds);
|
||||||
while (nowMillis.getAsLong() < deadline) {
|
while (nowMillis.getAsLong() < deadline) {
|
||||||
AgentStatus status;
|
AgentStatus status;
|
||||||
try {
|
try {
|
||||||
@@ -488,101 +1009,13 @@ public final class LeadRollover {
|
|||||||
status = null;
|
status = null;
|
||||||
}
|
}
|
||||||
if (status == AgentStatus.IDLE || status == AgentStatus.DONE) {
|
if (status == AgentStatus.IDLE || status == AgentStatus.DONE) {
|
||||||
return true;
|
return new TurnSettleResult(true, nowMillis.getAsLong() - startMillis);
|
||||||
}
|
}
|
||||||
settleSleeper.run();
|
pollSleeper.run();
|
||||||
}
|
}
|
||||||
return false;
|
return new TurnSettleResult(false, nowMillis.getAsLong() - startMillis);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/** The measured outcome of {@link #waitUntilAtTurnBoundary}. */
|
||||||
* The SECOND wait in {@link #runRollover} — after {@code /clear} has been sent, waits for it to
|
private record TurnSettleResult(boolean settled, long elapsedMillis) {}
|
||||||
* settle, bounded by {@code settleSeconds}. <strong>fleetd #489 — the paste-race fix.</strong>
|
|
||||||
* {@code /clear} does not start a real turn of its own, so a pane with no submit race simply
|
|
||||||
* stays {@link AgentStatus#IDLE} the whole time: {@link #waitUntilAtTurnBoundary} would (wrongly)
|
|
||||||
* call that "settled" on its very first poll, whether or not the {@code /clear} Enter actually
|
|
||||||
* landed. That was Fault 1, measured live on 2026-09-12 — the second gate was a no-op, so a
|
|
||||||
* {@code bootstrapText} send followed immediately, racing Fault 2: {@link AgentControl#submit}'s
|
|
||||||
* own javadoc already records that the submit accompanying a delivery "can race the paste —
|
|
||||||
* especially right as the worker's TUI becomes interactive — leaving the text unsubmitted"
|
|
||||||
* (CB-113). Because {@code runRollover} deliberately bypasses {@code Injector} for {@code
|
|
||||||
* /clear} (see this class's javadoc), it inherited none of {@code Injector}'s nudging — so the
|
|
||||||
* lost {@code /clear} Enter sat in the input box and {@code bootstrapText} was typed right after
|
|
||||||
* it, landing as one concatenated line.
|
|
||||||
*
|
|
||||||
* <p>This method copies the pickup-nudge pattern {@link dev.ltms.fleet.inject.Injector} already
|
|
||||||
* ships for exactly this, on its own post-turn {@code /clear} housekeeping (fleetd #306; see
|
|
||||||
* {@code Injector.java:288-340} and {@code Injector.java:437-442}):
|
|
||||||
* <ul>
|
|
||||||
* <li>an {@link AgentStatus#WORKING} sample means {@code /clear} was picked up as a real
|
|
||||||
* turn;</li>
|
|
||||||
* <li>until that happens, each poll that still reports {@link AgentStatus#IDLE} or {@link
|
|
||||||
* AgentStatus#DONE} re-sends the submit keystroke ({@link AgentControl#submit}) to nudge
|
|
||||||
* the raced Enter — for the first {@code PICKUP_GRACE_POLLS - 1} of {@link
|
|
||||||
* #PICKUP_GRACE_POLLS} consecutive such polls (i.e. {@code PICKUP_GRACE_POLLS - 1}
|
|
||||||
* nudges: 7, not 8, given {@code PICKUP_GRACE_POLLS = 8}). A second Enter on an empty
|
|
||||||
* Claude Code prompt is a no-op, so repeating it is safe;</li>
|
|
||||||
* <li>the {@code PICKUP_GRACE_POLLS}th consecutive such poll, with {@code WORKING} still never
|
|
||||||
* observed, releases rather than wedges the roll instead of nudging again — the same
|
|
||||||
* choice {@code Injector} makes — and returns {@code true} anyway, logged at {@code info}
|
|
||||||
* so an operator can see which path ran;</li>
|
|
||||||
* <li>once {@code WORKING} has been observed, nudging stops and this instead waits for a real
|
|
||||||
* {@code working → IDLE/DONE} completion boundary before returning {@code true}.</li>
|
|
||||||
* </ul>
|
|
||||||
*
|
|
||||||
* <p><strong>{@link AgentStatus#BLOCKED} is deliberately excluded from both the nudge and the
|
|
||||||
* boundary check</strong> — the same reasoning as {@link #waitUntilAtTurnBoundary}'s own
|
|
||||||
* javadoc: a paused live turn is not a settled one, and re-sending Enter into an open approval
|
|
||||||
* prompt could wrongly answer it. A {@code BLOCKED} sample (or an unreadable/{@link
|
|
||||||
* AgentStatus#UNKNOWN} one) simply keeps this polling, with no nudge and no release, until either
|
|
||||||
* a real boundary is reached or {@code settleSeconds} runs out.
|
|
||||||
*
|
|
||||||
* <p>{@link AgentControl#submit} can itself throw; a {@link RuntimeException} from it is
|
|
||||||
* swallowed and logged at {@code debug}, exactly like {@code Injector.java:437-442} — a failed
|
|
||||||
* nudge must not abort the roll.
|
|
||||||
*
|
|
||||||
* @return {@code true} once {@code /clear} has settled, or once the nudge budget was exhausted
|
|
||||||
* with no pickup ever observed (released rather than wedged); {@code false} if {@code
|
|
||||||
* settleSeconds} elapses first — the caller must NOT send {@code bootstrapText} in that
|
|
||||||
* case, exactly as before this fix
|
|
||||||
*/
|
|
||||||
private boolean waitForClearPickupAndSettle(String target, int settleSeconds) {
|
|
||||||
long deadline = nowMillis.getAsLong() + TimeUnit.SECONDS.toMillis(settleSeconds);
|
|
||||||
boolean pickedUp = false; // a WORKING sample has been observed since /clear was sent
|
|
||||||
int idlePollsAwaitingPickup = 0;
|
|
||||||
while (nowMillis.getAsLong() < deadline) {
|
|
||||||
AgentStatus status;
|
|
||||||
try {
|
|
||||||
status = agents.status(target);
|
|
||||||
} catch (RuntimeException e) {
|
|
||||||
log.debug("lead-rollover: status check failed while waiting for {} to settle after "
|
|
||||||
+ "/clear: {}", target, e.toString());
|
|
||||||
status = null;
|
|
||||||
}
|
|
||||||
if (status == AgentStatus.WORKING) {
|
|
||||||
pickedUp = true;
|
|
||||||
} else if (status == AgentStatus.IDLE || status == AgentStatus.DONE) {
|
|
||||||
if (pickedUp) {
|
|
||||||
return true; // a real WORKING -> IDLE/DONE completion boundary
|
|
||||||
}
|
|
||||||
if (++idlePollsAwaitingPickup >= PICKUP_GRACE_POLLS) {
|
|
||||||
log.info("lead-rollover: /clear on {} was never observed as WORKING after {} "
|
|
||||||
+ "consecutive IDLE/DONE polls ({} of those were nudged) — "
|
|
||||||
+ "releasing rather than wedging the roll",
|
|
||||||
target, PICKUP_GRACE_POLLS, PICKUP_GRACE_POLLS - 1);
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
try {
|
|
||||||
agents.submit(target); // nudge a raced Enter (CB-113) so /clear actually submits
|
|
||||||
} catch (RuntimeException e) {
|
|
||||||
log.debug("lead-rollover: resubmit to {} failed (will retry next poll): {}",
|
|
||||||
target, e.getMessage());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// AgentStatus.BLOCKED or UNKNOWN (or an unreadable status, above): neither a pickup
|
|
||||||
// signal nor a boundary — keep polling without nudging or releasing.
|
|
||||||
settleSleeper.run();
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -32,10 +32,25 @@ public final class ConnectionIdentity {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The caller resolved from the connection: its worker {@code terminal} (or {@code null} for the
|
* The {@link PaneLocator} this identity resolves callers against — fleetd #612 CB-185: lets a
|
||||||
* primary / an off-host client) and its {@code pid} (or {@code -1} if not resolvable).
|
* test drive the exact {@link PaneLocator} a real assembly wired up (e.g. {@code
|
||||||
|
* FleetdAssembly}'s {@code new ConnectionIdentity(new PaneLocator(herdr, memberHerdr), ...)})
|
||||||
|
* directly with a chosen pid, bypassing the OS-dependent {@link PeerPidLookup} that {@link
|
||||||
|
* #resolve} otherwise goes through. A full HTTP round trip cannot exercise this: {@code
|
||||||
|
* LsofPeerPidLookup} excludes its own pid, and an in-process test client and server share one
|
||||||
|
* JVM pid, so {@code pidForLocalPort} always returns {@code -1} and {@link PaneLocator} never
|
||||||
|
* gets called at all.
|
||||||
*/
|
*/
|
||||||
public record Caller(String terminal, long pid) {
|
public PaneLocator panes() {
|
||||||
|
return panes;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The caller resolved from the connection: its worker {@code terminal} (or {@code null} for the
|
||||||
|
* primary / an off-host client), its {@code pid} (or {@code -1} if not resolvable), and whether
|
||||||
|
* the pane scan behind {@code terminal} ran to completion ({@link #scanComplete}).
|
||||||
|
*/
|
||||||
|
public record Caller(String terminal, long pid, boolean scanComplete) {
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Whether the OS peer-PID lookup actually succeeded — {@code false} means {@code pid} is
|
* Whether the OS peer-PID lookup actually succeeded — {@code false} means {@code pid} is
|
||||||
@@ -51,6 +66,10 @@ public final class ConnectionIdentity {
|
|||||||
* {@link ConnectionIdentity#isLoopback} is centralised rather than left for each caller to
|
* {@link ConnectionIdentity#isLoopback} is centralised rather than left for each caller to
|
||||||
* reimplement: a raw {@code pid > 0} check duplicated at every call site is precisely the
|
* reimplement: a raw {@code pid > 0} check duplicated at every call site is precisely the
|
||||||
* "one rule, two copies" shape that let #305 drift.
|
* "one rule, two copies" shape that let #305 drift.
|
||||||
|
*
|
||||||
|
* <p>This method is deliberately NOT widened for fleetd #505's failure (a herdr error
|
||||||
|
* during the pane scan, not a failed lsof lookup) — it still tests only the sentinel it is
|
||||||
|
* named for. #505 is a different axis, carried separately in {@link #scanComplete}.
|
||||||
*/
|
*/
|
||||||
public boolean resolved() {
|
public boolean resolved() {
|
||||||
return pid > 0;
|
return pid > 0;
|
||||||
@@ -60,10 +79,11 @@ public final class ConnectionIdentity {
|
|||||||
/** Resolve the caller's terminal and PID from one peer-PID lookup. */
|
/** Resolve the caller's terminal and PID from one peer-PID lookup. */
|
||||||
public Caller resolve(String remoteAddr, int remotePort) {
|
public Caller resolve(String remoteAddr, int remotePort) {
|
||||||
if (!isLoopback(remoteAddr)) {
|
if (!isLoopback(remoteAddr)) {
|
||||||
return new Caller(null, -1); // only same-host callers can be workers
|
return new Caller(null, -1, true); // only same-host callers can be workers
|
||||||
}
|
}
|
||||||
long pid = pids.pidForLocalPort(remotePort);
|
long pid = pids.pidForLocalPort(remotePort);
|
||||||
return new Caller(panes.terminalForPid(pid), pid);
|
PaneLocator.Lookup lookup = panes.terminalForPid(pid);
|
||||||
|
return new Caller(lookup.terminal(), pid, lookup.complete());
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -8,6 +8,7 @@ import dev.ltms.fleet.guard.SubscriptionGuard;
|
|||||||
import dev.ltms.fleet.herdr.Agent;
|
import dev.ltms.fleet.herdr.Agent;
|
||||||
import dev.ltms.fleet.herdr.AgentControl;
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
|
import dev.ltms.fleet.launch.ClaudeCodeArguments;
|
||||||
import dev.ltms.fleet.peer.Capability;
|
import dev.ltms.fleet.peer.Capability;
|
||||||
import dev.ltms.fleet.peer.PeerLauncher;
|
import dev.ltms.fleet.peer.PeerLauncher;
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
@@ -298,7 +299,7 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
|||||||
// has neither MCP nor a charter — session flags must be added into a list we own.
|
// has neither MCP nor a charter — session flags must be added into a list we own.
|
||||||
List<String> argv = mutableArgv(argvWithFleet(cfg, spec));
|
List<String> argv = mutableArgv(argvWithFleet(cfg, spec));
|
||||||
String agentSessionId = applySessionIdentity(argv, spec.sessionName(), spec.resumeSessionId());
|
String agentSessionId = applySessionIdentity(argv, spec.sessionName(), spec.resumeSessionId());
|
||||||
return new Launch(workerEnv, argvWithAutoCompact(argvWithModel(argv, cfg), cfg), agentSessionId);
|
return new Launch(workerEnv, ClaudeCodeArguments.withAutoCompactWindow(argvWithModel(argv, cfg), cfg), agentSessionId);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -908,30 +909,6 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
|||||||
return withModel;
|
return withModel;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Pin a bounded auto-compaction window on the command line via {@code --autocompact <tokens>},
|
|
||||||
* opt-in per profile (CB-634's sibling ticket: a member that runs out of context dies mid-turn
|
|
||||||
* and its {@code fleet_reply} — the whole point of the turn — is lost with it; opencode already
|
|
||||||
* forces {@code compaction.auto: true} unconditionally, CB-523, but Claude Code has no equivalent
|
|
||||||
* and runs at the backend's own default window).
|
|
||||||
*
|
|
||||||
* <p>Mirrors {@link #argvWithModel}: appended after it, so it survives the {@code ccs <profile>}
|
|
||||||
* wrapper the same way {@code --model} does, and outranks env/settings and the operator's own
|
|
||||||
* {@code argv}. Verified: {@code claude 2.1.241 --help} lists {@code --autocompact <auto|tokens>}
|
|
||||||
* (either the literal {@code auto}, or an integer 100k–1M) — {@link FleetConfig#load} rejects a
|
|
||||||
* configured value outside that band before this ever runs, so the flag Claude Code receives here
|
|
||||||
* is always in range.
|
|
||||||
*/
|
|
||||||
private static List<String> argvWithAutoCompact(List<String> argv, FleetConfig.Profile cfg) {
|
|
||||||
if (cfg.autoCompactWindow() == null) {
|
|
||||||
return argv;
|
|
||||||
}
|
|
||||||
List<String> withAutoCompact = mutableArgv(argv);
|
|
||||||
withAutoCompact.add("--autocompact");
|
|
||||||
withAutoCompact.add(String.valueOf(cfg.autoCompactWindow()));
|
|
||||||
return withAutoCompact;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
|
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
|
||||||
|
|
||||||
/** Spawn a worker for the default profile in the resolved default cwd. */
|
/** Spawn a worker for the default profile in the resolved default cwd. */
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ import dev.ltms.fleet.herdr.AgentControl;
|
|||||||
import dev.ltms.fleet.herdr.AgentStatus;
|
import dev.ltms.fleet.herdr.AgentStatus;
|
||||||
import dev.ltms.fleet.herdr.HerdrClient;
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
import dev.ltms.fleet.herdr.HerdrException;
|
import dev.ltms.fleet.herdr.HerdrException;
|
||||||
|
import dev.ltms.fleet.herdr.ResilientAgentLaunch;
|
||||||
import dev.ltms.fleet.herdr.Tab;
|
import dev.ltms.fleet.herdr.Tab;
|
||||||
import dev.ltms.fleet.herdr.Workspace;
|
import dev.ltms.fleet.herdr.Workspace;
|
||||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
@@ -67,16 +68,6 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
|||||||
|
|
||||||
private static final Logger log = LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
private static final Logger log = LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||||
|
|
||||||
/** herdr rejects a duplicate agent {@code name}; we retry a bumped name this many times. */
|
|
||||||
private static final int NAME_RETRIES = 8;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Retries for {@code agent.start} against a seed pane whose shell has not reached its prompt
|
|
||||||
* yet — {@code tab.create}/{@code pane.split} return as soon as the pane exists, and herdr
|
|
||||||
* refuses to start an agent in a pane that is not "an available shell" ({@code agent_pane_busy}).
|
|
||||||
*/
|
|
||||||
private static final int SHELL_READY_RETRIES = 20;
|
|
||||||
|
|
||||||
private final String namePrefix; // label prefix: naming + reap scheme
|
private final String namePrefix; // label prefix: naming + reap scheme
|
||||||
private final AgentControl agents;
|
private final AgentControl agents;
|
||||||
private final WorkspaceControl spaces;
|
private final WorkspaceControl spaces;
|
||||||
@@ -766,90 +757,26 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
|||||||
// Protocol 19 resolves the executable from the agent kind (== namePrefix here), so
|
// Protocol 19 resolves the executable from the agent kind (== namePrefix here), so
|
||||||
// argv[0] — the configured executable — is dropped and only the extra args are passed.
|
// argv[0] — the configured executable — is dropped and only the extra args are passed.
|
||||||
List<String> args = argv.isEmpty() ? argv : argv.subList(1, argv.size());
|
List<String> args = argv.isEmpty() ? argv : argv.subList(1, argv.size());
|
||||||
checkPaneCommandFits(cfg, argv);
|
try {
|
||||||
HerdrException last = null;
|
ResilientAgentLaunch.checkFits(cfg.profile(), argv);
|
||||||
for (int attempt = 0; attempt < NAME_RETRIES; attempt++) {
|
} catch (ResilientAgentLaunch.TooLargeException e) {
|
||||||
long seq = nameSeq.incrementAndGet();
|
throw new PeerUnreachableException(e.getMessage());
|
||||||
String name = namePrefix + "-" + cfg.profile() + "-" + nameNonce + "-" + seq;
|
|
||||||
try {
|
|
||||||
return new Started(startAwaitingShellPrompt(name, args, paneId), seq);
|
|
||||||
} catch (HerdrException e) {
|
|
||||||
if (!"agent_name_taken".equals(e.code())) throw e;
|
|
||||||
log.debug("peer name '{}' taken, retrying", name);
|
|
||||||
last = e;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
throw last;
|
long[] lastSeq = {0};
|
||||||
}
|
Agent agent = ResilientAgentLaunch.startUniquelyNamed(agents, namePrefix, args, paneId,
|
||||||
|
attempt -> {
|
||||||
/**
|
lastSeq[0] = nameSeq.incrementAndGet();
|
||||||
* fleetd #220: herdr does not exec the launch command — it TYPES it into the pane as one line,
|
return namePrefix + "-" + cfg.profile() + "-" + nameNonce + "-" + lastSeq[0];
|
||||||
* and a pty line buffer holds only {@value #PANE_COMMAND_BYTE_LIMIT} bytes (BSD/macOS {@code
|
},
|
||||||
* MAX_CANON}). Everything past that byte is dropped. Nothing reports it: herdr answers "agent
|
ResilientAgentLaunch.NAME_RETRIES, ResilientAgentLaunch.SHELL_READY_RETRIES, sleeper);
|
||||||
* started", the backend exits on the mangled argument it was handed, the pane closes, and the
|
return new Started(agent, lastSeq[0]);
|
||||||
* only symptom is {@link #waitUntilInjectableOrThrow} timing out 20 seconds later with no
|
|
||||||
* reason. That is exactly how #214 broke every claude-code spawn — one 50-byte flag pushed a
|
|
||||||
* 978-byte command to 1028, and the tail that got cut was {@code --autocompact 250000}.
|
|
||||||
*
|
|
||||||
* <p>So measure it here and refuse, loudly and immediately, rather than spawn something that
|
|
||||||
* cannot work. The estimate is deliberately conservative: fleetd cannot see herdr's quoting, so
|
|
||||||
* every argument is charged its own bytes plus a separator and a quote pair. An over-estimate
|
|
||||||
* costs a clear error at a length that was already unsafe; an under-estimate would let the
|
|
||||||
* silent truncation back in.
|
|
||||||
*
|
|
||||||
* @throws PeerUnreachableException when the command cannot fit — the same failure the spawn
|
|
||||||
* would have hit anyway, named at the point it is still
|
|
||||||
* explainable
|
|
||||||
*/
|
|
||||||
private void checkPaneCommandFits(FleetConfig.Profile cfg, List<String> argv) {
|
|
||||||
int bytes = 0;
|
|
||||||
String longest = null;
|
|
||||||
int longestBytes = 0;
|
|
||||||
for (String arg : argv) {
|
|
||||||
int argBytes = arg == null ? 0 : arg.getBytes(java.nio.charset.StandardCharsets.UTF_8).length;
|
|
||||||
bytes += argBytes + QUOTING_OVERHEAD_PER_ARG;
|
|
||||||
if (argBytes > longestBytes) {
|
|
||||||
longestBytes = argBytes;
|
|
||||||
longest = arg;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (bytes <= PANE_COMMAND_BYTE_LIMIT) {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
String culprit = longest == null ? "<none>"
|
|
||||||
: longest.substring(0, Math.min(longest.length(), 60)) + (longest.length() > 60 ? "…" : "");
|
|
||||||
throw new PeerUnreachableException(
|
|
||||||
"launch command for profile " + cfg.profile() + " is about " + bytes + " bytes, over the "
|
|
||||||
+ PANE_COMMAND_BYTE_LIMIT + "-byte limit of the pane line herdr types it into. "
|
|
||||||
+ "The pty would drop the tail silently and the backend would exit on a mangled "
|
|
||||||
+ "argument. Longest argument is " + longestBytes + " bytes: " + culprit
|
|
||||||
+ " — move it off the command line (a file flag) or shorten it.");
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The pty line buffer herdr types a launch command into: BSD/macOS {@code MAX_CANON}. Not a
|
* The pty line buffer herdr types a launch command into: BSD/macOS {@code MAX_CANON}. Not a
|
||||||
* fleetd choice and not configurable — see {@link #checkPaneCommandFits}.
|
* fleetd choice and not configurable — see {@link ResilientAgentLaunch#checkFits}.
|
||||||
*/
|
*/
|
||||||
static final int PANE_COMMAND_BYTE_LIMIT = 1024;
|
static final int PANE_COMMAND_BYTE_LIMIT = ResilientAgentLaunch.PANE_COMMAND_BYTE_LIMIT;
|
||||||
|
|
||||||
/** Per-argument allowance for the separating space and a shell quote pair fleetd cannot see. */
|
|
||||||
private static final int QUOTING_OVERHEAD_PER_ARG = 3;
|
|
||||||
|
|
||||||
/** Start the agent into {@code paneId}, waiting out the seed shell's boot with the sleeper. */
|
|
||||||
private Agent startAwaitingShellPrompt(String name, List<String> args, String paneId) {
|
|
||||||
HerdrException busy = null;
|
|
||||||
for (int attempt = 0; attempt < SHELL_READY_RETRIES; attempt++) {
|
|
||||||
try {
|
|
||||||
return agents.start(name, namePrefix, args, paneId);
|
|
||||||
} catch (HerdrException e) {
|
|
||||||
if (!"agent_pane_busy".equals(e.code())) throw e;
|
|
||||||
log.debug("pane {} not at its shell prompt yet, retrying agent.start", paneId);
|
|
||||||
busy = e;
|
|
||||||
sleeper.run();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
throw busy;
|
|
||||||
}
|
|
||||||
|
|
||||||
// --- discovery + reap ----------------------------------------------------------------------
|
// --- discovery + reap ----------------------------------------------------------------------
|
||||||
|
|
||||||
|
|||||||
@@ -246,7 +246,7 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
|||||||
* neither. So before this method existed with a requeue step, it dropped {@link #held}'s entries
|
* neither. So before this method existed with a requeue step, it dropped {@link #held}'s entries
|
||||||
* for {@code target} while the broker still considered them outstanding: never acked, never
|
* for {@code target} while the broker still considered them outstanding: never acked, never
|
||||||
* nacked, never requeued, and no longer reachable by {@link #peek} — permanently invisible. This
|
* nacked, never requeued, and no longer reachable by {@link #peek} — permanently invisible. This
|
||||||
* is unlike {@link #handleRecovery} and {@link #close()}, whose bare {@code held.clear()} is
|
* is unlike {@link RecoveryListener#handleRecovery(Recoverable)} and {@link #close()}, whose bare {@code held.clear()} is
|
||||||
* correct because each has already made the broker requeue (a real connection drop, or
|
* correct because each has already made the broker requeue (a real connection drop, or
|
||||||
* {@code channel.close()} respectively) before clearing local state.
|
* {@code channel.close()} respectively) before clearing local state.
|
||||||
*
|
*
|
||||||
|
|||||||
@@ -0,0 +1,24 @@
|
|||||||
|
package dev.ltms.fleet.msg;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 A-gaps (gap 1): a {@link LeadChannel} that its owner can also close.
|
||||||
|
*
|
||||||
|
* <p>{@link LeadChannel}'s own javadoc says plainly that {@code close()} is deliberately left out
|
||||||
|
* of that interface — draining is a caller convenience nobody uses, and closing is the
|
||||||
|
* <em>owner's</em> job. This interface is that owner's own, wider view: whoever opens the
|
||||||
|
* coordination mailbox (the assembly that builds the daemon) also needs to close it from the
|
||||||
|
* shutdown path, and a test standing in for a real broker connection needs a fake it can mark
|
||||||
|
* closed, without ever holding a live connection. Every ordinary consumer ({@code FleetMcp},
|
||||||
|
* {@link LeadCoordLoop}) keeps taking the narrower {@link LeadChannel} exactly as before — only
|
||||||
|
* the owner speaks this wider one.
|
||||||
|
*
|
||||||
|
* <p>{@link LeadMailbox} is still the only production implementation. This only generalises the
|
||||||
|
* TYPE its owner holds it as (previously the concrete class), so a test can substitute a fake
|
||||||
|
* closeable channel instead of a real AMQP connection.
|
||||||
|
*/
|
||||||
|
public interface LeadChannelHandle extends LeadChannel, AutoCloseable {
|
||||||
|
|
||||||
|
/** Release the underlying connection. Declared with no checked exception, unlike the plain {@link AutoCloseable#close()}. */
|
||||||
|
@Override
|
||||||
|
void close();
|
||||||
|
}
|
||||||
@@ -2,6 +2,7 @@ package dev.ltms.fleet.msg;
|
|||||||
|
|
||||||
import dev.ltms.fleet.herdr.AgentControl;
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
import dev.ltms.fleet.herdr.AgentStatus;
|
import dev.ltms.fleet.herdr.AgentStatus;
|
||||||
|
import dev.ltms.fleet.lead.LeadContextGauge;
|
||||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||||
import dev.ltms.fleet.metrics.Metrics;
|
import dev.ltms.fleet.metrics.Metrics;
|
||||||
@@ -13,6 +14,7 @@ import java.util.ArrayList;
|
|||||||
import java.util.List;
|
import java.util.List;
|
||||||
import java.util.concurrent.ScheduledExecutorService;
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
import java.util.concurrent.TimeUnit;
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.function.Function;
|
||||||
import java.util.function.LongSupplier;
|
import java.util.function.LongSupplier;
|
||||||
import java.util.function.Supplier;
|
import java.util.function.Supplier;
|
||||||
|
|
||||||
@@ -44,6 +46,15 @@ import java.util.function.Supplier;
|
|||||||
* loop stands down, so two competing injections never start two turns in the same pane
|
* loop stands down, so two competing injections never start two turns in the same pane
|
||||||
* (constraint 6).</li>
|
* (constraint 6).</li>
|
||||||
* </ol>
|
* </ol>
|
||||||
|
*
|
||||||
|
* <p><b>fleetd #609 — context-high notice.</b> Optionally ({@code contextHighNudge}, opt-in like the
|
||||||
|
* loop itself), a tick that finds the lead's own {@link LeadContextGauge} reading at {@link
|
||||||
|
* LeadContextGauge.State#HIGH} appends a text notice to whatever nudge it sends, telling the lead to
|
||||||
|
* consider {@code fleet_handover}. This is text only — it never rolls a pane itself. It fires once per
|
||||||
|
* HIGH stretch (a latch, cleared only by a later {@code OK} reading — {@code UNKNOWN} neither sets nor
|
||||||
|
* clears it, since "I could not look" must not be read as "it got better"), and it never spends the
|
||||||
|
* quiet-nudge budget: an idle, quiet, HIGH-context lead is exactly the case {@link Action#QUIET_DONE}
|
||||||
|
* would otherwise swallow, and it is the one case most worth interrupting the quiet cap for.
|
||||||
*/
|
*/
|
||||||
public final class LeadHeartbeatLoop {
|
public final class LeadHeartbeatLoop {
|
||||||
|
|
||||||
@@ -63,11 +74,16 @@ public final class LeadHeartbeatLoop {
|
|||||||
private final long backoffMs;
|
private final long backoffMs;
|
||||||
private final int quietNudgeCap;
|
private final int quietNudgeCap;
|
||||||
private final Metrics metrics; // CB-512 pattern: nullable — no registry in unit tests
|
private final Metrics metrics; // CB-512 pattern: nullable — no registry in unit tests
|
||||||
|
private final LeadContextSource contextSource; // fleetd #609
|
||||||
|
private final boolean contextHighNudge; // fleetd #609: opt-in, like the loop itself
|
||||||
|
private final boolean requireOperatorConfirm; // fleetd #621: mirrors leadRollover.requireOperatorConfirm
|
||||||
|
|
||||||
/** When the current idle stretch began (nanos), or {@link #NOT_IDLE}. Single scheduler thread only. */
|
/** When the current idle stretch began (nanos), or {@link #NOT_IDLE}. Single scheduler thread only. */
|
||||||
private long idleSinceNanos = NOT_IDLE;
|
private long idleSinceNanos = NOT_IDLE;
|
||||||
/** Consecutive nudges that found no pending fleet state. Single scheduler thread only. */
|
/** Consecutive nudges that found no pending fleet state. Single scheduler thread only. */
|
||||||
private int quietCount = 0;
|
private int quietCount = 0;
|
||||||
|
/** fleetd #609: latched "already told this lead about this HIGH stretch". Single scheduler thread only. */
|
||||||
|
private boolean contextNotified = false;
|
||||||
|
|
||||||
/** Constructor with an injectable clock and no metric registry (unit tests, or wiring that opts out). */
|
/** Constructor with an injectable clock and no metric registry (unit tests, or wiring that opts out). */
|
||||||
public LeadHeartbeatLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
public LeadHeartbeatLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||||
@@ -75,7 +91,7 @@ public final class LeadHeartbeatLoop {
|
|||||||
ScheduledExecutorService scheduler, LongSupplier clock,
|
ScheduledExecutorService scheduler, LongSupplier clock,
|
||||||
long idleAfterNanos, long backoffMs, int quietNudgeCap) {
|
long idleAfterNanos, long backoffMs, int quietNudgeCap) {
|
||||||
this(primaryRegistry, agents, inbox, roster, pushLoop, scheduler, clock,
|
this(primaryRegistry, agents, inbox, roster, pushLoop, scheduler, clock,
|
||||||
idleAfterNanos, backoffMs, quietNudgeCap, null);
|
idleAfterNanos, backoffMs, quietNudgeCap, null, LeadContextSource.none(), false);
|
||||||
}
|
}
|
||||||
|
|
||||||
/** As above, with a metric registry (the CB-512 pattern) so nudge outcomes are counted. */
|
/** As above, with a metric registry (the CB-512 pattern) so nudge outcomes are counted. */
|
||||||
@@ -83,6 +99,40 @@ public final class LeadHeartbeatLoop {
|
|||||||
Supplier<List<MemberSession>> roster, ReplyPushLoop pushLoop,
|
Supplier<List<MemberSession>> roster, ReplyPushLoop pushLoop,
|
||||||
ScheduledExecutorService scheduler, LongSupplier clock,
|
ScheduledExecutorService scheduler, LongSupplier clock,
|
||||||
long idleAfterNanos, long backoffMs, int quietNudgeCap, Metrics metrics) {
|
long idleAfterNanos, long backoffMs, int quietNudgeCap, Metrics metrics) {
|
||||||
|
this(primaryRegistry, agents, inbox, roster, pushLoop, scheduler, clock,
|
||||||
|
idleAfterNanos, backoffMs, quietNudgeCap, metrics, LeadContextSource.none(), false);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #609: as above, plus the lead's own context source and whether a HIGH reading should
|
||||||
|
* append a hand-over notice to the loop's nudge. Pass {@link LeadContextSource#none()} and
|
||||||
|
* {@code false} to keep the pre-#609 behaviour exactly (both existing public constructors do).
|
||||||
|
*
|
||||||
|
* <p>fleetd #621: delegates to the full constructor with {@code requireOperatorConfirm=true} —
|
||||||
|
* the pre-#621 wording ("ask the operator ... only the operator can approve the roll") assumed
|
||||||
|
* the config default, so every caller of this overload keeps that text byte-identical.
|
||||||
|
*/
|
||||||
|
public LeadHeartbeatLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||||
|
Supplier<List<MemberSession>> roster, ReplyPushLoop pushLoop,
|
||||||
|
ScheduledExecutorService scheduler, LongSupplier clock,
|
||||||
|
long idleAfterNanos, long backoffMs, int quietNudgeCap, Metrics metrics,
|
||||||
|
LeadContextSource contextSource, boolean contextHighNudge) {
|
||||||
|
this(primaryRegistry, agents, inbox, roster, pushLoop, scheduler, clock,
|
||||||
|
idleAfterNanos, backoffMs, quietNudgeCap, metrics, contextSource, contextHighNudge, true);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #621: as above, plus the daemon's effective {@code leadRollover.requireOperatorConfirm}
|
||||||
|
* value — threaded into {@link #contextNotice(boolean, LeadContextGauge.Reading, boolean, boolean)}
|
||||||
|
* so the notice's wording tracks the config the daemon actually enforces (see {@code
|
||||||
|
* LeadRollover.confirm}) instead of always asserting the operator gate is on.
|
||||||
|
*/
|
||||||
|
public LeadHeartbeatLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||||
|
Supplier<List<MemberSession>> roster, ReplyPushLoop pushLoop,
|
||||||
|
ScheduledExecutorService scheduler, LongSupplier clock,
|
||||||
|
long idleAfterNanos, long backoffMs, int quietNudgeCap, Metrics metrics,
|
||||||
|
LeadContextSource contextSource, boolean contextHighNudge,
|
||||||
|
boolean requireOperatorConfirm) {
|
||||||
this.primaryRegistry = primaryRegistry;
|
this.primaryRegistry = primaryRegistry;
|
||||||
this.agents = agents;
|
this.agents = agents;
|
||||||
this.inbox = inbox;
|
this.inbox = inbox;
|
||||||
@@ -94,6 +144,20 @@ public final class LeadHeartbeatLoop {
|
|||||||
this.backoffMs = backoffMs;
|
this.backoffMs = backoffMs;
|
||||||
this.quietNudgeCap = quietNudgeCap;
|
this.quietNudgeCap = quietNudgeCap;
|
||||||
this.metrics = metrics;
|
this.metrics = metrics;
|
||||||
|
this.contextSource = contextSource;
|
||||||
|
this.contextHighNudge = contextHighNudge;
|
||||||
|
this.requireOperatorConfirm = requireOperatorConfirm;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #609: one lead's own context reading, keyed by its terminal id — the same injected-source
|
||||||
|
* idiom {@code FleetMcp.LeadSeatSource}/{@code FleetMcp.LeadConfigDirSource} already use.
|
||||||
|
*/
|
||||||
|
public record LeadContextSource(Function<String, LeadContextGauge.Reading> readingFor) {
|
||||||
|
/** Inert source — every lead reads UNKNOWN, so the context notice can never fire. */
|
||||||
|
public static LeadContextSource none() {
|
||||||
|
return new LeadContextSource(_ -> LeadContextGauge.Reading.unknown());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -126,8 +190,11 @@ public final class LeadHeartbeatLoop {
|
|||||||
STAND_DOWN
|
STAND_DOWN
|
||||||
}
|
}
|
||||||
|
|
||||||
/** The outcome of one decision: the action plus the state to persist for the next tick. */
|
/**
|
||||||
record Decision(Action action, Long idleSinceNanos, int quietCount) {}
|
* The outcome of one decision: the action, the state to persist for the next tick, and (fleetd
|
||||||
|
* #609) whether the lead has now been told about the current HIGH context stretch.
|
||||||
|
*/
|
||||||
|
record Decision(Action action, Long idleSinceNanos, int quietCount, boolean contextNotified) {}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Pure decision function: given the current loop state and fleet/lead facts, return what to do
|
* Pure decision function: given the current loop state and fleet/lead facts, return what to do
|
||||||
@@ -142,34 +209,47 @@ public final class LeadHeartbeatLoop {
|
|||||||
* @param pushLoopActive whether {@link ReplyPushLoop} is currently nudging some target (constraint 6)
|
* @param pushLoopActive whether {@link ReplyPushLoop} is currently nudging some target (constraint 6)
|
||||||
* @param leadKnown whether a lead terminal is known to nudge at all
|
* @param leadKnown whether a lead terminal is known to nudge at all
|
||||||
* @param fleet a snapshot of the pending fleet state (constraint 5)
|
* @param fleet a snapshot of the pending fleet state (constraint 5)
|
||||||
|
* @param context fleetd #609: the lead's own {@link LeadContextGauge} reading for this tick
|
||||||
|
* @param contextNotified fleetd #609: whether the lead has already been told about the current HIGH
|
||||||
|
* stretch — a latch, carried forward by {@link #applyDecision}
|
||||||
* @return the action to take and the state to persist
|
* @return the action to take and the state to persist
|
||||||
*/
|
*/
|
||||||
Decision decide(long nowNanos, Long idleSinceNanos, int quietCount, AgentStatus status,
|
Decision decide(long nowNanos, Long idleSinceNanos, int quietCount, AgentStatus status,
|
||||||
boolean pushLoopActive, boolean leadKnown, FleetState fleet) {
|
boolean pushLoopActive, boolean leadKnown, FleetState fleet,
|
||||||
|
LeadContextGauge.State context, boolean contextNotified) {
|
||||||
|
// fleetd #609: re-arm the latch only on a positive OK reading. UNKNOWN means "I could not
|
||||||
|
// look", not "it got better" — re-arming on UNKNOWN would let a flapping gauge (a transcript
|
||||||
|
// read that misses one tick) nudge a full lead again on every recovery, defeating the "once
|
||||||
|
// per HIGH stretch" promise. Computed once, up front, so every gate below carries it forward
|
||||||
|
// unchanged unless it is the gate that actually discharges it.
|
||||||
|
boolean latch = context == LeadContextGauge.State.OK ? false : contextNotified;
|
||||||
|
boolean contextHigh = contextHighNudge && context == LeadContextGauge.State.HIGH;
|
||||||
|
|
||||||
// Constraint 6: while ReplyPushLoop is actively nudging the lead, injecting a second,
|
// Constraint 6: while ReplyPushLoop is actively nudging the lead, injecting a second,
|
||||||
// competing prompt into the same pane would start a second turn — racing loops multiply
|
// competing prompt into the same pane would start a second turn — racing loops multiply
|
||||||
// turns and context burn. Stand aside, and treat the active push as real state (re-arm the
|
// turns and context burn. Stand aside, and treat the active push as real state (re-arm the
|
||||||
// quiet counter), because the reply that drove it is exactly the kind of new state that
|
// quiet counter), because the reply that drove it is exactly the kind of new state that
|
||||||
// should reset the cap.
|
// should reset the cap. The context latch is untouched: standing down must not spend the
|
||||||
|
// one notice this stretch gets.
|
||||||
if (pushLoopActive) {
|
if (pushLoopActive) {
|
||||||
return new Decision(Action.STAND_DOWN, idleSinceNanos, 0);
|
return new Decision(Action.STAND_DOWN, idleSinceNanos, 0, latch);
|
||||||
}
|
}
|
||||||
// Constraint 2: a WORKING lead is making progress and must NOT be touched; an unreadable
|
// Constraint 2: a WORKING lead is making progress and must NOT be touched; an unreadable
|
||||||
// status (read failure, or the agent is gone) is safest treated the same way — never inject
|
// status (read failure, or the agent is gone) is safest treated the same way — never inject
|
||||||
// into a state we cannot read. Either way, reset the idle window and the quiet counter: the
|
// into a state we cannot read. Either way, reset the idle window and the quiet counter: the
|
||||||
// lead was / may be active, so the next idle stretch must count its own quiet period fresh.
|
// lead was / may be active, so the next idle stretch must count its own quiet period fresh.
|
||||||
if (status == null || !status.injectable()) {
|
if (status == null || !status.injectable()) {
|
||||||
return new Decision(Action.LEAD_BUSY, null, 0);
|
return new Decision(Action.LEAD_BUSY, null, 0, latch);
|
||||||
}
|
}
|
||||||
if (idleSinceNanos == null) {
|
if (idleSinceNanos == null) {
|
||||||
// The lead just became injectable — record the start of an idle stretch and wait out the
|
// The lead just became injectable — record the start of an idle stretch and wait out the
|
||||||
// debounce quiet period before ever nudging (constraint 3).
|
// debounce quiet period before ever nudging (constraint 3).
|
||||||
return new Decision(Action.WAIT_IDLE, nowNanos, quietCount);
|
return new Decision(Action.WAIT_IDLE, nowNanos, quietCount, latch);
|
||||||
}
|
}
|
||||||
if (nowNanos - idleSinceNanos < idleAfterNanos) {
|
if (nowNanos - idleSinceNanos < idleAfterNanos) {
|
||||||
// Still within the quiet period: the lead that just finished a turn sits momentarily idle
|
// Still within the quiet period: the lead that just finished a turn sits momentarily idle
|
||||||
// and must not be re-prompted into every natural pause.
|
// and must not be re-prompted into every natural pause.
|
||||||
return new Decision(Action.WAIT_IDLE, idleSinceNanos, quietCount);
|
return new Decision(Action.WAIT_IDLE, idleSinceNanos, quietCount, latch);
|
||||||
}
|
}
|
||||||
// Past the quiet period with an injectable lead: it is a genuine candidate for a nudge. Two
|
// Past the quiet period with an injectable lead: it is a genuine candidate for a nudge. Two
|
||||||
// gating facts decide whether and how:
|
// gating facts decide whether and how:
|
||||||
@@ -177,23 +257,33 @@ public final class LeadHeartbeatLoop {
|
|||||||
// No lead terminal is known yet (e.g. the registry has not learned one) — there is nobody
|
// No lead terminal is known yet (e.g. the registry has not learned one) — there is nobody
|
||||||
// to nudge. Keep waiting; the window stays open so discovery re-arms it without a fresh
|
// to nudge. Keep waiting; the window stays open so discovery re-arms it without a fresh
|
||||||
// quiet period.
|
// quiet period.
|
||||||
return new Decision(Action.WAIT_IDLE, idleSinceNanos, quietCount);
|
return new Decision(Action.WAIT_IDLE, idleSinceNanos, quietCount, latch);
|
||||||
}
|
}
|
||||||
if (fleet.hasPending()) {
|
if (fleet.hasPending()) {
|
||||||
// Real fleet state is waiting — a worker reply or a DONE session. This is new state, so
|
// Real fleet state is waiting — a worker reply or a DONE session. This is new state, so
|
||||||
// it resets the quiet counter (constraint 4) and the lead is nudged to go collect it.
|
// it resets the quiet counter (constraint 4) and the lead is nudged to go collect it. The
|
||||||
return new Decision(Action.INJECT, idleSinceNanos, 0);
|
// nudge text carries the context notice too when contextHigh — see injectNudge/contextNotice
|
||||||
|
// — so this route discharges the same duty and must set the latch.
|
||||||
|
return new Decision(Action.INJECT, idleSinceNanos, 0, latch || contextHigh);
|
||||||
|
}
|
||||||
|
if (contextHigh && !latch) {
|
||||||
|
// fleetd #609: the lead is idle, its context is full, and nothing is pending. This is the
|
||||||
|
// one case the quiet cap would otherwise swallow, and it is exactly when the lead most
|
||||||
|
// needs to hear it. Fire once per HIGH stretch, and do NOT spend the quiet budget on it:
|
||||||
|
// this is an event notice, not a "are you still there" nudge.
|
||||||
|
return new Decision(Action.INJECT, idleSinceNanos, quietCount, true);
|
||||||
}
|
}
|
||||||
if (quietCount < quietNudgeCap) {
|
if (quietCount < quietNudgeCap) {
|
||||||
// Nothing is pending, but the cap is not exhausted: nudge anyway, telling the lead
|
// Nothing is pending, but the cap is not exhausted: nudge anyway, telling the lead
|
||||||
// exactly that nothing is waiting so it can choose to stand down rather than hunt
|
// exactly that nothing is waiting so it can choose to stand down rather than hunt
|
||||||
// (constraint 5). Count it toward the consecutive-quiet cap.
|
// (constraint 5). Count it toward the consecutive-quiet cap. This nudge also carries the
|
||||||
return new Decision(Action.INJECT, idleSinceNanos, quietCount + 1);
|
// context notice when contextHigh (already latched above, or being latched now).
|
||||||
|
return new Decision(Action.INJECT, idleSinceNanos, quietCount + 1, latch || contextHigh);
|
||||||
}
|
}
|
||||||
// Nothing pending and the cap is exhausted: stop nudging until real state appears again
|
// Nothing pending and the cap is exhausted: stop nudging until real state appears again
|
||||||
// (constraint 4). The loop still ticks on backoff so a genuinely new reply or session change
|
// (constraint 4). The loop still ticks on backoff so a genuinely new reply or session change
|
||||||
// re-arms it — QUIET_DONE stops injection, not observation.
|
// re-arms it — QUIET_DONE stops injection, not observation.
|
||||||
return new Decision(Action.QUIET_DONE, idleSinceNanos, quietCount);
|
return new Decision(Action.QUIET_DONE, idleSinceNanos, quietCount, latch);
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- loop ----------------------------------------------------------------------------------
|
// --- loop ----------------------------------------------------------------------------------
|
||||||
@@ -203,62 +293,195 @@ public final class LeadHeartbeatLoop {
|
|||||||
* does not evaluate the lead's idle state before the fleet has settled.
|
* does not evaluate the lead's idle state before the fleet has settled.
|
||||||
*/
|
*/
|
||||||
public void start() {
|
public void start() {
|
||||||
log.info("idle-lead heartbeat: on — nudge lead after {}s idle (recheck {}ms, quiet cap {})",
|
// fleetd #613: contextHighNudge added alongside the three settings already here — an
|
||||||
TimeUnit.NANOSECONDS.toSeconds(idleAfterNanos), backoffMs, quietNudgeCap);
|
// operator otherwise cannot tell from the boot log whether the #609 handover notice is
|
||||||
|
// armed, and had to load the deployed jar's config to confirm it.
|
||||||
|
log.info("idle-lead heartbeat: on — nudge lead after {}s idle (recheck {}ms, quiet cap {}, "
|
||||||
|
+ "context-high nudge {})",
|
||||||
|
TimeUnit.NANOSECONDS.toSeconds(idleAfterNanos), backoffMs, quietNudgeCap,
|
||||||
|
contextHighNudge);
|
||||||
scheduler.schedule(this::tick, backoffMs, TimeUnit.MILLISECONDS);
|
scheduler.schedule(this::tick, backoffMs, TimeUnit.MILLISECONDS);
|
||||||
}
|
}
|
||||||
|
|
||||||
/** One loop tick, every {@link #backoffMs} — the thin scheduler around {@link #decide}. */
|
/** One loop tick, every {@link #backoffMs} — the thin scheduler around {@link #decide}. Package-private
|
||||||
private void tick() {
|
* (mirroring {@link ReplyPushLoop#tick(String)}) so tests can drive it directly with a fake clock and a
|
||||||
|
* fake {@link AgentControl} instead of racing the scheduler thread. */
|
||||||
|
void tick() {
|
||||||
boolean leadKnown = primaryRegistry.primaryTerminal().isPresent();
|
boolean leadKnown = primaryRegistry.primaryTerminal().isPresent();
|
||||||
FleetState fleet = snapshot(inbox, roster);
|
FleetState fleet = snapshot(inbox, roster);
|
||||||
AgentStatus status = AgentStatus.UNKNOWN;
|
AgentStatus status = AgentStatus.UNKNOWN;
|
||||||
|
LeadContextGauge.Reading reading = LeadContextGauge.Reading.unknown();
|
||||||
if (leadKnown) {
|
if (leadKnown) {
|
||||||
|
String leadTerminal = primaryRegistry.primaryTerminal().orElseThrow();
|
||||||
try {
|
try {
|
||||||
status = agents.status(primaryRegistry.primaryTerminal().orElseThrow());
|
status = agents.status(leadTerminal);
|
||||||
} catch (RuntimeException e) {
|
} catch (RuntimeException e) {
|
||||||
// A failed status read degrades to "unknown" — decide() treats that like a busy lead
|
// A failed status read degrades to "unknown" — decide() treats that like a busy lead
|
||||||
// and never injects into a state it cannot read. Retry on the next backoff.
|
// and never injects into a state it cannot read. Retry on the next backoff.
|
||||||
log.debug("idle-heartbeat: status check failed for lead, will retry: {}", e.toString());
|
log.debug("idle-heartbeat: status check failed for lead, will retry: {}", e.toString());
|
||||||
}
|
}
|
||||||
|
// fleetd #609: read the lead's own context regardless of status — decide() still gates on
|
||||||
|
// status first (constraint 2), so this is harmless work on a WORKING lead and lets the
|
||||||
|
// latch state stay accurate for whenever the lead does go idle.
|
||||||
|
reading = contextSource.readingFor().apply(leadTerminal);
|
||||||
}
|
}
|
||||||
|
|
||||||
Decision d = decide(clock.getAsLong(),
|
Decision d = decide(clock.getAsLong(),
|
||||||
idleSinceNanos == NOT_IDLE ? null : idleSinceNanos,
|
idleSinceNanos == NOT_IDLE ? null : idleSinceNanos,
|
||||||
quietCount, status, pushLoop.isActive(), leadKnown, fleet);
|
quietCount, status, pushLoop.isActive(), leadKnown, fleet,
|
||||||
|
reading.state(), contextNotified);
|
||||||
applyDecision(d);
|
applyDecision(d);
|
||||||
switch (d.action()) {
|
switch (d.action()) {
|
||||||
case INJECT -> injectNudge(fleet);
|
case INJECT -> injectNudge(d, fleet, reading);
|
||||||
case QUIET_DONE -> countNudge("exhausted");
|
case QUIET_DONE -> {
|
||||||
case WAIT_IDLE, LEAD_BUSY, STAND_DOWN -> { /* nothing to inject, nothing to count */ }
|
countNudge("exhausted");
|
||||||
|
contextNotified = d.contextNotified();
|
||||||
|
}
|
||||||
|
case WAIT_IDLE, LEAD_BUSY, STAND_DOWN -> contextNotified = d.contextNotified();
|
||||||
}
|
}
|
||||||
scheduleNext();
|
scheduleNext();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Persist the state a decision returned, so the next tick starts from it. */
|
/**
|
||||||
|
* Persist the idle/quiet state a decision returned, so the next tick starts from it.
|
||||||
|
*
|
||||||
|
* <p>fleetd #609 review: the context latch ({@link #contextNotified}) is deliberately <em>not</em>
|
||||||
|
* set here any more. Setting it from the decision unconditionally — before {@link #injectNudge} even
|
||||||
|
* tries to send — is exactly the review's blocker: a decision to notify is not the same fact as "the
|
||||||
|
* notice reached the pane". Every branch of {@link #tick} now assigns {@link #contextNotified} itself,
|
||||||
|
* once it knows whether a send happened and whether it carried the notice (see {@link #injectNudge}).
|
||||||
|
*/
|
||||||
private void applyDecision(Decision d) {
|
private void applyDecision(Decision d) {
|
||||||
idleSinceNanos = d.idleSinceNanos() == null ? NOT_IDLE : d.idleSinceNanos();
|
idleSinceNanos = d.idleSinceNanos() == null ? NOT_IDLE : d.idleSinceNanos();
|
||||||
quietCount = d.quietCount();
|
quietCount = d.quietCount();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Send the nudge to the known lead. */
|
/**
|
||||||
private void injectNudge(FleetState fleet) {
|
* Send the nudge to the known lead, with the fleetd #609 context notice appended when it applies, and
|
||||||
|
* persist the context latch based on what actually happened this tick — not merely what {@code d}
|
||||||
|
* chose to attempt.
|
||||||
|
*/
|
||||||
|
private void injectNudge(Decision d, FleetState fleet, LeadContextGauge.Reading reading) {
|
||||||
|
// fleetd #609 review: build the notice from the latch as it stood BEFORE this tick's decision —
|
||||||
|
// d.contextNotified() is the value to persist once delivery is confirmed, not the value the text
|
||||||
|
// itself should be built from. Otherwise a HIGH stretch that is still latched would never see the
|
||||||
|
// notice at all, defeating the very check this fixes.
|
||||||
|
String notice = contextNotice(contextHighNudge, reading, contextNotified, requireOperatorConfirm);
|
||||||
var lead = primaryRegistry.primaryTerminal();
|
var lead = primaryRegistry.primaryTerminal();
|
||||||
if (lead.isEmpty()) {
|
boolean sent = lead.isPresent() && trySend(lead.get(), fleet.nudgeText() + notice, notice);
|
||||||
return; // the lead disappeared between the decision and the injection
|
// The latch becomes true only when all three hold: decide() chose to notify, a notice was
|
||||||
}
|
// actually included in the text, and the send reached the pane without throwing. Whenever no
|
||||||
String leadTerminal = lead.get();
|
// notice was attempted (disabled, not HIGH, or already latched), nothing was promised to the lead
|
||||||
|
// this tick, so apply the decision's own carried-forward value unconditionally — that is how the
|
||||||
|
// OK-only re-arm rule and STAND_DOWN's "don't burn the notice" rule keep working through this path
|
||||||
|
// too. A lead that disappeared between the decision and the send (lead.isEmpty()) is treated the
|
||||||
|
// same as a failed send: nothing reached the pane, so the latch must not be set.
|
||||||
|
contextNotified = notice.isEmpty() ? d.contextNotified() : sent;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Attempt one herdr send and count its outcome. Returns whether {@code agents.send} returned without
|
||||||
|
* throwing — the caller ({@link #injectNudge}) needs this to decide whether the fleetd #609 context
|
||||||
|
* latch may be persisted as set.
|
||||||
|
*/
|
||||||
|
private boolean trySend(String leadTerminal, String text, String notice) {
|
||||||
try {
|
try {
|
||||||
agents.send(leadTerminal, fleet.nudgeText());
|
agents.send(leadTerminal, text);
|
||||||
log.debug("idle-heartbeat: nudge sent to lead {} (quiet nudges so far in this stretch: {})",
|
log.debug("idle-heartbeat: nudge sent to lead {} (quiet nudges so far in this stretch: {})",
|
||||||
leadTerminal, quietCount);
|
leadTerminal, quietCount);
|
||||||
countNudge("sent");
|
// fleetd #609: a nudge that carries the context notice is counted under its own outcome so
|
||||||
|
// it is visible in /metrics — one count per nudge either way, never two.
|
||||||
|
countNudge(notice.isEmpty() ? "sent" : "sent_context");
|
||||||
|
return true;
|
||||||
} catch (RuntimeException e) {
|
} catch (RuntimeException e) {
|
||||||
log.warn("idle-heartbeat: failed to nudge lead {}: {}", leadTerminal, e.toString());
|
log.warn("idle-heartbeat: failed to nudge lead {}: {}", leadTerminal, e.toString());
|
||||||
countNudge("failed");
|
countNudge("failed");
|
||||||
|
return false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #609: the text appended to a nudge when the lead's own context is full — {@code ""}
|
||||||
|
* whenever the notice does not apply, so callers can unconditionally append this without an extra
|
||||||
|
* branch. Wording stays plain (CEFR B1) and honest about who actually gates the roll — see the
|
||||||
|
* {@code requireOperatorConfirm} overload (fleetd #621) for which check that is. This loop only
|
||||||
|
* ever prints text, it never calls {@code fleet_handover} itself.
|
||||||
|
*
|
||||||
|
* @param enabled the {@code leadHeartbeat.contextHighNudge} config flag
|
||||||
|
* @param reading the lead's current {@link LeadContextGauge} reading
|
||||||
|
* @return the notice text (starting with a leading space, to append directly after {@link
|
||||||
|
* FleetState#nudgeText()}), or {@code ""} when disabled or the state is not {@code HIGH}
|
||||||
|
*/
|
||||||
|
static String contextNotice(boolean enabled, LeadContextGauge.Reading reading) {
|
||||||
|
return contextNotice(enabled, reading, false);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #609 review: as {@link #contextNotice(boolean, LeadContextGauge.Reading)}, but also gated on
|
||||||
|
* {@code alreadyNotified} — the context latch as it stood <em>before</em> the current tick's decision.
|
||||||
|
* Without this gate, every pending-driven {@code INJECT} that lands while the context stays {@code
|
||||||
|
* HIGH} would re-append the full notice on top of an already-latched stretch, making the notice's own
|
||||||
|
* closing sentence ("You will not be told again until your context reads ok.") false. {@link
|
||||||
|
* #injectNudge} is the only caller that passes a non-default {@code alreadyNotified}.
|
||||||
|
*
|
||||||
|
* <p>fleetd #621: delegates with {@code requireOperatorConfirm=true} — the pre-#621 default and the
|
||||||
|
* value every existing caller of this overload (including every test written before #621) already
|
||||||
|
* assumed, so the text this overload returns stays byte-identical.
|
||||||
|
*
|
||||||
|
* @param alreadyNotified whether the lead has already been told about the current HIGH stretch
|
||||||
|
*/
|
||||||
|
static String contextNotice(boolean enabled, LeadContextGauge.Reading reading, boolean alreadyNotified) {
|
||||||
|
return contextNotice(enabled, reading, alreadyNotified, true);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #621: as {@link #contextNotice(boolean, LeadContextGauge.Reading, boolean)}, but the closing
|
||||||
|
* instructions also track the daemon's effective {@code leadRollover.requireOperatorConfirm} value,
|
||||||
|
* instead of always asserting that only the operator can approve the roll.
|
||||||
|
*
|
||||||
|
* <p>{@code LeadRollover.confirm(...)} already honours this flag: when it is {@code false}, the daemon
|
||||||
|
* itself gates the roll on the three handover-file checks alone (exists, modified after the {@code
|
||||||
|
* open()} request, and no older than {@code maxDocAgeSeconds}) and never consults {@code
|
||||||
|
* operatorConfirmed}. Before this parameter existed, this notice told the lead to ask the operator
|
||||||
|
* regardless — so a lead that followed its own instructions asked anyway, and setting the config knob
|
||||||
|
* to {@code false} stopped the daemon refusing the roll without stopping the operator being
|
||||||
|
* interrupted. This parameter is how the text is kept honest about which gate is actually live.
|
||||||
|
*
|
||||||
|
* @param requireOperatorConfirm the effective {@code leadRollover.requireOperatorConfirm} value
|
||||||
|
*/
|
||||||
|
static String contextNotice(boolean enabled, LeadContextGauge.Reading reading, boolean alreadyNotified,
|
||||||
|
boolean requireOperatorConfirm) {
|
||||||
|
if (!enabled || alreadyNotified || reading.state() != LeadContextGauge.State.HIGH) {
|
||||||
|
return "";
|
||||||
|
}
|
||||||
|
StringBuilder sb = new StringBuilder(" Your own context is nearly full");
|
||||||
|
String compactionWord = reading.compactions() == 1 ? "compaction" : "compactions";
|
||||||
|
if (reading.tokens() != null) {
|
||||||
|
sb.append(": ").append(reading.tokens()).append(" tokens used, ")
|
||||||
|
.append(reading.compactions()).append(' ').append(compactionWord).append(" so far.");
|
||||||
|
} else {
|
||||||
|
// A HIGH reading always carries a non-null token count today: LeadContextGauge only
|
||||||
|
// reaches HIGH by comparing a number against a threshold. That invariant lives in
|
||||||
|
// another class and nothing asserts it, so this branch does not rely on it — it drops
|
||||||
|
// the token clause rather than printing "null tokens".
|
||||||
|
sb.append(" (").append(reading.compactions()).append(' ').append(compactionWord)
|
||||||
|
.append(" so far).");
|
||||||
|
}
|
||||||
|
if (requireOperatorConfirm) {
|
||||||
|
sb.append(" A fresh session would work better. To hand over: call fleet_handover(action=\"open\"), "
|
||||||
|
+ "write the file it names, ask the operator, then call fleet_handover(action=\"confirm\", "
|
||||||
|
+ "token, operatorConfirmed). Only the operator can approve the roll. You will not be told "
|
||||||
|
+ "again until your context reads ok.");
|
||||||
|
} else {
|
||||||
|
sb.append(" A fresh session would work better. To hand over: call fleet_handover(action=\"open\"), "
|
||||||
|
+ "write the file it names, then call fleet_handover(action=\"confirm\", token). Decide for "
|
||||||
|
+ "yourself when to confirm: the roll goes through if the handover file exists, was "
|
||||||
|
+ "changed after you opened it, and is not older than maxDocAgeSeconds. You will not be "
|
||||||
|
+ "told again until your context reads ok.");
|
||||||
|
}
|
||||||
|
return sb.toString();
|
||||||
|
}
|
||||||
|
|
||||||
/** Schedule the next tick on the scheduler thread pool. */
|
/** Schedule the next tick on the scheduler thread pool. */
|
||||||
private void scheduleNext() {
|
private void scheduleNext() {
|
||||||
scheduler.schedule(this::tick, backoffMs, TimeUnit.MILLISECONDS);
|
scheduler.schedule(this::tick, backoffMs, TimeUnit.MILLISECONDS);
|
||||||
|
|||||||
@@ -63,7 +63,7 @@ import java.util.concurrent.TimeoutException;
|
|||||||
* which messages reached a lead. Any publish still awaiting its confirm is failed rather than left to idle out
|
* which messages reached a lead. Any publish still awaiting its confirm is failed rather than left to idle out
|
||||||
* the confirm timeout against a sequence number that means nothing on the new channel.
|
* the confirm timeout against a sequence number that means nothing on the new channel.
|
||||||
*/
|
*/
|
||||||
public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
public final class LeadMailbox implements LeadChannelHandle {
|
||||||
|
|
||||||
private static final Logger log = LoggerFactory.getLogger(LeadMailbox.class);
|
private static final Logger log = LoggerFactory.getLogger(LeadMailbox.class);
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
package dev.ltms.fleet.msg;
|
package dev.ltms.fleet.msg;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.auth.Principal;
|
||||||
import dev.ltms.fleet.herdr.AgentControl;
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
import dev.ltms.fleet.herdr.AgentStatus;
|
import dev.ltms.fleet.herdr.AgentStatus;
|
||||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||||
@@ -87,20 +88,49 @@ public final class MessageService {
|
|||||||
/**
|
/**
|
||||||
* The worker paused mid-turn to ask the primary a question (CB-205); {@code text} is the
|
* The worker paused mid-turn to ask the primary a question (CB-205); {@code text} is the
|
||||||
* question and {@code turnId} correlates the answer. Not terminal — the primary answers with
|
* question and {@code turnId} correlates the answer. Not terminal — the primary answers with
|
||||||
* {@link #answer(String, String, long)} and the turn resumes.
|
* {@link #answer(String, String, long, String)} and the turn resumes.
|
||||||
*/
|
*/
|
||||||
QUESTION,
|
QUESTION,
|
||||||
/** Timed out after the message was delivered — the worker is still working. */
|
/** Timed out after the message was delivered — the worker is still working. */
|
||||||
TIMED_OUT_WORKING,
|
TIMED_OUT_WORKING,
|
||||||
/** Timed out before delivery — the message is still queued for the worker. */
|
/**
|
||||||
|
* Timed out with no confirmed delivery, and the target saw nothing — the message will not
|
||||||
|
* arrive later, so a caller may resend. {@link #send} reaches this outcome through {@link
|
||||||
|
* Injector#cancel} reporting one of two routes: {@link Injector.Cancellation#CANCELLED}
|
||||||
|
* means the message was still queued and this call removed it; {@link
|
||||||
|
* Injector.Cancellation#NOT_DELIVERED} means nothing was ever sent — the queue was cleared
|
||||||
|
* because the target never became ready or was abandoned, or the injector's call to the
|
||||||
|
* target's terminal ({@link dev.ltms.fleet.herdr.AgentControl#send}) failed with a herdr
|
||||||
|
* error that this codebase already treats as a confirmed absence. A third route,
|
||||||
|
* {@link Injector.Cancellation#ATTEMPTED}, used to be folded into this same outcome
|
||||||
|
* (fleetd #571) — it no longer is; see {@link #TIMED_OUT_UNCONFIRMED}.
|
||||||
|
*/
|
||||||
TIMED_OUT_QUEUED,
|
TIMED_OUT_QUEUED,
|
||||||
|
/**
|
||||||
|
* Timed out with delivery unknown. {@link #send} reaches this outcome when {@link
|
||||||
|
* Injector#cancel} reports {@link Injector.Cancellation#ATTEMPTED} (fleetd #551): the call
|
||||||
|
* to the target's terminal ({@link dev.ltms.fleet.herdr.AgentControl#send}) was made, but
|
||||||
|
* this caller never observed whether it reached the pane. {@code agent.prompt} pastes
|
||||||
|
* <em>and submits</em> in one call, so the target may already hold a complete, submitted
|
||||||
|
* turn and be working on it right now — the same reality as {@link #TIMED_OUT_WORKING},
|
||||||
|
* just not confirmed. The message may or may not have arrived. Treat this as neither a
|
||||||
|
* confirmed delivery nor a confirmed absence: a caller that resends on this outcome risks a
|
||||||
|
* double delivery — the same brief typed into the pane twice (fleetd #571).
|
||||||
|
*/
|
||||||
|
TIMED_OUT_UNCONFIRMED,
|
||||||
/** Another send to this session was in flight for the whole window. */
|
/** Another send to this session was in flight for the whole window. */
|
||||||
BUSY,
|
BUSY,
|
||||||
/**
|
/**
|
||||||
* An answer ({@link #answer(String, String, long)}) referenced a {@code turnId} that is no
|
* An answer ({@link #answer(String, String, long, String)}) referenced a {@code turnId}
|
||||||
* longer open — the worker's {@code fleet_ask} already timed out or was answered.
|
* that is no longer open — the worker's {@code fleet_ask} already timed out or was answered.
|
||||||
*/
|
*/
|
||||||
STALE_TURN
|
STALE_TURN,
|
||||||
|
/**
|
||||||
|
* An answer ({@link #answer(String, String, long, String)}) named a {@code turnId} that is
|
||||||
|
* still open, but the answering caller is not the caller whose accepted delegation opened
|
||||||
|
* it. Distinct from {@link #STALE_TURN} so a refusal is never reported as a lapsed turn.
|
||||||
|
*/
|
||||||
|
NOT_TURN_OWNER
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -110,7 +140,7 @@ public final class MessageService {
|
|||||||
* {@link Outcome#COMPLETED_UNREPLIED}), or the question for {@link Outcome#QUESTION},
|
* {@link Outcome#COMPLETED_UNREPLIED}), or the question for {@link Outcome#QUESTION},
|
||||||
* else {@code null}
|
* else {@code null}
|
||||||
* @param turnId correlation id for a {@link Outcome#QUESTION} (answered via
|
* @param turnId correlation id for a {@link Outcome#QUESTION} (answered via
|
||||||
* {@link #answer(String, String, long)}), else {@code null}
|
* {@link #answer(String, String, long, String)}), else {@code null}
|
||||||
*/
|
*/
|
||||||
public record Reply(Outcome outcome, String text, String turnId) {
|
public record Reply(Outcome outcome, String text, String turnId) {
|
||||||
/** A reply with no correlation id (the common terminal outcomes). */
|
/** A reply with no correlation id (the common terminal outcomes). */
|
||||||
@@ -252,10 +282,16 @@ public final class MessageService {
|
|||||||
* {@link #abandon}) can never match again regardless of this flag's value.
|
* {@link #abandon}) can never match again regardless of this flag's value.
|
||||||
*/
|
*/
|
||||||
private volatile boolean askTimedOut;
|
private volatile boolean askTimedOut;
|
||||||
|
/**
|
||||||
|
* The owner key of the caller whose {@code fleet_send{wait:false}} created this ticket, or
|
||||||
|
* {@code null} for the unnamed primary and overloads that do not record a caller.
|
||||||
|
*/
|
||||||
|
private final String creatorOwner;
|
||||||
|
|
||||||
private Task(String ticket, String target, LongSupplier nowNanos) {
|
private Task(String ticket, String target, LongSupplier nowNanos, String creatorOwner) {
|
||||||
this.ticket = ticket;
|
this.ticket = ticket;
|
||||||
this.target = target;
|
this.target = target;
|
||||||
|
this.creatorOwner = creatorOwner;
|
||||||
this.createdNanos = nowNanos.getAsLong();
|
this.createdNanos = nowNanos.getAsLong();
|
||||||
future.whenComplete((reply, ex) -> completedNanos = nowNanos.getAsLong());
|
future.whenComplete((reply, ex) -> completedNanos = nowNanos.getAsLong());
|
||||||
}
|
}
|
||||||
@@ -299,15 +335,32 @@ public final class MessageService {
|
|||||||
*/
|
*/
|
||||||
private final ConcurrentHashMap<String, Boolean> strandedReplies = new ConcurrentHashMap<>();
|
private final ConcurrentHashMap<String, Boolean> strandedReplies = new ConcurrentHashMap<>();
|
||||||
/**
|
/**
|
||||||
* Targets whose last send timed out with {@link Outcome#TIMED_OUT_QUEUED} (CB-640) — the
|
* Targets whose last send timed out with no confirmed delivery (CB-640) — {@link #send} called
|
||||||
* message never reached the {@link Injector} delivery window before the caller's deadline, so
|
* {@link Injector#cancel} and got back something other than {@code DELIVERED}. That covers
|
||||||
* it is still sitting in the injector's own per-target queue. Set where {@link #send} already
|
* three histories, not one: {@link Injector.Cancellation#CANCELLED} — the message was still
|
||||||
* computes {@code wasDelivered} for that outcome; no new queue is kept here, only the fact.
|
* queued and {@code cancel} removed it right there; {@link Injector.Cancellation#NOT_DELIVERED}
|
||||||
* Cleared the same way as {@link #strandedReplies}: the next accepted delivery for the target
|
* — nothing was ever sent, because the target never became ready, was torn down, or the call to
|
||||||
* ({@link #send} opening a fresh waiter) or a teardown ({@link #abandon}).
|
* its terminal failed with a herdr error this codebase already treats as a confirmed absence; or
|
||||||
|
* {@link Injector.Cancellation#ATTEMPTED} (fleetd #551) — the call to the target's terminal was
|
||||||
|
* made and its outcome is unknown, so the target may already hold a complete, submitted turn.
|
||||||
|
* Only the first two mean the message will not arrive later and the target saw nothing; on the
|
||||||
|
* third it may already have arrived in full — and the caller sees a different outcome for it
|
||||||
|
* ({@link Outcome#TIMED_OUT_UNCONFIRMED}, fleetd #571) than for the first two ({@link
|
||||||
|
* Outcome#TIMED_OUT_QUEUED}). Set where {@link #send} already computes {@code wasDelivered} for
|
||||||
|
* that outcome; no queue is kept here, only the fact that the send ended with no confirmed
|
||||||
|
* delivery. Cleared the same way as {@link #strandedReplies}: the next accepted delivery for the
|
||||||
|
* target ({@link #send} opening a fresh waiter) or a teardown ({@link #abandon}).
|
||||||
*/
|
*/
|
||||||
private final ConcurrentHashMap<String, Boolean> queuedDeliveries = new ConcurrentHashMap<>();
|
private final ConcurrentHashMap<String, Boolean> queuedDeliveries = new ConcurrentHashMap<>();
|
||||||
private final AtomicLong ticketSeq = new AtomicLong();
|
private final AtomicLong ticketSeq = new AtomicLong();
|
||||||
|
/**
|
||||||
|
* Minted once per {@code MessageService} instance and folded into every ticket id (see
|
||||||
|
* {@link #sendAsync(String, String, Runnable, Principal)}). {@link #ticketSeq} alone restarts at
|
||||||
|
* zero for every instance, so without this a ticket id can be reused across instances and
|
||||||
|
* resolve to an unrelated {@link Task} with no error; this nonce makes that impossible, because
|
||||||
|
* an id minted by one instance can never match the id space of another.
|
||||||
|
*/
|
||||||
|
private final String ticketBootNonce = UUID.randomUUID().toString().substring(0, 6);
|
||||||
private final ExecutorService asyncExecutor = Executors.newThreadPerTaskExecutor(
|
private final ExecutorService asyncExecutor = Executors.newThreadPerTaskExecutor(
|
||||||
Thread.ofVirtual().name("bridge-async-", 0).factory());
|
Thread.ofVirtual().name("bridge-async-", 0).factory());
|
||||||
|
|
||||||
@@ -391,12 +444,22 @@ public final class MessageService {
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* Read-only delegation fact for fleet views (CB-640): {@code target}'s last send timed out
|
* Read-only delegation fact for fleet views (CB-640): {@code target}'s last send timed out
|
||||||
* before the {@link Injector} ever delivered it — the caller saw
|
* with no confirmed delivery — the caller saw {@link Outcome#TIMED_OUT_QUEUED} or {@link
|
||||||
* {@link Outcome#TIMED_OUT_QUEUED} (see the {@code TimeoutException} branch of {@link #send}),
|
* Outcome#TIMED_OUT_UNCONFIRMED} (fleetd #571; see the {@code TimeoutException} branch of
|
||||||
* and the message is still sitting in the injector's per-target queue waiting for the worker
|
* {@link #send}). Despite the method's name, this is not proof that a message is sitting in a
|
||||||
* to go idle. Distinct from {@link Outcome#TIMED_OUT_WORKING}, where delivery already happened
|
* queue: {@link Injector#cancel} reports this outcome through three routes. {@link
|
||||||
* and only the reply is outstanding. Cleared the next time this target's delivery is accepted
|
* Injector.Cancellation#CANCELLED} means the message was still queued and got removed right
|
||||||
* or the target is abandoned — see {@link #queuedDeliveries}.
|
* there. {@link Injector.Cancellation#NOT_DELIVERED} means nothing was ever sent — the target
|
||||||
|
* never became ready, was torn down, or the call to its terminal failed with a herdr error this
|
||||||
|
* codebase already treats as a confirmed absence. Only these two routes mean the message will
|
||||||
|
* not arrive later, and both report {@code TIMED_OUT_QUEUED}. {@link
|
||||||
|
* Injector.Cancellation#ATTEMPTED} (fleetd #551) means the call to the target's terminal was
|
||||||
|
* made and its outcome is unknown: {@code agent.prompt} pastes <em>and submits</em> in one
|
||||||
|
* call, so on this route the target may already hold a complete, submitted turn and be
|
||||||
|
* working on it right now — it does NOT follow that the target saw nothing, and this route
|
||||||
|
* reports {@code TIMED_OUT_UNCONFIRMED} instead. Distinct from {@link Outcome#TIMED_OUT_WORKING},
|
||||||
|
* where delivery already happened and only the reply is outstanding. Cleared the next time this
|
||||||
|
* target's delivery is accepted or the target is abandoned — see {@link #queuedDeliveries}.
|
||||||
*/
|
*/
|
||||||
public boolean hasQueuedDelivery(String target) {
|
public boolean hasQueuedDelivery(String target) {
|
||||||
return target != null && queuedDeliveries.containsKey(target);
|
return target != null && queuedDeliveries.containsKey(target);
|
||||||
@@ -416,15 +479,15 @@ public final class MessageService {
|
|||||||
/**
|
/**
|
||||||
* Read-only delegation fact for fleet views (CB-640): an async ticket is still
|
* Read-only delegation fact for fleet views (CB-640): an async ticket is still
|
||||||
* {@link Phase#PENDING} against {@code target}, yet nothing is actually in flight for it — no
|
* {@link Phase#PENDING} against {@code target}, yet nothing is actually in flight for it — no
|
||||||
* open rendezvous waiter ({@link #hasAcceptedDelivery}) and no message still sitting in the
|
* open rendezvous waiter ({@link #hasAcceptedDelivery}) and no record of a send that ended with
|
||||||
* injector's queue ({@link #hasQueuedDelivery}). A healthy PENDING ticket can briefly look this
|
* no confirmed delivery ({@link #hasQueuedDelivery}). A healthy PENDING ticket can briefly look
|
||||||
* way while its virtual thread has not yet been scheduled or is blocked on the session lock
|
* this way while its virtual thread has not yet been scheduled or is blocked on the session
|
||||||
* behind another send to the same target, so this is a snapshot fact for the health classifier
|
* lock behind another send to the same target, so this is a snapshot fact for the health
|
||||||
* to weigh across ticks, not proof on its own that the ticket is stuck. It also genuinely
|
* classifier to weigh across ticks, not proof on its own that the ticket is stuck. It also
|
||||||
* persists — not just as a passing race — once an async {@code fleet_ask} lapses unanswered:
|
* genuinely persists — not just as a passing race — once an async {@code fleet_ask} lapses
|
||||||
* {@link #ask} clears the ticket's question and returns it to {@code PENDING}, but {@link #send}
|
* unanswered: {@link #ask} clears the ticket's question and returns it to {@code PENDING}, but
|
||||||
* already closed the forward waiter the instant the question surfaced, so the target has
|
* {@link #send} already closed the forward waiter the instant the question surfaced, so the
|
||||||
* neither an accepted nor a queued delivery left to show for it.
|
* target has neither an accepted nor a queued delivery left to show for it.
|
||||||
*
|
*
|
||||||
* <p><strong>Deliberately still {@code question == null} only (fleetd #275).</strong> This
|
* <p><strong>Deliberately still {@code question == null} only (fleetd #275).</strong> This
|
||||||
* method must not also report a still-{@link Phase#ASKING} task as orphaned: the worker may
|
* method must not also report a still-{@link Phase#ASKING} task as orphaned: the worker may
|
||||||
@@ -611,10 +674,10 @@ public final class MessageService {
|
|||||||
return switch (o) {
|
return switch (o) {
|
||||||
case REPLIED -> "replied";
|
case REPLIED -> "replied";
|
||||||
case COMPLETED_UNREPLIED -> "completion_fallback";
|
case COMPLETED_UNREPLIED -> "completion_fallback";
|
||||||
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, BUSY -> "timeout";
|
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, TIMED_OUT_UNCONFIRMED, BUSY -> "timeout";
|
||||||
case WORKER_FAILED -> "failed";
|
case WORKER_FAILED -> "failed";
|
||||||
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
||||||
case STALE_TURN, QUESTION -> null; // not a completed delegation
|
case STALE_TURN, QUESTION, NOT_TURN_OWNER -> null; // not a completed delegation
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -713,7 +776,7 @@ public final class MessageService {
|
|||||||
* failure.
|
* failure.
|
||||||
*
|
*
|
||||||
* <p><strong>Without {@code sweepAsking} on the release path, a target torn down while
|
* <p><strong>Without {@code sweepAsking} on the release path, a target torn down while
|
||||||
* genuinely {@code ASKING} was unrecoverable.</strong> {@link #resolveQuestion} had already
|
* genuinely {@code ASKING} was unrecoverable.</strong> {@link Rendezvous#resolveQuestion(String, String, String)} had already
|
||||||
* closed the forward waiter the instant the question surfaced (so the {@code waiter} branch
|
* closed the forward waiter the instant the question surfaced (so the {@code waiter} branch
|
||||||
* below finds nothing to fail), the {@code question == null} guard excluded the task from
|
* below finds nothing to fail), the {@code question == null} guard excluded the task from
|
||||||
* {@code matching} (so the loop below skipped it too), and the worker's own {@code fleet_ask}
|
* {@code matching} (so the loop below skipped it too), and the worker's own {@code fleet_ask}
|
||||||
@@ -873,14 +936,17 @@ public final class MessageService {
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* Deliver {@code content} to {@code target} (a herdr {@code terminal_id}) and block until the
|
* Deliver {@code content} to {@code target} (a herdr {@code terminal_id}) and block until the
|
||||||
* worker replies via {@link Rendezvous} or {@code timeoutMillis} elapses.
|
* worker replies via {@link Rendezvous} or {@code timeoutMillis} elapses. {@code callerOwner}
|
||||||
|
* identifies the caller making this call and is recorded as the turn's owner. It is the only
|
||||||
|
* caller {@link #answer(String, String, long, String)} will
|
||||||
|
* later accept an answer from if the worker pauses mid-turn to ask.
|
||||||
*/
|
*/
|
||||||
public Reply send(String target, String content, long timeoutMillis) {
|
public Reply send(String target, String content, long timeoutMillis, String callerOwner) {
|
||||||
return send(target, content, timeoutMillis, null);
|
return send(target, content, timeoutMillis, null, callerOwner);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* As {@link #send(String, String, long)}, but with an accepted-delivery hook.
|
* As {@link #send(String, String, long, String)}, but with an accepted-delivery hook.
|
||||||
*
|
*
|
||||||
* <p>{@code onAccepted} is invoked exactly once, once this send has won {@code target}'s send
|
* <p>{@code onAccepted} is invoked exactly once, once this send has won {@code target}'s send
|
||||||
* lock and so become the <em>accepted target turn</em> — it runs <em>before</em> delivery is
|
* lock and so become the <em>accepted target turn</em> — it runs <em>before</em> delivery is
|
||||||
@@ -891,12 +957,13 @@ public final class MessageService {
|
|||||||
* acceptance means a concurrent sender that times out {@code BUSY} can never steal ownership it
|
* acceptance means a concurrent sender that times out {@code BUSY} can never steal ownership it
|
||||||
* never earned. {@code null} disables the hook.
|
* never earned. {@code null} disables the hook.
|
||||||
*/
|
*/
|
||||||
public Reply send(String target, String content, long timeoutMillis, Runnable onAccepted) {
|
public Reply send(String target, String content, long timeoutMillis, Runnable onAccepted, String callerOwner) {
|
||||||
return send(target, content, timeoutMillis, onAccepted, null);
|
return send(target, content, timeoutMillis, onAccepted, null, callerOwner);
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Run a send, optionally stopping an async task that teardown already failed before acceptance. */
|
/** Run a send, optionally stopping an async task that teardown already failed before acceptance. */
|
||||||
private Reply send(String target, String content, long timeoutMillis, Runnable onAccepted, Task task) {
|
private Reply send(String target, String content, long timeoutMillis, Runnable onAccepted, Task task,
|
||||||
|
String callerOwner) {
|
||||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||||
ReentrantLock lock = sessionLocks.computeIfAbsent(target, _ -> new ReentrantLock());
|
ReentrantLock lock = sessionLocks.computeIfAbsent(target, _ -> new ReentrantLock());
|
||||||
|
|
||||||
@@ -916,7 +983,7 @@ public final class MessageService {
|
|||||||
// open race). Opening first also means a throwing onAccepted (fired before enqueue) or an
|
// open race). Opening first also means a throwing onAccepted (fired before enqueue) or an
|
||||||
// enqueue failure is safely closed by the finally below: nothing is left queued, and the
|
// enqueue failure is safely closed by the finally below: nothing is left queued, and the
|
||||||
// failed send leaves no stale waiter behind.
|
// failed send leaves no stale waiter behind.
|
||||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(target);
|
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(target, Rendezvous.Owner.of(callerOwner));
|
||||||
// CB-640: this send now owns target's delivery, so any earlier stranded-reply or
|
// CB-640: this send now owns target's delivery, so any earlier stranded-reply or
|
||||||
// still-queued fact no longer describes the live state — clear both rather than let
|
// still-queued fact no longer describes the live state — clear both rather than let
|
||||||
// them outlive the send that supersedes them.
|
// them outlive the send that supersedes them.
|
||||||
@@ -940,6 +1007,7 @@ public final class MessageService {
|
|||||||
} catch (TimeoutException e) {
|
} catch (TimeoutException e) {
|
||||||
boolean wasDelivered = delivery.completion().isDone()
|
boolean wasDelivered = delivery.completion().isDone()
|
||||||
&& !delivery.completion().isCompletedExceptionally();
|
&& !delivery.completion().isCompletedExceptionally();
|
||||||
|
Injector.Cancellation cancellation = null;
|
||||||
if (!wasDelivered) {
|
if (!wasDelivered) {
|
||||||
if (timeoutCancellationRaceHookForTest != null) {
|
if (timeoutCancellationRaceHookForTest != null) {
|
||||||
// Test-only (fleetd #345): see the field's own javadoc.
|
// Test-only (fleetd #345): see the field's own javadoc.
|
||||||
@@ -947,16 +1015,27 @@ public final class MessageService {
|
|||||||
}
|
}
|
||||||
// The target monitor makes cancellation atomic with onStatus picking this
|
// The target monitor makes cancellation atomic with onStatus picking this
|
||||||
// Pending up. If pickup won, report TIMED_OUT_WORKING because the text landed.
|
// Pending up. If pickup won, report TIMED_OUT_WORKING because the text landed.
|
||||||
wasDelivered = injector.cancel(delivery) == Injector.Cancellation.DELIVERED;
|
cancellation = injector.cancel(delivery);
|
||||||
|
wasDelivered = cancellation == Injector.Cancellation.DELIVERED;
|
||||||
}
|
}
|
||||||
log.debug("send to {} timed out (delivered={})", target, wasDelivered);
|
log.debug("send to {} timed out (delivered={})", target, wasDelivered);
|
||||||
if (!wasDelivered) {
|
Outcome outcome;
|
||||||
// CB-640: record that delivery did not happen for fleet health (see
|
if (wasDelivered) {
|
||||||
// queuedDeliveries). The exact Pending was cancelled, so it cannot arrive later.
|
outcome = Outcome.TIMED_OUT_WORKING;
|
||||||
|
} else if (cancellation == Injector.Cancellation.ATTEMPTED) {
|
||||||
|
// fleetd #571: the call to the target's terminal was made and its outcome is
|
||||||
|
// unknown — the message may already have arrived in full, so this must not
|
||||||
|
// be reported as TIMED_OUT_QUEUED, which promises it never will.
|
||||||
|
outcome = Outcome.TIMED_OUT_UNCONFIRMED;
|
||||||
|
} else {
|
||||||
|
// CB-640: record that delivery is not confirmed, for fleet health (see
|
||||||
|
// queuedDeliveries). cancellation is CANCELLED (this call removed a
|
||||||
|
// still-queued Pending) or NOT_DELIVERED (an earlier attempt already failed
|
||||||
|
// with a confirmed absence) — both mean the target saw nothing.
|
||||||
queuedDeliveries.put(target, Boolean.TRUE);
|
queuedDeliveries.put(target, Boolean.TRUE);
|
||||||
|
outcome = Outcome.TIMED_OUT_QUEUED;
|
||||||
}
|
}
|
||||||
return recorded(new Reply(
|
return recorded(new Reply(outcome, null));
|
||||||
wasDelivered ? Outcome.TIMED_OUT_WORKING : Outcome.TIMED_OUT_QUEUED, null));
|
|
||||||
} catch (ExecutionException e) {
|
} catch (ExecutionException e) {
|
||||||
Throwable cause = e.getCause();
|
Throwable cause = e.getCause();
|
||||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||||
@@ -1083,34 +1162,57 @@ public final class MessageService {
|
|||||||
* mid-turn (already picked up), so the answer flows back through its own open {@code fleet_ask}
|
* mid-turn (already picked up), so the answer flows back through its own open {@code fleet_ask}
|
||||||
* call, not a new status-gated delivery. The forward waiter is opened <em>before</em> the worker
|
* call, not a new status-gated delivery. The forward waiter is opened <em>before</em> the worker
|
||||||
* is unblocked so a reply that lands the instant it resumes is not lost.
|
* is unblocked so a reply that lands the instant it resumes is not lost.
|
||||||
|
*
|
||||||
|
* <p>{@code callerOwner} identifies the caller making this call. It is checked against the
|
||||||
|
* turn's recorded owner (the caller whose
|
||||||
|
* accepted delegation opened it, see {@link #send(String, String, long, String)} and
|
||||||
|
* {@link #sendAsync(String, String, Runnable, Principal)}) before anything else runs: a mismatch,
|
||||||
|
* including a turn with no owner on record at all, returns {@link Outcome#NOT_TURN_OWNER}
|
||||||
|
* without touching the rendezvous, the session lock, or any async task bookkeeping.
|
||||||
*/
|
*/
|
||||||
public Reply answer(String turnId, String content, long timeoutMillis) {
|
public Reply answer(String turnId, String content, long timeoutMillis, String callerOwner) {
|
||||||
String workerSession = rendezvous.askSession(turnId);
|
String workerSession = rendezvous.askSession(turnId);
|
||||||
if (workerSession == null) {
|
if (workerSession == null) {
|
||||||
return new Reply(Outcome.STALE_TURN, null); // the ask lapsed (timed out or already answered)
|
return new Reply(Outcome.STALE_TURN, null); // the ask lapsed (timed out or already answered)
|
||||||
}
|
}
|
||||||
|
Rendezvous.Owner owner = rendezvous.askOwner(turnId);
|
||||||
|
if (!Rendezvous.Owner.permits(owner, callerOwner)) {
|
||||||
|
return new Reply(Outcome.NOT_TURN_OWNER, null);
|
||||||
|
}
|
||||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||||
ReentrantLock lock = sessionLocks.computeIfAbsent(workerSession, _ -> new ReentrantLock());
|
ReentrantLock lock = sessionLocks.computeIfAbsent(workerSession, _ -> new ReentrantLock());
|
||||||
if (!tryLock(lock, remainingMillis(deadlineNanos))) {
|
if (!tryLock(lock, remainingMillis(deadlineNanos))) {
|
||||||
return new Reply(Outcome.BUSY, null);
|
return new Reply(Outcome.BUSY, null);
|
||||||
}
|
}
|
||||||
try {
|
try {
|
||||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession);
|
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession, owner);
|
||||||
// #282: mirror send()'s registration (:802) so a SECOND fleet_ask inside this same
|
// fleetd #575: this try used to open below, AFTER the Task lookup/registration and the
|
||||||
// resumed turn can re-associate the async ticket with its new turnId via
|
// STALE_TURN early return that follows it — so that return was covered only by a
|
||||||
// markAsyncQuestion — without this, that second ask has no Task to attach to, and
|
// hand-rolled copy of the finally's own cleanup pair, not the finally itself. Widening the
|
||||||
// markAsyncQuestion silently returns null.
|
// try up to wrap the registration closes the gap structurally: every exit from here on,
|
||||||
Task task = asyncTasksByTurn.get(turnId);
|
// STALE_TURN included, now runs through the one finally below exactly once, and the
|
||||||
if (task != null) {
|
// duplicated pair is gone. This did not fix a live leak — see the ticket: neither
|
||||||
asyncTasksByWaiter.put(reply, task);
|
// rendezvous.answerAsk nor clearAsyncQuestion(turnId, false) can throw, so nothing ever
|
||||||
}
|
// actually left through the old gap uncovered — but #572 found this exact drift (one
|
||||||
if (!rendezvous.answerAsk(turnId, content)) {
|
// finally asserted, an identical sibling not) on this same file, and the hand-rolled copy
|
||||||
asyncTasksByWaiter.remove(reply);
|
// was the wrong shape to keep regardless of whether it was ever exercised.
|
||||||
rendezvous.close(workerSession, reply);
|
|
||||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
|
||||||
}
|
|
||||||
clearAsyncQuestion(turnId, false);
|
|
||||||
try {
|
try {
|
||||||
|
// #282: mirror send()'s registration (:802) so a SECOND fleet_ask inside this same
|
||||||
|
// resumed turn can re-associate the async ticket with its new turnId via
|
||||||
|
// markAsyncQuestion — without this, that second ask has no Task to attach to, and
|
||||||
|
// markAsyncQuestion silently returns null.
|
||||||
|
Task task = asyncTasksByTurn.get(turnId);
|
||||||
|
if (task != null) {
|
||||||
|
asyncTasksByWaiter.put(reply, task);
|
||||||
|
}
|
||||||
|
if (answerAskLapseRaceHookForTest != null) {
|
||||||
|
// Test-only (fleetd #575): see the field's own javadoc.
|
||||||
|
answerAskLapseRaceHookForTest.run();
|
||||||
|
}
|
||||||
|
if (!rendezvous.answerAsk(turnId, content)) {
|
||||||
|
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||||
|
}
|
||||||
|
clearAsyncQuestion(turnId, false);
|
||||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||||
Reply result = new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
Reply result = new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||||
// #282: this waiter can resolve with a FRESH question rather than a terminal reply —
|
// #282: this waiter can resolve with a FRESH question rather than a terminal reply —
|
||||||
@@ -1213,8 +1315,20 @@ public final class MessageService {
|
|||||||
* @return the ticket to poll for the eventual result
|
* @return the ticket to poll for the eventual result
|
||||||
*/
|
*/
|
||||||
public String sendAsync(String target, String content, Runnable onAccepted) {
|
public String sendAsync(String target, String content, Runnable onAccepted) {
|
||||||
String ticket = "task-" + ticketSeq.incrementAndGet();
|
return sendAsync(target, content, onAccepted, null);
|
||||||
Task task = new Task(ticket, target, nowNanos);
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* As {@link #sendAsync(String, String, Runnable)}, recording {@code creator}'s owner key as this
|
||||||
|
* ticket's owner. The key is derived here from the resolved principal so callers cannot pass a
|
||||||
|
* terminal address where an owner identity is required.
|
||||||
|
*
|
||||||
|
* @return the ticket to poll for the eventual result
|
||||||
|
*/
|
||||||
|
public String sendAsync(String target, String content, Runnable onAccepted, Principal creator) {
|
||||||
|
String ticket = "task-" + ticketBootNonce + "-" + ticketSeq.incrementAndGet();
|
||||||
|
String creatorOwner = creator == null ? null : creator.ownerKey();
|
||||||
|
Task task = new Task(ticket, target, nowNanos, creatorOwner);
|
||||||
tasks.put(ticket, task);
|
tasks.put(ticket, task);
|
||||||
if (pushLoop != null) {
|
if (pushLoop != null) {
|
||||||
// CB-588: task.future only ever completes on a terminal phase (DONE or a failure) — a
|
// CB-588: task.future only ever completes on a terminal phase (DONE or a failure) — a
|
||||||
@@ -1245,7 +1359,7 @@ public final class MessageService {
|
|||||||
}
|
}
|
||||||
asyncExecutor.submit(() -> {
|
asyncExecutor.submit(() -> {
|
||||||
try {
|
try {
|
||||||
Reply result = send(target, content, ASYNC_TIMEOUT_MS, onAccepted, task);
|
Reply result = send(target, content, ASYNC_TIMEOUT_MS, onAccepted, task, creatorOwner);
|
||||||
if (result.outcome() == Outcome.QUESTION) {
|
if (result.outcome() == Outcome.QUESTION) {
|
||||||
// Keep the accepted owner until answer() finishes it. markAsyncQuestion may run
|
// Keep the accepted owner until answer() finishes it. markAsyncQuestion may run
|
||||||
// just after resolveQuestion wakes this thread.
|
// just after resolveQuestion wakes this thread.
|
||||||
@@ -1277,15 +1391,31 @@ public final class MessageService {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Snapshot the state of an async delegation. Returns {@code null} for an unknown/expired ticket;
|
* As {@link #poll(String, String)}, with no caller owner key — the unnamed primary's ticket rule
|
||||||
* otherwise a {@link Phase#PENDING} view (with the live worker status as detail), a
|
* checked, so this overload must only be used where the caller's identity is otherwise
|
||||||
* {@link Phase#DONE} view carrying the reply, or a {@link Phase#FAILED} view with the reason.
|
* irrelevant.
|
||||||
*/
|
*/
|
||||||
public TaskView poll(String ticket) {
|
public TaskView poll(String ticket) {
|
||||||
|
return poll(ticket, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Snapshot the state of an async delegation. Returns {@code null} for an unknown/expired ticket.
|
||||||
|
* Refuses a {@code callerOwner} that differs from the owner that created the ticket (see
|
||||||
|
* {@link #sendAsync(String, String, Runnable, Principal)}) with a {@link Phase#FAILED} view that
|
||||||
|
* carries no reply text. The unnamed primary has a {@code null} owner key and is never refused.
|
||||||
|
* Otherwise returns a {@link Phase#PENDING} view (with the live worker status as detail), a
|
||||||
|
* {@link Phase#DONE} view carrying the reply, or a {@link Phase#FAILED} view with the reason.
|
||||||
|
*/
|
||||||
|
public TaskView poll(String ticket, String callerOwner) {
|
||||||
Task task = tasks.get(ticket);
|
Task task = tasks.get(ticket);
|
||||||
if (task == null) {
|
if (task == null) {
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
|
if (!ownsTicket(task, callerOwner)) {
|
||||||
|
return new TaskView(ticket, Phase.FAILED, null, null,
|
||||||
|
"forbidden: this ticket was created by a different session", null);
|
||||||
|
}
|
||||||
CompletableFuture<Reply> f = task.future;
|
CompletableFuture<Reply> f = task.future;
|
||||||
if (!f.isDone()) {
|
if (!f.isDone()) {
|
||||||
Reply question = task.question;
|
Reply question = task.question;
|
||||||
@@ -1321,6 +1451,16 @@ public final class MessageService {
|
|||||||
return new TaskView(ticket, Phase.FAILED, null, null, detail, null);
|
return new TaskView(ticket, Phase.FAILED, null, null, detail, null);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Whether {@code callerOwner} may read {@code task}'s state. A {@code null} caller key is the
|
||||||
|
* unnamed primary and may read every ticket. Other callers must match the task's owner key. This
|
||||||
|
* differs from {@link Rendezvous.Owner#permits}: a missing rendezvous owner is not an authenticated
|
||||||
|
* unnamed primary, so that gate refuses every caller when no owner was recorded.
|
||||||
|
*/
|
||||||
|
private static boolean ownsTicket(Task task, String callerOwner) {
|
||||||
|
return callerOwner == null || callerOwner.equals(task.creatorOwner);
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Test seam only — carries no production behaviour, and nothing in this class calls it;
|
* Test seam only — carries no production behaviour, and nothing in this class calls it;
|
||||||
* {@link #pruneTerminalTickets} still reads {@link Task#completedNanos} directly.
|
* {@link #pruneTerminalTickets} still reads {@link Task#completedNanos} directly.
|
||||||
@@ -1610,22 +1750,47 @@ public final class MessageService {
|
|||||||
this.askTimeoutRaceHookForTest = hook;
|
this.askTimeoutRaceHookForTest = hook;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Null in production; test seam for fleetd #575 — invoked from {@link #answer}, right after this
|
||||||
|
* call's own {@code Task} registration and right before its {@code rendezvous.answerAsk(turnId,
|
||||||
|
* content)} call. A test installs this to complete the SAME turnId's ask directly via {@link
|
||||||
|
* Rendezvous#answerAsk} from inside that exact window, deterministically reproducing what a
|
||||||
|
* second, concurrent {@code answer()} call racing to unblock the same ask can otherwise only
|
||||||
|
* win by timing luck: this call's own {@code askSession(turnId)} lookup at the top already saw
|
||||||
|
* the ask as open, but by the time it reaches {@code rendezvous.answerAsk} here, the other call
|
||||||
|
* already completed it (or the worker's own {@code ask()} teardown already closed it) — so this
|
||||||
|
* call must see {@code false} and return {@link Outcome#STALE_TURN}, exactly the "lapsed between
|
||||||
|
* the lookup and the unblock" case named at that call site. Proves the #575 fix (widening this
|
||||||
|
* method's try so a single finally covers this exit) does not change that outcome and still
|
||||||
|
* cleans this call's own {@code reply} up exactly once.
|
||||||
|
*/
|
||||||
|
private volatile Runnable answerAskLapseRaceHookForTest;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Test-only (fleetd #575): install {@link #answerAskLapseRaceHookForTest}. Package-private so the
|
||||||
|
* test, in the same package, can reach it without widening any production API.
|
||||||
|
*/
|
||||||
|
void setAnswerAskLapseRaceHookForTest(Runnable hook) {
|
||||||
|
this.answerAskLapseRaceHookForTest = hook;
|
||||||
|
}
|
||||||
|
|
||||||
/** A new send must not open a waiter while an async ticket owns this worker's paused turn. */
|
/** A new send must not open a waiter while an async ticket owns this worker's paused turn. */
|
||||||
private boolean hasAsyncQuestion(String target) {
|
private boolean hasAsyncQuestion(String target) {
|
||||||
return asyncTasksByTurn.values().stream().anyMatch(task -> target.equals(task.target));
|
return asyncTasksByTurn.values().stream().anyMatch(task -> target.equals(task.target));
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The question {@code workerSession} is currently paused on via {@code fleet_ask}, if any
|
* The question {@code workerSession} is currently paused on via {@code fleet_ask}, if any —
|
||||||
* (CB-582) — {@code fleet_status} uses this to show a pending question without the caller
|
* {@code fleet_status} uses this to show a pending question without the caller needing the
|
||||||
* needing the ticket. {@code null} when the session has no open async question (including a
|
* ticket. {@code null} when the session has no open async question (including a session mid a
|
||||||
* session mid a <em>blocking</em> {@code fleet_ask}, which has no {@link Task} to look up — see
|
* <em>blocking</em> {@code fleet_ask}, which has no {@link Task} to look up — see
|
||||||
* {@link PendingAsk}).
|
* {@link PendingAsk}), or when {@code callerOwner} does not own the task the question
|
||||||
|
* belongs to (see {@link #ownsTicket(Task, String)}).
|
||||||
*/
|
*/
|
||||||
public PendingAsk pendingAsk(String workerSession) {
|
public PendingAsk pendingAsk(String workerSession, String callerOwner) {
|
||||||
for (Task task : tasks.values()) {
|
for (Task task : tasks.values()) {
|
||||||
Reply q = task.question;
|
Reply q = task.question;
|
||||||
if (q != null && workerSession.equals(task.target)) {
|
if (q != null && workerSession.equals(task.target) && ownsTicket(task, callerOwner)) {
|
||||||
return new PendingAsk(task.ticket, q.text(), q.turnId());
|
return new PendingAsk(task.ticket, q.text(), q.turnId());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
package dev.ltms.fleet.msg;
|
package dev.ltms.fleet.msg;
|
||||||
|
|
||||||
|
import java.util.UUID;
|
||||||
import java.util.concurrent.CompletableFuture;
|
import java.util.concurrent.CompletableFuture;
|
||||||
import java.util.concurrent.ConcurrentHashMap;
|
import java.util.concurrent.ConcurrentHashMap;
|
||||||
import java.util.concurrent.atomic.AtomicLong;
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
@@ -60,7 +61,7 @@ public final class Rendezvous {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/** A worker's open mid-turn question: the worker session it belongs to and the answer future. */
|
/** A worker's open mid-turn question: the worker session it belongs to and the answer future. */
|
||||||
private record AskWaiter(String session, CompletableFuture<String> answer) {
|
private record AskWaiter(String session, CompletableFuture<String> answer, Owner owner) {
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -70,11 +71,44 @@ public final class Rendezvous {
|
|||||||
public record AskTicket(String turnId, CompletableFuture<String> answer, boolean fresh) {
|
public record AskTicket(String turnId, CompletableFuture<String> answer, boolean fresh) {
|
||||||
}
|
}
|
||||||
|
|
||||||
private final ConcurrentHashMap<String, CompletableFuture<Resolution>> waiters = new ConcurrentHashMap<>();
|
/**
|
||||||
|
* The caller whose accepted delegation opened a turn — the only caller allowed to answer it.
|
||||||
|
* A {@code null} owner key means the unnamed primary.
|
||||||
|
*/
|
||||||
|
public record Owner(String ownerKey) {
|
||||||
|
public static final Owner UNNAMED_PRIMARY = new Owner(null);
|
||||||
|
|
||||||
|
public static Owner of(String ownerKey) {
|
||||||
|
return ownerKey == null ? UNNAMED_PRIMARY : new Owner(ownerKey);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Whether {@code callerOwner} matches {@code owner}. A {@code null} owner means no owner was
|
||||||
|
* recorded, so it matches no caller. {@link #UNNAMED_PRIMARY} records the unnamed primary
|
||||||
|
* with an owner object whose key is {@code null}.
|
||||||
|
*/
|
||||||
|
public static boolean permits(Owner owner, String callerOwner) {
|
||||||
|
return owner != null && java.util.Objects.equals(owner.ownerKey(), callerOwner);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** A registered forward waiter together with the owner its delegation was opened under. */
|
||||||
|
private record ForwardWaiter(Owner owner, CompletableFuture<Resolution> future) {
|
||||||
|
}
|
||||||
|
|
||||||
|
private final ConcurrentHashMap<String, ForwardWaiter> waiters = new ConcurrentHashMap<>();
|
||||||
|
|
||||||
/** Reverse rendezvous (CB-205): worker questions awaiting the primary's answer, keyed by {@code turnId}. */
|
/** Reverse rendezvous (CB-205): worker questions awaiting the primary's answer, keyed by {@code turnId}. */
|
||||||
private final ConcurrentHashMap<String, AskWaiter> asks = new ConcurrentHashMap<>();
|
private final ConcurrentHashMap<String, AskWaiter> asks = new ConcurrentHashMap<>();
|
||||||
private final AtomicLong askSeq = new AtomicLong();
|
private final AtomicLong askSeq = new AtomicLong();
|
||||||
|
/**
|
||||||
|
* Minted once per {@code Rendezvous} instance and folded into every {@code turnId} (see
|
||||||
|
* {@link #openAsk(String)}). {@link #askSeq} alone restarts at zero for every instance, so
|
||||||
|
* without this a {@code turnId} minted by one instance could be minted again by another and
|
||||||
|
* resolve to an unrelated ask with no error; this nonce makes that impossible, because an id
|
||||||
|
* minted by one instance can never match the id space of another.
|
||||||
|
*/
|
||||||
|
private final String askBootNonce = UUID.randomUUID().toString().substring(0, 6);
|
||||||
/** Per-session index of the currently-open ask, so duplicate fleet_ask calls coalesce onto one turn. */
|
/** Per-session index of the currently-open ask, so duplicate fleet_ask calls coalesce onto one turn. */
|
||||||
private final ConcurrentHashMap<String, String> openAsksBySession = new ConcurrentHashMap<>();
|
private final ConcurrentHashMap<String, String> openAsksBySession = new ConcurrentHashMap<>();
|
||||||
|
|
||||||
@@ -89,8 +123,17 @@ public final class Rendezvous {
|
|||||||
* code a double open is impossible; this is a tripwire for the day that no longer holds.
|
* code a double open is impossible; this is a tripwire for the day that no longer holds.
|
||||||
*/
|
*/
|
||||||
public CompletableFuture<Resolution> open(String session) {
|
public CompletableFuture<Resolution> open(String session) {
|
||||||
|
return open(session, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Same as {@link #open(String)}, additionally recording {@code owner} as the caller whose
|
||||||
|
* delegation opened this waiter. A {@code null} owner records no owner at all — the
|
||||||
|
* fail-closed default {@link Owner#permits} refuses to everyone.
|
||||||
|
*/
|
||||||
|
public CompletableFuture<Resolution> open(String session, Owner owner) {
|
||||||
CompletableFuture<Resolution> waiter = new CompletableFuture<>();
|
CompletableFuture<Resolution> waiter = new CompletableFuture<>();
|
||||||
CompletableFuture<Resolution> existing = waiters.putIfAbsent(session, waiter);
|
ForwardWaiter existing = waiters.putIfAbsent(session, new ForwardWaiter(owner, waiter));
|
||||||
if (existing != null) {
|
if (existing != null) {
|
||||||
throw new IllegalStateException(
|
throw new IllegalStateException(
|
||||||
"rendezvous double-open for session " + session + " — a waiter is already registered");
|
"rendezvous double-open for session " + session + " — a waiter is already registered");
|
||||||
@@ -105,7 +148,13 @@ public final class Rendezvous {
|
|||||||
* successful {@code open} after a finished turn requires this close to have happened first).
|
* successful {@code open} after a finished turn requires this close to have happened first).
|
||||||
*/
|
*/
|
||||||
public void close(String session, CompletableFuture<Resolution> waiter) {
|
public void close(String session, CompletableFuture<Resolution> waiter) {
|
||||||
waiters.remove(session, waiter);
|
waiters.computeIfPresent(session, (s, w) -> w.future() == waiter ? null : w);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The owner recorded for {@code session}'s open waiter, or {@code null} if none is open. */
|
||||||
|
public Owner ownerOf(String session) {
|
||||||
|
ForwardWaiter w = waiters.get(session);
|
||||||
|
return w == null ? null : w.owner();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Whether a send is currently awaiting a resolution for {@code session}. */
|
/** Whether a send is currently awaiting a resolution for {@code session}. */
|
||||||
@@ -119,7 +168,8 @@ public final class Rendezvous {
|
|||||||
* send (see the CB-116 note above) rather than whichever send happens to be waiting when they fire.
|
* send (see the CB-116 note above) rather than whichever send happens to be waiting when they fire.
|
||||||
*/
|
*/
|
||||||
public CompletableFuture<Resolution> currentWaiter(String session) {
|
public CompletableFuture<Resolution> currentWaiter(String session) {
|
||||||
return waiters.get(session);
|
ForwardWaiter w = waiters.get(session);
|
||||||
|
return w == null ? null : w.future();
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -145,9 +195,9 @@ public final class Rendezvous {
|
|||||||
while (true) {
|
while (true) {
|
||||||
AskWaiter[] minted = { null };
|
AskWaiter[] minted = { null };
|
||||||
String turnId = openAsksBySession.computeIfAbsent(session, _ -> {
|
String turnId = openAsksBySession.computeIfAbsent(session, _ -> {
|
||||||
String newTurnId = session + "#" + askSeq.incrementAndGet();
|
String newTurnId = session + "#" + askBootNonce + "-" + askSeq.incrementAndGet();
|
||||||
CompletableFuture<String> answer = new CompletableFuture<>();
|
CompletableFuture<String> answer = new CompletableFuture<>();
|
||||||
AskWaiter waiter = new AskWaiter(session, answer);
|
AskWaiter waiter = new AskWaiter(session, answer, ownerOf(session));
|
||||||
asks.put(newTurnId, waiter);
|
asks.put(newTurnId, waiter);
|
||||||
minted[0] = waiter;
|
minted[0] = waiter;
|
||||||
return newTurnId;
|
return newTurnId;
|
||||||
@@ -183,6 +233,16 @@ public final class Rendezvous {
|
|||||||
return w == null ? null : w.session();
|
return w == null ? null : w.session();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The owner recorded for {@code turnId} when its ask turn was freshly opened — the caller
|
||||||
|
* whose delegation {@link #answerAsk} must match. {@code null} if {@code turnId} is unknown or
|
||||||
|
* lapsed, or if the ask opened with no forward waiter owner on record.
|
||||||
|
*/
|
||||||
|
public Owner askOwner(String turnId) {
|
||||||
|
AskWaiter w = asks.get(turnId);
|
||||||
|
return w == null ? null : w.owner();
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Resolve a worker's blocked {@code fleet_ask} with the primary's {@code answer}, unblocking it
|
* Resolve a worker's blocked {@code fleet_ask} with the primary's {@code answer}, unblocking it
|
||||||
* to resume its turn.
|
* to resume its turn.
|
||||||
@@ -245,7 +305,7 @@ public final class Rendezvous {
|
|||||||
}
|
}
|
||||||
|
|
||||||
private boolean complete(String session, Resolution resolution) {
|
private boolean complete(String session, Resolution resolution) {
|
||||||
CompletableFuture<Resolution> waiter = waiters.get(session);
|
ForwardWaiter waiter = waiters.get(session);
|
||||||
return waiter != null && waiter.complete(resolution);
|
return waiter != null && waiter.future().complete(resolution);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -44,7 +44,7 @@ import java.util.stream.Collectors;
|
|||||||
* still holds an unacked message ({@link #pendingReplies}), tickets not yet collected
|
* still holds an unacked message ({@link #pendingReplies}), tickets not yet collected
|
||||||
* ({@link #pendingTickets}), and open questions not yet answered or lapsed
|
* ({@link #pendingTickets}), and open questions not yet answered or lapsed
|
||||||
* ({@link #pendingQuestions}) — and sends at most one combined nudge per tick
|
* ({@link #pendingQuestions}) — and sends at most one combined nudge per tick
|
||||||
* ({@link #injectNudge(String, int, int, int)}). Work that arrives while the lead is busy is
|
* ({@link #injectNudge(String, int, int, int, int, int)}). Work that arrives while the lead is busy is
|
||||||
* never lost: it is re-read fresh on every tick until the lead is injectable or its own reminder
|
* never lost: it is re-read fresh on every tick until the lead is injectable or its own reminder
|
||||||
* cap ({@link #maxReminders}) is reached — each source spends from its own budget, so one source
|
* cap ({@link #maxReminders}) is reached — each source spends from its own budget, so one source
|
||||||
* exhausting its cap does not stop nudges about the others (post-CB-590 regression fix; see
|
* exhausting its cap does not stop nudges about the others (post-CB-590 regression fix; see
|
||||||
|
|||||||
@@ -29,8 +29,8 @@ public enum MemberRole {
|
|||||||
* <p>Reads the repo and writes analysis. Never commits code and never opens a pull request —
|
* <p>Reads the repo and writes analysis. Never commits code and never opens a pull request —
|
||||||
* an architect that starts implementing has stopped doing the job that makes it useful.
|
* an architect that starts implementing has stopped doing the job that makes it useful.
|
||||||
*
|
*
|
||||||
* <p>Architects are the one member kind declared in config, because a lead addresses the same
|
* <p>Architects are the one member kind with live slot binding, because a lead addresses the
|
||||||
* slots across many tickets and needs a stable name for them.
|
* same slots across many tickets and needs a stable name for them.
|
||||||
*/
|
*/
|
||||||
ARCHITECT,
|
ARCHITECT,
|
||||||
|
|
||||||
@@ -43,6 +43,14 @@ public enum MemberRole {
|
|||||||
*/
|
*/
|
||||||
DEV,
|
DEV,
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Sweeps an assigned package for defects and reports several ranked findings.
|
||||||
|
*
|
||||||
|
* <p>Never changes code, commits, or opens a pull request. A hunt gathers evidence, which can
|
||||||
|
* include running the build, but leaves every fix to a later implementation unit.
|
||||||
|
*/
|
||||||
|
HUNTER,
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Reviews a diff it did not write and reports one structured finding.
|
* Reviews a diff it did not write and reports one structured finding.
|
||||||
*
|
*
|
||||||
@@ -59,7 +67,7 @@ public enum MemberRole {
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* The {@code fleet:} block that holds this role's pool — {@code architects},
|
* The {@code fleet:} block that holds this role's pool — {@code architects},
|
||||||
* {@code developers}, {@code reviewers}.
|
* {@code developers}, {@code hunters}, {@code reviewers}.
|
||||||
*
|
*
|
||||||
* <p>Plural, and not always the wire name: the pool of things a {@code dev} may run on reads
|
* <p>Plural, and not always the wire name: the pool of things a {@code dev} may run on reads
|
||||||
* naturally as {@code developers:}. The wire name stays the singular {@code dev}, because that
|
* naturally as {@code developers:}. The wire name stays the singular {@code dev}, because that
|
||||||
@@ -69,6 +77,7 @@ public enum MemberRole {
|
|||||||
return switch (this) {
|
return switch (this) {
|
||||||
case ARCHITECT -> "architects";
|
case ARCHITECT -> "architects";
|
||||||
case DEV -> "developers";
|
case DEV -> "developers";
|
||||||
|
case HUNTER -> "hunters";
|
||||||
case REVIEWER -> "reviewers";
|
case REVIEWER -> "reviewers";
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -49,22 +49,52 @@ import java.util.stream.Collectors;
|
|||||||
*/
|
*/
|
||||||
public final class FleetApp {
|
public final class FleetApp {
|
||||||
|
|
||||||
/** The authorization action the matching route handler hands to {@link #allow}. */
|
/**
|
||||||
|
* The authorization action the matching route handler hands to {@link #allow}, for a route
|
||||||
|
* whose action does not depend on the request body.
|
||||||
|
*/
|
||||||
static Authz.Action routeAction(String route) {
|
static Authz.Action routeAction(String route) {
|
||||||
|
return routeAction(route, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* As above, plus the one route whose action depends on the body: {@code POST
|
||||||
|
* /sessions/{id}/message} carries a {@code turnId} (the answer-a-blocked-worker shape) or not
|
||||||
|
* (a plain delivery), mirroring {@code FleetMcp#sendAction}'s split of the same two call
|
||||||
|
* shapes over MCP. {@code turnId} is ignored by every other route.
|
||||||
|
*
|
||||||
|
* @param turnId the request body's {@code turnId}, or {@code null}/blank when absent or not
|
||||||
|
* applicable to this route
|
||||||
|
*/
|
||||||
|
static Authz.Action routeAction(String route, String turnId) {
|
||||||
return switch (route) {
|
return switch (route) {
|
||||||
case "GET /metrics" -> Authz.Action.METRICS;
|
case "GET /metrics" -> Authz.Action.METRICS;
|
||||||
case "POST /members" -> Authz.Action.SPAWN;
|
case "POST /members" -> Authz.Action.SPAWN;
|
||||||
case "DELETE /members/{paneId}" -> Authz.Action.STOP;
|
case "DELETE /members/{paneId}" -> Authz.Action.STOP;
|
||||||
case "POST /sessions/{id}/message" -> Authz.Action.SEND;
|
case "POST /sessions/{id}/message" -> turnId == null || turnId.isBlank()
|
||||||
|
? Authz.Action.SEND : Authz.Action.ANSWER;
|
||||||
case "POST /sessions/{id}/reply" -> Authz.Action.REPLY;
|
case "POST /sessions/{id}/reply" -> Authz.Action.REPLY;
|
||||||
case "GET /sessions/{id}/replies" -> Authz.Action.DRAIN;
|
case "GET /sessions/{id}/replies" -> Authz.Action.DRAIN;
|
||||||
case "POST /sessions/{id}/ask" -> Authz.Action.ASK;
|
case "POST /sessions/{id}/ask" -> Authz.Action.ASK;
|
||||||
case "GET /sessions", "GET /agents", "GET /members", "GET /profiles",
|
case "GET /sessions", "GET /agents", "GET /members", "GET /profiles",
|
||||||
"GET /member-credentials", "GET /sessions/{id}/status", "GET /tasks/{ticket}" -> Authz.Action.READ;
|
"GET /member-credentials" -> Authz.Action.READ;
|
||||||
|
case "GET /sessions/{id}/status", "GET /tasks/{ticket}" -> Authz.Action.TASK_READ;
|
||||||
default -> throw new IllegalArgumentException("route has no authorization gate: " + route);
|
default -> throw new IllegalArgumentException("route has no authorization gate: " + route);
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The second gate for {@code POST /sessions/{id}/message}: checked only when {@code turnId}
|
||||||
|
* is present and non-blank, against {@link Authz.Action#ANSWER}. A request with no {@code
|
||||||
|
* turnId} passes this gate unconditionally, without consulting {@code permit} at all, having
|
||||||
|
* already cleared the coarse {@link Authz.Action#SEND} grant checked ahead of it.
|
||||||
|
*
|
||||||
|
* @param permit reports whether the caller holds the named grant
|
||||||
|
*/
|
||||||
|
static boolean answerGatePasses(String turnId, Predicate<Authz.Action> permit) {
|
||||||
|
return turnId == null || turnId.isBlank() || permit.test(Authz.Action.ANSWER);
|
||||||
|
}
|
||||||
|
|
||||||
/** Default blocking window for a message; kept under typical HTTP idle timeouts. */
|
/** Default blocking window for a message; kept under typical HTTP idle timeouts. */
|
||||||
private static final long DEFAULT_MESSAGE_TIMEOUT_MS = 25_000;
|
private static final long DEFAULT_MESSAGE_TIMEOUT_MS = 25_000;
|
||||||
private static final long MAX_MESSAGE_TIMEOUT_MS = 120_000;
|
private static final long MAX_MESSAGE_TIMEOUT_MS = 120_000;
|
||||||
@@ -94,6 +124,7 @@ public final class FleetApp {
|
|||||||
// .none() (the honest "feature not wired" view) for every constructor that does not pass one.
|
// .none() (the honest "feature not wired" view) for every constructor that does not pass one.
|
||||||
private final FleetMcp.QuarantineSource quarantine;
|
private final FleetMcp.QuarantineSource quarantine;
|
||||||
private final FleetMcp.OutageSource outage;
|
private final FleetMcp.OutageSource outage;
|
||||||
|
private final FleetMcp.LoopHealthSource loopHealth;
|
||||||
private final ObjectMapper mapper = new ObjectMapper();
|
private final ObjectMapper mapper = new ObjectMapper();
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -155,7 +186,8 @@ public final class FleetApp {
|
|||||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials) {
|
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials) {
|
||||||
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics,
|
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics,
|
||||||
deliverable, memberCredentials, FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none());
|
deliverable, memberCredentials, FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none(),
|
||||||
|
FleetMcp.LoopHealthSource.none());
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -169,8 +201,18 @@ public final class FleetApp {
|
|||||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||||
MessageService messages, MemberPresence presence,
|
MessageService messages, MemberPresence presence,
|
||||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials,
|
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials,
|
||||||
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage) {
|
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage) {
|
||||||
|
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics, deliverable,
|
||||||
|
memberCredentials, quarantine, outage, FleetMcp.LoopHealthSource.none());
|
||||||
|
}
|
||||||
|
|
||||||
|
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||||
|
MessageService messages, MemberPresence presence,
|
||||||
|
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||||
|
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials,
|
||||||
|
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage,
|
||||||
|
FleetMcp.LoopHealthSource loopHealth) {
|
||||||
this.herdr = herdr;
|
this.herdr = herdr;
|
||||||
this.memberHerdr = memberHerdr != null ? memberHerdr : herdr;
|
this.memberHerdr = memberHerdr != null ? memberHerdr : herdr;
|
||||||
this.workers = workers;
|
this.workers = workers;
|
||||||
@@ -183,6 +225,7 @@ public final class FleetApp {
|
|||||||
this.memberCredentials = memberCredentials != null ? memberCredentials : MemberCredentialPolicyView::absent;
|
this.memberCredentials = memberCredentials != null ? memberCredentials : MemberCredentialPolicyView::absent;
|
||||||
this.quarantine = quarantine != null ? quarantine : FleetMcp.QuarantineSource.none();
|
this.quarantine = quarantine != null ? quarantine : FleetMcp.QuarantineSource.none();
|
||||||
this.outage = outage != null ? outage : FleetMcp.OutageSource.none();
|
this.outage = outage != null ? outage : FleetMcp.OutageSource.none();
|
||||||
|
this.loopHealth = loopHealth != null ? loopHealth : FleetMcp.LoopHealthSource.none();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Wire routes onto a fresh, unstarted Javalin instance. Caller starts it. */
|
/** Wire routes onto a fresh, unstarted Javalin instance. Caller starts it. */
|
||||||
@@ -223,6 +266,21 @@ public final class FleetApp {
|
|||||||
return app;
|
return app;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The authorization decision behind {@link #allow}, taking the caller directly rather than
|
||||||
|
* pulling it from a servlet {@link Context} — unit-testable without fabricating a live
|
||||||
|
* request, the same reason {@code FleetMcp#denyFor} is split from {@code FleetMcp#deny}.
|
||||||
|
*
|
||||||
|
* @param knownLeadOrCollaborator the classifier a collaborator's {@code SEND} is checked
|
||||||
|
* against; pass {@link #auth}'s own {@code
|
||||||
|
* knownLeadOrCollaborator()} to exercise the real production
|
||||||
|
* gate, as {@link #allow} does
|
||||||
|
*/
|
||||||
|
static boolean permitsFor(Principal caller, Authz.Action action, String target,
|
||||||
|
Predicate<String> knownLeadOrCollaborator) {
|
||||||
|
return Authz.permits(caller, action, target, knownLeadOrCollaborator);
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Gate a handler on the CB-505 authorization table. Returns {@code true} when the request may
|
* Gate a handler on the CB-505 authorization table. Returns {@code true} when the request may
|
||||||
* proceed; otherwise writes the error response and returns {@code false}.
|
* proceed; otherwise writes the error response and returns {@code false}.
|
||||||
@@ -236,8 +294,9 @@ public final class FleetApp {
|
|||||||
return true; // legacy: authorization not enforced
|
return true; // legacy: authorization not enforced
|
||||||
}
|
}
|
||||||
Principal caller = ctx.attribute(CALLER);
|
Principal caller = ctx.attribute(CALLER);
|
||||||
if (Authz.permits(caller, action, target)) {
|
if (permitsFor(caller, action, target, auth.knownLeadOrCollaborator())) {
|
||||||
if (action != Authz.Action.READ && action != Authz.Action.METRICS) {
|
if (action != Authz.Action.READ && action != Authz.Action.METRICS
|
||||||
|
&& action != Authz.Action.TASK_READ) {
|
||||||
AuditLog.allowed(caller, action, target); // reads would drown the trail
|
AuditLog.allowed(caller, action, target); // reads would drown the trail
|
||||||
}
|
}
|
||||||
return true;
|
return true;
|
||||||
@@ -291,15 +350,19 @@ public final class FleetApp {
|
|||||||
* spotted by comparing two numbers by eye.
|
* spotted by comparing two numbers by eye.
|
||||||
*/
|
*/
|
||||||
private void healthz(Context ctx) {
|
private void healthz(Context ctx) {
|
||||||
|
HealthzResponse response = healthzResponse(herdr, memberHerdr, loopHealth);
|
||||||
|
ctx.status(response.status()).json(response.body());
|
||||||
|
}
|
||||||
|
|
||||||
|
record HealthzResponse(int status, Map<String, Object> body) { }
|
||||||
|
|
||||||
|
static HealthzResponse healthzResponse(HerdrClient herdr, HerdrClient memberHerdr,
|
||||||
|
FleetMcp.LoopHealthSource loopHealth) {
|
||||||
JsonNode pong;
|
JsonNode pong;
|
||||||
try {
|
try {
|
||||||
pong = herdr.call("ping");
|
pong = herdr.call("ping");
|
||||||
} catch (HerdrException e) {
|
} catch (HerdrException e) {
|
||||||
ctx.status(503).json(Map.of(
|
return degradedResponse("unreachable", e.getMessage(), loopHealth);
|
||||||
"status", "degraded",
|
|
||||||
"herdr", "unreachable",
|
|
||||||
"detail", e.getMessage()));
|
|
||||||
return;
|
|
||||||
}
|
}
|
||||||
Map<String, Object> body = new LinkedHashMap<>();
|
Map<String, Object> body = new LinkedHashMap<>();
|
||||||
body.put("status", "ok");
|
body.put("status", "ok");
|
||||||
@@ -311,11 +374,11 @@ public final class FleetApp {
|
|||||||
try {
|
try {
|
||||||
memberPong = memberHerdr.call("ping");
|
memberPong = memberHerdr.call("ping");
|
||||||
} catch (HerdrException e) {
|
} catch (HerdrException e) {
|
||||||
ctx.status(503).json(Map.of(
|
return new HealthzResponse(503, Map.of(
|
||||||
"status", "degraded",
|
"status", "degraded",
|
||||||
"herdr", "member unreachable",
|
"herdr", "member unreachable",
|
||||||
"detail", e.getMessage()));
|
"detail", e.getMessage(),
|
||||||
return;
|
"loopHealth", loopHealthView(loopHealth)));
|
||||||
}
|
}
|
||||||
int leadProtocol = pong.path("protocol").asInt();
|
int leadProtocol = pong.path("protocol").asInt();
|
||||||
int memberProtocol = memberPong.path("protocol").asInt();
|
int memberProtocol = memberPong.path("protocol").asInt();
|
||||||
@@ -326,7 +389,20 @@ public final class FleetApp {
|
|||||||
body.put("protocolMismatch", true);
|
body.put("protocolMismatch", true);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
ctx.status(200).json(body);
|
body.put("loopHealth", loopHealthView(loopHealth));
|
||||||
|
return new HealthzResponse(200, body);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Map<String, String> loopHealthView(FleetMcp.LoopHealthSource loopHealth) {
|
||||||
|
return Map.of("statusPoller", loopHealth.statusPoller().get().name(),
|
||||||
|
"sessionReaper", loopHealth.sessionReaper().get().name());
|
||||||
|
}
|
||||||
|
|
||||||
|
private static HealthzResponse degradedResponse(String herdr, String detail,
|
||||||
|
FleetMcp.LoopHealthSource loopHealth) {
|
||||||
|
return new HealthzResponse(503, Map.of(
|
||||||
|
"status", "degraded", "herdr", herdr, "detail", detail,
|
||||||
|
"loopHealth", loopHealthView(loopHealth)));
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -573,26 +649,39 @@ public final class FleetApp {
|
|||||||
* status-gated injector and block until the worker returns a structured {@code fleet_reply}.
|
* status-gated injector and block until the worker returns a structured {@code fleet_reply}.
|
||||||
* Times out with a typed 202 (working / queued / busy) rather than an error — the message may
|
* Times out with a typed 202 (working / queued / busy) rather than an error — the message may
|
||||||
* still land.
|
* still land.
|
||||||
|
*
|
||||||
|
* <p>Two call shapes share this route, exactly as {@code fleet_send} does over MCP (see
|
||||||
|
* {@code FleetMcp#sendAction}): a plain delivery to {@code id}, and -- when the body carries
|
||||||
|
* {@code turnId} -- resolving a worker's blocked question. The coarse {@link
|
||||||
|
* Authz.Action#SEND} grant is checked first, before the body is read at all; only once that
|
||||||
|
* passes is the body parsed, and a present {@code turnId} is then checked again against
|
||||||
|
* {@link Authz.Action#ANSWER}. A body that fails to parse is rejected with 400 and reaches
|
||||||
|
* neither {@code messages.answer} nor {@code messages.send}.
|
||||||
*/
|
*/
|
||||||
private void sendMessage(Context ctx) {
|
private void sendMessage(Context ctx) {
|
||||||
String id = ctx.pathParam("id");
|
String id = ctx.pathParam("id");
|
||||||
if (!allow(ctx, routeAction("POST /sessions/{id}/message"), id)) {
|
if (!allow(ctx, routeAction("POST /sessions/{id}/message"), id)) {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
String content;
|
Principal caller = ctx.attribute(CALLER);
|
||||||
String turnId;
|
String callerOwner = caller == null ? null : caller.ownerKey();
|
||||||
long timeout;
|
JsonNode body;
|
||||||
boolean wait;
|
|
||||||
try {
|
try {
|
||||||
JsonNode body = mapper.readTree(ctx.body());
|
body = mapper.readTree(ctx.body());
|
||||||
content = body.path("content").asText("");
|
|
||||||
turnId = body.path("turnId").asText(null);
|
|
||||||
timeout = body.path("timeoutMs").asLong(DEFAULT_MESSAGE_TIMEOUT_MS);
|
|
||||||
wait = body.path("wait").asBoolean(true); // default: block for the reply (CB-104)
|
|
||||||
} catch (Exception e) {
|
} catch (Exception e) {
|
||||||
|
body = null;
|
||||||
|
}
|
||||||
|
if (body == null) {
|
||||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "body must be JSON"));
|
ctx.status(400).json(Map.of("error", "bad_request", "detail", "body must be JSON"));
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
String turnId = body.path("turnId").asText(null);
|
||||||
|
if (!answerGatePasses(turnId, action -> allow(ctx, action, id))) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
String content = body.path("content").asText("");
|
||||||
|
long timeout = body.path("timeoutMs").asLong(DEFAULT_MESSAGE_TIMEOUT_MS);
|
||||||
|
boolean wait = body.path("wait").asBoolean(true); // default: block for the reply (CB-104)
|
||||||
if (content.isBlank()) {
|
if (content.isBlank()) {
|
||||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "content is required"));
|
ctx.status(400).json(Map.of("error", "bad_request", "detail", "content is required"));
|
||||||
return;
|
return;
|
||||||
@@ -601,19 +690,19 @@ public final class FleetApp {
|
|||||||
|
|
||||||
// Answering a worker's fleet_ask (CB-205): always blocks, and derives the worker from turnId.
|
// Answering a worker's fleet_ask (CB-205): always blocks, and derives the worker from turnId.
|
||||||
if (turnId != null && !turnId.isBlank()) {
|
if (turnId != null && !turnId.isBlank()) {
|
||||||
writeReply(ctx, id, messages.answer(turnId, content, timeout), timeout);
|
writeReply(ctx, id, messages.answer(turnId, content, timeout, callerOwner), timeout);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (!wait) {
|
if (!wait) {
|
||||||
// Fire-and-poll (CB-107): return a ticket immediately; the caller polls GET /tasks/{ticket}.
|
// Fire-and-poll (CB-107): return a ticket immediately; the caller polls GET /tasks/{ticket}.
|
||||||
String ticket = messages.sendAsync(id, content);
|
String ticket = messages.sendAsync(id, content, null, caller);
|
||||||
ctx.status(202).json(Map.of("sessionId", id, "ticket", ticket, "status", "accepted"));
|
ctx.status(202).json(Map.of("sessionId", id, "ticket", ticket, "status", "accepted"));
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
try {
|
try {
|
||||||
writeReply(ctx, id, messages.send(id, content, timeout), timeout);
|
writeReply(ctx, id, messages.send(id, content, timeout, callerOwner), timeout);
|
||||||
} catch (HerdrException e) {
|
} catch (HerdrException e) {
|
||||||
herdrError(ctx, e);
|
herdrError(ctx, e);
|
||||||
}
|
}
|
||||||
@@ -632,6 +721,10 @@ public final class FleetApp {
|
|||||||
case STALE_TURN -> ctx.status(409).json(Map.of(
|
case STALE_TURN -> ctx.status(409).json(Map.of(
|
||||||
"sessionId", id, "error", "stale_turn",
|
"sessionId", id, "error", "stale_turn",
|
||||||
"detail", "that question is no longer open (timed out or already answered)"));
|
"detail", "that question is no longer open (timed out or already answered)"));
|
||||||
|
case NOT_TURN_OWNER -> ctx.status(403).json(Map.of(
|
||||||
|
"sessionId", id, "error", "not_turn_owner",
|
||||||
|
"detail", "this turn belongs to a different delegation — only the caller that "
|
||||||
|
+ "opened it may answer it"));
|
||||||
case REPLIED, COMPLETED_UNREPLIED -> {
|
case REPLIED, COMPLETED_UNREPLIED -> {
|
||||||
// replySource distinguishes a structured fleet_reply from the CB-106 completion
|
// replySource distinguishes a structured fleet_reply from the CB-106 completion
|
||||||
// fallback (a scrape of the worker's transcript when it finished without replying).
|
// fallback (a scrape of the worker's transcript when it finished without replying).
|
||||||
@@ -640,15 +733,32 @@ public final class FleetApp {
|
|||||||
}
|
}
|
||||||
default -> ctx.status(202).json(Map.of(
|
default -> ctx.status(202).json(Map.of(
|
||||||
"sessionId", id,
|
"sessionId", id,
|
||||||
|
// fleetd #571 (ticket comment 17126): no `default` here on purpose. This switch
|
||||||
|
// is an expression, so the compiler already demands every Outcome constant have
|
||||||
|
// an arm — adding an 11th constant to Outcome is a compile error here, not a
|
||||||
|
// silent fall-through. That is exactly the bug this ticket exists to fix:
|
||||||
|
// `default -> "done"` used to sit here and would have told a REST caller the
|
||||||
|
// delegation completed for TIMED_OUT_UNCONFIRMED, the one outcome where delivery
|
||||||
|
// is unknown. REPLIED, COMPLETED_UNREPLIED, QUESTION, STALE_TURN and
|
||||||
|
// NOT_TURN_OWNER can never actually reach this inner switch — the outer switch
|
||||||
|
// above always dispatches them first — but they still need an arm to keep this
|
||||||
|
// switch exhaustive.
|
||||||
"status", switch (reply.outcome()) {
|
"status", switch (reply.outcome()) {
|
||||||
case TIMED_OUT_WORKING -> "working";
|
case TIMED_OUT_WORKING -> "working";
|
||||||
case TIMED_OUT_QUEUED -> "queued";
|
case TIMED_OUT_QUEUED -> "queued";
|
||||||
|
// Delivery here is unknown, not merely still queued — see
|
||||||
|
// Outcome#TIMED_OUT_UNCONFIRMED's own javadoc.
|
||||||
|
case TIMED_OUT_UNCONFIRMED -> "unconfirmed";
|
||||||
case BUSY -> "busy";
|
case BUSY -> "busy";
|
||||||
case WORKER_FAILED -> "failed";
|
case WORKER_FAILED -> "failed";
|
||||||
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
||||||
default -> "done"; // unreachable (terminal outcomes handled above)
|
case REPLIED, COMPLETED_UNREPLIED, QUESTION, STALE_TURN, NOT_TURN_OWNER -> "done"; // unreachable
|
||||||
},
|
},
|
||||||
"detail", (reply.outcome() == MessageService.Outcome.WORKER_FAILED
|
"detail", reply.outcome() == MessageService.Outcome.TIMED_OUT_UNCONFIRMED
|
||||||
|
? "no reply within " + timeout + "ms; delivery is unconfirmed — the "
|
||||||
|
+ "message may already have reached the worker, so a resend "
|
||||||
|
+ "risks sending it twice; poll status first"
|
||||||
|
: (reply.outcome() == MessageService.Outcome.WORKER_FAILED
|
||||||
|| reply.outcome() == MessageService.Outcome.BACKEND_EXHAUSTED)
|
|| reply.outcome() == MessageService.Outcome.BACKEND_EXHAUSTED)
|
||||||
&& reply.text() != null
|
&& reply.text() != null
|
||||||
? reply.text()
|
? reply.text()
|
||||||
@@ -771,10 +881,12 @@ public final class FleetApp {
|
|||||||
body.put("sessionId", id);
|
body.put("sessionId", id);
|
||||||
body.put("status", messages.status(id).name().toLowerCase());
|
body.put("status", messages.status(id).name().toLowerCase());
|
||||||
body.put("ready", deliverable.test(id));
|
body.put("ready", deliverable.test(id));
|
||||||
// CB-582: a worker paused mid-turn in an async fleet_ask is otherwise invisible to a
|
// A worker paused mid-turn in an async fleet_ask is otherwise invisible to a status
|
||||||
// status poll — surface the open question and how to answer it, same as fleet_poll's
|
// poll — surface the open question and how to answer it, same as fleet_poll's
|
||||||
// Phase.ASKING view.
|
// Phase.ASKING view, but only to the caller whose owner key created that delegation, or
|
||||||
MessageService.PendingAsk ask = messages.pendingAsk(id);
|
// to the unnamed primary.
|
||||||
|
Principal caller = ctx.attribute(CALLER);
|
||||||
|
MessageService.PendingAsk ask = messages.pendingAsk(id, caller == null ? null : caller.ownerKey());
|
||||||
if (ask != null) {
|
if (ask != null) {
|
||||||
body.put("question", ask.question());
|
body.put("question", ask.question());
|
||||||
body.put("turnId", ask.turnId());
|
body.put("turnId", ask.turnId());
|
||||||
@@ -791,7 +903,8 @@ public final class FleetApp {
|
|||||||
if (!allow(ctx, routeAction("GET /tasks/{ticket}"), null)) {
|
if (!allow(ctx, routeAction("GET /tasks/{ticket}"), null)) {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
MessageService.TaskView v = messages.poll(ctx.pathParam("ticket"));
|
Principal caller = ctx.attribute(CALLER);
|
||||||
|
MessageService.TaskView v = messages.poll(ctx.pathParam("ticket"), caller == null ? null : caller.ownerKey());
|
||||||
if (v == null) {
|
if (v == null) {
|
||||||
ctx.status(404).json(Map.of("error", "unknown_ticket", "detail", "no such task (or it has expired)"));
|
ctx.status(404).json(Map.of("error", "unknown_ticket", "detail", "no such task (or it has expired)"));
|
||||||
return;
|
return;
|
||||||
|
|||||||
@@ -26,6 +26,7 @@ import java.util.concurrent.atomic.AtomicBoolean;
|
|||||||
import java.util.concurrent.atomic.AtomicLong;
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
import java.util.function.Consumer;
|
import java.util.function.Consumer;
|
||||||
import java.util.function.LongSupplier;
|
import java.util.function.LongSupplier;
|
||||||
|
import java.util.function.Supplier;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Authoritative in-daemon registry of the worker sessions this {@code fleetd} process spawned.
|
* Authoritative in-daemon registry of the worker sessions this {@code fleetd} process spawned.
|
||||||
@@ -57,6 +58,18 @@ public final class SessionManager implements TurnListener {
|
|||||||
* Populated on every spawn path, removed on {@link #release}.
|
* Populated on every spawn path, removed on {@link #release}.
|
||||||
*/
|
*/
|
||||||
private final ConcurrentHashMap<String /*paneId*/, PeerHandle> handles = new ConcurrentHashMap<>();
|
private final ConcurrentHashMap<String /*paneId*/, PeerHandle> handles = new ConcurrentHashMap<>();
|
||||||
|
/**
|
||||||
|
* fleetd #702: a pane mid-teardown, keyed by paneId, held from just before its registry entry
|
||||||
|
* is removed until {@link #releaseRemoved} finishes. {@link #spawnedMemberRole} consults this
|
||||||
|
* alongside the registry, so a caller resolving the pane's terminal during that window still
|
||||||
|
* sees a live member and never falls through to a tab map.
|
||||||
|
*
|
||||||
|
* <p>Depth-counted rather than a plain set: two threads can be tearing down the same pane at
|
||||||
|
* once (the CAS in {@link #releaseIfCurrent} exists for exactly that race), and with a set the
|
||||||
|
* loser's {@code finally} would unmark the pane while the winner is still mid-teardown,
|
||||||
|
* reopening the window this exists to close.
|
||||||
|
*/
|
||||||
|
private final ConcurrentHashMap<String /*paneId*/, Releasing> releasing = new ConcurrentHashMap<>();
|
||||||
private final MemberPresence presence;
|
private final MemberPresence presence;
|
||||||
private final SecureRandom nonceRandom = new SecureRandom();
|
private final SecureRandom nonceRandom = new SecureRandom();
|
||||||
private final AtomicLong nonceSeq = new AtomicLong();
|
private final AtomicLong nonceSeq = new AtomicLong();
|
||||||
@@ -242,6 +255,10 @@ public final class SessionManager implements TurnListener {
|
|||||||
handle.id(), handle.terminalId(), resolvedProfile, actualRole, cwd, ownerTerminal, now, now, 0,
|
handle.id(), handle.terminalId(), resolvedProfile, actualRole, cwd, ownerTerminal, now, now, 0,
|
||||||
MemberSession.State.SPAWNING, null, null, handle.charterReceipt(), handle.agentSessionId());
|
MemberSession.State.SPAWNING, null, null, handle.charterReceipt(), handle.agentSessionId());
|
||||||
registry.put(handle.id(), session);
|
registry.put(handle.id(), session);
|
||||||
|
// A presence contact that already arrived for this terminal found no registry
|
||||||
|
// entry to transition and gave up silently. Retry it now that one exists; remove
|
||||||
|
// this call and such a session stays in SPAWNING even though it is present.
|
||||||
|
reconcilePresence(handle.terminalId());
|
||||||
handles.put(handle.id(), handle);
|
handles.put(handle.id(), handle);
|
||||||
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
||||||
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
||||||
@@ -302,9 +319,12 @@ public final class SessionManager implements TurnListener {
|
|||||||
* with no copy and no error. Do NOT fuse these back together; the cost of an orphaned worktree
|
* with no copy and no error. Do NOT fuse these back together; the cost of an orphaned worktree
|
||||||
* is a logged path an operator can reclaim, the cost of a deleted one is unrecoverable work.
|
* is a logged path an operator can reclaim, the cost of a deleted one is unrecoverable work.
|
||||||
*/
|
*/
|
||||||
private void release(String paneId, ReleaseCause cause) {
|
private MemberSession release(String paneId, ReleaseCause cause) {
|
||||||
MemberSession removed = registry.remove(paneId);
|
return releaseWindow(paneId, registry.get(paneId), () -> {
|
||||||
releaseRemoved(paneId, removed, handles.remove(paneId), cause);
|
MemberSession removed = registry.remove(paneId);
|
||||||
|
releaseRemoved(paneId, removed, handles.remove(paneId), cause);
|
||||||
|
return removed;
|
||||||
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -313,16 +333,85 @@ public final class SessionManager implements TurnListener {
|
|||||||
* DONE record from stopping a worker that delivery has made BUSY.
|
* DONE record from stopping a worker that delivery has made BUSY.
|
||||||
*/
|
*/
|
||||||
private boolean releaseIfCurrent(MemberSession expected, ReleaseCause cause) {
|
private boolean releaseIfCurrent(MemberSession expected, ReleaseCause cause) {
|
||||||
if (!registry.remove(expected.paneId(), expected)) {
|
return releaseWindow(expected.paneId(), expected, () -> {
|
||||||
// A lifecycle transition replaced the record between the caller's check and this remove.
|
if (!registry.remove(expected.paneId(), expected)) {
|
||||||
// Log it: this race is by definition unobservable otherwise, and a reaper that silently
|
// A lifecycle transition replaced the record between the caller's check and this
|
||||||
// declines to reap is the hardest kind of behaviour to diagnose after the fact.
|
// remove. Log it: this race is by definition unobservable otherwise, and a reaper
|
||||||
log.debug("skipping reap of pane={}: its registry record changed after the idle check "
|
// that silently declines to reap is the hardest kind of behaviour to diagnose
|
||||||
+ "(most likely a delivery made it BUSY)", expected.paneId());
|
// after the fact.
|
||||||
return false;
|
log.debug("skipping reap of pane={}: its registry record changed after the idle "
|
||||||
|
+ "check (most likely a delivery made it BUSY)", expected.paneId());
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
releaseRemoved(expected.paneId(), expected, handles.remove(expected.paneId()), cause);
|
||||||
|
return true;
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #702: mark {@code paneId} as mid-teardown — using {@code known}'s terminal/role when
|
||||||
|
* it is available — for the whole of {@code teardown}, which removes the registry entry and
|
||||||
|
* then runs {@link #releaseRemoved}. Shared by both registry-removal sites ({@link #release}'s
|
||||||
|
* unconditional remove and {@link #releaseIfCurrent}'s CAS remove) so neither can leave the
|
||||||
|
* other's window unmarked.
|
||||||
|
*
|
||||||
|
* <p>The mark is written before {@code teardown} runs — so it covers the removal itself, not
|
||||||
|
* only what comes after it — and cleared in a {@code finally}, so an unchecked throw out of
|
||||||
|
* {@code teardown} (including one from {@link PeerLauncher#stop}, which declares nothing) can
|
||||||
|
* never leave the pane marked for the rest of the daemon's life.
|
||||||
|
*/
|
||||||
|
private <T> T releaseWindow(String paneId, MemberSession known, Supplier<T> teardown) {
|
||||||
|
releasing.compute(paneId, (_, prior) -> Releasing.enter(prior, known));
|
||||||
|
try {
|
||||||
|
return teardown.get();
|
||||||
|
} finally {
|
||||||
|
releasing.compute(paneId, (_, prior) -> prior == null ? null : prior.leave());
|
||||||
}
|
}
|
||||||
releaseRemoved(expected.paneId(), expected, handles.remove(expected.paneId()), cause);
|
}
|
||||||
return true;
|
|
||||||
|
/**
|
||||||
|
* Depth count plus the terminal/role a mid-teardown pane belongs to, for
|
||||||
|
* {@link #spawnedMemberRole}. The terminal/role come from whichever call into
|
||||||
|
* {@link #releaseWindow} first knew them: a call that finds the registry entry already gone
|
||||||
|
* passes a {@code null} session, and must not blank out what the first call recorded.
|
||||||
|
*/
|
||||||
|
record Releasing(int depth, String terminalId, MemberRole role) {
|
||||||
|
static Releasing enter(Releasing prior, MemberSession known) {
|
||||||
|
int depth = (prior == null ? 0 : prior.depth()) + 1;
|
||||||
|
String terminalId = known != null ? known.terminalId() : prior == null ? null : prior.terminalId();
|
||||||
|
MemberRole role = known != null ? known.role() : prior == null ? null : prior.role();
|
||||||
|
return new Releasing(depth, terminalId, role);
|
||||||
|
}
|
||||||
|
|
||||||
|
Releasing leave() {
|
||||||
|
return depth <= 1 ? null : new Releasing(depth - 1, terminalId, role);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The role of the live spawned member occupying {@code terminal} — whether it is currently in
|
||||||
|
* the registry, or mid-teardown between {@link #release} removing its registry entry and
|
||||||
|
* {@link #releaseRemoved} actually stopping its pane (fleetd #702). {@code null} for a terminal
|
||||||
|
* that is neither: this method is the one reader a caller resolver consults before any tab
|
||||||
|
* map, so a live or releasing member's identity never falls back to a tab label.
|
||||||
|
*
|
||||||
|
* <p>Checks the registry directly via {@link #findByTerminal} rather than {@link #roster()},
|
||||||
|
* so this hot-path lookup (consulted on every resolve) never pays for a list copy or a stream.
|
||||||
|
*/
|
||||||
|
public MemberRole spawnedMemberRole(String terminal) {
|
||||||
|
MemberSession session = findByTerminal(terminal);
|
||||||
|
if (session != null) {
|
||||||
|
return session.role();
|
||||||
|
}
|
||||||
|
if (terminal == null) {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
for (Releasing r : releasing.values()) {
|
||||||
|
if (terminal.equals(r.terminalId())) {
|
||||||
|
return r.role();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return null;
|
||||||
}
|
}
|
||||||
|
|
||||||
private void releaseRemoved(String paneId, MemberSession removed, PeerHandle removedHandle,
|
private void releaseRemoved(String paneId, MemberSession removed, PeerHandle removedHandle,
|
||||||
@@ -381,6 +470,12 @@ public final class SessionManager implements TurnListener {
|
|||||||
MemberSession resolved = resolveAgentSessionId(removed, removedHandle);
|
MemberSession resolved = resolveAgentSessionId(removed, removedHandle);
|
||||||
notifyReleased(new ReleaseDetail(resolved.terminalId(), resolved.worktree(),
|
notifyReleased(new ReleaseDetail(resolved.terminalId(), resolved.worktree(),
|
||||||
resolved.branch(), snapshotRef, resolved.agentSessionId()));
|
resolved.branch(), snapshotRef, resolved.agentSessionId()));
|
||||||
|
String terminal = removed.terminalId();
|
||||||
|
if (terminal != null && !terminal.isBlank()) {
|
||||||
|
// Without this, a terminal stays marked present after its pane is gone, so a
|
||||||
|
// later send to the same id would read as deliverable instead of refused.
|
||||||
|
presence.forget(terminal);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// CB-581: the pane must always stop, even if the dirty check above threw. A session removed
|
// CB-581: the pane must always stop, even if the dirty check above threw. A session removed
|
||||||
@@ -724,6 +819,10 @@ public final class SessionManager implements TurnListener {
|
|||||||
handle.charterReceipt(),
|
handle.charterReceipt(),
|
||||||
handle.agentSessionId());
|
handle.agentSessionId());
|
||||||
registry.put(handle.id(), session);
|
registry.put(handle.id(), session);
|
||||||
|
// A presence contact that already arrived for this terminal found no registry entry to
|
||||||
|
// transition and gave up silently. Retry it now that one exists; remove this call and
|
||||||
|
// such a session stays in SPAWNING even though it is present.
|
||||||
|
reconcilePresence(handle.terminalId());
|
||||||
handles.put(handle.id(), handle);
|
handles.put(handle.id(), handle);
|
||||||
log.debug("acquired worktree session id={} terminal={} profile={} branch={} path={}",
|
log.debug("acquired worktree session id={} terminal={} profile={} branch={} path={}",
|
||||||
handle.id(), handle.terminalId(), session.profile(), session.branch(), session.worktree());
|
handle.id(), handle.terminalId(), session.profile(), session.branch(), session.worktree());
|
||||||
@@ -870,10 +969,23 @@ public final class SessionManager implements TurnListener {
|
|||||||
// digest lets a lead tell at a glance whether all members got the same charter; the source
|
// digest lets a lead tell at a glance whether all members got the same charter; the source
|
||||||
// records whether a role charter was configured ("fleet.charters.<role>") or only the reply
|
// records whether a role charter was configured ("fleet.charters.<role>") or only the reply
|
||||||
// charter was composed ("none").
|
// charter was composed ("none").
|
||||||
|
// #604: charterBytes rides along with charterSha256, not with charterSource — it is only
|
||||||
|
// meaningful as the digest's companion (a length turns "they differ" into "by how much").
|
||||||
|
// A member with no composed charter reports charterSource and nothing else, as before.
|
||||||
|
//
|
||||||
|
// Gating on the DIGEST rather than on the receipt is deliberate, and the two are not
|
||||||
|
// always null together. CharterReceipt.compose() derives the digest with digestOf(), which
|
||||||
|
// returns null for BLANK text, while the byte count is composed.getBytes().length, which
|
||||||
|
// does not. So a whitespace-only charter (a blank fleet.charters.<role> on a profile with
|
||||||
|
// no MCP, so no reply charter is appended) yields a null digest beside a non-zero size.
|
||||||
|
// Reporting a size with no digest would say "they differ by N bytes" about an artifact we
|
||||||
|
// cannot fingerprint, so this gate omits both. Never widen it to the receipt-level null
|
||||||
|
// check without deciding what that case should report.
|
||||||
if (session.charterReceipt() != null) {
|
if (session.charterReceipt() != null) {
|
||||||
m.put("charterSource", session.charterReceipt().charterSource());
|
m.put("charterSource", session.charterReceipt().charterSource());
|
||||||
if (session.charterReceipt().charterSha256() != null) {
|
if (session.charterReceipt().charterSha256() != null) {
|
||||||
m.put("charterSha256", session.charterReceipt().charterSha256());
|
m.put("charterSha256", session.charterReceipt().charterSha256());
|
||||||
|
m.put("charterBytes", session.charterReceipt().charterBytes());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
m.put("liveStatus", live == null ? "unknown" : live.status().name().toLowerCase());
|
m.put("liveStatus", live == null ? "unknown" : live.status().name().toLowerCase());
|
||||||
@@ -885,6 +997,20 @@ public final class SessionManager implements TurnListener {
|
|||||||
transitionByTerminal(terminalId, MemberSession.State.SPAWNING, MemberSession.State.READY);
|
transitionByTerminal(terminalId, MemberSession.State.SPAWNING, MemberSession.State.READY);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Completes a newly registered session's {@code SPAWNING -> READY} transition when {@code
|
||||||
|
* terminalId} was already marked present before this ran. A terminal never marked present is
|
||||||
|
* left in {@code SPAWNING}; it reaches {@code READY} normally through {@link #onReady} once
|
||||||
|
* its own contact arrives. Callers must run this only once the session's registry entry is
|
||||||
|
* already visible — {@link #onReady}'s transition matches against that entry, and reconciling
|
||||||
|
* before the entry exists finds nothing to transition.
|
||||||
|
*/
|
||||||
|
private void reconcilePresence(String terminalId) {
|
||||||
|
if (terminalId != null && !terminalId.isBlank() && presence.isPresent(terminalId)) {
|
||||||
|
onReady(terminalId);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Lifecycle hook: a message was delivered into the worker — it is now busy on a turn.
|
* Lifecycle hook: a message was delivered into the worker — it is now busy on a turn.
|
||||||
* The turn count is bumped and the activity timestamp is refreshed. A {@code DONE} session
|
* The turn count is bumped and the activity timestamp is refreshed. A {@code DONE} session
|
||||||
@@ -1064,16 +1190,62 @@ public final class SessionManager implements TurnListener {
|
|||||||
* drain (see above), and a straggler must not buy the drain more time than the flag it lost the
|
* drain (see above), and a straggler must not buy the drain more time than the flag it lost the
|
||||||
* race against would have. In the ordinary case the sweep finds nothing and costs one empty
|
* race against would have. In the ordinary case the sweep finds nothing and costs one empty
|
||||||
* {@link #roster()} call.
|
* {@link #roster()} call.
|
||||||
|
*
|
||||||
|
* <p>fleetd #512: a drain that releases every session cleanly used to log nothing at all — the
|
||||||
|
* only log calls in this method and {@link #drainSnapshot} sit on abnormal paths, so "nothing
|
||||||
|
* logged" was indistinguishable from "died on the first session". The {@code log.info} at the
|
||||||
|
* end below is a positive assertion that the drain actually finished, on the normal path,
|
||||||
|
* every time — including the all-zero case, which is a common and legitimate outcome (no
|
||||||
|
* members were live) and must still produce the line. Both {@link #drainSnapshot} passes (the
|
||||||
|
* main snapshot and the straggler sweep) are folded into the one line: a caller reading two
|
||||||
|
* lines could not tell a two-pass drain from two separate drains.
|
||||||
|
*
|
||||||
|
* <p><strong>Non-goal: this line must never move into a {@code finally} block, and this method
|
||||||
|
* must never grow one around it.</strong> "Every time" above means every time the drain
|
||||||
|
* <em>finishes</em>, not every time this method exits. The absence of the line is the signal
|
||||||
|
* that the drain died, so a {@code finally} would destroy the signal and print confident
|
||||||
|
* partial counts in the same edit — the line would appear after a drain that threw, carrying
|
||||||
|
* whatever {@code tally} it had reached. Both halves of the value are lost at once. The line
|
||||||
|
* has to be the last statement of the successful path and reachable only from it.
|
||||||
|
*
|
||||||
|
* <p>This is written down because it is the obvious review comment ("shouldn't we always log
|
||||||
|
* the drain result?"), it sounds like thoroughness, and the paragraph above reads as an
|
||||||
|
* invitation to it. Raised by the fleet01 lead on 2026-09-12, from their 2026-09-10 incident:
|
||||||
|
* we only know that drain died on that host because it <em>threw</em>, and a
|
||||||
|
* {@code NoClassDefFoundError} reached the JVM's uncaught handler. A drain that hung on one
|
||||||
|
* session, or returned early on a condition rather than an exception, would leave no stack
|
||||||
|
* trace, no {@code ERROR} token and no priority — only a missing line. That makes the loud
|
||||||
|
* variant the one we have seen and the quiet variants the ones this line exists to catch.
|
||||||
|
*
|
||||||
|
* <p>Related: {@code released} and {@code abandoned} are counted incrementally inside {@link
|
||||||
|
* #drainSnapshot}'s loop and folded with {@link DrainTally#plus}, rather than derived from a
|
||||||
|
* collection read at the end, for the same reason. If a partial report is ever wanted it must
|
||||||
|
* be a different line with a different verb. One line must not serve both, or a reader cannot
|
||||||
|
* tell a finished drain from an interrupted one by its wording.
|
||||||
*/
|
*/
|
||||||
void drainAll(long timeoutNanos) {
|
void drainAll(long timeoutNanos) {
|
||||||
long deadline = System.nanoTime() + timeoutNanos;
|
long deadline = System.nanoTime() + timeoutNanos;
|
||||||
draining.set(true);
|
draining.set(true);
|
||||||
drainSnapshot(roster(), deadline);
|
DrainTally tally = drainSnapshot(roster(), deadline);
|
||||||
List<MemberSession> stragglers = roster();
|
List<MemberSession> stragglers = roster();
|
||||||
if (!stragglers.isEmpty()) {
|
if (!stragglers.isEmpty()) {
|
||||||
log.warn("drain sweep found {} session(s) registered after the drain snapshot was "
|
log.warn("drain sweep found {} session(s) registered after the drain snapshot was "
|
||||||
+ "taken (raced past the shutdown guard); draining them too", stragglers.size());
|
+ "taken (raced past the shutdown guard); draining them too", stragglers.size());
|
||||||
drainSnapshot(stragglers, deadline);
|
tally = tally.plus(drainSnapshot(stragglers, deadline));
|
||||||
|
}
|
||||||
|
log.info("drain complete: released={} abandoned={} (still BUSY at the shutdown deadline)",
|
||||||
|
tally.released(), tally.abandoned());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Running count for one {@link #drainAll} invocation, folded across both {@link #drainSnapshot}
|
||||||
|
* passes (fleetd #512). {@code abandoned} counts sessions that were still {@code BUSY} at the
|
||||||
|
* moment they were released — i.e. the whole-drain deadline passed before they left {@code BUSY}
|
||||||
|
* on their own (see {@link #drainSnapshot}) — a subset of {@code released}, not additional to it.
|
||||||
|
*/
|
||||||
|
private record DrainTally(int released, int abandoned) {
|
||||||
|
private DrainTally plus(DrainTally other) {
|
||||||
|
return new DrainTally(released + other.released, abandoned + other.abandoned);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1081,8 +1253,12 @@ public final class SessionManager implements TurnListener {
|
|||||||
* Drain exactly the sessions in {@code snapshot}, waiting out a {@code BUSY} one against the
|
* Drain exactly the sessions in {@code snapshot}, waiting out a {@code BUSY} one against the
|
||||||
* shared whole-drain {@code deadline} before releasing it. Shared by {@link #drainAll}'s main
|
* shared whole-drain {@code deadline} before releasing it. Shared by {@link #drainAll}'s main
|
||||||
* pass and its post-loop straggler sweep (fleetd #308) so both honor the same one budget.
|
* pass and its post-loop straggler sweep (fleetd #308) so both honor the same one budget.
|
||||||
|
* Returns how many sessions this pass released, and how many of those were still {@code BUSY}
|
||||||
|
* (abandoned mid-turn) at the moment of release.
|
||||||
*/
|
*/
|
||||||
private void drainSnapshot(List<MemberSession> snapshot, long deadline) {
|
private DrainTally drainSnapshot(List<MemberSession> snapshot, long deadline) {
|
||||||
|
int released = 0;
|
||||||
|
int abandoned = 0;
|
||||||
for (MemberSession s : snapshot) {
|
for (MemberSession s : snapshot) {
|
||||||
try {
|
try {
|
||||||
if (s.state() == MemberSession.State.BUSY) {
|
if (s.state() == MemberSession.State.BUSY) {
|
||||||
@@ -1100,11 +1276,16 @@ public final class SessionManager implements TurnListener {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
release(s.paneId(), ReleaseCause.SHUTDOWN);
|
MemberSession removed = release(s.paneId(), ReleaseCause.SHUTDOWN);
|
||||||
|
released++;
|
||||||
|
if (removed != null && removed.state() == MemberSession.State.BUSY) {
|
||||||
|
abandoned++;
|
||||||
|
}
|
||||||
} catch (RuntimeException e) {
|
} catch (RuntimeException e) {
|
||||||
log.warn("drain failed for pane={}; continuing with remaining sessions", s.paneId(), e);
|
log.warn("drain failed for pane={}; continuing with remaining sessions", s.paneId(), e);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
return new DrainTally(released, abandoned);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
@@ -1,9 +1,11 @@
|
|||||||
package dev.ltms.fleet.session;
|
package dev.ltms.fleet.session;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import java.util.concurrent.TimeUnit;
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Periodic virtual-thread reaper that tears down {@code READY}/{@code DONE} sessions which have
|
* Periodic virtual-thread reaper that tears down {@code READY}/{@code DONE} sessions which have
|
||||||
@@ -14,6 +16,16 @@ public final class SessionReaper {
|
|||||||
|
|
||||||
private static final Logger log = LoggerFactory.getLogger(SessionReaper.class);
|
private static final Logger log = LoggerFactory.getLogger(SessionReaper.class);
|
||||||
private static final long DEFAULT_INTERVAL_MILLIS = 5000;
|
private static final long DEFAULT_INTERVAL_MILLIS = 5000;
|
||||||
|
/**
|
||||||
|
* How many multiples of {@code intervalMillis} the last-completed round may age before
|
||||||
|
* {@link #health()} reports {@link LoopWatchdog.State#STALLED} (fleetd #544). One round here
|
||||||
|
* is {@code sessions.reapIdle} plus the (rare, best-effort) WIP sweep — both normally finish
|
||||||
|
* in a small fraction of one interval. At the default 5s interval this puts the threshold at
|
||||||
|
* 60s: generous enough that an occasional slow git call in the WIP sweep never trips it, short
|
||||||
|
* enough that a genuinely wedged reap round (mirroring the herdr-read-with-no-deadline hang
|
||||||
|
* {@code StatusPoller} guards against — see fleetd #544) is caught inside about a minute.
|
||||||
|
*/
|
||||||
|
private static final long STALE_AFTER_INTERVAL_MULTIPLIER = 12;
|
||||||
/** CB-586: the refs/wip age floor — never sweep a snapshot younger than 24h (the CB-586 rule). */
|
/** CB-586: the refs/wip age floor — never sweep a snapshot younger than 24h (the CB-586 rule). */
|
||||||
private static final long WIP_MIN_AGE_MILLIS = TimeUnit.HOURS.toMillis(24);
|
private static final long WIP_MIN_AGE_MILLIS = TimeUnit.HOURS.toMillis(24);
|
||||||
/**
|
/**
|
||||||
@@ -25,6 +37,7 @@ public final class SessionReaper {
|
|||||||
private final SessionManager sessions;
|
private final SessionManager sessions;
|
||||||
private final long idleTtlNanos;
|
private final long idleTtlNanos;
|
||||||
private final long intervalMillis;
|
private final long intervalMillis;
|
||||||
|
private final LoopWatchdog watchdog;
|
||||||
private volatile boolean running;
|
private volatile boolean running;
|
||||||
private Thread thread;
|
private Thread thread;
|
||||||
/**
|
/**
|
||||||
@@ -45,29 +58,61 @@ public final class SessionReaper {
|
|||||||
|
|
||||||
/** Construct a reaper with an explicit polling interval (useful for tests). */
|
/** Construct a reaper with an explicit polling interval (useful for tests). */
|
||||||
public SessionReaper(SessionManager sessions, long idleTtlSeconds, long intervalMillis) {
|
public SessionReaper(SessionManager sessions, long idleTtlSeconds, long intervalMillis) {
|
||||||
|
this(sessions, idleTtlSeconds, intervalMillis, System::nanoTime);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Full constructor — for tests: an injectable monotonic clock (fleetd #544). */
|
||||||
|
SessionReaper(SessionManager sessions, long idleTtlSeconds, long intervalMillis, LongSupplier nowNanos) {
|
||||||
this.sessions = sessions;
|
this.sessions = sessions;
|
||||||
this.idleTtlNanos = TimeUnit.SECONDS.toNanos(idleTtlSeconds);
|
this.idleTtlNanos = TimeUnit.SECONDS.toNanos(idleTtlSeconds);
|
||||||
this.intervalMillis = intervalMillis;
|
this.intervalMillis = intervalMillis;
|
||||||
|
this.watchdog = new LoopWatchdog(nowNanos, staleAfterNanos(intervalMillis));
|
||||||
|
}
|
||||||
|
|
||||||
|
private static long staleAfterNanos(long intervalMillis) {
|
||||||
|
return TimeUnit.MILLISECONDS.toNanos(Math.max(intervalMillis, 1)) * STALE_AFTER_INTERVAL_MULTIPLIER;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* This loop's progress fact (fleetd #544): {@code RUNNING}, {@code STALLED} (dead or parked —
|
||||||
|
* indistinguishable from outside, and never observed on purpose), or {@code STOPPED}
|
||||||
|
* ({@link #stop()} was called). See {@link LoopWatchdog} for why staleness — not thread
|
||||||
|
* liveness — is the signal.
|
||||||
|
*/
|
||||||
|
public LoopWatchdog.State health() {
|
||||||
|
return watchdog.state();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Start the reaper loop on a virtual thread. Idempotent. */
|
/** Start the reaper loop on a virtual thread. Idempotent. */
|
||||||
public synchronized void start() {
|
public synchronized void start() {
|
||||||
if (running) return;
|
if (running) return;
|
||||||
running = true;
|
running = true;
|
||||||
|
watchdog.reset(); // fleetd #544: a fresh grace period, not last run's stale mark/timestamp.
|
||||||
thread = Thread.ofVirtual().name("session-reaper").start(this::loop);
|
thread = Thread.ofVirtual().name("session-reaper").start(this::loop);
|
||||||
log.info("session reaper started (idle ttl {}s, interval {}ms)",
|
log.info("session reaper started (idle ttl {}s, interval {}ms)",
|
||||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos), intervalMillis);
|
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos), intervalMillis);
|
||||||
}
|
}
|
||||||
|
|
||||||
private void loop() {
|
private void loop() {
|
||||||
while (running) {
|
try {
|
||||||
try {
|
while (running) {
|
||||||
sessions.reapIdle(idleTtlNanos);
|
try {
|
||||||
} catch (RuntimeException e) {
|
sessions.reapIdle(idleTtlNanos);
|
||||||
log.warn("session reaper iteration failed; continuing", e);
|
} catch (Throwable e) {
|
||||||
|
log.error("session reaper iteration failed; continuing", e);
|
||||||
|
}
|
||||||
|
maybeSweepWipRefs();
|
||||||
|
// fleetd #544: a round is "complete" only after reapIdle AND the WIP-sweep gate have
|
||||||
|
// both returned — a hang in either (e.g. a wedged git call) means this line is never
|
||||||
|
// reached, so the watchdog goes stale exactly like a dead loop would.
|
||||||
|
watchdog.recordRoundComplete();
|
||||||
|
sleep();
|
||||||
}
|
}
|
||||||
maybeSweepWipRefs();
|
} finally {
|
||||||
sleep();
|
if (running) {
|
||||||
|
log.error("session reaper loop exited unexpectedly; it can be restarted");
|
||||||
|
}
|
||||||
|
running = false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -90,8 +135,8 @@ public final class SessionReaper {
|
|||||||
log.info("refs/wip retention sweep deleted {} snapshot ref(s) older than 24h whose "
|
log.info("refs/wip retention sweep deleted {} snapshot ref(s) older than 24h whose "
|
||||||
+ "content was already reachable from main", deleted);
|
+ "content was already reachable from main", deleted);
|
||||||
}
|
}
|
||||||
} catch (RuntimeException e) {
|
} catch (Throwable e) {
|
||||||
log.warn("refs/wip retention sweep failed; continuing", e);
|
log.error("refs/wip retention sweep failed; continuing", e);
|
||||||
}
|
}
|
||||||
// Set even when the sweep threw, so a broken repo is retried on the slow cadence rather
|
// Set even when the sweep threw, so a broken repo is retried on the slow cadence rather
|
||||||
// than hammering git on every 5-second iteration.
|
// than hammering git on every 5-second iteration.
|
||||||
@@ -111,6 +156,7 @@ public final class SessionReaper {
|
|||||||
/** Stop the reaper loop. Idempotent. */
|
/** Stop the reaper loop. Idempotent. */
|
||||||
public synchronized void stop() {
|
public synchronized void stop() {
|
||||||
running = false;
|
running = false;
|
||||||
|
watchdog.markStoppedByCaller(); // fleetd #544: this halt is on purpose — health() must say STOPPED, not STALLED.
|
||||||
if (thread != null) thread.interrupt();
|
if (thread != null) thread.interrupt();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -14,10 +14,12 @@ import static org.junit.jupiter.api.Assertions.assertTrue;
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* CB-534: the injector's readiness gate must open for a lead as well as for a present worker.
|
* CB-534: the injector's readiness gate must open for a lead as well as for a present worker.
|
||||||
|
* fleetd #669 follow-up: the same gate must also open for a collaborator, which — like a lead —
|
||||||
|
* is never enrolled in {@link MemberPresence} and never discovered by the lead scan.
|
||||||
*
|
*
|
||||||
* <p>The bug these cover was silent and slow: a lead was never marked present (only workers are), so
|
* <p>The bug these cover was silent and slow: a lead (and later a collaborator) was never marked
|
||||||
* every lead→lead delivery sat on the gate for the full readiness grace and failed ~60s later without
|
* present (only workers are) and never counted as a lead, so every send to one sat on the gate for
|
||||||
* a keystroke ever reaching the pane.
|
* the full readiness grace and failed ~60s later without a keystroke ever reaching the pane.
|
||||||
*/
|
*/
|
||||||
class FleetDeliverabilityTest {
|
class FleetDeliverabilityTest {
|
||||||
|
|
||||||
@@ -25,19 +27,25 @@ class FleetDeliverabilityTest {
|
|||||||
return () -> m;
|
return () -> m;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private static Supplier<Map<String, String>> collaborators(Map<String, String> m) {
|
||||||
|
return () -> m;
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@DisplayName("a worker that has connected its MCP is deliverable")
|
@DisplayName("a worker that has connected its MCP is deliverable")
|
||||||
void presentWorkerIsDeliverable() {
|
void presentWorkerIsDeliverable() {
|
||||||
MemberPresence presence = new MemberPresence();
|
MemberPresence presence = new MemberPresence();
|
||||||
presence.markPresent("term_worker");
|
presence.markPresent("term_worker");
|
||||||
|
|
||||||
assertTrue(Fleetd.deliverableTo(presence, leads(Map.of())).test("term_worker"));
|
assertTrue(Fleetd.deliverableTo(presence, leads(Map.of()), collaborators(Map.of()))
|
||||||
|
.test("term_worker"));
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@DisplayName("a worker still in its boot window is held back")
|
@DisplayName("a worker still in its boot window is held back")
|
||||||
void absentWorkerIsNotDeliverable() {
|
void absentWorkerIsNotDeliverable() {
|
||||||
assertFalse(Fleetd.deliverableTo(new MemberPresence(), leads(Map.of())).test("term_booting"));
|
assertFalse(Fleetd.deliverableTo(new MemberPresence(), leads(Map.of()), collaborators(Map.of()))
|
||||||
|
.test("term_booting"));
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -45,19 +53,31 @@ class FleetDeliverabilityTest {
|
|||||||
void leadIsDeliverableWithoutPresence() {
|
void leadIsDeliverableWithoutPresence() {
|
||||||
MemberPresence presence = new MemberPresence();
|
MemberPresence presence = new MemberPresence();
|
||||||
Predicate<String> deliverable =
|
Predicate<String> deliverable =
|
||||||
Fleetd.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")));
|
Fleetd.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")), collaborators(Map.of()));
|
||||||
|
|
||||||
assertFalse(presence.isPresent("term_lead"), "a lead is never enrolled in worker presence");
|
assertFalse(presence.isPresent("term_lead"), "a lead is never enrolled in worker presence");
|
||||||
assertTrue(deliverable.test("term_lead"), "…and must be deliverable anyway");
|
assertTrue(deliverable.test("term_lead"), "…and must be deliverable anyway");
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@DisplayName("an unknown terminal is deliverable to neither")
|
@DisplayName("a collaborator is deliverable without ever being marked present or scanned as a lead")
|
||||||
|
void collaboratorIsDeliverableWithoutPresenceOrLeadStatus() {
|
||||||
|
MemberPresence presence = new MemberPresence();
|
||||||
|
Predicate<String> deliverable = Fleetd.deliverableTo(presence, leads(Map.of()),
|
||||||
|
collaborators(Map.of("term_collab", "kevin")));
|
||||||
|
|
||||||
|
assertFalse(presence.isPresent("term_collab"), "a collaborator is never enrolled in worker presence");
|
||||||
|
assertTrue(deliverable.test("term_collab"), "…and must be deliverable anyway");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("an unknown terminal is deliverable to none of presence, leads, or collaborators")
|
||||||
void strangerIsNotDeliverable() {
|
void strangerIsNotDeliverable() {
|
||||||
MemberPresence presence = new MemberPresence();
|
MemberPresence presence = new MemberPresence();
|
||||||
presence.markPresent("term_worker");
|
presence.markPresent("term_worker");
|
||||||
|
|
||||||
assertFalse(Fleetd.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")))
|
assertFalse(Fleetd.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")),
|
||||||
|
collaborators(Map.of("term_collab", "kevin")))
|
||||||
.test("term_stranger"));
|
.test("term_stranger"));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -65,22 +85,47 @@ class FleetDeliverabilityTest {
|
|||||||
@DisplayName("a lead discovered after startup becomes deliverable with no restart")
|
@DisplayName("a lead discovered after startup becomes deliverable with no restart")
|
||||||
void leadSetIsReadThroughOnEveryCall() {
|
void leadSetIsReadThroughOnEveryCall() {
|
||||||
Map<String, String> discovered = new HashMap<>();
|
Map<String, String> discovered = new HashMap<>();
|
||||||
Predicate<String> deliverable = Fleetd.deliverableTo(new MemberPresence(), leads(discovered));
|
Predicate<String> deliverable =
|
||||||
|
Fleetd.deliverableTo(new MemberPresence(), leads(discovered), collaborators(Map.of()));
|
||||||
|
|
||||||
assertFalse(deliverable.test("term_late"));
|
assertFalse(deliverable.test("term_late"));
|
||||||
discovered.put("term_late", "gpt-sol-5.6"); // leadScan picks up a newly labelled tab
|
discovered.put("term_late", "gpt-sol-5.6"); // leadScan picks up a newly labelled tab
|
||||||
assertTrue(deliverable.test("term_late"), "the supplier must be re-read, not snapshotted");
|
assertTrue(deliverable.test("term_late"), "the supplier must be re-read, not snapshotted");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a collaborator discovered after startup becomes deliverable with no restart")
|
||||||
|
void collaboratorSetIsReadThroughOnEveryCall() {
|
||||||
|
Map<String, String> discovered = new HashMap<>();
|
||||||
|
Predicate<String> deliverable =
|
||||||
|
Fleetd.deliverableTo(new MemberPresence(), leads(Map.of()), collaborators(discovered));
|
||||||
|
|
||||||
|
assertFalse(deliverable.test("term_late_collab"));
|
||||||
|
discovered.put("term_late_collab", "kevin"); // the same tab scan picks up a newly labelled collaborator tab
|
||||||
|
assertTrue(deliverable.test("term_late_collab"), "the supplier must be re-read, not snapshotted");
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@DisplayName("forgetting a torn-down worker does not strip a lead of its deliverability")
|
@DisplayName("forgetting a torn-down worker does not strip a lead of its deliverability")
|
||||||
void forgetDoesNotDisarmALead() {
|
void forgetDoesNotDisarmALead() {
|
||||||
MemberPresence presence = new MemberPresence();
|
MemberPresence presence = new MemberPresence();
|
||||||
Predicate<String> deliverable =
|
Predicate<String> deliverable =
|
||||||
Fleetd.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")));
|
Fleetd.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")), collaborators(Map.of()));
|
||||||
|
|
||||||
presence.forget("term_lead"); // the injector's cleanup path runs against every target
|
presence.forget("term_lead"); // the injector's cleanup path runs against every target
|
||||||
|
|
||||||
assertTrue(deliverable.test("term_lead"));
|
assertTrue(deliverable.test("term_lead"));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("forgetting a torn-down worker does not strip a collaborator of its deliverability")
|
||||||
|
void forgetDoesNotDisarmACollaborator() {
|
||||||
|
MemberPresence presence = new MemberPresence();
|
||||||
|
Predicate<String> deliverable = Fleetd.deliverableTo(presence, leads(Map.of()),
|
||||||
|
collaborators(Map.of("term_collab", "kevin")));
|
||||||
|
|
||||||
|
presence.forget("term_collab"); // the injector's cleanup path runs against every target
|
||||||
|
|
||||||
|
assertTrue(deliverable.test("term_collab"));
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,183 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import ch.qos.logback.classic.Level;
|
||||||
|
import ch.qos.logback.classic.Logger;
|
||||||
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
|
import ch.qos.logback.core.read.ListAppender;
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||||
|
import dev.ltms.fleet.msg.LeadChannel;
|
||||||
|
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||||
|
import dev.ltms.fleet.msg.LeadMessage;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.concurrent.atomic.AtomicInteger;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertSame;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 step 4 ranks 3 and 8: the assembled daemon must use the AMQP openers from {@link
|
||||||
|
* ResourcePorts}, and its startup report must describe the object the runtime actually owns. These
|
||||||
|
* fakes never open a socket.
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyAmqpOpenersTest {
|
||||||
|
|
||||||
|
private static final String COORD_ID = "assembly-test";
|
||||||
|
|
||||||
|
private static final class DurableReplyInbox implements ReplyInbox {
|
||||||
|
@Override public void own(String target) { }
|
||||||
|
@Override public void release(String target) { }
|
||||||
|
@Override public void publish(String target, String msgId, String content) { }
|
||||||
|
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||||
|
@Override public boolean ack(String target, String msgId) { return false; }
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class DurableLeadMailbox implements LeadChannelHandle {
|
||||||
|
@Override public void publish(String toCoordId, LeadMessage message) { }
|
||||||
|
@Override public List<LeadMessage> peek() { return List.of(); }
|
||||||
|
@Override public void ack(String msgId) { }
|
||||||
|
@Override public String selfCoordId() { return COORD_ID; }
|
||||||
|
@Override public boolean heldDurable() { return true; }
|
||||||
|
@Override public MailboxState inspect(String coordId) { return MailboxState.unknown(coordId); }
|
||||||
|
@Override public void close() { }
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class RecordingPorts implements ResourcePorts {
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
final DurableReplyInbox replyInbox = new DurableReplyInbox();
|
||||||
|
final DurableLeadMailbox leadMailbox = new DurableLeadMailbox();
|
||||||
|
final AtomicInteger replyOpenCalls = new AtomicInteger();
|
||||||
|
final AtomicInteger mailboxOpenCalls = new AtomicInteger();
|
||||||
|
final boolean openSucceeds;
|
||||||
|
|
||||||
|
RecordingPorts(boolean openSucceeds) {
|
||||||
|
this.openSucceeds = openSucceeds;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override public Map<String, String> environment() { return Map.of(); }
|
||||||
|
@Override public HerdrClient connectHerdr(Path socketPath) { return herdr; }
|
||||||
|
@Override public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> {
|
||||||
|
replyOpenCalls.incrementAndGet();
|
||||||
|
if (!openSucceeds) throw new IllegalStateException("fake reply broker is down");
|
||||||
|
return replyInbox;
|
||||||
|
};
|
||||||
|
}
|
||||||
|
@Override public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfId, prefetch) -> {
|
||||||
|
mailboxOpenCalls.incrementAndGet();
|
||||||
|
if (!openSucceeds) throw new IllegalStateException("fake coordination broker is down");
|
||||||
|
return leadMailbox;
|
||||||
|
};
|
||||||
|
}
|
||||||
|
@Override public LongSupplier nanoClock() { return System::nanoTime; }
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
@Override public LongSupplier wallClockNanos() { return System::nanoTime; }
|
||||||
|
@Override public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
@Override public void addShutdownHook(Runnable hook) { }
|
||||||
|
@Override public void startHttp(Javalin app, String host, int port) { }
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Files.createDirectories(dir);
|
||||||
|
Path config = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(config, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
broker:
|
||||||
|
uri: "amqp://fake-reply-broker/vh"
|
||||||
|
coordinator:
|
||||||
|
uri: "amqp://fake-coordination-broker/vh"
|
||||||
|
selfId: "assembly-test"
|
||||||
|
""");
|
||||||
|
return FleetConfig.load(config);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetdRuntime assemble(Path dir, RecordingPorts ports) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
return FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, new ConfigRef(dir.resolve("fleetd.yaml"), cfg),
|
||||||
|
new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static boolean reportContains(ListAppender<ILoggingEvent> appender, String text) {
|
||||||
|
return appender.list.stream().map(ILoggingEvent::getFormattedMessage).anyMatch(message -> message.contains(text));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void assembledAmqpOpenersAndTheirReportsAgreeOnDurableAndFallbackStates(@TempDir Path dir) throws Exception {
|
||||||
|
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||||
|
Level oldLevel = logger.getLevel();
|
||||||
|
ListAppender<ILoggingEvent> reports = new ListAppender<>();
|
||||||
|
reports.start();
|
||||||
|
logger.setLevel(Level.INFO);
|
||||||
|
logger.addAppender(reports);
|
||||||
|
try {
|
||||||
|
RecordingPorts durablePorts = new RecordingPorts(true);
|
||||||
|
FleetdRuntime durable = assemble(dir.resolve("durable"), durablePorts);
|
||||||
|
try {
|
||||||
|
// Control: this fails loudly if the assembly did not run or used an inert opener.
|
||||||
|
assertEquals(1, durablePorts.replyOpenCalls.get(), "assembly must call replyInboxOpener once");
|
||||||
|
assertEquals(1, durablePorts.mailboxOpenCalls.get(), "assembly must call leadMailboxOpener once");
|
||||||
|
assertSame(durablePorts.replyInbox, durable.replyInbox(),
|
||||||
|
"the durable reply report must describe the exact inbox the runtime owns");
|
||||||
|
assertSame(durablePorts.leadMailbox, durable.leadMailbox(),
|
||||||
|
"the coordination-on report must describe the exact mailbox the runtime owns");
|
||||||
|
assertNotNull(durable.leadCoordLoop(), "a durable mailbox must start lead coordination");
|
||||||
|
assertTrue(reportContains(reports, "reply inbox: AMQP broker (durable)"));
|
||||||
|
assertTrue(reportContains(reports, "lead coordination: ON as coord-id " + COORD_ID));
|
||||||
|
} finally {
|
||||||
|
durable.close();
|
||||||
|
}
|
||||||
|
|
||||||
|
reports.list.clear();
|
||||||
|
RecordingPorts fallbackPorts = new RecordingPorts(false);
|
||||||
|
FleetdRuntime fallback = assemble(dir.resolve("fallback"), fallbackPorts);
|
||||||
|
try {
|
||||||
|
assertEquals(1, fallbackPorts.replyOpenCalls.get(), "assembly must call the failing reply opener once");
|
||||||
|
assertEquals(1, fallbackPorts.mailboxOpenCalls.get(), "assembly must call the failing mailbox opener once");
|
||||||
|
assertTrue(fallback.replyInbox() instanceof InMemoryReplyInbox,
|
||||||
|
"a failed reply opener must make the runtime own the in-memory fallback");
|
||||||
|
assertNull(fallback.leadMailbox(), "a failed mailbox opener must leave coordination off");
|
||||||
|
assertNull(fallback.leadCoordLoop(), "coordination must not start without a mailbox");
|
||||||
|
assertTrue(reportContains(reports, "reply inbox: in-memory (soft-state)"));
|
||||||
|
assertTrue(reportContains(reports, "lead-to-lead messaging is OFF"));
|
||||||
|
} finally {
|
||||||
|
fallback.close();
|
||||||
|
}
|
||||||
|
} finally {
|
||||||
|
logger.detachAppender(reports);
|
||||||
|
logger.setLevel(oldLevel);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,166 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.auth.Authz;
|
||||||
|
import dev.ltms.fleet.auth.Principal;
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.mcp.FleetMcp;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import io.modelcontextprotocol.spec.McpSchema;
|
||||||
|
import org.junit.jupiter.api.AfterEach;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.lang.reflect.Method;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Asserts that the {@link FleetMcp} built by {@link FleetdAssembly#assembleAndStart} applies the
|
||||||
|
* authorization table: a worker is refused {@code SPAWN}, and the primary is allowed it.
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyAuthorizationModeTest {
|
||||||
|
|
||||||
|
private static final class TestResourcePorts implements ResourcePorts {
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> new ReplyInbox() {
|
||||||
|
@Override public void own(String target) { }
|
||||||
|
@Override public void release(String target) { }
|
||||||
|
@Override public void publish(String target, String msgId, String content) { }
|
||||||
|
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||||
|
@Override public boolean ack(String target, String msgId) { return false; }
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException(
|
||||||
|
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
// Binding a real port would clash with any daemon already listening on it.
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("FakeHerdr is healthy; no poll wait is expected");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private TestResourcePorts ports;
|
||||||
|
|
||||||
|
@AfterEach
|
||||||
|
void tearDown() {
|
||||||
|
if (ports != null && ports.shutdownHook != null) {
|
||||||
|
ports.shutdownHook.run();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
health:
|
||||||
|
enabled: false
|
||||||
|
broker:
|
||||||
|
uri: "amqp://fake-test-broker/vh"
|
||||||
|
""");
|
||||||
|
return FleetConfig.load(file);
|
||||||
|
}
|
||||||
|
|
||||||
|
private FleetMcp assemble(Path dir) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
ports = new TestResourcePorts();
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg,
|
||||||
|
new ConfigRef(dir.resolve("fleetd.yaml"), cfg), new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||||
|
return runtime.mcp();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Invokes {@code FleetMcp#denyFor}, which is package-private to {@code dev.ltms.fleet.mcp}
|
||||||
|
* while this test is in {@code dev.ltms.fleet}. Nothing here catches a missing method: if
|
||||||
|
* {@code denyFor} is renamed or removed, {@link NoSuchMethodException} propagates and the
|
||||||
|
* test fails.
|
||||||
|
*/
|
||||||
|
private static McpSchema.CallToolResult denyFor(FleetMcp mcp, Principal caller, Authz.Action action,
|
||||||
|
String target) throws Exception {
|
||||||
|
Method m = FleetMcp.class.getDeclaredMethod("denyFor", Principal.class, Authz.Action.class, String.class);
|
||||||
|
m.setAccessible(true);
|
||||||
|
return (McpSchema.CallToolResult) m.invoke(mcp, caller, action, target);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void productionBootPathRefusesAnUnauthorizedCallerThroughTheAssembledFleetMcp(@TempDir Path dir)
|
||||||
|
throws Exception {
|
||||||
|
FleetMcp mcp = assemble(dir);
|
||||||
|
|
||||||
|
McpSchema.CallToolResult deniedForWorker = denyFor(mcp, Principal.worker("term_a", 200),
|
||||||
|
Authz.Action.SPAWN, "term_a");
|
||||||
|
assertNotNull(deniedForWorker,
|
||||||
|
"a worker must not be able to fleet_spawn through the assembled FleetMcp");
|
||||||
|
assertTrue(deniedForWorker.isError(), "a refusal is returned as an MCP tool error");
|
||||||
|
|
||||||
|
McpSchema.CallToolResult allowedForPrimary = denyFor(mcp, Principal.primary(100),
|
||||||
|
Authz.Action.SPAWN, "term_a");
|
||||||
|
assertNull(allowedForPrimary,
|
||||||
|
"control: the primary must still be allowed to fleet_spawn — otherwise the worker "
|
||||||
|
+ "refusal above would pass even with the gate wired backwards");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,165 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertSame;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #669 Unit E. Reaches the real {@link dev.ltms.fleet.herdr.HerdrRouter} that {@link
|
||||||
|
* FleetdAssembly#assembleAndStart} builds and wires — not a copy built for this test — and proves
|
||||||
|
* that a configured collaborator's terminal routes to the LEAD herdr daemon.
|
||||||
|
*
|
||||||
|
* <p>Two distinct {@link FakeHerdr} instances are required, the same pattern {@code
|
||||||
|
* FleetdAssemblyConnectionIdentityTest} and {@code FleetdLeadRolloverAssemblyTest} already use:
|
||||||
|
* with one client shared between {@code herdrSocket} and {@code memberHerdrSocket},
|
||||||
|
* {@code HerdrRouter} folds {@code leadAgents} and {@code memberAgents} into the same instance
|
||||||
|
* (see its constructor), and {@code agentsFor} would return that one object regardless of whether
|
||||||
|
* the collaborator map was ever consulted — invisible to a mutation of the predicate this ticket
|
||||||
|
* fixes. This test's two sockets resolve to two different fakes, so the assertion only passes when
|
||||||
|
* the collaborator's terminal is actually recognised and routed to the lead one.
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyCollaboratorHerdrRoutingTest {
|
||||||
|
|
||||||
|
private static final Path LEAD_SOCKET = Path.of("/fake/lead-herdr.sock");
|
||||||
|
private static final Path MEMBER_SOCKET = Path.of("/fake/member-herdr.sock");
|
||||||
|
|
||||||
|
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||||
|
final Map<Path, HerdrClient> herdrsBySocket = new LinkedHashMap<>();
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
HerdrClient client = herdrsBySocket.get(socketPath);
|
||||||
|
if (client == null) {
|
||||||
|
throw new IllegalStateException("no fake herdr registered for socket " + socketPath);
|
||||||
|
}
|
||||||
|
return client;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> new ReplyInbox() {
|
||||||
|
@Override public void own(String target) { }
|
||||||
|
@Override public void release(String target) { }
|
||||||
|
@Override public void publish(String target, String msgId, String content) { }
|
||||||
|
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||||
|
@Override public boolean ack(String target, String msgId) { return false; }
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException(
|
||||||
|
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
// Do not bind a real port in this assembly test.
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("FakeHerdr is healthy; no poll wait is expected");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
herdrSocket: "%s"
|
||||||
|
memberHerdrSocket: "%s"
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
fleet:
|
||||||
|
collaborators:
|
||||||
|
reviewer-alex:
|
||||||
|
tab: "collab: alex"
|
||||||
|
profiles:
|
||||||
|
sonnet:
|
||||||
|
subscription: true
|
||||||
|
argv: ["ccs", "sonnet"]
|
||||||
|
""".formatted(LEAD_SOCKET, MEMBER_SOCKET));
|
||||||
|
return FleetConfig.load(file);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void assembledRouterRoutesACollaboratorTerminalToTheLeadDaemon(@TempDir Path dir) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||||
|
|
||||||
|
// The fixed FakeHerdr fixture already ties terminal "term_a" to a live agent on tab
|
||||||
|
// "w2:t7" (pane "w2:p7") — seeding only the tab LABEL to match the configured collaborator
|
||||||
|
// is enough to make LeadTabScanner resolve "term_a" as that collaborator. Seeded on the
|
||||||
|
// LEAD fake only: a collaborator's pane lives in the lead daemon, exactly like a lead's.
|
||||||
|
FakeHerdr lead = new FakeHerdr().withTab("w2", "w2:t7", "collab: alex");
|
||||||
|
FakeHerdr member = new FakeHerdr();
|
||||||
|
ports.herdrsBySocket.put(LEAD_SOCKET, lead);
|
||||||
|
ports.herdrsBySocket.put(MEMBER_SOCKET, member);
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg,
|
||||||
|
new ConfigRef(dir.resolve("fleetd.yaml"), cfg), new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||||
|
try {
|
||||||
|
assertSame(runtime.router().leadAgents(), runtime.router().agentsFor("term_a"),
|
||||||
|
"a configured collaborator's terminal must route to the LEAD daemon — "
|
||||||
|
+ "FleetdAssembly must wire the collaborator map into the router's "
|
||||||
|
+ "predicate, not just LeadTabScanner.get()");
|
||||||
|
assertSame(runtime.router().memberAgents(), runtime.router().agentsFor("term_shell"),
|
||||||
|
"control: a terminal naming neither a lead nor a collaborator (term_shell, on "
|
||||||
|
+ "the unlabelled tab w2:t8) must still route to the member daemon");
|
||||||
|
} finally {
|
||||||
|
assertNotNull(ports.shutdownHook, "control: assembly must capture its shutdown hook");
|
||||||
|
ports.shutdownHook.run();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,230 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.herdr.PaneLocator;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.AfterEach;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.CopyOnWriteArrayList;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 step 2, unit B2 (CB-185, identity half). Replaces the deleted
|
||||||
|
* {@code FleetdConnectionIdentityConstructionTest}, which pinned this claim by reading {@code
|
||||||
|
* Fleetd.java}'s source text for {@code "new PaneLocator(herdr, memberHerdr)"}. That claim moved
|
||||||
|
* to {@code FleetdAssembly.java} (fleetd #612 Unit A) and is pinned here instead, by driving the
|
||||||
|
* real {@link ConnectionIdentity} — via {@code runtime.mcp().identity()}, not a copy — that {@link
|
||||||
|
* FleetdAssembly#assembleAndStart} built.
|
||||||
|
*
|
||||||
|
* <p><strong>What this guards against</strong> (from the deleted test's own javadoc): pinning
|
||||||
|
* {@code PaneLocator} to {@code memberHerdr} alone leaves every LEAD's own MCP connection
|
||||||
|
* unresolvable ({@code callerTerminal == null}) the moment {@code memberHerdrSocket} names a
|
||||||
|
* second daemon, which breaks {@code fleet_reply}/{@code fleet_ask}/{@code fleet_whoami} for a
|
||||||
|
* lead. {@code PaneLocatorTest} already proves {@link PaneLocator} itself can search two clients
|
||||||
|
* given two — the gap this pins is that the assembly actually passes it two, and in the right
|
||||||
|
* order (lead first).
|
||||||
|
*
|
||||||
|
* <p><strong>Why this cannot be driven through a real MCP/HTTP round trip.</strong> The natural
|
||||||
|
* way to observe {@code ConnectionIdentity} would be a real {@code fleet_whoami} call over the
|
||||||
|
* built {@code FleetMcp}, the way {@code FleetMcpContextExtractorTest} drives its own
|
||||||
|
* hand-built one. That does not work for the REAL assembly, because {@code FleetdAssembly} wires
|
||||||
|
* {@code ConnectionIdentity} with a hardcoded {@code new LsofPeerPidLookup()} (see {@code
|
||||||
|
* FleetdAssembly.java:444}), and {@code LsofPeerPidLookup} explicitly excludes its own PID — see
|
||||||
|
* its javadoc: "we exclude our own PID and take the other end". In a JUnit test the HTTP client
|
||||||
|
* and the daemon under test run in the very same JVM, so the "client" and "server" ends of the
|
||||||
|
* loopback connection ARE the same PID, and {@code pidForLocalPort} always returns {@code -1}
|
||||||
|
* before {@link PaneLocator} is ever reached — proving nothing about which daemon(s) got searched.
|
||||||
|
* This test instead reaches the real {@link PaneLocator} the assembly built (through {@link
|
||||||
|
* ConnectionIdentity#panes()}, added for exactly this) and drives it with a chosen pid directly,
|
||||||
|
* bypassing the OS-dependent PID lookup entirely — a legitimate substitute, since the pid lookup
|
||||||
|
* is not what CB-185 is about.
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyConnectionIdentityTest {
|
||||||
|
|
||||||
|
/** Same shape as {@code FleetdAssemblyLifecycleTest}'s fake, but keys {@code connectHerdr} by
|
||||||
|
* socket path so the lead and member daemons can be two DIFFERENT {@link FakeHerdr}s. */
|
||||||
|
private static final class TwoHerdrResourcePorts implements ResourcePorts {
|
||||||
|
|
||||||
|
final Map<Path, HerdrClient> herdrsBySocket = new LinkedHashMap<>();
|
||||||
|
final CopyOnWriteArrayList<ScheduledExecutorService> schedulers = new CopyOnWriteArrayList<>();
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
HerdrClient client = herdrsBySocket.get(socketPath);
|
||||||
|
if (client == null) {
|
||||||
|
throw new IllegalStateException("no fake herdr registered for socket " + socketPath);
|
||||||
|
}
|
||||||
|
return client;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> new dev.ltms.fleet.msg.InMemoryReplyInbox();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException(
|
||||||
|
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
ScheduledExecutorService scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||||
|
schedulers.add(scheduler);
|
||||||
|
return scheduler;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
this.shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
// Deliberately never bind — this test never issues a real HTTP request.
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private FleetdRuntime runtime;
|
||||||
|
private TwoHerdrResourcePorts ports;
|
||||||
|
|
||||||
|
@AfterEach
|
||||||
|
void tearDown() {
|
||||||
|
if (ports != null && ports.shutdownHook != null) {
|
||||||
|
ports.shutdownHook.run();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final Path LEAD_SOCKET = Path.of("/fake/lead-herdr.sock");
|
||||||
|
private static final Path MEMBER_SOCKET = Path.of("/fake/member-herdr.sock");
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
herdrSocket: "%s"
|
||||||
|
memberHerdrSocket: "%s"
|
||||||
|
lifecycle:
|
||||||
|
idleTtlSeconds: 600
|
||||||
|
health:
|
||||||
|
enabled: false
|
||||||
|
broker:
|
||||||
|
uri: "amqp://fake-test-broker/vh"
|
||||||
|
""".formatted(LEAD_SOCKET, MEMBER_SOCKET));
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
private FleetdRuntime assemble(Path dir, FakeHerdr lead, FakeHerdr member) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
ports = new TwoHerdrResourcePorts();
|
||||||
|
ports.herdrsBySocket.put(LEAD_SOCKET, lead);
|
||||||
|
ports.herdrsBySocket.put(MEMBER_SOCKET, member);
|
||||||
|
runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
return runtime;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The pin. {@code lead} carries the one pane {@link FakeHerdr}'s canned {@code
|
||||||
|
* pane.process_info} ties to {@link FakeHerdr#WORKER_PID} (pane {@code w2:p7}); {@code member}
|
||||||
|
* reports NO panes at all ({@link FakeHerdr#withNoPanes()}) — modelling a second daemon that
|
||||||
|
* simply does not host the caller's pane, exactly the CB-185 javadoc's scenario for a lead's
|
||||||
|
* own connection. If {@code PaneLocator} only ever searches the member daemon (the bug), this
|
||||||
|
* pid resolves to nothing, because the pane that owns it lives on the LEAD daemon the bug
|
||||||
|
* skips.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void connectionIdentitySearchesTheLeadDaemonNotJustTheMemberOne(@TempDir Path dir) throws Exception {
|
||||||
|
FakeHerdr lead = new FakeHerdr();
|
||||||
|
FakeHerdr member = new FakeHerdr().withNoPanes();
|
||||||
|
|
||||||
|
assemble(dir, lead, member);
|
||||||
|
|
||||||
|
PaneLocator panes = runtime.mcp().identity().panes();
|
||||||
|
PaneLocator.Lookup lookup = panes.terminalForPid(FakeHerdr.WORKER_PID);
|
||||||
|
|
||||||
|
assertEquals("term_a", lookup.terminal(),
|
||||||
|
"the pane owning WORKER_PID lives on the LEAD daemon only (the member fake reports "
|
||||||
|
+ "no panes) — PaneLocator must still find it, which is only possible if it "
|
||||||
|
+ "searches the lead client and not just the member one");
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The mirror control: when the pane instead lives ONLY on the member daemon (the lead reports
|
||||||
|
* no panes), the lookup must still find it — proving the member client is genuinely searched
|
||||||
|
* too, not merely tolerated as a second, always-losing argument.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void connectionIdentityAlsoSearchesTheMemberDaemon(@TempDir Path dir) throws Exception {
|
||||||
|
FakeHerdr lead = new FakeHerdr().withNoPanes();
|
||||||
|
FakeHerdr member = new FakeHerdr();
|
||||||
|
|
||||||
|
assemble(dir, lead, member);
|
||||||
|
|
||||||
|
PaneLocator panes = runtime.mcp().identity().panes();
|
||||||
|
PaneLocator.Lookup lookup = panes.terminalForPid(FakeHerdr.WORKER_PID);
|
||||||
|
|
||||||
|
assertEquals("term_a", lookup.terminal(),
|
||||||
|
"the pane owning WORKER_PID lives on the MEMBER daemon only — PaneLocator must "
|
||||||
|
+ "find it there too");
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Sanity control: a pid nobody owns resolves to nothing on either daemon. */
|
||||||
|
@Test
|
||||||
|
void aPidNoPaneOwnsResolvesToNoTerminalOnEitherDaemon(@TempDir Path dir) throws Exception {
|
||||||
|
FakeHerdr lead = new FakeHerdr();
|
||||||
|
FakeHerdr member = new FakeHerdr();
|
||||||
|
|
||||||
|
assemble(dir, lead, member);
|
||||||
|
|
||||||
|
PaneLocator panes = runtime.mcp().identity().panes();
|
||||||
|
PaneLocator.Lookup lookup = panes.terminalForPid(999_999L);
|
||||||
|
|
||||||
|
assertNull(lookup.terminal());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,219 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.msg.LeadChannel;
|
||||||
|
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||||
|
import dev.ltms.fleet.msg.LeadMessage;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertSame;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 A-gaps (gap 1): {@code FleetdAssemblyLifecycleTest}'s own class javadoc says plainly
|
||||||
|
* that it leaves {@code coordinator:} unset, so {@code leadMailbox} and {@code leadCoordLoop} stay
|
||||||
|
* {@code null} throughout — the configured-coordinator path is never exercised by Unit A's own
|
||||||
|
* test. This class drives that path instead: a real {@code coordinator:} block, a fake {@link
|
||||||
|
* Fleetd.LeadMailboxOpener} returning a fake closeable channel (never a real broker connection),
|
||||||
|
* and proof that {@link FleetdAssembly#assembleAndStart} both builds it and, on shutdown, closes it.
|
||||||
|
*
|
||||||
|
* <p>Made possible by generalising {@code Fleetd.LeadMailboxOpener}'s return type (and {@code
|
||||||
|
* FleetdRuntime}'s field) from the concrete {@code LeadMailbox} to {@link LeadChannelHandle} — a
|
||||||
|
* {@link LeadChannel} its owner can also close. {@code FleetMcp} and {@code LeadCoordLoop} already
|
||||||
|
* consumed the narrower {@link LeadChannel}; this only widens the one seam that owns and closes it.
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyCoordinatorLifecycleTest {
|
||||||
|
|
||||||
|
private static final String SELF_COORD_ID = "test-lead";
|
||||||
|
|
||||||
|
/** A fake {@link LeadChannelHandle}: never touches a broker, and records whether it was closed. */
|
||||||
|
private static final class FakeLeadChannel implements LeadChannelHandle {
|
||||||
|
volatile boolean closed = false;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void publish(String toCoordId, LeadMessage m) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<LeadMessage> peek() {
|
||||||
|
return List.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void ack(String msgId) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public String selfCoordId() {
|
||||||
|
return SELF_COORD_ID;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean heldDurable() {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public MailboxState inspect(String coordId) {
|
||||||
|
return MailboxState.unknown(coordId);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
closed = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Minimal fake {@link ResourcePorts}: a real herdr fake, a fake reply inbox, and a real, offered fake lead channel. */
|
||||||
|
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||||
|
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||||
|
final FakeLeadChannel leadChannel = new FakeLeadChannel();
|
||||||
|
String offeredUri;
|
||||||
|
String offeredSelfId;
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> replyInbox;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
offeredUri = uri;
|
||||||
|
offeredSelfId = selfCoordId;
|
||||||
|
return leadChannel;
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
this.shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
// No real HTTP bind in a unit test.
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class SentinelReplyInbox implements ReplyInbox {
|
||||||
|
@Override
|
||||||
|
public void own(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void release(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void publish(String target, String msgId, String content) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<InboxMessage> peek(String target) {
|
||||||
|
return List.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean ack(String target, String msgId) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
coordinator:
|
||||||
|
uri: "amqp://fake-lead-broker/vh"
|
||||||
|
selfId: "test-lead"
|
||||||
|
""");
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void configuredCoordinatorIsBuiltByTheAssemblyAndClosedOnShutdown(@TempDir Path dir) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
|
||||||
|
// --- the assembly actually calls the configured opener and builds the coordinator path ---
|
||||||
|
assertEquals("amqp://fake-lead-broker/vh", ports.offeredUri,
|
||||||
|
"the assembly must open the mailbox at the configured broker uri");
|
||||||
|
assertEquals(SELF_COORD_ID, ports.offeredSelfId,
|
||||||
|
"the assembly must open the mailbox under the configured selfId");
|
||||||
|
assertSame(ports.leadChannel, runtime.leadMailbox(),
|
||||||
|
"FleetdRuntime must own the exact LeadChannelHandle the opener returned, not a copy");
|
||||||
|
assertNotNull(runtime.leadCoordLoop(),
|
||||||
|
"a configured coordinator: block must build the receiving LeadCoordLoop too");
|
||||||
|
assertFalse(ports.leadChannel.closed, "the channel must still be open while the daemon is running");
|
||||||
|
|
||||||
|
// --- shutting the assembly down closes it -------------------------------------------------
|
||||||
|
assertNotNull(ports.shutdownHook, "FleetdAssembly must have registered a shutdown hook");
|
||||||
|
ports.shutdownHook.run();
|
||||||
|
|
||||||
|
assertTrue(ports.leadChannel.closed,
|
||||||
|
"FleetdRuntime.close() must close the configured LeadChannelHandle");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,278 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.AfterEach;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.Timeout;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.net.URI;
|
||||||
|
import java.net.http.HttpClient;
|
||||||
|
import java.net.http.HttpRequest;
|
||||||
|
import java.net.http.HttpResponse;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.CopyOnWriteArrayList;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 step 2, unit B2 (CB-185, {@code FleetApp} half). Replaces the deleted {@code
|
||||||
|
* FleetdFleetAppConstructionTest}, which pinned this claim by reading {@code Fleetd.java}'s
|
||||||
|
* source text for {@code "new FleetApp(herdr, memberHerdr, workers,"}. That claim moved to {@code
|
||||||
|
* FleetdAssembly.java} (fleetd #612 Unit A) and is pinned here instead, by driving the real {@code
|
||||||
|
* Javalin} app — via {@code runtime.app()}, not a copy — that {@link
|
||||||
|
* FleetdAssembly#assembleAndStart} built and handed to {@link FleetdRuntime}.
|
||||||
|
*
|
||||||
|
* <p><strong>What this guards against</strong> (from the deleted test's own javadoc): constructing
|
||||||
|
* {@code FleetApp} with the lead-only {@code herdr} client (dropping {@code memberHerdr}) makes
|
||||||
|
* {@code GET /healthz} report green while the MEMBER daemon is down — so every spawn fails
|
||||||
|
* invisibly — and silently drops every member workspace from {@code GET /sessions}. {@code
|
||||||
|
* FleetAppTwoDaemonTest} already proves {@code FleetApp} itself merges/gates correctly given two
|
||||||
|
* clients; the gap this pins is that the assembly actually passes it two.
|
||||||
|
*
|
||||||
|
* <p><strong>Both directions, not just one</strong> (fleetd #612 issue comment 17525): the deleted
|
||||||
|
* guard's positive assertion required the exact pair {@code "new FleetApp(herdr, memberHerdr,
|
||||||
|
* workers,"}, which does not survive EITHER daemon being dropped. An earlier version of this class
|
||||||
|
* only proved the member-dropped direction, which left {@code new FleetApp(memberHerdr,
|
||||||
|
* memberHerdr, ...)} — the symmetric bug, {@code /healthz} green while the LEAD daemon is down —
|
||||||
|
* an undetected regression. {@link #healthzGoesRedWhenTheLeadDaemonIsDownEvenThoughTheMemberIsUp}
|
||||||
|
* closes that.
|
||||||
|
*
|
||||||
|
* <p>Unlike the {@code ConnectionIdentity} half of CB-185 ({@code
|
||||||
|
* FleetdAssemblyConnectionIdentityTest}), {@code /healthz} needs no caller identity at all, so
|
||||||
|
* this test can bind {@link FleetdRuntime#app()} to a REAL ephemeral port (exactly {@code
|
||||||
|
* FleetAppTwoDaemonTest} does for its own hand-built {@code FleetApp}) and drive it with a real
|
||||||
|
* {@code HttpClient} — no accessor needed for this half.
|
||||||
|
*
|
||||||
|
* <p><strong>{@code GET /sessions} could not be driven the same way</strong>, so this class does
|
||||||
|
* not pin the merge half of the deleted test's javadoc. This class configures no {@code auth:}
|
||||||
|
* block, so it runs under the default {@code loopback-trust} mode ({@code FleetConfig}). Under
|
||||||
|
* that mode, {@code /sessions} requires {@code Authz.Action.READ}, which — through the REAL
|
||||||
|
* assembly's real {@code CallerResolver}/{@code ConnectionIdentity} (built with a hardcoded
|
||||||
|
* {@code new LsofPeerPidLookup()}) — needs {@code Caller.resolved()}, i.e. a real positive pid
|
||||||
|
* from {@code lsof}. {@code LsofPeerPidLookup} excludes its own pid (see its javadoc), and a
|
||||||
|
* JUnit test's HTTP client and the daemon under test share one JVM pid, so the resolved pid is
|
||||||
|
* always {@code -1} and every such request is refused as {@code ANONYMOUS} (fleetd #317's
|
||||||
|
* fail-closed rule) before the route handler — and its {@code memberHerdr} merge — is ever
|
||||||
|
* reached. Verified directly: driving {@code GET /sessions} here returns {@code 401
|
||||||
|
* unauthenticated}, not the merged body. {@code FleetAppTwoDaemonTest} avoids this because it
|
||||||
|
* builds {@code FleetApp} with {@code callers: null}, which is not what the real assembly
|
||||||
|
* passes. The {@code /healthz} pin below is what this class relies on for CB-185's {@code
|
||||||
|
* FleetApp} half; {@code FleetAppTwoDaemonTest} remains the full behavioural proof that
|
||||||
|
* {@code FleetApp} itself merges {@code /sessions} correctly once handed two clients.
|
||||||
|
*
|
||||||
|
* <p><strong>This refusal is {@code loopback-trust}-specific, not a property of {@code
|
||||||
|
* CallerResolver} in general.</strong> Under {@code auth.mode: token}, {@code
|
||||||
|
* CallerResolver#resolve} returns before ever consulting {@code Caller.resolved()} or {@code
|
||||||
|
* Caller.scanComplete()}: a request carrying a valid bearer token in its {@code Authorization}
|
||||||
|
* header resolves to {@code Role#PRIMARY} with no pid lookup at all, so the same-JVM-pid
|
||||||
|
* exclusion above never comes into play. {@code FleetdQuarantineOutageDualWindowAssemblyTest}
|
||||||
|
* and {@code FleetdListReportingSourcesAssemblyTest} both drive {@code Authz.Action.READ} this
|
||||||
|
* way, over a real {@code McpSyncClient}/{@code HttpClient} against a real {@code
|
||||||
|
* FleetdAssembly#assembleAndStart}, and both get the real response rather than a refusal.
|
||||||
|
*
|
||||||
|
* <p><strong>fleetd #629 follow-up.</strong> The fix below (see {@link TwoHerdrResourcePorts})
|
||||||
|
* makes {@link #healthzGoesRedWhenTheLeadDaemonIsDownEvenThoughTheMemberIsUp}'s fake {@code
|
||||||
|
* nanoClock()} frozen unless {@code herdrPollWait()} itself advances it. That is a sharper pin
|
||||||
|
* than an assertion — if a future edit to {@code FleetdAssembly} ever bypasses {@code
|
||||||
|
* ports.herdrPollWait()} again (e.g. reverting to a hardcoded {@code Thread.sleep}), the clock
|
||||||
|
* never advances, {@code Fleetd#awaitHerdr}'s deadline is never reached, and this test hangs
|
||||||
|
* forever instead of failing — proven by deliberately reintroducing that exact regression while
|
||||||
|
* fixing this ticket. {@code @Timeout} turns that silent hang into a bounded, named test failure:
|
||||||
|
* {@code SEPARATE_THREAD} so JUnit's timeout governor can actually interrupt a thread stuck in a
|
||||||
|
* real {@code Thread.sleep} loop (the default {@code SAME_THREAD} mode cannot — it only measures
|
||||||
|
* elapsed time after the test method returns on its own, which never happens here). 10 seconds is
|
||||||
|
* roughly 150x the real passing times measured here (~0.06s), so a slow CI machine has no reason
|
||||||
|
* to flake, and it is still 3x faster than discovering the regression by burning a CI job's whole
|
||||||
|
* wall-clock budget.
|
||||||
|
*/
|
||||||
|
@Timeout(value = 10, unit = TimeUnit.SECONDS, threadMode = Timeout.ThreadMode.SEPARATE_THREAD)
|
||||||
|
class FleetdAssemblyFleetAppTest {
|
||||||
|
|
||||||
|
private static final class TwoHerdrResourcePorts implements ResourcePorts {
|
||||||
|
|
||||||
|
final Map<Path, HerdrClient> herdrsBySocket = new LinkedHashMap<>();
|
||||||
|
final CopyOnWriteArrayList<ScheduledExecutorService> schedulers = new CopyOnWriteArrayList<>();
|
||||||
|
// fleetd #629: a fake, advanceable clock — NOT System::nanoTime. awaitHerdr's poll wait
|
||||||
|
// (herdrPollWait() below) advances this on every poll instead of sleeping for real, so the
|
||||||
|
// down-lead test below reaches awaitHerdr's deadline without burning real wall-clock time.
|
||||||
|
final AtomicLong nowNanos = new AtomicLong(1_000_000_000L); // arbitrary non-zero start
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
HerdrClient client = herdrsBySocket.get(socketPath);
|
||||||
|
if (client == null) {
|
||||||
|
throw new IllegalStateException("no fake herdr registered for socket " + socketPath);
|
||||||
|
}
|
||||||
|
return client;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> new dev.ltms.fleet.msg.InMemoryReplyInbox();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException(
|
||||||
|
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return nowNanos::get;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// fleetd #629: advance the fake clock instead of a real Thread.sleep, so awaitHerdr's
|
||||||
|
// deadline is reached in real time regardless of the configured poll interval.
|
||||||
|
return () -> nowNanos.addAndGet(TimeUnit.SECONDS.toNanos(1));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
ScheduledExecutorService scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||||
|
schedulers.add(scheduler);
|
||||||
|
return scheduler;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
this.shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
// Deliberately never bind here — this test binds runtime.app() itself, for real, below.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final Path LEAD_SOCKET = Path.of("/fake/lead-herdr.sock");
|
||||||
|
private static final Path MEMBER_SOCKET = Path.of("/fake/member-herdr.sock");
|
||||||
|
|
||||||
|
private final HttpClient http = HttpClient.newHttpClient();
|
||||||
|
private FleetdRuntime runtime;
|
||||||
|
private TwoHerdrResourcePorts ports;
|
||||||
|
private Javalin boundApp;
|
||||||
|
|
||||||
|
@AfterEach
|
||||||
|
void tearDown() {
|
||||||
|
if (boundApp != null) {
|
||||||
|
boundApp.stop();
|
||||||
|
}
|
||||||
|
if (ports != null && ports.shutdownHook != null) {
|
||||||
|
ports.shutdownHook.run();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
herdrSocket: "%s"
|
||||||
|
memberHerdrSocket: "%s"
|
||||||
|
lifecycle:
|
||||||
|
idleTtlSeconds: 600
|
||||||
|
health:
|
||||||
|
enabled: false
|
||||||
|
broker:
|
||||||
|
uri: "amqp://fake-test-broker/vh"
|
||||||
|
""".formatted(LEAD_SOCKET, MEMBER_SOCKET));
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Assembles the real graph, then binds the real {@code Javalin app} to an ephemeral port. */
|
||||||
|
private int assembleAndBind(Path dir, FakeHerdr lead, FakeHerdr member) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
ports = new TwoHerdrResourcePorts();
|
||||||
|
ports.herdrsBySocket.put(LEAD_SOCKET, lead);
|
||||||
|
ports.herdrsBySocket.put(MEMBER_SOCKET, member);
|
||||||
|
runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
boundApp = runtime.app().start("127.0.0.1", 0);
|
||||||
|
return boundApp.port();
|
||||||
|
}
|
||||||
|
|
||||||
|
private HttpResponse<String> get(int port, String path) throws Exception {
|
||||||
|
HttpRequest req = HttpRequest.newBuilder(URI.create("http://127.0.0.1:" + port + path)).GET().build();
|
||||||
|
return http.send(req, HttpResponse.BodyHandlers.ofString());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The pin. The MEMBER daemon is down; the LEAD daemon is healthy. If the assembly built
|
||||||
|
* {@code FleetApp} with only the lead client (the bug: passing {@code herdr} where {@code
|
||||||
|
* memberHerdr} is expected), the down member is invisible and {@code /healthz} stays 200.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void healthzGoesRedWhenTheMemberDaemonIsDownEvenThoughTheLeadIsUp(@TempDir Path dir) throws Exception {
|
||||||
|
FakeHerdr lead = new FakeHerdr();
|
||||||
|
FakeHerdr member = new FakeHerdr().healthy(false);
|
||||||
|
|
||||||
|
int port = assembleAndBind(dir, lead, member);
|
||||||
|
|
||||||
|
HttpResponse<String> res = get(port, "/healthz");
|
||||||
|
assertEquals(503, res.statusCode(),
|
||||||
|
"a down MEMBER daemon must not be masked by a healthy lead: " + res.body());
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Sanity control: both daemons healthy must still be green through the real assembly. */
|
||||||
|
@Test
|
||||||
|
void healthzIsGreenWhenBothDaemonsAreUp(@TempDir Path dir) throws Exception {
|
||||||
|
FakeHerdr lead = new FakeHerdr();
|
||||||
|
FakeHerdr member = new FakeHerdr();
|
||||||
|
|
||||||
|
int port = assembleAndBind(dir, lead, member);
|
||||||
|
|
||||||
|
assertEquals(200, get(port, "/healthz").statusCode());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The symmetric pin (fleetd #612 issue comment 17525): the LEAD daemon is down; the MEMBER
|
||||||
|
* daemon is healthy. If the assembly built {@code FleetApp} with only the member client
|
||||||
|
* (dropping {@code herdr} — the mirror of the bug above, {@code new FleetApp(memberHerdr,
|
||||||
|
* memberHerdr, ...)}), the down LEAD is invisible and {@code /healthz} stays 200. Without this
|
||||||
|
* case the pair above is one-directional and does not cover the deleted guard's positive
|
||||||
|
* assertion (it required BOTH {@code herdr,} and {@code memberHerdr,} in that order).
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void healthzGoesRedWhenTheLeadDaemonIsDownEvenThoughTheMemberIsUp(@TempDir Path dir) throws Exception {
|
||||||
|
FakeHerdr lead = new FakeHerdr().healthy(false);
|
||||||
|
FakeHerdr member = new FakeHerdr();
|
||||||
|
|
||||||
|
int port = assembleAndBind(dir, lead, member);
|
||||||
|
|
||||||
|
HttpResponse<String> res = get(port, "/healthz");
|
||||||
|
assertEquals(503, res.statusCode(),
|
||||||
|
"a down LEAD daemon must not be masked by a healthy member: " + res.body());
|
||||||
|
}
|
||||||
|
}
|
||||||
+198
@@ -0,0 +1,198 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.health.FleetHealthMonitor;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.msg.MessageService;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.BiConsumer;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 rank 7 — {@code FleetdAssembly.java:429} wires {@link FleetHealthMonitor}'s {@code
|
||||||
|
* failTarget} callback with {@code Fleetd.healthFailTarget(messages)}. {@link
|
||||||
|
* FleetdHealthFailTargetWiringTest} already pins that the FACTORY itself delegates to {@code
|
||||||
|
* messages::abandon}, but it calls {@code Fleetd.healthFailTarget} directly — it never drives {@code
|
||||||
|
* FleetdAssembly.assembleAndStart} and so cannot see whether the real call site at {@code :429}
|
||||||
|
* still passes it the real, assembled {@link MessageService}. Swapping that argument for a no-op
|
||||||
|
* {@code (a, b) -> {}} compiles clean and leaves the whole suite — including the factory-level test
|
||||||
|
* — green: a dead member's waiting ticket then sits {@code PENDING} for the full 30-minute async
|
||||||
|
* timeout instead of failing immediately.
|
||||||
|
*
|
||||||
|
* <p>This test assembles the real daemon with {@code health.enabled: true}, pulls the REAL {@code
|
||||||
|
* failTarget} {@link BiConsumer} out of the REAL, assembled {@link FleetHealthMonitor} (via
|
||||||
|
* reflection — the field is package-private to {@code dev.ltms.fleet.health}, and nothing public
|
||||||
|
* exposes it; {@code StatusPollerResilienceTest} already uses the same technique in this suite), and
|
||||||
|
* invokes it directly against the REAL {@link MessageService} {@link FleetdRuntime#messages()}
|
||||||
|
* returns. A no-op lambda swapped in at the call site leaves the ticket {@code PENDING} forever,
|
||||||
|
* which this test catches; the real one fails it.
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyHealthFailTargetBehaviouralTest {
|
||||||
|
|
||||||
|
private static final String TARGET = "term_a";
|
||||||
|
|
||||||
|
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException("no broker: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException("no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
this.shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
health:
|
||||||
|
enabled: true
|
||||||
|
intervalSeconds: 30
|
||||||
|
""");
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
@Test
|
||||||
|
@DisplayName("[BEHAVIOURAL] the real assembled FleetHealthMonitor's failTarget reaches the real "
|
||||||
|
+ "MessageService.abandon, not a no-op")
|
||||||
|
void assembledHealthFailTargetReachesRealMessages(@TempDir Path dir) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
assertNotNull(ports.shutdownHook, "FleetdAssembly must have registered a shutdown hook");
|
||||||
|
|
||||||
|
// Surefire runs the whole suite in one JVM fork, so the scheduler/loops this assembly starts
|
||||||
|
// (SessionReaper, StatusPoller, the health monitor) must be torn down here, on the failure
|
||||||
|
// path too — hence the try/finally, not just a statement at the end of the happy path.
|
||||||
|
try {
|
||||||
|
FleetHealthMonitor healthMonitor = runtime.healthMonitor();
|
||||||
|
assertNotNull(healthMonitor, "health.enabled: true in this test's config, so "
|
||||||
|
+ "FleetdAssembly.assembleAndStart must have built a real FleetHealthMonitor");
|
||||||
|
|
||||||
|
Field field = FleetHealthMonitor.class.getDeclaredField("failTarget");
|
||||||
|
field.setAccessible(true);
|
||||||
|
BiConsumer<String, String> failTarget = (BiConsumer<String, String>) field.get(healthMonitor);
|
||||||
|
assertNotNull(failTarget, "FleetHealthMonitor's failTarget must never be null — the "
|
||||||
|
+ "constructor itself requires it");
|
||||||
|
|
||||||
|
MessageService messages = runtime.messages();
|
||||||
|
|
||||||
|
// --- loud control: prove the assembled MessageService is actually wired up and a ticket is
|
||||||
|
// genuinely PENDING before failTarget ever runs. If this fails, the test below would pass
|
||||||
|
// vacuously on a MessageService that never got a ticket in the first place. TARGET has no
|
||||||
|
// live agent behind it (no session was ever acquired), so nothing resolves this ticket on
|
||||||
|
// its own — it stays PENDING until failTarget (or a timeout) ends it.
|
||||||
|
String ticket = messages.sendAsync(TARGET, "long task");
|
||||||
|
MessageService.TaskView before = messages.poll(ticket);
|
||||||
|
assertEquals(MessageService.Phase.PENDING, before.phase(),
|
||||||
|
"control: the async ticket must be PENDING before failTarget runs");
|
||||||
|
|
||||||
|
failTarget.accept(TARGET, "member unreachable (health monitor)");
|
||||||
|
|
||||||
|
MessageService.TaskView after = awaitTerminal(messages, ticket);
|
||||||
|
assertEquals(MessageService.Phase.FAILED, after.phase(),
|
||||||
|
"FleetdAssembly.java:429 must pass Fleetd.healthFailTarget(messages) built from the "
|
||||||
|
+ "SAME assembled MessageService — a no-op BiConsumer at that call site leaves "
|
||||||
|
+ "this ticket PENDING for the full 30-minute async timeout instead of failing it");
|
||||||
|
assertTrue(after.detail() != null && after.detail().contains("member unreachable"),
|
||||||
|
"the failure reason passed to failTarget.accept must reach MessageService.abandon and "
|
||||||
|
+ "end up in the ticket's detail");
|
||||||
|
} finally {
|
||||||
|
// Proof the teardown actually ran, not just an assurance that a finally was added: the
|
||||||
|
// captured shutdown hook's close order (FleetdAssemblyLifecycleTest) closes the herdr
|
||||||
|
// client last, so ports.herdr.closed flips to true only if this hook really executed.
|
||||||
|
ports.shutdownHook.run();
|
||||||
|
assertTrue(ports.herdr.closed, "the captured shutdown hook must have run and closed herdr — "
|
||||||
|
+ "proof this test's assembled background loops/scheduler were torn down");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static MessageService.TaskView awaitTerminal(MessageService messages, String ticket)
|
||||||
|
throws InterruptedException {
|
||||||
|
long deadline = System.currentTimeMillis() + 5000;
|
||||||
|
MessageService.TaskView view = messages.poll(ticket);
|
||||||
|
while (view.phase() == MessageService.Phase.PENDING && System.currentTimeMillis() < deadline) {
|
||||||
|
//noinspection BusyWait
|
||||||
|
Thread.sleep(10);
|
||||||
|
view = messages.poll(ticket);
|
||||||
|
}
|
||||||
|
return view;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,200 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.herdr.LeadTabScanner;
|
||||||
|
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||||
|
import dev.ltms.fleet.msg.LeadCoordLoop;
|
||||||
|
import dev.ltms.fleet.msg.LeadMessage;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Set;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
import java.util.function.Supplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertInstanceOf;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #670 — pins the {@code excludedWorkspaceLabels} argument {@link FleetdAssembly}'s
|
||||||
|
* production boot path passes to {@link LeadTabScanner} ({@code Set.of()}).
|
||||||
|
*
|
||||||
|
* <p>{@code LeadTabScannerTest} already covers this constructor parameter, but it builds its own
|
||||||
|
* {@link LeadTabScanner} with its own set, so it tests the seam and proves nothing about the
|
||||||
|
* producer. This test instead reaches the exact object {@link FleetdAssembly#assembleAndStart}
|
||||||
|
* builds: a {@code fleet.leaders:} block makes the assembly construct a real
|
||||||
|
* {@link LeadTabScanner} for its local {@code leads} supplier, and a {@code coordinator:} block
|
||||||
|
* makes it hand that same supplier instance to {@link LeadCoordLoop} (fleetd #637), which stores
|
||||||
|
* it as a field. Reflection recovers it from there, and then from the scanner itself, so the
|
||||||
|
* assertion is against the real production argument rather than a copy built for this test.
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyLeadTabScannerExclusionTest {
|
||||||
|
|
||||||
|
private static final class FakeLeadChannel implements LeadChannelHandle {
|
||||||
|
@Override
|
||||||
|
public void publish(String toCoordId, LeadMessage message) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<LeadMessage> peek() {
|
||||||
|
return List.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void ack(String msgId) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public String selfCoordId() {
|
||||||
|
return "test-lead";
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean heldDurable() {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public MailboxState inspect(String coordId) {
|
||||||
|
return MailboxState.unknown(coordId);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class TestResourcePorts implements ResourcePorts {
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> new ReplyInbox() {
|
||||||
|
@Override public void own(String target) { }
|
||||||
|
@Override public void release(String target) { }
|
||||||
|
@Override public void publish(String target, String msgId, String content) { }
|
||||||
|
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||||
|
@Override public boolean ack(String target, String msgId) { return false; }
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> new FakeLeadChannel();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
// Do not bind a real port in this assembly test.
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("FakeHerdr is healthy; no poll wait is expected");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
coordinator:
|
||||||
|
uri: "amqp://fake-lead-broker/vh"
|
||||||
|
selfId: "test-lead"
|
||||||
|
fleet:
|
||||||
|
leaders:
|
||||||
|
primary:
|
||||||
|
tab: "lead: primary"
|
||||||
|
profile: sonnet
|
||||||
|
profiles:
|
||||||
|
sonnet:
|
||||||
|
subscription: true
|
||||||
|
argv: ["ccs", "sonnet"]
|
||||||
|
""");
|
||||||
|
return FleetConfig.load(file);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void productionBootPathPassesNoExcludedWorkspaceLabels(@TempDir Path dir) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
TestResourcePorts ports = new TestResourcePorts();
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg,
|
||||||
|
new ConfigRef(dir.resolve("fleetd.yaml"), cfg), new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||||
|
try {
|
||||||
|
LeadCoordLoop coordLoop = runtime.leadCoordLoop();
|
||||||
|
assertNotNull(coordLoop, "control: a configured coordinator: block must build LeadCoordLoop");
|
||||||
|
|
||||||
|
Field leadsField = LeadCoordLoop.class.getDeclaredField("leads");
|
||||||
|
leadsField.setAccessible(true);
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
Supplier<Map<String, String>> leads = (Supplier<Map<String, String>>) leadsField.get(coordLoop);
|
||||||
|
|
||||||
|
assertInstanceOf(LeadTabScanner.class, leads,
|
||||||
|
"control: a non-empty fleet.leaders: block must make FleetdAssembly build a real "
|
||||||
|
+ "LeadTabScanner for its `leads` supplier, not the Map::of fallback — "
|
||||||
|
+ "otherwise this test would pass for the wrong reason");
|
||||||
|
|
||||||
|
Field excludedField = LeadTabScanner.class.getDeclaredField("excludedWorkspaceLabels");
|
||||||
|
excludedField.setAccessible(true);
|
||||||
|
Set<?> excluded = (Set<?>) excludedField.get(leads);
|
||||||
|
|
||||||
|
assertTrue(excluded.isEmpty(),
|
||||||
|
"FleetdAssembly must pass an empty excludedWorkspaceLabels to "
|
||||||
|
+ "LeadTabScanner — scanning member tabs would demote the lead to a worker");
|
||||||
|
} finally {
|
||||||
|
assertNotNull(ports.shutdownHook, "control: assembly must capture its shutdown hook");
|
||||||
|
ports.shutdownHook.run();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,266 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.CopyOnWriteArrayList;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 Unit A: proves {@link FleetdAssembly#assembleAndStart} — not a copy of its logic —
|
||||||
|
* against a real {@link FleetConfig}, a {@link FakeHerdr} and a fake {@link ResourcePorts}, with no
|
||||||
|
* real herdr socket, no real broker, and no real HTTP bind.
|
||||||
|
*
|
||||||
|
* <p><strong>The canonical order this test asserts against was recorded from {@code Fleetd.java}
|
||||||
|
* BEFORE any code moved</strong> (fleetd #612 Unit A's mandated order of work), by reading the
|
||||||
|
* original {@code main}'s body and its shutdown-hook {@code Thread}:
|
||||||
|
*
|
||||||
|
* <p>Start order: {@code SessionReaper.start()} → {@code StatusPoller.start()} →
|
||||||
|
* {@code LeadHeartbeatLoop.start()} (opt-in) → {@code FleetHealthMonitor.start()} (opt-in) →
|
||||||
|
* {@code LeadCoordLoop.start()} (opt-in) → {@code ConfigWatcher.start()} (opt-in) →
|
||||||
|
* {@code app.start()} (HTTP), always last.
|
||||||
|
*
|
||||||
|
* <p>Close order (from the original shutdown hook body): {@code sessions.close(drainTimeoutSeconds)}
|
||||||
|
* → {@code poller.stop()} → {@code messages.close()} → {@code pushLoop.close()} →
|
||||||
|
* {@code heartbeat.close()} (if present) → {@code leadCoordLoop.close()} (if present) →
|
||||||
|
* {@code leadCoordScheduler.shutdownNow()} (if present) → {@code healthMonitor.stop()} (if present)
|
||||||
|
* → {@code configWatcher.stop()} (if present) → {@code mcp.close()} → {@code reaper.stop()} (if
|
||||||
|
* present) → {@code idleSleepGuard.close()} (if present) → {@code replyInbox.close()} (if
|
||||||
|
* {@code AutoCloseable}) → {@code leadMailbox.close()} (if present) → {@code router.close()}.
|
||||||
|
*
|
||||||
|
* <p>This test's config deliberately leaves {@code coordinator:} unset, so {@code leadMailbox} and
|
||||||
|
* {@code leadCoordLoop} stay {@code null} throughout — the lead-mailbox resource-ledger criterion is
|
||||||
|
* NOT exercised here; see the class-level caveat in the implementer's hand-off. {@code
|
||||||
|
* idleSleepGuard.enabled: false} is set for the same kind of reason: it would otherwise try to spawn
|
||||||
|
* a real {@code caffeinate} subprocess, which is not one of the resources the ticket's acceptance
|
||||||
|
* criteria names (scheduler/inbox/mailbox/loop/MCP server/herdr router).
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyLifecycleTest {
|
||||||
|
|
||||||
|
/** The one {@link HerdrClient} both {@code herdrSocket} and {@code memberHerdrSocket} resolve to. */
|
||||||
|
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||||
|
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||||
|
final List<String> ledger = new CopyOnWriteArrayList<>();
|
||||||
|
final List<ScheduledExecutorService> schedulers = new CopyOnWriteArrayList<>();
|
||||||
|
final List<String> schedulerPurposes = new ArrayList<>();
|
||||||
|
Runnable shutdownHook;
|
||||||
|
Javalin startedApp;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
ledger.add("connectHerdr");
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
ledger.add("replyInboxOpener");
|
||||||
|
return (uri, prefetch) -> replyInbox;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
// Never invoked: this test's config has no `coordinator:` block, so
|
||||||
|
// Fleetd.openLeadMailbox returns null before calling the opener at all.
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException(
|
||||||
|
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
ledger.add("newScheduler:" + purpose);
|
||||||
|
schedulerPurposes.add(purpose);
|
||||||
|
ScheduledExecutorService scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||||
|
schedulers.add(scheduler);
|
||||||
|
return scheduler;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
ledger.add("addShutdownHook");
|
||||||
|
this.shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
// Deliberately never call app.start(host, port): no real HTTP bind in a unit test.
|
||||||
|
ledger.add("startHttp");
|
||||||
|
this.startedApp = app;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** A fake {@link ReplyInbox} that is also {@link AutoCloseable}, so the ledger can prove it closes. */
|
||||||
|
private static final class SentinelReplyInbox implements ReplyInbox, AutoCloseable {
|
||||||
|
volatile boolean closed = false;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void own(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void release(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void publish(String target, String msgId, String content) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<InboxMessage> peek(String target) {
|
||||||
|
return List.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean ack(String target, String msgId) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
closed = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
lifecycle:
|
||||||
|
idleTtlSeconds: 600
|
||||||
|
health:
|
||||||
|
enabled: true
|
||||||
|
intervalSeconds: 30
|
||||||
|
leadHeartbeat:
|
||||||
|
idleAfterSeconds: 600
|
||||||
|
backoffMs: 15000
|
||||||
|
quietNudgeCap: 5
|
||||||
|
configReload:
|
||||||
|
enabled: true
|
||||||
|
intervalSeconds: 30
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
broker:
|
||||||
|
uri: "amqp://fake-test-broker/vh"
|
||||||
|
""");
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void assemblesTheRealBootGraphWithoutTouchingAnyRealSocketBrokerOrPort(@TempDir Path dir)
|
||||||
|
throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
|
||||||
|
// --- start order: the herdr connect, then the recurring background loops, in the recorded
|
||||||
|
// order, then the shutdown hook is registered, then (finally) HTTP "starts" -------------
|
||||||
|
assertTrue(ports.ledger.indexOf("connectHerdr") < ports.ledger.indexOf("newScheduler:bridge-push-"),
|
||||||
|
"herdr must connect before the push scheduler is created: " + ports.ledger);
|
||||||
|
assertEquals(List.of("bridge-push-", "bridge-heartbeat-", "bridge-health-"), ports.schedulerPurposes,
|
||||||
|
"the three always-created schedulers must be requested in exactly this order: "
|
||||||
|
+ ports.schedulerPurposes);
|
||||||
|
assertTrue(ports.ledger.indexOf("addShutdownHook") < ports.ledger.indexOf("startHttp"),
|
||||||
|
"the shutdown hook must be registered before HTTP starts — the one statement that "
|
||||||
|
+ "could not be reordered without changing FleetdRuntime's constructor shape, "
|
||||||
|
+ "see FleetdAssembly's javadoc: " + ports.ledger);
|
||||||
|
assertEquals(ports.ledger.size() - 1, ports.ledger.indexOf("startHttp"),
|
||||||
|
"HTTP must start LAST of everything this fake observes: " + ports.ledger);
|
||||||
|
assertNotNull(ports.startedApp, "FleetdAssembly must have built and handed off a real FleetApp");
|
||||||
|
|
||||||
|
// No real HTTP bind and no real herdr socket: this call returning at all, plus the ledger
|
||||||
|
// above, is the proof — a real bind or a real UnixSocketHerdrClient.connect would have
|
||||||
|
// thrown or hung against the sockets/ports this test never opened.
|
||||||
|
assertNotNull(runtime.app(), "FleetdRuntime must own the same Javalin app that was built");
|
||||||
|
assertEquals(ports.startedApp, runtime.app(), "attachApp must hand FleetdRuntime the SAME instance");
|
||||||
|
|
||||||
|
// --- the runtime owns the real, live objects — not a copy --------------------------------
|
||||||
|
assertEquals(LoopWatchdog.State.RUNNING, runtime.reaper().health(),
|
||||||
|
"SessionReaper must be running: lifecycle.idleTtlSeconds is configured");
|
||||||
|
assertEquals(LoopWatchdog.State.RUNNING, runtime.poller().health(), "StatusPoller must be running");
|
||||||
|
assertNotNull(runtime.heartbeat(), "leadHeartbeat: is configured, so the loop must be built and started");
|
||||||
|
assertNotNull(runtime.healthMonitor(), "health.enabled: true, so the monitor must be built and started");
|
||||||
|
assertNotNull(runtime.configWatcher(), "configReload.enabled: true, so the watcher must be built and started");
|
||||||
|
// Documented gap (see class javadoc): no coordinator: block, so these stay null.
|
||||||
|
assertEquals(null, runtime.leadCoordLoop(), "no coordinator: block — leadCoordLoop must stay unbuilt");
|
||||||
|
assertEquals(null, runtime.leadMailbox(), "no coordinator: block — leadMailbox must stay unbuilt");
|
||||||
|
assertEquals(runtime.replyInbox(), ports.replyInbox,
|
||||||
|
"FleetdRuntime must own the exact ReplyInbox instance this fake's AmqpOpener returned");
|
||||||
|
assertFalse(ports.herdr.closed, "herdr must still be open while the daemon is running");
|
||||||
|
assertFalse(ports.replyInbox.closed, "the reply inbox must still be open while the daemon is running");
|
||||||
|
for (ScheduledExecutorService scheduler : ports.schedulers) {
|
||||||
|
assertFalse(scheduler.isShutdown(), "a scheduler must still be running while the daemon is up");
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- close order: invoke the captured shutdown-hook Runnable directly (no real JVM shutdown
|
||||||
|
// happens in a unit test) and prove every resource this fake can observe is released --------
|
||||||
|
assertNotNull(ports.shutdownHook, "FleetdAssembly must have registered a shutdown hook");
|
||||||
|
ports.shutdownHook.run();
|
||||||
|
|
||||||
|
assertEquals(LoopWatchdog.State.STOPPED, runtime.reaper().health(), "SessionReaper must stop on close");
|
||||||
|
assertEquals(LoopWatchdog.State.STOPPED, runtime.poller().health(), "StatusPoller must stop on close");
|
||||||
|
assertTrue(ports.herdr.closed, "router.close() must close the herdr client last");
|
||||||
|
assertTrue(ports.replyInbox.closed, "the AutoCloseable reply inbox must be closed");
|
||||||
|
for (int i = 0; i < ports.schedulers.size(); i++) {
|
||||||
|
assertTrue(ports.schedulers.get(i).isShutdown(),
|
||||||
|
"scheduler for '" + ports.schedulerPurposes.get(i) + "' must be shut down by close(): "
|
||||||
|
+ "LeadHeartbeatLoop/FleetHealthMonitor/ReplyPushLoop each call "
|
||||||
|
+ "scheduler.shutdownNow() on the exact instance ports.newScheduler(...) handed them");
|
||||||
|
}
|
||||||
|
|
||||||
|
// Calling the captured hook a second time must never happen for a real JVM shutdown hook,
|
||||||
|
// but nothing above should have thrown — that already proves every accessed field's close()
|
||||||
|
// tolerated running once, in the recorded order, without an exception escaping.
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,180 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||||
|
import dev.ltms.fleet.mcp.FleetMcp;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.AfterEach;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.net.URI;
|
||||||
|
import java.net.http.HttpClient;
|
||||||
|
import java.net.http.HttpRequest;
|
||||||
|
import java.net.http.HttpResponse;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 Shape A, rank 10: {@link FleetdAssembly} creates one {@link FleetMcp.LoopHealthSource}
|
||||||
|
* from the real started {@code StatusPoller} and {@code SessionReaper}, then gives it to two operator
|
||||||
|
* windows. These tests reach the real assembled objects through {@link FleetdRuntime}, rather than
|
||||||
|
* building a second source beside them. A hardcoded {@code RUNNING} source would pass a simple
|
||||||
|
* "running" test, so the mutation proof also mis-wires the source to never-started loops: both
|
||||||
|
* windows must then report {@code STOPPED} and these assertions go red.
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyLoopHealthTest {
|
||||||
|
|
||||||
|
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> new dev.ltms.fleet.msg.InMemoryReplyInbox();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException("no coordinator is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("healthy FakeHerdr must not be polled");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
// Bind runtime.app() to an ephemeral port only in the REST assertion below.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private final HttpClient http = HttpClient.newHttpClient();
|
||||||
|
private RecordingResourcePorts ports;
|
||||||
|
private FleetdRuntime runtime;
|
||||||
|
private Javalin boundApp;
|
||||||
|
|
||||||
|
@AfterEach
|
||||||
|
void tearDown() {
|
||||||
|
if (boundApp != null) {
|
||||||
|
boundApp.stop();
|
||||||
|
}
|
||||||
|
if (runtime != null) {
|
||||||
|
runtime.close();
|
||||||
|
}
|
||||||
|
if (ports != null) {
|
||||||
|
assertNotNull(ports.shutdownHook, "the assembly must register its shutdown hook");
|
||||||
|
assertTrue(ports.herdr.closed, "FleetdRuntime.close must close the real assembled herdr client");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
lifecycle:
|
||||||
|
idleTtlSeconds: 600
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
health:
|
||||||
|
enabled: false
|
||||||
|
broker:
|
||||||
|
uri: "amqp://fake-test-broker/vh"
|
||||||
|
""");
|
||||||
|
return FleetConfig.load(file);
|
||||||
|
}
|
||||||
|
|
||||||
|
private void assemble(Path dir) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
ports = new RecordingResourcePorts();
|
||||||
|
runtime = FleetdAssembly.assembleAndStart(
|
||||||
|
new AssemblyInputs(cfg, new ConfigRef(dir.resolve("fleetd.yaml"), cfg),
|
||||||
|
new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void fleetListUsesTheRunningLoopsInTheRealAssembledMcp(@TempDir Path dir) throws Exception {
|
||||||
|
assemble(dir);
|
||||||
|
|
||||||
|
// The field is the exact source captured by FleetMcp's fleet_list handler. Reflection is
|
||||||
|
// necessary because FleetMcp has no public source accessor; it is not a source-text check.
|
||||||
|
FleetMcp.LoopHealthSource loopHealth = loopHealthOf(runtime.mcp());
|
||||||
|
assertRunning(loopHealth, "FleetMcp's real fleet_list source");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void healthzUsesTheRunningLoopsInTheRealAssembledApp(@TempDir Path dir) throws Exception {
|
||||||
|
assemble(dir);
|
||||||
|
boundApp = runtime.app().start("127.0.0.1", 0);
|
||||||
|
|
||||||
|
HttpRequest request = HttpRequest.newBuilder(
|
||||||
|
URI.create("http://127.0.0.1:" + boundApp.port() + "/healthz")).GET().build();
|
||||||
|
HttpResponse<String> response = http.send(request, HttpResponse.BodyHandlers.ofString());
|
||||||
|
assertEquals(200, response.statusCode(), response.body());
|
||||||
|
assertTrue(response.body().contains("\"statusPoller\":\"RUNNING\""), response.body());
|
||||||
|
assertTrue(response.body().contains("\"sessionReaper\":\"RUNNING\""), response.body());
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetMcp.LoopHealthSource loopHealthOf(FleetMcp mcp) throws Exception {
|
||||||
|
Field field = FleetMcp.class.getDeclaredField("loopHealth");
|
||||||
|
field.setAccessible(true);
|
||||||
|
return (FleetMcp.LoopHealthSource) field.get(mcp);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void assertRunning(FleetMcp.LoopHealthSource source, String consumer) {
|
||||||
|
assertEquals(LoopWatchdog.State.RUNNING, source.statusPoller().get(),
|
||||||
|
consumer + " must report the started real StatusPoller as RUNNING");
|
||||||
|
assertEquals(LoopWatchdog.State.RUNNING, source.sessionReaper().get(),
|
||||||
|
consumer + " must report the started real SessionReaper as RUNNING");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,232 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||||
|
import dev.ltms.fleet.msg.MessageService;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||||
|
import dev.ltms.fleet.session.SessionManager;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.Consumer;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 rank 6 — {@code FleetdAssembly.java:447} wires {@code
|
||||||
|
* sessions.onRelease(Fleetd.releaseCleanup(messages, replyInbox, primaryRegistry))}. {@link
|
||||||
|
* FleetdReleaseCleanupWiringTest} already pins that the FACTORY {@code Fleetd.releaseCleanup}
|
||||||
|
* itself reaches all three collaborators — but it calls the factory directly, never {@code
|
||||||
|
* FleetdAssembly.assembleAndStart}, so it cannot see whether the real call site at {@code :447}
|
||||||
|
* still registers it (as opposed to a no-op {@code detail -> { }}) or still passes it the REAL,
|
||||||
|
* assembled {@code messages}/{@code replyInbox}/{@code primaryRegistry}. Swapping the registered
|
||||||
|
* listener for a no-op at that call site compiles clean and leaves the whole suite — including the
|
||||||
|
* factory-level test — green: EVERY teardown then leaks a stuck rendezvous waiter, an unreleased
|
||||||
|
* reply-inbox consumer, and a stale lead binding, all three at once.
|
||||||
|
*
|
||||||
|
* <p>This test assembles the real daemon with {@code idleSleepGuard.enabled: false} — the ONLY
|
||||||
|
* other {@code onRelease} registration in {@code FleetdAssembly} (see {@code
|
||||||
|
* dev.ltms.fleet.power.IdleSleepGuard}'s own wiring at {@code FleetdAssembly.java:228}) — so the
|
||||||
|
* real {@link SessionManager}'s release-listener list holds exactly the one listener this call site
|
||||||
|
* registers. It pulls that REAL listener out via reflection (the list itself is private, like
|
||||||
|
* {@code StatusPollerResilienceTest}'s use of the same technique elsewhere in this suite), invokes
|
||||||
|
* it directly, and asserts all three collaborator effects against the REAL, assembled {@link
|
||||||
|
* MessageService} ({@link FleetdRuntime#messages()}), the REAL {@link ReplyInbox} ({@link
|
||||||
|
* FleetdRuntime#replyInbox()}), and the REAL {@link PrimaryRegistry} — reached through {@link
|
||||||
|
* FleetdRuntime#pushLoop()}, the only other accessor that was handed the same {@code
|
||||||
|
* primaryRegistry} instance ({@code FleetdAssembly.java:382}), since {@code FleetMcp} never exposes
|
||||||
|
* it.
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyReleaseCleanupBehaviouralTest {
|
||||||
|
|
||||||
|
private static final String TARGET = "term_a";
|
||||||
|
|
||||||
|
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException("no broker: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException("no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
this.shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
""");
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
@Test
|
||||||
|
@DisplayName("[BEHAVIOURAL] the real assembled release listener reaches messages.abandon, "
|
||||||
|
+ "replyInbox.release, AND primaryRegistry.forgetDelegation — all three leaks at once")
|
||||||
|
void assembledReleaseListenerReachesAllThreeCollaborators(@TempDir Path dir) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
assertNotNull(ports.shutdownHook, "FleetdAssembly must have registered a shutdown hook");
|
||||||
|
|
||||||
|
// Surefire runs the whole suite in one JVM fork, so the scheduler/loops this assembly starts
|
||||||
|
// must be torn down here, on the failure path too — hence the try/finally, not just a
|
||||||
|
// statement at the end of the happy path.
|
||||||
|
try {
|
||||||
|
// --- reach into SessionManager's private release-listener list. idleSleepGuard.enabled:
|
||||||
|
// false above means FleetdAssembly.java:228 never registers, so this list must hold EXACTLY
|
||||||
|
// the one listener :447 registers.
|
||||||
|
Field listenersField = SessionManager.class.getDeclaredField("releaseListeners");
|
||||||
|
listenersField.setAccessible(true);
|
||||||
|
List<Consumer<SessionManager.ReleaseDetail>> releaseListeners =
|
||||||
|
(List<Consumer<SessionManager.ReleaseDetail>>) listenersField.get(runtime.sessions());
|
||||||
|
assertEquals(1, releaseListeners.size(), "control: with idleSleepGuard.enabled: false, "
|
||||||
|
+ "FleetdAssembly.java:447 must be the ONLY onRelease registration — a different "
|
||||||
|
+ "count means this test is no longer isolating the call site it claims to pin");
|
||||||
|
Consumer<SessionManager.ReleaseDetail> releaseListener = releaseListeners.get(0);
|
||||||
|
|
||||||
|
MessageService messages = runtime.messages();
|
||||||
|
ReplyInbox replyInbox = runtime.replyInbox();
|
||||||
|
|
||||||
|
// primaryRegistry is never exposed by FleetdRuntime directly — ReplyPushLoop is the other
|
||||||
|
// collaborator FleetdAssembly.java:382 hands the SAME instance to, so reach it from there.
|
||||||
|
Field primaryRegistryField = ReplyPushLoop.class.getDeclaredField("primaryRegistry");
|
||||||
|
primaryRegistryField.setAccessible(true);
|
||||||
|
PrimaryRegistry primaryRegistry = (PrimaryRegistry) primaryRegistryField.get(runtime.pushLoop());
|
||||||
|
assertNotNull(primaryRegistry, "control: the assembled ReplyPushLoop must hold a real "
|
||||||
|
+ "PrimaryRegistry instance");
|
||||||
|
|
||||||
|
// --- loud controls: set up the "before" state each collaborator's effect is measured
|
||||||
|
// against, against the REAL assembled objects. If any of these three fails, the test below
|
||||||
|
// would pass vacuously because the subject it claims to observe never existed in the first
|
||||||
|
// place.
|
||||||
|
String ticket = messages.sendAsync(TARGET, "long task");
|
||||||
|
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket).phase(),
|
||||||
|
"control: the async ticket must be PENDING before the release listener runs");
|
||||||
|
|
||||||
|
replyInbox.own(TARGET);
|
||||||
|
replyInbox.publish(TARGET, "msg-1", "hello");
|
||||||
|
assertEquals(1, replyInbox.peek(TARGET).size(),
|
||||||
|
"control: the reply inbox must own TARGET and hold one message before the release "
|
||||||
|
+ "listener runs");
|
||||||
|
|
||||||
|
primaryRegistry.recordDelegation(TARGET, "lead-1");
|
||||||
|
assertEquals("lead-1", primaryRegistry.nudgeTargetFor(TARGET).orElse(null),
|
||||||
|
"control: the delegation must be recorded before the release listener runs");
|
||||||
|
|
||||||
|
// --- the one call under test: invoke the REAL, assembled release listener directly, the
|
||||||
|
// same way SessionManager.release(...) would on a real teardown.
|
||||||
|
releaseListener.accept(new SessionManager.ReleaseDetail(TARGET, null, null, null, null));
|
||||||
|
|
||||||
|
MessageService.TaskView after = awaitTerminal(messages, ticket);
|
||||||
|
assertEquals(MessageService.Phase.FAILED, after.phase(),
|
||||||
|
"FleetdAssembly.java:447 must register a listener that calls messages.abandon(...) "
|
||||||
|
+ "on the SAME assembled MessageService — an inert listener leaves this "
|
||||||
|
+ "ticket PENDING for the full 30-minute async timeout");
|
||||||
|
assertTrue(after.detail() != null && after.detail().contains("released"),
|
||||||
|
"the abandon reason must say the worker session was released");
|
||||||
|
|
||||||
|
assertTrue(replyInbox.peek(TARGET).isEmpty(),
|
||||||
|
"FleetdAssembly.java:447 must register a listener that calls replyInbox.release(...) "
|
||||||
|
+ "— an inert listener leaves the inbox still owning TARGET with its message");
|
||||||
|
|
||||||
|
assertTrue(primaryRegistry.nudgeTargetFor(TARGET).isEmpty(),
|
||||||
|
"FleetdAssembly.java:447 must register a listener that calls "
|
||||||
|
+ "primaryRegistry.forgetDelegation(...) — an inert listener leaves the stale "
|
||||||
|
+ "delegation in place");
|
||||||
|
} finally {
|
||||||
|
// Proof the teardown actually ran, not just an assurance that a finally was added: the
|
||||||
|
// captured shutdown hook's close order (FleetdAssemblyLifecycleTest) closes the herdr
|
||||||
|
// client last, so ports.herdr.closed flips to true only if this hook really executed.
|
||||||
|
ports.shutdownHook.run();
|
||||||
|
assertTrue(ports.herdr.closed, "the captured shutdown hook must have run and closed herdr — "
|
||||||
|
+ "proof this test's assembled background loops/scheduler were torn down");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static MessageService.TaskView awaitTerminal(MessageService messages, String ticket)
|
||||||
|
throws InterruptedException {
|
||||||
|
long deadline = System.currentTimeMillis() + 5000;
|
||||||
|
MessageService.TaskView view = messages.poll(ticket);
|
||||||
|
while (view.phase() == MessageService.Phase.PENDING && System.currentTimeMillis() < deadline) {
|
||||||
|
//noinspection BusyWait
|
||||||
|
Thread.sleep(10);
|
||||||
|
view = messages.poll(ticket);
|
||||||
|
}
|
||||||
|
return view;
|
||||||
|
}
|
||||||
|
}
|
||||||
+241
@@ -0,0 +1,241 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.lead.LeadContextGauge;
|
||||||
|
import dev.ltms.fleet.msg.LeadHeartbeatLoop;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.lang.reflect.Method;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #630, and fleetd #612's ranks for this call site. {@code FleetdAssembly.java:402}
|
||||||
|
* computes {@code requireOperatorConfirm} from the effective {@code leadRollover.requireOperatorConfirm}
|
||||||
|
* config, and {@code :409} threads it as the 14th argument into the full {@link LeadHeartbeatLoop}
|
||||||
|
* constructor. Measured on 26f1986: dropping that one argument so the 13-argument overload is
|
||||||
|
* selected instead (it delegates with {@code true} hardcoded — see that overload's own javadoc,
|
||||||
|
* fleetd #621) compiles with 0 errors and leaves all 1883 tests green, both with and without the
|
||||||
|
* argument. In production this means the daemon keeps starting and keeps nudging, but the
|
||||||
|
* context-high notice silently goes back to telling EVERY lead to ask the operator before a
|
||||||
|
* context roll — on a host that set {@code requireOperatorConfirm: false} specifically so it would
|
||||||
|
* not have to. That is the operator's own fix silently reverting, with a fully green suite.
|
||||||
|
*
|
||||||
|
* <p>{@code LeadHeartbeatLoopTest} already proves {@link LeadHeartbeatLoop}'s package-private
|
||||||
|
* {@code contextNotice(boolean, LeadContextGauge.Reading, boolean, boolean)} branches correctly on
|
||||||
|
* its own {@code requireOperatorConfirm} argument — that the METHOD works. It says nothing about
|
||||||
|
* which value {@code FleetdAssembly} actually passes into the constructed loop, so it is not
|
||||||
|
* reused here as coverage for the call site.
|
||||||
|
*
|
||||||
|
* <p>This test assembles the real daemon TWICE — once with {@code leadRollover.requireOperatorConfirm:
|
||||||
|
* false}, once with {@code true} — pulls the REAL {@code requireOperatorConfirm} field out of the
|
||||||
|
* REAL, assembled {@link LeadHeartbeatLoop} each time (reflection: the field, and {@code
|
||||||
|
* contextNotice} itself, are package-private to {@code dev.ltms.fleet.msg}, and nothing public
|
||||||
|
* exposes either — the same technique {@code StatusPollerResilienceTest} already uses in this
|
||||||
|
* suite), and calls the REAL {@code contextNotice} method with that field's value to produce the
|
||||||
|
* actual notice text the assembled loop would append to a nudge. Both directions are asserted: a
|
||||||
|
* one-directional test here would pass on a constant.
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyRequireOperatorConfirmBehaviouralTest {
|
||||||
|
|
||||||
|
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException("no broker: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException("no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
this.shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir, boolean requireOperatorConfirm) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
leadHeartbeat:
|
||||||
|
idleAfterSeconds: 600
|
||||||
|
backoffMs: 15000
|
||||||
|
quietNudgeCap: 5
|
||||||
|
leadRollover:
|
||||||
|
handoverPath: handover.md
|
||||||
|
requireOperatorConfirm: %s
|
||||||
|
""".formatted(requireOperatorConfirm));
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Carries both the assembled loop under test AND its {@link RecordingResourcePorts}, so the
|
||||||
|
* caller can tear the assembly down (this test assembles the real daemon TWICE — see the class
|
||||||
|
* javadoc — and each assembly needs its own teardown, not just the last one). */
|
||||||
|
private record Assembled(LeadHeartbeatLoop heartbeat, RecordingResourcePorts ports) {
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Assembled assembleHeartbeat(Path dir, boolean requireOperatorConfirm) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir, requireOperatorConfirm);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
assertNotNull(ports.shutdownHook, "FleetdAssembly must have registered a shutdown hook");
|
||||||
|
|
||||||
|
LeadHeartbeatLoop heartbeat = runtime.heartbeat();
|
||||||
|
assertNotNull(heartbeat, "control: leadHeartbeat: is configured, so FleetdAssembly.assembleAndStart "
|
||||||
|
+ "must have built a real LeadHeartbeatLoop");
|
||||||
|
return new Assembled(heartbeat, ports);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Pulls the REAL {@code requireOperatorConfirm} field off the REAL, assembled loop. */
|
||||||
|
private static boolean assembledRequireOperatorConfirm(LeadHeartbeatLoop heartbeat) throws Exception {
|
||||||
|
Field field = LeadHeartbeatLoop.class.getDeclaredField("requireOperatorConfirm");
|
||||||
|
field.setAccessible(true);
|
||||||
|
return field.getBoolean(heartbeat);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Calls the REAL, package-private {@code contextNotice(boolean, Reading, boolean, boolean)} via reflection. */
|
||||||
|
private static String contextNotice(boolean enabled, LeadContextGauge.Reading reading, boolean alreadyNotified,
|
||||||
|
boolean requireOperatorConfirm) throws Exception {
|
||||||
|
Method method = LeadHeartbeatLoop.class.getDeclaredMethod("contextNotice", boolean.class,
|
||||||
|
LeadContextGauge.Reading.class, boolean.class, boolean.class);
|
||||||
|
method.setAccessible(true);
|
||||||
|
return (String) method.invoke(null, enabled, reading, alreadyNotified, requireOperatorConfirm);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("[BEHAVIOURAL] the real assembled LeadHeartbeatLoop's context-high notice tracks "
|
||||||
|
+ "leadRollover.requireOperatorConfirm — BOTH directions")
|
||||||
|
void assembledRequireOperatorConfirmControlsNoticeWording(@TempDir Path dir) throws Exception {
|
||||||
|
LeadContextGauge.Reading highReading = new LeadContextGauge.Reading(LeadContextGauge.State.HIGH,
|
||||||
|
250_000L, 2);
|
||||||
|
|
||||||
|
String noticeFalse;
|
||||||
|
String noticeTrue;
|
||||||
|
|
||||||
|
// --- direction 1: requireOperatorConfirm: false -----------------------------------------
|
||||||
|
Path falseDir = dir.resolve("false");
|
||||||
|
Files.createDirectories(falseDir);
|
||||||
|
Assembled assembledFalse = assembleHeartbeat(falseDir, false);
|
||||||
|
// Surefire runs the whole suite in one JVM fork, so each assembly's scheduler/loops must be
|
||||||
|
// torn down here, on the failure path too — hence try/finally per assembly (this test
|
||||||
|
// assembles TWICE, so both need their own teardown, not just the last one).
|
||||||
|
try {
|
||||||
|
boolean fieldFalse = assembledRequireOperatorConfirm(assembledFalse.heartbeat());
|
||||||
|
assertFalse(fieldFalse, "FleetdAssembly.java:402/:409 must thread leadRollover."
|
||||||
|
+ "requireOperatorConfirm: false into the assembled LeadHeartbeatLoop's own field — "
|
||||||
|
+ "dropping the 14th constructor argument selects the 13-argument overload, which "
|
||||||
|
+ "hardcodes true regardless of config (fleetd #621), and this would read true instead");
|
||||||
|
|
||||||
|
noticeFalse = contextNotice(true, highReading, false, fieldFalse);
|
||||||
|
assertTrue(noticeFalse.contains("Decide for yourself when to confirm"),
|
||||||
|
"with requireOperatorConfirm: false, the assembled loop's own notice must tell the "
|
||||||
|
+ "lead it can decide for itself — got: " + noticeFalse);
|
||||||
|
assertFalse(noticeFalse.contains("ask the operator") || noticeFalse.contains("Only the operator"),
|
||||||
|
"with requireOperatorConfirm: false, the assembled loop's own notice must NOT ask the "
|
||||||
|
+ "operator — got: " + noticeFalse);
|
||||||
|
} finally {
|
||||||
|
// Proof the teardown actually ran, not just an assurance that a finally was added: the
|
||||||
|
// captured shutdown hook's close order (FleetdAssemblyLifecycleTest) closes the herdr
|
||||||
|
// client last, so ports.herdr.closed flips to true only if this hook really executed.
|
||||||
|
assembledFalse.ports().shutdownHook.run();
|
||||||
|
assertTrue(assembledFalse.ports().herdr.closed, "the captured shutdown hook must have run "
|
||||||
|
+ "and closed herdr — proof this assembly's background loops/scheduler were torn down");
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- direction 2: requireOperatorConfirm: true -------------------------------------------
|
||||||
|
Path trueDir = dir.resolve("true");
|
||||||
|
Files.createDirectories(trueDir);
|
||||||
|
Assembled assembledTrue = assembleHeartbeat(trueDir, true);
|
||||||
|
try {
|
||||||
|
boolean fieldTrue = assembledRequireOperatorConfirm(assembledTrue.heartbeat());
|
||||||
|
assertTrue(fieldTrue, "FleetdAssembly.java:402/:409 must thread leadRollover."
|
||||||
|
+ "requireOperatorConfirm: true into the assembled LeadHeartbeatLoop's own field");
|
||||||
|
|
||||||
|
noticeTrue = contextNotice(true, highReading, false, fieldTrue);
|
||||||
|
assertTrue(noticeTrue.contains("ask the operator") && noticeTrue.contains("Only the operator can approve the roll"),
|
||||||
|
"with requireOperatorConfirm: true, the assembled loop's own notice must ask the "
|
||||||
|
+ "operator — got: " + noticeTrue);
|
||||||
|
assertFalse(noticeTrue.contains("Decide for yourself when to confirm"),
|
||||||
|
"with requireOperatorConfirm: true, the assembled loop's own notice must NOT tell "
|
||||||
|
+ "the lead it can decide for itself — got: " + noticeTrue);
|
||||||
|
} finally {
|
||||||
|
assembledTrue.ports().shutdownHook.run();
|
||||||
|
assertTrue(assembledTrue.ports().herdr.closed, "the captured shutdown hook must have run "
|
||||||
|
+ "and closed herdr — proof this assembly's background loops/scheduler were torn down");
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- the two directions must actually differ: a constant return would pass both assertion
|
||||||
|
// blocks above vacuously if they happened to share wording, so compare them directly too.
|
||||||
|
assertTrue(!noticeFalse.equals(noticeTrue),
|
||||||
|
"the two directions must produce genuinely different notice text — got the same "
|
||||||
|
+ "text for both: " + noticeFalse);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,173 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import ch.qos.logback.classic.Level;
|
||||||
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 A-gaps (gap 2): {@code Fleetd.java:185} used to call {@code
|
||||||
|
* reportRoleFallbackGaps(cfg)} <em>outside</em> the boundary {@code FleetdAssembly.assembleAndStart}
|
||||||
|
* — the assembly call itself sat at line 201, after it — so nothing that drives the assembly (the
|
||||||
|
* seam {@code FleetdAssemblyLifecycleTest} exercises) could ever notice the call being deleted.
|
||||||
|
* {@code Fleetd.main} still refuses a bad config end to end (see {@code
|
||||||
|
* FleetdStartupValidationTest}), but that only pins {@code validateAll()} and {@code
|
||||||
|
* assertChartersNameOnlyRegisteredTools} — a validator that <em>throws</em>. {@code
|
||||||
|
* reportRoleFallbackGaps} only logs; nothing about {@code main} throwing or not throwing can
|
||||||
|
* observe whether that particular call ran.
|
||||||
|
*
|
||||||
|
* <p>This drives {@link FleetdAssembly#assembleAndStart} directly — never a copy of its logic —
|
||||||
|
* with a config that has no {@code fleet:} pools or charters configured for any role, so every role
|
||||||
|
* trips both of {@code reportRoleFallbackGaps}' log branches, and asserts on the real log line a
|
||||||
|
* {@code ListAppender} attached to the shared {@code Fleetd}/{@code FleetdAssembly} logger
|
||||||
|
* captures. Deleting the call from {@code FleetdAssembly} (verified by hand, see the ticket) turns
|
||||||
|
* this test red; deleting it from {@code Fleetd.main} instead (its old location) would not, which
|
||||||
|
* is exactly the gap this test closes.
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyRoleFallbackBoundaryTest {
|
||||||
|
|
||||||
|
/** Minimal fake {@link ResourcePorts}: enough for {@code assembleAndStart} to run with no real I/O. */
|
||||||
|
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||||
|
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> replyInbox;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException(
|
||||||
|
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
this.shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
// No real HTTP bind in a unit test.
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class SentinelReplyInbox implements ReplyInbox {
|
||||||
|
@Override
|
||||||
|
public void own(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void release(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void publish(String target, String msgId, String content) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<InboxMessage> peek(String target) {
|
||||||
|
return List.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean ack(String target, String msgId) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
// Deliberately no `fleet:` block at all: every MemberRole has neither a pool nor a
|
||||||
|
// charter, so reportRoleFallbackGaps' "role fallback: no fleet.<role>s: pool for ..."
|
||||||
|
// branch is guaranteed to log something to assert on.
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
""");
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void assembleAndStartItselfReportsTheRoleFallbackGaps(@TempDir Path dir) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||||
|
|
||||||
|
try (var captured = CapturedLog.at(Fleetd.class, Level.INFO)) {
|
||||||
|
FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
|
||||||
|
String infoLines = captured.events().stream()
|
||||||
|
.filter(e -> e.getLevel() == Level.INFO)
|
||||||
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
|
.reduce("", (a, b) -> a + "\n" + b);
|
||||||
|
assertTrue(infoLines.contains("role fallback: no fleet.<role>s: pool for"),
|
||||||
|
() -> "FleetdAssembly.assembleAndStart itself must call reportRoleFallbackGaps "
|
||||||
|
+ "(fleetd #612 A-gaps gap 2) — captured INFO lines: " + infoLines);
|
||||||
|
} finally {
|
||||||
|
if (ports.shutdownHook != null) {
|
||||||
|
ports.shutdownHook.run();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,190 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.inject.Injector;
|
||||||
|
import dev.ltms.fleet.inject.StatusPoller;
|
||||||
|
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||||
|
import dev.ltms.fleet.msg.LeadCoordLoop;
|
||||||
|
import dev.ltms.fleet.msg.LeadMessage;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.AfterEach;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Asserts that the assembled loops use the production reminder, coordination, and delivery timing
|
||||||
|
* defaults when no {@code primary:} block configures the reply-push values.
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyTimingDefaultsTest {
|
||||||
|
|
||||||
|
private static final class FakeLeadChannel implements LeadChannelHandle {
|
||||||
|
@Override
|
||||||
|
public void publish(String toCoordId, LeadMessage message) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<LeadMessage> peek() {
|
||||||
|
return List.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void ack(String msgId) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public String selfCoordId() {
|
||||||
|
return "test-lead";
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean heldDurable() {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public MailboxState inspect(String coordId) {
|
||||||
|
return MailboxState.unknown(coordId);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class TestResourcePorts implements ResourcePorts {
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> new ReplyInbox() {
|
||||||
|
@Override public void own(String target) { }
|
||||||
|
@Override public void release(String target) { }
|
||||||
|
@Override public void publish(String target, String msgId, String content) { }
|
||||||
|
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||||
|
@Override public boolean ack(String target, String msgId) { return false; }
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> new FakeLeadChannel();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("FakeHerdr is healthy; no poll wait is expected");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private TestResourcePorts ports;
|
||||||
|
|
||||||
|
@AfterEach
|
||||||
|
void tearDown() {
|
||||||
|
if (ports != null && ports.shutdownHook != null) {
|
||||||
|
ports.shutdownHook.run();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
coordinator:
|
||||||
|
uri: "amqp://fake-lead-broker/vh"
|
||||||
|
selfId: "test-lead"
|
||||||
|
""");
|
||||||
|
return FleetConfig.load(file);
|
||||||
|
}
|
||||||
|
|
||||||
|
private FleetdRuntime assemble(Path dir) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
ports = new TestResourcePorts();
|
||||||
|
return FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg,
|
||||||
|
new ConfigRef(dir.resolve("fleetd.yaml"), cfg), new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static long longField(Object target, String name) throws Exception {
|
||||||
|
Field field = target.getClass().getDeclaredField(name);
|
||||||
|
field.setAccessible(true);
|
||||||
|
return field.getLong(target);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void productionBootPathUsesTheExpectedLoopTimingDefaults(@TempDir Path dir) throws Exception {
|
||||||
|
FleetdRuntime runtime = assemble(dir);
|
||||||
|
|
||||||
|
ReplyPushLoop pushLoop = runtime.pushLoop();
|
||||||
|
assertEquals(5, longField(pushLoop, "maxReminders"),
|
||||||
|
"without primary:, ReplyPushLoop must stop after five reminder attempts");
|
||||||
|
assertEquals(15_000L, longField(pushLoop, "backoffMs"),
|
||||||
|
"without primary:, ReplyPushLoop must wait fifteen seconds before the next reminder");
|
||||||
|
|
||||||
|
LeadCoordLoop leadCoordLoop = runtime.leadCoordLoop();
|
||||||
|
assertNotNull(leadCoordLoop, "control: coordinator: must build LeadCoordLoop");
|
||||||
|
assertEquals(3_000L, longField(leadCoordLoop, "intervalMs"),
|
||||||
|
"LeadCoordLoop must poll for peer-lead mail every three seconds");
|
||||||
|
|
||||||
|
StatusPoller poller = runtime.poller();
|
||||||
|
assertEquals(Injector.POLL_INTERVAL_MILLIS, longField(poller, "intervalMillis"),
|
||||||
|
"StatusPoller must use Injector's delivery poll interval");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,184 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.AgentStatus;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.inject.Injector;
|
||||||
|
import dev.ltms.fleet.inject.TurnListener;
|
||||||
|
import dev.ltms.fleet.msg.Rendezvous;
|
||||||
|
import dev.ltms.fleet.msg.TurnToken;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.CompletableFuture;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 rank 12 — {@code FleetdAssembly} supplies the {@link Injector}'s {@code TurnRegistrar}
|
||||||
|
* with {@code Fleetd.turnRegistrar(completion)}. The normal delivery path cannot distinguish that
|
||||||
|
* registrar from {@code TurnRegistrar.NOOP}: {@link dev.ltms.fleet.inject.CompletionResolver#onDelivered}
|
||||||
|
* registers the same turn shortly afterwards. The distinction matters when a delivery listener throws
|
||||||
|
* after the pane received the message but before normal completion runs. The registrar must already have
|
||||||
|
* registered the waiter with the REAL assembled resolver, so a later completion can still resolve it.
|
||||||
|
*
|
||||||
|
* <p>The test installs a throwing wrapper around the real assembled listener. It does not replace the
|
||||||
|
* registrar or resolver. This constructs the narrow failure condition without a sleep, then observes the
|
||||||
|
* real resolver through {@link FleetdRuntime#completion()}. An inert registrar, or a registrar wired to a
|
||||||
|
* throwaway resolver, leaves this waiter's turn absent from the real resolver and makes the final assertion
|
||||||
|
* fail.
|
||||||
|
*/
|
||||||
|
class FleetdAssemblyTurnRegistrarBehaviouralTest {
|
||||||
|
|
||||||
|
private static final String TARGET = "term_a";
|
||||||
|
|
||||||
|
private static final class ControllableResourcePorts implements ResourcePorts {
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
final AtomicLong nowNanos = new AtomicLong(1_000_000_000L);
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
void advanceSeconds(long seconds) {
|
||||||
|
nowNanos.addAndGet(TimeUnit.SECONDS.toNanos(seconds));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException("no broker: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException("no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return nowNanos::get;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return nowNanos::get;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
// Do not bind a real port in this assembly test.
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("FakeHerdr is healthy; no poll wait is expected");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
fleet:
|
||||||
|
leaders:
|
||||||
|
primary:
|
||||||
|
tab: "lead: primary"
|
||||||
|
profile: sonnet
|
||||||
|
profiles:
|
||||||
|
sonnet:
|
||||||
|
subscription: true
|
||||||
|
argv: ["ccs", "sonnet"]
|
||||||
|
""");
|
||||||
|
return FleetConfig.load(file);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void realAssembledResolverStillResolvesAfterDeliveredListenerThrows(@TempDir Path dir) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
ControllableResourcePorts ports = new ControllableResourcePorts();
|
||||||
|
ports.herdr.withTab("w2", "w2:t7", "lead: primary");
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg,
|
||||||
|
new ConfigRef(dir.resolve("fleetd.yaml"), cfg), new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||||
|
try {
|
||||||
|
Injector injector = runtime.injector();
|
||||||
|
installThrowingDeliveredListener(injector);
|
||||||
|
|
||||||
|
CompletableFuture<Rendezvous.Resolution> waiter = new CompletableFuture<>();
|
||||||
|
injector.enqueue(TARGET, "brief", new TurnToken(TARGET, waiter));
|
||||||
|
|
||||||
|
IllegalStateException thrown = assertThrows(IllegalStateException.class,
|
||||||
|
() -> injector.onStatus(TARGET, AgentStatus.IDLE),
|
||||||
|
"control: delivery must reach the installed listener and it must throw after delivery");
|
||||||
|
assertTrue(thrown.getMessage().contains("listener failure"), thrown::getMessage);
|
||||||
|
|
||||||
|
ports.herdr.readText("worker report after the listener failure");
|
||||||
|
ports.advanceSeconds(3);
|
||||||
|
runtime.completion().resolveBeforePostAction(TARGET);
|
||||||
|
|
||||||
|
assertTrue(waiter.isDone(),
|
||||||
|
"FleetdAssembly must wire the Injector registrar to this runtime's real CompletionResolver: "
|
||||||
|
+ "after a delivered listener throws, resolveBeforePostAction must still find and "
|
||||||
|
+ "resolve the registered waiter");
|
||||||
|
} finally {
|
||||||
|
assertTrue(ports.shutdownHook != null, "control: assembly must capture its shutdown hook");
|
||||||
|
ports.shutdownHook.run();
|
||||||
|
assertTrue(ports.herdr.closed, "teardown control: the captured shutdown hook must close herdr");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void installThrowingDeliveredListener(Injector injector) throws Exception {
|
||||||
|
Field field = Injector.class.getDeclaredField("turnListener");
|
||||||
|
field.setAccessible(true);
|
||||||
|
field.set(injector, new TurnListener() {
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onDelivered(String target, TurnToken token) {
|
||||||
|
throw new IllegalStateException("listener failure after delivery");
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,192 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import ch.qos.logback.classic.Level;
|
||||||
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrException;
|
||||||
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.util.concurrent.atomic.AtomicBoolean;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
import static org.junit.jupiter.api.Assertions.fail;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #498: {@code Fleetd.awaitHerdr} used to return a bare {@code boolean}, collapsing "the
|
||||||
|
* configured wait budget genuinely ran out" and "the waiting thread was interrupted, possibly
|
||||||
|
* milliseconds in" onto the same {@code false} — and the caller's log line printed only the
|
||||||
|
* configured budget, never how long the wait actually ran. This class covers both halves of the
|
||||||
|
* fix:
|
||||||
|
* <ul>
|
||||||
|
* <li>the seam — {@link Fleetd#awaitHerdr} itself, driven with an injected clock and a stub
|
||||||
|
* {@link HerdrClient}, one test per {@link Fleetd.HerdrWaitResult};</li>
|
||||||
|
* <li>the call site — {@link Fleetd#logHerdrWaitOutcomeAndShouldReap}, the exact decision {@code
|
||||||
|
* main} calls (extracted here because {@code main} itself boots the whole daemon and cannot
|
||||||
|
* be driven from a unit test), pinning the three distinct log messages it emits.</li>
|
||||||
|
* </ul>
|
||||||
|
* Every expected message below is a plain literal, not built from {@code HERDR_WAIT_SECONDS} or
|
||||||
|
* any other production constant — a test that derives its expectation the way the code does
|
||||||
|
* cannot see a change to either (fleetd #496's identical trap).
|
||||||
|
*/
|
||||||
|
class FleetdAwaitHerdrTest {
|
||||||
|
|
||||||
|
// ---- the seam: Fleetd.awaitHerdr ----------------------------------------------------------
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void answeredReturnsImmediatelyWithZeroElapsedAndNeverPolls() {
|
||||||
|
HerdrStub herdr = new HerdrStub(0); // succeeds on the very first call
|
||||||
|
LongSupplier clock = fixedClock(1_000L);
|
||||||
|
AtomicBoolean polled = new AtomicBoolean(false);
|
||||||
|
Runnable poller = () -> polled.set(true);
|
||||||
|
|
||||||
|
Fleetd.HerdrAwaitOutcome outcome = Fleetd.awaitHerdr(herdr, clock, poller);
|
||||||
|
|
||||||
|
assertEquals(Fleetd.HerdrWaitResult.ANSWERED, outcome.result());
|
||||||
|
assertEquals(0L, outcome.elapsedNanos(), "a fixed clock must measure zero elapsed time");
|
||||||
|
assertFalse(polled.get(), "herdr answering on the first try must never poll");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void deadlinePassedIsMeasuredNotAssumed() {
|
||||||
|
HerdrStub herdr = new HerdrStub(-1); // never succeeds
|
||||||
|
// call order inside awaitHerdr: start, then per failed attempt: deadline-check, elapsed-calc
|
||||||
|
ScriptedClock clock = new ScriptedClock(0L, 30_500_000_000L, 30_500_000_000L);
|
||||||
|
Runnable poller = () -> fail("the deadline was already exceeded on the first attempt — must not poll");
|
||||||
|
|
||||||
|
Fleetd.HerdrAwaitOutcome outcome = Fleetd.awaitHerdr(herdr, clock, poller);
|
||||||
|
|
||||||
|
assertEquals(Fleetd.HerdrWaitResult.DEADLINE_PASSED, outcome.result());
|
||||||
|
assertEquals(30_500_000_000L, outcome.elapsedNanos(),
|
||||||
|
"elapsed must be the MEASURED clock delta, not the configured budget");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void interruptedIsDistinctFromDeadlinePassedAndPreservesTheInterruptFlag() {
|
||||||
|
HerdrStub herdr = new HerdrStub(-1); // never succeeds
|
||||||
|
// start=0, deadline-check returns 500ms (well under the 30s budget) -> not deadline-passed,
|
||||||
|
// then the poller interrupts, and the elapsed-calc call returns 750ms.
|
||||||
|
ScriptedClock clock = new ScriptedClock(0L, 500_000_000L, 750_000_000L);
|
||||||
|
Runnable poller = () -> Thread.currentThread().interrupt();
|
||||||
|
|
||||||
|
try {
|
||||||
|
Fleetd.HerdrAwaitOutcome outcome = Fleetd.awaitHerdr(herdr, clock, poller);
|
||||||
|
|
||||||
|
assertEquals(Fleetd.HerdrWaitResult.INTERRUPTED, outcome.result());
|
||||||
|
assertEquals(750_000_000L, outcome.elapsedNanos(),
|
||||||
|
"elapsed must be measured even when the wait ends via interruption, not the deadline");
|
||||||
|
assertTrue(Thread.currentThread().isInterrupted(),
|
||||||
|
"the interrupt flag the old code re-set must still be set on return");
|
||||||
|
} finally {
|
||||||
|
Thread.interrupted(); // clear it so it cannot leak into another test on this thread
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- the call site: Fleetd.logHerdrWaitOutcomeAndShouldReap -------------------------------
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void answeredLogsNothingAndSaysReap() {
|
||||||
|
try (CapturedLog log = attach()) {
|
||||||
|
boolean shouldReap = Fleetd.logHerdrWaitOutcomeAndShouldReap(
|
||||||
|
new Fleetd.HerdrAwaitOutcome(Fleetd.HerdrWaitResult.ANSWERED, 0L));
|
||||||
|
|
||||||
|
assertTrue(shouldReap, "only ANSWERED should tell main to reap orphan workers");
|
||||||
|
assertEquals(0, log.events().size(), "the answered path logs nothing itself");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void deadlinePassedLogsConfiguredAndMeasuredElapsedTogether() {
|
||||||
|
try (CapturedLog log = attach()) {
|
||||||
|
boolean shouldReap = Fleetd.logHerdrWaitOutcomeAndShouldReap(
|
||||||
|
new Fleetd.HerdrAwaitOutcome(Fleetd.HerdrWaitResult.DEADLINE_PASSED, 30_500_000_000L));
|
||||||
|
|
||||||
|
assertFalse(shouldReap, "a deadline-passed wait must not tell main to reap");
|
||||||
|
assertEquals(1, log.events().size());
|
||||||
|
ILoggingEvent event = log.events().getFirst();
|
||||||
|
assertEquals(Level.WARN, event.getLevel());
|
||||||
|
assertEquals("herdr did not answer within the configured wait (configured=30s "
|
||||||
|
+ "elapsed=30500ms) — starting anyway; /healthz will report degraded until it "
|
||||||
|
+ "comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||||
|
event.getFormattedMessage());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void interruptedLogsItsOwnMessageAndNeverClaimsTheBudgetElapsed() {
|
||||||
|
try (CapturedLog log = attach()) {
|
||||||
|
// 3ms: the ticket's own example of "a few milliseconds in", not the 30s budget.
|
||||||
|
boolean shouldReap = Fleetd.logHerdrWaitOutcomeAndShouldReap(
|
||||||
|
new Fleetd.HerdrAwaitOutcome(Fleetd.HerdrWaitResult.INTERRUPTED, 3_000_000L));
|
||||||
|
|
||||||
|
assertFalse(shouldReap, "an interrupted wait must not tell main to reap");
|
||||||
|
assertEquals(1, log.events().size());
|
||||||
|
ILoggingEvent event = log.events().getFirst();
|
||||||
|
assertEquals(Level.WARN, event.getLevel());
|
||||||
|
String message = event.getFormattedMessage();
|
||||||
|
assertEquals("herdr wait was interrupted before the configured wait ran out "
|
||||||
|
+ "(configured=30s elapsed=3ms) — starting anyway; /healthz will report "
|
||||||
|
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||||
|
message);
|
||||||
|
assertFalse(message.contains("did not answer"),
|
||||||
|
"an interrupted wait must not be reported as if herdr failed to answer within the budget");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- fixtures --------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/** Always returns the same value, i.e. a clock that measures zero elapsed time. */
|
||||||
|
private static LongSupplier fixedClock(long value) {
|
||||||
|
return () -> value;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Returns each value in order, then repeats the last one for any call beyond the list. */
|
||||||
|
private static final class ScriptedClock implements LongSupplier {
|
||||||
|
private final long[] values;
|
||||||
|
private int index;
|
||||||
|
|
||||||
|
ScriptedClock(long... values) {
|
||||||
|
this.values = values;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public long getAsLong() {
|
||||||
|
long v = values[Math.min(index, values.length - 1)];
|
||||||
|
if (index < values.length - 1) {
|
||||||
|
index++;
|
||||||
|
}
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Fails {@code failuresBeforeSuccess} times, then succeeds forever; {@code -1} never succeeds. */
|
||||||
|
private static final class HerdrStub implements HerdrClient {
|
||||||
|
private final int failuresBeforeSuccess;
|
||||||
|
private int calls;
|
||||||
|
|
||||||
|
HerdrStub(int failuresBeforeSuccess) {
|
||||||
|
this.failuresBeforeSuccess = failuresBeforeSuccess;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public JsonNode call(String method, Object params) throws HerdrException {
|
||||||
|
calls++;
|
||||||
|
if (failuresBeforeSuccess < 0 || calls <= failuresBeforeSuccess) {
|
||||||
|
throw new HerdrException("herdr not up yet");
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static CapturedLog attach() {
|
||||||
|
return CapturedLog.at(Fleetd.class, Level.DEBUG);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,207 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.OptionalLong;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 B3 — replaces {@code FleetdBackendQuarantineWiringTest} (fleetd #466), a source-text
|
||||||
|
* test that scraped {@code Fleetd.java} (now {@code FleetdAssembly.java}, moved there by fleetd #612
|
||||||
|
* Unit A) for the {@code BackendQuarantine.withEscalation(...)} call, and separately asserted the
|
||||||
|
* flat two-argument constructor's text was ABSENT. That proves the right method NAME appears in
|
||||||
|
* source; it proves nothing about what the constructed object actually DOES.
|
||||||
|
*
|
||||||
|
* <p>This test instead drives the REAL {@link BackendQuarantine} the real {@link
|
||||||
|
* FleetdAssembly#assembleAndStart} builds — reached through {@link
|
||||||
|
* dev.ltms.fleet.mcp.FleetMcp#quarantineSource()} on the real, live {@code FleetMcp} {@code
|
||||||
|
* FleetdRuntime} owns — and asserts the ONE behavioural difference {@code withEscalation} and the
|
||||||
|
* flat constructor actually produce (see {@link BackendQuarantine}'s own class doc, "Mechanism"):
|
||||||
|
* quarantining the same credential twice in a row, within one base cooldown of the first deadline,
|
||||||
|
* must escalate the second cooldown past the first. A flat instance reports the identical cooldown
|
||||||
|
* both times.
|
||||||
|
*/
|
||||||
|
class FleetdBackendQuarantineAssemblyTest {
|
||||||
|
|
||||||
|
/** Base cooldown used throughout — long enough that rounding never blurs the 2x escalation. */
|
||||||
|
private static final int COOLDOWN_SECONDS = 100;
|
||||||
|
|
||||||
|
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||||
|
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||||
|
final AtomicLong clockNanos = new AtomicLong(0L);
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> replyInbox;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException(
|
||||||
|
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
// Controllable: the SAME LongSupplier instance BackendQuarantine.withEscalation(...) is
|
||||||
|
// built with, so advancing clockNanos after assembly moves the quarantine tracker's own
|
||||||
|
// clock, with no real sleep needed to observe escalation.
|
||||||
|
return clockNanos::get;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return clockNanos::get;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class SentinelReplyInbox implements ReplyInbox, AutoCloseable {
|
||||||
|
@Override
|
||||||
|
public void own(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void release(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void publish(String target, String msgId, String content) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<InboxMessage> peek(String target) {
|
||||||
|
return List.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean ack(String target, String msgId) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
broker:
|
||||||
|
uri: "amqp://fake-test-broker/vh"
|
||||||
|
quarantineCooldownSeconds: %d
|
||||||
|
""".formatted(COOLDOWN_SECONDS));
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("[BEHAVIOURAL] the real assembled BackendQuarantine escalates a repeated exhaustion, "
|
||||||
|
+ "which the flat two-argument constructor can never do")
|
||||||
|
void assembledQuarantineEscalatesOnARepeatedExhaustion(@TempDir Path dir) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
|
||||||
|
BackendQuarantine quarantine = runtime.mcp().quarantineSource().quarantine();
|
||||||
|
|
||||||
|
// First exhaustion, at clock=0: a fresh occurrence, blocked for exactly the base cooldown.
|
||||||
|
quarantine.quarantine("cred-x");
|
||||||
|
BackendQuarantine.Status first = quarantine.status("cred-x").orElseThrow(
|
||||||
|
() -> new AssertionError("credential must be quarantined immediately after quarantine()"));
|
||||||
|
assertEquals(1, first.repeatCount(), "the first call is repeat #1");
|
||||||
|
assertEquals(COOLDOWN_SECONDS, first.remainingSeconds(),
|
||||||
|
"a fresh quarantine blocks for exactly the base cooldown");
|
||||||
|
|
||||||
|
// Second exhaustion, arriving just after the first deadline — well within one base cooldown
|
||||||
|
// of it, so this is a CONTINUATION of the same streak (repeat #2), not a fresh occurrence.
|
||||||
|
long firstDeadlineNanos = COOLDOWN_SECONDS * 1_000_000_000L;
|
||||||
|
ports.clockNanos.set(firstDeadlineNanos + 1);
|
||||||
|
quarantine.quarantine("cred-x");
|
||||||
|
BackendQuarantine.Status second = quarantine.status("cred-x").orElseThrow(
|
||||||
|
() -> new AssertionError("credential must be quarantined immediately after the second "
|
||||||
|
+ "quarantine() call"));
|
||||||
|
assertEquals(2, second.repeatCount(), "the second call, arriving within one base cooldown of "
|
||||||
|
+ "the first deadline, continues the streak as repeat #2");
|
||||||
|
|
||||||
|
// The one behavioural difference: withEscalation doubles the cooldown on repeat #2 (capped
|
||||||
|
// well above this at 12x base), the flat two-argument constructor never grows past the base
|
||||||
|
// cooldown no matter how many times quarantine() is called in a row.
|
||||||
|
assertEquals(2 * COOLDOWN_SECONDS, second.remainingSeconds(),
|
||||||
|
"withEscalation's default backoff doubles the cooldown on the second consecutive "
|
||||||
|
+ "exhaustion — this is the exact call FleetdAssembly.java makes at the "
|
||||||
|
+ "BackendQuarantine.withEscalation(...) call site");
|
||||||
|
assertTrue(second.remainingSeconds() > first.remainingSeconds(),
|
||||||
|
"the flat two-argument BackendQuarantine constructor would report the SAME remaining "
|
||||||
|
+ "seconds both times — this inequality is what a mutation to the flat "
|
||||||
|
+ "constructor at that call site must fail");
|
||||||
|
|
||||||
|
// Also confirm isQuarantined/remainingSeconds agree, exercising the accessors a real caller
|
||||||
|
// (fleet_profiles / fleet_list, per BackendQuarantine's own class doc) actually reads.
|
||||||
|
assertTrue(quarantine.isQuarantined("cred-x"));
|
||||||
|
OptionalLong remaining = quarantine.remainingSeconds("cred-x");
|
||||||
|
assertTrue(remaining.isPresent());
|
||||||
|
assertEquals(2 * COOLDOWN_SECONDS, remaining.getAsLong());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,65 +0,0 @@
|
|||||||
package dev.ltms.fleet;
|
|
||||||
|
|
||||||
import java.nio.file.Files;
|
|
||||||
import java.nio.file.Path;
|
|
||||||
import org.junit.jupiter.api.DisplayName;
|
|
||||||
import org.junit.jupiter.api.Test;
|
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* fleetd #466 follow-up: {@code Fleetd.main} builds the daemon's one {@code BackendQuarantine}
|
|
||||||
* from {@link dev.ltms.fleet.placement.BackendQuarantine#withEscalation(java.util.function.LongSupplier,
|
|
||||||
* long)} — the escalating factory — rather than the plain two-argument constructor, which is still a
|
|
||||||
* flat cooldown (kept for backward compatibility, see that class's doc). {@code
|
|
||||||
* BackendQuarantineTest} proves {@code withEscalation} itself escalates, is ceilinged, and resets;
|
|
||||||
* it says nothing about which one {@code main} actually calls.
|
|
||||||
*
|
|
||||||
* <p>Measured directly: reverting {@code main} to {@code new BackendQuarantine(System::nanoTime,
|
|
||||||
* TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()))} — the pre-#466 flat call — compiles
|
|
||||||
* with 0 errors and leaves the entire 1608-test suite (including every {@code BackendQuarantineTest}
|
|
||||||
* case) green, because no other test constructs its {@code BackendQuarantine} through {@code main};
|
|
||||||
* every one of them builds its own instance directly. That silent regression is exactly the shape
|
|
||||||
* {@link FleetdLeadSeatWiringTest} and {@link FleetdCompletionResolverWiringTest} already guard
|
|
||||||
* against for their own constructor arguments — this is the same class of gap for fleetd #466's
|
|
||||||
* factory choice, following their approach.
|
|
||||||
*
|
|
||||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
|
||||||
* BackendQuarantine} and never runs {@code main} — a green result here proves only that the exact
|
|
||||||
* text {@code main} calls {@code BackendQuarantine.withEscalation(...)} rather than the flat
|
|
||||||
* constructor. It does not prove that call actually executes at startup (no test here starts the
|
|
||||||
* daemon), and it does not prove the escalation reaches a real backend or credential — only
|
|
||||||
* {@code BackendQuarantineTest} proves the factory's own behaviour, and only a live daemon proves
|
|
||||||
* the wiring runs.
|
|
||||||
*/
|
|
||||||
class FleetdBackendQuarantineWiringTest {
|
|
||||||
|
|
||||||
private static String fleetdSource() throws Exception {
|
|
||||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
@DisplayName("[SOURCE TEXT] main's BackendQuarantine local is still built from BackendQuarantine.withEscalation(...)")
|
|
||||||
void mainStillWiresTheEscalatingQuarantineFactory() throws Exception {
|
|
||||||
String source = fleetdSource();
|
|
||||||
assertTrue(source.contains(
|
|
||||||
"BackendQuarantine quarantine = BackendQuarantine.withEscalation(System::nanoTime,\n"
|
|
||||||
+ " TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));"),
|
|
||||||
"Fleetd.main's BackendQuarantine local must still be built from "
|
|
||||||
+ "BackendQuarantine.withEscalation(System::nanoTime, "
|
|
||||||
+ "TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds())). Reverting to the flat "
|
|
||||||
+ "two-argument constructor (fleetd #466's measured regression) compiles with 0 errors "
|
|
||||||
+ "and leaves the whole suite green, including every BackendQuarantineTest case that "
|
|
||||||
+ "proves the escalation itself works — this source check is what must go red instead. "
|
|
||||||
+ "A reverted daemon would go back to retrying a weekly subscription limit on every "
|
|
||||||
+ "flat ~30-minute cooldown, about 336 times across the week.");
|
|
||||||
|
|
||||||
// Negative form of the same check: the pre-#466 flat call, if it ever reappears at this
|
|
||||||
// declaration, must not be mistaken for the escalating one by a looser positive-only check.
|
|
||||||
assertFalse(source.contains(
|
|
||||||
"BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,\n"
|
|
||||||
+ " TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));"),
|
|
||||||
"main's BackendQuarantine local must never regress to the flat two-argument constructor");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,84 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
|
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Set;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #589 Group 2: {@link Fleetd#claudeCodeLauncher} is the factory that replaced {@code
|
||||||
|
* main}'s inline {@code new ClaudeCodeLauncher(...)} call, whose 10th argument is the CB-596 {@code
|
||||||
|
* memberCredentials} policy supplier ({@code () -> config.get().memberCredentials()}). Before this
|
||||||
|
* ticket that argument was untestable wiring: replacing it with {@code () -> null} compiled with 0
|
||||||
|
* errors and left every existing test green, since no existing test builds the exact object {@code
|
||||||
|
* main} wires and then spawns it. {@code memberCredentials} being a lambda is never itself {@code
|
||||||
|
* null}, so {@link dev.ltms.fleet.member.HerdrPeerLauncher#applyMemberCredentialPolicy} sees {@code
|
||||||
|
* memberCredentials.get() == null} and silently shadows nothing — reopening the exact CB-592
|
||||||
|
* exposure gap CB-596's policy closed (gitea issue #82).
|
||||||
|
*
|
||||||
|
* <p>This test drives the factory with a real {@link ConfigRef} carrying a {@code
|
||||||
|
* memberCredentials:} block, spawns through the resulting launcher, and inspects what {@code
|
||||||
|
* tab.create} actually carried — the same observable surface {@code ClaudeCodeLauncherTest}'s
|
||||||
|
* {@code everyKnownNameNotAllowedIsShadowedWithTheSentinel} uses for the launcher's own credential
|
||||||
|
* policy, applied here to prove {@code main}'s wiring reaches it.
|
||||||
|
*/
|
||||||
|
class FleetdClaudeCodeLauncherCredentialWiringTest {
|
||||||
|
|
||||||
|
private static final String YAML = """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
herdrSocket: ~/.config/herdr/herdr.sock
|
||||||
|
profiles:
|
||||||
|
ltms-local:
|
||||||
|
baseUrl: http://gx00.gw:8000
|
||||||
|
model: coder
|
||||||
|
memberCredentials:
|
||||||
|
policy: deny-by-default
|
||||||
|
known:
|
||||||
|
- GITEA_ACCESS_TOKEN
|
||||||
|
""";
|
||||||
|
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
private static Map<String, String> startEnv(FakeHerdr herdr) {
|
||||||
|
return (Map<String, String>) ((Map<String, Object>) herdr.lastCall("tab.create").params()).get("env");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("main's memberCredentials wiring reaches ClaudeCodeLauncher: a known-but-not-allowed "
|
||||||
|
+ "name is shadowed on spawn")
|
||||||
|
void memberCredentialsWiringReachesClaudeCodeLauncher(@TempDir Path dir) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, YAML);
|
||||||
|
FleetConfig cfg = FleetConfig.load(file);
|
||||||
|
ConfigRef config = ConfigRef.fixed(cfg);
|
||||||
|
FakeHerdr herdr = new FakeHerdr();
|
||||||
|
|
||||||
|
ClaudeCodeLauncher launcher = Fleetd.claudeCodeLauncher(new AgentControl(herdr),
|
||||||
|
new WorkspaceControl(herdr), new SubscriptionGuard(Set.of("gx00.gw")),
|
||||||
|
cfg.profiles(), cfg, config);
|
||||||
|
launcher.spawn();
|
||||||
|
|
||||||
|
String shadowed = startEnv(herdr).get("GITEA_ACCESS_TOKEN");
|
||||||
|
assertNotNull(shadowed,
|
||||||
|
"GITEA_ACCESS_TOKEN is 'known' but not 'allow'-ed in the loaded config — it must be "
|
||||||
|
+ "explicitly shadowed on spawn; replacing the memberCredentials supplier with "
|
||||||
|
+ "() -> null at the Fleetd.claudeCodeLauncher call site must fail this "
|
||||||
|
+ "assertion, since a null policy shadows nothing");
|
||||||
|
assertFalse(shadowed.isBlank(), "the overlay value must be non-blank");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,315 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.inject.CompletionResolver;
|
||||||
|
import dev.ltms.fleet.msg.Rendezvous;
|
||||||
|
import dev.ltms.fleet.msg.TurnToken;
|
||||||
|
import dev.ltms.fleet.placement.PlacementException;
|
||||||
|
import dev.ltms.fleet.session.MemberSession;
|
||||||
|
import dev.ltms.fleet.session.WorktreeRequest;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.CompletableFuture;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 step 2, Unit B1 — replaces {@code FleetdCompletionResolverWiringTest} (deleted in
|
||||||
|
* this same commit), whose four tests read {@code Fleetd.java}'s source text and asserted the
|
||||||
|
* {@code CompletionResolver} construction call still named the right arguments. That proved the
|
||||||
|
* call site's spelling, never that the assembled resolver actually behaves differently when an
|
||||||
|
* argument is dropped.
|
||||||
|
*
|
||||||
|
* <p>These tests drive {@link FleetdAssembly#assembleAndStart} — the real boot composition,
|
||||||
|
* fleetd #612 Unit A — and read {@link FleetdRuntime#completion()}: the exact {@link
|
||||||
|
* CompletionResolver} instance the assembled daemon uses, never a copy built alongside it for the
|
||||||
|
* test's benefit. Two behaviours are pinned, matching the ticket's own two measured mutations:
|
||||||
|
*
|
||||||
|
* <ul>
|
||||||
|
* <li>the 8th constructor argument ({@code Fleetd.worktreeBranchLookup(sessions::roster)}) —
|
||||||
|
* {@link #assembledResolverReportsTheMembersWorktreeAndBranchInAFallbackReport}; and</li>
|
||||||
|
* <li>the 5th/6th arguments ({@code backendErrorPatterns}, {@code backendErrorSink}, both
|
||||||
|
* assigned from {@code Fleetd}'s extracted factories rather than an inline lambda) —
|
||||||
|
* {@link #assembledResolverClassifiesAndCoolsOffOnAConfiguredBackendErrorPattern}.</li>
|
||||||
|
* </ul>
|
||||||
|
*
|
||||||
|
* <p>Both tests bypass {@link dev.ltms.fleet.inject.StatusPoller} and drive {@link
|
||||||
|
* CompletionResolver#onDelivered} / {@link CompletionResolver#resolveBeforePostAction} directly —
|
||||||
|
* the same public, synchronous entry points {@code CompletionResolverTest} uses — with a
|
||||||
|
* hand-built {@link CompletableFuture} waiter, so no real poller loop or herdr status poll is
|
||||||
|
* needed. The pane scrape comes from {@link FakeHerdr#readText}; the elapsed-time floor
|
||||||
|
* ({@code CompletionResolver.MIN_TURN_NANOS}) is controlled via a fake, advanceable {@link
|
||||||
|
* ResourcePorts#nanoClock()} rather than a real sleep.
|
||||||
|
*/
|
||||||
|
class FleetdCompletionResolverAssemblyTest {
|
||||||
|
|
||||||
|
/** Same shape as {@code FleetdAssemblyLifecycleTest}'s fake, plus a nanoClock this test can advance. */
|
||||||
|
private static final class ControllableResourcePorts implements ResourcePorts {
|
||||||
|
|
||||||
|
final FakeHerdr herdr;
|
||||||
|
final AtomicLong nowNanos = new AtomicLong(1_000_000_000L); // arbitrary non-zero start
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
ControllableResourcePorts(FakeHerdr herdr) {
|
||||||
|
this.herdr = herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
void advanceSeconds(long seconds) {
|
||||||
|
nowNanos.addAndGet(TimeUnit.SECONDS.toNanos(seconds));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
// Never invoked: this test's config has no `broker:` block, so Fleetd.selectReplyInbox
|
||||||
|
// returns the in-memory inbox before calling the opener at all.
|
||||||
|
return (uri, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException("replyInboxOpener must not be called — no broker: block");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
// Never invoked: no `coordinator:` block configured either.
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException("leadMailboxOpener must not be called — no coordinator: block");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return nowNanos::get;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return nowNanos::get;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
this.shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
// Deliberately never bind a real port.
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir, String profilesYaml, String extraGuardHost,
|
||||||
|
String worktreeRootYamlLine) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
%s
|
||||||
|
profiles:
|
||||||
|
%s
|
||||||
|
guard:
|
||||||
|
offSubscriptionHosts:
|
||||||
|
- %s
|
||||||
|
""".formatted(worktreeRootYamlLine == null ? "" : worktreeRootYamlLine, profilesYaml, extraGuardHost));
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void gitQuiet(Path cwd, String... args) throws Exception {
|
||||||
|
List<String> cmd = new java.util.ArrayList<>(List.of("git"));
|
||||||
|
cmd.addAll(List.of(args));
|
||||||
|
Process p = new ProcessBuilder(cmd).directory(cwd.toFile()).redirectErrorStream(true).start();
|
||||||
|
String out = new String(p.getInputStream().readAllBytes());
|
||||||
|
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git timed out: git " + String.join(" ", args));
|
||||||
|
assertEquals(0, p.exitValue(), "git " + String.join(" ", args) + " failed:\n" + out);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Path initRepo(Path dir) throws Exception {
|
||||||
|
Files.createDirectories(dir);
|
||||||
|
gitQuiet(dir, "init", "-q", "-b", "main");
|
||||||
|
gitQuiet(dir, "config", "user.email", "test@example.invalid");
|
||||||
|
gitQuiet(dir, "config", "user.name", "Test");
|
||||||
|
Files.writeString(dir.resolve("README.md"), "seed\n");
|
||||||
|
gitQuiet(dir, "add", "README.md");
|
||||||
|
gitQuiet(dir, "commit", "-q", "-m", "seed");
|
||||||
|
return dir;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #248's first measured mutation: replacing {@code CompletionResolver}'s 8th constructor
|
||||||
|
* argument with the inert {@code _ -> null} compiles clean and leaves every existing test green
|
||||||
|
* — it silently drops fleetd #241's fallback-report location. This drives the real assembled
|
||||||
|
* resolver through a member echoing its own injected brief back (no {@code fleet_reply}), which
|
||||||
|
* resolves via {@code noReportMessage(target)}, and proves the real member's {@code branch} —
|
||||||
|
* only obtainable via {@code Fleetd.worktreeBranchLookup(sessions::roster)} reading the real,
|
||||||
|
* worktree-provisioned {@link MemberSession} — appears in the reported text.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void assembledResolverReportsTheMembersWorktreeAndBranchInAFallbackReport(@TempDir Path dir) throws Exception {
|
||||||
|
Path repo = initRepo(dir.resolve("repo"));
|
||||||
|
FleetConfig cfg = writeConfig(dir, """
|
||||||
|
wtprofile:
|
||||||
|
baseUrl: http://wthost.local:8000
|
||||||
|
model: sonnet
|
||||||
|
""", "wthost.local", "worktreeRoot: " + dir.resolve("wts"));
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
ControllableResourcePorts ports = new ControllableResourcePorts(new FakeHerdr());
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
try {
|
||||||
|
MemberSession session = runtime.sessions().acquire("wtprofile", repo.toString(), repo.toString(),
|
||||||
|
null, new WorktreeRequest("fleetd-612-b1", null));
|
||||||
|
String target = session.terminalId();
|
||||||
|
String branch = session.branch();
|
||||||
|
assertTrue(branch != null && branch.startsWith("worker/"),
|
||||||
|
"sanity: a worktree-provisioned session must carry a real branch, got: " + branch);
|
||||||
|
|
||||||
|
CompletionResolver completion = runtime.completion();
|
||||||
|
CompletableFuture<Rendezvous.Resolution> waiter = new CompletableFuture<>();
|
||||||
|
String echoedBrief = "z".repeat(450); // >= CompletionResolver.ECHO_MIN_CHARS normalised chars
|
||||||
|
|
||||||
|
ports.herdr.readText("idle, nothing yet");
|
||||||
|
completion.onDelivered(target, new TurnToken(target, waiter, echoedBrief));
|
||||||
|
|
||||||
|
ports.herdr.readText(echoedBrief); // the pane just echoes the injected brief back — no real report
|
||||||
|
ports.advanceSeconds(3); // clear CompletionResolver.MIN_TURN_NANOS (2s) without a real sleep
|
||||||
|
completion.resolveBeforePostAction(target);
|
||||||
|
|
||||||
|
Rendezvous.Resolution resolution = waiter.getNow(null);
|
||||||
|
assertTrue(resolution != null, "the waiter must have resolved synchronously");
|
||||||
|
assertEquals(Rendezvous.Kind.COMPLETION, resolution.kind());
|
||||||
|
assertTrue(resolution.text().contains(CompletionResolver.NO_REPORT_PREFIX),
|
||||||
|
"sanity: must have gone down the noReportMessage sub-path: " + resolution.text());
|
||||||
|
assertTrue(resolution.text().contains("branch=" + branch),
|
||||||
|
"the assembled resolver must report the member's real branch (fleetd #241 via "
|
||||||
|
+ "fleetd #248's worktreeBranchLookup wiring); got: " + resolution.text());
|
||||||
|
} finally {
|
||||||
|
if (ports.shutdownHook != null) ports.shutdownHook.run();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #248's second measured mutation, and fleetd#201 Unit 5's own gap: replacing {@code
|
||||||
|
* backendErrorPatterns}/{@code backendErrorSink} with {@code BackendErrorPatternLookup.legacy()}
|
||||||
|
* / {@code BackendErrorSink.none()} compiles clean and leaves every existing behavioural test
|
||||||
|
* green.
|
||||||
|
*
|
||||||
|
* <p>Classification proof: this test's profile configures {@code errorPattern: "credential
|
||||||
|
* outage"} — text the built-in {@code (?i)\bAPI Error\s*:} fallback ({@code legacy()}'s only
|
||||||
|
* behaviour) never matches. So a real {@code Fleetd.backendErrorPatternLookup(...)} wiring
|
||||||
|
* classifies the send as {@code FAILED}; {@code legacy()} would fall through to the plain
|
||||||
|
* completion path instead ({@code Kind.COMPLETION}).
|
||||||
|
*
|
||||||
|
* <p>Cool-off proof: two distinct targets on the same profile/credential each classified as a
|
||||||
|
* backend error inside the 60s window must cool the credential off ({@link
|
||||||
|
* dev.ltms.fleet.placement.BackendOutagePolicy}, fleetd#201 Unit 5) — observable two ways: (1)
|
||||||
|
* the real {@code Fleetd.backendErrorSink(...)} marks each session {@code BACKEND_ERROR} (only
|
||||||
|
* the real sink calls {@code sessions.onBackendError}; {@code BackendErrorSink.none()} never
|
||||||
|
* does), and (2) a third explicit-profile spawn attempt is refused with a {@link
|
||||||
|
* PlacementException} naming the cool-off — only reachable because the real sink's {@code
|
||||||
|
* outagePolicy.record(...)} call actually ran.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void assembledResolverClassifiesAndCoolsOffOnAConfiguredBackendErrorPattern(@TempDir Path dir) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir, """
|
||||||
|
coolprofile:
|
||||||
|
baseUrl: http://coolhost.local:8000
|
||||||
|
model: sonnet
|
||||||
|
errorPattern: "credential outage"
|
||||||
|
""", "coolhost.local", null);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
ControllableResourcePorts ports = new ControllableResourcePorts(new FakeHerdr());
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
try {
|
||||||
|
MemberSession session1 = runtime.sessions().acquire("coolprofile", null, dir.toString(), null);
|
||||||
|
MemberSession session2 = runtime.sessions().acquire("coolprofile", null, dir.toString(), null);
|
||||||
|
String target1 = session1.terminalId();
|
||||||
|
String target2 = session2.terminalId();
|
||||||
|
assertTrue(!target1.equals(target2), "sanity: the two spawns must be distinct targets");
|
||||||
|
|
||||||
|
CompletionResolver completion = runtime.completion();
|
||||||
|
|
||||||
|
CompletableFuture<Rendezvous.Resolution> waiter1 = new CompletableFuture<>();
|
||||||
|
ports.herdr.readText("idle 1");
|
||||||
|
completion.onDelivered(target1, new TurnToken(target1, waiter1, null));
|
||||||
|
ports.herdr.readText("credential outage: upstream 503");
|
||||||
|
ports.advanceSeconds(3);
|
||||||
|
completion.resolveBeforePostAction(target1);
|
||||||
|
Rendezvous.Resolution resolution1 = waiter1.getNow(null);
|
||||||
|
assertTrue(resolution1 != null, "target1's waiter must have resolved synchronously");
|
||||||
|
assertEquals(Rendezvous.Kind.FAILED, resolution1.kind(),
|
||||||
|
"a configured errorPattern the built-in fallback never matches must classify as "
|
||||||
|
+ "a backend error, not a plain completion; got: " + resolution1);
|
||||||
|
assertTrue(resolution1.text().contains("credential outage: upstream 503"), resolution1.text());
|
||||||
|
|
||||||
|
CompletableFuture<Rendezvous.Resolution> waiter2 = new CompletableFuture<>();
|
||||||
|
ports.herdr.readText("idle 2");
|
||||||
|
completion.onDelivered(target2, new TurnToken(target2, waiter2, null));
|
||||||
|
ports.herdr.readText("credential outage: upstream 503 again");
|
||||||
|
ports.advanceSeconds(3);
|
||||||
|
completion.resolveBeforePostAction(target2);
|
||||||
|
Rendezvous.Resolution resolution2 = waiter2.getNow(null);
|
||||||
|
assertTrue(resolution2 != null, "target2's waiter must have resolved synchronously");
|
||||||
|
assertEquals(Rendezvous.Kind.FAILED, resolution2.kind());
|
||||||
|
|
||||||
|
List<MemberSession> roster = runtime.sessions().roster();
|
||||||
|
assertTrue(roster.stream().anyMatch(s -> target1.equals(s.terminalId())
|
||||||
|
&& s.state() == MemberSession.State.BACKEND_ERROR),
|
||||||
|
"the real backendErrorSink must have transitioned target1 to BACKEND_ERROR: " + roster);
|
||||||
|
assertTrue(roster.stream().anyMatch(s -> target2.equals(s.terminalId())
|
||||||
|
&& s.state() == MemberSession.State.BACKEND_ERROR),
|
||||||
|
"the real backendErrorSink must have transitioned target2 to BACKEND_ERROR: " + roster);
|
||||||
|
|
||||||
|
PlacementException coolOff = assertThrows(PlacementException.class,
|
||||||
|
() -> runtime.sessions().acquire("coolprofile", null, dir.toString(), null),
|
||||||
|
"two distinct targets classified within the 60s window must cool the credential "
|
||||||
|
+ "off (BackendOutagePolicy), refusing a third explicit-profile spawn");
|
||||||
|
assertTrue(coolOff.getMessage().contains("cooling off"), coolOff.getMessage());
|
||||||
|
} finally {
|
||||||
|
if (ports.shutdownHook != null) ports.shutdownHook.run();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,89 +0,0 @@
|
|||||||
package dev.ltms.fleet;
|
|
||||||
|
|
||||||
import java.nio.file.Files;
|
|
||||||
import java.nio.file.Path;
|
|
||||||
import org.junit.jupiter.api.DisplayName;
|
|
||||||
import org.junit.jupiter.api.Test;
|
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* fleetd #248: this is the test that was actually missing. {@code Fleetd.main} builds its {@code
|
|
||||||
* CompletionResolver} from an 8-argument constructor, and the ticket's own measurement proved two
|
|
||||||
* ways to silently unwire it — both compiled with 0 errors and left every existing test green:
|
|
||||||
*
|
|
||||||
* <ul>
|
|
||||||
* <li>replacing the worktree/branch argument (the 8th) with {@code _ -> null} — drops
|
|
||||||
* fleetd#241's fallback-report location entirely;</li>
|
|
||||||
* <li>replacing {@code backendErrorPatterns, backendErrorSink} (5th/6th) with {@code
|
|
||||||
* BackendErrorPatternLookup.legacy(), BackendErrorSink.none()} — drops fleetd#201 Unit 5's
|
|
||||||
* backend-error classification and cool-off entirely.</li>
|
|
||||||
* </ul>
|
|
||||||
*
|
|
||||||
* <p>Neither mutation could be caught by any test that constructs its own {@code
|
|
||||||
* CompletionResolver} (every test before this one did exactly that) or by a test of {@link
|
|
||||||
* Fleetd#worktreeBranchLookup}, {@link Fleetd#backendErrorPatternLookup}, or {@link
|
|
||||||
* Fleetd#backendErrorSink} in isolation (see {@code FleetdWorktreeBranchLookupTest}, {@code
|
|
||||||
* FleetdBackendErrorPatternLookupTest}, {@code FleetdBackendErrorSinkTest}) — those prove the
|
|
||||||
* factories work, never that {@code main} still calls them. This class is a plain source-text
|
|
||||||
* assertion on {@code Fleetd.java} — crude, but honest about what it checks, and it turns red the
|
|
||||||
* instant the wiring is dropped, mirroring the same fallback shape {@link
|
|
||||||
* FleetdFleetAppConstructionTest} already uses for a different constructor argument.
|
|
||||||
*
|
|
||||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
|
||||||
* CompletionResolver} and never runs {@code main}.
|
|
||||||
*/
|
|
||||||
class FleetdCompletionResolverWiringTest {
|
|
||||||
|
|
||||||
private static String fleetdSource() throws Exception {
|
|
||||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
@DisplayName("[SOURCE TEXT] CompletionResolver's construction call still names backendErrorPatterns and backendErrorSink")
|
|
||||||
void backendErrorArgumentsAreStillNamedAtTheCallSite() throws Exception {
|
|
||||||
String source = fleetdSource();
|
|
||||||
assertTrue(source.contains(
|
|
||||||
"exhaustionSink, backendErrorPatterns, backendErrorSink, System::nanoTime,"),
|
|
||||||
"CompletionResolver's construction call must still pass backendErrorPatterns and "
|
|
||||||
+ "backendErrorSink as its 5th/6th arguments. Replacing them with "
|
|
||||||
+ "BackendErrorPatternLookup.legacy()/BackendErrorSink.none() (fleetd #248's measured "
|
|
||||||
+ "mutation) compiles with 0 errors and leaves every behavioural test green — this "
|
|
||||||
+ "source check is what must go red instead.");
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
@DisplayName("[SOURCE TEXT] CompletionResolver's construction call still passes worktreeBranchLookup(sessions::roster)")
|
|
||||||
void worktreeBranchLookupIsStillPassedAtTheCallSite() throws Exception {
|
|
||||||
String source = fleetdSource();
|
|
||||||
assertTrue(source.contains("worktreeBranchLookup(sessions::roster)"),
|
|
||||||
"CompletionResolver's construction call must still pass worktreeBranchLookup(sessions::roster) "
|
|
||||||
+ "as its 8th (last) argument. Replacing it with the inert `_ -> null` (fleetd #248's "
|
|
||||||
+ "other measured mutation) compiles with 0 errors and leaves every behavioural test "
|
|
||||||
+ "green — this source check is what must go red instead.");
|
|
||||||
assertFalse(source.contains("System::nanoTime,\n _ -> null"),
|
|
||||||
"the worktree/branch argument must never regress to the inert `_ -> null` literal");
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
@DisplayName("[SOURCE TEXT] backendErrorPatterns is assigned from the extracted backendErrorPatternLookup(...) factory")
|
|
||||||
void backendErrorPatternsComesFromTheFactory() throws Exception {
|
|
||||||
String source = fleetdSource();
|
|
||||||
assertTrue(source.contains(
|
|
||||||
"BackendErrorPatternLookup backendErrorPatterns = backendErrorPatternLookup(sessions::roster,"),
|
|
||||||
"backendErrorPatterns must be assigned from Fleetd.backendErrorPatternLookup(...), not an "
|
|
||||||
+ "inline lambda that a source check on the CompletionResolver call alone cannot see "
|
|
||||||
+ "through");
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
@DisplayName("[SOURCE TEXT] backendErrorSink is assigned from the extracted backendErrorSink(...) factory")
|
|
||||||
void backendErrorSinkComesFromTheFactory() throws Exception {
|
|
||||||
String source = fleetdSource();
|
|
||||||
assertTrue(source.contains(
|
|
||||||
"BackendErrorSink backendErrorSink = backendErrorSink(sessions, () -> config.get().profiles(),"),
|
|
||||||
"backendErrorSink must be assigned from Fleetd.backendErrorSink(...), not an inline lambda "
|
|
||||||
+ "that a source check on the CompletionResolver call alone cannot see through");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -24,8 +24,8 @@ import static org.junit.jupiter.api.Assertions.assertTrue;
|
|||||||
* {@code ConfigRefTest} and {@code FleetdConfigRefCharterToolSurfaceWiringTest} case — because
|
* {@code ConfigRefTest} and {@code FleetdConfigRefCharterToolSurfaceWiringTest} case — because
|
||||||
* neither of those tests constructs its {@code ConfigRef} through {@code main}; both build their own
|
* neither of those tests constructs its {@code ConfigRef} through {@code main}; both build their own
|
||||||
* instance directly, wired with the check by hand. That silent regression is exactly the shape
|
* instance directly, wired with the check by hand. That silent regression is exactly the shape
|
||||||
* {@link FleetdBackendQuarantineWiringTest}, {@link FleetdLeadSeatWiringTest} and {@link
|
* {@link FleetdBackendQuarantineAssemblyTest}, {@link FleetdLeadSeatAssemblyTest} and {@link
|
||||||
* FleetdCompletionResolverWiringTest} already guard against for their own constructor arguments —
|
* FleetdCompletionResolverAssemblyTest} already guard against for their own constructor arguments —
|
||||||
* this class is the same class of gap for fleetd #474's {@code extraValidation} argument, following
|
* this class is the same class of gap for fleetd #474's {@code extraValidation} argument, following
|
||||||
* their approach.
|
* their approach.
|
||||||
*
|
*
|
||||||
|
|||||||
@@ -1,31 +0,0 @@
|
|||||||
package dev.ltms.fleet;
|
|
||||||
|
|
||||||
import java.nio.file.Files;
|
|
||||||
import java.nio.file.Path;
|
|
||||||
import org.junit.jupiter.api.Test;
|
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* CB-185: {@code ConnectionIdentity} must resolve a caller's pane on EITHER herdr daemon (a
|
|
||||||
* lead's MCP connection resolves against the lead daemon; a member's against the member daemon).
|
|
||||||
* Pinning {@code PaneLocator} to {@code memberHerdr} alone — the bug this guards against — leaves
|
|
||||||
* every lead's own connection unresolvable ({@code callerTerminal == null}) the moment
|
|
||||||
* {@code memberHerdrSocket} names a second daemon, which breaks {@code fleet_reply}/{@code
|
|
||||||
* fleet_ask} and {@code fleet_whoami} for a lead. A unit test on {@link
|
|
||||||
* dev.ltms.fleet.herdr.PaneLocator} alone (see {@code PaneLocatorTest}) proves the class CAN
|
|
||||||
* search two clients, but not that {@code Fleetd.main} actually wires it that way — hence this
|
|
||||||
* source-level assertion, the same technique {@code FleetdHerdrControlConstructionTest} uses.
|
|
||||||
*/
|
|
||||||
class FleetdConnectionIdentityConstructionTest {
|
|
||||||
@Test
|
|
||||||
void connectionIdentitySearchesBothDaemonsNotJustTheMemberOne() throws Exception {
|
|
||||||
String source = Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
|
||||||
assertFalse(source.contains("new PaneLocator(memberHerdr)"),
|
|
||||||
"PaneLocator must not be pinned to the member daemon alone — a lead's own "
|
|
||||||
+ "connection resolves against the LEAD daemon and would never be found");
|
|
||||||
assertTrue(source.contains("new PaneLocator(herdr, memberHerdr)"),
|
|
||||||
"PaneLocator must search the lead daemon first, then the member daemon");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,218 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.inject.CompletionResolver;
|
||||||
|
import dev.ltms.fleet.msg.Rendezvous;
|
||||||
|
import dev.ltms.fleet.msg.TurnToken;
|
||||||
|
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||||
|
import dev.ltms.fleet.session.MemberSession;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.OptionalLong;
|
||||||
|
import java.util.concurrent.CompletableFuture;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 step 4, ranks 1 and 2 (publish side) — {@link FleetdAssembly} lines
|
||||||
|
* {@code liveExhaustedPatterns}/{@code exhaustedPatterns} (CB-578 stage A, the ticket's own "worst
|
||||||
|
* consequence in the whole sweep": a genuine usage-limit refusal handed back to a waiting caller
|
||||||
|
* AS REAL COMPLETED WORK) and {@code Fleetd.publishExhaustionSink(...)} (CB-578 stage B: the
|
||||||
|
* credential that hit the limit is never quarantined). None of these three lines is driven by an
|
||||||
|
* existing test through the real assembly: {@code FleetdExhaustedPatternLookupWiringTest} and
|
||||||
|
* {@code FleetdLiveExhaustedPatternsWiringTest} (fleetd #589) call {@code Fleetd.liveExhaustedPatterns}
|
||||||
|
* / {@code Fleetd.exhaustedPatternLookup} directly as factories, never through {@link
|
||||||
|
* FleetdAssembly#assembleAndStart} — they prove the FACTORY classifies correctly, never that THIS
|
||||||
|
* call site is the one that actually got wired into the running {@link CompletionResolver}. {@link
|
||||||
|
* FleetdBackendQuarantineAssemblyTest} drives {@code BackendQuarantine.withEscalation(...)}
|
||||||
|
* directly, a different call site from {@code publishExhaustionSink} here.
|
||||||
|
*
|
||||||
|
* <p>This test drives the REAL assembled {@link CompletionResolver} ({@link
|
||||||
|
* FleetdRuntime#completion()}) with a profile carrying a configured {@code exhaustedPattern},
|
||||||
|
* through a pane scrape that matches it, and asserts both halves of the production consequence:
|
||||||
|
* (1) the resolution is {@link Rendezvous.Kind#BACKEND_EXHAUSTED}, never a plain completion handed
|
||||||
|
* back as real work, and (2) the profile's credential is actually quarantined afterward, through
|
||||||
|
* the REAL {@link BackendQuarantine} the same assembly built ({@link
|
||||||
|
* FleetdRuntime#mcp()}{@code .quarantineSource().quarantine()}) — never a copy.
|
||||||
|
*
|
||||||
|
* <p>Same {@code ControllableResourcePorts} shape as {@code FleetdCompletionResolverAssemblyTest}:
|
||||||
|
* a fake, advanceable {@code nanoClock} so {@code CompletionResolver.MIN_TURN_NANOS} clears without
|
||||||
|
* a real sleep, and {@link FakeHerdr#readText} to drive the pane scrape.
|
||||||
|
*/
|
||||||
|
class FleetdExhaustedPatternAssemblyTest {
|
||||||
|
|
||||||
|
private static final class ControllableResourcePorts implements ResourcePorts {
|
||||||
|
|
||||||
|
final FakeHerdr herdr;
|
||||||
|
final AtomicLong nowNanos = new AtomicLong(1_000_000_000L); // arbitrary non-zero start
|
||||||
|
Runnable shutdownHook;
|
||||||
|
|
||||||
|
ControllableResourcePorts(FakeHerdr herdr) {
|
||||||
|
this.herdr = herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
void advanceSeconds(long seconds) {
|
||||||
|
nowNanos.addAndGet(TimeUnit.SECONDS.toNanos(seconds));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException("replyInboxOpener must not be called — no broker: block");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException("leadMailboxOpener must not be called — no coordinator: block");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return nowNanos::get;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return nowNanos::get;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
this.shutdownHook = hook;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
// Deliberately never bind a real port.
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir, int cooldownSeconds) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
quarantineCooldownSeconds: %d
|
||||||
|
profiles:
|
||||||
|
exhaustprofile:
|
||||||
|
baseUrl: http://exhausthost.local:8000
|
||||||
|
model: sonnet
|
||||||
|
exhaustedPattern: "usage limit reached"
|
||||||
|
guard:
|
||||||
|
offSubscriptionHosts:
|
||||||
|
- exhausthost.local
|
||||||
|
""".formatted(cooldownSeconds));
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #589's own description of this gap ({@code Fleetd#exhaustedPatternLookup}'s javadoc):
|
||||||
|
* "the worst consequence in the whole #589 sweep" — a genuine usage-limit refusal stops being
|
||||||
|
* classified as {@code BACKEND_EXHAUSTED} and is handed back to a waiting {@code fleet_send} as
|
||||||
|
* if it were real completed work. Pins {@code FleetdAssembly}'s {@code liveExhaustedPatterns}
|
||||||
|
* AND {@code exhaustedPatterns} lines (rank 1) together with {@code publishExhaustionSink}
|
||||||
|
* (rank 2, the non-OpenCode half) in one flow: classify, then quarantine.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
@DisplayName("[BEHAVIOURAL] a scrape matching the profile's exhaustedPattern resolves "
|
||||||
|
+ "BACKEND_EXHAUSTED (never a plain completion) and quarantines the credential")
|
||||||
|
void assembledResolverClassifiesExhaustionAndQuarantinesTheCredential(@TempDir Path dir) throws Exception {
|
||||||
|
int cooldownSeconds = 120;
|
||||||
|
FleetConfig cfg = writeConfig(dir, cooldownSeconds);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
ControllableResourcePorts ports = new ControllableResourcePorts(new FakeHerdr());
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
try {
|
||||||
|
MemberSession session = runtime.sessions().acquire("exhaustprofile", null, dir.toString(), null);
|
||||||
|
String target = session.terminalId();
|
||||||
|
|
||||||
|
CompletionResolver completion = runtime.completion();
|
||||||
|
CompletableFuture<Rendezvous.Resolution> waiter = new CompletableFuture<>();
|
||||||
|
|
||||||
|
ports.herdr.readText("idle, nothing yet");
|
||||||
|
completion.onDelivered(target, new TurnToken(target, waiter, null));
|
||||||
|
// The matched text must START the pane line (CompletionResolver.startsWithExhaustion) —
|
||||||
|
// no preceding sentence — for the quarantine side-effect to fire, same as production.
|
||||||
|
ports.herdr.readText("usage limit reached: try again in a few hours");
|
||||||
|
ports.advanceSeconds(3); // clear CompletionResolver.MIN_TURN_NANOS (2s), no real sleep
|
||||||
|
completion.resolveBeforePostAction(target);
|
||||||
|
|
||||||
|
// CONTROL: the waiter must have resolved synchronously at all — if the assembled
|
||||||
|
// CompletionResolver were never actually driven (e.g. a wiring break upstream silently
|
||||||
|
// left the resolver unreachable), this fails loudly before the real assertions below
|
||||||
|
// ever run, rather than passing on an untouched waiter.
|
||||||
|
Rendezvous.Resolution resolution = waiter.getNow(null);
|
||||||
|
assertTrue(resolution != null, "CONTROL: the waiter must have resolved synchronously — "
|
||||||
|
+ "if this is null, the assembled resolver was never actually exercised");
|
||||||
|
|
||||||
|
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, resolution.kind(),
|
||||||
|
"a scrape matching the profile's configured exhaustedPattern must classify as "
|
||||||
|
+ "BACKEND_EXHAUSTED, not a plain completion handed back as real work — "
|
||||||
|
+ "replacing FleetdAssembly's liveExhaustedPatterns/exhaustedPatterns "
|
||||||
|
+ "lines with their inert forms (Map.of() / target -> null) must fail "
|
||||||
|
+ "this assertion; got: " + resolution);
|
||||||
|
assertTrue(resolution.text().contains("usage limit reached"), resolution.text());
|
||||||
|
|
||||||
|
BackendQuarantine quarantine = runtime.mcp().quarantineSource().quarantine();
|
||||||
|
assertTrue(quarantine.isQuarantined("exhaustprofile"),
|
||||||
|
"the real publishExhaustionSink-built sink must have quarantined the profile's "
|
||||||
|
+ "credential (effectiveCredentialId() == the profile name here, no "
|
||||||
|
+ "credentialId configured) — replacing FleetdAssembly's "
|
||||||
|
+ "publishExhaustionSink call site with a hardcoded ExhaustionSink.none() "
|
||||||
|
+ "must fail this assertion, since nothing would ever call "
|
||||||
|
+ "quarantine.quarantine(...)");
|
||||||
|
OptionalLong remaining = quarantine.remainingSeconds("exhaustprofile");
|
||||||
|
assertTrue(remaining.isPresent() && remaining.getAsLong() > 0
|
||||||
|
&& remaining.getAsLong() <= cooldownSeconds,
|
||||||
|
"a fresh quarantine must block for at most the configured base cooldown: " + remaining);
|
||||||
|
} finally {
|
||||||
|
if (ports.shutdownHook != null) ports.shutdownHook.run();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,86 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.inject.ExhaustedPatternLookup;
|
||||||
|
import dev.ltms.fleet.inject.LiveExhaustedPatterns;
|
||||||
|
import dev.ltms.fleet.peer.MemberRole;
|
||||||
|
import dev.ltms.fleet.session.MemberSession;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.regex.Pattern;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #589 Group 1: {@link Fleetd#exhaustedPatternLookup} is the factory that replaced {@code
|
||||||
|
* main}'s inline lambda — resolve a herdr {@code target} to its session's profile, then to that
|
||||||
|
* profile's live {@link LiveExhaustedPatterns#patternFor}. Same shape as {@link
|
||||||
|
* Fleetd#worktreeBranchLookup} (which {@code FleetdWorktreeBranchLookupTest} pins the same way).
|
||||||
|
*
|
||||||
|
* <p>Before this ticket the lambda was built inline in {@code main} and untestable: replacing it
|
||||||
|
* with {@code target -> null} — the exact shape of {@link ExhaustedPatternLookup#none()} — compiled
|
||||||
|
* with 0 errors and left every existing test green. Per the ticket, this is the worst consequence
|
||||||
|
* in the whole #589 sweep: a genuine usage-limit refusal would stop being classified as {@code
|
||||||
|
* BACKEND_EXHAUSTED} and would be handed back to a waiting {@code fleet_send} as if it were real
|
||||||
|
* completed work.
|
||||||
|
*/
|
||||||
|
class FleetdExhaustedPatternLookupWiringTest {
|
||||||
|
|
||||||
|
private static final String YAML = """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
herdrSocket: ~/.config/herdr/herdr.sock
|
||||||
|
profiles:
|
||||||
|
terra:
|
||||||
|
baseUrl: http://gx00.gw:8000
|
||||||
|
model: claude-opus-5
|
||||||
|
exhaustedPattern: "usage limit"
|
||||||
|
""";
|
||||||
|
|
||||||
|
private static MemberSession session(String terminal, String profile) {
|
||||||
|
return new MemberSession("pane-" + terminal, terminal, profile, MemberRole.DEV,
|
||||||
|
"/cwd", null, 0L, 0L, 0, MemberSession.State.READY, null, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static LiveExhaustedPatterns liveExhaustedPatterns(Path dir) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, YAML);
|
||||||
|
FleetConfig cfg = FleetConfig.load(file);
|
||||||
|
return new LiveExhaustedPatterns(() -> ConfigRef.fixed(cfg).get().profiles());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a known target resolves through its session's profile to that profile's live pattern")
|
||||||
|
void knownTargetResolvesThroughItsProfile(@TempDir Path dir) throws Exception {
|
||||||
|
LiveExhaustedPatterns patterns = liveExhaustedPatterns(dir);
|
||||||
|
ExhaustedPatternLookup lookup = Fleetd.exhaustedPatternLookup(
|
||||||
|
() -> List.of(session("term1", "terra")), patterns);
|
||||||
|
|
||||||
|
Pattern resolved = lookup.patternFor("term1");
|
||||||
|
|
||||||
|
assertNotNull(resolved,
|
||||||
|
"the lookup must resolve term1 -> profile 'terra' -> LiveExhaustedPatterns.patternFor("
|
||||||
|
+ "'terra') — replacing the lambda body with 'target -> null' at the "
|
||||||
|
+ "Fleetd.exhaustedPatternLookup call site must fail this assertion");
|
||||||
|
assertTrue(resolved.matcher("the usage limit has been reached").find());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("an unknown target resolves to null, not a thrown exception")
|
||||||
|
void unknownTargetResolvesToNull(@TempDir Path dir) throws Exception {
|
||||||
|
LiveExhaustedPatterns patterns = liveExhaustedPatterns(dir);
|
||||||
|
ExhaustedPatternLookup lookup = Fleetd.exhaustedPatternLookup(
|
||||||
|
() -> List.of(session("term1", "terra")), patterns);
|
||||||
|
|
||||||
|
assertNull(lookup.patternFor("term_stranger"));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,62 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.util.concurrent.atomic.AtomicBoolean;
|
||||||
|
import java.util.concurrent.atomic.AtomicReference;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #589 Group 1: {@link Fleetd#forwardingExhaustionSink} is the factory that replaced
|
||||||
|
* {@code main}'s inline {@code ExhaustionSink.forwardingTo(exhaustionSinkRef::get)} (fleetd #175's
|
||||||
|
* construction-order break: the adapters need a sink before {@code sessions} exists to build the
|
||||||
|
* real one). Before this ticket that call site was untestable wiring: replacing the supplier
|
||||||
|
* argument with a hardcoded {@code () -> ExhaustionSink.none()} compiled with 0 errors and left
|
||||||
|
* every existing test green, because no test builds the object {@code main} actually wires and
|
||||||
|
* then mutates the reference afterward — every existing {@code ExhaustionSink.forwardingTo} caller
|
||||||
|
* in this codebase reads and writes the SAME reference within one test, so a hardcoded-none supplier
|
||||||
|
* and a correctly-forwarding one are indistinguishable to them.
|
||||||
|
*
|
||||||
|
* <p>This test builds the reference, builds the forwarder from it, and only THEN repoints the
|
||||||
|
* reference at a spy sink — the discriminating order fleetd #175's whole design depends on
|
||||||
|
* ({@code exhaustionSinkRef} starts at {@code none()} and is repointed once {@code sessions}
|
||||||
|
* exists). A forwarder that captured a fixed target at construction time (the inert form) can never
|
||||||
|
* see that later repoint.
|
||||||
|
*/
|
||||||
|
class FleetdExhaustionSinkForwardingWiringTest {
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("the forwarder reads the reference live: repointing it AFTER construction is honoured")
|
||||||
|
void forwarderReadsTheReferenceLiveNotAFixedTargetCapturedAtConstruction() {
|
||||||
|
AtomicReference<ExhaustionSink> exhaustionSinkRef = new AtomicReference<>(ExhaustionSink.none());
|
||||||
|
ExhaustionSink forwarder = Fleetd.forwardingExhaustionSink(exhaustionSinkRef);
|
||||||
|
|
||||||
|
AtomicBoolean spyCalled = new AtomicBoolean(false);
|
||||||
|
exhaustionSinkRef.set((target, reason, profile) -> spyCalled.set(true));
|
||||||
|
|
||||||
|
forwarder.onExhausted("term_x", "usage limit reached", "terra");
|
||||||
|
|
||||||
|
assertTrue(spyCalled.get(),
|
||||||
|
"forwardingExhaustionSink must delegate to whatever exhaustionSinkRef currently "
|
||||||
|
+ "holds — hardcoding the supplier to () -> ExhaustionSink.none() at the "
|
||||||
|
+ "Fleetd.forwardingExhaustionSink call site must fail this assertion, "
|
||||||
|
+ "since the spy set into the reference after construction would never run");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("before any repoint, the forwarder is inert — it starts at none(), not a crash")
|
||||||
|
void beforeAnyRepointTheForwarderIsInert() {
|
||||||
|
AtomicReference<ExhaustionSink> exhaustionSinkRef = new AtomicReference<>(ExhaustionSink.none());
|
||||||
|
ExhaustionSink forwarder = Fleetd.forwardingExhaustionSink(exhaustionSinkRef);
|
||||||
|
|
||||||
|
AtomicBoolean spyCalled = new AtomicBoolean(false);
|
||||||
|
forwarder.onExhausted("term_x", "usage limit reached", "terra");
|
||||||
|
|
||||||
|
assertFalse(spyCalled.get(), "nothing was ever wired to be called here — this only pins "
|
||||||
|
+ "that the factory does not throw before a real sink is published");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,110 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
|
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||||
|
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||||
|
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||||
|
import dev.ltms.fleet.session.SessionManager;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.HashMap;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Set;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.concurrent.atomic.AtomicReference;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #589 Group 1: {@link Fleetd#publishExhaustionSink} is the factory that replaced {@code
|
||||||
|
* main}'s previously untested two-statement sequence — build the real {@link
|
||||||
|
* Fleetd#exhaustionSink}, then {@code exhaustionSinkRef.set(exhaustionSink)}. {@link
|
||||||
|
* Fleetd#exhaustionSink} itself is already pinned by {@code FleetdExhaustionSinkWarningTest} (its
|
||||||
|
* log text) — what was NEVER pinned is the {@code .set(...)} call: {@code main} could replace it
|
||||||
|
* with {@code exhaustionSinkRef.set(ExhaustionSink.none())} and compile with 0 errors, leaving
|
||||||
|
* every existing test green, because {@link Fleetd#exhaustionSink}'s own tests build and call the
|
||||||
|
* sink directly, never through the reference {@code main} publishes it into.
|
||||||
|
*
|
||||||
|
* <p>This test proves the PUBLISHED reference — not a freshly rebuilt sink — is the one that
|
||||||
|
* actually quarantines a credential, by reading {@link BackendQuarantine#isQuarantined} after
|
||||||
|
* calling {@code exhaustionSinkRef.get().onExhausted(...)}, the same object {@link
|
||||||
|
* Fleetd#forwardingExhaustionSink} forwards to in production.
|
||||||
|
*/
|
||||||
|
class FleetdExhaustionSinkPublishWiringTest {
|
||||||
|
|
||||||
|
private static final String YAML = """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
herdrSocket: ~/.config/herdr/herdr.sock
|
||||||
|
profiles:
|
||||||
|
terra:
|
||||||
|
baseUrl: http://gx00.gw:8000
|
||||||
|
model: claude-opus-5
|
||||||
|
guard:
|
||||||
|
offSubscriptionHosts:
|
||||||
|
- gx00.gw
|
||||||
|
""";
|
||||||
|
|
||||||
|
private static SessionManager emptyRosterSessions() {
|
||||||
|
FakeHerdr h = new FakeHerdr();
|
||||||
|
FleetConfig.Profile dummy = new FleetConfig.Profile(
|
||||||
|
"dummy", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN", null,
|
||||||
|
"tab", "fleetd-workers", "worker: {profile} #{n}", null, null, null);
|
||||||
|
ClaudeCodeLauncher launcher = new ClaudeCodeLauncher(new AgentControl(h), new WorkspaceControl(h),
|
||||||
|
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(dummy.profile(), dummy), dummy.profile(), _ -> "tok");
|
||||||
|
// Never acquires a session — publishExhaustionSink's built sink resolves target -> profile
|
||||||
|
// via the profileHint fallback (fleetd #234), exactly like OpenCodeLauncher's real call
|
||||||
|
// site does, so this never needs a populated roster.
|
||||||
|
return new SessionManager(launcher);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("the published reference actually quarantines — not a rebuilt-but-never-set sink")
|
||||||
|
void publishedReferenceActuallyQuarantines(@TempDir Path dir) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, YAML);
|
||||||
|
FleetConfig cfg = FleetConfig.load(file);
|
||||||
|
ConfigRef config = ConfigRef.fixed(cfg);
|
||||||
|
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||||
|
Map<String, String> reasonByCredential = new HashMap<>();
|
||||||
|
AtomicReference<ExhaustionSink> exhaustionSinkRef = new AtomicReference<>(ExhaustionSink.none());
|
||||||
|
|
||||||
|
Fleetd.publishExhaustionSink(exhaustionSinkRef, emptyRosterSessions(), config, quarantine,
|
||||||
|
reasonByCredential, cfg);
|
||||||
|
exhaustionSinkRef.get().onExhausted("term_x", "The usage limit has been reached", "terra");
|
||||||
|
|
||||||
|
assertTrue(quarantine.isQuarantined("terra"),
|
||||||
|
"publishExhaustionSink must repoint exhaustionSinkRef at the REAL sink — "
|
||||||
|
+ "replacing the .set(...) call with exhaustionSinkRef.set(ExhaustionSink.none()) "
|
||||||
|
+ "at the Fleetd.publishExhaustionSink call site must fail this assertion, "
|
||||||
|
+ "since none()'s onExhausted does nothing");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("before publishing, the reference is still inert — no quarantine, no crash")
|
||||||
|
void beforePublishingTheReferenceIsStillInert(@TempDir Path dir) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, YAML);
|
||||||
|
FleetConfig cfg = FleetConfig.load(file);
|
||||||
|
ConfigRef config = ConfigRef.fixed(cfg);
|
||||||
|
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||||
|
AtomicReference<ExhaustionSink> exhaustionSinkRef = new AtomicReference<>(ExhaustionSink.none());
|
||||||
|
|
||||||
|
exhaustionSinkRef.get().onExhausted("term_x", "The usage limit has been reached", "terra");
|
||||||
|
|
||||||
|
assertFalse(quarantine.isQuarantined("terra"),
|
||||||
|
"nothing was published yet — this only pins the starting state the other test's "
|
||||||
|
+ "assertion actually distinguishes from");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,29 +0,0 @@
|
|||||||
package dev.ltms.fleet;
|
|
||||||
|
|
||||||
import java.nio.file.Files;
|
|
||||||
import java.nio.file.Path;
|
|
||||||
import org.junit.jupiter.api.Test;
|
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* CB-185: {@code FleetApp} must be constructed with BOTH herdr clients (the lead's and the
|
|
||||||
* member's), never the raw lead-only {@code herdr}. Passing only {@code herdr} — the bug this
|
|
||||||
* guards against — makes {@code GET /healthz} green while the member daemon is down (so every
|
|
||||||
* spawn fails invisibly) and silently drops every member workspace from {@code GET /sessions}.
|
|
||||||
* A behavioural test on {@code FleetApp} alone (see {@code FleetAppTwoDaemonTest}) proves the
|
|
||||||
* class merges/gates correctly when given two clients, but not that {@code Fleetd.main} actually
|
|
||||||
* passes it two — hence this source-level assertion, mirroring
|
|
||||||
* {@code FleetdHerdrControlConstructionTest}.
|
|
||||||
*/
|
|
||||||
class FleetdFleetAppConstructionTest {
|
|
||||||
@Test
|
|
||||||
void fleetAppIsConstructedWithBothHerdrDaemons() throws Exception {
|
|
||||||
String source = Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
|
||||||
assertFalse(source.contains("new FleetApp(herdr, workers,"),
|
|
||||||
"FleetApp must not be constructed with the lead-only herdr client");
|
|
||||||
assertTrue(source.contains("new FleetApp(herdr, memberHerdr, workers,"),
|
|
||||||
"FleetApp must be constructed with both the lead and the member herdr client");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,160 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.mcp.FleetMcp;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #426: {@code FleetHealthMonitor.coverage} had zero references anywhere in the test
|
||||||
|
* tree — not the method, not either output string, not the field it populates. {@code
|
||||||
|
* FleetHealthMonitorCoverageTest} (package {@code dev.ltms.fleet.health}) pins the three-branch
|
||||||
|
* method itself; that is the easy half.
|
||||||
|
*
|
||||||
|
* <p>The half that actually matters is this one: {@link Fleetd#healthCoverageSource} is the exact
|
||||||
|
* call site {@code Fleetd.main} wires into {@code FleetMcp}'s constructor, and it is what feeds
|
||||||
|
* {@code fleet_list}'s {@code healthCoverage} field (see {@code FleetMcp#listFleet}'s {@code
|
||||||
|
* result.put("healthCoverage", healthCoverage.value().get())}). Measured precedent on fleetd #423
|
||||||
|
* (for #415): swapping two arguments at a call site like this one — recreating #415's defect with
|
||||||
|
* the keys exchanged — compiled with 0 errors and ran the ENTIRE suite (1506 tests) green. A
|
||||||
|
* thoroughly-tested method proves nothing about whether the call site pairs its arguments correctly;
|
||||||
|
* only a test that drives the call site itself can catch that.
|
||||||
|
*
|
||||||
|
* <p><strong>Why this is not driven through a real {@code Fleetd.main} the way #407 drives its five
|
||||||
|
* reporters</strong> (keep the config invalid, assert on the log line emitted before {@code
|
||||||
|
* cfg.validateAll()} throws): all five of #407's reporters run in {@code main} before {@code
|
||||||
|
* validateAll()} (line ~171). {@link Fleetd#healthCoverageSource} is built during {@code FleetMcp}
|
||||||
|
* construction, which happens only after {@code UnixSocketHerdrClient.connect} has already opened a
|
||||||
|
* real herdr socket (line ~188) and after {@code sessions}/{@code workers} are constructed. Reaching
|
||||||
|
* this call site by actually running {@code main} would require a real socket connect — banned by
|
||||||
|
* this ticket's hard constraints — so #407's option 1 does not apply here. Instead {@link
|
||||||
|
* Fleetd#healthCoverageSource} is extracted to a directly-callable package-private factory, the same
|
||||||
|
* shape {@link Fleetd#capacitySource} and {@link Fleetd#quarantineSource} already use for the same
|
||||||
|
* reason (see {@code FleetdCapacitySourceWiringTest}, the direct precedent this test follows).
|
||||||
|
*
|
||||||
|
* <p>Uses a real {@link FleetConfig#load} + {@link ConfigRef} (no socket, no port bind, no spawn,
|
||||||
|
* nothing written outside {@code @TempDir}) so the fixture goes through the actual YAML parser and
|
||||||
|
* {@code FleetConfig.Health}/{@code Notifications} records, not a hand-built stand-in that could
|
||||||
|
* silently drift from what the parser actually produces.
|
||||||
|
*/
|
||||||
|
class FleetdHealthCoverageSourceWiringTest {
|
||||||
|
|
||||||
|
private static final String BASE = """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
herdrSocket: ~/.config/herdr/herdr.sock
|
||||||
|
""";
|
||||||
|
|
||||||
|
private static final String HEALTH_DISABLED_WITH_WEBHOOK = BASE + """
|
||||||
|
health:
|
||||||
|
enabled: false
|
||||||
|
notifications:
|
||||||
|
mode: webhook
|
||||||
|
""";
|
||||||
|
|
||||||
|
private static final String HEALTH_DETECTION_ONLY = BASE + """
|
||||||
|
health:
|
||||||
|
enabled: true
|
||||||
|
""";
|
||||||
|
|
||||||
|
private static final String HEALTH_FULL = BASE + """
|
||||||
|
health:
|
||||||
|
enabled: true
|
||||||
|
notifications:
|
||||||
|
mode: webhook
|
||||||
|
""";
|
||||||
|
|
||||||
|
private static final String HEALTH_ABSENT = BASE;
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("enabled: false reports off, even with a webhook configured")
|
||||||
|
void disabledHealthReportsOff(@TempDir Path dir) throws Exception {
|
||||||
|
FleetMcp.HealthCoverageSource source = sourceFor(dir, HEALTH_DISABLED_WITH_WEBHOOK);
|
||||||
|
|
||||||
|
assertEquals("off", source.value().get(),
|
||||||
|
"health.enabled: false must report 'off' regardless of notifications — flipping "
|
||||||
|
+ "the 'enabled' argument at the HealthCoverageSource call site would report "
|
||||||
|
+ "'full' here instead");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("enabled with no notifications reports detection-only")
|
||||||
|
void enabledWithoutNotificationsReportsDetectionOnly(@TempDir Path dir) throws Exception {
|
||||||
|
FleetMcp.HealthCoverageSource source = sourceFor(dir, HEALTH_DETECTION_ONLY);
|
||||||
|
|
||||||
|
assertEquals("detection-only", source.value().get());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("enabled with a webhook configured reports full")
|
||||||
|
void enabledWithNotificationsReportsFull(@TempDir Path dir) throws Exception {
|
||||||
|
FleetMcp.HealthCoverageSource source = sourceFor(dir, HEALTH_FULL);
|
||||||
|
|
||||||
|
assertEquals("full", source.value().get(),
|
||||||
|
"health.enabled: true with notifications.mode: webhook must report 'full' — "
|
||||||
|
+ "swapping 'full' and 'detection-only' at the call site, or breaking the "
|
||||||
|
+ "enabled/notificationConfigured argument pairing, would report "
|
||||||
|
+ "'detection-only' here instead");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("an absent health: block reports off")
|
||||||
|
void absentHealthBlockReportsOff(@TempDir Path dir) throws Exception {
|
||||||
|
FleetMcp.HealthCoverageSource source = sourceFor(dir, HEALTH_ABSENT);
|
||||||
|
|
||||||
|
assertEquals("off", source.value().get());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #426, the live-wiring half: {@code health.notifications} is a {@code
|
||||||
|
* ConfigRef.SPLIT_KEYS} entry, and {@link Fleetd#healthCoverageSource} reads {@code
|
||||||
|
* config.get().health()} live (not the frozen startup {@code cfg}) — exactly like {@link
|
||||||
|
* Fleetd#capacitySource}'s {@code maxLoad} ({@code FleetdCapacitySourceWiringTest}'s {@code
|
||||||
|
* reloadedMaxLoadStillChangesWhatFleetListReports}). A hot-reloaded notifications block must
|
||||||
|
* change what {@code fleet_list} reports without a restart; a fix that froze the whole source
|
||||||
|
* against the startup snapshot would silently break that and every other test above would stay
|
||||||
|
* green, since none of them reload.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
@DisplayName("a hot notifications reload still changes what fleet_list reports")
|
||||||
|
void reloadedNotificationsStillChangeWhatFleetListReports(@TempDir Path dir) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, HEALTH_DETECTION_ONLY);
|
||||||
|
FleetConfig cfg = FleetConfig.load(file);
|
||||||
|
ConfigRef config = new ConfigRef(file, cfg);
|
||||||
|
|
||||||
|
FleetMcp.HealthCoverageSource source = Fleetd.healthCoverageSource(config);
|
||||||
|
assertEquals("detection-only", source.value().get(),
|
||||||
|
"sanity: detection-only before any reload");
|
||||||
|
|
||||||
|
Files.writeString(file, HEALTH_FULL);
|
||||||
|
assertTrue(config.reload().applied());
|
||||||
|
// The live snapshot now carries the webhook — proves the reload really happened and this
|
||||||
|
// test is not accidentally passing because nothing changed.
|
||||||
|
assertTrue(config.get().health().notifications() != null
|
||||||
|
&& config.get().health().notifications().configured(),
|
||||||
|
"sanity: the reloaded config really carries a configured webhook");
|
||||||
|
|
||||||
|
assertEquals("full", source.value().get(),
|
||||||
|
"the SAME HealthCoverageSource instance must reflect a reloaded notifications "
|
||||||
|
+ "block without a restart — health.notifications is read live off "
|
||||||
|
+ "config.get(), exactly like capacitySource's maxLoad");
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetMcp.HealthCoverageSource sourceFor(Path dir, String yaml) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, yaml);
|
||||||
|
FleetConfig cfg = FleetConfig.load(file);
|
||||||
|
ConfigRef config = new ConfigRef(file, cfg);
|
||||||
|
return Fleetd.healthCoverageSource(config);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,80 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.inject.Injector;
|
||||||
|
import dev.ltms.fleet.msg.MessageService;
|
||||||
|
import dev.ltms.fleet.msg.Rendezvous;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.util.function.BiConsumer;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #589 Group 3, site {@code :591}. {@code Fleetd.main} wires {@link
|
||||||
|
* dev.ltms.fleet.health.FleetHealthMonitor}'s {@code failTarget} callback with {@code
|
||||||
|
* messages::abandon} — before this ticket that was an inline argument to {@code new
|
||||||
|
* FleetHealthMonitor(...)}. Measured: replacing it with a no-op {@code BiConsumer} at the call
|
||||||
|
* site compiles with 0 errors and leaves the full suite green, because nothing else in the tree
|
||||||
|
* ever drives that specific constructor argument. In production it means a member the monitor
|
||||||
|
* classifies {@code GONE}/{@code NEVER_READY} never has its pending ticket failed — the caller
|
||||||
|
* keeps reporting {@code PENDING} for the full 30-minute async timeout instead of the immediate,
|
||||||
|
* accurate failure CB-580 exists to give it.
|
||||||
|
*
|
||||||
|
* <p>This test calls {@link Fleetd#healthFailTarget} directly — never {@code FleetHealthMonitor}
|
||||||
|
* or {@code main} — against a real {@link MessageService}, using the same {@code sendAsync} +
|
||||||
|
* {@code poll} observable {@link MessageServiceTest} already relies on to pin {@code
|
||||||
|
* MessageService.abandon} itself.
|
||||||
|
*/
|
||||||
|
class FleetdHealthFailTargetWiringTest {
|
||||||
|
|
||||||
|
private static final String T = "term_a";
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("Fleetd.healthFailTarget delegates to the real MessageService.abandon, not a no-op")
|
||||||
|
void healthFailTargetDelegatesToMessagesAbandon() throws Exception {
|
||||||
|
FakeHerdr herdr = new FakeHerdr().readText("$ prompt");
|
||||||
|
AgentControl agents = new AgentControl(herdr);
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
Injector injector = new Injector(agents);
|
||||||
|
MessageService messages = new MessageService(agents, injector, rendezvous);
|
||||||
|
|
||||||
|
BiConsumer<String, String> failTarget = Fleetd.healthFailTarget(messages);
|
||||||
|
|
||||||
|
String ticket = messages.sendAsync(T, "long task");
|
||||||
|
awaitWaiting(rendezvous);
|
||||||
|
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket).phase());
|
||||||
|
|
||||||
|
failTarget.accept(T, "member unreachable (health monitor)");
|
||||||
|
|
||||||
|
MessageService.TaskView view = null;
|
||||||
|
long deadline = System.currentTimeMillis() + 3000;
|
||||||
|
while (System.currentTimeMillis() < deadline) {
|
||||||
|
view = messages.poll(ticket);
|
||||||
|
if (view.phase() != MessageService.Phase.PENDING) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
//noinspection BusyWait
|
||||||
|
Thread.sleep(10);
|
||||||
|
}
|
||||||
|
assertEquals(MessageService.Phase.FAILED, view == null ? null : view.phase(),
|
||||||
|
"Fleetd.healthFailTarget(messages) must return messages::abandon — replacing it "
|
||||||
|
+ "with a no-op BiConsumer at the Fleetd.healthFailTarget call site means "
|
||||||
|
+ "this ticket is never failed and keeps polling as PENDING");
|
||||||
|
assertTrue(view.detail() != null && view.detail().contains("member unreachable"),
|
||||||
|
"the failure reason passed to failTarget.accept must reach MessageService.abandon "
|
||||||
|
+ "and end up in the ticket's detail");
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void awaitWaiting(Rendezvous rendezvous) throws InterruptedException {
|
||||||
|
long deadline = System.currentTimeMillis() + 2000;
|
||||||
|
while (!rendezvous.isWaiting(T) && System.currentTimeMillis() < deadline) {
|
||||||
|
//noinspection BusyWait
|
||||||
|
Thread.sleep(5);
|
||||||
|
}
|
||||||
|
assertTrue(rendezvous.isWaiting(T), "send should have opened its rendezvous waiter");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -2,16 +2,67 @@ package dev.ltms.fleet;
|
|||||||
|
|
||||||
import java.nio.file.Files;
|
import java.nio.file.Files;
|
||||||
import java.nio.file.Path;
|
import java.nio.file.Path;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* {@code AgentControl} caches {@code paneByTerminal}, so {@code HerdrRouter} must be its only
|
||||||
|
* production factory — a second instance means a second cache; the same reasoning applies to
|
||||||
|
* {@code WorkspaceControl}. {@code HerdrRouter}'s constructor is the one place both are built.
|
||||||
|
*
|
||||||
|
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||||
|
* HerdrRouter} and never runs {@code FleetdAssembly.assembleAndStart} — a green result proves only
|
||||||
|
* that neither watched file's text contains {@code new AgentControl(} or {@code new
|
||||||
|
* WorkspaceControl(}. It does not prove the instances {@code HerdrRouter} does build are the ones
|
||||||
|
* actually wired through the rest of the daemon, and it does not cover a bypass written into a
|
||||||
|
* production file other than the two this test reads.
|
||||||
|
*/
|
||||||
class FleetdHerdrControlConstructionTest {
|
class FleetdHerdrControlConstructionTest {
|
||||||
|
|
||||||
|
private static String source(String relativePath) throws Exception {
|
||||||
|
return Files.readString(Path.of(relativePath));
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
|
@DisplayName("[SOURCE TEXT] Fleetd.java never constructs AgentControl or WorkspaceControl directly")
|
||||||
void fleetdDelegatesStatefulControlsToTheRouter() throws Exception {
|
void fleetdDelegatesStatefulControlsToTheRouter() throws Exception {
|
||||||
// AgentControl caches paneByTerminal, so the router must be its only production factory.
|
String source = source("src/main/java/dev/ltms/fleet/Fleetd.java");
|
||||||
String source = Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
|
||||||
assertFalse(source.contains("new AgentControl("));
|
// A broken read (wrong working directory, wrong path, a file that came back empty) would
|
||||||
assertFalse(source.contains("new WorkspaceControl("));
|
// make the assertFalse checks below pass vacuously — a "clean" negative check that actually
|
||||||
|
// checked nothing. Guard against that first, with an anchor that has nothing to do with
|
||||||
|
// this mutation, so a bad read fails loudly here instead of silently proving nothing below.
|
||||||
|
assertTrue(source.contains("public final class Fleetd"),
|
||||||
|
"the read of Fleetd.java did not come back containing its own class declaration — "
|
||||||
|
+ "the assertFalse checks below would pass vacuously on a broken read; fix the "
|
||||||
|
+ "read before trusting this test.");
|
||||||
|
|
||||||
|
assertFalse(source.contains("new AgentControl("),
|
||||||
|
"Fleetd.java must not construct AgentControl directly — HerdrRouter is its only "
|
||||||
|
+ "production factory");
|
||||||
|
assertFalse(source.contains("new WorkspaceControl("),
|
||||||
|
"Fleetd.java must not construct WorkspaceControl directly — HerdrRouter is its only "
|
||||||
|
+ "production factory");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("[SOURCE TEXT] FleetdAssembly.java never constructs AgentControl or WorkspaceControl directly")
|
||||||
|
void fleetdAssemblyDelegatesStatefulControlsToTheRouter() throws Exception {
|
||||||
|
String source = source("src/main/java/dev/ltms/fleet/FleetdAssembly.java");
|
||||||
|
|
||||||
|
assertTrue(source.contains("final class FleetdAssembly"),
|
||||||
|
"the read of FleetdAssembly.java did not come back containing its own class "
|
||||||
|
+ "declaration — the assertFalse checks below would pass vacuously on a broken "
|
||||||
|
+ "read; fix the read before trusting this test.");
|
||||||
|
|
||||||
|
assertFalse(source.contains("new AgentControl("),
|
||||||
|
"FleetdAssembly.java must not construct AgentControl directly — HerdrRouter is its "
|
||||||
|
+ "only production factory");
|
||||||
|
assertFalse(source.contains("new WorkspaceControl("),
|
||||||
|
"FleetdAssembly.java must not construct WorkspaceControl directly — HerdrRouter is "
|
||||||
|
+ "its only production factory");
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,100 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.function.Function;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #602 gauge-wiring: {@link Fleetd#leadConfigDirLookup} is the factory {@code Fleetd.main}
|
||||||
|
* wires into {@code FleetMcp.LeadConfigDirSource} so {@code fleet_list}'s {@code context} row reads
|
||||||
|
* the transcript directory a lead's OWN profile actually writes to, instead of always falling back
|
||||||
|
* to the built-in {@code <user.home>/.claude} default (see {@code LeadContextGauge}).
|
||||||
|
*
|
||||||
|
* <p>{@code FleetMcpLeadContextGaugeWiringTest} proves the directory this factory returns is what
|
||||||
|
* actually gets read; this class proves the factory's own matching logic — the same
|
||||||
|
* {@code fleet.leaders.<name>.profile} link {@code Fleetd.leadSeatLookup} already follows (see
|
||||||
|
* {@code FleetdLeadSeatLookupTest}), one step further to that profile's own {@code configDir:}.
|
||||||
|
*/
|
||||||
|
class FleetdLeadConfigDirLookupTest {
|
||||||
|
|
||||||
|
private static FleetConfig.Profile profileWithConfigDir(String name, String configDir) {
|
||||||
|
return new FleetConfig.Profile(name, null, "claude-sonnet-5", configDir, null, null,
|
||||||
|
"tab", "fleet", "w #{n}", null, null, null, null, null, null, null,
|
||||||
|
null, 3, true, null, null, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig.Leader leadOnProfile(String profile) {
|
||||||
|
return new FleetConfig.Leader(profile, "lead: primary", 1, "lead:", 10, "claude", "claude-sonnet-5");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a lead on a profile that sets configDir resolves to that directory")
|
||||||
|
void leadOnAProfileWithConfigDirResolvesToIt() {
|
||||||
|
Map<String, FleetConfig.Profile> profiles = Map.of("opus", profileWithConfigDir("opus", "/mnt/opus-claude"));
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||||
|
Function<String, String> lookup = Fleetd.leadConfigDirLookup(() -> profiles, leaders);
|
||||||
|
|
||||||
|
assertEquals("/mnt/opus-claude", lookup.apply("primary"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("changing the config from directory A to directory B changes what the lookup reports")
|
||||||
|
void configChangeFromDirectoryAToDirectoryBChangesTheAnswer() {
|
||||||
|
java.util.concurrent.atomic.AtomicReference<Map<String, FleetConfig.Profile>> profilesRef =
|
||||||
|
new java.util.concurrent.atomic.AtomicReference<>(
|
||||||
|
Map.of("opus", profileWithConfigDir("opus", "/mnt/dir-a")));
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||||
|
Function<String, String> lookup = Fleetd.leadConfigDirLookup(profilesRef::get, leaders);
|
||||||
|
|
||||||
|
assertEquals("/mnt/dir-a", lookup.apply("primary"), "must read directory A before the config changes");
|
||||||
|
|
||||||
|
profilesRef.set(Map.of("opus", profileWithConfigDir("opus", "/mnt/dir-b")));
|
||||||
|
assertEquals("/mnt/dir-b", lookup.apply("primary"), "must read directory B once the LIVE config changes — "
|
||||||
|
+ "a lookup that snapshotted the profile map at construction would still answer directory A here");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a lead entry with no `profile:` (recognise-only) resolves to null, not a thrown exception")
|
||||||
|
void recogniseOnlyLeadWithNoProfileResolvesToNull() {
|
||||||
|
Map<String, FleetConfig.Profile> profiles = Map.of("opus", profileWithConfigDir("opus", "/mnt/opus-claude"));
|
||||||
|
FleetConfig.Leader recogniseOnly = new FleetConfig.Leader(null, "lead: primary", 1, "lead:", 10,
|
||||||
|
"claude", "claude-sonnet-5");
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", recogniseOnly);
|
||||||
|
Function<String, String> lookup = Fleetd.leadConfigDirLookup(() -> profiles, leaders);
|
||||||
|
|
||||||
|
assertNull(lookup.apply("primary"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a lead naming a profile that is not configured resolves to null, not a thrown exception")
|
||||||
|
void leadOnAnUnconfiguredProfileResolvesToNull() {
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("ghost-profile"));
|
||||||
|
Function<String, String> lookup = Fleetd.leadConfigDirLookup(Map::of, leaders);
|
||||||
|
|
||||||
|
assertNull(lookup.apply("primary"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a lead on a profile that sets no configDir override resolves to null")
|
||||||
|
void leadOnAProfileWithNoConfigDirResolvesToNull() {
|
||||||
|
Map<String, FleetConfig.Profile> profiles = Map.of("opus", profileWithConfigDir("opus", null));
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||||
|
Function<String, String> lookup = Fleetd.leadConfigDirLookup(() -> profiles, leaders);
|
||||||
|
|
||||||
|
assertNull(lookup.apply("primary"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("an unrecognised lead name resolves to null, not a thrown exception")
|
||||||
|
void unrecognisedLeadNameResolvesToNull() {
|
||||||
|
Function<String, String> lookup = Fleetd.leadConfigDirLookup(Map::of, Map.of());
|
||||||
|
|
||||||
|
assertNull(lookup.apply("ghost-lead"));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,213 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.mcp.FleetMcp;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 Shape A, unit r5 — {@code FleetdAssembly.java:488} wires {@link
|
||||||
|
* FleetMcp.LeadConfigDirSource} with {@code Fleetd.leadConfigDirSource(() -> config.get().profiles(),
|
||||||
|
* leaders)}. {@link FleetdLeadConfigDirSourceWiringTest} already pins that the FACTORY itself
|
||||||
|
* delegates to the real {@link Fleetd#leadConfigDirLookup} — but, by its own javadoc, it "does not
|
||||||
|
* and structurally cannot cover" whether the real call site in {@code FleetdAssembly} still calls
|
||||||
|
* that factory at all. Measured there: swapping that one-line call for a bare {@code
|
||||||
|
* FleetMcp.LeadConfigDirSource.none()} compiles with 0 errors and leaves the full suite green.
|
||||||
|
*
|
||||||
|
* <p>This is the literal fleetd #602/#606 defect, one call site away from its own fix: {@code main}
|
||||||
|
* (now {@code FleetdAssembly}) used to build {@code LeadConfigDirSource.none()} inline, the whole
|
||||||
|
* suite passed, and the live daemon reported {@code "state":"unknown"} for every lead's context,
|
||||||
|
* forever, with no test noticing. The fix extracted the factory; this test is the one that proves
|
||||||
|
* {@code FleetdAssembly}'s own call site still reaches it.
|
||||||
|
*
|
||||||
|
* <p>This test drives the REAL {@link FleetMcp} the real {@link FleetdAssembly#assembleAndStart}
|
||||||
|
* builds, reached through {@link FleetdRuntime#mcp()}, and reads the {@code leadConfigDirs} field it
|
||||||
|
* was constructed with via reflection — {@code FleetMcp} exposes no public accessor for it (unlike
|
||||||
|
* {@code quarantineSource()}/{@code leadSeatSource()}), so there is no non-reflective route to the
|
||||||
|
* live instance. The assertion resolves a REAL lead name against a REAL configured {@code
|
||||||
|
* configDir:}: {@link FleetMcp.LeadConfigDirSource#none()} (the historical defect, and the
|
||||||
|
* mis-wire this test's mutation cycles reintroduce) always returns {@code null} regardless of the
|
||||||
|
* input, so a non-null, config-matching answer is a property {@code none()} can never produce by
|
||||||
|
* accident.
|
||||||
|
*/
|
||||||
|
class FleetdLeadConfigDirSourceAssemblyTest {
|
||||||
|
|
||||||
|
private static final String LEAD_NAME = "opus";
|
||||||
|
private static final String LEAD_TAB = "lead: opus";
|
||||||
|
private static final String LEAD_PROFILE = "sonnet";
|
||||||
|
|
||||||
|
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||||
|
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> replyInbox;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException(
|
||||||
|
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class SentinelReplyInbox implements ReplyInbox, AutoCloseable {
|
||||||
|
@Override
|
||||||
|
public void own(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void release(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void publish(String target, String msgId, String content) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<InboxMessage> peek(String target) {
|
||||||
|
return List.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean ack(String target, String msgId) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir, String configDir) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
broker:
|
||||||
|
uri: "amqp://fake-test-broker/vh"
|
||||||
|
fleet:
|
||||||
|
leaders:
|
||||||
|
%s:
|
||||||
|
tab: "%s"
|
||||||
|
profile: %s
|
||||||
|
profiles:
|
||||||
|
%s:
|
||||||
|
subscription: true
|
||||||
|
argv: ["ccs", "sonnet"]
|
||||||
|
configDir: "%s"
|
||||||
|
""".formatted(LEAD_NAME, LEAD_TAB, LEAD_PROFILE, LEAD_PROFILE, configDir));
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
private static FleetMcp.LeadConfigDirSource leadConfigDirSourceOf(FleetMcp mcp) throws Exception {
|
||||||
|
Field field = FleetMcp.class.getDeclaredField("leadConfigDirs");
|
||||||
|
field.setAccessible(true);
|
||||||
|
return (FleetMcp.LeadConfigDirSource) field.get(mcp);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("[BEHAVIOURAL] the real assembled LeadConfigDirSource resolves a lead's REAL "
|
||||||
|
+ "configured configDir, not the none() stand-in's hardcoded null")
|
||||||
|
void assembledLeadConfigDirSourceResolvesTheRealConfiguredConfigDir(@TempDir Path dir) throws Exception {
|
||||||
|
String configuredConfigDir = "/mnt/fake-lead-configdir";
|
||||||
|
FleetConfig cfg = writeConfig(dir, configuredConfigDir);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||||
|
// Label FakeHerdr's own default pane's tab (term_a / w2:p7 / w2:t7, already carrying a live
|
||||||
|
// agent) to match fleet.leaders.opus.tab exactly, so LeadLauncher.ensureLeads() sees the
|
||||||
|
// lead as already live and does not try to auto-launch a second one.
|
||||||
|
ports.herdr.withTab("w2", "w2:t7", LEAD_TAB);
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
try {
|
||||||
|
FleetMcp.LeadConfigDirSource source = leadConfigDirSourceOf(runtime.mcp());
|
||||||
|
|
||||||
|
assertEquals(configuredConfigDir, source.configDirFor().apply(LEAD_NAME),
|
||||||
|
"fleet.leaders." + LEAD_NAME + ".profile (" + LEAD_PROFILE + ") configures "
|
||||||
|
+ "configDir: " + configuredConfigDir + " — the real assembled source must "
|
||||||
|
+ "resolve it. FleetMcp.LeadConfigDirSource.none() (the inert stand-in "
|
||||||
|
+ "this test's mutation cycles swap the call site for, and the historical "
|
||||||
|
+ "fleetd #602/#606 defect) always reports null here, whatever the input");
|
||||||
|
|
||||||
|
// A lead name the config does not recognise still resolves to null, not a crash — the
|
||||||
|
// same source, applied to an input that must stay at the inert answer even on the real,
|
||||||
|
// non-inert instance.
|
||||||
|
assertNull(source.configDirFor().apply("no-such-lead"));
|
||||||
|
} finally {
|
||||||
|
// Surefire runs the whole suite in one JVM fork (fleetd/pom.xml sets no forkCount /
|
||||||
|
// reuseForks), so the scheduler/loops this assembly starts must be torn down here, on the
|
||||||
|
// failure path too — hence try/finally rather than a bare statement at the end.
|
||||||
|
runtime.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,63 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.mcp.FleetMcp;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.util.Map;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Pins {@link Fleetd#leadConfigDirSource}'s own wiring of the window lookup into the returned
|
||||||
|
* {@link FleetMcp.LeadConfigDirSource}, not only the detached {@link Fleetd#leadContextWindowLookup}
|
||||||
|
* factory it delegates to. Calls the producer directly, with real {@link FleetConfig.Profile}/
|
||||||
|
* {@link FleetConfig.Leader} fixtures, and asserts on {@code windowFor()} — the companion of
|
||||||
|
* {@link FleetdLeadConfigDirSourceWiringTest}, which pins the same factory's {@code configDirFor()}.
|
||||||
|
*/
|
||||||
|
class FleetdLeadConfigDirSourceWindowWiringTest {
|
||||||
|
|
||||||
|
private static FleetConfig.Profile profileWithWindow(String name, Integer autoCompactWindow) {
|
||||||
|
return new FleetConfig.Profile(name, null, "claude-sonnet-5", null, null, null,
|
||||||
|
"tab", "fleet", "w #{n}", null, null, null, null, null, null, null,
|
||||||
|
null, null, true, null, null, null, null, null, autoCompactWindow, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig.Leader leadOnProfile(String profile) {
|
||||||
|
return new FleetConfig.Leader(profile, "lead: primary", 1, "lead:", 10, "claude", "claude-sonnet-5");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("the returned source resolves the lead's REAL configured effective window, not a hardcoded null")
|
||||||
|
void resolvesTheRealConfiguredWindow() {
|
||||||
|
Map<String, FleetConfig.Profile> profiles = Map.of("opus", profileWithWindow("opus", 250_000));
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||||
|
|
||||||
|
FleetMcp.LeadConfigDirSource source = Fleetd.leadConfigDirSource(() -> profiles, leaders);
|
||||||
|
|
||||||
|
assertEquals(250_000L, source.windowFor().apply("primary"),
|
||||||
|
"windowFor must delegate to the real leadContextWindowLookup, not a stub that always "
|
||||||
|
+ "returns null");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a lead on a profile with no window configured still resolves to null, not a crash")
|
||||||
|
void leadWithNoWindowConfiguredResolvesToNull() {
|
||||||
|
Map<String, FleetConfig.Profile> profiles = Map.of("opus", profileWithWindow("opus", null));
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||||
|
|
||||||
|
FleetMcp.LeadConfigDirSource source = Fleetd.leadConfigDirSource(() -> profiles, leaders);
|
||||||
|
|
||||||
|
assertNull(source.windowFor().apply("primary"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("an unrecognised lead name resolves to null, not a thrown exception")
|
||||||
|
void unrecognisedLeadNameResolvesToNull() {
|
||||||
|
FleetMcp.LeadConfigDirSource source = Fleetd.leadConfigDirSource(Map::of, Map.of());
|
||||||
|
|
||||||
|
assertNull(source.windowFor().apply("ghost-lead"));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,98 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.mcp.FleetMcp;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.util.Map;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #602 gauge-wiring follow-up (PR #606 review comment 17353): {@code Fleetd.main}'s {@code
|
||||||
|
* LeadConfigDirSource} local used to be a bare {@code new FleetMcp.LeadConfigDirSource(
|
||||||
|
* leadConfigDirLookup(...))} built inline, with nothing a test could call directly. Measured on
|
||||||
|
* that shape: replacing the whole expression with {@code FleetMcp.LeadConfigDirSource.none()} at
|
||||||
|
* the call site compiled with 0 errors and left the full 1822-test suite green — the daemon could
|
||||||
|
* be changed to always report every lead's context as {@code UNKNOWN}, forever, and no test would
|
||||||
|
* notice. That is the same hand-built-vs-config-wired shape as fleetd #561/#248/#426/#562
|
||||||
|
* ({@code FleetdLoopHealthSourceWiringTest}).
|
||||||
|
*
|
||||||
|
* <p>{@code FleetMcpLeadContextGaugeWiringTest} and {@code FleetdLeadConfigDirLookupTest} both
|
||||||
|
* predate this class and are both still correct — but neither can catch the mutation above. One
|
||||||
|
* builds its own {@code FleetMcp} and hands it its own {@code LeadConfigDirSource}; the other
|
||||||
|
* builds its own lookup and calls {@link Fleetd#leadConfigDirLookup} directly. Neither one ever
|
||||||
|
* calls the thing {@code Fleetd.main} actually calls.
|
||||||
|
*
|
||||||
|
* <p>The fix extracts the inline {@code new} into {@link Fleetd#leadConfigDirSource}, a
|
||||||
|
* package-private factory in the same style as {@link Fleetd#loopHealthSource}/{@link
|
||||||
|
* Fleetd#capacitySource}/{@link Fleetd#healthCoverageSource} — which is exactly what makes it
|
||||||
|
* directly callable here. This test calls that factory with real {@link FleetConfig.Profile}/
|
||||||
|
* {@link FleetConfig.Leader} fixtures (the same shapes {@code FleetdLeadConfigDirLookupTest}
|
||||||
|
* already uses) and asserts the returned source resolves a real {@code configDir} — a property
|
||||||
|
* that would be false if {@link Fleetd#leadConfigDirSource} were mutated to {@code return
|
||||||
|
* FleetMcp.LeadConfigDirSource.none();}. Measured: mutating exactly that line makes
|
||||||
|
* {@link #resolvesTheRealConfiguredConfigDir()} fail ({@code expected: </mnt/opus-claude> but was:
|
||||||
|
* <null>}); restoring it makes the whole suite green again.
|
||||||
|
*
|
||||||
|
* <p><b>What this class does not and cannot cover.</b> {@code main}'s own line —
|
||||||
|
* {@code leadConfigDirSource(() -> config.get().profiles(), leaders)} — could itself be swapped
|
||||||
|
* for a bare {@code FleetMcp.LeadConfigDirSource.none()}, bypassing this factory entirely. Measured:
|
||||||
|
* that exact mutation compiles with 0 errors and leaves every test in this file, and the full
|
||||||
|
* 1825-test suite, green. {@link Fleetd#loopHealthSource}'s own wiring test has the identical gap
|
||||||
|
* for its own one-line call in {@code main} — no test in this codebase calls {@code Fleetd.main}
|
||||||
|
* far enough to observe which factory call it made. This class narrows the gap from "nothing tests
|
||||||
|
* the wiring" (the pre-extraction state this ticket found) to "the factory's own logic is pinned,
|
||||||
|
* and main's call to it is a one-line, visually-verifiable delegation" — the same standard already
|
||||||
|
* accepted for {@code loopHealthSource}/{@code capacitySource}/{@code healthCoverageSource}.
|
||||||
|
*/
|
||||||
|
class FleetdLeadConfigDirSourceWiringTest {
|
||||||
|
|
||||||
|
private static FleetConfig.Profile profileWithConfigDir(String name, String configDir) {
|
||||||
|
return new FleetConfig.Profile(name, null, "claude-sonnet-5", configDir, null, null,
|
||||||
|
"tab", "fleet", "w #{n}", null, null, null, null, null, null, null,
|
||||||
|
null, 3, true, null, null, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig.Leader leadOnProfile(String profile) {
|
||||||
|
return new FleetConfig.Leader(profile, "lead: primary", 1, "lead:", 10, "claude", "claude-sonnet-5");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("the returned source resolves the lead's REAL configured configDir, not a hardcoded null")
|
||||||
|
void resolvesTheRealConfiguredConfigDir() {
|
||||||
|
Map<String, FleetConfig.Profile> profiles = Map.of("opus", profileWithConfigDir("opus", "/mnt/opus-claude"));
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||||
|
|
||||||
|
FleetMcp.LeadConfigDirSource source = Fleetd.leadConfigDirSource(() -> profiles, leaders);
|
||||||
|
|
||||||
|
assertEquals("/mnt/opus-claude", source.configDirFor().apply("primary"),
|
||||||
|
"the configDirFor function must delegate to the real leadConfigDirLookup — mutating "
|
||||||
|
+ "Fleetd.leadConfigDirSource's own body to `return FleetMcp.LeadConfigDirSource.none();` "
|
||||||
|
+ "must fail this assertion (measured: it does — see this test's class javadoc for the "
|
||||||
|
+ "companion measurement on main's one-line call to this factory, which this assertion "
|
||||||
|
+ "does not and structurally cannot cover)");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a lead on a profile with no configDir override still resolves to null, not a crash")
|
||||||
|
void leadWithNoConfigDirOverrideResolvesToNull() {
|
||||||
|
Map<String, FleetConfig.Profile> profiles = Map.of("opus", profileWithConfigDir("opus", null));
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||||
|
|
||||||
|
FleetMcp.LeadConfigDirSource source = Fleetd.leadConfigDirSource(() -> profiles, leaders);
|
||||||
|
|
||||||
|
assertNull(source.configDirFor().apply("primary"),
|
||||||
|
"no configDir: override configured ⇒ null, so LeadContextGauge falls back to its own default");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("an unrecognised lead name resolves to null, not a thrown exception")
|
||||||
|
void unrecognisedLeadNameResolvesToNull() {
|
||||||
|
FleetMcp.LeadConfigDirSource source = Fleetd.leadConfigDirSource(Map::of, Map.of());
|
||||||
|
|
||||||
|
assertNull(source.configDirFor().apply("ghost-lead"));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,156 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrException;
|
||||||
|
import dev.ltms.fleet.lead.LeadContextGauge;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.io.IOException;
|
||||||
|
import java.nio.charset.StandardCharsets;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.function.Function;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertDoesNotThrow;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #609: {@link Fleetd#leadContextLookup} is the factory {@code Fleetd.main} wires into
|
||||||
|
* {@code LeadHeartbeatLoop.LeadContextSource} so the idle-lead heartbeat can read a lead's own
|
||||||
|
* {@link LeadContextGauge} reading by its terminal id. Three hops: terminal → lead name (unknown ⇒
|
||||||
|
* UNKNOWN), lead name → configDir, and {@code agents.get(terminal)} for the live session id and
|
||||||
|
* agent type the gauge itself needs — a herdr failure on that last hop must degrade to UNKNOWN, not
|
||||||
|
* throw and kill the heartbeat's own tick.
|
||||||
|
*
|
||||||
|
* <p>{@code AgentControl.agentCall} resolves a {@code term_}-prefixed target's pane id via a first
|
||||||
|
* {@code agent.list} round trip (see its own javadoc); this class's terminal id deliberately does
|
||||||
|
* NOT start with {@code term_} so the stub {@link HerdrClient} below only needs to answer
|
||||||
|
* {@code agent.get} — the one call this factory actually depends on.
|
||||||
|
*/
|
||||||
|
class FleetdLeadContextLookupTest {
|
||||||
|
|
||||||
|
private static final String LEAD_TERMINAL = "leadpane1";
|
||||||
|
private static final String LEAD_NAME = "opus";
|
||||||
|
private static final String SESSION_ID = "sess-609-happy-path";
|
||||||
|
|
||||||
|
private static String usageLine(long tokens) {
|
||||||
|
return "{\"type\":\"assistant\",\"message\":{\"role\":\"assistant\",\"usage\":{"
|
||||||
|
+ "\"input_tokens\":" + tokens + ",\"cache_read_input_tokens\":0,\"cache_creation_input_tokens\":0}}}";
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Lays out {@code <configDir>/projects/<anySlug>/<sessionId>.jsonl} carrying one usage record. */
|
||||||
|
private static void writeTranscript(Path configDir, String sessionId, long tokens) throws IOException {
|
||||||
|
Path projectDir = configDir.resolve("projects").resolve("some-project-slug");
|
||||||
|
Files.createDirectories(projectDir);
|
||||||
|
Files.writeString(projectDir.resolve(sessionId + ".jsonl"), usageLine(tokens) + "\n", StandardCharsets.UTF_8);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** An {@link AgentControl} whose every {@code agent.get} answers with the given session/type/status. */
|
||||||
|
private static AgentControl agentControlStub(String sessionId, String agentType, String status) {
|
||||||
|
HerdrClient client = new HerdrClient() {
|
||||||
|
@Override
|
||||||
|
public JsonNode call(String method, Object params) {
|
||||||
|
if (!"agent.get".equals(method)) {
|
||||||
|
throw new HerdrException("stub has no canned response for " + method);
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
String sessionField = sessionId == null ? ""
|
||||||
|
: ",\"agent_session\":{\"kind\":\"id\",\"value\":\"" + sessionId + "\"}";
|
||||||
|
String agentField = agentType == null ? "null" : "\"" + agentType + "\"";
|
||||||
|
return new ObjectMapper().readTree(("""
|
||||||
|
{"type":"agent_info","agent":{"terminal_id":"%s","agent":%s,
|
||||||
|
"agent_status":"%s"%s}}""")
|
||||||
|
.formatted(LEAD_TERMINAL, agentField, status, sessionField));
|
||||||
|
} catch (Exception e) {
|
||||||
|
throw new HerdrException("stub decode failed", e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
}
|
||||||
|
};
|
||||||
|
return new AgentControl(client);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** An {@link AgentControl} whose every herdr call fails — models a herdr hiccup mid-tick. */
|
||||||
|
private static AgentControl throwingAgentControl() {
|
||||||
|
HerdrClient client = new HerdrClient() {
|
||||||
|
@Override
|
||||||
|
public JsonNode call(String method, Object params) {
|
||||||
|
throw new HerdrException("herdr unreachable (stub)");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
}
|
||||||
|
};
|
||||||
|
return new AgentControl(client);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("an unrecognised terminal resolves to UNKNOWN, not a thrown exception")
|
||||||
|
void unrecognisedTerminalResolvesToUnknown() {
|
||||||
|
Function<String, LeadContextGauge.Reading> lookup = Fleetd.leadContextLookup(
|
||||||
|
new LeadContextGauge(), throwingAgentControl(), Map::of, name -> null, name -> null);
|
||||||
|
|
||||||
|
LeadContextGauge.Reading reading = assertDoesNotThrow(() -> lookup.apply("ghost-terminal"));
|
||||||
|
|
||||||
|
assertEquals(LeadContextGauge.State.UNKNOWN, reading.state());
|
||||||
|
assertNull(reading.tokens());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("agents.get throwing degrades to UNKNOWN, no exception escapes")
|
||||||
|
void agentsGetThrowingDegradesToUnknown() {
|
||||||
|
Map<String, String> liveLeadTerminals = Map.of(LEAD_TERMINAL, LEAD_NAME);
|
||||||
|
Function<String, LeadContextGauge.Reading> lookup = Fleetd.leadContextLookup(
|
||||||
|
new LeadContextGauge(), throwingAgentControl(), () -> liveLeadTerminals, name -> null, name -> null);
|
||||||
|
|
||||||
|
LeadContextGauge.Reading reading = assertDoesNotThrow(() -> lookup.apply(LEAD_TERMINAL));
|
||||||
|
|
||||||
|
assertEquals(LeadContextGauge.State.UNKNOWN, reading.state(),
|
||||||
|
"a herdr failure resolving the live agent must degrade to UNKNOWN, never kill the heartbeat's tick");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("the happy path resolves the configDir, sessionId and agentType through to the gauge")
|
||||||
|
void happyPathPassesResolvedFactsThroughToTheGauge(@TempDir Path tmp) throws IOException {
|
||||||
|
writeTranscript(tmp, SESSION_ID, 12_345);
|
||||||
|
Map<String, String> liveLeadTerminals = Map.of(LEAD_TERMINAL, LEAD_NAME);
|
||||||
|
AgentControl agents = agentControlStub(SESSION_ID, "claude", "idle");
|
||||||
|
Function<String, LeadContextGauge.Reading> lookup = Fleetd.leadContextLookup(
|
||||||
|
new LeadContextGauge(), agents, () -> liveLeadTerminals,
|
||||||
|
name -> LEAD_NAME.equals(name) ? tmp.toString() : null, name -> null);
|
||||||
|
|
||||||
|
LeadContextGauge.Reading reading = lookup.apply(LEAD_TERMINAL);
|
||||||
|
|
||||||
|
assertEquals(LeadContextGauge.State.OK, reading.state(),
|
||||||
|
"the resolved configDir + the live agent's own sessionId/agentType must reach the gauge — "
|
||||||
|
+ "a wrong hop anywhere in the chain would read no transcript and report UNKNOWN instead");
|
||||||
|
assertEquals(12_345L, reading.tokens());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a lead whose agent type is not claude still resolves to UNKNOWN, never a crash")
|
||||||
|
void nonClaudeAgentTypeResolvesToUnknown(@TempDir Path tmp) throws IOException {
|
||||||
|
writeTranscript(tmp, SESSION_ID, 12_345);
|
||||||
|
Map<String, String> liveLeadTerminals = Map.of(LEAD_TERMINAL, LEAD_NAME);
|
||||||
|
AgentControl agents = agentControlStub(SESSION_ID, "opencode", "idle");
|
||||||
|
Function<String, LeadContextGauge.Reading> lookup = Fleetd.leadContextLookup(
|
||||||
|
new LeadContextGauge(), agents, () -> liveLeadTerminals,
|
||||||
|
name -> LEAD_NAME.equals(name) ? tmp.toString() : null, name -> null);
|
||||||
|
|
||||||
|
LeadContextGauge.Reading reading = lookup.apply(LEAD_TERMINAL);
|
||||||
|
|
||||||
|
assertEquals(LeadContextGauge.State.UNKNOWN, reading.state(),
|
||||||
|
"the agentType hop must reach the gauge too — a non-claude peer must not be misread as claude");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,223 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.lead.LeadContextGauge;
|
||||||
|
import dev.ltms.fleet.msg.LeadHeartbeatLoop;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.io.IOException;
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.nio.charset.StandardCharsets;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #659: {@code FleetdAssembly.java} wires {@code Fleetd.leadContextSource}'s window-lookup
|
||||||
|
* argument with {@code Fleetd.leadContextWindowLookup(() -> config.get().profiles(), leaders)} —
|
||||||
|
* but nothing called the real assembled {@link LeadHeartbeatLoop} far enough to prove that
|
||||||
|
* argument is the one the live heartbeat reads through. Measured: swapping that one call-site
|
||||||
|
* argument for {@code _ -> null} compiles with 0 errors and leaves the full suite green.
|
||||||
|
*
|
||||||
|
* <p>This test drives the REAL {@link LeadHeartbeatLoop} the real {@link
|
||||||
|
* FleetdAssembly#assembleAndStart} builds, reached through {@link FleetdRuntime#heartbeat()}, and
|
||||||
|
* reads its private {@code contextSource} field via reflection — the loop exposes no public
|
||||||
|
* accessor for it, the same reason {@link FleetdLeadConfigDirSourceAssemblyTest} reflects on
|
||||||
|
* {@code FleetMcp.leadConfigDirs}. The configured profile's {@code autoCompactWindow: 100000}
|
||||||
|
* resolves a HIGH threshold of {@code 66666} ({@link LeadContextGauge}'s {@code 2/3} fraction) —
|
||||||
|
* far below the fixed {@code 200000} fallback a lost window argument would silently revert to.
|
||||||
|
* {@code 90000} live tokens sits between the two: HIGH under the real window, OK under the
|
||||||
|
* fallback — a property the fallback can never produce by accident.
|
||||||
|
*/
|
||||||
|
class FleetdLeadContextSourceWindowAssemblyTest {
|
||||||
|
|
||||||
|
private static final String LEAD_NAME = "opus";
|
||||||
|
private static final String LEAD_TAB = "lead: opus";
|
||||||
|
private static final String LEAD_PROFILE = "sonnet";
|
||||||
|
/** {@code FakeHerdr}'s own default {@code agent.list} entry: terminal {@code term_a}, session {@code sess-1111}. */
|
||||||
|
private static final String LEAD_TERMINAL = "term_a";
|
||||||
|
private static final String LEAD_SESSION_ID = "sess-1111";
|
||||||
|
|
||||||
|
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||||
|
|
||||||
|
final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
return herdr;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> replyInbox;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException(
|
||||||
|
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class SentinelReplyInbox implements ReplyInbox, AutoCloseable {
|
||||||
|
@Override
|
||||||
|
public void own(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void release(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void publish(String target, String msgId, String content) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<InboxMessage> peek(String target) {
|
||||||
|
return List.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean ack(String target, String msgId) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static String usageLine(long tokens) {
|
||||||
|
return "{\"type\":\"assistant\",\"message\":{\"role\":\"assistant\",\"usage\":{"
|
||||||
|
+ "\"input_tokens\":" + tokens + ",\"cache_read_input_tokens\":0,\"cache_creation_input_tokens\":0}}}";
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Lays out {@code <configDir>/projects/<anySlug>/<sessionId>.jsonl} carrying one usage record. */
|
||||||
|
private static void writeTranscript(Path configDir, String sessionId, long tokens) throws IOException {
|
||||||
|
Path projectDir = configDir.resolve("projects").resolve("some-project-slug");
|
||||||
|
Files.createDirectories(projectDir);
|
||||||
|
Files.writeString(projectDir.resolve(sessionId + ".jsonl"), usageLine(tokens) + "\n", StandardCharsets.UTF_8);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir, String configDir) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
broker:
|
||||||
|
uri: "amqp://fake-test-broker/vh"
|
||||||
|
leadHeartbeat:
|
||||||
|
idleAfterSeconds: 600
|
||||||
|
backoffMs: 15000
|
||||||
|
quietNudgeCap: 5
|
||||||
|
fleet:
|
||||||
|
leaders:
|
||||||
|
%s:
|
||||||
|
tab: "%s"
|
||||||
|
profile: %s
|
||||||
|
profiles:
|
||||||
|
%s:
|
||||||
|
subscription: true
|
||||||
|
argv: ["ccs", "sonnet"]
|
||||||
|
configDir: "%s"
|
||||||
|
autoCompactWindow: 100000
|
||||||
|
""".formatted(LEAD_NAME, LEAD_TAB, LEAD_PROFILE, LEAD_PROFILE, configDir));
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
private static LeadHeartbeatLoop.LeadContextSource contextSourceOf(LeadHeartbeatLoop heartbeat) throws Exception {
|
||||||
|
Field field = LeadHeartbeatLoop.class.getDeclaredField("contextSource");
|
||||||
|
field.setAccessible(true);
|
||||||
|
return (LeadHeartbeatLoop.LeadContextSource) field.get(heartbeat);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("[BEHAVIOURAL] the real assembled heartbeat loop resolves HIGH against the lead's "
|
||||||
|
+ "REAL configured window, not the fixed 200000 fallback a lost window argument reverts to")
|
||||||
|
void assembledHeartbeatContextSourceResolvesTheRealConfiguredWindow(@TempDir Path dir) throws Exception {
|
||||||
|
FleetConfig cfg = writeConfig(dir, dir.toString());
|
||||||
|
writeTranscript(dir, LEAD_SESSION_ID, 90_000);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||||
|
// Label FakeHerdr's own default pane's tab (term_a / w2:p7 / w2:t7, already carrying a live
|
||||||
|
// agent on session sess-1111) to match fleet.leaders.opus.tab exactly, so LeadTabScanner
|
||||||
|
// recognises it as the live "opus" lead without a second auto-launched pane.
|
||||||
|
ports.herdr.withTab("w2", "w2:t7", LEAD_TAB).agentSessionId(LEAD_SESSION_ID);
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
try {
|
||||||
|
LeadHeartbeatLoop.LeadContextSource source = contextSourceOf(runtime.heartbeat());
|
||||||
|
|
||||||
|
LeadContextGauge.Reading reading = source.readingFor().apply(LEAD_TERMINAL);
|
||||||
|
|
||||||
|
assertEquals(LeadContextGauge.State.HIGH, reading.state(),
|
||||||
|
"profiles." + LEAD_PROFILE + ".autoCompactWindow: 100000 resolves a HIGH threshold "
|
||||||
|
+ "of 66666 tokens — 90000 live tokens must read HIGH against it. Mutating "
|
||||||
|
+ "FleetdAssembly's window-lookup argument to `_ -> null` falls back to the "
|
||||||
|
+ "fixed 200000 threshold, under which 90000 reads OK instead: " + reading);
|
||||||
|
} finally {
|
||||||
|
// Surefire runs the whole suite in one JVM fork (fleetd/pom.xml sets no forkCount /
|
||||||
|
// reuseForks), so the scheduler/loops this assembly starts must be torn down here, on the
|
||||||
|
// failure path too — hence try/finally rather than a bare statement at the end.
|
||||||
|
runtime.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,109 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrException;
|
||||||
|
import dev.ltms.fleet.lead.LeadContextGauge;
|
||||||
|
import dev.ltms.fleet.msg.LeadHeartbeatLoop;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.io.IOException;
|
||||||
|
import java.nio.charset.StandardCharsets;
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.Map;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #609: {@code Fleetd.main}'s {@code LeadHeartbeatLoop.LeadContextSource} local could be
|
||||||
|
* swapped for a bare {@code LeadHeartbeatLoop.LeadContextSource.none()} at the call site — compiling
|
||||||
|
* with 0 errors and leaving every pre-existing test green — exactly the shape #602/#606 already found
|
||||||
|
* for {@code LeadConfigDirSource} (see {@code FleetdLeadConfigDirSourceWiringTest}'s own javadoc for
|
||||||
|
* the measured version of that gap).
|
||||||
|
*
|
||||||
|
* <p>The fix follows the same pattern: {@link Fleetd#leadContextSource} is the extracted,
|
||||||
|
* directly-callable factory {@code main} calls to build the source it hands {@code
|
||||||
|
* LeadHeartbeatLoop}'s constructor. This test calls that exact factory and asserts it resolves a
|
||||||
|
* REAL reading off a real transcript file — a property that would be false if {@link
|
||||||
|
* Fleetd#leadContextSource} were mutated to {@code return LeadHeartbeatLoop.LeadContextSource.none();}.
|
||||||
|
*
|
||||||
|
* <p>What this class does not and cannot cover: {@code main}'s own one-line call to this factory
|
||||||
|
* could itself be swapped for {@code LeadHeartbeatLoop.LeadContextSource.none()}, bypassing this
|
||||||
|
* factory entirely — the same structural gap {@code FleetdLeadConfigDirSourceWiringTest} names for
|
||||||
|
* its own factory, and for the same reason (no test in this codebase calls {@code Fleetd.main} far
|
||||||
|
* enough to observe which factory call it made).
|
||||||
|
*/
|
||||||
|
class FleetdLeadContextSourceWiringTest {
|
||||||
|
|
||||||
|
/** Deliberately not {@code term_}-prefixed — see {@code FleetdLeadContextLookupTest}'s class doc. */
|
||||||
|
private static final String LEAD_TERMINAL = "leadpane1";
|
||||||
|
private static final String LEAD_NAME = "opus";
|
||||||
|
private static final String SESSION_ID = "sess-609-wiring";
|
||||||
|
|
||||||
|
private static String usageLine(long tokens) {
|
||||||
|
return "{\"type\":\"assistant\",\"message\":{\"role\":\"assistant\",\"usage\":{"
|
||||||
|
+ "\"input_tokens\":" + tokens + ",\"cache_read_input_tokens\":0,\"cache_creation_input_tokens\":0}}}";
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void writeTranscript(Path configDir, String sessionId, long tokens) throws IOException {
|
||||||
|
Path projectDir = configDir.resolve("projects").resolve("some-project-slug");
|
||||||
|
Files.createDirectories(projectDir);
|
||||||
|
Files.writeString(projectDir.resolve(sessionId + ".jsonl"), usageLine(tokens) + "\n", StandardCharsets.UTF_8);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static AgentControl agentControlStub() {
|
||||||
|
HerdrClient client = new HerdrClient() {
|
||||||
|
@Override
|
||||||
|
public JsonNode call(String method, Object params) {
|
||||||
|
if (!"agent.get".equals(method)) {
|
||||||
|
throw new HerdrException("stub has no canned response for " + method);
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
return new ObjectMapper().readTree(("""
|
||||||
|
{"type":"agent_info","agent":{"terminal_id":"%s","agent":"claude",
|
||||||
|
"agent_status":"idle","agent_session":{"kind":"id","value":"%s"}}}""")
|
||||||
|
.formatted(LEAD_TERMINAL, SESSION_ID));
|
||||||
|
} catch (Exception e) {
|
||||||
|
throw new HerdrException("stub decode failed", e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
}
|
||||||
|
};
|
||||||
|
return new AgentControl(client);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("main's factory resolves a REAL reading, not the inert none() answer")
|
||||||
|
void resolvesARealReadingNotTheInertNoneAnswer(@TempDir Path tmp) throws IOException {
|
||||||
|
writeTranscript(tmp, SESSION_ID, 54_321);
|
||||||
|
Map<String, String> liveLeadTerminals = Map.of(LEAD_TERMINAL, LEAD_NAME);
|
||||||
|
|
||||||
|
LeadHeartbeatLoop.LeadContextSource source = Fleetd.leadContextSource(new LeadContextGauge(),
|
||||||
|
agentControlStub(), () -> liveLeadTerminals,
|
||||||
|
name -> LEAD_NAME.equals(name) ? tmp.toString() : null, name -> null);
|
||||||
|
|
||||||
|
LeadContextGauge.Reading reading = source.readingFor().apply(LEAD_TERMINAL);
|
||||||
|
|
||||||
|
assertEquals(LeadContextGauge.State.OK, reading.state(),
|
||||||
|
"mutating Fleetd.leadContextSource's own body to `return LeadHeartbeatLoop.LeadContextSource.none();` "
|
||||||
|
+ "must fail this assertion");
|
||||||
|
assertEquals(54_321L, reading.tokens());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("an unrecognised lead terminal resolves to UNKNOWN, not a thrown exception")
|
||||||
|
void unrecognisedTerminalResolvesToUnknown() {
|
||||||
|
LeadHeartbeatLoop.LeadContextSource source = Fleetd.leadContextSource(new LeadContextGauge(),
|
||||||
|
agentControlStub(), Map::of, name -> null, name -> null);
|
||||||
|
|
||||||
|
assertEquals(LeadContextGauge.State.UNKNOWN, source.readingFor().apply("ghost-terminal").state());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,96 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.function.Function;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* {@link Fleetd#leadContextWindowLookup} is the factory wired into {@code
|
||||||
|
* FleetMcp.LeadConfigDirSource} and {@code LeadHeartbeatLoop.LeadContextSource} so {@link
|
||||||
|
* dev.ltms.fleet.lead.LeadContextGauge} scales its HIGH threshold against a lead's own profile's
|
||||||
|
* effective auto-compact window instead of always the gauge's fixed fallback — the same {@code
|
||||||
|
* fleet.leaders.<name>.profile} link {@link Fleetd#leadConfigDirLookup} already follows, one step
|
||||||
|
* further to {@link FleetConfig.Profile#effectiveAutoCompactWindow()}.
|
||||||
|
*/
|
||||||
|
class FleetdLeadContextWindowLookupTest {
|
||||||
|
|
||||||
|
private static FleetConfig.Profile profileWithWindow(String name, Integer autoCompactWindow,
|
||||||
|
Map<String, String> env) {
|
||||||
|
return new FleetConfig.Profile(name, null, "claude-sonnet-5", null, null, null,
|
||||||
|
"tab", "fleet", "w #{n}", null, null, null, null, null, null, env,
|
||||||
|
null, null, true, null, null, null, null, null, autoCompactWindow, null);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig.Leader leadOnProfile(String profile) {
|
||||||
|
return new FleetConfig.Leader(profile, "lead: primary", 1, "lead:", 10, "claude", "claude-sonnet-5");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a lead on a profile that sets autoCompactWindow resolves to that window")
|
||||||
|
void leadOnAProfileWithAutoCompactWindowResolvesToIt() {
|
||||||
|
Map<String, FleetConfig.Profile> profiles =
|
||||||
|
Map.of("opus", profileWithWindow("opus", 250_000, Map.of()));
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||||
|
Function<String, Long> lookup = Fleetd.leadContextWindowLookup(() -> profiles, leaders);
|
||||||
|
|
||||||
|
assertEquals(250_000L, lookup.apply("primary"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("yaml and env disagree: the lookup resolves the env value, not the yaml one")
|
||||||
|
void yamlAndEnvDisagreeLookupResolvesTheEnvValue() {
|
||||||
|
Map<String, FleetConfig.Profile> profiles = Map.of("opus",
|
||||||
|
profileWithWindow("opus", 250_000, Map.of("CLAUDE_CODE_AUTO_COMPACT_WINDOW", "150000")));
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||||
|
Function<String, Long> lookup = Fleetd.leadContextWindowLookup(() -> profiles, leaders);
|
||||||
|
|
||||||
|
assertEquals(150_000L, lookup.apply("primary"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a lead entry with no `profile:` resolves to null, not a thrown exception")
|
||||||
|
void recogniseOnlyLeadWithNoProfileResolvesToNull() {
|
||||||
|
Map<String, FleetConfig.Profile> profiles =
|
||||||
|
Map.of("opus", profileWithWindow("opus", 250_000, Map.of()));
|
||||||
|
FleetConfig.Leader recogniseOnly = new FleetConfig.Leader(null, "lead: primary", 1, "lead:", 10,
|
||||||
|
"claude", "claude-sonnet-5");
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", recogniseOnly);
|
||||||
|
Function<String, Long> lookup = Fleetd.leadContextWindowLookup(() -> profiles, leaders);
|
||||||
|
|
||||||
|
assertNull(lookup.apply("primary"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a lead naming a profile that is not configured resolves to null, not a thrown exception")
|
||||||
|
void leadOnAnUnconfiguredProfileResolvesToNull() {
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("ghost-profile"));
|
||||||
|
Function<String, Long> lookup = Fleetd.leadContextWindowLookup(Map::of, leaders);
|
||||||
|
|
||||||
|
assertNull(lookup.apply("primary"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a lead on a profile that resolves no window at all resolves to null")
|
||||||
|
void leadOnAProfileWithNoWindowResolvesToNull() {
|
||||||
|
Map<String, FleetConfig.Profile> profiles =
|
||||||
|
Map.of("opus", profileWithWindow("opus", null, Map.of()));
|
||||||
|
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||||
|
Function<String, Long> lookup = Fleetd.leadContextWindowLookup(() -> profiles, leaders);
|
||||||
|
|
||||||
|
assertNull(lookup.apply("primary"));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("an unrecognised lead name resolves to null, not a thrown exception")
|
||||||
|
void unrecognisedLeadNameResolvesToNull() {
|
||||||
|
Function<String, Long> lookup = Fleetd.leadContextWindowLookup(Map::of, Map.of());
|
||||||
|
|
||||||
|
assertNull(lookup.apply("ghost-lead"));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,48 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.net.ServerSocket;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #589 Group 3, site {@code :518}. {@code Fleetd.main} passes {@code LeadMailbox::open} as
|
||||||
|
* the {@link Fleetd.LeadMailboxOpener} argument to {@code openLeadMailbox(...)} — before this
|
||||||
|
* ticket that method reference was inline at the call site. Measured: replacing it with the inert
|
||||||
|
* {@code (uri, selfCoordId, prefetch) -> null} compiles with 0 errors and leaves the full suite
|
||||||
|
* green — {@code FleetdLeadMailboxSelectionTest} drives {@code openLeadMailbox} with its own
|
||||||
|
* injected opener and never observes what {@code main} itself actually passes.
|
||||||
|
*
|
||||||
|
* <p>This test calls {@link Fleetd#leadMailboxOpener} directly and proves it is the real,
|
||||||
|
* network-attempting opener rather than a stub: pointed at a guaranteed-closed local port, it must
|
||||||
|
* throw, exactly mirroring {@link FleetdReplyInboxOpenerWiringTest} for the reply-inbox opener.
|
||||||
|
* The inert form never attempts a connection and returns {@code null} without throwing, so it
|
||||||
|
* fails this assertion silently.
|
||||||
|
*/
|
||||||
|
class FleetdLeadMailboxOpenerWiringTest {
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("Fleetd.leadMailboxOpener is the real LeadMailbox::open, not a stub that never connects")
|
||||||
|
void leadMailboxOpenerAttemptsARealConnection() throws Exception {
|
||||||
|
int closedPort;
|
||||||
|
try (ServerSocket socket = new ServerSocket(0)) {
|
||||||
|
closedPort = socket.getLocalPort();
|
||||||
|
} // released immediately — connecting to it now is a guaranteed refusal, not a fluke
|
||||||
|
|
||||||
|
Fleetd.LeadMailboxOpener opener = Fleetd.leadMailboxOpener();
|
||||||
|
|
||||||
|
IllegalStateException thrown = assertThrows(IllegalStateException.class,
|
||||||
|
() -> opener.open("amqp://user:pw@127.0.0.1:" + closedPort + "/coord", "coord-1", 50),
|
||||||
|
"Fleetd.leadMailboxOpener() must be LeadMailbox::open — a real network attempt "
|
||||||
|
+ "against a genuinely unreachable broker must throw. The inert form "
|
||||||
|
+ "(uri, selfCoordId, prefetch) -> null never attempts a connection and "
|
||||||
|
+ "returns null instead of throwing, so it would fail this assertion "
|
||||||
|
+ "silently.");
|
||||||
|
assertTrue(thrown.getMessage().contains("cannot connect to AMQP coordination broker"),
|
||||||
|
"must be LeadMailbox.open's own real failure message, not a different exception "
|
||||||
|
+ "shape standing in for it");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,15 +1,13 @@
|
|||||||
package dev.ltms.fleet;
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
import ch.qos.logback.classic.Level;
|
import ch.qos.logback.classic.Level;
|
||||||
import ch.qos.logback.classic.Logger;
|
|
||||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
import ch.qos.logback.core.read.ListAppender;
|
|
||||||
import dev.ltms.fleet.config.FleetConfig;
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||||
import dev.ltms.fleet.msg.LeadMailbox;
|
import dev.ltms.fleet.msg.LeadMailbox;
|
||||||
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
import org.slf4j.LoggerFactory;
|
|
||||||
|
|
||||||
import java.util.List;
|
|
||||||
import java.util.Map;
|
import java.util.Map;
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.*;
|
import static org.junit.jupiter.api.Assertions.*;
|
||||||
@@ -38,7 +36,7 @@ class FleetdLeadMailboxSelectionTest {
|
|||||||
boolean unreachable;
|
boolean unreachable;
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public LeadMailbox open(String uri, String selfCoordId, int prefetch) {
|
public LeadChannelHandle open(String uri, String selfCoordId, int prefetch) {
|
||||||
this.offeredUri = uri;
|
this.offeredUri = uri;
|
||||||
this.offeredSelfId = selfCoordId;
|
this.offeredSelfId = selfCoordId;
|
||||||
this.offeredPrefetch = prefetch;
|
this.offeredPrefetch = prefetch;
|
||||||
@@ -52,29 +50,22 @@ class FleetdLeadMailboxSelectionTest {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
private static ListAppender<ILoggingEvent> captureFleetdLogs() {
|
private static String joined(CapturedLog captured, Level level) {
|
||||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
return captured.events().stream().filter(e -> e.getLevel() == level)
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.start();
|
|
||||||
logger.addAppender(appender);
|
|
||||||
return appender;
|
|
||||||
}
|
|
||||||
|
|
||||||
private static String joined(ListAppender<ILoggingEvent> appender, Level level) {
|
|
||||||
return appender.list.stream().filter(e -> e.getLevel() == level)
|
|
||||||
.map(ILoggingEvent::getFormattedMessage).reduce("", (a, b) -> a + "\n" + b);
|
.map(ILoggingEvent::getFormattedMessage).reduce("", (a, b) -> a + "\n" + b);
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void noCoordinatorBlockLeavesTheFeatureOffSilently() {
|
void noCoordinatorBlockLeavesTheFeatureOffSilently() {
|
||||||
var appender = captureFleetdLogs();
|
try (var captured = CapturedLog.of(Fleetd.class)) {
|
||||||
var opener = new RecordingOpener();
|
var opener = new RecordingOpener();
|
||||||
|
|
||||||
assertNull(Fleetd.openLeadMailbox(null, Map.of(), opener));
|
assertNull(Fleetd.openLeadMailbox(null, Map.of(), opener));
|
||||||
|
|
||||||
assertNull(opener.offeredUri, "nothing configured means nothing is opened");
|
assertNull(opener.offeredUri, "nothing configured means nothing is opened");
|
||||||
assertEquals("", joined(appender, Level.WARN),
|
assertEquals("", joined(captured, Level.WARN),
|
||||||
"an opt-in feature nobody asked for must not warn on every boot");
|
"an opt-in feature nobody asked for must not warn on every boot");
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -114,31 +105,33 @@ class FleetdLeadMailboxSelectionTest {
|
|||||||
|
|
||||||
@Test
|
@Test
|
||||||
void warnsAndStaysOffWhenSelfIdIsMissing() {
|
void warnsAndStaysOffWhenSelfIdIsMissing() {
|
||||||
var appender = captureFleetdLogs();
|
try (var captured = CapturedLog.of(Fleetd.class)) {
|
||||||
var opener = new RecordingOpener();
|
var opener = new RecordingOpener();
|
||||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, null, null, null);
|
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, null, null, null);
|
||||||
|
|
||||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener));
|
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener));
|
||||||
|
|
||||||
assertNull(opener.offeredUri, "a mailbox with no owning coord-id has no queue to declare");
|
assertNull(opener.offeredUri, "a mailbox with no owning coord-id has no queue to declare");
|
||||||
String warns = joined(appender, Level.WARN);
|
String warns = joined(captured, Level.WARN);
|
||||||
assertTrue(warns.contains("coordinator.selfId"), () -> "say which key is missing: " + warns);
|
assertTrue(warns.contains("coordinator.selfId"), () -> "say which key is missing: " + warns);
|
||||||
assertFalse(warns.contains(SECRET), () -> "the URI's password must never be logged: " + warns);
|
assertFalse(warns.contains(SECRET), () -> "the URI's password must never be logged: " + warns);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void warnsAndStaysOffWhenTheBrokerIsUnreachableAtBoot() {
|
void warnsAndStaysOffWhenTheBrokerIsUnreachableAtBoot() {
|
||||||
var appender = captureFleetdLogs();
|
try (var captured = CapturedLog.of(Fleetd.class)) {
|
||||||
var opener = new RecordingOpener();
|
var opener = new RecordingOpener();
|
||||||
opener.unreachable = true;
|
opener.unreachable = true;
|
||||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null, null);
|
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null, null);
|
||||||
|
|
||||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener),
|
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener),
|
||||||
"a down coordination broker turns the feature off; it must never take the daemon down");
|
"a down coordination broker turns the feature off; it must never take the daemon down");
|
||||||
|
|
||||||
String warns = joined(appender, Level.WARN);
|
String warns = joined(captured, Level.WARN);
|
||||||
assertTrue(warns.contains("coord.example"), () -> "name the host that failed: " + warns);
|
assertTrue(warns.contains("coord.example"), () -> "name the host that failed: " + warns);
|
||||||
assertFalse(warns.contains(SECRET), () -> "with credentials stripped: " + warns);
|
assertFalse(warns.contains(SECRET), () -> "with credentials stripped: " + warns);
|
||||||
assertTrue(warns.contains("Connection refused"), () -> "and the real reason: " + warns);
|
assertTrue(warns.contains("Connection refused"), () -> "and the real reason: " + warns);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,407 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
|
import dev.ltms.fleet.lead.LeadLauncher;
|
||||||
|
import dev.ltms.fleet.lead.LeadRollover;
|
||||||
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import io.javalin.Javalin;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
import java.util.LinkedHashMap;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.Executors;
|
||||||
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
import static org.junit.jupiter.api.Assertions.fail;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #612 B3 — replaces {@code FleetdLeadRolloverWiringTest} (fleetd #480). That class was a
|
||||||
|
* source-text test scraping {@code Fleetd.java} (now {@code FleetdAssembly.java}, moved there by
|
||||||
|
* fleetd #612 Unit A) with three methods: {@code unrelatedAnchorStillPresent} (a scaffold anchor,
|
||||||
|
* not an independent claim — needs no replacement of its own), {@code
|
||||||
|
* mainStillCallsTheLeadRolloverFactory} (the call-site pin replaced by {@link
|
||||||
|
* #assembledLeadRolloverEndsTheOldPaneThroughTheRealHerdrRouter}), and {@code
|
||||||
|
* factoryGatesOnConfigPresence} (the absent-config claim replaced by {@link
|
||||||
|
* #absentLeadRolloverConfigMeansNoRolloverIsBuilt} — a claim this ticket found was NOT actually
|
||||||
|
* covered behaviourally anywhere else: {@code LeadRolloverTest}'s only related assertion is
|
||||||
|
* vacuous, {@code assertNull(null)}, and never calls the real factory).
|
||||||
|
*
|
||||||
|
* <p><strong>fleetd #612 B3 correction (ticket comment 17553):</strong> the first version of this
|
||||||
|
* test configured a single shared {@link FakeHerdr} for both the lead and member herdr sockets.
|
||||||
|
* {@code FleetdAssembly.java:140-142} falls back to {@code memberHerdr = herdr} whenever no
|
||||||
|
* distinct {@code memberHerdrSocket} is configured, so with one fake, {@code
|
||||||
|
* router.leadAgents()} and {@code router.memberAgents()} wrapped the identical client — a
|
||||||
|
* mutation swapping {@code Fleetd.leadRollover(cfg, router.leadAgents(), config, leads)} for
|
||||||
|
* {@code ..., router.memberAgents(), ...} at {@code FleetdAssembly.java:408} was therefore
|
||||||
|
* invisible to this test, even though the two are genuinely different daemons in production. This
|
||||||
|
* version configures two distinct sockets and two distinct {@link FakeHerdr} instances (the same
|
||||||
|
* pattern {@code FleetdAssemblyConnectionIdentityTest}, fleetd #612 B2, already uses to separate
|
||||||
|
* lead from member) and asserts the roll's {@code pane.close} call lands on the LEAD fake and
|
||||||
|
* never on the MEMBER one.
|
||||||
|
*/
|
||||||
|
class FleetdLeadRolloverAssemblyTest {
|
||||||
|
|
||||||
|
private static final Path LEAD_SOCKET = Path.of("/fake/lead-herdr.sock");
|
||||||
|
private static final Path MEMBER_SOCKET = Path.of("/fake/member-herdr.sock");
|
||||||
|
|
||||||
|
/** Keys {@code connectHerdr} by socket path so the lead and member daemons can be two
|
||||||
|
* DIFFERENT {@link FakeHerdr}s — same shape as B2's {@code FleetdAssemblyConnectionIdentityTest
|
||||||
|
* .TwoHerdrResourcePorts}. */
|
||||||
|
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||||
|
|
||||||
|
final Map<Path, HerdrClient> herdrsBySocket = new LinkedHashMap<>();
|
||||||
|
final SentinelReplyInbox replyInbox = new SentinelReplyInbox();
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Map<String, String> environment() {
|
||||||
|
return Map.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public HerdrClient connectHerdr(Path socketPath) {
|
||||||
|
HerdrClient client = herdrsBySocket.get(socketPath);
|
||||||
|
if (client == null) {
|
||||||
|
throw new IllegalStateException("no fake herdr registered for socket " + socketPath);
|
||||||
|
}
|
||||||
|
return client;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||||
|
return (uri, prefetch) -> replyInbox;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||||
|
return (uri, selfCoordId, prefetch) -> {
|
||||||
|
throw new UnsupportedOperationException(
|
||||||
|
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier nanoClock() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public LongSupplier wallClockNanos() {
|
||||||
|
return System::nanoTime;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public ScheduledExecutorService newScheduler(String purpose) {
|
||||||
|
return Executors.newSingleThreadScheduledExecutor();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void addShutdownHook(Runnable hook) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void startHttp(Javalin app, String host, int port) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Runnable herdrPollWait() {
|
||||||
|
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||||
|
return () -> {
|
||||||
|
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class SentinelReplyInbox implements ReplyInbox, AutoCloseable {
|
||||||
|
@Override
|
||||||
|
public void own(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void release(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void publish(String target, String msgId, String content) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<InboxMessage> peek(String target) {
|
||||||
|
return List.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean ack(String target, String msgId) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetConfig writeConfig(Path dir, Path leadCwd) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
herdrSocket: "%s"
|
||||||
|
memberHerdrSocket: "%s"
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
broker:
|
||||||
|
uri: "amqp://fake-test-broker/vh"
|
||||||
|
fleet:
|
||||||
|
leaders:
|
||||||
|
opus:
|
||||||
|
tab: "lead: opus"
|
||||||
|
cwd: "%s"
|
||||||
|
leadRollover:
|
||||||
|
handoverPath: handover.md
|
||||||
|
requireOperatorConfirm: false
|
||||||
|
""".formatted(LEAD_SOCKET, MEMBER_SOCKET, leadCwd.toString()));
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Unlike {@link #writeConfig}, this names a {@code profile:} for the lead and declares it
|
||||||
|
* under {@code profiles:}, so {@code LeadLauncher#relaunch} can actually start a fresh agent
|
||||||
|
* instead of refusing with "names no profile". {@code relaunchReadySeconds} is cut to 2s so
|
||||||
|
* the recognition wait (expected to time out — see the test) does not cost real test seconds.
|
||||||
|
*/
|
||||||
|
private static FleetConfig writeConfigWithRelaunchableLead(Path dir, Path leadCwd) throws Exception {
|
||||||
|
Path f = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(f, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
herdrSocket: "%s"
|
||||||
|
memberHerdrSocket: "%s"
|
||||||
|
idleSleepGuard:
|
||||||
|
enabled: false
|
||||||
|
broker:
|
||||||
|
uri: "amqp://fake-test-broker/vh"
|
||||||
|
fleet:
|
||||||
|
leaders:
|
||||||
|
opus:
|
||||||
|
tab: "lead: opus"
|
||||||
|
cwd: "%s"
|
||||||
|
profile: opus
|
||||||
|
profiles:
|
||||||
|
opus:
|
||||||
|
subscription: true
|
||||||
|
argv: ["ccs", "opus"]
|
||||||
|
leadRollover:
|
||||||
|
handoverPath: handover.md
|
||||||
|
requireOperatorConfirm: false
|
||||||
|
relaunchReadySeconds: 2
|
||||||
|
""".formatted(LEAD_SOCKET, MEMBER_SOCKET, leadCwd.toString()));
|
||||||
|
return FleetConfig.load(f);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("[BEHAVIOURAL] the real assembled LeadRollover runs the open/confirm/continuation "
|
||||||
|
+ "sequence through the real herdr router — ending the old pane, then giving up once it "
|
||||||
|
+ "never reports gone")
|
||||||
|
void assembledLeadRolloverEndsTheOldPaneThroughTheRealHerdrRouter(@TempDir Path dir) throws Exception {
|
||||||
|
Path leadCwd = dir.resolve("lead-workspace");
|
||||||
|
Files.createDirectories(leadCwd);
|
||||||
|
FleetConfig cfg = writeConfig(dir, leadCwd);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||||
|
// Two DISTINCT fakes — one per configured socket — so leadAgents()/memberAgents() wrap
|
||||||
|
// genuinely different clients, exactly like production when memberHerdrSocket is set.
|
||||||
|
FakeHerdr lead = new FakeHerdr();
|
||||||
|
lead.withTab("w2", "w2:t7", "lead: opus");
|
||||||
|
FakeHerdr member = new FakeHerdr();
|
||||||
|
ports.herdrsBySocket.put(LEAD_SOCKET, lead);
|
||||||
|
ports.herdrsBySocket.put(MEMBER_SOCKET, member);
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
|
||||||
|
LeadRollover rollover = runtime.mcp().leadRollover();
|
||||||
|
assertNotNull(rollover, "leadRollover: is present in this test's config, so "
|
||||||
|
+ "FleetdAssembly.assembleAndStart must have built a real LeadRollover through the "
|
||||||
|
+ "Fleetd.leadRollover(...) call site — a mutation to `LeadRollover leadRollover = "
|
||||||
|
+ "null;` at that call site can never pass this");
|
||||||
|
|
||||||
|
// assembleAndStart's own boot work (the orphan-worker reap) makes a real call on the
|
||||||
|
// member daemon before the roll ever starts. Clear it here so the assertion below measures
|
||||||
|
// only what the roll itself does, not what daemon startup does.
|
||||||
|
member.calls.clear();
|
||||||
|
|
||||||
|
LeadRollover.PendingRollover pending = rollover.open("term_a", "fleetd #612 B3 test");
|
||||||
|
String expectedHandoverPath = leadCwd.resolve("handover.md").normalize().toString();
|
||||||
|
assertEquals(expectedHandoverPath, pending.handoverPath());
|
||||||
|
|
||||||
|
// Ensure the handover file's mtime lands strictly AFTER open()'s requestedAtMillis —
|
||||||
|
// LeadRollover.checkHandover refuses on mtime <= requestedAt (HANDOVER_STALE).
|
||||||
|
Thread.sleep(50);
|
||||||
|
Files.writeString(Path.of(pending.handoverPath()), "handover content for fleetd #612 B3");
|
||||||
|
|
||||||
|
LeadRollover.RollDecision decision = rollover.confirm("term_a", pending.token(), true);
|
||||||
|
assertTrue(decision.accepted(), "confirm() must approve: requireOperatorConfirm is false, "
|
||||||
|
+ "the caller terminal matches open()'s, and the handover file exists, is non-empty "
|
||||||
|
+ "and fresh — got: " + decision);
|
||||||
|
|
||||||
|
// The production LeadRollover constructor always runs the post-confirm continuation on a
|
||||||
|
// real virtual thread (see Fleetd.leadRollover, which never passes the package-private test
|
||||||
|
// constructor), so this polls the real FleetMcp.leadRollover() instance's status(token)
|
||||||
|
// until the real continuation finishes.
|
||||||
|
LeadRollover.RollStatus status = pollUntilTerminal(rollover, pending.token());
|
||||||
|
|
||||||
|
// FakeHerdr's pane.get is a fixed canned response that never reports a pane as gone, so the
|
||||||
|
// real router's death poll runs out its whole budget and the roll stops here — proving the
|
||||||
|
// real teardown call landed on the real LEAD pane without ever reaching a relaunch or a send.
|
||||||
|
assertEquals(LeadRollover.RollState.OLD_PANE_NEVER_DIED, status.state(),
|
||||||
|
"the old pane never reports gone against this fake, so the roll must stop with "
|
||||||
|
+ "OLD_PANE_NEVER_DIED rather than ever relaunching or sending anything — "
|
||||||
|
+ "detail: " + status.detail());
|
||||||
|
|
||||||
|
// Prove the real herdr router actually closed the real LEAD pane — this is the one thing a
|
||||||
|
// source-text pin on the call site could never show.
|
||||||
|
boolean closedOldPane = lead.calls.stream()
|
||||||
|
.anyMatch(c -> c.method().equals("pane.close")
|
||||||
|
&& c.params() instanceof Map<?, ?> m && "w2:p7".equals(m.get("pane_id")));
|
||||||
|
assertTrue(closedOldPane, "endOldSession must close the real old pane (w2:p7) through the "
|
||||||
|
+ "real LEAD herdr client, got calls: " + lead.calls);
|
||||||
|
|
||||||
|
// No agent.prompt is ever sent on this path: the roll stops at the pane-death wait, strictly
|
||||||
|
// before the relaunch and the final send step.
|
||||||
|
List<FakeHerdr.Call> prompts = lead.calls.stream()
|
||||||
|
.filter(c -> c.method().equals("agent.prompt"))
|
||||||
|
.toList();
|
||||||
|
assertTrue(prompts.isEmpty(), "a roll that stops at OLD_PANE_NEVER_DIED must never reach the "
|
||||||
|
+ "send step, got agent.prompt call(s) on the LEAD daemon: " + prompts);
|
||||||
|
|
||||||
|
// fleetd #612 B3 correction: prove the roll never touches the MEMBER daemon. A mutation
|
||||||
|
// swapping router.leadAgents() for router.memberAgents() at the real call site would move
|
||||||
|
// the pane.close call above onto `member` instead, which this assertion catches — the thing
|
||||||
|
// the single-fake version of this test could never see, because both wrapped the same client.
|
||||||
|
assertTrue(member.calls.isEmpty(), "the roll must be wired to the LEAD daemon only — got "
|
||||||
|
+ member.calls.size() + " call(s) recorded on the MEMBER daemon since the roll began: "
|
||||||
|
+ member.calls);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Exercises the relaunch site {@link #assembledLeadRolloverEndsTheOldPaneThroughTheRealHerdrRouter}
|
||||||
|
* never reaches: with the old pane confirmed gone, the roll relaunches a fresh lead, and
|
||||||
|
* {@code bootstrapText} must reach it even though recognition times out (FakeHerdr's
|
||||||
|
* {@code tab.list} is a fixed canned response that never reflects the relaunch's own
|
||||||
|
* {@code tab.rename}, so the fresh terminal is never recognised as a live lead). Same
|
||||||
|
* dual-socket shape as the sibling test: two distinct {@link FakeHerdr} instances, so a
|
||||||
|
* {@code bootstrapText} send wired to the wrong daemon is visible.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
@DisplayName("[BEHAVIOURAL] bootstrapText reaches the fresh LEAD terminal even when recognition "
|
||||||
|
+ "times out, and the MEMBER daemon never sees it")
|
||||||
|
void bootstrapTextReachesTheFreshLeadTerminalEvenWhenRecognitionTimesOut(@TempDir Path dir) throws Exception {
|
||||||
|
Path leadCwd = dir.resolve("lead-workspace");
|
||||||
|
Files.createDirectories(leadCwd);
|
||||||
|
FleetConfig cfg = writeConfigWithRelaunchableLead(dir, leadCwd);
|
||||||
|
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||||
|
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||||
|
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||||
|
FakeHerdr lead = new FakeHerdr();
|
||||||
|
lead.withTab("w2", "w2:t7", "lead: opus");
|
||||||
|
// Lets the old pane (w2:p7) report gone once pane.close actually reaches it, so the roll
|
||||||
|
// proceeds to relaunch instead of stopping at OLD_PANE_NEVER_DIED.
|
||||||
|
lead.paneGoneAfterClose("w2:p7");
|
||||||
|
FakeHerdr member = new FakeHerdr();
|
||||||
|
ports.herdrsBySocket.put(LEAD_SOCKET, lead);
|
||||||
|
ports.herdrsBySocket.put(MEMBER_SOCKET, member);
|
||||||
|
|
||||||
|
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||||
|
|
||||||
|
LeadRollover rollover = runtime.mcp().leadRollover();
|
||||||
|
assertNotNull(rollover, "leadRollover: is present in this test's config, so a real "
|
||||||
|
+ "LeadRollover must have been built");
|
||||||
|
|
||||||
|
LeadRollover.PendingRollover pending = rollover.open("term_a", "bootstrapText relaunch test");
|
||||||
|
Thread.sleep(50);
|
||||||
|
Files.writeString(Path.of(pending.handoverPath()), "handover content for bootstrapText test");
|
||||||
|
|
||||||
|
LeadRollover.RollDecision decision = rollover.confirm("term_a", pending.token(), true);
|
||||||
|
assertTrue(decision.accepted(), "confirm() must approve — got: " + decision);
|
||||||
|
|
||||||
|
LeadRollover.RollStatus status = pollUntilTerminal(rollover, pending.token());
|
||||||
|
assertEquals(LeadRollover.RollState.RELAUNCH_NOT_RECOGNISED, status.state(),
|
||||||
|
"the fresh terminal is never recognised against this fake's static tab.list, so the "
|
||||||
|
+ "roll must reach RELAUNCH_NOT_RECOGNISED — not an earlier failure state and "
|
||||||
|
+ "not ROLLED — detail: " + status.detail());
|
||||||
|
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
List<FakeHerdr.Call> leadPrompts = lead.calls.stream()
|
||||||
|
.filter(c -> c.method().equals("agent.prompt"))
|
||||||
|
.toList();
|
||||||
|
assertEquals(1, leadPrompts.size(), "exactly one bootstrapText send is expected, on the LEAD "
|
||||||
|
+ "daemon, once recognition gives up — got: " + leadPrompts);
|
||||||
|
Object text = ((Map<String, Object>) leadPrompts.get(0).params()).get("text");
|
||||||
|
assertTrue(text instanceof String && ((String) text).contains("handover.md"),
|
||||||
|
"the send must be bootstrapText naming the resolved handover path, got: " + text);
|
||||||
|
|
||||||
|
// Scoped to agent.prompt specifically, not every MEMBER call: the orphan-worker reap also
|
||||||
|
// talks to the MEMBER daemon once, unconditionally, at daemon boot — unrelated to this roll.
|
||||||
|
List<FakeHerdr.Call> memberPrompts = member.calls.stream()
|
||||||
|
.filter(c -> c.method().equals("agent.prompt"))
|
||||||
|
.toList();
|
||||||
|
assertTrue(memberPrompts.isEmpty(), "bootstrapText must never be sent to the MEMBER daemon, "
|
||||||
|
+ "got: " + memberPrompts);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static LeadRollover.RollStatus pollUntilTerminal(LeadRollover rollover, String token)
|
||||||
|
throws InterruptedException {
|
||||||
|
long deadline = System.nanoTime() + java.util.concurrent.TimeUnit.SECONDS.toNanos(15);
|
||||||
|
while (System.nanoTime() < deadline) {
|
||||||
|
LeadRollover.RollStatus status = rollover.status(token);
|
||||||
|
if (status.state() != LeadRollover.RollState.PENDING
|
||||||
|
&& status.state() != LeadRollover.RollState.IN_PROGRESS) {
|
||||||
|
return status;
|
||||||
|
}
|
||||||
|
Thread.sleep(50);
|
||||||
|
}
|
||||||
|
fail("the real continuation did not reach a terminal state within 10s — last status: "
|
||||||
|
+ rollover.status(token));
|
||||||
|
throw new AssertionError("unreachable");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("[BEHAVIOURAL] Fleetd.leadRollover(...) returns null when leadRollover: is absent "
|
||||||
|
+ "from config — the opt-in gate FleetdLeadRolloverWiringTest's "
|
||||||
|
+ "factoryGatesOnConfigPresence pinned by source text alone")
|
||||||
|
void absentLeadRolloverConfigMeansNoRolloverIsBuilt(@TempDir Path dir) throws Exception {
|
||||||
|
Path yaml = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(yaml, """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
""");
|
||||||
|
ConfigRef config = new ConfigRef(yaml, FleetConfig.load(yaml));
|
||||||
|
AgentControl agents = new AgentControl(new FakeHerdr());
|
||||||
|
WorkspaceControl spaces = new WorkspaceControl(new FakeHerdr());
|
||||||
|
LeadLauncher launcher = new LeadLauncher(agents, spaces, config.get());
|
||||||
|
|
||||||
|
LeadRollover rollover = Fleetd.leadRollover(config.get(), agents, spaces, launcher, config, Map::of);
|
||||||
|
|
||||||
|
assertNull(rollover, "leadRollover: is absent from this config, so the factory's opt-in "
|
||||||
|
+ "gate (`if (cfg.leadRollover() == null) return null;`) must fire and no "
|
||||||
|
+ "LeadRollover must be constructed at all");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,89 +0,0 @@
|
|||||||
package dev.ltms.fleet;
|
|
||||||
|
|
||||||
import java.nio.file.Files;
|
|
||||||
import java.nio.file.Path;
|
|
||||||
import org.junit.jupiter.api.DisplayName;
|
|
||||||
import org.junit.jupiter.api.Test;
|
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* fleetd #480 Unit A, hard requirement 6: pin {@code Fleetd.main}'s construction of {@link
|
|
||||||
* dev.ltms.fleet.lead.LeadRollover} with a source-text assertion, mirroring {@code
|
|
||||||
* FleetdCompletionResolverWiringTest}'s pattern — five log-only reporters in {@code Fleetd.main}
|
|
||||||
* already survived mutation batteries this exact way (fleetd #415's extraction antidote note).
|
|
||||||
*
|
|
||||||
* <p>What this class still covers, and what it never claimed to. {@code LeadRolloverTest}
|
|
||||||
* constructs its own {@code LeadRollover} directly (as every prior test of an extracted factory
|
|
||||||
* does) with a hand-built lookup, so a mutation that deletes the {@code leadRollover(...)} call
|
|
||||||
* from {@code main} — or replaces one of its arguments with something that still compiles, e.g.
|
|
||||||
* {@code router.leadAgents()} swapped for {@code null}, or the whole assignment swapped for a bare
|
|
||||||
* {@code null} literal — leaves every behavioural test green. This is a plain string read, guarded
|
|
||||||
* by an unrelated anchor assertion so a broken or empty file read cannot pass as a real change.
|
|
||||||
*
|
|
||||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
|
||||||
* LeadRollover} and never runs {@code main}. It pins the {@code leadRollover(...)} CALL SITE's
|
|
||||||
* argument list — that {@code main} still passes {@code leads} at all — never what the factory
|
|
||||||
* DOES with that argument once inside its own body.
|
|
||||||
*
|
|
||||||
* <p><b>Correction (fleetd #480 relative-handover-path follow-up): that gap used to be real, and
|
|
||||||
* now is not — but not here.</b> This class's javadoc previously claimed "no behavioural test can
|
|
||||||
* catch this wiring dropping out" for the whole factory, including the lambda {@code
|
|
||||||
* leadRollover(...)} builds internally (terminal → lead name → {@code Leader.cwd()}). That claim
|
|
||||||
* was proven true at the time — mutating that lambda's body to {@code String leadName = null;}
|
|
||||||
* (always "no lead found", which silently reintroduces the daemon-cwd bug this ticket fixes) left
|
|
||||||
* the full suite green, {@code Tests run: 1669, Failures: 0}. It is no longer true: {@code
|
|
||||||
* FleetdLeadRolloverWorkspaceLookupTest} now calls {@code Fleetd.leadRollover(...)} directly with a
|
|
||||||
* real {@link dev.ltms.fleet.config.ConfigRef} built from a temp {@code fleetd.yaml}, and fails
|
|
||||||
* against that exact one-line mutation. So: THIS class still covers only the call site's argument
|
|
||||||
* list; {@code FleetdLeadRolloverWorkspaceLookupTest} is what now covers the lambda's body. Neither
|
|
||||||
* one subsumes the other — keep both.
|
|
||||||
*/
|
|
||||||
class FleetdLeadRolloverWiringTest {
|
|
||||||
|
|
||||||
private static String fleetdSource() throws Exception {
|
|
||||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
@DisplayName("[SOURCE TEXT] unrelated anchor: Fleetd.java still declares the Fleetd class")
|
|
||||||
void unrelatedAnchorStillPresent() throws Exception {
|
|
||||||
// Guards the two assertions below: without this, a bad read (empty string, wrong file,
|
|
||||||
// truncated file) could vacuously fail to contain the leadRollover(...) call too, and a
|
|
||||||
// test that only asserts "contains X" would report a false pass for the wrong reason if X
|
|
||||||
// happened to match. Asserting an unrelated, structurally distant string first proves the
|
|
||||||
// read actually pulled real file content.
|
|
||||||
String source = fleetdSource();
|
|
||||||
assertTrue(source.contains("public final class Fleetd"),
|
|
||||||
"sanity anchor failed — the file read did not return real Fleetd.java source; the "
|
|
||||||
+ "leadRollover(...) wiring assertions below cannot be trusted until this passes");
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
@DisplayName("[SOURCE TEXT] main still constructs LeadRollover via the leadRollover(...) factory, exactly as heartbeat is constructed")
|
|
||||||
void mainStillCallsTheLeadRolloverFactory() throws Exception {
|
|
||||||
String source = fleetdSource();
|
|
||||||
assertTrue(source.contains(
|
|
||||||
"LeadRollover leadRollover = leadRollover(cfg, router.leadAgents(), config, leads);"),
|
|
||||||
"Fleetd.main must still assign `LeadRollover leadRollover = leadRollover(cfg, "
|
|
||||||
+ "router.leadAgents(), config, leads);`. Dropping this call, or swapping one of "
|
|
||||||
+ "its arguments for something that still compiles (e.g. null in place of "
|
|
||||||
+ "router.leadAgents()), leaves every behavioural test green — this source check is "
|
|
||||||
+ "what must go red instead. fleetd #480 correction 2 deliberately dropped "
|
|
||||||
+ "primaryRegistry from this call — see LeadRollover's class javadoc for why a "
|
|
||||||
+ "single-slot lookup was wrong here. The fleetd #480 relative-handover-path "
|
|
||||||
+ "follow-up added `leads` (terminal → lead name) so the factory can resolve a "
|
|
||||||
+ "relative handoverPath against the calling lead's own workspace.");
|
|
||||||
}
|
|
||||||
|
|
||||||
@Test
|
|
||||||
@DisplayName("[SOURCE TEXT] the leadRollover(...) factory itself gates construction on cfg.leadRollover() != null")
|
|
||||||
void factoryGatesOnConfigPresence() throws Exception {
|
|
||||||
String source = fleetdSource();
|
|
||||||
assertTrue(source.contains("if (cfg.leadRollover() == null) {"),
|
|
||||||
"Fleetd.leadRollover(...) must refuse to construct a LeadRollover when the "
|
|
||||||
+ "leadRollover: block is absent — an upgraded daemon must never silently acquire "
|
|
||||||
+ "the ability to clear the lead's own pane. See LeadHeartbeatLoop's construction "
|
|
||||||
+ "gate (cfg.leadHeartbeat() != null) for the pattern this mirrors.");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -4,6 +4,8 @@ import dev.ltms.fleet.config.ConfigRef;
|
|||||||
import dev.ltms.fleet.config.FleetConfig;
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
import dev.ltms.fleet.herdr.AgentControl;
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
|
import dev.ltms.fleet.lead.LeadLauncher;
|
||||||
import dev.ltms.fleet.lead.LeadRollover;
|
import dev.ltms.fleet.lead.LeadRollover;
|
||||||
import org.junit.jupiter.api.DisplayName;
|
import org.junit.jupiter.api.DisplayName;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
@@ -54,6 +56,11 @@ class FleetdLeadRolloverWorkspaceLookupTest {
|
|||||||
return new AgentControl(new FakeHerdr());
|
return new AgentControl(new FakeHerdr());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** None of this class's tests reach the deferred continuation, so a plain fake is enough. */
|
||||||
|
private static LeadLauncher fakeLauncher(FleetConfig cfg) {
|
||||||
|
return new LeadLauncher(fakeAgents(), new WorkspaceControl(new FakeHerdr()), cfg);
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@DisplayName("[BEHAVIOURAL] Fleetd.leadRollover(...) resolves a relative handoverPath against "
|
@DisplayName("[BEHAVIOURAL] Fleetd.leadRollover(...) resolves a relative handoverPath against "
|
||||||
+ "the CALLING lead's configured cwd, not the daemon's own working directory")
|
+ "the CALLING lead's configured cwd, not the daemon's own working directory")
|
||||||
@@ -74,7 +81,8 @@ class FleetdLeadRolloverWorkspaceLookupTest {
|
|||||||
""".formatted(leadCwd.toString()));
|
""".formatted(leadCwd.toString()));
|
||||||
ConfigRef config = new ConfigRef(yaml, FleetConfig.load(yaml));
|
ConfigRef config = new ConfigRef(yaml, FleetConfig.load(yaml));
|
||||||
|
|
||||||
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(), config,
|
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(),
|
||||||
|
new WorkspaceControl(new FakeHerdr()), fakeLauncher(config.get()), config,
|
||||||
() -> Map.of("term_opus", "opus"));
|
() -> Map.of("term_opus", "opus"));
|
||||||
assertNotNull(rollover, "leadRollover: is present in the loaded config, so the factory "
|
assertNotNull(rollover, "leadRollover: is present in the loaded config, so the factory "
|
||||||
+ "must construct an object");
|
+ "must construct an object");
|
||||||
@@ -105,7 +113,8 @@ class FleetdLeadRolloverWorkspaceLookupTest {
|
|||||||
|
|
||||||
// No lead has been discovered yet — exactly the real shape of a lead the live tab scan
|
// No lead has been discovered yet — exactly the real shape of a lead the live tab scan
|
||||||
// has not yet scanned, or one with no fleet.leaders entry at all.
|
// has not yet scanned, or one with no fleet.leaders entry at all.
|
||||||
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(), config, Map::of);
|
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(),
|
||||||
|
new WorkspaceControl(new FakeHerdr()), fakeLauncher(config.get()), config, Map::of);
|
||||||
assertNotNull(rollover);
|
assertNotNull(rollover);
|
||||||
|
|
||||||
LeadRollover.PendingRollover pending = rollover.open("term_unknown", "test");
|
LeadRollover.PendingRollover pending = rollover.open("term_unknown", "test");
|
||||||
@@ -144,7 +153,8 @@ class FleetdLeadRolloverWorkspaceLookupTest {
|
|||||||
// below — exactly the natural mistake to make, since leads are discovered by a live tab
|
// below — exactly the natural mistake to make, since leads are discovered by a live tab
|
||||||
// scan that runs AFTER this factory is constructed at startup.
|
// scan that runs AFTER this factory is constructed at startup.
|
||||||
Map<String, String> liveLeadTerminals = new HashMap<>();
|
Map<String, String> liveLeadTerminals = new HashMap<>();
|
||||||
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(), config,
|
LeadRollover rollover = Fleetd.leadRollover(config.get(), fakeAgents(),
|
||||||
|
new WorkspaceControl(new FakeHerdr()), fakeLauncher(config.get()), config,
|
||||||
() -> liveLeadTerminals);
|
() -> liveLeadTerminals);
|
||||||
assertNotNull(rollover);
|
assertNotNull(rollover);
|
||||||
|
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user