Compare commits
202 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 2926cd1784 | |||
| b9c2cf69f4 | |||
| 8beae50fe7 | |||
| b2a58cb966 | |||
| b8b25cf74c | |||
| f159ca7d27 | |||
| 83f2aea60f | |||
| 002329adb5 | |||
| a49671ceb9 | |||
| 76672ff016 | |||
| 9379f92c23 | |||
| 21ff63b11d | |||
| 21844b54d7 | |||
| 4769481515 | |||
| bbbb4c1eb3 | |||
| 38c248e617 | |||
| 9debc0de27 | |||
| f34361b263 | |||
| be5ba22c75 | |||
| 85c90d440a | |||
| e97502d550 | |||
| aef14ff46e | |||
| cba516bda4 | |||
| ba51e0c6cc | |||
| 086c59848e | |||
| 0c10079755 | |||
| ece2091b53 | |||
| b3f917e6f5 | |||
| fa39a5f55e | |||
| d5128a1d35 | |||
| 61097e5cf0 | |||
| ef507bcd12 | |||
| 94f50e507a | |||
| 75b15086b0 | |||
| dab9645906 | |||
| e93b5f6512 | |||
| bfabe13e8f | |||
| 7e49c6eca2 | |||
| 11cbfa79b4 | |||
| c2c2746922 | |||
| 6ed70700a0 | |||
| 0087645da4 | |||
| 19cacf5b62 | |||
| 4dd12083ab | |||
| 30e21adec7 | |||
| 719b79f892 | |||
| 66e5247b6d | |||
| 1e41bd63b4 | |||
| 4887d03d88 | |||
| 7d5434455d | |||
| f71ee4926e | |||
| 282a2fc2b8 | |||
| f04e934b94 | |||
| 3fae35c357 | |||
| 18aecbfe67 | |||
| 5d75f72473 | |||
| e028a0ae54 | |||
| 2fa673d4c0 | |||
| 9020d01b40 | |||
| 1006805027 | |||
| 27aefbf9a0 | |||
| d42c2bc204 | |||
| 3916adc372 | |||
| fa97f598dd | |||
| ea9aa4fd77 | |||
| a2b8caf6b5 | |||
| b5ddbe5757 | |||
| d223a93039 | |||
| 96d8191149 | |||
| f0e7ac73d6 | |||
| e2fe861b4d | |||
| 279d6f5fbd | |||
| 34480cebef | |||
| 9d0bf14c46 | |||
| 5a3ab5764c | |||
| 3bad9f5785 | |||
| 8308c0b68f | |||
| 3b3063eb2b | |||
| df9086263d | |||
| eb568ff451 | |||
| ac790e4cce | |||
| 6b5f3f472f | |||
| 38dec72152 | |||
| 0e8bfb74fc | |||
| 51f7b0a3ca | |||
| 21c539f22e | |||
| bd2774b5f1 | |||
| c50f5b2d61 | |||
| 01a840cc14 | |||
| c4deef08be | |||
| c796eac09c | |||
| e897e5257b | |||
| 2afa3652bb | |||
| 80092ff359 | |||
| 9d37f3aa29 | |||
| eaf89abaf6 | |||
| d895f02bc1 | |||
| 43206cac2f | |||
| 8bba3a8184 | |||
| 5cf3ca9a89 | |||
| ac474981e4 | |||
| ba04b2359b | |||
| 2e5b63f6f6 | |||
| 3437d6313d | |||
| 743377d6cd | |||
| 5952d559c7 | |||
| 205ad823b0 | |||
| d654ccb818 | |||
| a89dcc9b7e | |||
| 0d7b4fb026 | |||
| 321d8dcbb5 | |||
| ef8c97871e | |||
| 26bafe824b | |||
| 838a701109 | |||
| e5eb3534c7 | |||
| 959c83534f | |||
| 31b028e860 | |||
| 7840e9adf6 | |||
| c3672f5472 | |||
| cf54aed451 | |||
| bbf68f3e3c | |||
| 776743cbe2 | |||
| 826e0aeb2a | |||
| c935b181dd | |||
| ee932fd85b | |||
| fe2e5ede34 | |||
| c325054242 | |||
| 7662e2d0c8 | |||
| 4877992a70 | |||
| 1c051c4e47 | |||
| adb7a67880 | |||
| 723fe494e9 | |||
| 0f08b93659 | |||
| 432c1d92d1 | |||
| 60b7e67b42 | |||
| 3fbd43fe3f | |||
| 32ebf065ac | |||
| 1178b3f684 | |||
| 2a434ced2f | |||
| cc919aa2b6 | |||
| 6417b0edd9 | |||
| 39c7ce76f3 | |||
| b40f477210 | |||
| 5c56cb347f | |||
| e694deace3 | |||
| 748367b7d6 | |||
| dcf5fb3be3 | |||
| f9d2ee2a2b | |||
| 866c7f2e9a | |||
| d9168de43e | |||
| c3fa1136d4 | |||
| cabcd87b66 | |||
| bf0e09b1a2 | |||
| e1eb50ce65 | |||
| 049e7d9d54 | |||
| ff3b49cd1e | |||
| 0373b6c41b | |||
| 445a45f6e1 | |||
| 966c58a3b8 | |||
| de026b8f8a | |||
| 97f6c33a45 | |||
| 735c837604 | |||
| 457dc0330d | |||
| a1a9015217 | |||
| 5a811a3695 | |||
| 1fdaa74eb3 | |||
| 7d4a4339c2 | |||
| 847e8bd3fa | |||
| f9fb387427 | |||
| a55079afbd | |||
| e18ad4723b | |||
| 6de8ac8972 | |||
| a052975420 | |||
| 6d82ca95a4 | |||
| 8067ee4ec4 | |||
| 3743789e8d | |||
| 388aba7632 | |||
| ad587eafa3 | |||
| 08ce9aef11 | |||
| ee5f8b932b | |||
| 4ac688b6d9 | |||
| a814d1ef00 | |||
| ea98856130 | |||
| a49e96835a | |||
| 63c19dcba7 | |||
| 2823349c8e | |||
| 6fc301d62c | |||
| bc99d64786 | |||
| 615af4ed0a | |||
| 045d229728 | |||
| d1fd5700f5 | |||
| 2d55b0b9a5 | |||
| d89ae94a2e | |||
| e6193c4098 | |||
| b66f0677ed | |||
| 23ada1981e | |||
| a22480c117 | |||
| 24f404f989 | |||
| a237fbff9d | |||
| 17af61e8dd | |||
| 6af87b6ad6 | |||
| fc655e78c2 |
@@ -39,6 +39,19 @@ a worker made all 59 of its edits in the primary's tree and never noticed.
|
||||
test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-toplevel)"
|
||||
```
|
||||
|
||||
**Never run `git stash` (or `git stash pop`/`apply`/`drop`).** Your worktree is isolated, but the
|
||||
stash is **not**: `refs/stash` is one stack shared by the primary's checkout and every other
|
||||
worker's worktree of this repo. Measured on 2026-09-04 — `git stash list` from a worker's worktree
|
||||
and from the primary's tree returned byte-identical output. So a `git stash` you run can be popped
|
||||
into someone else's tree, and a `git stash pop` you run can drop **another worker's** uncommitted
|
||||
edits on top of yours. This has already happened here: two workers were running in parallel and one
|
||||
of them had its in-progress edit silently overwritten by the other's stash.
|
||||
|
||||
The branch is your isolation, so use it instead. To set work aside, commit it on your own branch
|
||||
(`git commit -m "wip: ..."`) and carry on; to try something and back out, use
|
||||
`git diff > /tmp/<your-branch>.patch` then `git checkout -- <file>`. Both stay inside your worktree.
|
||||
If you find a stash entry you did not create, leave it alone and say so in your report.
|
||||
|
||||
## 2. Implement
|
||||
|
||||
- Implement exactly the scope the lead named. Keep the diff focused; note anything out of scope
|
||||
|
||||
@@ -188,6 +188,18 @@ must obey belongs in the charter, not here.
|
||||
- **This repo is the bridge.** The daemon is `fleetd`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
and the code behind the rules above is `mcp/FleetMcp` (tools), `auth/Authz` (the role table),
|
||||
`mcp/ConnectionIdentity` (connection→role), and `worker/*Launcher` (`REPLY_CHARTER`).
|
||||
- **`fleet_profiles`/`fleet_list` report two separate outage states, and they are not the same
|
||||
thing.** *Quarantined* (CB-578) means the backend told us it is out of capacity — a long,
|
||||
1800s-default cooldown. *Cooling off* (fleetd #201/#227) means a profile's credential threw two
|
||||
distinct backend errors (a non-exhaustion failure such as an HTTP 5xx) within 60 seconds — a
|
||||
short, fixed 60s cooldown, not configurable per profile. Each check runs independently, so a
|
||||
profile can show both at once. In the JSON: a cooling profile carries `credentialId` and
|
||||
`coolingOffForSeconds`; a quarantined profile carries `quarantinedForSeconds`; a profile hit by
|
||||
both carries all three fields, and either state alone already sets that profile's `free` to `0`.
|
||||
A `fleet_spawn` naming a cooling-off profile is refused before it ever reaches the backend
|
||||
adapter, with a message naming the credential and the remaining seconds ("cooling off after
|
||||
repeated backend errors") — distinct wording from a quarantine refusal, so don't conflate the
|
||||
two when reading a spawn failure.
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR) and
|
||||
`reviewer` (scoped review → one structured finding). Name one in every delegation.
|
||||
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
||||
@@ -195,11 +207,22 @@ must obey belongs in the charter, not here.
|
||||
`fleets-status` (report every fleet that shares one LavinMQ instance).
|
||||
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
||||
(a submodule with its own remote).
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, turn-done fallback —
|
||||
are diagrammed in `docs/MCP-Contract.md` **§6 only**. The rest of that page is a pre-build design
|
||||
doc whose tool names, parameter names and REST paths never caught up with the code, so do not use
|
||||
it as the tool reference (CB-609). Section 6 is kept out of this file because this file loads into
|
||||
every session's context.
|
||||
- **A provisioned worktree neutralizes `.mcp.json`, `opencode.json` and `.autoenv`** — the repo's
|
||||
committed copies would otherwise mount the primary's IDE and forge servers (fleetd #134). The
|
||||
worktree's copy of each is a stub, **not** the repo's real file, so a worker that reads one and
|
||||
reports what it found is reporting on the stub. The daemon logs a per-spawn summary, but the
|
||||
worker cannot see that log. From inside its own worktree a worker — or a lead debugging one —
|
||||
reads the list with `git config --worktree --get-all fleet.neutralizedConfig`, and the
|
||||
consequence with `git config --worktree --get fleet.neutralizedConfigNote`. Never brief a worker
|
||||
to edit one of these files: the edit cannot be committed, and it will not tell you so.
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, the turn-done
|
||||
fallback and status gating — are diagrammed in `docs/MCP-Contract.md`. That page is now flows
|
||||
only: its pre-build tool catalogue, parameter tables and REST paths were deleted rather than
|
||||
corrected, because a hand-maintained second copy of the tool surface is what drifted for a month
|
||||
while this line pointed every session at it (CB-609 / #114). **The live MCP schema is the tool
|
||||
reference**, with the intent→tool table above as the short form. `McpContractDocTest` fails if
|
||||
that page names a `fleet_*` tool the server does not register. The flows are kept out of this
|
||||
file because this file loads into every session's context.
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
|
||||
+20
-4
@@ -31,11 +31,27 @@ Environment=HERDR_SOCKET_PATH=%h/.config/herdr/herdr.sock
|
||||
# spawns, so this line decides whether the fleet can run a build at all. systemd does not source a
|
||||
# login shell, so without it the daemon — and every worker — gets a bare default with no JDK/Maven.
|
||||
Environment=PATH=/usr/lib/jvm/temurin-25-jdk/bin:/usr/share/maven/bin:/usr/local/bin:/usr/bin:/bin
|
||||
# Secrets are NOT set here — this file is committed. Put the API/worker tokens in a private
|
||||
# drop-in that systemd reads with restrictive permissions:
|
||||
# systemctl --user edit fleetd → [Service] / Environment=FLEETD_API_TOKEN=...
|
||||
# or point EnvironmentFile at a 0600 file:
|
||||
# Secrets are NOT set here — this file is committed. Put ALL three tokens in a private drop-in
|
||||
# that systemd reads with restrictive permissions. In `systemctl --user edit fleetd`, add:
|
||||
# [Service]
|
||||
# Environment=FLEETD_API_TOKEN=...
|
||||
# Environment=WORKER_GITEA_TOKEN=...
|
||||
# Environment=AI_GATEWAY_TOKEN=...
|
||||
# FLEETD_API_TOKEN protects fleetd's API. WORKER_GITEA_TOKEN lets members open pull requests; if
|
||||
# it is missing, fleetd still starts, but a member fails when it later tries to open a pull request.
|
||||
# AI_GATEWAY_TOKEN authenticates gateway profiles; if it is missing, fleetd still starts, but a
|
||||
# gateway profile later returns HTTP 401. Or, put the same three variables in a 0600 file and add:
|
||||
# EnvironmentFile=%h/.config/fleetd/env
|
||||
# After starting, check which of them actually resolved. The daemon reports every secret a
|
||||
# configured profile references, by name, never by value:
|
||||
# journalctl --user -u fleetd | grep 'startup secret'
|
||||
# A resolved one logs "startup secret NAME: set (profile 'x' tokenEnv)". A missing one logs
|
||||
# "startup secret NAME: MISSING" at WARN — and the daemon starts anyway, which is the whole
|
||||
# problem: without this grep the first sign is a member that cannot open a pull request, hours
|
||||
# later and in a different component.
|
||||
# Note what the report can and cannot tell you. It lists only names some profile actually
|
||||
# references (tokenEnv, gitTokenEnv, and the broker uriEnv). A secret nothing references is never
|
||||
# reported, because nothing needs it.
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=10s
|
||||
|
||||
@@ -0,0 +1,531 @@
|
||||
# CB-201 and CB-227 refinement
|
||||
|
||||
Date: 2026-09-03
|
||||
|
||||
## Decision
|
||||
|
||||
#201 and #227 are one delivery program, but they are not one implementation unit.
|
||||
|
||||
#201 has a real seam: `CompletionResolver` can publish a typed backend-error event only after its
|
||||
waiter resolution wins. #227 can consume that event without knowing any pane text. The classifier
|
||||
must land before the final #227 wiring. However, the policy engine, roster state, and lead nudge can
|
||||
be built in parallel with the classifier.
|
||||
|
||||
I propose five units. Units 1 to 4 own separate files and can run in parallel. Unit 5 owns all
|
||||
composition files and lands after them. It also depends on the #234 defect 2 fix named in the task.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
U1["Unit 1: typed backend-error classification"] --> U5["Unit 5: wire policy, spawn gate, and fleet views"]
|
||||
U2["Unit 2: credential outage policy"] --> U5
|
||||
U3["Unit 3: lead outage nudge"] --> U5
|
||||
U4["Unit 4: durable member outcome"] --> U5
|
||||
D234["#234 defect 2: fail-loud target resolution"] --> U5
|
||||
```
|
||||
|
||||
*Figure 1. Four file-disjoint foundations feed one composition unit.*
|
||||
|
||||
This split keeps `Fleetd.java` under one owner. It also keeps every other production file under one
|
||||
unit in this plan.
|
||||
|
||||
## Evidence checked in the current branch
|
||||
|
||||
I read both issue pages in full. Each page reports zero comments.
|
||||
|
||||
| Evidence | What the code says now |
|
||||
|---|---|
|
||||
| `inject/CompletionResolver.java:229-237` | A turn below two seconds fails before normal scrape classification. A matching fast backend error is therefore only a generic failure today. |
|
||||
| `inject/CompletionResolver.java:260-276` and `:332-367` | #211 already added raw-screen classification when `lastAssistantBlock` is empty. The “dead-code question” in #201 is stale on this branch. |
|
||||
| `inject/CompletionResolver.java:288-317` | Exhaustion wins before the hard-coded `API Error:` match. A backend error then goes through generic `fail(...)`. |
|
||||
| `inject/CompletionResolver.java:311-316` | The code admits that the pattern is a heuristic. A member report which quotes an API error may match it. |
|
||||
| `inject/CompletionResolver.java:449-467` | Startup coverage exists only for `exhaustedPattern`. |
|
||||
| `inject/ExhaustedPatternLookup.java:13-25` | The current lookup and explicit `none()` value are a good shape for the new classifier seam. |
|
||||
| `Fleetd.java:196-207` | One `BackendQuarantine` is shared by placement and the exhaustion sink. Its cooldown comes from `quarantineCooldownSeconds`. |
|
||||
| `Fleetd.java:322-363` | Pattern compilation, target-to-profile lookup, and the live `ExhaustionSink` are composed in `Fleetd.main`. The sink on this branch still ends in `.ifPresent(...)`. This plan assumes #234 replaces that silent path. |
|
||||
| `placement/BackendQuarantine.java:60-87` | A repeated exhaustion restarts one long quarantine. The store is credential-keyed and uses an injected monotonic clock. |
|
||||
| `member/CompositePeerLauncher.java:260-317` | Explicit and policy-selected spawns have separate gates. Both paths must learn about outage cool-off. |
|
||||
| `member/CompositePeerLauncher.java:347-379` | Exhaustion refusal already checks a credential for explicit spawns and filters policy candidates. Its error text says “exhausted”. |
|
||||
| `placement/PlacementContext.java:10-22` and `PlacementPolicyUtil.java:14-83` | Automatic placement has only one transient exclusion set named `quarantined`. Reusing it would make outage errors say “backend exhausted”. |
|
||||
| `mcp/FleetMcp.java:913-1025` | `fleet_list` sets `free: 0` and adds `credentialId` plus `quarantinedForSeconds` when quarantine is active. |
|
||||
| `session/MemberSession.java:51-59` | The roster has `DONE` and generic `FAILED`, but no backend-error state or stored reason. |
|
||||
| `session/SessionManager.java:695-773` | A normal boundary moves `BUSY` to `DONE`. A failure moves any non-released session to `FAILED`. The async completion resolver can race the `DONE` update. |
|
||||
| `session/SessionManager.java:648-687` | `rosterView` reports the session state, but it reports no terminal reason. |
|
||||
| `msg/MessageService.java:922-940` | CB-588 already nudges for every terminal async ticket, including failures. Current code would report failed tickets, but it would not report one correlated outage. |
|
||||
| `msg/ReplyPushLoop.java:20-48` | Replies, terminal tickets, and questions share one per-lead schedule. This prevents two push sources from injecting competing turns. |
|
||||
| `msg/ReplyPushLoop.java:305-395` | Each push entry point resolves the owning lead through `PrimaryRegistry`. Missing ownership is logged and the durable or pending item remains the backstop. |
|
||||
| `msg/ReplyPushLoop.java:496-547` | One tick builds one combined nudge. Pending items have separate reminder counts. |
|
||||
| `health/FleetHealthMonitor.java:91-143` | Health is a slow periodic observer of members and message-layer facts. It does not receive completion classifications. |
|
||||
| `health/FleetHealthMonitor.java:206-208` | `healthCoverage` means health enabled plus webhook configured. It does not describe lead-pane alerts. |
|
||||
| `Fleetd.java:465-486` | Health stays `detection-only` without the webhook notification setting. |
|
||||
|
||||
I also read the related unit tests for `CompletionResolver`, `ReplyPushLoop`, `BackendQuarantine`,
|
||||
`CompositePeerLauncher`, `PlacementPolicyUtil`, `SessionManager`, `MessageService`, and `FleetMcp`.
|
||||
|
||||
I did not inspect the in-progress #234 branch. I only used the two measured facts in the task. No
|
||||
peer architect was named, so I did not exchange a design with one.
|
||||
|
||||
## Required behaviour
|
||||
|
||||
The policy should use these first values:
|
||||
|
||||
- Threshold: **2** classified backend errors.
|
||||
- Window: **60 seconds**, measured from the first error to the second.
|
||||
- Cool-off: **60 seconds**, starting when the threshold is reached.
|
||||
- Correlation key: `credentialId`, never profile name and never error text.
|
||||
- Incident rule: one active incident per credential. Errors during its cool-off do not extend it and
|
||||
do not create more lead notices.
|
||||
- Rearm rule: after cool-off ends, two fresh errors are needed for another incident.
|
||||
|
||||
Two errors are the smallest threshold which protects the honest one-turn failure. A 60-second window
|
||||
fits the measured two-member outage. A 60-second cool-off blocks immediate repeat spawns without
|
||||
turning a short backend fault into the default 1,800-second exhaustion quarantine.
|
||||
|
||||
A single classified error still fails its send and marks its member `backend_error`. It does not
|
||||
cool a credential and does not send an outage notice. This is what “a single error changes nothing”
|
||||
must mean at the credential level. It cannot mean that the failed member still looks successful.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant R1 as Resolver for member A
|
||||
participant R2 as Resolver for member B
|
||||
participant P as Outage policy
|
||||
participant S as Spawn gate
|
||||
participant N as Lead push loop
|
||||
participant L as Lead pane
|
||||
|
||||
R1->>P: backend error for credential C
|
||||
Note over P: Count 1, no cool-off
|
||||
R2->>P: backend error for credential C within 60s
|
||||
P->>P: Start one 60s incident
|
||||
P->>S: Credential C is cooling off
|
||||
P->>N: Queue one incident notice
|
||||
N->>L: Inject when lead is idle, blocked, or done
|
||||
L->>S: Request another spawn on credential C
|
||||
S-->>L: Refuse and report remaining cool-off
|
||||
```
|
||||
|
||||
*Figure 2. The second independent classification creates the fleet-level event.*
|
||||
|
||||
Against the 2026-09-01 case, the second failed member would start cool-off. `fleet_list` would show
|
||||
zero free capacity and both members as `backend_error`. The push loop would inject one outage notice
|
||||
even if the lead had not polled either ticket yet. The design reports the outage. It does not recover
|
||||
uncommitted work from the members.
|
||||
|
||||
## Unit 1 — Typed backend-error classification
|
||||
|
||||
### Scope
|
||||
|
||||
Replace the direct hard-coded check inside `CompletionResolver` with a lookup and a sink. Keep the
|
||||
public send result as a failed send. The typed internal event is the seam #227 consumes.
|
||||
|
||||
The lookup returns the pattern for a target. The sink receives the target, matched line, and full
|
||||
failure reason. It fires only after `Rendezvous.resolveFailure(...)` wins for that exact captured
|
||||
waiter. This copies the race rule already used by `ExhaustionSink`.
|
||||
|
||||
The classifier must run in all three current paths:
|
||||
|
||||
1. a normal non-empty assistant block;
|
||||
2. the #211 raw scrape fallback;
|
||||
3. a turn inside `MIN_TURN_NANOS`, before it becomes a generic too-fast failure.
|
||||
|
||||
In every path, the order stays: stale-baseline guard, exhaustion, backend error, then generic
|
||||
failure or completion. A fast turn still fails when no configured pattern matches.
|
||||
|
||||
Keep `(?i)\bAPI Error\s*:` as a compatibility pattern for profiles without `errorPattern` until the
|
||||
operator config is updated. Do not call this full coverage. Startup reporting in Unit 5 must name
|
||||
profiles using this weaker legacy default.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/inject/BackendErrorPatternLookup.java`.
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/inject/BackendErrorSink.java`.
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/inject/CompletionResolver.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/inject/CompletionResolverTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. A target-specific error pattern matches a normal assistant block and resolves the send as failed.
|
||||
2. The same match calls `BackendErrorSink` exactly once after the waiter resolution wins.
|
||||
3. A late classification which loses to `fleet_reply` does not call the sink.
|
||||
4. An exhausted line that also matches the generic error pattern stays `BACKEND_EXHAUSTED`. It calls
|
||||
only `ExhaustionSink`.
|
||||
5. A raw pane with leading Terminal User Interface (TUI) chrome and no assistant marker still uses
|
||||
the #211 fallback and calls the backend-error sink.
|
||||
6. A matching error inside the two-second floor is typed and sent to the sink. A non-matching fast
|
||||
turn stays a generic failure.
|
||||
7. An unchanged delivery baseline which contains old backend-error text is suppressed. It never
|
||||
increments outage evidence.
|
||||
8. A non-match keeps the existing completion result and text.
|
||||
9. Constructors used by current callers keep compiling. They use the legacy default lookup and an
|
||||
explicit inert sink until Unit 5 supplies the production objects.
|
||||
10. Unit tests pass. The developer runs the focused test first, then `mvn clean install` from
|
||||
`fleetd/`.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 1 can run with Units 2, 3, and 4.
|
||||
|
||||
Unit 5 depends on its new lookup, sink, and constructor.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact classifier order in all three paths.
|
||||
- The focused test command and result.
|
||||
- The test which proves a losing waiter race does not publish an event.
|
||||
- The test which proves a fast matching failure is typed.
|
||||
- The final `mvn clean install` result.
|
||||
- Any constructor kept only for transition and where Unit 5 replaces it.
|
||||
|
||||
## Unit 2 — Credential outage policy
|
||||
|
||||
### Scope
|
||||
|
||||
Build a small credential-keyed state machine. It accepts already-classified backend-error events.
|
||||
It does not read pane text, profiles, sessions, or lead state.
|
||||
|
||||
Use an injected monotonic clock. A call records `credentialId`, target, and reason. It returns a new
|
||||
incident only on the threshold crossing. The incident contains a stable event id, credential id,
|
||||
the distinct affected targets, evidence count, window, and remaining cool-off.
|
||||
|
||||
This class owns both correlation and short cool-off. Keeping them together makes threshold crossing
|
||||
and the cool-off deadline one atomic state change.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/placement/BackendOutagePolicy.java`.
|
||||
- Add `fleetd/src/test/java/dev/ltms/fleet/placement/BackendOutagePolicyTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. One error creates no incident and no cool-off.
|
||||
2. Two errors for one credential within 60 seconds create exactly one incident and a 60-second
|
||||
cool-off.
|
||||
3. Two errors more than 60 seconds apart do not create an incident.
|
||||
4. The exact 60-second boundary has a pinned result. Use inclusive `<= 60s` so scheduler delay does
|
||||
not discard evidence at the boundary.
|
||||
5. Different credentials never share evidence.
|
||||
6. Different profiles which supply the same credential id do share evidence. The policy itself only
|
||||
sees the credential id.
|
||||
7. More errors during active cool-off do not extend its deadline and do not return another incident.
|
||||
8. After expiry, old evidence is cleared. Two fresh errors are needed to create the next incident.
|
||||
9. Remaining seconds round up, matching `BackendQuarantine` reporting.
|
||||
10. Concurrent second and third errors cannot return two incidents.
|
||||
11. The focused tests and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 2 can run with Units 1, 3, and 4.
|
||||
|
||||
Unit 5 depends on the policy API.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The state transition table and locking method.
|
||||
- The exact threshold, window, cool-off, and boundary rule.
|
||||
- The test which proves one incident under concurrent calls.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 3 — Lead outage nudge
|
||||
|
||||
### Scope
|
||||
|
||||
Add backend incidents as a fourth pending source in `ReplyPushLoop`. Do not create another scheduler
|
||||
or call `AgentControl.send` from `Fleetd`. The existing combined per-lead schedule is the control
|
||||
which prevents competing injected turns.
|
||||
|
||||
The entry point takes an incident id, affected worker targets, credential id, affected profile
|
||||
names, and remaining cool-off. It resolves distinct owning leads through `PrimaryRegistry`.
|
||||
|
||||
Each `(incidentId, lead)` item is one-shot. It waits while the lead is not injectable. After one
|
||||
successful `agents.send`, remove it. A send exception keeps it pending for a bounded retry. It never
|
||||
uses the repeated reminder behaviour of an uncollected ticket.
|
||||
|
||||
Also add a fail-loud entry point for a classified target that Unit 5 cannot map to a credential. It
|
||||
uses `PrimaryRegistry.nudgeTargetFor(target)` and says that correlation could not run. If no lead is
|
||||
known, log at `WARN`, not `DEBUG`.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/msg/ReplyPushLoop.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/msg/ReplyPushLoopTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. One incident affecting two workers owned by one lead causes one successful pane injection.
|
||||
2. Two affected workers owned by two leads cause one successful injection per affected lead. This
|
||||
is one notice per event per lead, not one notice per member.
|
||||
3. Repeating the same incident id is idempotent.
|
||||
4. A busy or unknown lead is not injected. The item stays pending until the lead becomes injectable
|
||||
or its attempt cap is reached.
|
||||
5. After one successful injection, later ticks do not mention that incident again.
|
||||
6. A failed `agents.send` is retried within the existing bound. A successful retry still gives only
|
||||
one successful send.
|
||||
7. A pending ticket and an outage incident for one lead appear in one combined nudge, not two
|
||||
competing turns.
|
||||
8. The text names the credential, profiles, affected workers, and remaining cool-off. It tells the
|
||||
lead to run `fleet_list`.
|
||||
9. An unmapped target produces a direct warning notice when a lead is known. If no lead is known,
|
||||
the code logs a `WARN` naming the target and reason.
|
||||
10. `stop()` clears incident state as it clears other push state.
|
||||
11. Existing reply, ticket, and question tests stay green. The focused tests and
|
||||
`mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. The API uses plain values, not the Unit 2 incident class. This lets Unit 3 run in parallel.
|
||||
|
||||
Unit 5 adapts the Unit 2 incident into this entry point.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact one-shot and retry rules.
|
||||
- The test showing one combined nudge with a failed ticket.
|
||||
- The test showing one successful send for two affected workers.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 4 — Durable member backend-error outcome
|
||||
|
||||
### Scope
|
||||
|
||||
Make a classified backend failure remain visible after its ticket is collected or expires.
|
||||
|
||||
Add `BACKEND_ERROR` to `MemberSession.State`. Add a nullable failure detail to `MemberSession` and
|
||||
render it as `failureReason` in `SessionManager.rosterView`. Add
|
||||
`SessionManager.onBackendError(target, reason)`.
|
||||
|
||||
The transition must handle both completion orderings:
|
||||
|
||||
- `BUSY -> BACKEND_ERROR` when classification wins before the normal completion state update;
|
||||
- `DONE -> BACKEND_ERROR` when the async resolver runs after `SessionManager.onTurnComplete`.
|
||||
|
||||
It must use a compare-and-set retry or another atomic update. `RELEASED` must never return to the
|
||||
roster. A backend-error member is terminal and cannot accept another delivery.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/session/MemberSession.java`.
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/session/SessionManager.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/session/SessionManagerTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. `onBackendError` moves a `BUSY` member to `BACKEND_ERROR` and stores the reason.
|
||||
2. It also moves `DONE` to `BACKEND_ERROR`, covering the resolver race.
|
||||
3. A later normal `onTurnComplete` cannot change `BACKEND_ERROR` back to `DONE`.
|
||||
4. A released or unknown member is not recreated. The unknown case logs at `WARN` and returns an
|
||||
explicit false result to its caller.
|
||||
5. `onDelivered` refuses a `BACKEND_ERROR` member, just as it refuses generic `FAILED`.
|
||||
6. `rosterView` reports `state: backend_error` and `failureReason` after the send ticket is gone.
|
||||
7. Ordinary members do not gain a blank or invented `failureReason` field.
|
||||
8. Existing constructors keep source compatibility for tests and adapters.
|
||||
9. The focused tests and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 4 can run with Units 1, 2, and 3.
|
||||
|
||||
Unit 5 calls the new session method from the production sink.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The two race orderings and the tests for both.
|
||||
- The exact roster JSON shape.
|
||||
- The unknown-target result and log level.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 5 — Production wiring, spawn gate, and fleet views
|
||||
|
||||
### Scope
|
||||
|
||||
Compose Units 1 to 4 in production. This is the only unit which edits `Fleetd.java`.
|
||||
|
||||
Add per-profile `errorPattern` config beside `exhaustedPattern`. Compile both once at startup. A
|
||||
configured pattern wins over the legacy default. Report configured profiles and legacy-default
|
||||
profiles separately at startup. A bad regex must stop startup with the profile and key in the
|
||||
message.
|
||||
|
||||
Wire one production `BackendErrorSink` with this order:
|
||||
|
||||
1. mark the member `backend_error` with its reason;
|
||||
2. resolve the profile and its current `effectiveCredentialId()` through the fail-loud #234 seam;
|
||||
3. record the error in `BackendOutagePolicy`;
|
||||
4. on a new incident, submit one event to `ReplyPushLoop`.
|
||||
|
||||
If target metadata cannot be resolved, do not end in `Optional.ifPresent`. Log an error and call the
|
||||
Unit 3 unmapped-target notice. The failed send still reaches its ticket through CB-588.
|
||||
|
||||
Teach both spawn paths about a separate cool-off source. Exhaustion quarantine has priority when
|
||||
both states are active. Automatic placement needs a distinct `coolingOff` set so its refusal does
|
||||
not say “exhausted”.
|
||||
|
||||
Extend the MCP (Model Context Protocol) views:
|
||||
|
||||
- A cooling profile has `free: 0`, `credentialId`, and `coolingOffForSeconds` in `fleet_list`.
|
||||
- It does not have `quarantinedForSeconds` unless exhaustion quarantine is also active.
|
||||
- `fleet_profiles` has a separate `coolingOff` map, not an entry in `quarantined`.
|
||||
- A direct spawn refusal says the credential is cooling off after repeated backend errors and gives
|
||||
the remaining seconds.
|
||||
|
||||
Do not change `FleetHealthMonitor.coverage`. It still describes the periodic health webhook path.
|
||||
Lead-pane outage delivery is a separate capability.
|
||||
|
||||
### Files owned
|
||||
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/Fleetd.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/config/ConfigRef.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/member/CompositePeerLauncher.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/placement/PlacementContext.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/placement/PlacementPolicyUtil.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/config/FleetConfigTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/config/ConfigRefTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/member/CompositePeerLauncherTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/placement/PlacementPolicyTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/mcp/FleetMcpTest.java`
|
||||
- Add `fleetd/src/test/java/dev/ltms/fleet/BackendOutageFlowTest.java`.
|
||||
- `fleetd/fleetd.example.yaml`
|
||||
- `CLAUDE.md`
|
||||
|
||||
No earlier unit edits these files.
|
||||
|
||||
The lead, not a worker, must update `wiki/7-Use-Cases.md`, `wiki/9-Implementation.md`, and
|
||||
`wiki/11-Features.md`. Project rules forbid workers from committing `wiki/`. The portable block in
|
||||
`CLAUDE.md` and `wiki/7-Use-Cases.md` must remain byte-identical.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. `errorPattern` binds per profile. Blank uses the legacy default and is reported as degraded
|
||||
coverage. A config reload which changes it is reported as deferred because patterns are compiled
|
||||
at startup.
|
||||
2. A malformed `errorPattern` stops startup and names `profiles.<name>.errorPattern`.
|
||||
3. The real production sink never silently drops an unknown target. A test captures its error log
|
||||
and the fallback notice call.
|
||||
4. One real backend-error classification through `CompletionResolver` marks only its member. It does
|
||||
not cool the credential and does not send an outage notice.
|
||||
5. Two real classifications for one credential within 60 seconds start one incident.
|
||||
6. The integration test then calls the real explicit-profile spawn gate. It is refused before any
|
||||
adapter spawn call, with “cooling off” and remaining seconds in the message.
|
||||
7. The same test calls an automatic placement path. A cooling candidate is skipped. If every
|
||||
candidate is cooling, the error names cool-off rather than exhaustion.
|
||||
8. Profiles sharing the credential are all blocked. A profile on another credential stays usable.
|
||||
9. `fleet_list` from the same fixture shows both members as `backend_error`, preserves each failure
|
||||
reason, and reports `free: 0`, the credential, and `coolingOffForSeconds`.
|
||||
10. `fleet_profiles` reports cool-off separately from quarantine.
|
||||
11. The real `ReplyPushLoop` receives one incident and makes one successful lead-pane send. Existing
|
||||
failed-ticket notice content may share that same combined send.
|
||||
12. Exhaustion still wins when a line matches both patterns. A simultaneous exhaustion quarantine
|
||||
also wins in spawn errors and fleet views.
|
||||
13. After the 60-second cool-off, spawn is allowed again. A new incident needs two fresh errors.
|
||||
14. `healthCoverage` has the same value before and after this change for the same health config.
|
||||
15. `fleetd.example.yaml` explains `errorPattern`, the legacy fallback, 2/60/60 policy, and the
|
||||
difference between cool-off and exhaustion quarantine.
|
||||
16. `CLAUDE.md` tells leads how `fleet_profiles` and `fleet_list` report cool-off. The lead later
|
||||
applies the matching wiki updates and runs the documented byte-sync check.
|
||||
17. The developer records the new end-to-end test failing before implementation, then passing. The
|
||||
focused suites and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
Unit 5 starts only after Units 1 to 4 are merged or rebased into its branch. It also starts after the
|
||||
#234 defect 2 fix lands, because both areas touch the same target-resolution control path.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact commits used for Units 1 to 4 and #234.
|
||||
- The startup coverage line with one configured and one legacy-default profile.
|
||||
- The red test output before the implementation and its green result after.
|
||||
- The explicit and automatic spawn refusal text.
|
||||
- Sample `fleet_list` and `fleet_profiles` JSON for cool-off and exhaustion.
|
||||
- The number and text of lead-pane sends in the real-path test.
|
||||
- The focused test commands and final `mvn clean install` result.
|
||||
- The exact `CLAUDE.md` change and the wiki edits the lead must apply.
|
||||
|
||||
## File ownership summary
|
||||
|
||||
| Area | Unit | Shared edit risk |
|
||||
|---|---:|---|
|
||||
| Completion classification | 1 | Only Unit 1 edits `CompletionResolver` and its test. |
|
||||
| Correlation and cool-off state | 2 | New files only. |
|
||||
| Lead push scheduling | 3 | Only Unit 3 edits `ReplyPushLoop` and its test. |
|
||||
| Member terminal state | 4 | Only Unit 4 edits `MemberSession`, `SessionManager`, and their test. |
|
||||
| Main composition, config, placement, MCP views, shipped prompt | 5 | Only Unit 5 edits `Fleetd`, `FleetConfig`, `CompositePeerLauncher`, placement context, `FleetMcp`, and `CLAUDE.md`. |
|
||||
| Wiki propagation | Lead after Unit 5 | Workers do not commit the wiki submodule. |
|
||||
|
||||
## What I would not build
|
||||
|
||||
1. **Do not reuse `BackendQuarantine` for outages.** Its repeat call restarts a long credential
|
||||
quarantine. Its fields and errors say “exhausted”. That is wrong for a short outage.
|
||||
2. **Do not merge the exhaustion and generic error patterns.** Exhaustion must win because it has a
|
||||
different policy and duration.
|
||||
3. **Do not group by error string.** One outage can produce different text. The shared operational
|
||||
limit is the credential.
|
||||
4. **Do not mark a profile unusable until config changes.** The current classifier cannot safely
|
||||
tell a permanent malformed request from a transient service fault. A permanent state would need
|
||||
a stronger error taxonomy first.
|
||||
5. **Do not quarantine on the first generic backend error.** That would turn one bad request or one
|
||||
false pattern match into a fleet-wide capacity loss.
|
||||
6. **Do not add this to `FleetHealthMonitor`.** The monitor samples slow member health. The exact
|
||||
backend event already exists at completion resolution, and moving it to polling would lose type
|
||||
and time.
|
||||
7. **Do not add another direct lead injector.** `ReplyPushLoop` already owns status gating,
|
||||
per-lead coalescing, retry bounds, and heartbeat stand-down.
|
||||
8. **Do not change `healthCoverage` to `full`.** That field still means a webhook notification sink
|
||||
exists for periodic health. A backend outage nudge does not make every health event visible.
|
||||
9. **Do not persist incident history across daemon restart in this work.** Existing exhaustion
|
||||
quarantine is also in memory. A 60-second state does not justify a new durable store.
|
||||
10. **Do not build work recovery.** The PR-body survival story proves why checkpoint-first work is
|
||||
useful, but these tickets are about detection, capacity, and signalling.
|
||||
11. **Do not remove the legacy `API Error:` fallback in the first release.** Doing so would turn an
|
||||
unedited config back into a false successful completion. Report it as degraded coverage instead.
|
||||
12. **Do not reorder or add the old `visibleTurn` fallback from #201.** #211 already implemented the
|
||||
narrow raw-scrape fallback at `CompletionResolver.classifyRawScrapeFallback`.
|
||||
|
||||
## Riskiest assumption and cheapest experiment
|
||||
|
||||
The riskiest assumption is that a configured error regex means “the backend failed this turn”. The
|
||||
current code and test already show the counterexample: a worker may quote `API Error:` while writing
|
||||
a valid report. Two such false matches on one credential would now remove capacity for 60 seconds.
|
||||
|
||||
The cheapest experiment is a replay corpus before Unit 5 ships:
|
||||
|
||||
1. Save the full pane text from the measured 2026-09-01 outage.
|
||||
2. Produce one safe failure per backend with a disposable invalid endpoint or request.
|
||||
3. Save one valid member report which quotes each error line.
|
||||
4. Replay all samples through the real `CompletionResolver` test fixture.
|
||||
5. Require outage samples to match and quoted-report samples not to match after assistant-block
|
||||
extraction and baseline checks.
|
||||
|
||||
This costs no outage deployment and no real sleep. If quoted reports still match, narrow the profile
|
||||
patterns before enabling correlation. Do not raise the threshold to hide a bad classifier.
|
||||
|
||||
## Sequencing with three developers
|
||||
|
||||
First wave:
|
||||
|
||||
1. Developer A: Unit 1, typed classification.
|
||||
2. Developer B: Unit 2, credential outage policy.
|
||||
3. Developer C: Unit 3, lead outage nudge.
|
||||
|
||||
As soon as one slot is free, start Unit 4. It is file-disjoint from every first-wave unit. Merge and
|
||||
review Units 1 to 4 independently.
|
||||
|
||||
Start Unit 5 only after all four foundations and #234 are available. Unit 5 is the only high-conflict
|
||||
integration branch, so no other active unit should touch its file list.
|
||||
|
||||
## Checks performed for this refinement
|
||||
|
||||
- Read issue #201 and issue #227 through their Gitea pages. Both showed zero comments.
|
||||
- Read the source and tests named in the evidence section.
|
||||
- Ran `git status --short --branch`; the branch was clean before this document was added.
|
||||
- Ran `git log --oneline -12` to identify the branch base.
|
||||
- I did not run Maven because this change adds only a design document.
|
||||
- Rendered both Mermaid blocks with `npx @mermaid-js/mermaid-cli`; both commands succeeded.
|
||||
+123
-324
@@ -1,311 +1,162 @@
|
||||
# MCP Contract — `fleetd`'s unified gateway
|
||||
# MCP flows and error model — `fleetd`
|
||||
|
||||
> **Status: 🔴 HISTORICAL DESIGN — do NOT use as the tool reference.** Written 2026-07-14, before
|
||||
> any MCP code existed. The system shipped and this page never caught up, so **its tool names,
|
||||
> parameter names and REST paths are wrong today**. Audited 2026-08-17; the specific drift:
|
||||
> **What this page is.** The **flows**: how a delegation, a clarification, a detached task and a
|
||||
> silent member each travel through `fleetd`. These shapes are what shipped, and they are hard to
|
||||
> read off the code because they span the MCP face, the rendezvous registry, the `Injector` and
|
||||
> herdr.
|
||||
>
|
||||
> - **Tools it names that do not exist:** `fleet_read`, `fleet_cancel`.
|
||||
> - **Shipped tools it omits:** `fleet_poll`, `fleet_ack`, `fleet_profiles`, `fleet_whoami`.
|
||||
> - **Parameter names are wrong nearly everywhere** — it says `message`/`target`/`timeout_seconds`/
|
||||
> `block` where the code takes `content`/`sessionId`/`timeoutMs`/`wait`; `text` where
|
||||
> `fleet_reply` takes `content`; `target` where `fleet_stop` takes `paneId`.
|
||||
> - **REST paths are wrong:** it says `POST /workers` and `DELETE /workers/{paneId}`; the daemon
|
||||
> serves `POST /members` and `DELETE /members/{paneId}`.
|
||||
> **What this page is NOT: a tool reference.** It deliberately holds no tool catalogue, no
|
||||
> parameter tables and no REST paths. **The live MCP schema is the authority** — each tool's own
|
||||
> description and parameters, as mounted — with the intent→tool table in `CLAUDE.md` as the short
|
||||
> form.
|
||||
>
|
||||
> **The authoritative tool surface is the live MCP schema** (each tool's own description and
|
||||
> parameters, as mounted), with the intent→tool table in `CLAUDE.md` as the short form. Both were
|
||||
> checked against `mcp/FleetMcp.java` on 2026-08-17 and are accurate.
|
||||
> That absence is the fix for fleetd #114 (CB-609), and it is worth stating why. This page used to
|
||||
> carry a full tool catalogue written in July 2026, before any MCP code existed. The code shipped;
|
||||
> the page did not follow. By August it named two tools that do not exist, omitted five that do,
|
||||
> had the wrong name for nearly every parameter, pointed at REST paths the daemon does not serve,
|
||||
> and — worst — still described an identity model (*"any connection that does not map to a known
|
||||
> worker is treated as a primary"*) that was a real privilege bug, fixed since by the ancestry
|
||||
> walk in fleetd #161. Every one of those errors is the same error: **a second, hand-maintained
|
||||
> copy of something the code already states**. So the second copy is gone rather than corrected.
|
||||
> Only the flows remain, because a flow is a shape rather than a name, and shapes are what this
|
||||
> page was ever good for.
|
||||
>
|
||||
> What is still worth reading here is **§6 — the flows and the error model** (rendezvous,
|
||||
> `fleet_ask`, detached delivery, the turn-done fallback). The shapes it describes are the ones
|
||||
> that shipped; only the names around them drifted. Rewriting this page is tracked as **CB-609**.
|
||||
|
||||
`fleetd` is the **sole communication gateway** for every Claude session in the bridge. Both
|
||||
the **primary** (Opus, on subscription) and every **worker** (off-subscription Claude Code)
|
||||
mount the *same* MCP server with a single `claude mcp add` line, and talk only through its
|
||||
tools. No Claude session ever addresses a broker, a peer, or the network directly.
|
||||
|
||||
This document defines every MCP tool that face must expose, who may call it, its blocking
|
||||
semantics, and how it maps onto the code already in the tree.
|
||||
> The names that do appear below are checked by `McpContractDocTest`, which fails if this page
|
||||
> names a `fleet_*` tool the server does not register. That test is the whole reason it is safe to
|
||||
> write a tool name here at all.
|
||||
|
||||
---
|
||||
|
||||
## 1. Design constraints (non-negotiable)
|
||||
## 1. Rendezvous flows
|
||||
|
||||
These come from the project's core invariants and bound every decision below.
|
||||
### 1.1 Delegation — happy path
|
||||
|
||||
1. **One server, both roles.** The primary and all workers mount an identical server. The
|
||||
catalog must serve both, and `fleetd` must decide *who is calling* from the connection —
|
||||
never from a caller-supplied argument that could be spoofed.
|
||||
2. **Subscription-safe by construction.** No MCP tool ever reads, sets, or forwards
|
||||
`ANTHROPIC_BASE_URL`. Mounting the bridge cannot move a session off subscription.
|
||||
Enforced today by [`SubscriptionGuard`](1-Architecture).
|
||||
3. **Blocking rendezvous, no busy-poll.** The primary consumes a worker's reply through a
|
||||
*single* MCP call that `fleetd` holds open — never a cross-turn poll loop that would burn
|
||||
subscription quota.
|
||||
4. **Status-gated delivery.** Anything that puts text into a worker flows through the existing
|
||||
[`Injector`](1-Architecture): delivered only when the worker is `idle`/`blocked`, at most
|
||||
one message per turn.
|
||||
5. **`fleetd` owns policy; herdr owns PTYs.** MCP tools express *intent*; `fleetd`
|
||||
translates it into guard checks, rendezvous bookkeeping, and herdr `agent.*` calls.
|
||||
|
||||
---
|
||||
|
||||
## 2. Topology
|
||||
|
||||
Both faces live in the one daemon. The **north face** is MCP (this document); the **south
|
||||
face** is the herdr Unix socket. REST/SSE remains only for non-Claude clients and dashboards.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
OPUS["Opus — primary<br/>(Claude Code, env CLEAN)<br/>MCP client"]
|
||||
subgraph BD["fleetd — standalone daemon"]
|
||||
MCP["MCP server (north face)<br/>fleet_send · fleet_reply<br/>fleet_ask · fleet_status · lifecycle"]
|
||||
RDV["rendezvous registry<br/>(blocking-call waiters)"]
|
||||
INJ["Injector + StatusPoller<br/>(status-gated writer)"]
|
||||
SOCK["herdr socket client (south face)"]
|
||||
MCP --> RDV
|
||||
RDV --> INJ
|
||||
INJ --> SOCK
|
||||
MCP --> SOCK
|
||||
end
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
|
||||
OPUS -->|"fleet_send (blocks)"| MCP
|
||||
W -.->|"fleet_reply / fleet_ask"| MCP
|
||||
SOCK -->|"agent.start · agent.send<br/>agent.get · pane.close"| HERDR
|
||||
HERDR -->|"drives PTY"| W
|
||||
|
||||
classDef ext fill:#2b6cb0,stroke:#1a365d,color:#ffffff;
|
||||
classDef core fill:#2f855a,stroke:#22543d,color:#ffffff;
|
||||
class OPUS,W ext
|
||||
class MCP,RDV,INJ,SOCK core
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. Identity & addressing
|
||||
|
||||
Because the same server is mounted by everyone, `fleetd` resolves the caller's role on every
|
||||
request — this is the linchpin of the whole contract and has no code yet.
|
||||
|
||||
- **Workers are known.** `fleetd` spawns every worker
|
||||
([`WorkerService`](1-Architecture)) and records its herdr session UUID / `terminal_id` on
|
||||
the returned [`Agent`]. When a call arrives on a connection that maps to a known worker,
|
||||
the caller is *that* worker — so **workers never pass a target**; routing is implicit.
|
||||
- **The primary is "not a worker".** Any connection that does not map to a known worker is
|
||||
treated as a primary. It addresses workers **explicitly** by `target` — a session UUID,
|
||||
a `terminal_id`, or a friendly `profile` name.
|
||||
- **Turn correlation.** A blocking `fleet_send` registers a *waiter* keyed by worker
|
||||
identity. A worker's later `fleet_reply` / `fleet_ask` on the same identity resolves that
|
||||
waiter. A `turn_id` is minted per exchange so a clarification round-trip
|
||||
(§6.2) rejoins the right turn.
|
||||
|
||||
---
|
||||
|
||||
## 4. Transport
|
||||
|
||||
`fleetd` is a long-lived daemon serving **multiple** concurrent clients (one primary + N
|
||||
workers), so a per-client stdio child is the wrong shape. The recommended transport is
|
||||
**streamable-HTTP / SSE** on the same bind as the REST face:
|
||||
|
||||
```bash
|
||||
# identical on primary and every worker
|
||||
claude mcp add --transport http fleetd http://127.0.0.1:8080/mcp
|
||||
```
|
||||
|
||||
This adds an MCP-server dependency the pom does not yet carry. See [Open decisions](#10-open-decisions).
|
||||
|
||||
---
|
||||
|
||||
## 5. Tool catalog
|
||||
|
||||
| Tool | Caller | Blocks? | Backing (exists today?) |
|
||||
|---|---|---|---|
|
||||
| [`fleet_send`](#fleet_send) | primary | yes (default) | `Injector.enqueue` ✅ · rendezvous registry ❌ (CB-104) |
|
||||
| [`fleet_reply`](#fleet_reply) | worker | no | rendezvous ❌ · pane injection via `Injector` ✅ |
|
||||
| [`fleet_ask`](#fleet_ask) | worker | yes | reverse rendezvous ❌ |
|
||||
| [`fleet_status`](#fleet_status) | either | no | `AgentControl.status` ✅ · `Injector.activeTargets` ✅ |
|
||||
| [`fleet_spawn`](#lifecycle) | primary | no | `WorkerService.spawn` ✅ (`POST /workers`) |
|
||||
| [`fleet_list`](#lifecycle) | either | no | `WorkerService.list` ✅ (`/agents`) |
|
||||
| [`fleet_stop`](#lifecycle) | primary | no | `WorkerService.stop` ✅ (`DELETE /workers/{paneId}`) |
|
||||
| [`fleet_read`](#fleet_read) | primary | no | `AgentControl.read` ✅ |
|
||||
| [`fleet_cancel`](#fleet_cancel) | primary | no | — ❌ (future) |
|
||||
|
||||
### Core: delegation & rendezvous
|
||||
|
||||
#### `fleet_send`
|
||||
*(primary → worker — the headline tool, CB-104)*
|
||||
|
||||
- **Params:** `message` (required); `target` (optional — defaults to the sole worker / default
|
||||
profile); `timeout_seconds` (default 600); `block` (default `true`); `auto_spawn`
|
||||
(default `true`); `turn_id` (optional — supplied when answering a worker's `fleet_ask`).
|
||||
- **Blocking (`block:true`):** enqueue `message` via the `Injector`, then hold the call open
|
||||
until exactly one of:
|
||||
- worker calls `fleet_reply` → `{ outcome:"reply", text }`
|
||||
- worker calls `fleet_ask` → `{ outcome:"question", text, turn_id }`
|
||||
- worker's `agent_status` reaches done/idle with no reply → `{ outcome:"turn_done", text:<terminal tail> }`
|
||||
- deadline elapses → `{ outcome:"timeout" }`
|
||||
- worker gone → error `worker_gone`
|
||||
- **Detached (`block:false`):** enqueue and return `{ outcome:"dispatched", dispatch_id }`
|
||||
immediately. The eventual reply is injected into the primary's idle pane (§6.3), or drained
|
||||
via `fleet_status` on a split-host primary.
|
||||
|
||||
#### `fleet_reply`
|
||||
*(worker → primary)*
|
||||
|
||||
- **Params:** `text` (required); `final` (default `true`).
|
||||
- **Behavior:** resolve the primary waiter registered against this worker with `text`. If no
|
||||
waiter exists (detached delegation), `fleetd` **injects the primary's idle pane** instead.
|
||||
Returns `{ delivered:true, mode:"resolved"|"injected" }`. No `target` — identity is implicit.
|
||||
|
||||
#### `fleet_ask`
|
||||
*(worker → primary — the reverse rendezvous)*
|
||||
|
||||
- **Params:** `question` (required); `timeout_seconds`.
|
||||
- **Behavior:** blocks the *worker's* call. Surfaces the question to the primary (resolving its
|
||||
open `fleet_send` with `outcome:"question"`, or injecting its pane). When the primary
|
||||
answers — a `fleet_send` carrying the matching `turn_id` — that unblocks this call and
|
||||
returns `{ answer }` to the worker, which continues **in the same turn**.
|
||||
|
||||
### Worker lifecycle
|
||||
<a id="lifecycle"></a>
|
||||
Thin adapters over [`WorkerService`](1-Architecture) — parity with the existing REST routes.
|
||||
|
||||
- **`fleet_spawn`** — `{ profile? }` → worker view (`sessionId`, `terminalId`, `paneId`,
|
||||
`status`). Guard-checked; a boundary breach returns error `subscription_boundary` (the
|
||||
REST `403`).
|
||||
- **`fleet_list`** — no params → all workers + `agent_status`. Read-only, either role.
|
||||
- **`fleet_stop`** — `{ target }` → tears down the pane and its dedicated tab. Idempotent.
|
||||
|
||||
### Observability
|
||||
|
||||
#### `fleet_status`
|
||||
*(either role — the README's 4th named tool)*
|
||||
|
||||
- **Params:** `target?`.
|
||||
- **Behavior:** per-worker `agent_status`, queue depth (`Injector.activeTargets`), whether a
|
||||
rendezvous is open, and ids. For the *calling* session it also reports/drains **pending
|
||||
messages addressed to me** — the path a split-host primary's `Stop`-hook uses to wake and
|
||||
collect replies without being injectable. Read-only, non-blocking.
|
||||
|
||||
#### `fleet_read`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`; `source` ∈ `visible | recent | recent_unwrapped | detection`.
|
||||
- **Behavior:** returns the worker's terminal text so the primary can peek at a *detached*
|
||||
worker's progress. Adapter over `AgentControl.read`.
|
||||
|
||||
### Control (future)
|
||||
|
||||
#### `fleet_cancel`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`. Interrupt the worker's current turn / abandon the rendezvous. No
|
||||
backing code yet.
|
||||
|
||||
---
|
||||
|
||||
## 6. Rendezvous flows
|
||||
|
||||
### 6.1 Delegation — happy path
|
||||
|
||||
One blocking call, zero polls.
|
||||
One blocking call, zero polls. The lead's call is held open by `fleetd` until the member answers.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary (Opus)
|
||||
participant B as fleetd (MCP + Injector)
|
||||
participant P as "Lead (primary)"
|
||||
participant B as "fleetd (MCP + Injector)"
|
||||
participant H as herdr
|
||||
participant W as Worker (Claude)
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B->>B: register waiter(w)
|
||||
B->>H: agent.send(w, "do X") (idle window)
|
||||
H-->>W: prompt injected
|
||||
W->>W: works the turn
|
||||
W->>B: fleet_reply("result")
|
||||
B->>B: resolve waiter(w)
|
||||
B-->>P: { outcome:"reply", text:"result" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B->>B: "register waiter(sessionId)"
|
||||
B->>H: "agent.send — only in an injectable window"
|
||||
H-->>W: "prompt injected"
|
||||
W->>W: "works the turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B->>B: "resolve waiter"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.2 Clarification — reverse rendezvous (`fleet_ask`)
|
||||
**The cap that matters:** a blocking `fleet_send` is bounded by the *caller's own* MCP client
|
||||
timeout, about 60 seconds — not by the task. Anything slower than that must use the detached flow
|
||||
in §1.3, or the lead's call returns while the member is still working.
|
||||
|
||||
The worker pauses mid-turn to ask; the primary answers; the worker resumes in the same turn.
|
||||
### 1.2 Clarification — reverse rendezvous
|
||||
|
||||
The member pauses mid-turn to ask, the lead answers, and the member resumes **the same turn** with
|
||||
its context intact.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>B: fleet_ask("which config?") — worker blocks
|
||||
B-->>P: { outcome:"question", text:"which config?", turn_id }
|
||||
P->>B: fleet_send("config.yaml", target=w, turn_id) — blocks again
|
||||
B-->>W: resolve fleet_ask → { answer:"config.yaml" }
|
||||
W->>W: resumes same turn
|
||||
W->>B: fleet_reply("done")
|
||||
B-->>P: { outcome:"reply", text:"done" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B-->>W: "content injected"
|
||||
W->>B: "fleet_ask{question} — member blocks"
|
||||
B-->>P: "{ outcome: question, turnId }"
|
||||
P->>B: "fleet_send{turnId, content} — answers THIS turn"
|
||||
B-->>W: "fleet_ask returns the answer"
|
||||
W->>W: "resumes the same turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.3 Detached delegation — pane injection
|
||||
**Answer with `turnId`, never `sessionId`.** A `sessionId` send starts a new turn; it does not
|
||||
resolve the waiting `fleet_ask`.
|
||||
|
||||
The primary does not block; the reply arrives later in its idle pane.
|
||||
**The window is about 55 seconds and no nudge extends it.** So never brief a member to "ask me":
|
||||
decide before delegating, or give the member an explicit default to fall back on.
|
||||
|
||||
### 1.3 Detached delegation — the lead does not block
|
||||
|
||||
The lead gets a ticket immediately and collects the answer later. This is the flow for any real
|
||||
task, because of the ~60s cap in §1.1.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w, block=false)
|
||||
B-->>P: { outcome:"dispatched", dispatch_id }
|
||||
P->>P: continues its own work
|
||||
W->>B: fleet_reply("result")
|
||||
Note over B: no waiter → detached path
|
||||
B->>B: Injector.enqueue(primary_pane, "result")
|
||||
B-->>P: injected into idle pane (status-gated)
|
||||
P->>B: "fleet_send{sessionId, content, wait:false}"
|
||||
B-->>P: "accepted — ticket"
|
||||
P->>P: "continues its own work"
|
||||
W->>B: "fleet_reply{content}"
|
||||
Note over B: "no waiter is blocked — the reply is held"
|
||||
B->>B: "nudge the lead's own pane (status-gated)"
|
||||
P->>B: "fleet_poll{ticket}"
|
||||
B-->>P: "the member's report"
|
||||
P->>B: "fleet_ack{target, msgId}"
|
||||
```
|
||||
|
||||
### 6.4 Uncooperative worker — turn-done fallback
|
||||
A terminal ticket nudges the lead's pane by itself, so a detached task does not need watching. The
|
||||
nudge needs an injectable lead pane and is capped, so it is a convenience rather than a guarantee.
|
||||
|
||||
A worker that never calls `fleet_reply` still returns a result: `fleetd` reads its terminal
|
||||
tail when the turn completes.
|
||||
### 1.4 The member never replies — turn-done fallback
|
||||
|
||||
A member that ends its turn without `fleet_reply` still produces something: `fleetd` reads its
|
||||
pane tail. This is a **fallback, not a channel** — it is lossy in three separate ways, and every
|
||||
one of them has produced a wrong answer in practice.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>W: works, never calls fleet_reply
|
||||
B->>B: StatusPoller sees agent_status → idle/done
|
||||
B->>B: AgentControl.read(w, "recent")
|
||||
B-->>P: { outcome:"turn_done", text:<terminal tail> }
|
||||
P->>B: "fleet_send — blocks or detaches"
|
||||
B-->>W: "content injected"
|
||||
W->>W: "works, never calls fleet_reply"
|
||||
B->>B: "StatusPoller sees the turn end"
|
||||
B->>B: "read the pane tail"
|
||||
B->>B: "classify: exhausted? echoed brief? real report?"
|
||||
B-->>P: "{ outcome: turn_done } or a named failure"
|
||||
```
|
||||
|
||||
The three ways it goes wrong, and what each looks like now:
|
||||
|
||||
| What happened | What the lead used to get | What it gets today |
|
||||
|---|---|---|
|
||||
| The report is longer than the scrape window | The **end** silently cut off | Still clipped, but marked partial |
|
||||
| The member never started — spent credential | The lead's **own brief** echoed back as a report | A named failure: backend exhausted |
|
||||
| The member is simply slow | A tail of work in progress | Unchanged — read it as a hint, not a result |
|
||||
|
||||
The echoed-brief case is the one to remember: it reads as a long, on-topic report with nothing in
|
||||
it from the member. It is suppressed now, but the general rule stands — **check the member's
|
||||
worktree with `git log` before believing a report you did not watch arrive.**
|
||||
|
||||
---
|
||||
|
||||
## 7. Status gating
|
||||
## 2. Status gating
|
||||
|
||||
Delivery only happens in a safe window. This is the state machine the `Injector` already
|
||||
enforces via `AgentStatus.injectable()`; MCP `fleet_send` is simply its producer.
|
||||
Delivery only happens in a safe window. `fleet_send` is a producer for the `Injector`, which
|
||||
already enforces this through `AgentStatus.injectable()`.
|
||||
|
||||
```mermaid
|
||||
stateDiagram-v2
|
||||
[*] --> IDLE
|
||||
IDLE --> WORKING: message delivered / picks up
|
||||
WORKING --> IDLE: turn done
|
||||
WORKING --> BLOCKED: awaits input
|
||||
BLOCKED --> WORKING: input delivered
|
||||
IDLE --> UNKNOWN: detection glitch
|
||||
BLOCKED --> UNKNOWN: detection glitch
|
||||
UNKNOWN --> IDLE: re-detected
|
||||
IDLE --> WORKING: "message delivered, picked up"
|
||||
WORKING --> IDLE: "turn done"
|
||||
WORKING --> BLOCKED: "awaits input"
|
||||
BLOCKED --> WORKING: "input delivered"
|
||||
IDLE --> UNKNOWN: "detection glitch"
|
||||
BLOCKED --> UNKNOWN: "detection glitch"
|
||||
UNKNOWN --> IDLE: "re-detected"
|
||||
|
||||
note right of IDLE
|
||||
injectable — deliver head of FIFO
|
||||
@@ -321,69 +172,17 @@ stateDiagram-v2
|
||||
end note
|
||||
```
|
||||
|
||||
At most one message is delivered per turn: after a send the `Injector` waits for a `WORKING`
|
||||
pickup before delivering the next, with a `PICKUP_GRACE_POLLS` fallback for turns faster than
|
||||
the poll interval. A herdr `events.subscribe` stream can later replace the sampling without
|
||||
touching this state machine.
|
||||
**At most one message per turn.** After a send, the `Injector` waits for a `WORKING` pickup before
|
||||
delivering the next, with a grace-poll fallback for turns that finish faster than the poll
|
||||
interval.
|
||||
|
||||
---
|
||||
Two consequences a lead feels directly:
|
||||
|
||||
## 8. Error model
|
||||
- **A second send to a busy member never lands.** It reports as queued and times out. The member
|
||||
is fine; the message simply waits, and then restarts the member when it next goes idle.
|
||||
- **A spawned member is not deliverable until it has mounted the MCP.** Until then a send waits on
|
||||
that gate for about 60 seconds and then fails without ever reaching the pane.
|
||||
|
||||
| Condition | `fleet_send` result | Notes |
|
||||
|---|---|---|
|
||||
| Worker replies | `{ outcome:"reply" }` | normal |
|
||||
| Worker asks | `{ outcome:"question", turn_id }` | answer with `fleet_send(turn_id)` |
|
||||
| Turn ends, no reply | `{ outcome:"turn_done" }` | terminal tail as text |
|
||||
| Deadline elapsed | `{ outcome:"timeout" }` | message may still be queued/delivered |
|
||||
| Worker vanished | error `worker_gone` | `Injector.drop` fails the queued future |
|
||||
| Guard breach on spawn | error `subscription_boundary` | REST `403` parity |
|
||||
| Delivery failed at herdr | error, message dropped | poisoned message not left blocking the FIFO |
|
||||
|
||||
`fleet_reply` from a worker with no open waiter is **not** an error — it falls through to
|
||||
detached pane injection (§6.3).
|
||||
|
||||
---
|
||||
|
||||
## 9. Mapping to existing code
|
||||
|
||||
The MCP face is a thin adapter layer; nearly every capability already exists behind the REST
|
||||
seam. Only the **rendezvous registry** and the **caller-identity resolver** are new.
|
||||
|
||||
| MCP tool | Existing collaborator | New work |
|
||||
|---|---|---|
|
||||
| `fleet_send` | `Injector.enqueue`, `AgentControl.send` | waiter registry, timeout, outcome mux (CB-104) |
|
||||
| `fleet_reply` / `fleet_ask` | `Injector` (pane injection) | reverse rendezvous, identity resolver |
|
||||
| `fleet_status` | `AgentControl.status`, `Injector.activeTargets` | pending-drain projection |
|
||||
| `fleet_spawn` / `list` / `stop` | `WorkerService.{spawn,list,stop}` | MCP adapter only |
|
||||
| `fleet_read` | `AgentControl.read` | MCP adapter only |
|
||||
|
||||
Because the REST routes in `FleetApp` already exercise the collaborators, MCP tools are
|
||||
validated by **parity** against those routes, not by re-testing behavior.
|
||||
|
||||
---
|
||||
|
||||
## 10. Open decisions
|
||||
|
||||
1. **`fleet_ask` direction.** This page defines it as *worker-asks-primary* (a genuine reverse
|
||||
channel, matching the "inject the primary's pane" language). The alternative — a synonym for
|
||||
a blocking primary→worker send — is weaker and produces different plumbing. **Recommend
|
||||
worker-asks-primary.**
|
||||
2. **Detached delivery shape.** A `block:false` param on `fleet_send` (keeps the catalog
|
||||
small) vs. a separate `fleet_dispatch` tool. **Recommend the param.**
|
||||
3. **Auto-spawn on send.** `fleet_send` provisions a worker per profile when none exists
|
||||
(simplest primary UX) vs. requiring an explicit `fleet_spawn` first. **Recommend
|
||||
auto-spawn, defaulting on.**
|
||||
4. **Transport & SDK.** Streamable-HTTP/SSE co-located with the REST bind (recommended) vs.
|
||||
stdio. Requires choosing a Java MCP server SDK and adding it to the pom.
|
||||
|
||||
---
|
||||
|
||||
## 11. Implementation staging
|
||||
|
||||
- **CB-104** — blocking `fleet_send` + rendezvous registry + caller-identity resolver
|
||||
(the producer that finally drives the inert `StatusPoller`).
|
||||
- **CB-1xx** — `fleet_reply` / `fleet_ask` reverse rendezvous + detached pane injection.
|
||||
- **CB-1xx** — lifecycle + observability adapters (`fleet_spawn/list/stop/status/read`).
|
||||
- **CB-1xx** — transport wiring + `claude mcp add` docs; parity tests vs. REST.
|
||||
- **Later** — `fleet_cancel`; swap `StatusPoller` for herdr `events.subscribe`.
|
||||
`UNKNOWN` is deliberately neither injectable nor a pickup. A pane whose status cannot be read is
|
||||
not a pane that is safe to write to — see fleetd #176 for what happens when a gate treats an
|
||||
unreadable pane as a ready one.
|
||||
|
||||
@@ -0,0 +1,208 @@
|
||||
# Wiki audit for #168
|
||||
|
||||
**Source checked:** `.wiki-snapshot/` at `68e32c6` (2026-08-31). I did not use
|
||||
`wiki/`. Code references below are from the current `fleetd` source tree. A quoted
|
||||
line is a concrete claim that needs correction, unless the table says `KEEP`.
|
||||
|
||||
| Page | Verdict | One-line reason |
|
||||
|---|---|---|
|
||||
| `Home.md` | REVISE | Good overview, but it still names the retired product. |
|
||||
| `_Sidebar.md` | REVISE | The heading still says `claude-bridge`. |
|
||||
| `1-Architecture.md` | REBUILD | Its component contract mixes current names with removed tools, routes, and planned backends. |
|
||||
| `2-Message-Server.md` | REBUILD | The claimed MCP schema, mount command, REST/SSE surface, and fallback paths are pre-build design. |
|
||||
| `3-Approaches.md` | REVISE | Useful research history, but it presents unbuilt AgentAPI as a selectable fallback. |
|
||||
| `4-Setup.md` | RETIRE | It is an intentional stub that only redirects to chapter 13. |
|
||||
| `5-Operations.md` | RETIRE | It is an intentional stub that only redirects to chapter 13. |
|
||||
| `6-Team.md` | REBUILD | It teaches role-addressed sends and a Claude-only team model that the shipped API does not have. |
|
||||
| `7-Use-Cases.md` | REBUILD | Its flagship flow depends on removed `ccs` profiles and removed send parameters. |
|
||||
| `8-Roadmap.md` | REBUILD | It is a historical plan, but it presents old implementation choices and planned work as the current stack. |
|
||||
| `9-Implementation.md` | REBUILD | Its package, class, endpoint, and outcome map has drifted from the source. |
|
||||
| `10-Cross-Host-Messaging.md` | REVISE | It labels most federation work proposed, but misses the shipped `coordinator:` lead channel. |
|
||||
| `11-Features.md` | REVISE | It is the right catalogue, but code-path names are old and it misses the second-herdr-daemon capability. |
|
||||
| `12-Claude-to-OpenCode.md` | REVISE | The porting guide is mostly current, but calls the product and spawned-member path a bridge. |
|
||||
| `13-User-Guide.md` | REVISE | It is the best operator page, but needs the product rename and the second-herdr-daemon setup. |
|
||||
|
||||
## Pages needing work
|
||||
|
||||
### `Home.md` — REVISE
|
||||
|
||||
- Quote: `# claude-bridge` (line 1) and `` `claude-bridge` keeps`` (line 11).
|
||||
The product is `fleet` / `fleetd`. The MCP server identifies itself as `fleet` in
|
||||
`fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java:313-315`.
|
||||
- Quote: `AgentAPI ... swappable fallback injector` (lines 73-76).
|
||||
There is no AgentAPI implementation under `fleetd/src/main/java`; the actual
|
||||
launchers are selected by `Profile.kind` in
|
||||
`fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java:265-270`.
|
||||
|
||||
### `_Sidebar.md` — REVISE
|
||||
|
||||
- Quote: `### 📖 claude-bridge` (line 1).
|
||||
Rename it to `fleet`. `FleetMcp` registers the current product-facing tool set at
|
||||
`fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java:301-326`.
|
||||
|
||||
### `1-Architecture.md` — REBUILD
|
||||
|
||||
- Quote: `` `claude-bridge` lets`` (line 3). The product was renamed; the MCP
|
||||
server name is `fleet` (`FleetMcp.java:313-315`).
|
||||
- Quote: ``fleet_read`` in the tool list (line 102). No such tool is registered.
|
||||
The complete registered list is `fleet_send` through `fleet_whoami` at
|
||||
`FleetMcp.java:301-326`; `fleet_read` is absent.
|
||||
- Quote: `SSE (GET /events)` (line 143). `FleetApp.build()` registers no `/events`
|
||||
route; its routes are listed at `FleetApp.java:143-159`.
|
||||
- Quote: `Redis Streams / NATS JetStream, or an embedded queue` (line 106).
|
||||
The shipped durable inbox is AMQP, configured by `broker`, at
|
||||
`FleetConfig.java:49-50` and `FleetConfig.java:655-714`.
|
||||
- Quote: `AgentAPI (fallback)` (line 107). No AgentAPI adapter exists; shipped
|
||||
launcher kinds are `claude-code` and `opencode` (`FleetConfig.java:265-270`).
|
||||
|
||||
### `2-Message-Server.md` — REBUILD
|
||||
|
||||
- Quote: `claude mcp add --transport http bridge http://127.0.0.1:8080/mcp`
|
||||
(line 67). The daemon defaults to port `8765` in `FleetConfig.java:183-187`,
|
||||
and identifies its server as `fleet` at `FleetMcp.java:313-315`.
|
||||
- Quote: ``fleet_send(message, target?, {block, timeout_seconds, auto_spawn,
|
||||
turn_id})`` (line 80). The real parameters are `sessionId`, `content`,
|
||||
`timeoutMs`, `wait`, `turnId`, and `coordId` (`FleetMcp.java:1096-1108`).
|
||||
- Quote: ``fleet_read(target, source)`` (line 85). It is not registered; see the
|
||||
complete registration at `FleetMcp.java:301-326`.
|
||||
- Quote: `docs/MCP-Contract.md ... normative` (lines 87-88). That is not a valid
|
||||
reference: only §6 is current, as the current operator guide itself says at
|
||||
`.wiki-snapshot/13-User-Guide.md:466`.
|
||||
- Quote: `SSE (GET /events)` (line 45). No route exists in the built REST surface,
|
||||
`FleetApp.java:143-159`.
|
||||
|
||||
### `3-Approaches.md` — REVISE
|
||||
|
||||
- Quote: `AgentAPI ... remains a swappable fallback injector` (lines 78-84).
|
||||
It was never built. The shipped adapter selection is only `claude-code` or
|
||||
`opencode` (`FleetConfig.java:265-270`). Keep it as discarded research, not an
|
||||
operational fallback.
|
||||
- Quote: `claude-bridge` (line 109). Rename the product to `fleet`; the runtime
|
||||
package is `dev.ltms.fleet`, for example `FleetMcp.java:1`.
|
||||
|
||||
### `4-Setup.md` — RETIRE
|
||||
|
||||
It is a 25-line redirect and says its procedure was never written (lines 3-9).
|
||||
Chapter 13 is the maintained install procedure. Keeping a second navigation page
|
||||
adds no working documentation.
|
||||
|
||||
### `5-Operations.md` — RETIRE
|
||||
|
||||
It is a 35-line redirect and says its runbook was never written (lines 3-14).
|
||||
Chapter 13 now owns run and recovery instructions.
|
||||
|
||||
### `6-Team.md` — REBUILD
|
||||
|
||||
- Quote: `fleet_send {role: w-claude, prompt: A}` (line 98). `fleet_send` accepts
|
||||
`sessionId` and `content`, not `role` or `prompt` (`FleetMcp.java:1096-1108`).
|
||||
- Quote: `some on Claude, some on the remote local LLM` (lines 3-5) and `Every
|
||||
worker is ... Claude Code` (line 25). `opencode` is a first-class launcher kind,
|
||||
not a Claude worker (`FleetConfig.java:265-270`).
|
||||
- Quote: `fleetd's concurrency policy` (line 121). The configured capacity control
|
||||
is per-profile `maxLoad` (`FleetConfig.java:251-264`), not the role routing model
|
||||
described here.
|
||||
|
||||
### `7-Use-Cases.md` — REBUILD
|
||||
|
||||
- Quote: `ccs profile` (line 10), `ccs + herdr` (line 22), and `ccs-spawn`
|
||||
(line 45). The configuration has `profiles` and `fleet`, not `ccs`:
|
||||
`FleetConfig.java:34-58` and `FleetConfig.java:81-101`.
|
||||
- Quote: `fleet_send({"to", "kind", "body", "block"})` (lines 55-62).
|
||||
None of those are the shipped send parameters. The schema is
|
||||
`FleetMcp.java:1096-1108`.
|
||||
- Quote: `fleet_list() → { "profiles": ... }` (lines 74-80). `fleet_list` is a
|
||||
roster view; `fleet_profiles` is the configured-backend view, as registered at
|
||||
`FleetMcp.java:307-311` and described at `FleetMcp.java:1176-1182`.
|
||||
|
||||
### `8-Roadmap.md` — REBUILD
|
||||
|
||||
- Quote: `Java 21+` (line 43). The current project guidance and source use Java 25;
|
||||
the `FleetConfig` source itself uses Java 25 unnamed lambda parameters, for
|
||||
example `FleetConfig.java:102`.
|
||||
- Quote: `herdr 0.7.0 / protocol 14` (line 46). The current REST health endpoint
|
||||
reports the live protocol returned by herdr (`FleetApp.java:240-244`), while the
|
||||
current operator guide records protocol 19 at
|
||||
`.wiki-snapshot/13-User-Guide.md:76-85`.
|
||||
- Quote: `ccs <profile> claude` and `ccs env <profile>` (lines 47-48). Shipped
|
||||
configuration uses `Profile` records and launcher `kind`,
|
||||
`FleetConfig.java:313-330` and `FleetConfig.java:265-270`.
|
||||
- Quote: `Redis Streams via Lettuce` (line 50). The actual durable inbox is AMQP
|
||||
`broker`, `FleetConfig.java:655-714`.
|
||||
|
||||
### `9-Implementation.md` — REBUILD
|
||||
|
||||
- Quote: `rest.FleetdApp` and `mcp.BridgeMcp` (lines 29-30). The classes are
|
||||
`rest.FleetApp` and `mcp.FleetMcp` (`FleetApp.java:46`; `FleetMcp.java:67`).
|
||||
- Quote: `dev.ltms.fleetd` (line 67). The source package is `dev.ltms.fleet`
|
||||
(`FleetMcp.java:1`).
|
||||
- Quote: `WorkerPresence` (line 110). The current class is `MemberPresence`, as
|
||||
imported and used by `FleetMcp` at `FleetMcp.java:12` and `465-469`.
|
||||
- Quote: the outcome list ending in `STALE_TURN` (lines 128-131). The code also
|
||||
has `BACKEND_EXHAUSTED` (`FleetMcp.java:550-554`) and async `ASKING` handling
|
||||
(`FleetMcp.java:664-668`).
|
||||
- Quote: `FleetdApp` (line 207) and `FleetdConfig` (line 211). These names do not
|
||||
resolve; current classes are `FleetApp` and `FleetConfig`.
|
||||
|
||||
### `10-Cross-Host-Messaging.md` — REVISE
|
||||
|
||||
- Quote: the chapter says the cross-host fabric is proposed except for the
|
||||
single-host inbox (lines 3-8). Cross-host **lead-to-lead** delivery shipped:
|
||||
`fleet_send` accepts `coordId` (`FleetMcp.java:1094-1107`) and publishes it at
|
||||
`FleetMcp.java:616-641`; configuration has `coordinator` at
|
||||
`FleetConfig.java:74-78` and `99-101`.
|
||||
- Quote: `bridge.dlx` (line 90). This product name is stale. The shipped lead path
|
||||
uses `LeadChannel`, not the proposed exchange flow (`FleetMcp.java:95-96` and
|
||||
`616-641`). Keep the proposed federation design, but add a clear shipped/proposed
|
||||
boundary for CB-637.
|
||||
|
||||
### `11-Features.md` — REVISE
|
||||
|
||||
- Quote: `mcp/BridgeMcp` (line 22), `config/FleetdConfig` (lines 25-27), and other
|
||||
index references. These paths no longer resolve; the source classes are
|
||||
`mcp/FleetMcp` (`FleetMcp.java:67`) and `config/FleetConfig`
|
||||
(`FleetConfig.java:81`).
|
||||
- Quote: `fleet_whoami` returns only `primary` or `worker` (lines 99-100).
|
||||
It also returns `architect` (`FleetMcp.java:1235-1244`).
|
||||
- The page needs the missing separate member-herdr-daemon feature listed below.
|
||||
|
||||
### `12-Claude-to-OpenCode.md` — REVISE
|
||||
|
||||
- Quote: `same bridge mount` (line 5) and `a bridge-spawned worker` (line 94).
|
||||
Rename the product path to `fleet`. The daemon exposes the MCP server as `fleet`
|
||||
(`FleetMcp.java:313-315`), and profiles select OpenCode with `kind: opencode`
|
||||
(`FleetConfig.java:332-335`).
|
||||
- Quote: the sample mount name is `fleetd` (line 67). The server name is `fleet`;
|
||||
update the sample to avoid teaching a second product name.
|
||||
|
||||
### `13-User-Guide.md` — REVISE
|
||||
|
||||
- Quote: `The bridge is the only channel` (line 63). The invariant is correct, but
|
||||
the product term needs the `fleet` rename. The daemon's MCP server name is
|
||||
`fleet` (`FleetMcp.java:313-315`).
|
||||
- Quote: it describes one herdr socket (lines 72-85). It needs the optional
|
||||
`memberHerdrSocket` setup and two-daemon health meaning. The config key is in
|
||||
`FleetConfig.java:34-37`, and `/healthz` checks both daemons when configured at
|
||||
`FleetApp.java:210-245`.
|
||||
|
||||
## MISSING
|
||||
|
||||
`11-Features.md` has a body section for **routing members through a separate herdr daemon**
|
||||
(`## memberHerdrSocket`, line 2174), but **no row in the index table** at the top of the page
|
||||
(lines 20-95). That table is how the page is meant to be read, so a capability absent from it is
|
||||
effectively undiscoverable. Lead note: this is my own omission — I added the section on 2026-08-31
|
||||
and did not add the matching row. Fixed in the wiki at `68e32c6`'s successor.
|
||||
|
||||
The original audit stated the feature had no entry at all. That was wrong: the section exists. The
|
||||
gap is the index row. Recorded here rather than silently corrected, because the difference matters —
|
||||
"undocumented" and "documented but unindexed" are different jobs.
|
||||
|
||||
Evidence for the feature itself: `FleetConfig.java:34-37` and `FleetApp.java:103-115`, `210-245`,
|
||||
and `247-263`.
|
||||
|
||||
## Audit method and coverage
|
||||
|
||||
I checked all 15 pages. I checked concrete tool, route, config, class, file, and
|
||||
product-name claims claim-by-claim on 11 pages: Home, Sidebar, 1, 2, 4, 5, 6, 7, 9,
|
||||
11, and 13. I skimmed the remaining four long historical or research pages (3, 8, 10,
|
||||
12), then checked their concrete claims that affect the verdict. This is an audit of
|
||||
the supplied snapshot, not a wiki rewrite.
|
||||
+155
-12
@@ -114,6 +114,18 @@ bind:
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
|
||||
# Optional socket for member panes. Omit this to use herdrSocket for both leads and members.
|
||||
# memberHerdrSocket: /Users/member/.config/herdr/herdr.sock
|
||||
|
||||
# fleetd #213: the login shell the member OS user (memberHerdrSocket above) actually runs. ONLY
|
||||
# read when memberHerdrSocket is set — fleetd's own $SHELL says nothing about a pane running
|
||||
# under a different OS user, and there is no channel to ask herdr for that user's shell, so this
|
||||
# must be told rather than guessed. Absent, blank, or anything not ending in "zsh" is treated the
|
||||
# same as "not zsh": the memberCredentials.policy: allow-list ZDOTDIR scrub (see worktreeGroup
|
||||
# below) is skipped in favour of the weaker CB-596 sentinel overlay — a degraded control, never a
|
||||
# refusal to spawn. When memberHerdrSocket is absent this key is never consulted at all.
|
||||
# memberLoginShell: /bin/zsh
|
||||
|
||||
# How member sessions are spawned. Define one or more named profiles (backends) under
|
||||
# `profiles`; each key is the profile name (also the ccs profile). A profile says only WHICH
|
||||
# BACKEND — model, CLI adapter, credentials, cost. It says nothing about what a member spawned on
|
||||
@@ -160,7 +172,11 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# skills/MCP/hooks. Omit to leave the worker on the host default.
|
||||
# parityOverlay → repo-relative paths copied primary→worktree so a worker in a provisioned
|
||||
# worktree sees the same local config (CB-301-ext). Omit for the default set:
|
||||
# [.env, .envrc]. (.claude/settings.local.json is NOT in the default — it
|
||||
# [.env] only (CB-148). .envrc is left out of the default on purpose: it is
|
||||
# executable shell that direnv runs on every cd, so copying it carries
|
||||
# behaviour into the worker, not just values, unlike .env. An operator who
|
||||
# wants it copied can still write parityOverlay: [.env, .envrc] explicitly.
|
||||
# (.claude/settings.local.json is NOT in the default — it
|
||||
# pre-approves IDE/tool grants a member must not hold ambiently; CB-525/CB-634.)
|
||||
#
|
||||
# Do NOT add .mcp.json (CB-525). A worker's tools are whatever its launcher
|
||||
@@ -194,6 +210,39 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# profile quarantines alone, under its own name, exactly as if the field did
|
||||
# not exist. Cooldown length is the top-level quarantineCooldownSeconds below.
|
||||
# HOT: read live at every spawn/exhaustion check — no restart needed.
|
||||
# errorPattern → fleetd #201 / #227: regex matched against a completion-fallback scrape to
|
||||
# classify a turn that ended with no fleet_reply as a BACKEND ERROR — a
|
||||
# credential outage or a provider 5xx — rather than a real answer or a
|
||||
# usage-limit exhaustion (exhaustedPattern above always wins when a line
|
||||
# matches both). Opt-in. Omit it and this profile falls back to fleetd's
|
||||
# built-in legacy pattern `(?i)\bAPI Error\s*:` — classification still
|
||||
# happens, just without a profile-specific match; every backend words its
|
||||
# failure differently, so a hardcoded sentence would only ever match one
|
||||
# of them.
|
||||
# DEFERRED: compiled once into a startup pattern map, same as exhaustedPattern
|
||||
# — editing it needs a daemon restart.
|
||||
# # errorPattern: "503 Service Unavailable" # opt-in: classify a backend outage
|
||||
#
|
||||
# What happens once a match fires (BackendOutagePolicy, credentialId-keyed,
|
||||
# SEPARATE from the CB-578 stage B quarantine above and never merged with it):
|
||||
# - threshold 2 — TWO DISTINCT TARGETS (never raw events) on the same
|
||||
# effective credential inside a 60-second window start an "incident" and a
|
||||
# 60-second cool-off for that credential. One member repeating the same
|
||||
# classified line twice never cools anything off — a real outage hits
|
||||
# every target on that credential, so requiring a second, independent
|
||||
# target loses nothing against the case this guards against, while
|
||||
# protecting against a heuristic misfire on one flaky member.
|
||||
# - a fresh error while a credential is already cooling off is ignored
|
||||
# outright: it neither extends the 60s deadline nor starts a new incident.
|
||||
# - `fleet_list`/`fleet_profiles` report a cooling credential with
|
||||
# `coolingOffForSeconds` (never `quarantinedForSeconds`, unless CB-578
|
||||
# exhaustion quarantine is ALSO independently active for the same
|
||||
# credential — the two checks can both fire at once). A spawn onto a
|
||||
# cooling profile is refused with a message naming the credential and
|
||||
# remaining seconds — "cooling off", never "exhausted", so an operator can
|
||||
# tell a short transient fault from a spent subscription at a glance.
|
||||
# - the lead gets ONE nudge per incident (not one per affected target), via
|
||||
# the same push loop that already delivers ticket/question reminders.
|
||||
# env → extra environment for this profile's workers, as a literal key/value map
|
||||
# (CB-511). Use it to give workers a toolchain.
|
||||
#
|
||||
@@ -257,13 +306,49 @@ profiles:
|
||||
# GOTCHA 2 — `maxLoad` is the ONLY throttle you have here. There is no metering, no budget
|
||||
# and no refusal on cost; the cap on live members is the single thing standing between a
|
||||
# fan-out and your monthly limit. Set it deliberately and keep it small.
|
||||
#
|
||||
# GOTCHA 3 (fleetd #176, corrected by fleetd #257) — `maxLoad` counts members, never the lead
|
||||
# itself. The lead is a live `claude` session on this SAME account (a lead is never moved
|
||||
# off-subscription, whatever its own profile says), so it already holds one seat before any
|
||||
# member spawns. If a lead's `fleet.leaders.<name>.profile` names THIS profile — or ANY OTHER
|
||||
# `subscription: true` profile that shares this one's account (see THE SENTINEL, just below,
|
||||
# next to `credentialId:`) — `fleet_list` reports that seat count under `leadSeats`; see
|
||||
# `profile:` under THE FLEET below. `free` itself is NEVER reduced by `leadSeats`: `free` means
|
||||
# "what the real placement gate (`CompositePeerLauncher#enforceMaxLoad`) will actually grant a
|
||||
# fresh `fleet_spawn` right now", and that gate only ever compares live members against
|
||||
# `maxLoad` — it has no notion of the lead's own seat. An earlier cut of this feature
|
||||
# subtracted `leadSeats` from `free` on the theory it made `free` describe the true ceiling on
|
||||
# the account, but no backend seat ceiling shared with the lead has ever actually been
|
||||
# measured, and the subtraction just made `free` disagree with the one thing it is supposed to
|
||||
# describe — the fleetd #257 fix. `maxLoad: 3` means 3 member slots, full stop; a lead sharing
|
||||
# the account is a fact you can see in `leadSeats`, not a reason `free` undercounts spawns that
|
||||
# will, in practice, succeed.
|
||||
#
|
||||
# THE SENTINEL (fleetd #176 stage 2, correcting an inert stage 1 fix): every `subscription:
|
||||
# true` profile that leaves `credentialId` unset shares ONE implicit account-wide credential
|
||||
# id with every other such profile on this host — because a subscription profile doesn't
|
||||
# authenticate with a credential of its own, it authenticates as the operator's own Claude
|
||||
# login, and there is exactly one of those. So on a typical host, `opus` (the lead's profile)
|
||||
# and `sonnet` (the members' profile) are linked automatically, with NOTHING to set here — that
|
||||
# is what makes GOTCHA 3 above work without also writing matching `credentialId:` values on
|
||||
# both. This linkage is not just cosmetic: it is the same key `BackendQuarantine`/cool-off use,
|
||||
# so a usage-limit hit on `opus` now quarantines `sonnet` too (and vice versa) — correct, since
|
||||
# they are one Claude account, but worth knowing before you wonder why an unrelated-looking
|
||||
# profile went quarantined.
|
||||
#
|
||||
# WHEN TO OVERRIDE — set explicit, DIFFERENT `credentialId:` values on two `subscription: true`
|
||||
# profiles only when they are genuinely two separate Claude logins on the same host (a real,
|
||||
# if unusual, setup). An explicit `credentialId` always wins over the sentinel, so this is the
|
||||
# one way to keep two subscription profiles from being treated as one account for lead-seat
|
||||
# counting AND for quarantine/cool-off grouping alike.
|
||||
# gitTokenEnv: GITEA_TOKEN # opt-in: let this profile's workers open their own PR (CB-302)
|
||||
# gitHostEnv: GITEA_HOST # defaults to GITEA_HOST; injected only with gitTokenEnv
|
||||
# exhaustedPattern: "usage limit has been reached" # opt-in: classify a usage-limit refusal (CB-578)
|
||||
# credentialId: shared-openai # opt-in: quarantine together with every other profile sharing this id (CB-578)
|
||||
# errorPattern: "503 Service Unavailable" # opt-in: classify a backend outage (fleetd #201/#227) — see the key doc above
|
||||
# configDir: /Users/me/.ccs/instances/gx10 # CLAUDE_CONFIG_DIR — inherit that profile's skills/MCP
|
||||
# cwd: /Users/me/src/myrepo # pin the working dir; omit to inherit the primary's
|
||||
# parityOverlay: [".env", ".envrc"] # the default; never add .mcp.json or .claude/settings.local.json — see above
|
||||
# parityOverlay: [".env"] # the default; add ".envrc" explicitly if you want it copied too (CB-148) — never add .mcp.json or .claude/settings.local.json — see above
|
||||
# ideMcpUrl: http://127.0.0.1:29170/index-mcp/streamable-http # opt-in (CB-634): IDE code intelligence, pinned to the worktree
|
||||
# ideProjectDir: fleetd # CB-634: module dir the IDE opens + the overlay pins (this repo's pom is in fleetd/)
|
||||
# ideOpenCommand: env DISPLAY=:10.0 idea {dir} # CB-634 auto-open: opens {dir} in the IDE at spawn; omit to open by hand
|
||||
@@ -357,6 +442,12 @@ placement: weighted
|
||||
# DEFERRED: baked once into the BackendQuarantine built at startup — a running quarantine keeps
|
||||
# its original cooldown regardless; a new value only applies to a quarantine that starts after a
|
||||
# restart. Editing this needs a daemon restart to take effect.
|
||||
#
|
||||
# This does NOT govern the fleetd #201 / #227 backend-error cool-off documented under errorPattern
|
||||
# above — that mechanism is a separate, shorter-lived, NOT-configurable policy (threshold 2 distinct
|
||||
# targets, 60-second window, 60-second cool-off), on purpose: it exists to survive a brief transient
|
||||
# fault, not to replace this 30-minute exhaustion quarantine. Do not conflate the two when reading
|
||||
# fleet_list/fleet_profiles — coolingOffForSeconds and quarantinedForSeconds are independent facts.
|
||||
# quarantineCooldownSeconds: 1800
|
||||
|
||||
# Re-read this file without restarting the daemon (CB-559). Off unless you add this block, so an
|
||||
@@ -383,8 +474,10 @@ placement: weighted
|
||||
# stage B — baked once into the quarantine tracker built at startup), ADDING or
|
||||
# REMOVING a profile (a new backend needs its own launcher, and launchers are built
|
||||
# once), AND an existing profile's launch settings — model, baseUrl, argv, env,
|
||||
# configDir, mcpUrl, tabLabel, exhaustedPattern. The launcher takes a copy of
|
||||
# `profiles:` at startup and resolves every spawn out of that copy, so those never
|
||||
# configDir, mcpUrl, tabLabel, exhaustedPattern, errorPattern (fleetd #201 / #227 —
|
||||
# compiled once into a startup pattern map the same way exhaustedPattern is). The
|
||||
# launcher takes a copy of `profiles:` at startup and resolves every spawn out of
|
||||
# that copy, so those never
|
||||
# reach a launch until you restart. The reload logs them by name rather than
|
||||
# pretending they applied.
|
||||
# COLD → cannot change at all: `bind:`, `herdrSocket:`, `broker:` and `auth:`. The socket is
|
||||
@@ -445,6 +538,23 @@ fleet:
|
||||
# recognised: give it a `profile:` and the daemon launches the shortfall when fewer than
|
||||
# `instances` are live. Omit `profile:` and it is recognise-only, as before.
|
||||
#
|
||||
# `profile:` has a SECOND job as of fleetd #176, even for a recognise-only lead you never want
|
||||
# auto-launched: it is also how fleetd learns which account this lead's own session shares. A
|
||||
# `subscription: true` profile bills the operator's Claude account, and the lead itself is always
|
||||
# a live `claude` session on that same account — `maxLoad` never counted that seat. If a lead
|
||||
# entry here names a profile that shares a worker profile's account, `fleet_list` reports the
|
||||
# lead's live seat(s) on that worker profile under `leadSeats` — informational only, as of fleetd
|
||||
# #257 it is NEVER subtracted from `free` (see GOTCHA 3, next to `maxLoad:`, in THE WORKERS above,
|
||||
# for why). "Shares the account" is decided by matching `effectiveCredentialId()`, which (fleetd
|
||||
# #176 stage 2 — see THE SENTINEL, next to `credentialId:`, in THE WORKERS above) means: an
|
||||
# explicit, matching `credentialId:` on both, OR — the common case, needing NO extra config — both
|
||||
# being `subscription: true` with `credentialId` left unset, since those all share one implicit
|
||||
# account-wide id. A lead on `opus` and workers on `sonnet` link automatically this way; they do
|
||||
# NOT need the same profile name. Setting `profile:` on an already-running, recognise-only lead is
|
||||
# safe — the daemon only launches the SHORTFALL below `instances`, so naming a profile here does
|
||||
# not, by itself, start anything. Omit it and fleetd has no way to derive the sharing — there is
|
||||
# no other reliable signal on the daemon's side — so that lead's seat never appears in `leadSeats`.
|
||||
#
|
||||
# `tab:` (CB-579) is REQUIRED and is the only field identity depends on — the exact label of the
|
||||
# tab hosting the lead, matched case-insensitively. Label the tab yourself and put that same
|
||||
# string here, and the pane is recognised on the next rescan. Reopen the tab later, or the session
|
||||
@@ -566,20 +676,35 @@ guard:
|
||||
# every name here NOT also in `allow` is overlaid with a non-secret sentinel value before
|
||||
# the pane's login shell runs — real protection only for names that shell does not itself
|
||||
# re-export (see the ROUND-2 CORRECTION note above). Under allow-list: reporting only.
|
||||
# sshAuthSock → whether SSH_AUTH_SOCK may pass through under allow-list ("allow") or must be
|
||||
# blanked like any other non-derived name ("block", the default). This is a decision you
|
||||
# have to make explicitly: SSH_AUTH_SOCK is a handle to YOUR ssh-agent, and a member
|
||||
# holding it can sign with your keys — it sits in no secret file and looks like no
|
||||
# credential, which is why it slipped past three earlier tickets (gitea #110). Blocking
|
||||
# it breaks git over SSH inside members (push/fetch authenticate as you); use HTTPS
|
||||
# remotes or scoped deploy keys instead of allowing it lightly.
|
||||
# sshAgentEnv → whether SSH_AUTH_SOCK may pass through under allow-list ("inherit") or is omitted
|
||||
# from the member environment ("omit", the default). Omitting it only omits the
|
||||
# inherited ssh-agent path. It discourages automatic use of the operator's agent.
|
||||
# It does not deny same-user access to that socket. It also does not block SSH keys that
|
||||
# are readable on disk. Git over SSH may still work from inside a member. Keep the block:
|
||||
# it is correct and costs nothing, but it is not a control. A member runs as the same OS
|
||||
# user as the lead. Inside one uid, ordinary Unix permissions provide no meaningful
|
||||
# confidentiality boundary. A real boundary needs a different OS user or OS-level
|
||||
# confinement, such as a container or VM. That is the open question in fleetd #184.
|
||||
#
|
||||
# Still do not set this to "inherit" casually. SSH_AUTH_SOCK is a live handle to YOUR
|
||||
# ssh-agent, so a member holding it can sign with EVERY key the agent holds. It sits in
|
||||
# no secret file and looks like no credential, which is why it slipped past three
|
||||
# earlier tickets (gitea #110). Blocking it does not contain a member, but allowing it
|
||||
# hands one a signing capability for no gain — the block costs nothing, so keep it.
|
||||
#
|
||||
# Both halves of this are measured, not argued. 2026-08-28: a member with
|
||||
# SSH_AUTH_SOCK blanked pushed to the forge over SSH successfully, because `ssh -G`
|
||||
# resolves an IdentityFile outside ~/.ssh that is readable and has no passphrase. An
|
||||
# earlier version of this comment claimed blocking the socket BREAKS git over SSH. It
|
||||
# does not. That claim came from looking only in ~/.ssh, which holds nothing but four
|
||||
# `Include` lines — looking in one place and concluding about the whole host.
|
||||
#
|
||||
# HOT-RELOADABLE the same way `fleet:` is (CB-559): read fresh on every spawn, so editing this list
|
||||
# and reloading config (or restarting) changes what the NEXT spawn inherits; already-running members
|
||||
# are unaffected either way.
|
||||
# memberCredentials:
|
||||
# policy: deny-by-default # or "deny-list", or "allow-list" (CB-633) — see above
|
||||
# sshAuthSock: block # allow-list only; see the sshAuthSock note above
|
||||
# sshAgentEnv: omit # allow-list only; see the sshAgentEnv note above
|
||||
# allow:
|
||||
# - AI_GATEWAY_TOKEN # named in a profile's tokenEnv (local/gx) — a member reaching the
|
||||
# # gateway is by design, not a leak
|
||||
@@ -631,6 +756,24 @@ guard:
|
||||
# to a sibling directory of the repo root.
|
||||
# worktreeRoot: /Users/me/src/.bridged-worktrees
|
||||
|
||||
# Worktree group sharing (fleetd #185 stage 3). OPTIONAL, off by default. Names an OS group
|
||||
# that a provisioned worktree's repo is made group-writable for (git config
|
||||
# core.sharedRepository group, plus a one-time chgrp/chmod/setgid fix-up), so a member spawned
|
||||
# under a DIFFERENT OS user (see memberHerdrSocket) can write its own worktree, its
|
||||
# per-worktree git metadata, and its own commit objects — without it, every file GitWorktrees
|
||||
# creates is owned by fleetd's own uid and unwritable by another user.
|
||||
# CAUTION: this isolates credentials, not the repository — a member in the group can still
|
||||
# write the operator's git objects and refs in the shared repo. The operator running fleetd
|
||||
# must already be a member of the named group, or every provisioning spawn fails loudly.
|
||||
#
|
||||
# fleetd #213: this is also the ONE group the memberCredentials.policy: allow-list ZDOTDIR scrub
|
||||
# reuses when memberHerdrSocket is set — deliberately not a second config key. Under
|
||||
# memberHerdrSocket, the scrub directory is generated under worktreeRoot (never java.io.tmpdir,
|
||||
# which the member OS user cannot reach) and shared read-only with this group. If worktreeGroup
|
||||
# is unset while memberHerdrSocket is set, the scrub cannot be guaranteed reachable by the member,
|
||||
# so fleetd falls back to the weaker CB-596 sentinel overlay instead (a WARN names the gap).
|
||||
# worktreeGroup: fleet-workers
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
# the feature disabled. By default the daemon never reaps, caps, or drains sessions.
|
||||
# idleTtlSeconds → reap READY/DONE sessions idle longer than this (never BUSY/SPAWNING)
|
||||
|
||||
@@ -28,6 +28,8 @@
|
||||
<testcontainers.version>1.20.4</testcontainers.version>
|
||||
<commons-compress.version>1.27.1</commons-compress.version>
|
||||
<commons-lang3.version>3.18.0</commons-lang3.version>
|
||||
<sqlite-jdbc.version>3.53.4.0</sqlite-jdbc.version>
|
||||
<archunit.version>1.5.0</archunit.version>
|
||||
</properties>
|
||||
|
||||
<!--
|
||||
@@ -44,6 +46,12 @@
|
||||
3.0-rc5; bumping Jackson 3 to the patched 3.2.x breaks the SDK (annotation mismatch).
|
||||
Only the loopback /mcp endpoint parses this JSON, from trusted local Claude clients.
|
||||
The 11.0.23 -> 11.0.25 bump did clear jetty CVE-2024-8184 (5.9) and CVE-2024-6763.
|
||||
|
||||
fleetd #206: org.xerial:sqlite-jdbc 3.53.4.0 (added for OpenCodeSessionDiscovery) — the
|
||||
only known advisory against this artifact is CVE-2023-32697 (RCE via an attacker-controlled
|
||||
JDBC URL), fixed in 3.41.2.2; 3.53.4.0 is well past that fix and OSV.dev reports no open
|
||||
advisory against it. Checked via the OSV.dev API (no Mend.io/JetBrains IDE MCP mount
|
||||
available from this worktree) on 2026-08-31.
|
||||
-->
|
||||
|
||||
<!-- Force the latest patched Jetty 11.x across all Javalin-pulled Jetty modules (no version
|
||||
@@ -123,6 +131,17 @@
|
||||
<version>${amqp.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- fleetd #206: opencode moved its session store from a JSON tree to SQLite
|
||||
(opencode.db). This is the JDBC driver OpenCodeSessionDiscovery uses to read it
|
||||
read-only. Ships bundled native libraries (linux/mac/windows, several archs), so it
|
||||
is a heavier jar than most deps here — see the pom's dependency-security note below
|
||||
for the size/CVE tradeoff actually measured. -->
|
||||
<dependency>
|
||||
<groupId>org.xerial</groupId>
|
||||
<artifactId>sqlite-jdbc</artifactId>
|
||||
<version>${sqlite-jdbc.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Logging -->
|
||||
<dependency>
|
||||
<groupId>org.slf4j</groupId>
|
||||
@@ -158,6 +177,14 @@
|
||||
<version>${testcontainers.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
|
||||
<!-- fleetd #131: package-boundary and cycle enforcement (PackageCyclesTest). -->
|
||||
<dependency>
|
||||
<groupId>com.tngtech.archunit</groupId>
|
||||
<artifactId>archunit-junit5</artifactId>
|
||||
<version>${archunit.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
|
||||
<build>
|
||||
|
||||
@@ -7,11 +7,14 @@ import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||
import dev.ltms.fleet.herdr.LeadTabScanner;
|
||||
import dev.ltms.fleet.lead.LeadLauncher;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.herdr.UnixSocketHerdrClient;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.BackendErrorPatternLookup;
|
||||
import dev.ltms.fleet.inject.BackendErrorSink;
|
||||
import dev.ltms.fleet.inject.CompletionResolver;
|
||||
import dev.ltms.fleet.inject.ExhaustedPatternLookup;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
@@ -48,7 +51,9 @@ import dev.ltms.fleet.session.SessionReaper;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.member.HerdrPeerLauncher;
|
||||
import dev.ltms.fleet.member.MemberCredentialPolicyView;
|
||||
import dev.ltms.fleet.member.OpenCodeLauncher;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import io.javalin.Javalin;
|
||||
import org.slf4j.Logger;
|
||||
@@ -60,6 +65,7 @@ import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
@@ -115,6 +121,7 @@ public final class Fleetd {
|
||||
// else can fail on a silently-empty one. A daemon started without a login shell (launchd)
|
||||
// boots fine either way — this is the only thing that says so out loud.
|
||||
reportRequiredSecrets(cfg);
|
||||
reportMemberTrustModel(cfg);
|
||||
// CB-596: an absent (or empty) memberCredentials: block blocks NOTHING — no credential
|
||||
// name is hardcoded any more to fall back on. Say so loudly, the same way a missing
|
||||
// secret is reported above, so upgrading past this commit never silently drops CB-592's
|
||||
@@ -149,9 +156,12 @@ public final class Fleetd {
|
||||
: UnixSocketHerdrClient.defaultSocketPath();
|
||||
|
||||
UnixSocketHerdrClient herdr = UnixSocketHerdrClient.connect(socket, new com.fasterxml.jackson.databind.ObjectMapper());
|
||||
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
WorkspaceControl spaces = new WorkspaceControl(herdr);
|
||||
UnixSocketHerdrClient memberHerdr = cfg.memberHerdrSocket() != null && !cfg.memberHerdrSocket().isBlank()
|
||||
? UnixSocketHerdrClient.connect(Path.of(cfg.memberHerdrSocket()), new com.fasterxml.jackson.databind.ObjectMapper())
|
||||
: herdr;
|
||||
AtomicReference<Supplier<Map<String, String>>> leadsRef = new AtomicReference<>(Map::of);
|
||||
HerdrRouter router = new HerdrRouter(herdr, memberHerdr,
|
||||
target -> leadsRef.get().get().containsKey(target));
|
||||
// CB-402: one adapter per configured peer kind, fronted by a composite router. A profile's
|
||||
// `kind:` selects its adapter — claude-code (the default) and opencode partition the profile
|
||||
// set — and the composite dispatches each SPI call to the adapter that owns the profile/pane.
|
||||
@@ -165,22 +175,34 @@ public final class Fleetd {
|
||||
}
|
||||
});
|
||||
List<HerdrPeerLauncher> adapters = new ArrayList<>();
|
||||
// fleetd #175: the daemon's real ExhaustionSink can only be built once `sessions` exists
|
||||
// (below), but `sessions` needs `workers`, which needs the adapters built right here — a
|
||||
// genuine cycle. Break it exactly like liveCountRef below: a forwarding sink built now,
|
||||
// pointed at the real one once it exists.
|
||||
AtomicReference<ExhaustionSink> exhaustionSinkRef = new AtomicReference<>(ExhaustionSink.none());
|
||||
// fleetd #234, round 4: routed through the shared ExhaustionSink.forwardingTo factory
|
||||
// rather than written inline here — not because a lambda at this call site is unsafe
|
||||
// anymore (it is not: the 3-arg overload is now the interface's single abstract method, so
|
||||
// there is no 2-arg overload left for any lambda to silently bind to instead), but so a
|
||||
// test can call the exact same object this line builds, instead of asserting a copy of its
|
||||
// shape (round 3's lesson).
|
||||
ExhaustionSink forwardingExhaustionSink = ExhaustionSink.forwardingTo(exhaustionSinkRef::get);
|
||||
// The claude-code adapter is the always-present default; keep it even with no profiles (so a
|
||||
// bridge configured with no workers, or opencode-only, still has a well-defined base adapter)
|
||||
// unless opencode is the only kind configured.
|
||||
if (!claudeProfiles.isEmpty() || opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new ClaudeCodeLauncher(agents, spaces, guard,
|
||||
adapters.add(new ClaudeCodeLauncher(router.memberAgents(), router.memberSpaces(), guard,
|
||||
claudeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet(),
|
||||
() -> config.get().memberCredentials()));
|
||||
() -> config.get().memberCredentials(), null, config::get));
|
||||
}
|
||||
if (!opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new OpenCodeLauncher(agents, spaces,
|
||||
adapters.add(new OpenCodeLauncher(router.memberAgents(), router.memberSpaces(),
|
||||
opencodeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet(),
|
||||
() -> config.get().memberCredentials()));
|
||||
() -> config.get().memberCredentials(), config::get, forwardingExhaustionSink));
|
||||
}
|
||||
AtomicReference<Function<String, Integer>> liveCountRef = new AtomicReference<>(_ -> 0);
|
||||
// CB-578 stage B: one quarantine tracker for the whole daemon, shared between the launcher
|
||||
@@ -189,12 +211,18 @@ public final class Fleetd {
|
||||
// here, at startup, and a config reload only changes it for a daemon restart.
|
||||
BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));
|
||||
// fleetd #201 Unit 5: one outage-cool-off tracker for the whole daemon, shared between the
|
||||
// launcher (checked at spawn, like `quarantine` above) and the backend-error sink wired in
|
||||
// below (written on a classified backend error). A SEPARATE, shorter-lived mechanism from
|
||||
// `quarantine` — see BackendOutagePolicy's class doc — never merged with it.
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(System::nanoTime);
|
||||
PeerLauncher workers = new CompositePeerLauncher(
|
||||
adapters,
|
||||
cfg.effectiveDefaultProfile(),
|
||||
config,
|
||||
profileName -> liveCountRef.get().apply(profileName),
|
||||
quarantine);
|
||||
quarantine,
|
||||
outagePolicy);
|
||||
// CB-504: under supervision (launchd/systemd) fleetd can start before herdr's socket
|
||||
// exists. The client itself is lazy — it connects per call — but the orphan reap below is
|
||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||
@@ -220,11 +248,9 @@ public final class Fleetd {
|
||||
contextCap = cfg.lifecycle().contextCap();
|
||||
}
|
||||
boolean clearAfterTurn = cfg.lifecycle() != null && cfg.lifecycle().clearAfterTurn();
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(cfg.worktreeRoot()),
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(cfg.worktreeRoot(), cfg.worktreeGroup()),
|
||||
System::nanoTime, contextCap, clearAfterTurn);
|
||||
liveCountRef.set(profileName -> (int) sessions.roster().stream()
|
||||
.filter(s -> profileName.equals(s.profile()))
|
||||
.count());
|
||||
liveCountRef.set(profileName -> liveSessionCount(sessions.roster(), profileName));
|
||||
|
||||
// CB-303 part 1: idle-ttl reaper — only when configured, defaults to disabled.
|
||||
final SessionReaper reaper;
|
||||
@@ -271,6 +297,7 @@ public final class Fleetd {
|
||||
// operational cadence, not identity, so there is no correctness reason to give every
|
||||
// lead its own scanner.
|
||||
int scanIntervalSeconds = leaders.values().iterator().next().scanIntervalSeconds();
|
||||
// This must use the lead daemon: scanning member tabs would demote the lead to a worker.
|
||||
leads = new LeadTabScanner(herdr, tabToName, Set.of(),
|
||||
TimeUnit.SECONDS.toNanos(scanIntervalSeconds), System::nanoTime);
|
||||
log.info("lead scan: tabs {} host a lead (rescan every {}s, shared fleet space)",
|
||||
@@ -278,13 +305,14 @@ public final class Fleetd {
|
||||
} else {
|
||||
leads = () -> leadTerminals;
|
||||
}
|
||||
leadsRef.set(leads);
|
||||
|
||||
// CB-558: start any declared lead that is not already running. After the scanner is built,
|
||||
// because both read the same tab labels and the ordering makes that dependency visible; and
|
||||
// only when herdr answered, because the launcher's whole safety property is that it can
|
||||
// count live leads first — it must never guess and risk a second orchestrator.
|
||||
if (herdrUp && !leaders.isEmpty()) {
|
||||
int launched = new LeadLauncher(agents, spaces, cfg).ensureLeads();
|
||||
int launched = new LeadLauncher(router.leadAgents(), router.leadSpaces(), cfg).ensureLeads();
|
||||
if (launched > 0) {
|
||||
log.info("lead auto-launch: {} lead(s) started", launched);
|
||||
}
|
||||
@@ -323,24 +351,100 @@ public final class Fleetd {
|
||||
.map(session -> exhaustedPatternsByProfile.get(session.profile()))
|
||||
.orElse(null);
|
||||
log.info("backend-exhausted classification (CB-578 stage A): {}",
|
||||
CompletionResolver.coverage(cfg.profiles().keySet(), exhaustedPatternsByProfile.keySet()));
|
||||
CompletionResolver.coverage("exhaustedPattern", cfg.profiles().keySet(),
|
||||
exhaustedPatternsByProfile.keySet()));
|
||||
// fleetd #201 Unit 5: classify a completion-fallback scrape that matches a profile's
|
||||
// configured backend-error refusal (a credential outage, a provider 5xx) as a backend error
|
||||
// rather than handing it back as a real answer. Compiled once at startup, keyed by profile
|
||||
// name, mirroring exhaustedPatternsByProfile above — a profile with no configured
|
||||
// errorPattern is simply absent here, so CompletionResolver falls back to its built-in
|
||||
// narrow {@code (?i)\bAPI Error\s*:} compatibility pattern for that profile's targets
|
||||
// (BackendErrorPatternLookup#legacy's contract — see backendErrorPatterns below).
|
||||
Map<String, Pattern> errorPatternsByProfile = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (profile.hasErrorPattern()) {
|
||||
errorPatternsByProfile.put(name, Pattern.compile(profile.errorPattern()));
|
||||
}
|
||||
});
|
||||
// fleetd #248: extracted to a static factory (see backendErrorPatternLookup below) so a
|
||||
// test can prove main() actually PASSES this into CompletionResolver, not only that the
|
||||
// lookup itself behaves correctly — the exact gap fleetd #248 exists to close.
|
||||
BackendErrorPatternLookup backendErrorPatterns = backendErrorPatternLookup(sessions::roster,
|
||||
errorPatternsByProfile);
|
||||
log.info("backend-error classification (fleetd #201 Unit 5): {}",
|
||||
CompletionResolver.coverage("errorPattern", cfg.profiles().keySet(),
|
||||
errorPatternsByProfile.keySet()));
|
||||
// CB-578 stage B: on a classification that actually wins, quarantine the exhausted profile's
|
||||
// CREDENTIAL — not the profile name — so a profile sharing that credential (e.g. two models
|
||||
// on one OpenAI account) is refused too, not just the one that happened to report it. Reads
|
||||
// the profile config live off `config`, so a credentialId edit is hot: no restart needed.
|
||||
ExhaustionSink exhaustionSink = (target, reason) -> sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(MemberSession::profile)
|
||||
.map(profileName -> config.get().profiles().get(profileName))
|
||||
.ifPresent(profile -> {
|
||||
String credentialId = profile.effectiveCredentialId();
|
||||
quarantine.quarantine(credentialId);
|
||||
log.warn("credential '{}' quarantined for {}s (profile '{}' classified "
|
||||
+ "BACKEND_EXHAUSTED): {}", credentialId,
|
||||
cfg.quarantineCooldownSeconds(), profile.profile(), reason);
|
||||
});
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous, exhaustedPatterns, exhaustionSink);
|
||||
//
|
||||
// fleetd #175: this sink is now also the quarantine target for OpenCodeLauncher's
|
||||
// model-mismatch check — it needs nothing profile-specific from the caller beyond `target`
|
||||
// (a herdr terminal id) and `reason`, so reusing it here is exactly "the existing
|
||||
// ExhaustionSink path", not a new mechanism.
|
||||
//
|
||||
// fleetd #234: that check fires from SessionAwareHandle.agentSessionId(), which runs during
|
||||
// SessionManager.acquire() BEFORE this session is registered in sessions.roster() — so the
|
||||
// roster-only lookup below used to find nothing, .ifPresent silently no-op'd, and the
|
||||
// ERROR the check had just logged ("quarantining this profile's credential") was a lie:
|
||||
// nothing was quarantined, and nothing said so. Two changes: (1) OpenCodeLauncher now
|
||||
// passes its OWN profile name via ExhaustionSink's 3-arg overload — it already has the
|
||||
// FleetConfig.Profile in hand and does not need the roster at all — used here as a
|
||||
// fallback whenever the roster lookup misses; (2) if a profile still cannot be resolved
|
||||
// (neither the roster nor the hint names a configured one), this logs loudly at ERROR
|
||||
// instead of silently doing nothing — a control that cannot act must say so.
|
||||
// fleetd #234, round 4: the 3-arg overload is now ExhaustionSink's single abstract method,
|
||||
// so this is safely a lambda — there is no separate 2-arg overload left for it to bind to
|
||||
// instead and silently drop profileHint (that was rounds 1-3's whole hazard).
|
||||
ExhaustionSink exhaustionSink = (target, reason, profileHint) -> {
|
||||
String profileName = sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(MemberSession::profile)
|
||||
.orElse(profileHint);
|
||||
FleetConfig.Profile profile = profileName == null ? null : config.get().profiles().get(profileName);
|
||||
if (profile == null) {
|
||||
log.error("quarantine requested for target '{}' ({}) but no profile could be "
|
||||
+ "resolved — the target is not (yet) in the roster, and {} — "
|
||||
+ "credential NOT quarantined (fleetd #234)",
|
||||
target, reason,
|
||||
profileHint == null ? "no profile hint was given"
|
||||
: "the hinted profile '" + profileHint + "' is not configured");
|
||||
return;
|
||||
}
|
||||
String credentialId = profile.effectiveCredentialId();
|
||||
quarantine.quarantine(credentialId);
|
||||
log.warn("credential '{}' quarantined for {}s (profile '{}'): {}", credentialId,
|
||||
cfg.quarantineCooldownSeconds(), profile.profile(), reason);
|
||||
};
|
||||
// fleetd #175: point the forwarding sink handed to OpenCodeLauncher above at the real one,
|
||||
// now that `sessions` exists to resolve target -> session -> profile.
|
||||
exhaustionSinkRef.set(exhaustionSink);
|
||||
// fleetd #201 Unit 5: the production BackendErrorSink needs `pushLoop` (built further below,
|
||||
// after `sessions`) to tell a lead about an incident or an unmapped target — the same
|
||||
// construction-order cycle `exhaustionSinkRef` breaks above, broken the same way: a mutable
|
||||
// holder set once `pushLoop` exists, read lazily from inside the lambda built here.
|
||||
AtomicReference<ReplyPushLoop> pushLoopRef = new AtomicReference<>();
|
||||
// fleetd #248: extracted to a static factory (see backendErrorSink below), public rather
|
||||
// than package-private like the other two factories here, so
|
||||
// dev.ltms.fleet.inject.BackendOutageFlowTest can exercise the REAL production sink
|
||||
// directly instead of a hand-mirrored copy of this lambda — that copy was precisely the
|
||||
// gap fleetd #248 exists to close (see that test's class doc for the history).
|
||||
BackendErrorSink backendErrorSink = backendErrorSink(sessions, () -> config.get().profiles(),
|
||||
outagePolicy, pushLoopRef::get);
|
||||
AgentControl agents = router.memberAgents();
|
||||
// Both fleetd#201 Unit 5 (backend-error patterns + sink) and fleetd#241 (the worktree/branch
|
||||
// lookup the fallback report names) land on this one call. The full constructor takes both,
|
||||
// so neither feature is dropped; nowNanos must be passed explicitly to reach it.
|
||||
//
|
||||
// fleetd #248: every argument built specifically for this call (backendErrorPatterns and
|
||||
// backendErrorSink above, and the worktree/branch lookup right here) now comes from a
|
||||
// static factory tested on its own; FleetdCompletionResolverWiringTest source-asserts that
|
||||
// THIS call actually passes them, which is the coverage that was missing before.
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous, exhaustedPatterns,
|
||||
exhaustionSink, backendErrorPatterns, backendErrorSink, System::nanoTime,
|
||||
worktreeBranchLookup(sessions::roster));
|
||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||
MemberPresence presence = sessions.asPresence();
|
||||
@@ -381,9 +485,9 @@ public final class Fleetd {
|
||||
}
|
||||
};
|
||||
Predicate<String> deliverable = deliverableTo(presence, leads);
|
||||
Injector injector = new Injector(agents, turnListener, deliverable,
|
||||
Injector injector = new Injector(router, turnListener, deliverable,
|
||||
presence::forget);
|
||||
StatusPoller poller = new StatusPoller(agents, injector, Injector.POLL_INTERVAL_MILLIS);
|
||||
StatusPoller poller = new StatusPoller(router, injector, Injector.POLL_INTERVAL_MILLIS);
|
||||
poller.start();
|
||||
|
||||
// CB-307: reply inbox. A broker: block selects the AMQP-backed durable adapter; absent (or
|
||||
@@ -420,8 +524,11 @@ public final class Fleetd {
|
||||
// are counted at their single funnel rather than at each of the two caller-facing surfaces.
|
||||
// CB-512: the push loop takes it too, so nudge outcomes (delivered|exhausted) are counted.
|
||||
Metrics metrics = FleetMetrics.create(sessions, replyInbox);
|
||||
var pushLoop = new ReplyPushLoop(primaryRegistry, agents, replyInbox,
|
||||
var pushLoop = new ReplyPushLoop(primaryRegistry, router.leadAgents(), replyInbox,
|
||||
pushScheduler, maxReminders, backoffMs, metrics);
|
||||
// fleetd #201 Unit 5: point the forwarding holder captured by the backendErrorSink lambda
|
||||
// above at the real push loop, now that it exists.
|
||||
pushLoopRef.set(pushLoop);
|
||||
// CB-551: idle-lead heartbeat. Opt-in; absent `leadHeartbeat:` this is never constructed, so
|
||||
// an upgraded daemon cannot silently start spending subscription on nudging an idle lead.
|
||||
// It has its own single-thread scheduler and holds its own scheduler shutdown via close().
|
||||
@@ -430,7 +537,7 @@ public final class Fleetd {
|
||||
Thread.ofVirtual().name("bridge-heartbeat-").unstarted(r));
|
||||
if (cfg.leadHeartbeat() != null) {
|
||||
var hb = cfg.leadHeartbeat();
|
||||
heartbeat = new LeadHeartbeatLoop(primaryRegistry, agents, replyInbox, sessions::roster,
|
||||
heartbeat = new LeadHeartbeatLoop(primaryRegistry, router.leadAgents(), replyInbox, sessions::roster,
|
||||
pushLoop, heartbeatScheduler, System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(hb.idleAfterSeconds()), hb.backoffMs(), hb.quietNudgeCap(),
|
||||
metrics);
|
||||
@@ -439,7 +546,7 @@ public final class Fleetd {
|
||||
heartbeat = null;
|
||||
heartbeatScheduler.shutdownNow();
|
||||
}
|
||||
MessageService messages = new MessageService(agents, injector, rendezvous, replyInbox,
|
||||
MessageService messages = new MessageService(router, injector, rendezvous, replyInbox,
|
||||
pushLoop, metrics);
|
||||
|
||||
// Health is a slow whole-fleet observer. Keep it separate from the 250ms delivery poller.
|
||||
@@ -484,15 +591,22 @@ public final class Fleetd {
|
||||
if (detail.agentSessionId() != null) {
|
||||
reason += " agentSessionId=" + detail.agentSessionId();
|
||||
}
|
||||
messages.abandon(detail.terminalId(), reason);
|
||||
// fleetd #275: this is an explicit teardown (fleet_stop, or the idle reaper) — the
|
||||
// worker's pane is being stopped right now, so an open fleet_ask has no turn left to
|
||||
// resume into. Sweep it too, unlike FleetHealthMonitor's health-classification call
|
||||
// (see MessageService.abandon's javadoc for why those two must differ).
|
||||
messages.abandon(detail.terminalId(), reason, true);
|
||||
replyInbox.release(detail.terminalId());
|
||||
primaryRegistry.forgetDelegation(detail.terminalId()); // CB-532: don't leak the lead binding
|
||||
});
|
||||
|
||||
// MCP server face (CB-105): fleet_send/fleet_reply/fleet_status, mounted at /mcp.
|
||||
// Caller identity is resolved from the connection (peer PID → herdr pane), not arguments.
|
||||
// CB-185: a caller's pane can live on either daemon (a lead's on the lead daemon, a
|
||||
// member's on the member daemon) — search both, lead first. Collapses to one scan when
|
||||
// memberHerdrSocket is unset (herdr == memberHerdr).
|
||||
ConnectionIdentity identity = new ConnectionIdentity(
|
||||
new PaneLocator(herdr), new LsofPeerPidLookup(), new LsofProcessCwdLookup());
|
||||
new PaneLocator(herdr, memberHerdr), new LsofPeerPidLookup(), new LsofProcessCwdLookup());
|
||||
|
||||
// CB-501: one resolver behind both entry paths. Worker identity still comes from the
|
||||
// connection and is never token-gated, so enabling token mode cannot lock the fleet out.
|
||||
@@ -511,6 +625,19 @@ public final class Fleetd {
|
||||
log.info("auth: loopback-trust (any loopback non-worker caller is the primary)");
|
||||
}
|
||||
|
||||
// fleetd #297: named once and reused verbatim below for FleetApp's GET /profiles, rather than
|
||||
// built a second time — two independently-constructed sources reading the SAME BackendQuarantine
|
||||
// / BackendOutagePolicy would still be able to drift (e.g. a future edit to the credentialIdFor
|
||||
// closure in only one of the two places), exactly the shape #284 was.
|
||||
FleetMcp.QuarantineSource quarantineSource = new FleetMcp.QuarantineSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, quarantine);
|
||||
FleetMcp.OutageSource outageSource = new FleetMcp.OutageSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, outagePolicy);
|
||||
|
||||
FleetMcp mcp = new FleetMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, metrics, new FleetMcp.CapacitySource(profile -> liveCountRef.get().apply(profile),
|
||||
profile -> {
|
||||
@@ -522,11 +649,10 @@ public final class Fleetd {
|
||||
return FleetHealthMonitor.coverage(health != null && health.isEnabled(),
|
||||
health != null && health.notifications() != null && health.notifications().configured());
|
||||
}),
|
||||
new FleetMcp.QuarantineSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, quarantine),
|
||||
leadMailbox);
|
||||
quarantineSource,
|
||||
leadMailbox,
|
||||
outageSource,
|
||||
new FleetMcp.LeadSeatSource(leadSeatLookup(() -> config.get().profiles(), leaders, leads)));
|
||||
|
||||
// CB-637: the receive half. Only constructed when a lead mailbox actually opened — with no
|
||||
// coordinator (or an unreachable one) there is nothing to deliver, so no scheduler is
|
||||
@@ -538,7 +664,7 @@ public final class Fleetd {
|
||||
if (leadMailbox != null) {
|
||||
var leadCoordScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-leadcoord-").unstarted(r));
|
||||
leadCoordLoop = new LeadCoordLoop(leadMailbox, agents, leads, leadCoordScheduler,
|
||||
leadCoordLoop = new LeadCoordLoop(leadMailbox, router.leadAgents(), leads, leadCoordScheduler,
|
||||
LEAD_COORD_INTERVAL_MS);
|
||||
leadCoordLoop.start();
|
||||
leadCoordSchedulerRef = leadCoordScheduler;
|
||||
@@ -589,11 +715,19 @@ public final class Fleetd {
|
||||
log.debug("lead mailbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
herdr.close();
|
||||
router.close();
|
||||
}));
|
||||
|
||||
Javalin app = new FleetApp(herdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics, deliverable).build();
|
||||
// CB-185: give FleetApp both daemons — /healthz must require both to answer and
|
||||
// GET /sessions must merge across both, or a down/unpolled member daemon is invisible.
|
||||
// fleetd #111: live (re-read-per-request) memberCredentials view for GET /member-credentials —
|
||||
// same hot-reload shape as the memberCredentials supplier passed to ClaudeCodeLauncher above.
|
||||
// fleetd #297: quarantineSource/outageSource are the SAME instances passed to FleetMcp above —
|
||||
// GET /profiles must report the identical quarantine/cool-off facts as fleet_profiles.
|
||||
Javalin app = new FleetApp(herdr, memberHerdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics, deliverable,
|
||||
() -> MemberCredentialPolicyView.of(config.get().memberCredentials()),
|
||||
quarantineSource, outageSource).build();
|
||||
app.start(cfg.bind().host(), cfg.bind().port());
|
||||
log.info("fleetd listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
@@ -622,6 +756,198 @@ public final class Fleetd {
|
||||
return target -> presence.isPresent(target) || leads.get().containsKey(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #248: package-private factory for the member worktree/branch lookup {@link
|
||||
* CompletionResolver} uses to name a fallback report's worktree and branch (fleetd#241).
|
||||
*
|
||||
* <p>Before this ticket the lookup was an anonymous lambda built inline inside {@code main}'s
|
||||
* {@code CompletionResolver} constructor call — provably untested wiring, the whole reason
|
||||
* fleetd #248 exists: dropping that one argument (passing {@code _ -> null} instead) compiled
|
||||
* clean and left every test green. Extracted here, {@code main} now calls this factory instead
|
||||
* of building the lambda inline, and a source assertion on that call site
|
||||
* ({@code FleetdCompletionResolverWiringTest}) proves the argument is still actually passed.
|
||||
*
|
||||
* <p>Takes the roster as a plain {@link Supplier} — not a {@link SessionManager} — so this is
|
||||
* directly testable with a hand-built session list; no real {@code SessionManager} (launcher,
|
||||
* worktrees, …) needs constructing. Follows the same {@code static} factory pattern as
|
||||
* {@link #deliverableTo} above.
|
||||
*
|
||||
* @param roster the live member roster, normally {@code sessions::roster}
|
||||
*/
|
||||
static Function<String, CompletionResolver.WorktreeBranch> worktreeBranchLookup(
|
||||
Supplier<List<MemberSession>> roster) {
|
||||
return target -> roster.get().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(session -> new CompletionResolver.WorktreeBranch(session.worktree(), session.branch()))
|
||||
.orElse(null);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176: per-profile factory for {@link FleetMcp.LeadSeatSource} — how many seats a
|
||||
* profile's own live LEAD session(s) hold on the same Claude subscription.
|
||||
*
|
||||
* <p>{@code maxLoad} counts only members; the lead itself is a live {@code claude} session that
|
||||
* is never moved off-subscription ({@code LeadLauncher} strips {@code ANTHROPIC_BASE_URL}/
|
||||
* {@code AUTH_TOKEN} from a lead's env whatever its profile says), so a {@code subscription:
|
||||
* true} profile's real ceiling is lower than its configured {@code maxLoad} by exactly the
|
||||
* number of lead seats sharing that same account.
|
||||
*
|
||||
* <p><b>The derivation, and why this route was chosen over a new config key.</b> The link is
|
||||
* {@code fleet.leaders.<name>.profile} — the field the operator already sets to name which
|
||||
* {@code profiles:} entry a lead runs on (CB-557; see {@code fleetd.example.yaml}) — matched
|
||||
* against the profile passed in here via {@link FleetConfig.Profile#effectiveCredentialId()},
|
||||
* the same grouping key {@link dev.ltms.fleet.placement.BackendQuarantine} already uses to say
|
||||
* two profiles share one account. Nothing new is added to the config schema: this reuses a field
|
||||
* that already exists and already means "the profile this lead's own session runs on". A lead
|
||||
* entry that names no {@code profile:} (recognise-only, CB-558) says nothing about which account
|
||||
* it shares, and there is no other reliable signal on the daemon's side to derive that from — so
|
||||
* such a lead contributes no seats, exactly as before this ticket. Making that lead's seat count
|
||||
* requires the operator to add one line (`profile: sonnet` under its {@code fleet.leaders} entry)
|
||||
* — a config statement, not a code change, and the smallest one available given the field
|
||||
* already exists for a closely related purpose.
|
||||
*
|
||||
* <p>Only counts leads {@code liveLeadTerminals} currently reports — CB-531's live tab scan (or
|
||||
* the legacy {@code primary.terminal} pin) — never every configured lead: an entry whose
|
||||
* {@code instances} nobody has actually started is not really competing for a seat, and must not
|
||||
* shrink capacity for one that was never live.
|
||||
*
|
||||
* @param profiles the live profile map, normally {@code () -> config.get().profiles()}
|
||||
* in {@code main} — hot, like every other {@code maxLoad}/
|
||||
* {@code credentialId} read {@link FleetMcp.CapacitySource} already does
|
||||
* @param leaders {@code fleet.leaders}, read once at startup like the rest of that
|
||||
* block ({@code fleetd.example.yaml} notes it is not hot) — passed as a
|
||||
* plain map, never re-read from {@code config.get()}
|
||||
* @param liveLeadTerminals terminal_id → lead name for every CURRENTLY recognised lead, normally
|
||||
* the same supplier {@link dev.ltms.fleet.auth.CallerResolver#leads()}
|
||||
* and {@code LeadCoordLoop} already consult
|
||||
*/
|
||||
static Function<String, Integer> leadSeatLookup(Supplier<Map<String, FleetConfig.Profile>> profiles,
|
||||
Map<String, FleetConfig.Leader> leaders, Supplier<Map<String, String>> liveLeadTerminals) {
|
||||
return profileName -> {
|
||||
FleetConfig.Profile target = profiles.get().get(profileName);
|
||||
if (target == null || !target.isSubscription()) {
|
||||
return 0;
|
||||
}
|
||||
String targetCredential = target.effectiveCredentialId();
|
||||
int seats = 0;
|
||||
for (String leadName : liveLeadTerminals.get().values()) {
|
||||
FleetConfig.Leader lead = leaders.get(leadName);
|
||||
if (lead == null || lead.profile() == null || lead.profile().isBlank()) {
|
||||
continue;
|
||||
}
|
||||
FleetConfig.Profile leadProfile = profiles.get().get(lead.profile());
|
||||
if (leadProfile != null && targetCredential.equals(leadProfile.effectiveCredentialId())) {
|
||||
seats++;
|
||||
}
|
||||
}
|
||||
return seats;
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #248 / fleetd#201 Unit 5: package-private factory for the per-target backend-error
|
||||
* pattern lookup {@link CompletionResolver} classifies a pane scrape against. Closes over the
|
||||
* live roster (to resolve a target to a profile) and {@code errorPatternsByProfile} (each
|
||||
* profile's configured {@code errorPattern}, already compiled by the caller — the same map also
|
||||
* feeds the coverage log next to where this is called) — nothing else, so it is directly
|
||||
* testable. See {@link #worktreeBranchLookup} above for why this ticket exists and why the
|
||||
* factory takes a roster {@link Supplier} rather than a {@link SessionManager}.
|
||||
*
|
||||
* @param roster the live member roster, normally {@code sessions::roster}
|
||||
* @param errorPatternsByProfile every profile that has an {@code errorPattern} configured,
|
||||
* keyed by profile name
|
||||
*/
|
||||
static BackendErrorPatternLookup backendErrorPatternLookup(Supplier<List<MemberSession>> roster,
|
||||
Map<String, Pattern> errorPatternsByProfile) {
|
||||
return target -> roster.get().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(session -> errorPatternsByProfile.get(session.profile()))
|
||||
.orElse(null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Count sessions that occupy a profile's spawn capacity. A {@code BACKEND_ERROR} or
|
||||
* {@code FAILED} session stays in the roster so {@code fleet_list} can show its failure, but a
|
||||
* member that cannot accept another delivery does not use a seat.
|
||||
*/
|
||||
static int liveSessionCount(List<MemberSession> roster, String profileName) {
|
||||
return (int) roster.stream()
|
||||
.filter(session -> profileName.equals(session.profile()))
|
||||
.filter(session -> session.state() != MemberSession.State.BACKEND_ERROR)
|
||||
.filter(session -> session.state() != MemberSession.State.FAILED)
|
||||
.count();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #248 / fleetd#201 Unit 5: factory for the production {@link BackendErrorSink} — the
|
||||
* collaborator {@link CompletionResolver} notifies when a pane-scrape classification actually
|
||||
* resolves a waiter as a backend error. Order: (1) mark the member BACKEND_ERROR; (2) resolve
|
||||
* profile/credential through the roster — fail loud (never {@code Optional.ifPresent}, the
|
||||
* fleetd #234 lesson applied to this sink) and notify the lead via {@code
|
||||
* onBackendTargetUnmapped} when it cannot be resolved; (3) record the error in {@code
|
||||
* outagePolicy}; (4) on a NEW incident (the {@code record()} call that actually crosses the
|
||||
* threshold), tell the lead via {@code onBackendIncident}.
|
||||
*
|
||||
* <p>{@code public}, unlike {@link #worktreeBranchLookup} and {@link #backendErrorPatternLookup}
|
||||
* above: {@code dev.ltms.fleet.inject.BackendOutageFlowTest} exercises this exact object as the
|
||||
* real, wired production path, replacing what its own class doc used to call out as a
|
||||
* hand-mirrored copy of this lambda ("mirrors {@code Fleetd.main}'s {@code backendErrorSink}
|
||||
* lambda line-for-line") — that copy proved only itself, never that {@code main} still wires the
|
||||
* real thing. That was precisely the gap fleetd #248 exists to close.
|
||||
*
|
||||
* @param sessions the session registry; both read (roster) and written (onBackendError)
|
||||
* @param profiles the live profile map, normally {@code () -> config.get().profiles()} in
|
||||
* {@code main}, or a fixed test map via {@code () -> profiles}
|
||||
* @param pushLoop the lead-nudge loop, read lazily: {@code main} builds this sink before the
|
||||
* real {@link ReplyPushLoop} exists (a genuine construction-order cycle, broken
|
||||
* the same way {@code exhaustionSinkRef} is a few lines above it), so a
|
||||
* {@link Supplier} reads whatever {@code main} has filled in by the time a real
|
||||
* backend error fires
|
||||
*/
|
||||
public static BackendErrorSink backendErrorSink(SessionManager sessions,
|
||||
Supplier<Map<String, FleetConfig.Profile>> profiles, BackendOutagePolicy outagePolicy,
|
||||
Supplier<ReplyPushLoop> pushLoop) {
|
||||
return (target, matchedLine, reason) -> {
|
||||
sessions.onBackendError(target, reason);
|
||||
|
||||
String profileName = sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(MemberSession::profile)
|
||||
.orElse(null);
|
||||
FleetConfig.Profile profile = profileName == null ? null : profiles.get().get(profileName);
|
||||
if (profile == null) {
|
||||
log.error("backend error on target '{}' ({}) but no profile could be resolved — the "
|
||||
+ "target is not (yet) in the roster — no cool-off applied (fleetd #201 Unit 5)",
|
||||
target, reason);
|
||||
ReplyPushLoop loop = pushLoop.get();
|
||||
if (loop != null) {
|
||||
loop.onBackendTargetUnmapped(target, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
String credentialId = profile.effectiveCredentialId();
|
||||
Optional<BackendOutagePolicy.Incident> incident = outagePolicy.record(credentialId, target, reason);
|
||||
incident.ifPresent(inc -> {
|
||||
List<String> affectedProfiles = profiles.get().values().stream()
|
||||
.filter(p -> credentialId.equals(p.effectiveCredentialId()))
|
||||
.map(FleetConfig.Profile::profile)
|
||||
.sorted()
|
||||
.toList();
|
||||
log.warn("credential '{}' cooling off for {}s after backend errors on {} distinct "
|
||||
+ "target(s) (profile '{}'): {}", credentialId,
|
||||
inc.remainingCoolOffSeconds(), inc.evidenceCount(), profile.profile(), reason);
|
||||
ReplyPushLoop loop = pushLoop.get();
|
||||
if (loop != null) {
|
||||
loop.onBackendIncident(inc.id(), inc.targets(), credentialId, affectedProfiles,
|
||||
(int) inc.remainingCoolOffSeconds());
|
||||
}
|
||||
});
|
||||
};
|
||||
}
|
||||
|
||||
/** Injection seam for {@link #selectReplyInbox}: production binds {@link AmqpReplyInbox#open}. */
|
||||
@FunctionalInterface
|
||||
interface AmqpOpener {
|
||||
@@ -770,7 +1096,7 @@ public final class Fleetd {
|
||||
|
||||
/**
|
||||
* CB-594: which env vars the loaded config actually needs, and why — every non-{@code
|
||||
* subscription} profile's {@code tokenEnv} (a subscription profile never reads one, see
|
||||
* subscription} profile's explicitly configured {@code tokenEnv} (a subscription profile never reads one, see
|
||||
* {@link FleetConfig.Profile#isSubscription()}), plus every profile's {@code gitTokenEnv}
|
||||
* where set (opt-in), plus a configured {@code broker.uriEnv} (CB-151). Derived from the
|
||||
* config, not hard-coded, so a new profile is covered for free. A var required by more than one
|
||||
@@ -787,7 +1113,9 @@ public final class Fleetd {
|
||||
static Map<String, List<String>> requiredSecretEnvVars(FleetConfig cfg) {
|
||||
Map<String, List<String>> requiredBy = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (!profile.isSubscription()) {
|
||||
// All kinds are checked when tokenEnv is explicit. OpenCode may use provider credentials,
|
||||
// but an explicit tokenEnv still declares a required host secret for its configured provider.
|
||||
if (!profile.isSubscription() && profile.hasTokenEnv()) {
|
||||
requiredBy.computeIfAbsent(profile.tokenEnv(), _ -> new ArrayList<>())
|
||||
.add("profile '" + name + "' tokenEnv");
|
||||
}
|
||||
@@ -813,7 +1141,7 @@ public final class Fleetd {
|
||||
* <p>A missing entry only warns — it must never refuse to start. A daemon that boots and says
|
||||
* what is wrong is strictly more useful than one that will not boot at all.
|
||||
*/
|
||||
private static void reportRequiredSecrets(FleetConfig cfg) {
|
||||
static void reportRequiredSecrets(FleetConfig cfg) {
|
||||
Map<String, List<String>> requiredBy = requiredSecretEnvVars(cfg);
|
||||
if (requiredBy.isEmpty()) {
|
||||
log.info("startup secrets: no profile references a token env var — nothing to check");
|
||||
@@ -834,6 +1162,24 @@ public final class Fleetd {
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #184: state the member trust model at startup. Environment controls and worktrees do
|
||||
* not make a sandbox when fleetd and its members use the same OS user. A separate herdr may
|
||||
* provide that boundary, but fleetd cannot inspect the uid at the other end of its socket.
|
||||
*/
|
||||
static void reportMemberTrustModel(FleetConfig cfg) {
|
||||
if (cfg.memberHerdrSocket() != null && !cfg.memberHerdrSocket().isBlank()) {
|
||||
log.info("member trust model: members are routed to a separate herdr through "
|
||||
+ "memberHerdrSocket. fleetd cannot see that herdr's uid, so confirm it runs "
|
||||
+ "as a different OS user before treating it as a boundary.");
|
||||
return;
|
||||
}
|
||||
log.info("member trust model: members run as the same OS user as fleetd, not in a sandbox. "
|
||||
+ "A member can read any file this user can read, including SSH keys and credential "
|
||||
+ "stores, whatever memberCredentials says. To add a real boundary, route members to "
|
||||
+ "a second herdr under a different OS user with memberHerdrSocket.");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-596: {@code known:} empty (block absent entirely, or present but empty) means {@link
|
||||
* FleetConfig.MemberCredentials#blockedSet()} is empty too — every member pane inherits the
|
||||
@@ -847,12 +1193,14 @@ public final class Fleetd {
|
||||
* #requiredSecretEnvVars} is exposed for {@link #reportRequiredSecrets}'s own test.
|
||||
*/
|
||||
static void reportMemberCredentialsGap(FleetConfig cfg) {
|
||||
FleetConfig.MemberCredentials creds = cfg.memberCredentials();
|
||||
if (creds != null && !creds.known().isEmpty()) {
|
||||
// fleetd #111: the counts below come from MemberCredentialPolicyView, the same class the
|
||||
// live GET /member-credentials endpoint reads — one place computes them, not two.
|
||||
MemberCredentialPolicyView view = MemberCredentialPolicyView.of(cfg.memberCredentials());
|
||||
if (view.present()) {
|
||||
log.info("memberCredentials: policy={}, {} known name(s), {} allowed — blocking {} on "
|
||||
+ "every spawn{}",
|
||||
creds.policy(), creds.known().size(), creds.allow().size(), creds.blockedSet().size(),
|
||||
creds.isAllowList()
|
||||
view.policy(), view.knownCount(), view.allowedCount(), view.blockedCount(),
|
||||
cfg.memberCredentials().isAllowList()
|
||||
? " (allow-list: known/allow are reporting only — the control is the derived ZDOTDIR scrub)"
|
||||
: "");
|
||||
return;
|
||||
|
||||
@@ -262,11 +262,15 @@ public final class CallerResolver {
|
||||
return token.isEmpty() ? null : token;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #305: delegates to {@link ConnectionIdentity#isLoopback}. This used to be a second,
|
||||
* independent copy of the same rule, and the two drifted: this one accepted all of
|
||||
* {@code 127.0.0.0/8}, {@code ConnectionIdentity}'s accepted only {@code 127.0.0.1}. A caller
|
||||
* from {@code 127.0.0.2} therefore had its identity skipped (so it had no terminal) and was
|
||||
* then read as loopback here — which under loopback-trust is the primary. Sharing the inputs
|
||||
* would not have prevented that; only sharing the computation does.
|
||||
*/
|
||||
private static boolean isLoopback(String remoteAddr) {
|
||||
if (remoteAddr == null) {
|
||||
return false;
|
||||
}
|
||||
return remoteAddr.equals("127.0.0.1") || remoteAddr.equals("::1")
|
||||
|| remoteAddr.equals("0:0:0:0:0:0:0:1") || remoteAddr.startsWith("127.");
|
||||
return ConnectionIdentity.isLoopback(remoteAddr);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,17 +5,70 @@ import dev.ltms.fleet.peer.MemberRole;
|
||||
/** Optional session lifecycle hook for live member-slot bindings. */
|
||||
public interface MemberLifecycle {
|
||||
|
||||
/** A slot held before a member process starts. */
|
||||
record SlotReservation(String slot, String profile) { }
|
||||
|
||||
MemberLifecycle NONE = new MemberLifecycle() {
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
return role; // no registry configured — nothing to bind against, so the request stands
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
// no registry configured — nothing to validate against, so nothing is refused
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
return null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
}
|
||||
};
|
||||
|
||||
void acquired(MemberRole role, String profile, String terminal);
|
||||
/**
|
||||
* Try to bind a newly spawned {@code terminal} into the role it was granted.
|
||||
*
|
||||
* @return the role this session actually holds: {@code role} unchanged for a role with no
|
||||
* live slot-binding semantics (dev, reviewer), or when the bind succeeded; a fallback
|
||||
* role — never {@code role} — when a slot-bound role (architect) could not be bound.
|
||||
* Callers must record THIS value on the session, never the requested {@code role}, so
|
||||
* a later roster read never reports a role the session does not hold (CB-619). In
|
||||
* normal operation this fallback should not happen once a reservation has been bound.
|
||||
* It remains the honest answer if a caller has no reservation, or if binding a
|
||||
* reservation unexpectedly fails.
|
||||
*/
|
||||
MemberRole acquired(MemberRole role, String profile, String terminal);
|
||||
|
||||
void released(String terminal);
|
||||
|
||||
/**
|
||||
* Refuse an acquire before anything spawns when {@code role} requires a live slot binding and
|
||||
* no configured slot carries {@code profile} (CB-619 / fleetd #123). A no-op for a role with
|
||||
* no slot-binding semantics.
|
||||
*
|
||||
* @throws IllegalArgumentException naming the role, the profile, and the pools that do carry it
|
||||
*/
|
||||
void requireSlotFor(MemberRole role, String profile);
|
||||
|
||||
/** Reserve a matching slot before launch, or refuse before a charter can be delivered. */
|
||||
SlotReservation reserve(MemberRole role, String profile);
|
||||
|
||||
/** Convert a reservation into a live terminal binding. */
|
||||
boolean bind(SlotReservation reservation, String terminal);
|
||||
|
||||
/** Return an unbound reservation after a failed launch. */
|
||||
void release(SlotReservation reservation);
|
||||
}
|
||||
|
||||
@@ -8,6 +8,7 @@ import org.slf4j.LoggerFactory;
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
@@ -55,8 +56,10 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
}
|
||||
|
||||
private final Map<String, Entry> slots;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code this}. */
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code terminalToSlot}. */
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
/** Slot keys held between reservation and the terminal binding. Guarded by terminalToSlot. */
|
||||
private final java.util.Set<String> reservedSlots = new java.util.HashSet<>();
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one registry. Leaders are not members. */
|
||||
public MemberRegistry(FleetConfig.Fleet fleet) {
|
||||
@@ -166,7 +169,7 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
if (existingSlot != null) {
|
||||
return slot.equals(existingSlot); // already this slot (idempotent) or a different one
|
||||
}
|
||||
if (terminalToSlot.containsValue(slot)) {
|
||||
if (terminalToSlot.containsValue(slot) || reservedSlots.contains(slot)) {
|
||||
return false; // slot already hosts a terminal — no second one
|
||||
}
|
||||
terminalToSlot.put(terminal, slot);
|
||||
@@ -206,19 +209,116 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
*
|
||||
* <p>The role check is lifecycle policy. {@link CallerResolver} repeats it when resolving a
|
||||
* binding, so a later lifecycle regression cannot turn a worker into an architect.
|
||||
*
|
||||
* <p>CB-619 / fleetd #123: the return value is the role this session actually holds, and the
|
||||
* caller is required to record THAT — never the requested {@code role} — on the session. Before
|
||||
* this fix the caller kept the requested role regardless of whether the bind below succeeded, so
|
||||
* a demoted session's {@code GET /members} row still said {@code "architect"} while
|
||||
* {@code fleet_whoami} (which reads the live binding, not the request) correctly said
|
||||
* {@code "worker"} — three sources of truth that disagreed about one live member, silently.
|
||||
*/
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
if (role != MemberRole.ARCHITECT || terminal == null || terminal.isBlank()) {
|
||||
return;
|
||||
return role;
|
||||
}
|
||||
// slotsFor preserves definition order, so duplicate-profile slots use the first free one.
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if (Objects.equals(profile, entry.profile()) && bind(entry.key(), terminal)) {
|
||||
return;
|
||||
return MemberRole.ARCHITECT;
|
||||
}
|
||||
}
|
||||
// fleetd #123: at least WARN — a role downgrade that the roster must now also reflect is
|
||||
// not routine bookkeeping. requireSlotFor already refuses the config-gap case (no slot at
|
||||
// all carries this profile) before a process ever spawns; reaching here means the config DID
|
||||
// carry a matching slot but every one of them was already bound to a different terminal — a
|
||||
// race this pre-spawn check cannot close on its own (see requireSlotFor's javadoc).
|
||||
log.warn("member slot: no free architect slot for profile={} terminal={}; holding the session "
|
||||
+ "as {} instead of the architect it asked for — every configured slot for this "
|
||||
+ "profile is already bound to a different terminal", profile, terminal,
|
||||
MemberRole.DEV.wireName());
|
||||
return MemberRole.DEV;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-619 / fleetd #123: refuse an architect acquire before anything spawns when no configured
|
||||
* slot carries {@code profile} — the config-gap case from the original defect report (a spawn
|
||||
* asked for {@code role=architect, profile=sonnet}, and {@code fleet.architects} carried only
|
||||
* {@code opus} and {@code sol}). A dev/reviewer acquire is always a no-op: those pools are
|
||||
* placement candidates only (see {@code CompositePeerLauncher}), never a live identity binding,
|
||||
* so there is nothing here to refuse — an explicit profile outside the pool for those roles is a
|
||||
* documented operator override, not a defect.
|
||||
*
|
||||
* <p>This closes the config-gap case, not the live-capacity case: a profile that DOES carry a
|
||||
* slot can still lose the race to a concurrent spawn between this check and the actual
|
||||
* {@link #bind}, which is why {@link #acquired} must still answer honestly even after this
|
||||
* check has passed.
|
||||
*/
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return;
|
||||
}
|
||||
boolean hasSlot = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.anyMatch(e -> Objects.equals(profile, e.profile()));
|
||||
if (hasSlot) {
|
||||
return;
|
||||
}
|
||||
List<String> pools = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.map(Entry::profile)
|
||||
.distinct()
|
||||
.toList();
|
||||
throw new IllegalArgumentException(
|
||||
"no " + role.wireName() + " slot for profile '" + profile + "' — an architect's "
|
||||
+ "identity IS the slot it is bound to, so there is nothing to bind this "
|
||||
+ "session's identity to. fleet." + role.configKey() + " carries profiles: "
|
||||
+ (pools.isEmpty() ? "(none configured)" : String.join(", ", pools))
|
||||
+ "; add profile '" + profile + "' there, or spawn " + role.wireName()
|
||||
+ " on one of those profiles instead");
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return null;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if ((profile == null || profile.isBlank() || Objects.equals(profile, entry.profile()))
|
||||
&& !terminalToSlot.containsValue(entry.key()) && reservedSlots.add(entry.key())) {
|
||||
return new SlotReservation(entry.key(), entry.profile());
|
||||
}
|
||||
}
|
||||
}
|
||||
throw new IllegalArgumentException("no free architect slot for profile '" + profile
|
||||
+ "' — every matching slot is already bound or reserved");
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
if (reservation == null || terminal == null || terminal.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
if (!reservedSlots.remove(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
if (!isSlot(reservation.slot()) || terminalToSlot.containsKey(terminal)
|
||||
|| terminalToSlot.containsValue(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
terminalToSlot.put(terminal, reservation.slot());
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
if (reservation != null) {
|
||||
synchronized (terminalToSlot) {
|
||||
reservedSlots.remove(reservation.slot());
|
||||
}
|
||||
}
|
||||
log.info("member slot: no free architect slot for profile={}; session remains a worker", profile);
|
||||
}
|
||||
|
||||
/** Unbind a released terminal using the compare-safe registry operation. */
|
||||
|
||||
@@ -43,7 +43,9 @@ import java.util.function.Supplier;
|
||||
* which is constructed once), <em>and an existing profile's launch settings</em> —
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl},
|
||||
* {@code exhaustedPattern} (CB-578 stage A — compiled once into {@code Fleetd.main}'s
|
||||
* pattern map at startup), and the rest. {@code credentialId} (CB-578 stage B) is NOT on
|
||||
* pattern map at startup), {@code errorPattern} (fleetd #201 Unit 5 — compiled once into
|
||||
* {@code Fleetd.main}'s backend-error pattern map at startup, the same way), and the rest.
|
||||
* {@code credentialId} (CB-578 stage B) is NOT on
|
||||
* this list — it is read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so it is hot instead.
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} at construction and resolves
|
||||
@@ -71,7 +73,7 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
|
||||
/** Keys that cannot change under a running daemon — see the class doc. */
|
||||
private static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "broker", "auth");
|
||||
Set.of("bind", "herdrSocket", "memberHerdrSocket", "broker", "auth");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<FleetConfig> current;
|
||||
@@ -190,6 +192,9 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
if (!Objects.equals(old.herdrSocket(), fresh.herdrSocket())) {
|
||||
changed.add("herdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.memberHerdrSocket(), fresh.memberHerdrSocket())) {
|
||||
changed.add("memberHerdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.broker(), fresh.broker())) {
|
||||
changed.add("broker");
|
||||
}
|
||||
@@ -289,6 +294,10 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
// CB-578 stage B: exhaustedPattern is compiled once into Fleetd.main's pattern map
|
||||
// at startup (see ExhaustedPatternLookup wiring) — a reload never re-reads it, so a
|
||||
// changed pattern must be reported as deferred, exactly like model/baseUrl/argv.
|
||||
&& Objects.equals(a.exhaustedPattern(), b.exhaustedPattern());
|
||||
&& Objects.equals(a.exhaustedPattern(), b.exhaustedPattern())
|
||||
// fleetd #201 Unit 5: errorPattern is compiled once into Fleetd.main's backend-error
|
||||
// pattern map at startup (see BackendErrorPatternLookup wiring), the same way
|
||||
// exhaustedPattern is — a reload never re-reads it either.
|
||||
&& Objects.equals(a.errorPattern(), b.errorPattern());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import com.fasterxml.jackson.annotation.JsonCreator;
|
||||
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
|
||||
import com.fasterxml.jackson.annotation.JsonProperty;
|
||||
import com.fasterxml.jackson.core.JsonParser;
|
||||
import com.fasterxml.jackson.core.JsonToken;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
@@ -24,6 +26,8 @@ import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.regex.PatternSyntaxException;
|
||||
|
||||
/**
|
||||
* {@code fleetd} configuration, loaded from a YAML file (see
|
||||
@@ -32,7 +36,8 @@ import java.util.Set;
|
||||
* silently dropping a whole block is indistinguishable from honouring it.
|
||||
*
|
||||
* @param bind REST/MCP listen host:port
|
||||
* @param herdrSocket path to herdr's Unix socket ({@code null} → client default)
|
||||
* @param herdrSocket path to the lead herdr Unix socket ({@code null} → client default)
|
||||
* @param memberHerdrSocket optional member herdr Unix socket ({@code null}/blank → lead socket)
|
||||
* @param profiles named backend profiles, keyed by profile name (multi-backend fleet). A
|
||||
* profile answers <em>which backend</em> — model, CLI adapter, credentials,
|
||||
* cost. It says nothing about what the member spawned on it is for; that is
|
||||
@@ -75,11 +80,38 @@ import java.util.Set;
|
||||
* stay on {@code broker}'s vhost). {@code null} → no lead mailbox is opened.
|
||||
* Config parsing + accessors only — nothing here wires it into a live
|
||||
* {@code LeadMailbox}; that is a separate ticket. See {@link Coordinator}.
|
||||
* @param worktreeGroup optional OS group name (fleetd #185 stage 3) that makes a provisioned
|
||||
* worktree's repo group-shared, so a member running as a different OS user
|
||||
* (see {@code memberHerdrSocket}) can write its own worktree, its per-worktree
|
||||
* git metadata, and its own commit objects. {@code null}/blank/empty ⇒ off,
|
||||
* today's behaviour unchanged (every file stays owned by fleetd's own uid).
|
||||
* <strong>This isolates credentials, not the repository</strong>: a member in
|
||||
* the group can still write the operator's git objects and refs in the shared
|
||||
* repo. See {@link dev.ltms.fleet.session.Worktrees#shareWithGroup}.
|
||||
* @param memberLoginShell fleetd #213: the login shell the member's OS user actually runs, ONLY
|
||||
* meaningful (and only ever read) when {@code memberHerdrSocket} is
|
||||
* configured — that mode spawns member panes under a different OS user than
|
||||
* fleetd's own process, so fleetd's own {@code $SHELL} says nothing about what
|
||||
* that pane runs. There is no channel to ask herdr for another user's shell, so
|
||||
* this must be told, never guessed. {@code null}/blank (or a value not ending
|
||||
* in {@code zsh}) is treated the same as "not zsh". fleetd #155: what that means
|
||||
* now depends on {@code memberCredentials.policy}. Under {@code deny-by-default}
|
||||
* it stays a degraded control, never a refusal to spawn — the pane-creation
|
||||
* overlay is unaffected by shell type, so the spawn proceeds with a WARN naming
|
||||
* the shell. Under {@code allow-list} the spawn is REFUSED instead: that policy's
|
||||
* whole point is a control a sourced file cannot undo, so silently falling back
|
||||
* to the weaker overlay would be the same "control silently does nothing" defect
|
||||
* #155 exists to remove — configure this field (or move the account to zsh, or
|
||||
* switch policy back to {@code deny-by-default}) to unblock the spawn.
|
||||
* When {@code memberHerdrSocket} is NOT configured this field is never
|
||||
* consulted at all; fleetd keeps reading its own {@code $SHELL}, exactly as
|
||||
* before this field existed.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record FleetConfig(
|
||||
Bind bind,
|
||||
String herdrSocket,
|
||||
String memberHerdrSocket,
|
||||
Map<String, Profile> profiles,
|
||||
Guard guard,
|
||||
String worktreeRoot,
|
||||
@@ -96,7 +128,33 @@ public record FleetConfig(
|
||||
ConfigReload configReload,
|
||||
Integer quarantineCooldownSeconds,
|
||||
MemberCredentials memberCredentials,
|
||||
Coordinator coordinator) {
|
||||
Coordinator coordinator,
|
||||
String worktreeGroup,
|
||||
String memberLoginShell) {
|
||||
|
||||
/** Back-compat form before the {@code memberLoginShell} key was added. */
|
||||
public FleetConfig(Bind bind, String herdrSocket, String memberHerdrSocket, Map<String, Profile> profiles,
|
||||
Guard guard, String worktreeRoot, Lifecycle lifecycle, Integer spawnReadyTimeoutMs,
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, Health health, String placement, Auth auth,
|
||||
ConfigReload configReload, Integer quarantineCooldownSeconds,
|
||||
MemberCredentials memberCredentials, Coordinator coordinator, String worktreeGroup) {
|
||||
this(bind, herdrSocket, memberHerdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, coordinator, worktreeGroup, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the {@code worktreeGroup} key was added. */
|
||||
public FleetConfig(Bind bind, String herdrSocket, String memberHerdrSocket, Map<String, Profile> profiles,
|
||||
Guard guard, String worktreeRoot, Lifecycle lifecycle, Integer spawnReadyTimeoutMs,
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, Health health, String placement, Auth auth,
|
||||
ConfigReload configReload, Integer quarantineCooldownSeconds,
|
||||
MemberCredentials memberCredentials, Coordinator coordinator) {
|
||||
this(bind, herdrSocket, memberHerdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, coordinator, null, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the {@code coordinator:} block was added. */
|
||||
public FleetConfig(Bind bind, String herdrSocket, Map<String, Profile> profiles, Guard guard,
|
||||
@@ -105,9 +163,9 @@ public record FleetConfig(
|
||||
LeadHeartbeat leadHeartbeat, Health health, String placement, Auth auth,
|
||||
ConfigReload configReload, Integer quarantineCooldownSeconds,
|
||||
MemberCredentials memberCredentials) {
|
||||
this(bind, herdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
this(bind, herdrSocket, null, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, null);
|
||||
configReload, quarantineCooldownSeconds, memberCredentials, null, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the CB-596 {@code memberCredentials:} block was added. */
|
||||
@@ -116,9 +174,9 @@ public record FleetConfig(
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, Health health, String placement, Auth auth,
|
||||
ConfigReload configReload, Integer quarantineCooldownSeconds) {
|
||||
this(bind, herdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
this(bind, herdrSocket, null, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
|
||||
configReload, quarantineCooldownSeconds, null);
|
||||
configReload, quarantineCooldownSeconds, null, null);
|
||||
}
|
||||
|
||||
/** Default cooldown (CB-578 stage B) when {@code quarantineCooldownSeconds} is absent/non-positive. */
|
||||
@@ -129,8 +187,8 @@ public record FleetConfig(
|
||||
String worktreeRoot, Lifecycle lifecycle, Integer spawnReadyTimeoutMs,
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, String placement, Auth auth) {
|
||||
this(bind, herdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, null, placement, auth, null, null);
|
||||
this(bind, herdrSocket, null, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, null, placement, auth, null, null, null, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the optional {@code health:} block was added. */
|
||||
@@ -138,8 +196,8 @@ public record FleetConfig(
|
||||
String worktreeRoot, Lifecycle lifecycle, Integer spawnReadyTimeoutMs,
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, String placement, Auth auth, ConfigReload configReload) {
|
||||
this(bind, herdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, null, placement, auth, configReload, null);
|
||||
this(bind, herdrSocket, null, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, null, placement, auth, configReload, null, null, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the CB-578 stage B {@code quarantineCooldownSeconds} field was added. */
|
||||
@@ -148,9 +206,9 @@ public record FleetConfig(
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, Health health, String placement, Auth auth,
|
||||
ConfigReload configReload) {
|
||||
this(bind, herdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
this(bind, herdrSocket, null, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
|
||||
configReload, null);
|
||||
configReload, null, null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -192,7 +250,8 @@ public record FleetConfig(
|
||||
* @param configDir {@code CLAUDE_CONFIG_DIR} so the worker inherits the profile's
|
||||
* skills/MCP/hooks (may be {@code null})
|
||||
* @param tokenEnv name of the host env var holding the worker's auth token; its value
|
||||
* is injected as {@code ANTHROPIC_AUTH_TOKEN} (never stored in config)
|
||||
* is injected as {@code ANTHROPIC_AUTH_TOKEN} (never stored in config).
|
||||
* {@code null}/blank means this profile needs no token.
|
||||
* @param argv launch command; defaults to {@code ["claude"]}
|
||||
* @param placement where a worker lands: {@code "tab"} (default — its own tab in the
|
||||
* worker space) or {@code "pane"} (legacy — split the focused tab)
|
||||
@@ -208,7 +267,13 @@ public record FleetConfig(
|
||||
* @param cwd fixed working directory for this profile's workers (CB-112 "told otherwise");
|
||||
* {@code null}/blank → inherit the primary's cwd, else the daemon's
|
||||
* @param parityOverlay repo-relative paths copied primary→worktree for config parity; null/empty
|
||||
* defaults to a sensible set of local config files.
|
||||
* defaults to {@code [.env]} only (CB-148 point 2). {@code .env} is data — a
|
||||
* copy of it can only carry values. {@code .envrc} is executable shell that
|
||||
* {@code direnv} evaluates on every {@code cd}, so copying it moves behaviour
|
||||
* into the worker, not just values, and that is a different risk from copying
|
||||
* data. It is deliberately left out of the default for that reason. The knob
|
||||
* still supports it: an operator who wants it copied writes
|
||||
* {@code parityOverlay: [.env, .envrc]} explicitly and owns that choice.
|
||||
* <p><strong>Every overlay path must be gitignored or tracked-and-skipped.</strong>
|
||||
* CB-576 made {@code release()} preserve a worktree that {@code git status
|
||||
* --porcelain} reports as dirty, and it deliberately counts untracked files —
|
||||
@@ -218,11 +283,10 @@ public record FleetConfig(
|
||||
* worktrees then accumulate with no error anywhere.
|
||||
* <p>Checked on 2026-08-15 (CB-581): inert as configured. Tracked overlay
|
||||
* files carry {@code --skip-worktree} so {@code --porcelain} cannot see them,
|
||||
* {@code fleetd.yaml} is gitignored, and the default pair {@code .env} /
|
||||
* {@code .envrc} does not exist in this repo. Note the default applies to
|
||||
* <em>every</em> profile, so creating either file at the repo root is enough
|
||||
* to make it live. Add a new overlay path to {@code .gitignore} in the same
|
||||
* change that adds it here.
|
||||
* {@code fleetd.yaml} is gitignored, and {@code .env} does not exist in this
|
||||
* repo. Note the default applies to <em>every</em> profile, so creating
|
||||
* {@code .env} at the repo root is enough to make it live. Add a new overlay
|
||||
* path to {@code .gitignore} in the same change that adds it here.
|
||||
* <p>CB-578 stage C raised the stakes: a preserved dirty worktree is now also
|
||||
* committed to {@code refs/wip/<branch>} via {@code git add -A}. The overlay
|
||||
* carries the primary's own environment files into the worktree, so a
|
||||
@@ -258,8 +322,9 @@ public record FleetConfig(
|
||||
* statement that does not stop being true just because the profile was
|
||||
* named directly. A negative value has no sane meaning (there is no
|
||||
* "excluded" to degrade to below zero) and is refused at config load
|
||||
* instead, naming the profile and the key. Live means any session the
|
||||
* registry still owns (acquired and not yet released), in any state.
|
||||
* instead, naming the profile and the key. Live means a session that can
|
||||
* receive another delivery. The roster keeps terminal {@code BACKEND_ERROR}
|
||||
* and {@code FAILED} sessions for diagnostics, but they do not use capacity.
|
||||
* @param kind which peer launcher spawns this profile: {@code "claude-code"} (default —
|
||||
* the {@link dev.ltms.fleet.member.ClaudeCodeLauncher}) or {@code "opencode"}.
|
||||
* The {@code CompositePeerLauncher} routes {@code spawn}/reap by this value, so
|
||||
@@ -307,6 +372,16 @@ public record FleetConfig(
|
||||
* {@code limit.context} instead, which bounds the window opencode compacts
|
||||
* <em>within</em>, and only when {@code model} resolves to a
|
||||
* {@code provider/model} pair.
|
||||
* @param errorPattern regex matched against a completion-fallback scrape (fleetd #201 Unit 5)
|
||||
* to classify a turn that ended with no {@code fleet_reply} as a backend
|
||||
* error (credential outage, provider 5xx) rather than a real answer or an
|
||||
* exhausted usage limit. {@code null}/blank ⇒ this profile relies on
|
||||
* {@link dev.ltms.fleet.inject.CompletionResolver}'s built-in
|
||||
* {@code (?i)\bAPI Error\s*:} compatibility pattern instead — classification
|
||||
* still happens, just without a profile-specific match (reported separately
|
||||
* at startup as legacy-default coverage, never full coverage). Compiled once
|
||||
* at startup ({@code Fleetd.main}), like {@code exhaustedPattern}, so a
|
||||
* reload of this key is DEFERRED (see {@code ConfigRef}).
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record Profile(String profile, String baseUrl, String model,
|
||||
@@ -325,7 +400,8 @@ public record FleetConfig(
|
||||
String ideMcpUrl,
|
||||
String ideProjectDir,
|
||||
String ideOpenCommand,
|
||||
Integer autoCompactWindow) {
|
||||
Integer autoCompactWindow,
|
||||
String errorPattern) {
|
||||
|
||||
/** Peer kind spawned by {@link dev.ltms.fleet.member.ClaudeCodeLauncher} (the default). */
|
||||
public static final String KIND_CLAUDE_CODE = "claude-code";
|
||||
@@ -341,7 +417,7 @@ public record FleetConfig(
|
||||
? (KIND_CLAUDE_CODE.equals(k) ? List.of("claude") : List.of(k))
|
||||
: List.copyOf(argv);
|
||||
kind = k;
|
||||
tokenEnv = (tokenEnv == null || tokenEnv.isBlank()) ? "FLEETD_WORKER_TOKEN" : tokenEnv;
|
||||
tokenEnv = (tokenEnv == null || tokenEnv.isBlank()) ? null : tokenEnv;
|
||||
placement = (placement == null || placement.isBlank()) ? "tab" : placement.toLowerCase();
|
||||
workspace = (workspace == null || workspace.isBlank()) ? "fleet" : workspace;
|
||||
// CB-557: no per-profile default any more. A label is generated from the member's ROLE
|
||||
@@ -359,7 +435,7 @@ public record FleetConfig(
|
||||
// servers do not exist there to be enabled. This is defence in depth, not a live fix: the
|
||||
// worker stays isolated only because this separate mechanism already removes the servers.
|
||||
parityOverlay = (parityOverlay == null || parityOverlay.isEmpty())
|
||||
? List.of(".env", ".envrc")
|
||||
? List.of(".env")
|
||||
: List.copyOf(parityOverlay);
|
||||
// gitTokenEnv stays null when unset (opt-in). gitHostEnv defaults so operators enabling
|
||||
// checkpoints need only set gitTokenEnv; it is injected only alongside a resolved token.
|
||||
@@ -399,6 +475,32 @@ public record FleetConfig(
|
||||
// there is no blank-string form to normalize (it's an Integer), and the [100000, 1000000]
|
||||
// range is enforced eagerly at config load (rejectAutoCompactWindowOutOfRange), naming the
|
||||
// profile, rather than silently clamped here. A profile that never sets it keeps null.
|
||||
// errorPattern stays null when unset/blank (opt-in) — same rule as exhaustedPattern: no
|
||||
// defaulting, no vendor wording. A profile that never sets it relies on
|
||||
// CompletionResolver's built-in compatibility pattern instead (never "off").
|
||||
errorPattern = (errorPattern == null || errorPattern.isBlank()) ? null : errorPattern;
|
||||
}
|
||||
|
||||
/**
|
||||
* Backward-compatible constructor without the fleetd #201 {@code errorPattern} field — the
|
||||
* profile relies on {@code CompletionResolver}'s built-in compatibility pattern instead
|
||||
* (legacy-default coverage). This is the shape the canonical constructor had before the
|
||||
* field was added; every pre-existing Java call site (and any YAML that omits the key) keeps
|
||||
* compiling and behaving identically. Jackson binds the canonical (longest) constructor, so
|
||||
* YAML omitting {@code errorPattern:} still lands here as {@code null} via that path too.
|
||||
*/
|
||||
public Profile(String profile, String baseUrl, String model,
|
||||
String configDir, String tokenEnv, List<String> argv,
|
||||
String placement, String workspace, String tabLabel, String mcpUrl,
|
||||
String cwd, List<String> parityOverlay, String gitTokenEnv, String gitHostEnv,
|
||||
String kind, Map<String, String> env, Float weight, Integer maxLoad,
|
||||
Boolean subscription, String exhaustedPattern, String credentialId,
|
||||
String ideMcpUrl, String ideProjectDir, String ideOpenCommand,
|
||||
Integer autoCompactWindow) {
|
||||
this(profile, baseUrl, model, configDir, tokenEnv, argv, placement, workspace, tabLabel,
|
||||
mcpUrl, cwd, parityOverlay, gitTokenEnv, gitHostEnv, kind, env, weight, maxLoad,
|
||||
subscription, exhaustedPattern, credentialId, ideMcpUrl, ideProjectDir, ideOpenCommand,
|
||||
autoCompactWindow, null);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -443,7 +545,8 @@ public record FleetConfig(
|
||||
public Profile withProfile(String p) {
|
||||
return new Profile(p, baseUrl, model, configDir, tokenEnv, argv, placement, workspace, tabLabel,
|
||||
mcpUrl, cwd, parityOverlay, gitTokenEnv, gitHostEnv, kind, env, weight, maxLoad, subscription,
|
||||
exhaustedPattern, credentialId, ideMcpUrl, ideProjectDir, ideOpenCommand, autoCompactWindow);
|
||||
exhaustedPattern, credentialId, ideMcpUrl, ideProjectDir, ideOpenCommand, autoCompactWindow,
|
||||
errorPattern);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -541,14 +644,55 @@ public record FleetConfig(
|
||||
}
|
||||
|
||||
/**
|
||||
* The credential group this profile quarantines with (CB-578 stage B): the configured
|
||||
* {@link #credentialId} when set, else this profile's own name — so an unconfigured profile
|
||||
* quarantines alone, exactly as it did before this field existed. Two profiles that set the
|
||||
* same non-blank {@code credentialId} share one quarantine: a {@code BACKEND_EXHAUSTED}
|
||||
* classification on either one quarantines both.
|
||||
* True when this profile configures its own fleetd #201 Unit 5 backend-error classification
|
||||
* pattern. {@code false} means this profile relies on {@code CompletionResolver}'s built-in
|
||||
* {@code (?i)\bAPI Error\s*:} compatibility pattern instead (legacy-default coverage).
|
||||
*/
|
||||
public boolean hasErrorPattern() {
|
||||
return errorPattern != null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The shared credential id every {@code subscription: true} profile falls back to when it
|
||||
* sets no explicit {@link #credentialId} (fleetd #176 stage 2, correcting an inert first cut
|
||||
* of that ticket). A subscription profile has no credential of its own to fall back to its
|
||||
* name for: it authenticates as the operator's own Claude login, and a host has exactly one
|
||||
* of those, whatever names the operator gives the profiles running on it. Falling back to the
|
||||
* profile's own name (the way an ordinary off-subscription profile does) would keep two
|
||||
* subscription profiles on one login apart from each other, which is the opposite of what
|
||||
* "one account" means.
|
||||
*
|
||||
* <p>Measured live and what it broke: a lead on profile {@code opus}, members on profile
|
||||
* {@code sonnet}, same Claude login, neither setting {@code credentialId}. Before this
|
||||
* sentinel, {@code opus.effectiveCredentialId()} was {@code "opus"} and {@code sonnet
|
||||
* .effectiveCredentialId()} was {@code "sonnet"} — so fleetd #176's lead-seat matcher (and,
|
||||
* this sentinel now also fixes, {@code CompositePeerLauncher}'s quarantine/cool-off spawn
|
||||
* refusal and {@code BackendOutagePolicy}'s incident grouping) silently never linked them: the
|
||||
* fix shipped, and stayed inert on the one host it was written for.
|
||||
*/
|
||||
public static final String SUBSCRIPTION_CREDENTIAL_ID = "<subscription>";
|
||||
|
||||
/**
|
||||
* The credential group this profile quarantines with (CB-578 stage B; extended fleetd #176
|
||||
* stage 2 — see {@link #SUBSCRIPTION_CREDENTIAL_ID}): the configured {@link #credentialId}
|
||||
* when set — that always wins, so an operator with two separate Claude logins on one host can
|
||||
* still keep them apart. Otherwise, a {@code subscription: true} profile falls back to
|
||||
* {@link #SUBSCRIPTION_CREDENTIAL_ID} rather than its own name; an ordinary off-subscription
|
||||
* profile falls back to its own name, exactly as it did before this field existed, so an
|
||||
* unconfigured off-subscription profile still quarantines alone.
|
||||
*
|
||||
* <p>A {@code BACKEND_EXHAUSTED} (or repeated backend-error) classification on any profile
|
||||
* sharing the result quarantines/cools off every profile that shares it — including, now,
|
||||
* every {@code subscription: true} profile with no explicit {@code credentialId}. That is
|
||||
* intended, not incidental: one Claude subscription hitting a usage limit really does take out
|
||||
* every profile running on it, the same way {@code credentialId: openai-shared} already lets
|
||||
* two OpenAI-backed profiles share one quarantine.
|
||||
*/
|
||||
public String effectiveCredentialId() {
|
||||
return (credentialId == null || credentialId.isBlank()) ? profile : credentialId;
|
||||
if (credentialId != null && !credentialId.isBlank()) {
|
||||
return credentialId;
|
||||
}
|
||||
return isSubscription() ? SUBSCRIPTION_CREDENTIAL_ID : profile;
|
||||
}
|
||||
|
||||
/** True when this profile's workers are granted a forge token to open their own PR (CB-302). */
|
||||
@@ -556,6 +700,11 @@ public record FleetConfig(
|
||||
return gitTokenEnv != null && !gitTokenEnv.isBlank();
|
||||
}
|
||||
|
||||
/** True when this profile explicitly names a host token environment variable. */
|
||||
public boolean hasTokenEnv() {
|
||||
return tokenEnv != null && !tokenEnv.isBlank();
|
||||
}
|
||||
|
||||
/**
|
||||
* True when this profile's {@code env:} block names a worker-side Anthropic binding variable.
|
||||
* Those two keys are the adapter's, never the operator's: on a claude-code worker
|
||||
@@ -838,7 +987,15 @@ public record FleetConfig(
|
||||
* survives restarts of the agent inside it — so identity is now the tab label alone.
|
||||
*
|
||||
* @param profile the {@code profiles:} entry to launch this lead on when one must
|
||||
* be created; {@code null} ⇒ recognise-only, never create
|
||||
* be created; {@code null} ⇒ recognise-only, never create.
|
||||
* <p>fleetd #176: also the field {@code Fleetd.leadSeatLookup} reads
|
||||
* to learn which account this lead's own live session shares — set it
|
||||
* (safely, even on an already-running recognise-only lead: naming a
|
||||
* profile here never starts anything beyond {@code instances}) so a
|
||||
* {@code subscription: true} worker profile sharing its
|
||||
* {@code effectiveCredentialId()} has this lead's seat subtracted from
|
||||
* {@code fleet_list}'s {@code free}. {@code null} here also means this
|
||||
* lead's seat cannot be derived and is not counted.
|
||||
* @param tab the exact tab label hosting this lead, matched case-insensitively;
|
||||
* the only field identity depends on. Required — a lead with no
|
||||
* {@code tab} can never be discovered, launched or not
|
||||
@@ -1203,19 +1360,19 @@ public record FleetConfig(
|
||||
* deny-list/deny-by-default every name here that is NOT also in {@link #allow} is
|
||||
* overlaid with a non-secret sentinel value. Under allow-list this list is
|
||||
* reporting only.
|
||||
* @param sshAuthSock whether the member may inherit {@code SSH_AUTH_SOCK} under the allow-list
|
||||
* policy ({@code "allow"}) or must have it blanked ({@code "block"}, the default).
|
||||
* @param sshAgentEnv whether the member may inherit {@code SSH_AUTH_SOCK} under the allow-list
|
||||
* policy ({@code "inherit"}) or must have it omitted ({@code "omit"}, the default).
|
||||
* This is a DECISION, never a default: {@code SSH_AUTH_SOCK} is a handle to the
|
||||
* operator's ssh-agent, and a member holding it can sign with the operator's own
|
||||
* keys — but it appears in no secret file and is credential-shaped like nothing on
|
||||
* any list, which is why three earlier tickets missed it (gitea #110 / CB-607).
|
||||
* Blocking it breaks git over SSH inside the member; allow it only when members do
|
||||
* Omitting it does not prevent git over SSH inside the member; inherit it only when members do
|
||||
* not need to authenticate as the operator over SSH. Ignored under deny-list /
|
||||
* deny-by-default, which never touch the name.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record MemberCredentials(String policy, List<String> allow, List<String> known,
|
||||
String sshAuthSock) {
|
||||
String sshAgentEnv) {
|
||||
|
||||
/** Default policy: block every {@code known} name not in {@code allow}, via the env overlay. */
|
||||
public static final String POLICY_DENY_BY_DEFAULT = "deny-by-default";
|
||||
@@ -1232,11 +1389,25 @@ public record FleetConfig(
|
||||
*/
|
||||
public static final String POLICY_ALLOW_LIST = "allow-list";
|
||||
|
||||
/** The pre-CB-633 three-field form — {@code sshAuthSock} defaults to blocked. */
|
||||
/** The pre-CB-633 three-field form — {@code sshAgentEnv} defaults to omitted. */
|
||||
public MemberCredentials(String policy, List<String> allow, List<String> known) {
|
||||
this(policy, allow, known, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Reads both the current {@code sshAgentEnv} key and the compatible {@code sshAuthSock} key.
|
||||
* When both keys are present, {@code sshAgentEnv} wins, even if its value is unrecognised.
|
||||
*/
|
||||
@JsonCreator
|
||||
public static MemberCredentials fromYaml(@JsonProperty("policy") String policy,
|
||||
@JsonProperty("allow") List<String> allow,
|
||||
@JsonProperty("known") List<String> known,
|
||||
@JsonProperty("sshAgentEnv") String sshAgentEnv,
|
||||
@JsonProperty("sshAuthSock") String sshAuthSock) {
|
||||
return new MemberCredentials(policy, allow, known,
|
||||
sshAgentEnv != null ? sshAgentEnv : sshAuthSock);
|
||||
}
|
||||
|
||||
public MemberCredentials {
|
||||
String normalizedPolicy = (policy == null || policy.isBlank())
|
||||
? POLICY_DENY_BY_DEFAULT : policy.toLowerCase(java.util.Locale.ROOT);
|
||||
@@ -1245,8 +1416,9 @@ public record FleetConfig(
|
||||
policy = POLICY_DENY_LIST.equals(normalizedPolicy) ? POLICY_DENY_BY_DEFAULT : normalizedPolicy;
|
||||
allow = allow == null ? List.of() : List.copyOf(allow);
|
||||
known = known == null ? List.of() : List.copyOf(known);
|
||||
sshAuthSock = (sshAuthSock != null && "allow".equalsIgnoreCase(sshAuthSock.trim()))
|
||||
? "allow" : "block";
|
||||
sshAgentEnv = (sshAgentEnv != null
|
||||
&& ("inherit".equalsIgnoreCase(sshAgentEnv.trim())
|
||||
|| "allow".equalsIgnoreCase(sshAgentEnv.trim()))) ? "inherit" : "omit";
|
||||
}
|
||||
|
||||
/** True when this block selects the CB-633 derived-allow-list policy. */
|
||||
@@ -1255,8 +1427,8 @@ public record FleetConfig(
|
||||
}
|
||||
|
||||
/** True when {@code SSH_AUTH_SOCK} may pass through under the allow-list policy. Default: no. */
|
||||
public boolean sshAuthSockAllowed() {
|
||||
return "allow".equals(sshAuthSock);
|
||||
public boolean sshAgentEnvInherited() {
|
||||
return "inherit".equals(sshAgentEnv);
|
||||
}
|
||||
|
||||
/** {@link #allow} as a set, for membership checks. */
|
||||
@@ -1320,10 +1492,10 @@ public record FleetConfig(
|
||||
* {@code fleetd.yaml} itself is gitignored.
|
||||
*/
|
||||
static final Set<String> KNOWN_TOP_LEVEL_KEYS = Set.of(
|
||||
"bind", "herdrSocket", "profiles", "guard", "worktreeRoot",
|
||||
"bind", "herdrSocket", "memberHerdrSocket", "profiles", "guard", "worktreeRoot",
|
||||
"lifecycle", "spawnReadyTimeoutMs", "spawnReadyPollMs", "broker", "primary", "fleet",
|
||||
"leadHeartbeat", "health", "placement", "auth", "configReload", "quarantineCooldownSeconds",
|
||||
"memberCredentials", "coordinator");
|
||||
"memberCredentials", "coordinator", "worktreeGroup", "memberLoginShell");
|
||||
|
||||
/** Load and validate config from {@code path}. */
|
||||
public static FleetConfig load(Path path) {
|
||||
@@ -1335,6 +1507,7 @@ public record FleetConfig(
|
||||
rejectDuplicateMemberSlots(yaml);
|
||||
rejectNegativeMaxLoad(yaml);
|
||||
rejectAutoCompactWindowOutOfRange(yaml);
|
||||
rejectMalformedProfilePatterns(yaml);
|
||||
rejectUnknownKind(yaml);
|
||||
rejectUnknownAuthMode(yaml);
|
||||
rejectUnknownPlacement(yaml);
|
||||
@@ -1683,6 +1856,59 @@ public record FleetConfig(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Reject a profile whose {@code errorPattern} (fleetd #201 Unit 5) or {@code exhaustedPattern}
|
||||
* (CB-578 stage A) is not a valid Java regex, naming the profile, the key, and the parser's own
|
||||
* message.
|
||||
*
|
||||
* <p>Unset/{@code null} means, for {@code errorPattern}, "use {@code CompletionResolver}'s
|
||||
* built-in {@code (?i)\bAPI Error\s*:} compatibility pattern", and for {@code exhaustedPattern},
|
||||
* "opt out of that classification" — either way it passes silently. A profile that DOES set
|
||||
* either key gets it compiled once at daemon startup ({@code Fleetd.main}) — an uncaught
|
||||
* {@link java.util.regex.PatternSyntaxException} there crashes startup without naming which
|
||||
* profile or key is at fault (fleetd #273: this happened for {@code exhaustedPattern}, which had
|
||||
* no validator here even though its sibling {@code errorPattern} did). Validate eagerly here
|
||||
* instead, at config load, the same "fail loud at load, not lazily later" reasoning as
|
||||
* {@link #rejectAutoCompactWindowOutOfRange}. Both keys are checked from a single load, and any
|
||||
* failures from either are collected together into one message.
|
||||
*
|
||||
* @param yaml the raw config text
|
||||
* @throws IllegalStateException when any profile's {@code errorPattern} or
|
||||
* {@code exhaustedPattern} fails to compile
|
||||
*/
|
||||
static void rejectMalformedProfilePatterns(String yaml) {
|
||||
Map<?, ?> raw;
|
||||
try {
|
||||
raw = YAML.readValue(yaml, Map.class);
|
||||
} catch (IOException | IllegalArgumentException e) {
|
||||
return; // a malformed file is reported by the real parse, not here
|
||||
}
|
||||
if (raw == null || !(raw.get("profiles") instanceof Map<?, ?> profiles)) {
|
||||
return;
|
||||
}
|
||||
List<String> bad = new ArrayList<>();
|
||||
for (Map.Entry<?, ?> e : profiles.entrySet()) {
|
||||
if (!(e.getValue() instanceof Map<?, ?> p)) {
|
||||
continue;
|
||||
}
|
||||
for (String key : List.of("errorPattern", "exhaustedPattern")) {
|
||||
if (!(p.get(key) instanceof String pattern) || pattern.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
try {
|
||||
Pattern.compile(pattern);
|
||||
} catch (PatternSyntaxException ex) {
|
||||
bad.add("profiles." + e.getKey() + "." + key + " (\"" + pattern + "\"): " + ex.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
bad.sort(String::compareTo);
|
||||
if (!bad.isEmpty()) {
|
||||
throw new IllegalStateException("refusing to start: malformed pattern — "
|
||||
+ String.join("; ", bad));
|
||||
}
|
||||
}
|
||||
|
||||
/** The peer kinds this build has an adapter for — {@link Profile#kind()}'s only valid values. */
|
||||
private static final Set<String> KNOWN_KINDS = Set.of(Profile.KIND_CLAUDE_CODE, Profile.KIND_OPENCODE);
|
||||
|
||||
@@ -1941,9 +2167,14 @@ public record FleetConfig(
|
||||
: new MemberCredentials(null, List.of(), List.of());
|
||||
// coordinator is left as-is, like broker/primary above: null keeps no LeadMailbox opened,
|
||||
// and this ticket's Coordinator is config-only anyway (nothing yet reads it at startup).
|
||||
return new FleetConfig(b, herdrSocket, profiles, g, worktreeRoot, l, timeout, pollMs,
|
||||
// worktreeGroup is left as-is (fleetd #185 stage 3): null/blank is "off", and there is no
|
||||
// sane non-null default — an OS group name is operator-specific.
|
||||
// memberLoginShell is left as-is (fleetd #213), like worktreeGroup: null/blank is "not
|
||||
// configured", and there is no sane non-null default — a member's login shell is
|
||||
// operator-specific and only meaningful when memberHerdrSocket is also set.
|
||||
return new FleetConfig(b, herdrSocket, memberHerdrSocket, profiles, g, worktreeRoot, l, timeout, pollMs,
|
||||
broker, primary, f, leadHeartbeat, health, placementOrDefault, a, configReload,
|
||||
quarantineCooldown, mc, coordinator);
|
||||
quarantineCooldown, mc, coordinator, worktreeGroup, memberLoginShell);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -14,6 +14,7 @@ import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiConsumer;
|
||||
@@ -28,6 +29,14 @@ public final class FleetHealthMonitor {
|
||||
static final int MAX_FAIL_TARGET_ATTEMPTS = 3;
|
||||
// CB-641: Match the injector's 60s readiness gate so health allows a full first boot.
|
||||
static final long READINESS_GRACE_NANOS = TimeUnit.SECONDS.toNanos(60);
|
||||
/**
|
||||
* fleetd #280: how long after a terminal transition to wait before the one bounded re-check
|
||||
* fires. Must exceed the worst-case reverse-rendezvous {@code fleet_ask} window (55-115s, see
|
||||
* {@code FleetMcp.ASK_DEFAULT_TIMEOUT_MS} / {@code FleetApp.MAX_ASK_TIMEOUT_MS}) so that, if the
|
||||
* target was genuinely {@code ASKING} when {@code state} was first observed, its own ask has had
|
||||
* time to lapse (clearing {@code Task#question} back to {@code null}) before this fires.
|
||||
*/
|
||||
static final long ASK_LAPSE_RECHECK_DELAY_SECONDS = 120;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
@@ -38,7 +47,16 @@ public final class FleetHealthMonitor {
|
||||
private final long workingSuspectAfterNanos;
|
||||
private final BiConsumer<String, String> failTarget;
|
||||
private final Map<String, HealthPrior> priors = new HashMap<>();
|
||||
private final Map<String, HealthState> states = new HashMap<>();
|
||||
/**
|
||||
* The live classification per member, and the only one of this class's three maps that more
|
||||
* than one scheduler task touches. {@code tick} writes it (and prunes it to the roster);
|
||||
* fleetd #280's delayed {@link #recheckTerminalTarget} reads it from its own separate scheduled
|
||||
* task. Both run on the single-threaded scheduler {@code Fleetd} passes in today, so they are
|
||||
* serialised — but nothing in this class enforces that, and an unsynchronised {@link HashMap}
|
||||
* read racing a resize can spin a CPU forever rather than fail visibly. {@code priors} and
|
||||
* {@code orphanStreaks} stay plain maps because {@code tick} is still their only toucher.
|
||||
*/
|
||||
private final Map<String, HealthState> states = new ConcurrentHashMap<>();
|
||||
/**
|
||||
* CB-643: consecutive ticks on which a target looked like an orphaned delegation. The fact
|
||||
* {@link MessageService#hasOrphanedDelegation} reports is a true snapshot, but it can read true
|
||||
@@ -171,6 +189,7 @@ public final class FleetHealthMonitor {
|
||||
// member stayed terminal.
|
||||
if (terminal(next)) {
|
||||
failTerminalTarget(target, next);
|
||||
scheduleTerminalRecheck(target, next);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -191,6 +210,48 @@ public final class FleetHealthMonitor {
|
||||
target, state, MAX_FAIL_TARGET_ATTEMPTS, last);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #280: schedule the one bounded, delayed follow-up for a terminal transition — never a
|
||||
* per-tick retry (CB-580 rejected that shape; {@link #reportTransition} still fires
|
||||
* {@link #failTerminalTarget} exactly once per transition, unconditionally on the tick loop).
|
||||
* This is a single one-shot task, scheduled once per transition into GONE/NEVER_READY, so a
|
||||
* member stuck terminal for the rest of its life gets exactly one extra attempt, not one per
|
||||
* tick. See {@link #recheckTerminalTarget} for why the extra attempt is safe.
|
||||
*/
|
||||
private void scheduleTerminalRecheck(String target, HealthState state) {
|
||||
if (scheduler.isShutdown()) return;
|
||||
try {
|
||||
scheduler.schedule(() -> recheckTerminalTarget(target, state),
|
||||
ASK_LAPSE_RECHECK_DELAY_SECONDS, TimeUnit.SECONDS);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("fleet health: could not schedule terminal re-check for member={} state={}",
|
||||
target, state, e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #280: the delayed re-check {@link #scheduleTerminalRecheck} scheduled for one terminal
|
||||
* transition. By now, a {@code fleet_ask} that was still open when {@code state} was first
|
||||
* observed has had time to lapse on its own (see {@link #ASK_LAPSE_RECHECK_DELAY_SECONDS}),
|
||||
* clearing {@code Task#question} back to {@code null} — which is exactly what
|
||||
* {@link MessageService#abandon(String, String, boolean)}'s {@code sweepAsking=false} filter
|
||||
* needs to finally match it. Calling {@link #failTerminalTarget} again is safe only because
|
||||
* {@code sweepAsking} stays {@code false}: a task genuinely still {@code ASKING} is skipped
|
||||
* exactly as it was on the very first attempt — this never fails a ticket whose ask has not yet
|
||||
* lapsed.
|
||||
*
|
||||
* <p><strong>Guarded on "target is still classified {@code state}."</strong> Without this guard,
|
||||
* a member that recovered (or was released and dropped from the roster) between the transition
|
||||
* and this re-check would still take a blind {@code failTarget} call — reaching into whatever
|
||||
* brand-new, unrelated turn it has since picked up and failing it too. {@link #states} already
|
||||
* carries the live classification (updated every tick, pruned to the current roster on release),
|
||||
* so a stale or recovered target simply reads as a mismatch here and this is a no-op.
|
||||
*/
|
||||
void recheckTerminalTarget(String target, HealthState state) {
|
||||
if (states.get(target) != state) return;
|
||||
failTerminalTarget(target, state);
|
||||
}
|
||||
|
||||
private static boolean terminal(HealthState state) {
|
||||
return state == HealthState.GONE || state == HealthState.NEVER_READY;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import java.util.Objects;
|
||||
import java.util.function.Predicate;
|
||||
|
||||
/** Routes lead operations and member operations to their owning herdr daemon. */
|
||||
public final class HerdrRouter implements AutoCloseable {
|
||||
private final HerdrClient lead;
|
||||
private final HerdrClient member;
|
||||
private final AgentControl leadAgents;
|
||||
private final AgentControl memberAgents;
|
||||
private final WorkspaceControl leadSpaces;
|
||||
private final WorkspaceControl memberSpaces;
|
||||
private final Predicate<String> isLead;
|
||||
|
||||
public HerdrRouter(HerdrClient lead, HerdrClient member, Predicate<String> isLead) {
|
||||
this.lead = Objects.requireNonNull(lead, "lead");
|
||||
this.member = member != null ? member : lead;
|
||||
this.isLead = Objects.requireNonNull(isLead, "isLead");
|
||||
leadAgents = new AgentControl(this.lead);
|
||||
memberAgents = this.member == this.lead ? leadAgents : new AgentControl(this.member);
|
||||
leadSpaces = new WorkspaceControl(this.lead);
|
||||
memberSpaces = this.member == this.lead ? leadSpaces : new WorkspaceControl(this.member);
|
||||
}
|
||||
|
||||
public AgentControl leadAgents() { return leadAgents; }
|
||||
public WorkspaceControl leadSpaces() { return leadSpaces; }
|
||||
public AgentControl memberAgents() { return memberAgents; }
|
||||
public WorkspaceControl memberSpaces() { return memberSpaces; }
|
||||
public AgentControl agentsFor(String targetId) { return isLead.test(targetId) ? leadAgents : memberAgents; }
|
||||
|
||||
HerdrClient leadClient() { return lead; }
|
||||
HerdrClient memberClient() { return member; }
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
lead.close();
|
||||
if (member != lead) member.close();
|
||||
}
|
||||
}
|
||||
@@ -2,7 +2,11 @@ package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* Resolves which herdr pane a process belongs to — the herdr half of connection-based MCP
|
||||
@@ -11,46 +15,130 @@ import java.util.Map;
|
||||
* is calling without the worker sending anything spoofable.
|
||||
*
|
||||
* <p>herdr owns the PID→pane truth: {@code pane.process_info} reports each pane's {@code shell_pid}
|
||||
* and foreground process PIDs. This scans agent panes; a spawn-time {@code pid→terminal} cache is
|
||||
* the obvious optimization once wired into {@code ClaudeCodeLauncher}.
|
||||
* and foreground process PIDs. A pid that is neither of those directly — e.g. a grandchild a
|
||||
* worker spawned, such as a {@code python3} or {@code curl} helper that opens its own MCP
|
||||
* connection — is resolved by walking its ancestry (via {@link ParentResolver}) up to the root and
|
||||
* matching any ancestor against a pane's {@code shell_pid} or foreground pids (CB-161). Without
|
||||
* this walk such a pid matches no pane, and the caller falls through to loopback-trust and is
|
||||
* resolved as the primary — a worker→primary privilege escalation.
|
||||
*
|
||||
* <p>This scans agent panes; a spawn-time {@code pid→terminal} cache is the obvious optimization
|
||||
* once wired into {@code ClaudeCodeLauncher}.
|
||||
*
|
||||
* <p>CB-185 split the fleet across two herdr daemons — lead operations on one, members on the
|
||||
* other ({@code memberHerdrSocket}). A caller's pane can live on <em>either</em> daemon (a lead's
|
||||
* MCP connection resolves against the lead daemon; a member's against the member daemon), so this
|
||||
* must be able to search more than one client. {@link #PaneLocator(HerdrClient, HerdrClient)}
|
||||
* searches the lead client first, then the member client, and collapses to a single scan when the
|
||||
* two are the same object (the historical single-daemon deployment). The caller's ancestor set is
|
||||
* computed once per {@link #terminalForPid} call and reused across every client searched — it
|
||||
* does not depend on which daemon a pane happens to live on.
|
||||
*/
|
||||
public final class PaneLocator {
|
||||
|
||||
private final HerdrClient herdr;
|
||||
/**
|
||||
* Bound on how many ancestor generations {@link #ancestorsOf} walks. This runs on every MCP
|
||||
* call, so a cycle or a pathologically deep process tree must not hang identity resolution;
|
||||
* 32 generations is far more than any real worker→helper process tree needs.
|
||||
*/
|
||||
private static final int MAX_ANCESTRY_DEPTH = 32;
|
||||
|
||||
private final List<HerdrClient> herdrs;
|
||||
private final ParentResolver parentResolver;
|
||||
|
||||
/** Search only this client — the single-daemon deployment. */
|
||||
public PaneLocator(HerdrClient herdr) {
|
||||
this.herdr = herdr;
|
||||
this(herdr, ParentResolver.PROCESS_HANDLE);
|
||||
}
|
||||
|
||||
/** Search only this client, resolving ancestry through {@code parentResolver} — for tests. */
|
||||
public PaneLocator(HerdrClient herdr, ParentResolver parentResolver) {
|
||||
this.herdrs = List.of(herdr);
|
||||
this.parentResolver = parentResolver;
|
||||
}
|
||||
|
||||
/**
|
||||
* Search {@code lead} first, then {@code member} — the two-daemon deployment (CB-185). When
|
||||
* the caller passes the same client for both (no {@code memberHerdrSocket} configured), this
|
||||
* collapses to one client and one scan, exactly {@link #PaneLocator(HerdrClient)}'s behaviour.
|
||||
*/
|
||||
public PaneLocator(HerdrClient lead, HerdrClient member) {
|
||||
this(lead, member, ParentResolver.PROCESS_HANDLE);
|
||||
}
|
||||
|
||||
/** Two-daemon deployment, resolving ancestry through {@code parentResolver} — for tests. */
|
||||
public PaneLocator(HerdrClient lead, HerdrClient member, ParentResolver parentResolver) {
|
||||
this.herdrs = lead == member ? List.of(lead) : List.of(lead, member);
|
||||
this.parentResolver = parentResolver;
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code terminal_id} of the agent pane whose process tree contains {@code pid}, or
|
||||
* {@code null} if no agent pane owns it (e.g. the caller is the primary, or off-host).
|
||||
* {@code null} if no agent pane on any searched daemon owns it (e.g. the caller is the
|
||||
* primary, or off-host).
|
||||
*/
|
||||
public String terminalForPid(long pid) {
|
||||
if (pid <= 0) {
|
||||
return null;
|
||||
}
|
||||
Set<Long> ancestry = ancestorsOf(pid);
|
||||
for (HerdrClient herdr : herdrs) {
|
||||
String terminal = terminalForPid(herdr, ancestry);
|
||||
if (terminal != null) {
|
||||
return terminal;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code pid} itself plus its ancestor chain, walked through {@link #parentResolver} up to
|
||||
* {@link #MAX_ANCESTRY_DEPTH} generations or pid 1, whichever comes first. A vanished ancestor
|
||||
* ({@link ParentResolver#parentOf} returning empty) ends the walk without error — it just means
|
||||
* the chain is shorter than the bound. A cycle in a fake resolver is caught by the "already
|
||||
* seen" check and also ends the walk, so this can never loop.
|
||||
*/
|
||||
private Set<Long> ancestorsOf(long pid) {
|
||||
Set<Long> ancestry = new LinkedHashSet<>();
|
||||
long current = pid;
|
||||
for (int depth = 0; depth < MAX_ANCESTRY_DEPTH; depth++) {
|
||||
if (current <= 0 || !ancestry.add(current)) {
|
||||
break; // vanished/invalid pid, or a cycle back to a pid already recorded
|
||||
}
|
||||
if (current == 1) {
|
||||
break; // reached the root of the process tree
|
||||
}
|
||||
OptionalLong parent = parentResolver.parentOf(current);
|
||||
if (parent.isEmpty()) {
|
||||
break; // vanished ancestor — not an error, just the end of the chain
|
||||
}
|
||||
current = parent.getAsLong();
|
||||
}
|
||||
return ancestry;
|
||||
}
|
||||
|
||||
private static String terminalForPid(HerdrClient herdr, Set<Long> ancestry) {
|
||||
for (JsonNode pane : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String paneId = pane.path("pane_id").asText(null);
|
||||
if (paneId != null && paneOwnsPid(paneId, pid)) {
|
||||
if (paneId != null && paneOwnsAnyOf(herdr, paneId, ancestry)) {
|
||||
return pane.path("terminal_id").asText(null);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private boolean paneOwnsPid(String paneId, long pid) {
|
||||
private static boolean paneOwnsAnyOf(HerdrClient herdr, String paneId, Set<Long> ancestry) {
|
||||
JsonNode info;
|
||||
try {
|
||||
info = herdr.call("pane.process_info", Map.of("pane_id", paneId)).path("process_info");
|
||||
} catch (HerdrException e) {
|
||||
return false; // pane vanished mid-scan — just skip it
|
||||
}
|
||||
if (info.path("shell_pid").asLong(-1) == pid) {
|
||||
if (ancestry.contains(info.path("shell_pid").asLong(-1))) {
|
||||
return true;
|
||||
}
|
||||
for (JsonNode p : info.path("foreground_processes")) {
|
||||
if (p.path("pid").asLong(-1) == pid) {
|
||||
if (ancestry.contains(p.path("pid").asLong(-1))) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import java.util.OptionalLong;
|
||||
|
||||
/**
|
||||
* Resolves a pid's parent pid — the seam {@link PaneLocator} walks a process's ancestry through,
|
||||
* so its tests can drive the walk from a fake pid→parent map instead of spawning real processes.
|
||||
*
|
||||
* <p>{@link #PROCESS_HANDLE} is the production implementation, backed by {@link ProcessHandle}.
|
||||
*/
|
||||
public interface ParentResolver {
|
||||
|
||||
/** The parent pid of {@code pid}, or empty if {@code pid} is gone or has no known parent. */
|
||||
OptionalLong parentOf(long pid);
|
||||
|
||||
/** Production resolver: asks the JVM's {@link ProcessHandle} view of the OS process tree. */
|
||||
ParentResolver PROCESS_HANDLE = pid -> ProcessHandle.of(pid)
|
||||
.flatMap(ProcessHandle::parent)
|
||||
.map(parent -> OptionalLong.of(parent.pid()))
|
||||
.orElse(OptionalLong.empty());
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Per-target lookup for a profile's configured backend-error pattern (fleetd#201 / #227): how
|
||||
* {@link CompletionResolver} recognizes a backend that failed a turn outright (a credential
|
||||
* outage, a provider 5xx) from whatever ends up on the member's pane, apart from a genuine
|
||||
* completion.
|
||||
*
|
||||
* <p>The pattern is always profile config, never a vendor string in Java source: every backend
|
||||
* words its failure differently, so a hardcoded sentence would only ever match one of them. A
|
||||
* target with no configured pattern is not "off" the way {@link ExhaustedPatternLookup#none()}
|
||||
* is — {@link CompletionResolver} falls back to its narrow {@code (?i)\bAPI Error\s*:}
|
||||
* compatibility pattern instead, so classification still happens, just without a profile-specific
|
||||
* match. {@link #legacy()} is the explicit stand-in existing (pre-fleetd#201) callers pass to get
|
||||
* exactly that fallback-only behavior until a caller wires a real, profile-driven lookup
|
||||
* (fleetd#201 Unit 5).
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface BackendErrorPatternLookup {
|
||||
|
||||
/** The compiled pattern configured for {@code target}'s profile, or {@code null} if none. */
|
||||
Pattern patternFor(String target);
|
||||
|
||||
/**
|
||||
* Legacy lookup — no profile has a configured pattern, so {@link CompletionResolver} classifies
|
||||
* every target using only its built-in {@code (?i)\bAPI Error\s*:} compatibility pattern. The
|
||||
* explicit stand-in existing constructors pass so their behavior is unchanged until a caller
|
||||
* wires a real, profile-driven lookup.
|
||||
*/
|
||||
static BackendErrorPatternLookup legacy() {
|
||||
return target -> null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
/**
|
||||
* Notified when {@link CompletionResolver} actually delivers a typed backend-error classification
|
||||
* to a waiting send (fleetd#201 / #227) — never on a race that lost. {@link CompletionResolver}
|
||||
* calls this only after {@code Rendezvous.resolveFailure} returns {@code true} for that exact
|
||||
* waiter, mirroring the win-only race rule {@link ExhaustionSink} already uses.
|
||||
*
|
||||
* <p>The public send result is unchanged by this classification — it is still a failed send
|
||||
* ({@code Rendezvous.Kind#FAILED}); this sink is the internal seam a later stage (fleetd#201 Unit
|
||||
* 5, the policy that cools a credential off after two such failures in 60 seconds and tells the
|
||||
* lead) consumes. {@link CompletionResolver} knows only {@code target} (a herdr terminal id); it
|
||||
* has no notion of profiles or credentials, so mapping {@code target} to whatever should be
|
||||
* quarantined is entirely the sink's job — exactly like {@link ExhaustionSink}.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface BackendErrorSink {
|
||||
|
||||
/**
|
||||
* @param target the herdr terminal id whose turn was classified as a backend error
|
||||
* @param matchedLine the single pane line that matched the backend-error pattern
|
||||
* @param reason the full failure reason carried by the classification (the matched line
|
||||
* plus whatever pane context the classifying path attaches)
|
||||
*/
|
||||
void onBackendError(String target, String matchedLine, String reason);
|
||||
|
||||
/**
|
||||
* Inert sink — nothing happens on a backend-error classification. The explicit stand-in a
|
||||
* caller (or a test not exercising this feature) passes instead of a defaulting overload,
|
||||
* exactly like {@link ExhaustionSink#none()}.
|
||||
*/
|
||||
static BackendErrorSink none() {
|
||||
return (target, matchedLine, reason) -> { };
|
||||
}
|
||||
}
|
||||
@@ -14,6 +14,7 @@ import java.util.TreeSet;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Function;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
@@ -74,14 +75,40 @@ public final class CompletionResolver implements TurnListener {
|
||||
*/
|
||||
public static final long MIN_TURN_NANOS = Duration.ofSeconds(2).toNanos();
|
||||
|
||||
/**
|
||||
* fleetd#164 (part 2): one stable, explicit backend-failure marker seen on a Claude Code pane
|
||||
* when the backend itself rejected the turn (e.g. {@code "API Error: 400 invalid request body"}).
|
||||
* Kept deliberately narrow — a growing list of ad-hoc error strings rots as backends change their
|
||||
* wording; broader backend-error surfacing is out of scope here (fleetd#164 point 3).
|
||||
*/
|
||||
private static final Pattern BACKEND_ERROR = Pattern.compile("(?i)\\bAPI Error\\s*:");
|
||||
|
||||
private static final String CLIPPED_PANE_TAIL_MARKER =
|
||||
"[Pane tail clipped: member did not call fleet_reply.]";
|
||||
|
||||
/** A pane echo must be this large before it can replace a completion report. */
|
||||
static final int ECHO_MIN_CHARS = 400;
|
||||
|
||||
/** Normalised TUI chrome may add this many characters to an otherwise echoed brief. */
|
||||
static final int MAX_ECHO_EXCESS_CHARS = 160;
|
||||
|
||||
/** The explicit result returned instead of a lead's echoed injected brief. */
|
||||
public static final String NO_REPORT_PREFIX = "[no report — the member ended its turn without fleet_reply, "
|
||||
+ "and the pane still shows the injected brief. Nothing was produced on the pane. Check the "
|
||||
+ "member's worktree and branch for committed work before re-delegating.";
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ExhaustedPatternLookup exhaustedPatterns;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
private final BackendErrorPatternLookup backendErrorPatterns;
|
||||
private final BackendErrorSink backendErrorSink;
|
||||
private final LongSupplier nowNanos;
|
||||
private final Function<String, WorktreeBranch> worktreeBranches;
|
||||
|
||||
/** Known member location, used only to guide a lead after an echoed brief. */
|
||||
public record WorktreeBranch(String worktree, String branch) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-target record of the turn currently in flight: the exact {@link Rendezvous} waiter its
|
||||
@@ -98,7 +125,8 @@ public final class CompletionResolver implements TurnListener {
|
||||
* the other half of the {@link #MIN_TURN_NANOS} floor check, compared against a fresh reading at
|
||||
* resolution time.
|
||||
*/
|
||||
record InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline, long deliveredAtNanos) {
|
||||
record InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline, long deliveredAtNanos,
|
||||
String injectedText) {
|
||||
|
||||
/**
|
||||
* Convenience for tests exercising scrape/suppression logic that don't care about turn
|
||||
@@ -106,7 +134,11 @@ public final class CompletionResolver implements TurnListener {
|
||||
* Not used by production code — {@link #captureBaseline} always records a real reading.
|
||||
*/
|
||||
InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline) {
|
||||
this(waiter, baseline, Long.MIN_VALUE / 2);
|
||||
this(waiter, baseline, Long.MIN_VALUE / 2, null);
|
||||
}
|
||||
|
||||
InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline, long deliveredAtNanos) {
|
||||
this(waiter, baseline, deliveredAtNanos, null);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -121,10 +153,55 @@ public final class CompletionResolver implements TurnListener {
|
||||
* classification actually resolves a waiter. Required for the same
|
||||
* reason as {@code exhaustedPatterns} — pass {@link ExhaustionSink#none()}
|
||||
* to opt out.
|
||||
*
|
||||
* <p>Transition constructor (fleetd#201 Unit 1): delegates to the full constructor below with
|
||||
* {@link BackendErrorPatternLookup#legacy()} and {@link BackendErrorSink#none()}, so every
|
||||
* existing caller keeps today's behavior — the narrow built-in {@code API Error:} match, no
|
||||
* sink notified — until a caller wires a real, profile-driven backend-error lookup and sink
|
||||
* (fleetd#201 Unit 5).
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, System::nanoTime);
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
|
||||
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), System::nanoTime);
|
||||
}
|
||||
|
||||
/** Production constructor with a lookup for the member worktree and branch. */
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, Function<String, WorktreeBranch> worktreeBranches) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
|
||||
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), System::nanoTime, worktreeBranches);
|
||||
}
|
||||
|
||||
/**
|
||||
* Transition constructor (fleetd#201 Unit 1): same legacy backend-error defaults as the 4-arg
|
||||
* constructor above, but with the injectable clock. Kept so existing fleetd#164 timing tests
|
||||
* (and any other caller that wants a controllable clock but not the backend-error feature)
|
||||
* keep compiling unchanged.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, LongSupplier nowNanos) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
|
||||
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), nowNanos);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor (fleetd#201 Unit 1): adds the target-keyed backend-error pattern
|
||||
* lookup and sink alongside the existing exhaustion pair. Required, like {@code exhaustedPatterns}
|
||||
* and {@code exhaustionSink} — pass {@link BackendErrorPatternLookup#legacy()} and
|
||||
* {@link BackendErrorSink#none()} to opt out.
|
||||
*
|
||||
* @param backendErrorPatterns fleetd#201: per-target lookup for a profile's configured
|
||||
* backend-error pattern; a target with none configured is matched
|
||||
* against the built-in compatibility pattern instead (never "off").
|
||||
* @param backendErrorSink fleetd#201: notified when a backend-error classification actually
|
||||
* resolves a waiter — never on a race that lost.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
|
||||
BackendErrorSink backendErrorSink) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, backendErrorPatterns, backendErrorSink,
|
||||
System::nanoTime, _ -> null);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -137,12 +214,25 @@ public final class CompletionResolver implements TurnListener {
|
||||
* package.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, LongSupplier nowNanos) {
|
||||
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
|
||||
BackendErrorSink backendErrorSink, LongSupplier nowNanos) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, backendErrorPatterns, backendErrorSink,
|
||||
nowNanos, _ -> null);
|
||||
}
|
||||
|
||||
/** Full constructor with injectable clock and member location lookup. */
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
|
||||
BackendErrorSink backendErrorSink, LongSupplier nowNanos,
|
||||
Function<String, WorktreeBranch> worktreeBranches) {
|
||||
this.agents = agents;
|
||||
this.rendezvous = rendezvous;
|
||||
this.exhaustedPatterns = Objects.requireNonNull(exhaustedPatterns, "exhaustedPatterns");
|
||||
this.exhaustionSink = Objects.requireNonNull(exhaustionSink, "exhaustionSink");
|
||||
this.backendErrorPatterns = Objects.requireNonNull(backendErrorPatterns, "backendErrorPatterns");
|
||||
this.backendErrorSink = Objects.requireNonNull(backendErrorSink, "backendErrorSink");
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
this.worktreeBranches = Objects.requireNonNull(worktreeBranches, "worktreeBranches");
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -172,7 +262,7 @@ public final class CompletionResolver implements TurnListener {
|
||||
baseline = null; // fail open: no baseline ⇒ no suppression
|
||||
log.debug("delivery baseline for {} failed: {}", target, e.getMessage());
|
||||
}
|
||||
inFlight.put(target, new InFlight(waiter, baseline, nowNanos.getAsLong()));
|
||||
inFlight.put(target, new InFlight(waiter, baseline, nowNanos.getAsLong(), token.injectedText()));
|
||||
}
|
||||
|
||||
/** The turn currently baselined for {@code target}, or {@code null} — a test hook for the captureBaseline path. */
|
||||
@@ -224,16 +314,18 @@ public final class CompletionResolver implements TurnListener {
|
||||
// screen, since that is usually the backend's own error.
|
||||
long elapsedNanos = nowNanos.getAsLong() - turn.deliveredAtNanos();
|
||||
if (elapsedNanos < MIN_TURN_NANOS) {
|
||||
fail(target, turn, tooFastReason(target, elapsedNanos));
|
||||
failTooFast(target, turn, waiter, elapsedNanos);
|
||||
return;
|
||||
}
|
||||
String tail;
|
||||
String assistantBlock = null;
|
||||
String rawScrape = null;
|
||||
int originalLength = 0;
|
||||
boolean clipped = false;
|
||||
boolean scrapeFailed = false;
|
||||
try {
|
||||
assistantBlock = lastAssistantBlock(agents.read(target, SCRAPE_SOURCE));
|
||||
rawScrape = agents.read(target, SCRAPE_SOURCE);
|
||||
assistantBlock = lastAssistantBlock(rawScrape);
|
||||
originalLength = assistantBlock.strip().length();
|
||||
clipped = originalLength > MAX_SCRAPE_CHARS;
|
||||
tail = clip(assistantBlock);
|
||||
@@ -248,6 +340,20 @@ public final class CompletionResolver implements TurnListener {
|
||||
// member, so a caller (including a lead deciding whether to delegate again) can tell a lost
|
||||
// turn from a real empty answer.
|
||||
if (scrapeFailed || tail.isEmpty()) {
|
||||
// fleetd#211: lastAssistantBlock() found nothing usable — most often a pane with no ⏺
|
||||
// marker at all, whose boundary scan then starts at the top of the raw screen and breaks
|
||||
// immediately on the first line of TUI chrome (╭, │, ❯, …). Before giving up as a lost
|
||||
// turn, run the same exhaustion/backend-error classification against the RAW scrape as a
|
||||
// fallback, ONLY here. A pane that already yielded a usable block never reaches this
|
||||
// branch, so the narrow (trimmed) match on the normal path below is completely unchanged
|
||||
// — zero new false positives there. Every pane this fallback examines was already headed
|
||||
// for the empty-scrape failure, so a wrong label here is strictly less bad than silently
|
||||
// losing an exhaustion signal: the alternative outcome is already a failure, just one that
|
||||
// never quarantines the credential. lastAssistantBlock stays the source of the reply
|
||||
// TEXT everywhere else; only classification ever consults the raw scrape, and only here.
|
||||
if (rawScrape != null && classifyRawScrapeFallback(target, turn, waiter, rawScrape)) {
|
||||
return;
|
||||
}
|
||||
fail(target, turn, emptyScrapeReason(target, scrapeFailed));
|
||||
return;
|
||||
}
|
||||
@@ -278,7 +384,33 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
return;
|
||||
}
|
||||
// fleetd#164 (part 2) / fleetd#201: a scrape that read cleanly and produced content still
|
||||
// isn't a real reply when that content is the backend's own rejection (e.g. an HTTP 400
|
||||
// before the worker did any work). Classify it as a failure naming the member, rather than
|
||||
// handing the caller a scrape that reads like a completed answer, and — only on the
|
||||
// resolution that actually wins the race, mirroring the exhaustion sink above — notify the
|
||||
// typed backend-error sink so a later stage can act on repeated failures.
|
||||
String backendError = firstMatchingLine(assistantBlock, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
// Carry the whole scrape, not just the matched line. The pattern is a heuristic: a member
|
||||
// that forgot fleet_reply while reporting *about* a backend error matches it too. Failing
|
||||
// is still right — the caller must not read a scrape as an answer — but dropping the rest
|
||||
// of the pane would destroy the report, which is the same defect fleetd#164 is about.
|
||||
String reason = "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + tail;
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
// fleetd#201 Unit 1: only on the resolution that actually won the race — a late
|
||||
// duplicate must never double-count one backend failure.
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
String completion = clipped ? tail + "\n" + CLIPPED_PANE_TAIL_MARKER : tail;
|
||||
if (echoesInjectedBrief(tail, turn.injectedText())) {
|
||||
completion = noReportMessage(target) + (clipped ? "\n" + CLIPPED_PANE_TAIL_MARKER : "");
|
||||
}
|
||||
if (rendezvous.resolveCompletion(waiter, completion)) {
|
||||
inFlight.remove(target, turn);
|
||||
if (clipped) {
|
||||
@@ -291,6 +423,83 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A full echoed brief is at least 400 normalised characters. A scrape that contains the brief may
|
||||
* add no more than 160 normalised characters of TUI chrome. This accepts harmless status text, but
|
||||
* preserves a real report that restates the full brief before adding substantive content.
|
||||
*/
|
||||
static boolean echoesInjectedBrief(String scrape, String injectedText) {
|
||||
String normalScrape = normalize(scrape);
|
||||
String normalInjected = normalize(injectedText);
|
||||
if (normalScrape.length() < ECHO_MIN_CHARS || normalInjected.length() < ECHO_MIN_CHARS) {
|
||||
return false;
|
||||
}
|
||||
if (normalInjected.contains(normalScrape)) {
|
||||
return true;
|
||||
}
|
||||
return normalScrape.contains(normalInjected)
|
||||
&& normalScrape.length() - normalInjected.length() <= MAX_ECHO_EXCESS_CHARS;
|
||||
}
|
||||
|
||||
private static String normalize(String text) {
|
||||
return text == null ? "" : text.toLowerCase().replaceAll("[^a-z0-9]+", "");
|
||||
}
|
||||
|
||||
private String noReportMessage(String target) {
|
||||
WorktreeBranch location = worktreeBranches.apply(target);
|
||||
if (location == null || (location.worktree() == null && location.branch() == null)) {
|
||||
return NO_REPORT_PREFIX + "]";
|
||||
}
|
||||
String locationText = location.worktree() == null ? "" : " worktree=" + location.worktree();
|
||||
if (location.branch() != null) {
|
||||
locationText += " branch=" + location.branch();
|
||||
}
|
||||
return NO_REPORT_PREFIX + locationText + "]";
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#211: the raw-scrape fallback classification, run only when {@link #lastAssistantBlock}
|
||||
* found nothing usable (see the call site in {@link #resolve}). Mirrors the two classifications
|
||||
* the normal path already applies to the trimmed assistant block — exhaustion first, then the
|
||||
* narrow {@link #BACKEND_ERROR} pattern — against {@code raw} instead, and reports whether one of
|
||||
* them handled the turn (resolved the waiter or failed it) so the caller skips the empty-scrape
|
||||
* failure. Never runs on the normal (non-empty-block) path, and never touches the reply text.
|
||||
*/
|
||||
private boolean classifyRawScrapeFallback(String target, InFlight turn,
|
||||
CompletableFuture<Rendezvous.Resolution> waiter, String raw) {
|
||||
Pattern exhausted = exhaustedPatterns.patternFor(target);
|
||||
String matchedLine = exhausted == null ? null : firstMatchingLine(raw, exhausted);
|
||||
if (matchedLine != null) {
|
||||
String reason = "backend exhausted (usage limit): " + matchedLine;
|
||||
if (rendezvous.resolveExhausted(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("completion for {} classified BACKEND_EXHAUSTED from the raw scrape (no "
|
||||
+ "usable assistant block; no fleet_reply): {}", target, reason);
|
||||
// CB-578 stage B: only on the resolution that actually won the race — a late
|
||||
// duplicate must never quarantine a credential twice for one refusal.
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
String backendError = firstMatchingLine(raw, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
// Carry the pane, not just the matched line — the same fleetd#164 rule the normal path
|
||||
// above applies. Here it matters more, not less: the trimmed block was empty, so the raw
|
||||
// scrape is the ONLY copy of whatever the member managed to say. Clipped to the same cap
|
||||
// the normal path uses, since a raw screen has no boundary trimming to bound it.
|
||||
String reason = "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + clip(raw);
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback from the raw scrape: {}", target, reason);
|
||||
// fleetd#201 Unit 1: only on the resolution that actually won the race.
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/** Synchronous fail (the unit-testable core of {@link #onTurnFailed}). */
|
||||
void fail(String target, InFlight turn) {
|
||||
fail(target, turn, null);
|
||||
@@ -327,22 +536,53 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#164: the failure reason for a turn that settled inside {@link #MIN_TURN_NANOS} — names
|
||||
* the member and both timings, and appends whatever the pane shows (usually the backend's own
|
||||
* error) so the caller sees the cause, not just "it failed".
|
||||
* fleetd#164 (floor) / fleetd#201 (classification): fail a turn that settled inside
|
||||
* {@link #MIN_TURN_NANOS} — a crash signature (e.g. a backend HTTP 400 before the worker did
|
||||
* anything) that a bare {@code BUSY -> DONE} transition cannot be told apart from a genuinely
|
||||
* fast completion. Runs the same backend-error classification the normal and raw-scrape paths
|
||||
* apply, against whatever is on screen right now: a match is a typed failure that notifies
|
||||
* {@link #backendErrorSink} (only on the resolution that wins the race); a non-match stays the
|
||||
* original generic too-fast failure, naming the member and both timings, with whatever the pane
|
||||
* shows appended so the caller sees the cause, not just "it failed".
|
||||
*/
|
||||
private String tooFastReason(String target, long elapsedNanos) {
|
||||
private void failTooFast(String target, InFlight turn, CompletableFuture<Rendezvous.Resolution> waiter,
|
||||
long elapsedNanos) {
|
||||
String scrape;
|
||||
try {
|
||||
scrape = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
scrape = agents.read(target, SCRAPE_SOURCE);
|
||||
} catch (RuntimeException e) {
|
||||
scrape = "";
|
||||
}
|
||||
String reason = String.format(
|
||||
String clippedScrape = clip(scrape);
|
||||
String baseReason = String.format(
|
||||
"member %s went BUSY -> DONE in %dms (floor %dms) — too fast to be real work, most "
|
||||
+ "likely a backend error before any work started",
|
||||
target, elapsedNanos / 1_000_000, MIN_TURN_NANOS / 1_000_000);
|
||||
return scrape.isBlank() ? reason : reason + ": " + scrape;
|
||||
String backendError = firstMatchingLine(scrape, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
String reason = baseReason + ": " + clippedScrape;
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
// fleetd#201 Unit 1: only on the resolution that actually won the race.
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
fail(target, turn, clippedScrape.isBlank() ? baseReason : baseReason + ": " + clippedScrape);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#201: the pattern to classify a backend error against for {@code target} — its
|
||||
* profile's configured {@link BackendErrorPatternLookup} entry when there is one, else the
|
||||
* built-in {@link #BACKEND_ERROR} compatibility pattern. A target with no configured pattern is
|
||||
* never "off": it always falls back to this narrow default, exactly like the classification
|
||||
* behaved before fleetd#201 (Unit 5 reports a target relying on this fallback separately, as
|
||||
* legacy-default coverage rather than full per-profile coverage).
|
||||
*/
|
||||
private Pattern backendErrorPatternOrFallback(String target) {
|
||||
Pattern configured = backendErrorPatterns.patternFor(target);
|
||||
return configured != null ? configured : BACKEND_ERROR;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -380,9 +620,9 @@ public final class CompletionResolver implements TurnListener {
|
||||
* @param allProfiles every configured profile name
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry an exhausted pattern
|
||||
*/
|
||||
public static String coverage(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
public static String coverage(String patternKey, Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
if (configuredProfiles.isEmpty()) {
|
||||
return "off (no profile has an exhaustedPattern configured; profiles: " + sorted(allProfiles) + ")";
|
||||
return "off (no profile has an " + patternKey + " configured; profiles: " + sorted(allProfiles) + ")";
|
||||
}
|
||||
Set<String> unconfigured = new TreeSet<>(allProfiles);
|
||||
unconfigured.removeAll(configuredProfiles);
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* Notified when {@link CompletionResolver} actually delivers a {@code BACKEND_EXHAUSTED}
|
||||
* classification to a waiting send (CB-578 stage B) — never on a race that lost (see
|
||||
@@ -10,22 +12,81 @@ package dev.ltms.fleet.inject;
|
||||
* profiles or credentials, so mapping {@code target} to whatever should be quarantined is entirely
|
||||
* the sink's job — see {@code Fleetd.main}'s wiring, which resolves target → session → profile →
|
||||
* {@code effectiveCredentialId()} and calls {@code BackendQuarantine.quarantine} on it.
|
||||
*
|
||||
* <p><strong>The 3-arg overload is the single abstract method — fleetd #234, round 4.</strong> Two
|
||||
* earlier rounds each shipped a caller that silently dropped the profile hint (see {@link
|
||||
* #onExhausted(String, String, String)}): a lambda written against this interface can only ever
|
||||
* implement whichever overload is abstract, and while the 2-arg form held that position, EVERY
|
||||
* lambda site — a call-site forwarder in {@code Fleetd.java}, a hand-built test double — bound to
|
||||
* it and silently inherited the profile-dropping default, whether or not its author remembered the
|
||||
* hazard. Making the 3-arg form abstract instead removes the shape entirely: a lambda declared
|
||||
* against this interface today is <em>forced</em> by the compiler to take {@code (target, reason,
|
||||
* profile)}, so there is no overload left for it to bind to that can drop the hint. This is a type
|
||||
* change, not a test — it holds even for a caller that has never heard of fleetd #234.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface ExhaustionSink {
|
||||
|
||||
/**
|
||||
* @param target the herdr terminal id whose turn was classified {@code BACKEND_EXHAUSTED}
|
||||
* @param reason the matched-line reason carried by the classification
|
||||
* @param target the herdr terminal id whose turn was classified {@code BACKEND_EXHAUSTED}
|
||||
* @param reason the matched-line reason carried by the classification
|
||||
* @param profile the profile the caller already knows should be quarantined, or {@code null}
|
||||
* when the caller has no better answer than {@code target} alone (fleetd #234):
|
||||
* a caller whose {@code target} is not yet resolvable through whatever roster
|
||||
* the sink's implementation consults — {@link
|
||||
* dev.ltms.fleet.member.OpenCodeLauncher}'s model-mismatch check fires from
|
||||
* {@code SessionAwareHandle.agentSessionId()}, which runs during {@code
|
||||
* SessionManager.acquire()} <em>before</em> that session is registered, so a
|
||||
* target -> session -> profile lookup finds nothing at that point. That launcher
|
||||
* already has its own {@code FleetConfig.Profile} in hand and does not need the
|
||||
* roster to know which profile to quarantine, so it supplies this directly
|
||||
* instead of leaving the sink to guess.
|
||||
*/
|
||||
void onExhausted(String target, String reason);
|
||||
void onExhausted(String target, String reason, String profile);
|
||||
|
||||
/**
|
||||
* Convenience for a caller with no profile to offer — every existing call site that predates
|
||||
* the hint (fleetd #234): {@link dev.ltms.fleet.inject.CompletionResolver}'s two call sites
|
||||
* always call with a {@code target} that IS live in the roster at the time of the call, so they
|
||||
* need no hint and keep working exactly as before, unchanged by this default.
|
||||
*
|
||||
* @param target as {@link #onExhausted(String, String, String)}
|
||||
* @param reason as {@link #onExhausted(String, String, String)}
|
||||
*/
|
||||
default void onExhausted(String target, String reason) {
|
||||
onExhausted(target, reason, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Inert sink — nothing happens on exhaustion. The explicit stand-in a caller (or a test not
|
||||
* exercising this feature) passes instead of a defaulting overload, exactly like
|
||||
* {@link ExhaustedPatternLookup#none()}.
|
||||
* {@link ExhaustedPatternLookup#none()}. Safe as a lambda: the 3-arg form is now the interface's
|
||||
* single abstract method, so a lambda here has no other overload to silently bind to instead —
|
||||
* it simply does nothing with all three arguments.
|
||||
*/
|
||||
static ExhaustionSink none() {
|
||||
return (target, reason) -> { };
|
||||
return (target, reason, profile) -> { };
|
||||
}
|
||||
|
||||
/**
|
||||
* A sink that forwards to whatever {@code target} currently supplies (fleetd #234). Exists to
|
||||
* break a genuine construction-order cycle: {@code Fleetd.main} builds its adapters (including
|
||||
* {@link dev.ltms.fleet.member.OpenCodeLauncher}) before {@code sessions} exists, so it cannot
|
||||
* hand them the real sink yet — it hands them a forwarder pointed at an {@code
|
||||
* AtomicReference<ExhaustionSink>} that starts at {@link #none()} and gets {@code .set()} to the
|
||||
* real sink once {@code sessions} is built. {@code target} is evaluated on every call, never
|
||||
* cached, so the forwarder keeps working after the reference is repointed.
|
||||
*
|
||||
* <p>Now safe as a one-line lambda (round 4): forwarding the single 3-arg abstract method
|
||||
* forwards everything a caller can supply — there is no separate 2-arg overload left for a
|
||||
* forwarder to bind to instead and silently lose the hint. Kept as a named factory rather than
|
||||
* written inline at each call site anyway, so a test can call the exact object {@code
|
||||
* Fleetd.java} builds instead of asserting a rebuilt copy of its shape (round 3's lesson).
|
||||
*
|
||||
* @param target supplies the sink to forward to, evaluated fresh on every call
|
||||
* @return a sink whose call delegates to {@code target.get()}
|
||||
*/
|
||||
static ExhaustionSink forwardingTo(Supplier<ExhaustionSink> target) {
|
||||
return (t, r, p) -> target.get().onExhausted(t, r, p);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@ package dev.ltms.fleet.inject;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
@@ -89,6 +90,7 @@ public final class Injector {
|
||||
public static final long POLL_INTERVAL_MILLIS = 250;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final HerdrRouter router;
|
||||
private final TurnListener turnListener;
|
||||
private final Predicate<String> ready; // CB-113: a target is deliverable only when available
|
||||
private final Consumer<String> forget; // CB-114: clear a gone worker's readiness/presence
|
||||
@@ -124,11 +126,25 @@ public final class Injector {
|
||||
public Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||
Consumer<String> forget) {
|
||||
this.agents = agents;
|
||||
this.router = null;
|
||||
this.turnListener = turnListener;
|
||||
this.ready = ready;
|
||||
this.forget = forget;
|
||||
}
|
||||
|
||||
public Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
||||
Consumer<String> forget) {
|
||||
this.agents = null;
|
||||
this.router = router;
|
||||
this.turnListener = turnListener;
|
||||
this.ready = ready;
|
||||
this.forget = forget;
|
||||
}
|
||||
|
||||
private AgentControl agentsFor(String target) {
|
||||
return router != null ? router.agentsFor(target) : agents;
|
||||
}
|
||||
|
||||
/** A pending message and the future that completes when it has been delivered. */
|
||||
private record Pending(String text, TurnToken token, CompletableFuture<Void> delivered) {
|
||||
}
|
||||
@@ -141,6 +157,7 @@ public final class Injector {
|
||||
boolean awaitingCompletion; // a delivered message's turn is not yet known-complete
|
||||
boolean turnObserved; // saw a real `working` sample since that delivery (turn ran)
|
||||
int unknownSinceTurn; // consecutive `unknown` samples while a delegation is outstanding (CB-109)
|
||||
int unknownSincePostTurn; // the same, for the post-turn housekeeping phase (fleetd #306)
|
||||
int notReadySincePoll; // consecutive injectable samples a queued message waited on the readiness gate (CB-114)
|
||||
boolean postTurnPending; // completion observed; adapter housekeeping has not started yet
|
||||
boolean awaitingPostTurnPickup;
|
||||
@@ -201,10 +218,12 @@ public final class Injector {
|
||||
t.awaitingPickup = false;
|
||||
t.injectableSincePickup = 0;
|
||||
t.unknownSinceTurn = 0;
|
||||
t.unknownSincePostTurn = 0;
|
||||
t.notReadySincePoll = 0;
|
||||
if (t.awaitingCompletion) t.turnObserved = true;
|
||||
} else if (status.injectable()) { // IDLE or BLOCKED
|
||||
t.unknownSinceTurn = 0;
|
||||
t.unknownSincePostTurn = 0;
|
||||
if (t.awaitingPostTurnPickup) {
|
||||
if (++t.injectableSincePostTurnPickup >= PICKUP_GRACE_POLLS) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
@@ -253,7 +272,7 @@ public final class Injector {
|
||||
if (p != null && ready.test(target)) {
|
||||
t.notReadySincePoll = 0;
|
||||
try {
|
||||
agents.send(target, p.text());
|
||||
agentsFor(target).send(target, p.text());
|
||||
t.queue.poll();
|
||||
t.awaitingPickup = true;
|
||||
t.awaitingCompletion = true;
|
||||
@@ -299,6 +318,24 @@ public final class Injector {
|
||||
t.unknownSinceTurn = 0;
|
||||
turnFailed = true;
|
||||
}
|
||||
// fleetd #306: the same escape for the post-turn housekeeping phase. Four latches
|
||||
// gate delivery (awaitingCompletion, postTurnPending, awaitingPostTurnPickup,
|
||||
// postTurnObserved) and only the first had a way out of a sustained unknown streak —
|
||||
// a gate that closed one direction only. The other two below are released here as
|
||||
// well; postTurnPending needs no escape because it is cleared unconditionally on the
|
||||
// line after the listener call that sets it.
|
||||
//
|
||||
// This does NOT set turnFailed. The delegated turn already completed and its waiter
|
||||
// already resolved — what is outstanding is adapter housekeeping (the `/clear`).
|
||||
// Reporting a turn failure here would drive SessionManager.onFailed on a session
|
||||
// that genuinely finished its work, which is a worse lie than the wedge.
|
||||
if ((t.awaitingPostTurnPickup || t.postTurnObserved)
|
||||
&& ++t.unknownSincePostTurn >= TURN_STALL_GRACE_POLLS) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
t.postTurnObserved = false;
|
||||
t.injectableSincePostTurnPickup = 0;
|
||||
t.unknownSincePostTurn = 0;
|
||||
}
|
||||
}
|
||||
|
||||
// Reclaim the entry once the worker is fully quiescent (nothing queued, no pickup or
|
||||
@@ -313,7 +350,7 @@ public final class Injector {
|
||||
// thread while it holds the target lock.
|
||||
if (resubmit) {
|
||||
try {
|
||||
agents.submit(target); // nudge a raced Enter so the pending paste submits
|
||||
agentsFor(target).submit(target); // nudge a raced Enter so the pending paste submits
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("resubmit to {} failed (will retry next poll): {}", target, e.getMessage());
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@ package dev.ltms.fleet.inject;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -22,6 +23,7 @@ public final class StatusPoller {
|
||||
private static final Logger log = LoggerFactory.getLogger(StatusPoller.class);
|
||||
|
||||
private final AgentControl agents;
|
||||
private final HerdrRouter router;
|
||||
private final Injector injector;
|
||||
private final StatusRefiner refiner;
|
||||
private final long intervalMillis;
|
||||
@@ -35,11 +37,24 @@ public final class StatusPoller {
|
||||
public StatusPoller(AgentControl agents, Injector injector, StatusRefiner refiner,
|
||||
long intervalMillis) {
|
||||
this.agents = agents;
|
||||
this.router = null;
|
||||
this.injector = injector;
|
||||
this.refiner = refiner;
|
||||
this.intervalMillis = intervalMillis;
|
||||
}
|
||||
|
||||
public StatusPoller(HerdrRouter router, Injector injector, long intervalMillis) {
|
||||
this.agents = null;
|
||||
this.router = router;
|
||||
this.injector = injector;
|
||||
// CB-185: this refiner's own AgentControl (member) is only a default for the legacy 2-arg
|
||||
// refine() overload — the loop below always calls the 3-arg refine(target, raw, control)
|
||||
// with the per-target control from router.agentsFor(target), so a lead target is refined
|
||||
// against the LEAD daemon even though this field points at the member one.
|
||||
this.refiner = new StatusRefiner(router.memberAgents());
|
||||
this.intervalMillis = intervalMillis;
|
||||
}
|
||||
|
||||
/** Start the polling loop on a virtual thread. Idempotent. */
|
||||
public synchronized void start() {
|
||||
if (running) return;
|
||||
@@ -56,7 +71,11 @@ public final class StatusPoller {
|
||||
try {
|
||||
// herdr's agent_status can misreport a settled worker as `unknown`; refine it
|
||||
// against the pane content before it drives delivery/completion (CB-115).
|
||||
AgentStatus status = refiner.refine(target, agents.status(target));
|
||||
// CB-185: refine THROUGH the same control the raw status came from — a router
|
||||
// splits lead/member targets across two herdr daemons, and reading a lead's pane
|
||||
// through the (fixed) member refiner never finds it, wedging that lead at UNKNOWN.
|
||||
AgentControl control = router != null ? router.agentsFor(target) : agents;
|
||||
AgentStatus status = refiner.refine(target, control.status(target), control);
|
||||
injector.onStatus(target, status);
|
||||
} catch (HerdrException e) {
|
||||
// The worker's agent is gone — stop trying and unblock its waiters.
|
||||
|
||||
@@ -41,15 +41,32 @@ public final class StatusRefiner {
|
||||
}
|
||||
|
||||
/**
|
||||
* Return a trustworthy status for {@code target}. Any non-{@code UNKNOWN} {@code raw} is returned
|
||||
* unchanged; an {@code UNKNOWN} triggers a pane read and content classification. A read failure
|
||||
* leaves it {@code UNKNOWN} (the safe default: no delivery, and the stall path still applies).
|
||||
* Return a trustworthy status for {@code target}, reading its pane through this refiner's own
|
||||
* {@link AgentControl}. Equivalent to {@link #refine(String, AgentStatus, AgentControl)} with
|
||||
* that control — kept for callers that only ever talk to one herdr daemon.
|
||||
*/
|
||||
public AgentStatus refine(String target, AgentStatus raw) {
|
||||
return refine(target, raw, agents);
|
||||
}
|
||||
|
||||
/**
|
||||
* Return a trustworthy status for {@code target}. Any non-{@code UNKNOWN} {@code raw} is returned
|
||||
* unchanged; an {@code UNKNOWN} triggers a pane read (through {@code control}) and content
|
||||
* classification. A read failure leaves it {@code UNKNOWN} (the safe default: no delivery, and
|
||||
* the stall path still applies).
|
||||
*
|
||||
* <p>CB-185: {@code control} must be the {@link AgentControl} for the <em>same</em> daemon the
|
||||
* raw status was sampled from — a router splits lead and member targets across two herdr
|
||||
* daemons, and reading a lead's pane through the member client (or vice versa) fails to find
|
||||
* the pane and leaves the target wedged at {@code UNKNOWN} forever. Callers that route per
|
||||
* target (e.g. {@code StatusPoller}) must pass that target's control explicitly rather than
|
||||
* relying on the control fixed at construction.
|
||||
*/
|
||||
public AgentStatus refine(String target, AgentStatus raw, AgentControl control) {
|
||||
if (raw != AgentStatus.UNKNOWN) return raw;
|
||||
String pane;
|
||||
try {
|
||||
pane = agents.read(target, PROBE_SOURCE);
|
||||
pane = control.read(target, PROBE_SOURCE);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("status refine read for {} failed; leaving UNKNOWN: {}", target, e.getMessage());
|
||||
return AgentStatus.UNKNOWN;
|
||||
|
||||
@@ -60,7 +60,30 @@ public final class ConnectionIdentity {
|
||||
return pid > 0 ? cwds.cwdForPid(pid) : null;
|
||||
}
|
||||
|
||||
private static boolean isLoopback(String addr) {
|
||||
return "127.0.0.1".equals(addr) || "::1".equals(addr) || "0:0:0:0:0:0:0:1".equals(addr);
|
||||
/**
|
||||
* Whether {@code addr} is a same-host address, and therefore one whose peer PID is worth
|
||||
* looking up. <strong>This is the one definition of loopback in the daemon</strong> —
|
||||
* {@code CallerResolver} calls it rather than keeping its own, because the two used to differ
|
||||
* and that difference was a privilege escalation (fleetd #305).
|
||||
*
|
||||
* <p>The whole of {@code 127.0.0.0/8} counts, not just {@code 127.0.0.1}. On Linux every
|
||||
* address in that range is bound to {@code lo} by default, so a process can connect to
|
||||
* {@code 127.0.0.1:8765} with a source address of {@code 127.0.0.2} — measured on the Linux
|
||||
* fleet host, where binding that source succeeds.
|
||||
*
|
||||
* <p><strong>Being strict here does not make the daemon safer; it makes it unsafe.</strong>
|
||||
* That reads backwards, so it is worth stating plainly. This predicate does not decide whether
|
||||
* a caller is trusted — it decides whether the caller's identity is <em>resolved at all</em>.
|
||||
* Returning false means {@link #resolve} answers "no terminal", and downstream a caller with no
|
||||
* terminal is treated as the primary under loopback-trust. So every address excluded here is an
|
||||
* address on which a worker silently becomes the lead. Widening a check normally weakens it;
|
||||
* widening this one is what closes the hole.
|
||||
*/
|
||||
public static boolean isLoopback(String addr) {
|
||||
if (addr == null) {
|
||||
return false;
|
||||
}
|
||||
String a = addr.startsWith("::ffff:") ? addr.substring(7) : addr; // IPv4-mapped IPv6
|
||||
return a.startsWith("127.") || "::1".equals(a) || "0:0:0:0:0:0:0:1".equals(a);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,16 +10,19 @@ import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
import dev.ltms.fleet.inject.MemberPresence;
|
||||
import dev.ltms.fleet.inject.CompletionResolver;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.msg.LeadChannel;
|
||||
import dev.ltms.fleet.msg.LeadMessage;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.ShuttingDownException;
|
||||
import dev.ltms.fleet.session.WorktreeRequest;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
@@ -92,6 +95,10 @@ public final class FleetMcp {
|
||||
private final CapacitySource capacity;
|
||||
private final HealthCoverageSource healthCoverage;
|
||||
private final QuarantineSource quarantine;
|
||||
/** fleetd #201 Unit 5: SEPARATE from {@link #quarantine} — see {@link OutageSource}'s doc. */
|
||||
private final OutageSource outage;
|
||||
/** fleetd #176: SEPARATE from both of the above — see {@link LeadSeatSource}'s doc. */
|
||||
private final LeadSeatSource leadSeats;
|
||||
/** CB-637: this daemon's lead-to-lead channel; {@code null} when no coordinator is configured. */
|
||||
private final LeadChannel leadChannel;
|
||||
|
||||
@@ -115,6 +122,48 @@ public final class FleetMcp {
|
||||
public static QuarantineSource none() { return new QuarantineSource(_ -> null, BackendQuarantine.none()); }
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #201 Unit 5 cool-off facts used by {@code fleet_list}/{@code fleet_profiles}: a
|
||||
* profile → credential id lookup, plus the shared {@link BackendOutagePolicy} to read remaining
|
||||
* cool-offs off. A SEPARATE source from {@link QuarantineSource} — never merged into it — so a
|
||||
* credential that is cooling off after repeated backend errors is reported independently of
|
||||
* whether it is ALSO under CB-578 stage B exhaustion quarantine; the two checks can both fire
|
||||
* for the same profile, and when they do, {@code fleet_list}/{@code fleet_profiles} report both
|
||||
* (see {@link #capacityView}/{@link #profiles}).
|
||||
*/
|
||||
public record OutageSource(Function<String, String> credentialIdFor, BackendOutagePolicy outagePolicy) {
|
||||
/** Inert source — no profile is ever reported cooling off. Explicit stand-in, not a default. */
|
||||
public static OutageSource none() {
|
||||
return new OutageSource(_ -> null, new BackendOutagePolicy(System::nanoTime));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176: the seats a profile's own live LEAD session(s) hold on the same Claude
|
||||
* subscription — a fact {@code fleet_list} reports alongside {@code free} via the
|
||||
* {@code leadSeats} key.
|
||||
*
|
||||
* <p>{@code maxLoad} counts only <em>members</em>, never the lead itself. But a
|
||||
* {@code subscription: true} profile bills the operator's own Claude account, and the lead is
|
||||
* always a live {@code claude} session on that same account (it is never moved off-subscription
|
||||
* — see {@code LeadLauncher}). See {@code Fleetd.leadSeatLookup} for how the count is derived —
|
||||
* from {@code fleet.leaders.<name>.profile} and each profile's {@code effectiveCredentialId()},
|
||||
* never a hardcoded constant.
|
||||
*
|
||||
* <p>fleetd #257: this count is reported, never subtracted from {@code free}. An earlier cut of
|
||||
* this feature subtracted it, on the theory that it made {@code free} describe the real ceiling
|
||||
* on the account — but the real spawn gate ({@code CompositePeerLauncher#enforceMaxLoad}) never
|
||||
* read this count at all, so the subtraction made {@code free} disagree with the one thing it is
|
||||
* supposed to describe: what a fresh {@code fleet_spawn} will actually get. No backend seat
|
||||
* ceiling shared with the lead has been measured either — see {@code fleetd.example.yaml}'s
|
||||
* {@code maxLoad} docs. {@code free} now always equals {@code max(0, maxLoad - live)}, and
|
||||
* {@code leadSeats} is reported purely as a fact the caller may act on however it likes.
|
||||
*/
|
||||
public record LeadSeatSource(Function<String, Integer> seatsFor) {
|
||||
/** Inert source — no profile is ever reported as sharing a seat with a lead. */
|
||||
public static LeadSeatSource none() { return new LeadSeatSource(_ -> 0); }
|
||||
}
|
||||
|
||||
/**
|
||||
* @param callers resolves each call's {@link Principal}; {@code null} disables authorization.
|
||||
* This surface needs its own enforcement: {@code /mcp} is a raw servlet on
|
||||
@@ -129,7 +178,7 @@ public final class FleetMcp {
|
||||
CallerResolver callers, Metrics metrics, CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine) {
|
||||
this(messages, workers, sessions, identity, presence, primaryRegistry, callers, metrics, capacity,
|
||||
healthCoverage, quarantine, null);
|
||||
healthCoverage, quarantine, null, OutageSource.none(), LeadSeatSource.none());
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -142,9 +191,43 @@ public final class FleetMcp {
|
||||
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||
CallerResolver callers, Metrics metrics, CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, LeadChannel leadChannel) {
|
||||
this(messages, workers, sessions, identity, presence, primaryRegistry, callers, metrics, capacity,
|
||||
healthCoverage, quarantine, leadChannel, OutageSource.none(), LeadSeatSource.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, with fleetd #201 Unit 5 cool-off facts for {@code fleet_list}/{@code fleet_profiles}
|
||||
* (see {@link OutageSource}).
|
||||
*
|
||||
* @param outage required — pass {@link OutageSource#none()} for a caller that does not want the
|
||||
* feature, never a defaulting overload (the same rule {@code quarantine} follows).
|
||||
*/
|
||||
public FleetMcp(MessageService messages, PeerLauncher workers, SessionManager sessions,
|
||||
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||
CallerResolver callers, Metrics metrics, CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, LeadChannel leadChannel, OutageSource outage) {
|
||||
this(messages, workers, sessions, identity, presence, primaryRegistry, callers, metrics, capacity,
|
||||
healthCoverage, quarantine, leadChannel, outage, LeadSeatSource.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, with fleetd #176 lead-seat facts (see {@link LeadSeatSource}). This is what
|
||||
* {@code Fleetd.main} actually wires up.
|
||||
*
|
||||
* @param leadSeats required — pass {@link LeadSeatSource#none()} for a caller that does not want
|
||||
* the feature, never a defaulting overload (the same rule {@code quarantine} and
|
||||
* {@code outage} follow).
|
||||
*/
|
||||
public FleetMcp(MessageService messages, PeerLauncher workers, SessionManager sessions,
|
||||
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||
CallerResolver callers, Metrics metrics, CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, LeadChannel leadChannel, OutageSource outage,
|
||||
LeadSeatSource leadSeats) {
|
||||
this.leadChannel = leadChannel;
|
||||
this.capacity = capacity;
|
||||
this.quarantine = Objects.requireNonNull(quarantine, "quarantine");
|
||||
this.outage = Objects.requireNonNull(outage, "outage");
|
||||
this.leadSeats = Objects.requireNonNull(leadSeats, "leadSeats");
|
||||
this.healthCoverage = healthCoverage;
|
||||
McpJsonMapper json = new JacksonMcpJsonMapperSupplier().get();
|
||||
this.transport = HttpServletStreamableServerTransportProvider.builder()
|
||||
@@ -175,7 +258,7 @@ public final class FleetMcp {
|
||||
// Each handler is built once and wired to its fleet_* tool below.
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> sendHandler =
|
||||
(exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.SEND,
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_send", req.arguments()),
|
||||
str(req.arguments(), "sessionId"));
|
||||
if (denied != null) return denied;
|
||||
String caller = callerTerminal(exchange);
|
||||
@@ -217,7 +300,7 @@ public final class FleetMcp {
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> replyHandler =
|
||||
(exchange, req) -> {
|
||||
String self = callerTerminal(exchange);
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.REPLY, self);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_reply", req.arguments()), self);
|
||||
if (denied != null) return denied;
|
||||
return reply(messages, self, str(req.arguments(), "content"));
|
||||
};
|
||||
@@ -225,36 +308,38 @@ public final class FleetMcp {
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> askHandler =
|
||||
(exchange, req) -> {
|
||||
String self = callerTerminal(exchange);
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.ASK, self);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_ask", req.arguments()), self);
|
||||
if (denied != null) return denied;
|
||||
return ask(messages, self, str(req.arguments(), "question"), timeoutMs(req.arguments()));
|
||||
};
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> statusHandler =
|
||||
(exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_status", req.arguments()), null);
|
||||
if (denied != null) return denied;
|
||||
return status(messages, str(req.arguments(), "sessionId"));
|
||||
};
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> pollHandler =
|
||||
(exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
Map<String, Object> a = req.arguments();
|
||||
return poll(messages, str(a, "ticket"), str(a, "target"));
|
||||
String target = str(a, "target");
|
||||
// The action depends on the ARGUMENTS, not on the tool name -- see pollAction.
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_poll", a), target);
|
||||
if (denied != null) return denied;
|
||||
return poll(messages, str(a, "ticket"), target);
|
||||
};
|
||||
// CB-307 Increment 3: per-msgId ack (not needed in v1 but supported by the inbox).
|
||||
// Acking removes a reply from the inbox, so it is a drain, not a read.
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> ackHandler =
|
||||
(exchange, req) -> {
|
||||
Map<String, Object> a = req.arguments();
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.DRAIN, str(a, "target"));
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_ack", a), str(a, "target"));
|
||||
if (denied != null) return denied;
|
||||
return ack(messages, str(a, "target"), str(a, "msgId"));
|
||||
};
|
||||
// Fleet management (CB-108): spawn/list/stop over ClaudeCodeLauncher.
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> spawnHandler =
|
||||
(exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.SPAWN, null);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_spawn", req.arguments()), null);
|
||||
if (denied != null) return denied;
|
||||
String caller = callerTerminal(exchange);
|
||||
// SPAWN is already auth-gated to PRIMARY (architects can never call it), but
|
||||
@@ -271,29 +356,29 @@ public final class FleetMcp {
|
||||
};
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> listHandler =
|
||||
(exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_list", Map.of()), null);
|
||||
if (denied != null) return denied;
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine,
|
||||
callers == null ? Map.of() : callers.leads(),
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
||||
leadSeats, callers == null ? Map.of() : callers.leads(),
|
||||
callerTerminal(exchange),
|
||||
leadChannel == null ? null : leadChannel.selfCoordId());
|
||||
};
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> stopHandler =
|
||||
(exchange, req) -> {
|
||||
String paneId = str(req.arguments(), "paneId");
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.STOP, paneId);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_stop", req.arguments()), paneId);
|
||||
if (denied != null) return denied;
|
||||
return stop(sessions, paneId);
|
||||
};
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> profilesHandler =
|
||||
(exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_profiles", Map.of()), null);
|
||||
if (denied != null) return denied;
|
||||
return profiles(workers, quarantine);
|
||||
return profiles(workers, quarantine, outage);
|
||||
};
|
||||
BiFunction<McpSyncServerExchange, McpSchema.CallToolRequest, McpSchema.CallToolResult> whoamiHandler =
|
||||
(exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_whoami", Map.of()), null);
|
||||
if (denied != null) return denied;
|
||||
return whoami(principal(exchange), sessions);
|
||||
};
|
||||
@@ -543,8 +628,9 @@ public final class FleetMcp {
|
||||
case REPLIED -> text(r.text());
|
||||
// The worker's turn finished but it never called fleet_reply — hand back the scraped
|
||||
// transcript tail, flagged so the primary knows it isn't a structured reply.
|
||||
case COMPLETED_UNREPLIED -> text(
|
||||
"[worker finished without a structured fleet_reply — transcript tail follows]\n" + r.text());
|
||||
case COMPLETED_UNREPLIED -> text(r.text().startsWith(CompletionResolver.NO_REPORT_PREFIX)
|
||||
? r.text()
|
||||
: "[worker finished without a structured fleet_reply — transcript tail follows]\n" + r.text());
|
||||
// The worker ran the turn then wedged (CB-109) — surface the error context.
|
||||
case WORKER_FAILED -> text("[worker failed — turn ended in an unrecoverable state]\n" + r.text());
|
||||
// The backend refused on a subscription usage limit (CB-578 stage A) — the worker's
|
||||
@@ -641,6 +727,53 @@ public final class FleetMcp {
|
||||
return text("delivered to peer lead " + coordId + " (msgId " + msg.msgId() + ")");
|
||||
}
|
||||
|
||||
/**
|
||||
* Which authorization action a {@code fleet_poll} call needs, decided by its arguments
|
||||
* (fleetd #272).
|
||||
*
|
||||
* <p>{@code fleet_poll} is <strong>two operations behind one tool name</strong>. With {@code
|
||||
* ticket} it observes an async delegation and changes nothing, which is a {@link
|
||||
* Authz.Action#READ}. With {@code target} it calls {@link MessageService#drainReplies} on that
|
||||
* session -- the replies are removed from the inbox and a second call returns nothing -- so it
|
||||
* is a {@link Authz.Action#DRAIN}, the same gate {@code fleet_ack} already uses for removing a
|
||||
* single message, and the same one the REST path uses at {@code FleetApp.drainReplies}.
|
||||
*
|
||||
* <p>Until this method existed the handler passed a constant {@code READ} for both branches.
|
||||
* {@code READ} is open to every authenticated role, so any worker could read a peer's id out of
|
||||
* {@code fleet_list} and destroy the replies that peer had queued for the primary. The gate
|
||||
* failed open, and it did so because the required action is a function of the arguments while
|
||||
* the handler chose it before looking at them.
|
||||
*
|
||||
* <p>The choice lives in this method, and not inline in the handler, so that a test can assert
|
||||
* the mapping the handler actually uses. {@code FleetMcpAuthzTest} already checked every
|
||||
* {@link Authz.Action} against every {@link Role} and passed throughout -- it tested the policy
|
||||
* table, which was correct, while the defect was in which action the caller handed it.
|
||||
*
|
||||
* @param target the {@code target} argument of the call, or {@code null}/blank when absent
|
||||
*/
|
||||
static Authz.Action pollAction(String target) {
|
||||
return isBlank(target) ? Authz.Action.READ : Authz.Action.DRAIN;
|
||||
}
|
||||
|
||||
/**
|
||||
* The action a registered tool handler actually hands to the authorization gate.
|
||||
* Keeping this choice beside the registered-tool inventory makes a new tool fail the coverage
|
||||
* test until its action is pinned.
|
||||
*/
|
||||
static Authz.Action toolAction(String toolName, Map<String, Object> arguments) {
|
||||
return switch (toolName) {
|
||||
case "fleet_send" -> Authz.Action.SEND;
|
||||
case "fleet_reply" -> Authz.Action.REPLY;
|
||||
case "fleet_ask" -> Authz.Action.ASK;
|
||||
case "fleet_status", "fleet_list", "fleet_profiles", "fleet_whoami" -> Authz.Action.READ;
|
||||
case "fleet_poll" -> pollAction(str(arguments, "target"));
|
||||
case "fleet_ack" -> Authz.Action.DRAIN;
|
||||
case "fleet_spawn" -> Authz.Action.SPAWN;
|
||||
case "fleet_stop" -> Authz.Action.STOP;
|
||||
default -> throw new IllegalArgumentException("unregistered tool: " + toolName);
|
||||
};
|
||||
}
|
||||
|
||||
/** {@code fleet_poll}: check an async delegation by ticket, or drain a worker's inbox by target. */
|
||||
static McpSchema.CallToolResult poll(MessageService messages, String ticket, String target) {
|
||||
if (!isBlank(target)) {
|
||||
@@ -659,6 +792,7 @@ public final class FleetMcp {
|
||||
}
|
||||
return switch (v.phase()) {
|
||||
case DONE -> text(v.replySource() != null && v.replySource().equals("transcript")
|
||||
&& !v.reply().startsWith(CompletionResolver.NO_REPORT_PREFIX)
|
||||
? "[done — worker finished without a structured fleet_reply; transcript tail follows]\n" + v.reply()
|
||||
: v.reply());
|
||||
case PENDING -> text("[pending — " + v.detail() + "]");
|
||||
@@ -680,7 +814,12 @@ public final class FleetMcp {
|
||||
return error("fleet_reply is for workers only — could not identify the calling worker "
|
||||
+ "from the connection");
|
||||
}
|
||||
if (content == null) {
|
||||
// fleetd #302: isBlank, not == null, to match fleet_send's own guard above. MessageService
|
||||
// .reply now REJECTS blank content, and this handler is a bare BiFunction with no try/catch
|
||||
// around it — so a whitespace-only fleet_reply would leave here as an uncaught
|
||||
// IllegalArgumentException instead of this clean tool error. Null and whitespace are the
|
||||
// same mistake by the caller and must get the same answer.
|
||||
if (isBlank(content)) {
|
||||
return error("content is required");
|
||||
}
|
||||
messages.reply(callerTerminal, content);
|
||||
@@ -824,6 +963,10 @@ public final class FleetMcp {
|
||||
return text(json(memberView(member)));
|
||||
} catch (GuardException e) {
|
||||
return error("subscription boundary: " + e.getMessage());
|
||||
} catch (ShuttingDownException e) {
|
||||
// fleetd #308: the daemon's shutdown drain has already started — refuse loudly rather
|
||||
// than register a session drainAll will never see again.
|
||||
return error("shutting down: " + e.getMessage());
|
||||
} catch (PlacementException e) {
|
||||
// CB-599: no candidate had capacity (maxLoad, quarantine, or all-exhausted) — distinct
|
||||
// from "profile does not exist" below.
|
||||
@@ -872,26 +1015,67 @@ public final class FleetMcp {
|
||||
* has ever been quarantined gets exactly the pre-stage-B shape.
|
||||
*/
|
||||
static McpSchema.CallToolResult profiles(PeerLauncher workers, QuarantineSource quarantine) {
|
||||
return profiles(workers, quarantine, OutageSource.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, plus a SEPARATE {@code coolingOff} map (fleetd #201 Unit 5) — never merged into
|
||||
* {@code quarantined} — for a profile whose credential is cooling off after repeated backend
|
||||
* errors ({@link BackendOutagePolicy}). The two checks are independent: a profile can appear in
|
||||
* both maps at once when it is both exhaustion-quarantined AND cooling off.
|
||||
*/
|
||||
static McpSchema.CallToolResult profiles(PeerLauncher workers, QuarantineSource quarantine, OutageSource outage) {
|
||||
return text(json(profilesView(workers, quarantine, outage)));
|
||||
}
|
||||
|
||||
/**
|
||||
* The body both front doors answer {@code profiles} with: the configured profile names, the
|
||||
* default, and the two independent outage states — {@code quarantined} (the backend reported it
|
||||
* out of capacity) and {@code coolingOff} (the credential threw repeated non-exhaustion backend
|
||||
* errors). Each map is present only when at least one profile is in that state, and a profile
|
||||
* can appear in both at once, because the two checks are separate.
|
||||
*
|
||||
* <p>fleetd #297: extracted so {@code fleet_profiles} and {@code GET /profiles} render from ONE
|
||||
* body builder rather than two copies. Passing both doors the same {@link QuarantineSource} and
|
||||
* {@link OutageSource} instances is necessary but not sufficient: with the loop written out
|
||||
* twice, a later edit to the row shape — a renamed key, an added field — lands on one door and
|
||||
* not the other, and the two then disagree about a live outage. That is exactly what fleetd
|
||||
* #284 was, where one rule computed in two places was widened in only one and a single response
|
||||
* contradicted itself. Shared inputs do not make duplicated computation safe.
|
||||
*/
|
||||
public static Map<String, Object> profilesView(PeerLauncher workers, QuarantineSource quarantine, OutageSource outage) {
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
result.put("profiles", workers.profiles());
|
||||
result.put("default", workers.defaultProfile() == null ? "" : workers.defaultProfile());
|
||||
Map<String, Object> quarantined = new LinkedHashMap<>();
|
||||
Map<String, Object> coolingOff = new LinkedHashMap<>();
|
||||
for (String profile : workers.profiles()) {
|
||||
String credentialId = quarantine.credentialIdFor().apply(profile);
|
||||
if (credentialId == null) {
|
||||
continue;
|
||||
if (credentialId != null) {
|
||||
quarantine.quarantine().remainingSeconds(credentialId).ifPresent(remaining -> {
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("credentialId", credentialId);
|
||||
row.put("quarantinedForSeconds", remaining);
|
||||
quarantined.put(profile, row);
|
||||
});
|
||||
}
|
||||
String outageCredentialId = outage.credentialIdFor().apply(profile);
|
||||
if (outageCredentialId != null) {
|
||||
outage.outagePolicy().remainingCoolOffSeconds(outageCredentialId).ifPresent(remaining -> {
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("credentialId", outageCredentialId);
|
||||
row.put("coolingOffForSeconds", remaining);
|
||||
coolingOff.put(profile, row);
|
||||
});
|
||||
}
|
||||
quarantine.quarantine().remainingSeconds(credentialId).ifPresent(remaining -> {
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("credentialId", credentialId);
|
||||
row.put("quarantinedForSeconds", remaining);
|
||||
quarantined.put(profile, row);
|
||||
});
|
||||
}
|
||||
if (!quarantined.isEmpty()) {
|
||||
result.put("quarantined", quarantined);
|
||||
}
|
||||
return text(json(result));
|
||||
if (!coolingOff.isEmpty()) {
|
||||
result.put("coolingOff", coolingOff);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -929,6 +1113,15 @@ public final class FleetMcp {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, leads, selfTerm, null);
|
||||
}
|
||||
|
||||
/** As above, plus fleetd #201 Unit 5 cool-off facts (see {@link OutageSource}). */
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, OutageSource outage,
|
||||
Map<String, String> leads, String selfTerm) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
||||
LeadSeatSource.none(), leads, selfTerm, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, additionally reporting this daemon's own lead coordination id (CB-637) when one is
|
||||
* configured and its channel opened. There is no peer-discovery surface yet — a lead addresses a
|
||||
@@ -942,6 +1135,25 @@ public final class FleetMcp {
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, Map<String, String> leads, String selfTerm,
|
||||
String selfCoordId) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, OutageSource.none(),
|
||||
LeadSeatSource.none(), leads, selfTerm, selfCoordId);
|
||||
}
|
||||
|
||||
/** As above, plus fleetd #201 Unit 5 cool-off facts (see {@link OutageSource}). */
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, OutageSource outage,
|
||||
Map<String, String> leads, String selfTerm, String selfCoordId) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
||||
LeadSeatSource.none(), leads, selfTerm, selfCoordId);
|
||||
}
|
||||
|
||||
/** As above, plus fleetd #176 lead-seat facts (see {@link LeadSeatSource}). */
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, OutageSource outage,
|
||||
LeadSeatSource leadSeats, Map<String, String> leads, String selfTerm,
|
||||
String selfCoordId) {
|
||||
try {
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.map(Agent.class::cast)
|
||||
@@ -951,7 +1163,10 @@ public final class FleetMcp {
|
||||
.sorted(Map.Entry.comparingByValue())
|
||||
.map(e -> leadView(e.getKey(), e.getValue(), live.get(e.getKey()), selfTerm))
|
||||
.toList();
|
||||
List<MemberSession> roster = sessions.roster();
|
||||
// fleetd #209: this is the caller-driven fleet_list read that actually reports
|
||||
// agentSessionId (via memberCapacityView -> SessionManager.rosterView), so it uses the
|
||||
// resolving roster; the heartbeat/health/metrics timers stay on the plain sessions.roster().
|
||||
List<MemberSession> roster = sessions.rosterResolved();
|
||||
List<Map<String, Object>> out = roster.stream()
|
||||
.map(s -> memberCapacityView(s, live.get(s.terminalId()), messages, capacity.clock().getAsLong()))
|
||||
.toList();
|
||||
@@ -965,7 +1180,7 @@ public final class FleetMcp {
|
||||
}
|
||||
if (capacity.available()) result.put("capacity", profiles.stream()
|
||||
.map(profile -> capacityView(profile, capacity.liveCount(), capacity.maxLoad(), roster, messages,
|
||||
capacity.clock().getAsLong(), quarantine)).toList());
|
||||
capacity.clock().getAsLong(), quarantine, outage, leadSeats)).toList());
|
||||
return text(json(result));
|
||||
} catch (HerdrException e) {
|
||||
return error("herdr error listing the fleet: " + e.getMessage());
|
||||
@@ -981,15 +1196,35 @@ public final class FleetMcp {
|
||||
private static Map<String, Object> memberCapacityView(MemberSession session, Agent live,
|
||||
MessageService messages, long nowNanos) {
|
||||
Map<String, Object> row = SessionManager.rosterView(session, live);
|
||||
boolean open = messages != null && messages.hasAcceptedDelivery(session.terminalId());
|
||||
boolean inbox = messages != null && messages.hasInboxMessage(session.terminalId());
|
||||
boolean reclaimable = (session.state() == MemberSession.State.READY || session.state() == MemberSession.State.DONE)
|
||||
&& !open && !inbox;
|
||||
boolean reclaimable = reclaimable(session, messages);
|
||||
row.put("reclaimable", reclaimable);
|
||||
row.put("idleForSeconds", reclaimable ? Math.max(0, (nowNanos - session.lastActivityAtNanos()) / 1_000_000_000L) : null);
|
||||
return row;
|
||||
}
|
||||
|
||||
/**
|
||||
* The one definition of {@code reclaimable}: this member holds a spawn seat, and has no open
|
||||
* bridge work, so stopping it gives the seat back. Both views in a single {@code fleet_list}
|
||||
* response call it — the per-member flag in {@link #memberCapacityView} and the per-profile
|
||||
* count in {@link #capacityView} — because two copies of this rule in one response is how the
|
||||
* two numbers come to disagree.
|
||||
*
|
||||
* <p>fleetd #284: {@code BACKEND_ERROR} and {@code FAILED} are deliberately NOT reclaimable.
|
||||
* The ticket asked for them to be, and that half of the ticket was wrong. Once
|
||||
* {@code Fleetd.liveSessionCount} stopped counting a terminal session as live, that seat is
|
||||
* ALREADY in {@code free}; counting it here too reports the same seat twice, and
|
||||
* {@code free + reclaimable} then reads as more capacity than {@code maxLoad} allows. The dead
|
||||
* session stays visible either way: its roster row still carries {@code state:
|
||||
* "backend_error"} or {@code "failed"}, which is what tells the lead to stop it.
|
||||
*/
|
||||
static boolean reclaimable(MemberSession session, MessageService messages) {
|
||||
boolean holdsSeat = session.state() == MemberSession.State.READY
|
||||
|| session.state() == MemberSession.State.DONE;
|
||||
return holdsSeat && (messages == null
|
||||
|| (!messages.hasAcceptedDelivery(session.terminalId())
|
||||
&& !messages.hasInboxMessage(session.terminalId())));
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-583: {@code free} alone cannot tell a lead "busy, will free up" from "refusing, and
|
||||
* nothing changes for N seconds" — those need different decisions. So a quarantined profile
|
||||
@@ -998,19 +1233,50 @@ public final class FleetMcp {
|
||||
* reports, reusing {@link QuarantineSource} rather than a second lookup. Both new keys are
|
||||
* added only when the profile is actually quarantined, so an ordinary fleet's rows are
|
||||
* byte-identical to before this change.
|
||||
*
|
||||
* <p>fleetd #201 Unit 5: a profile whose credential is cooling off (a SEPARATE, independent
|
||||
* check from quarantine — see {@link OutageSource}) also forces {@code free} to 0 and carries
|
||||
* {@code credentialId}/{@code coolingOffForSeconds}, but never {@code quarantinedForSeconds} —
|
||||
* that key is added only when exhaustion quarantine is ALSO active for this profile, since the
|
||||
* two checks are independent and either, both, or neither can be true.
|
||||
*
|
||||
* <p>fleetd #176: {@code maxLoad} counts panes, not subscription seats — it never counted the
|
||||
* lead's own seat on a {@code subscription: true} profile's account. {@link LeadSeatSource}
|
||||
* reports that count (0 for a non-subscription profile, or when no live lead shares its
|
||||
* credential) via the {@code leadSeats} key, added only when the count is positive, for the
|
||||
* same byte-identical-when-unused reason as the quarantine/cool-off keys above.
|
||||
*
|
||||
* <p>fleetd #257: {@code leadSeatCount} is reported, never subtracted from {@code free}.
|
||||
* {@code free} means "what a fresh {@code fleet_spawn} on this profile will actually get", and
|
||||
* the real gate ({@code CompositePeerLauncher#enforceMaxLoad}) only ever compares {@code live}
|
||||
* against {@code maxLoad} — it has no notion of a lead's own seat. Subtracting
|
||||
* {@code leadSeatCount} here made {@code free} disagree with the gate it is supposed to
|
||||
* describe: it could report {@code free: 0} while a spawn on that exact profile still
|
||||
* succeeded. {@code leadSeats} stays in the row as a fact the caller can act on however it
|
||||
* likes, but it no longer changes what {@code free} means.
|
||||
*/
|
||||
private static Map<String, Object> capacityView(String profile, Function<String, Integer> liveCount,
|
||||
Function<String, Integer> maxLoad, List<MemberSession> roster,
|
||||
MessageService messages, long nowNanos, QuarantineSource quarantine) {
|
||||
MessageService messages, long nowNanos, QuarantineSource quarantine,
|
||||
OutageSource outage, LeadSeatSource leadSeats) {
|
||||
Integer cap = maxLoad.apply(profile);
|
||||
int live = liveCount.apply(profile);
|
||||
int leadSeatCount = leadSeats.seatsFor().apply(profile);
|
||||
int reclaimable = (int) roster.stream().filter(s -> profile.equals(s.profile()))
|
||||
.filter(s -> (s.state() == MemberSession.State.READY || s.state() == MemberSession.State.DONE))
|
||||
.filter(s -> messages == null || (!messages.hasAcceptedDelivery(s.terminalId()) && !messages.hasInboxMessage(s.terminalId())))
|
||||
.filter(s -> reclaimable(s, messages))
|
||||
.count();
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("profile", profile); row.put("maxLoad", cap); row.put("live", live);
|
||||
row.put("free", cap == null ? null : Math.max(0, cap - live)); row.put("reclaimable", reclaimable);
|
||||
// fleetd #257: free must report what the real spawn gate (CompositePeerLauncher#enforceMaxLoad)
|
||||
// will actually grant, and that gate never reads leadSeatCount — only maxLoad and live. Not
|
||||
// subtracting the lead's seat here used to make free UNDERSTATE what a fresh fleet_spawn would
|
||||
// get, so a lead believing free:0 gave up on a profile the gate would still spawn onto.
|
||||
// leadSeatCount is still reported via the leadSeats key below, just never subtracted from free.
|
||||
row.put("free", cap == null ? null : Math.max(0, cap - live));
|
||||
row.put("reclaimable", reclaimable);
|
||||
if (leadSeatCount > 0) {
|
||||
row.put("leadSeats", leadSeatCount);
|
||||
}
|
||||
String credentialId = quarantine.credentialIdFor().apply(profile);
|
||||
if (credentialId != null) {
|
||||
quarantine.quarantine().remainingSeconds(credentialId).ifPresent(remaining -> {
|
||||
@@ -1019,6 +1285,14 @@ public final class FleetMcp {
|
||||
row.put("quarantinedForSeconds", remaining);
|
||||
});
|
||||
}
|
||||
String outageCredentialId = outage.credentialIdFor().apply(profile);
|
||||
if (outageCredentialId != null) {
|
||||
outage.outagePolicy().remainingCoolOffSeconds(outageCredentialId).ifPresent(remaining -> {
|
||||
row.put("free", 0);
|
||||
row.put("credentialId", outageCredentialId);
|
||||
row.put("coolingOffForSeconds", remaining);
|
||||
});
|
||||
}
|
||||
return row;
|
||||
}
|
||||
|
||||
@@ -1158,7 +1432,11 @@ public final class FleetMcp {
|
||||
+ "worktree:<ticket-slug> to provision an isolated git worktree. Pass resumeSessionId "
|
||||
+ "to relaunch onto a prior conversation instead of starting cold — this requires an "
|
||||
+ "explicit profile whose backend supports it (fleet_list shows agentSessionId for "
|
||||
+ "resumable members), and is refused otherwise rather than silently starting fresh. "
|
||||
+ "resumable members; it is absent for a member fleetd cannot reliably re-identify, "
|
||||
+ "e.g. an opencode member spawned without a worktree), and is refused otherwise "
|
||||
+ "rather than silently starting fresh. For an opencode profile, resumeSessionId "
|
||||
+ "itself also requires worktree:true/<slug> on THIS spawn — without one fleetd can "
|
||||
+ "never re-verify which conversation it actually resumed (fleetd #249). "
|
||||
+ "sessionName gives the member a display name in its own UI when the backend supports "
|
||||
+ "one. Returns the member's sessionId (use with fleet_send) and paneId (use with "
|
||||
+ "fleet_stop).",
|
||||
@@ -1169,7 +1447,7 @@ public final class FleetMcp {
|
||||
"worktree", Map.of("type", "string", "description", "'true' or a ticket slug — requests an isolated git worktree"),
|
||||
"ticket", stringProp("Ticket slug when worktree:true"),
|
||||
"sessionName", stringProp("Logical display name for the member's own session, when its backend supports one"),
|
||||
"resumeSessionId", stringProp("A prior member's agentSessionId (from fleet_list) to resume — requires an explicit profile that supports it")),
|
||||
"resumeSessionId", stringProp("A prior member's agentSessionId (from fleet_list) to resume — requires an explicit profile that supports it, and (for opencode) a worktree on this spawn too")),
|
||||
List.of()));
|
||||
}
|
||||
|
||||
@@ -1190,12 +1468,22 @@ public final class FleetMcp {
|
||||
+ "discover a peer lead without being told its address. 'members' are the "
|
||||
+ "sessions delegated to — each with sessionId, paneId, role (architect/dev/"
|
||||
+ "reviewer), profile (the backend it runs on), state, optional "
|
||||
+ "worktree/branch/owner/agentSessionId (the id to pass as fleet_spawn's "
|
||||
+ "resumeSessionId to relaunch onto that same conversation, when the backend "
|
||||
+ "supports it), and live herdr status. An empty 'members' "
|
||||
+ "worktree/branch/owner/agentSessionId, and live herdr status. agentSessionId, "
|
||||
+ "when present, is the id to pass as fleet_spawn's resumeSessionId to relaunch "
|
||||
+ "onto that same conversation. It is ABSENT — not a guess — for a member fleetd "
|
||||
+ "cannot reliably re-identify: some backends (e.g. opencode) resolve it from the "
|
||||
+ "member's working directory, which only uniquely identifies a member when it "
|
||||
+ "was spawned into its own fleetd-provisioned worktree (worktree:true/<slug>); a "
|
||||
+ "member spawned without one shares its directory with others and never reports "
|
||||
+ "an id, however long it runs (fleetd #249). An empty 'members' "
|
||||
+ "means no members are spawned; it says nothing about peers. When capacity "
|
||||
+ "facts are configured, a 'capacity' row per profile also reports free: 0 for "
|
||||
+ "a quarantined profile's credential (see fleet_profiles), whatever its "
|
||||
+ "facts are configured, a 'capacity' row per profile reports 'free' — the "
|
||||
+ "slots a fresh fleet_spawn on that profile will actually be granted right "
|
||||
+ "now (max(0, maxLoad - live)), the same check the spawn gate itself runs. A "
|
||||
+ "'leadSeats' key, when present, reports how many of those live slots are a "
|
||||
+ "lead session sharing this profile's subscription — informational only, "
|
||||
+ "already NOT subtracted from 'free' (fleetd #257). It also reports free: 0 "
|
||||
+ "for a quarantined profile's credential (see fleet_profiles), whatever its "
|
||||
+ "maxLoad/live — with credentialId and quarantinedForSeconds naming the "
|
||||
+ "quarantine, so 'free: 0, busy' can be told apart from 'free: 0, refusing "
|
||||
+ "for N seconds'.",
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.fasterxml.jackson.databind.node.ObjectNode;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
@@ -14,6 +17,9 @@ import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.nio.file.attribute.PosixFileAttributeView;
|
||||
import java.util.Arrays;
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
@@ -47,6 +53,21 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ClaudeCodeLauncher.class);
|
||||
|
||||
/** JSON codec for the additive workspace-trust seed (fleetd #149) — Jackson's default settings. */
|
||||
private static final ObjectMapper TRUST_JSON = new ObjectMapper();
|
||||
|
||||
/**
|
||||
* Serialises every {@link #seedTrustDialog} read-modify-write for the whole daemon process.
|
||||
* Two claude-code spawns starting at once are normal (fleetd runs several members in parallel
|
||||
* routinely) and both would otherwise read the same {@code .claude.json}, add their own entry
|
||||
* to their own in-memory copy, and write — the second write wins and the first spawn's trust
|
||||
* entry silently disappears. A single process-wide lock is enough because every spawn on this
|
||||
* daemon runs in this one JVM; it does not protect against a second daemon process or the
|
||||
* operator's own Claude Code process writing at the same instant, which {@link #writeAtomically}
|
||||
* covers instead (each writer only ever sees a fully-old or fully-new file, never a torn one).
|
||||
*/
|
||||
private static final Object TRUST_JSON_LOCK = new Object();
|
||||
|
||||
private final SubscriptionGuard guard;
|
||||
|
||||
/**
|
||||
@@ -94,13 +115,26 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env, spawnReadyTimeoutMs, spawnReadyPollMs,
|
||||
fleet, memberCredentials, null, null);
|
||||
}
|
||||
|
||||
/** Production constructor, plus the live config for URI environment exclusions. */
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<Set<String>> hostEnvNames,
|
||||
Supplier<FleetConfig> config) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
fleet, memberCredentials);
|
||||
fleet, memberCredentials, hostEnvNames, config);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -153,10 +187,22 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env, spawnReadyTimeoutMs, nowMillis, sleeper,
|
||||
fleet, memberCredentials, null, null);
|
||||
}
|
||||
|
||||
/** Full testability constructor, plus the live config for URI environment exclusions. */
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper, Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<Set<String>> hostEnvNames,
|
||||
Supplier<FleetConfig> config) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials);
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials, hostEnvNames, config);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
@@ -174,7 +220,7 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<Set<String>> hostEnvNames) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials, hostEnvNames);
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials, hostEnvNames, null);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
@@ -233,11 +279,15 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
workerEnv.remove("ANTHROPIC_AUTH_TOKEN");
|
||||
} else {
|
||||
workerEnv.put("ANTHROPIC_BASE_URL", baseUrl);
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", env.apply(cfg.tokenEnv()));
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", resolveEnv(cfg.tokenEnv()));
|
||||
}
|
||||
putIfPresent(workerEnv, "ANTHROPIC_MODEL", cfg.model());
|
||||
putIfPresent(workerEnv, "CLAUDE_CONFIG_DIR", cfg.configDir());
|
||||
applyGitToken(workerEnv, cfg);
|
||||
// fleetd #149: seed the workspace-trust entry BEFORE this spawn ever reaches herdr — see
|
||||
// seedTrustDialog for why, and isProvisionedWorktree for why this is gated to a worktree
|
||||
// fleetd itself provisioned (never a real checkout, never an un-configured fallback cwd).
|
||||
seedTrustDialog(cfg.configDir(), spec.cwd());
|
||||
|
||||
// CB-547a: Claude Code can MINT its own session id, so fleetd chooses it — a fresh spawn
|
||||
// gets a UUID we pass as --session-id and return from agentSessionId(), so the resume
|
||||
@@ -253,18 +303,22 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
/**
|
||||
* Add the Claude-specific session-identity flags to {@code argv} and return the peer's OWN
|
||||
* session id — the resume handle. A resume request passes the prior id via {@code -r} and
|
||||
* returns that id; a fresh named session mints a new UUID, passes it via {@code --session-id},
|
||||
* and returns the mint. The bridge's logical name rides along as {@code -n} when present. When
|
||||
* <em>no</em> identity is requested (sessionName and resumeSessionId both blank) this adds
|
||||
* nothing and returns {@code null}, keeping the legacy no-identity launch byte-identical.
|
||||
* session id — the resume handle. A resume passes the prior id via {@code -r} and returns
|
||||
* that id; every other spawn mints a new UUID, passes it via {@code --session-id}, and
|
||||
* returns the mint. The bridge's logical name rides along as {@code -n} when present.
|
||||
*
|
||||
* <p>fleetd #214: the mint is unconditional. A plain {@code fleet_spawn} passes neither
|
||||
* sessionName nor resumeSessionId, yet the member must still be resumable, and this id is
|
||||
* the only resume handle a claude-code member has — unlike opencode, nothing resolves it
|
||||
* after the launch. Checked against the real binary (claude 2.1.252): the flag is safe on
|
||||
* every spawn. The binary takes only a valid UUID — it refuses any other value at argument
|
||||
* parsing ("Invalid session ID. Must be a valid UUID.") — so the {@code UUID.randomUUID()}
|
||||
* mint is required, not incidental. The only flag interaction the binary documents is with
|
||||
* {@code -r} (both claim the session id), and the resume branch above never combines the two.
|
||||
*/
|
||||
private static String applySessionIdentity(List<String> argv, String sessionName, String resumeSessionId) {
|
||||
boolean resuming = resumeSessionId != null && !resumeSessionId.isBlank();
|
||||
boolean named = sessionName != null && !sessionName.isBlank();
|
||||
if (!resuming && !named) {
|
||||
return null; // no identity requested — keep the legacy launch byte-identical
|
||||
}
|
||||
if (named) {
|
||||
argv.add("-n");
|
||||
argv.add(sessionName);
|
||||
@@ -295,11 +349,19 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
*
|
||||
* <p>CB-618: Claude Code refuses to start when BOTH {@code --append-system-prompt} and
|
||||
* {@code --append-system-prompt-file} are on the command line ("Cannot use both ... Please use
|
||||
* only one"), so the two charters can never travel on separate flags. When both are present they
|
||||
* are concatenated into the one file, role charter first and reply charter last — last is where
|
||||
* the reply rule must sit, because it is the rule that must survive. When only the reply charter
|
||||
* is present it keeps its proven inline {@code --append-system-prompt} delivery, which is also
|
||||
* the only form that reaches a member with no repo checkout.
|
||||
* only one"), so the two charters can never travel on separate flags. They are concatenated
|
||||
* into the one file, role charter first and reply charter last — last is where the reply rule
|
||||
* must sit, because it is the rule that must survive.
|
||||
*
|
||||
* <p>fleetd #220: a lone reply charter used to ride inline on {@code --append-system-prompt},
|
||||
* which put ~800 bytes of prose on the command line herdr types into the pane. That line is
|
||||
* capped at {@value HerdrPeerLauncher#PANE_COMMAND_BYTE_LIMIT} bytes by the pty itself, and
|
||||
* everything past the cap is dropped with no error from any layer. The charter alone left about
|
||||
* 50 bytes of headroom, so adding one flag ({@code --session-id}, fleetd #214) truncated the
|
||||
* LAST argument instead — {@code --autocompact 250000} arrived as {@code --autocompact 25},
|
||||
* claude rejected it, and every claude-code spawn died as an unexplained readiness timeout.
|
||||
* The charter now always travels as a file, which takes the prose off the command line for
|
||||
* good; {@link HerdrPeerLauncher#checkPaneCommandFits} is the backstop for whatever grows next.
|
||||
*/
|
||||
private List<String> argvWithFleet(FleetConfig.Profile cfg, LaunchSpec spec) {
|
||||
String roleCharter = nonBlank(spec.roleCharter());
|
||||
@@ -316,16 +378,13 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
argv.add("--mcp-config");
|
||||
argv.add(mcpConfigJson(cfg));
|
||||
}
|
||||
// Combine the charters in order role -> reply, dropping any that are absent. When two
|
||||
// or more survive they must ride one --append-system-prompt-file (CB-618 forbids the inline
|
||||
// flag and the file flag together). A lone reply charter keeps its proven inline delivery.
|
||||
// Combine the charters in order role -> reply, dropping any that are absent. They ride one
|
||||
// --append-system-prompt-file (CB-618 forbids the inline flag and the file flag together),
|
||||
// always — fleetd #220: charter prose on the command line overruns the pane's byte cap.
|
||||
List<String> charters = new java.util.ArrayList<>(2);
|
||||
if (roleCharter != null) charters.add(roleCharter);
|
||||
if (replyCharter != null) charters.add(replyCharter);
|
||||
if (charters.size() == 1 && replyCharter != null && roleCharter == null) {
|
||||
argv.add("--append-system-prompt");
|
||||
argv.add(replyCharter);
|
||||
} else if (!charters.isEmpty()) {
|
||||
if (!charters.isEmpty()) {
|
||||
argv.add("--append-system-prompt-file");
|
||||
argv.add(writeCharterFile(String.join("\n\n", charters)).toString());
|
||||
}
|
||||
@@ -378,12 +437,12 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
*/
|
||||
private static void writeIdeOverlay(String cwd, String projectPath) {
|
||||
try {
|
||||
Path dotGit = Path.of(cwd, ".git");
|
||||
if (!Files.isRegularFile(dotGit)) {
|
||||
if (!isProvisionedWorktree(cwd)) {
|
||||
// Not a provisioned worktree (primary's real checkout has a .git directory, or the
|
||||
// cwd is not a repo at all). Never write into it.
|
||||
return;
|
||||
}
|
||||
Path dotGit = Path.of(cwd, ".git");
|
||||
// The overlay FILE lives at the worktree root (claude-code's cwd), but its CONTENT pins
|
||||
// project_path to the module dir the IDE opened (projectPath), not the worktree root.
|
||||
Files.writeString(Path.of(cwd, "CLAUDE.local.md"), PeerLauncher.ideOverlayText(projectPath));
|
||||
@@ -413,6 +472,321 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #149: pre-seed the workspace-trust entry for {@code cwd} in Claude Code's own config
|
||||
* file, BEFORE this launch ever reaches herdr (called from {@link #buildLaunch}, which always
|
||||
* runs before the base class starts the process). Claude Code asks an interactive, un-timed
|
||||
* "Is this a project you created or one you trust?" the first time it starts in a directory it
|
||||
* has not seen before, and every {@code worktree: true} spawn lands in a brand-new directory —
|
||||
* so without this seed the member sits on that dialog forever, never mounts the bridge MCP, and
|
||||
* never calls {@code fleet_reply}. herdr reports it as healthy the whole time ({@code
|
||||
* agent_status: blocked}, {@code interactive_ready: true}), so nothing else catches it. Measured
|
||||
* live on fleet01 2026-08-23: an unseeded fresh cwd sat on the dialog indefinitely; a seeded one
|
||||
* reached {@code idle} clean.
|
||||
*
|
||||
* <p>This is not a new grant — the operator already trusted this repo by configuring the
|
||||
* profile against it, and a worktree is a checkout of that same repo.
|
||||
*
|
||||
* <p>The key Claude Code reads is per-project, in {@code .claude.json}: {@code
|
||||
* projects.<cwd>.hasTrustDialogAccepted}. The file lives at {@code <configDir>/.claude.json}
|
||||
* when the profile sets {@code CLAUDE_CONFIG_DIR} (mirrors this launcher's own env var above),
|
||||
* else the default {@code ~/.claude.json} — the same file Claude Code itself would read either
|
||||
* way, so this seeds exactly what the spawned peer is about to open.
|
||||
*
|
||||
* <p><b>Additive, not a rewrite.</b> {@code .claude.json} is large (tens of KB, dozens of
|
||||
* projects) and Claude Code itself rewrites it while running, so this reads the file as a JSON
|
||||
* tree (missing or unreadable → treated as an empty object) and changes only
|
||||
* {@code projects.<cwd>.hasTrustDialogAccepted} (that key alone — see fleetd #247) —
|
||||
* every other top-level key and every other project entry is written back untouched. Only the
|
||||
* one project entry for {@code cwd} is replaced/created; an existing entry for a DIFFERENT cwd
|
||||
* (or the operator's own project history) is never touched.
|
||||
*
|
||||
* <p>Best-effort, like {@link #writeIdeOverlay}: a failure here (unwritable configDir, a
|
||||
* corrupt existing file, …) must never fail the spawn — it is logged at debug and swallowed. A
|
||||
* peer that starts without the seed still starts; it just may hit the dialog fleetd #149
|
||||
* describes.
|
||||
*
|
||||
* <p><b>Gated to a provisioned worktree</b> ({@link HerdrPeerLauncher#isProvisionedWorktree}) — see that
|
||||
* method's javadoc for the incident that made this gate mandatory, not optional: this must
|
||||
* never run against a real checkout or an un-configured fallback cwd, only the exact
|
||||
* always-fresh-directory population fleetd #149 describes.
|
||||
*
|
||||
* <p><b>Atomic and lock-protected.</b> {@code .claude.json} is a live file — Claude Code itself
|
||||
* rewrites it while running, and this daemon routinely spawns several members at once, each
|
||||
* calling this method for its own cwd. Every write goes through {@link #writeAtomically} (a
|
||||
* sibling-temp-file + {@code ATOMIC_MOVE}, never a truncate-in-place) so a crash mid-write or a
|
||||
* concurrent reader never observes a half-written file, and through {@link #TRUST_JSON_LOCK} so
|
||||
* two concurrent spawns' entries both survive instead of the second write silently discarding
|
||||
* the first. Both exist because of a real incident: see {@link HerdrPeerLauncher#isProvisionedWorktree}'s javadoc
|
||||
* and {@link #writeAtomically}'s javadoc.
|
||||
*
|
||||
* <p><b>Compare-and-swap against a writer the lock cannot reach (fleetd #247).</b>
|
||||
* {@code TRUST_JSON_LOCK} only serialises calls this launcher itself makes inside this one JVM.
|
||||
* It does nothing about a writer outside it — and on a host where the profile's
|
||||
* {@code configDir} is the operator's own {@code CLAUDE_CONFIG_DIR}, the file this method writes
|
||||
* IS the operator's own live Claude Code session's config file, being read and written by that
|
||||
* session while it runs. Measured 2026-09-03: its mtime moved minutes after a spawn while that
|
||||
* session was active. A plain read-modify-write there is a routine lost update, not a rare one:
|
||||
* fleetd reads v1, the operator's session reads v1 and writes v2 with their own change, fleetd's
|
||||
* {@code ATOMIC_MOVE} then lands v3 built from v1 — atomic, but v2's change is gone. So before
|
||||
* the move this method re-reads {@code target}'s exact bytes and compares them with the bytes it
|
||||
* built its update from; a mismatch means someone else wrote in between, and it discards its
|
||||
* work and rebuilds from the fresh bytes, up to {@link #MAX_TRUST_JSON_CAS_ATTEMPTS} times.
|
||||
* <b>Exhausting the retries writes nothing</b> — see the WARN at the end of the loop for why
|
||||
* that, not a last write-anyway, is the safe failure: the member shows the trust dialog and
|
||||
* fails to reach an injectable state, which is visible, logged and recoverable; overwriting the
|
||||
* operator's live config with a stale copy is neither. This narrows the lost-update window, it
|
||||
* does not close it — a write landing between the final re-read and the {@code ATOMIC_MOVE}
|
||||
* itself is still lost, because there is no OS-level compare-and-swap on a plain file, only this
|
||||
* cooperative narrowing of the gap.
|
||||
*
|
||||
* <p><b>fleetd #285: refuses under {@code memberHerdrSocket} rather than writing somewhere the
|
||||
* member cannot read.</b> Under {@code memberHerdrSocket:} the member pane runs as a
|
||||
* <em>different OS user with its own {@code $HOME}</em> — the same reason {@link
|
||||
* #writeCharterFile} routes the role/reply charter under {@code worktreeRoot} instead of
|
||||
* {@code java.io.tmpdir} and refuses the spawn when it cannot. This method has no equivalent
|
||||
* relocation available: unlike the charter (fleetd's own content, free to place anywhere and
|
||||
* hand to the peer via an argv flag), {@code .claude.json} is a file Claude Code looks up for
|
||||
* ITSELF at a fixed location — {@code CLAUDE_CONFIG_DIR/.claude.json}, or else the member OS
|
||||
* user's own {@code ~/.claude.json}, a path fleetd has no channel to learn. So when {@code
|
||||
* configDir} is unset, there is no member-readable target to seed at all — writing the
|
||||
* unqualified default would land in <em>fleetd's own</em> {@code ~/.claude.json} instead, the
|
||||
* exact defect this fix closes, not a workable fallback. And even with {@code configDir} set,
|
||||
* the file this method itself just wrote is {@code 0600} (owner-only — see {@link
|
||||
* #copyPosixPermissionsIfPresent}), unreadable by a different-uid member unless shared with
|
||||
* {@code worktreeGroup}, the same group {@link EnvAllowListScrub#shareWithGroup} already uses
|
||||
* for the ZDOTDIR scrub (fleetd #213) and the charter file (fleetd #219/#222). So under {@code
|
||||
* memberHerdrSocket} this method requires BOTH {@code configDir} and {@code worktreeGroup}
|
||||
* before it ever touches a file, and refuses the spawn — naming exactly which one is missing —
|
||||
* rather than silently corrupt fleetd's own home or hand the member an unreadable path. This
|
||||
* mirrors {@link #writeCharterFile}'s "refuse, don't degrade" decision: a member spawned without
|
||||
* a readable trust seed is not degraded, it sits on the interactive dialog forever and never
|
||||
* calls {@code fleet_reply} — exactly the failure fleetd #149 exists to prevent, so trading it
|
||||
* for "spawn something" is not worth it. When {@code configDir} and {@code worktreeGroup} are
|
||||
* both present, the write proceeds exactly as below and the resulting file is additionally
|
||||
* chgrp'd/chmod'd group-readable ({@code rw-r-----}) via {@link
|
||||
* EnvAllowListScrub#shareFileWithGroup(Path, String)} so the member's OS user can actually
|
||||
* open it — the
|
||||
* directory itself (unlike {@code worktreeRoot} or the charter's per-spawn directory) is not
|
||||
* fleetd-managed, so its own traversal permissions remain the operator's setup, same as they
|
||||
* already must be for the member to read anything else fleetd points {@code CLAUDE_CONFIG_DIR}
|
||||
* at. With {@code memberHerdrSocket} ABSENT (today's only live mode) every branch below is
|
||||
* byte-identical to before this fix.
|
||||
*
|
||||
* @param configDir the profile's {@code CLAUDE_CONFIG_DIR} ({@code cfg.configDir()}), or
|
||||
* {@code null}/blank to target the default {@code ~/.claude.json} — refused
|
||||
* outright when {@code memberHerdrSocket} is configured, see above
|
||||
* @param cwd the spawn's resolved working directory — the exact key Claude Code will look
|
||||
* up for itself once it starts there
|
||||
* @throws IllegalStateException when {@code memberHerdrSocket} is configured but {@code
|
||||
* configDir} and/or {@code worktreeGroup} is not — the same
|
||||
* refusal shape as {@link #writeCharterFile}
|
||||
*/
|
||||
private void seedTrustDialog(String configDir, String cwd) {
|
||||
if (!isProvisionedWorktree(cwd)) {
|
||||
return;
|
||||
}
|
||||
boolean unsetConfigDir = configDir == null || configDir.isBlank();
|
||||
boolean memberHerdrSocket = memberHerdrSocketConfigured();
|
||||
String group = memberHerdrSocket ? memberGroup() : null;
|
||||
if (memberHerdrSocket && (unsetConfigDir || group == null)) {
|
||||
throw new IllegalStateException("memberHerdrSocket is configured, so the workspace-trust "
|
||||
+ "seed (.claude.json, which gates Claude Code's interactive trust dialog) must be "
|
||||
+ "placed where the member's OS user can read it — configDir, shared via "
|
||||
+ "worktreeGroup — but " + (unsetConfigDir ? "configDir" : "worktreeGroup")
|
||||
+ " is not configured. Refusing to spawn rather than write fleetd's own default "
|
||||
+ "'~/.claude.json' or hand the member a config file it cannot read: that member "
|
||||
+ "would sit on the interactive trust dialog forever and never reach an "
|
||||
+ "injectable state. Configure configDir on this profile and worktreeGroup on the "
|
||||
+ "fleet to enable claude-code member spawns under memberHerdrSocket.");
|
||||
}
|
||||
Path target = unsetConfigDir
|
||||
? Path.of(System.getProperty("user.home"), ".claude.json")
|
||||
: Path.of(configDir, ".claude.json");
|
||||
if (unsetConfigDir) {
|
||||
// fleetd #247: configDir unset is the ONLY path that targets ~/.claude.json — the
|
||||
// operator's own home file, not a per-profile one — and it is the default, so a
|
||||
// profile that simply forgot to set configDir gets no signal at all short of the
|
||||
// operator noticing their own file changing. Say so loudly, every time it is about to
|
||||
// happen, rather than only once ever: each occurrence is a live write to a real
|
||||
// person's home config and deserves its own log line. (Reached only when
|
||||
// memberHerdrSocket is absent — the block above already refused otherwise.)
|
||||
log.warn("seedTrustDialog: profile has no configDir set, so the workspace-trust seed "
|
||||
+ "for cwd '{}' is about to write the operator's own default '{}' — set "
|
||||
+ "configDir on this profile to target a per-member config file instead",
|
||||
cwd, target);
|
||||
}
|
||||
synchronized (TRUST_JSON_LOCK) {
|
||||
boolean written = false;
|
||||
try {
|
||||
if (target.getParent() != null) {
|
||||
Files.createDirectories(target.getParent());
|
||||
}
|
||||
for (int attempt = 1; attempt <= MAX_TRUST_JSON_CAS_ATTEMPTS && !written; attempt++) {
|
||||
byte[] before = Files.isRegularFile(target) ? Files.readAllBytes(target) : null;
|
||||
ObjectNode root = parseTrustJsonOrEmpty(before);
|
||||
JsonNode projectsNode = root.get("projects");
|
||||
ObjectNode projects = projectsNode instanceof ObjectNode projectsObject
|
||||
? projectsObject : TRUST_JSON.createObjectNode();
|
||||
if (!(projectsNode instanceof ObjectNode)) {
|
||||
root.set("projects", projects);
|
||||
}
|
||||
JsonNode projectNode = projects.get(cwd);
|
||||
ObjectNode project = projectNode instanceof ObjectNode projectObject
|
||||
? projectObject : TRUST_JSON.createObjectNode();
|
||||
if (!(projectNode instanceof ObjectNode)) {
|
||||
projects.set(cwd, project);
|
||||
}
|
||||
// fleetd #247: ONLY hasTrustDialogAccepted. We used to write
|
||||
// hasCompletedProjectOnboarding beside it; do not put it back. Measured on
|
||||
// 2026-09-03, minutes after a live spawn seeded this file: 28 of 28 project
|
||||
// entries carried hasTrustDialogAccepted and 0 of 28 carried the onboarding key
|
||||
// — including the 27 entries Claude Code wrote for itself. Claude Code
|
||||
// normalises the whole file when it saves and drops that key every time, so
|
||||
// writing it achieved nothing except making the next reader think it mattered.
|
||||
// The member reached idle with the trust flag alone, which is the only outcome
|
||||
// this seed exists for. If a future Claude Code needs the second flag the
|
||||
// symptom returns as the trust dialog fleetd #149 describes — re-measure then,
|
||||
// do not restore it on a guess.
|
||||
project.put("hasTrustDialogAccepted", true);
|
||||
String newContent = TRUST_JSON.writerWithDefaultPrettyPrinter().writeValueAsString(root);
|
||||
|
||||
trustJsonCasTestHook.run();
|
||||
|
||||
// fleetd #247 CAS: re-read immediately before the move and compare with what
|
||||
// this attempt built its update from. A mismatch means another writer (most
|
||||
// plausibly the operator's own live Claude Code — see this method's javadoc)
|
||||
// landed a change in between; discard this attempt's work and rebuild from the
|
||||
// fresh bytes rather than blindly overwriting it.
|
||||
byte[] atMove = Files.isRegularFile(target) ? Files.readAllBytes(target) : null;
|
||||
if (!Arrays.equals(before, atMove)) {
|
||||
continue;
|
||||
}
|
||||
writeAtomically(target, newContent);
|
||||
written = true;
|
||||
}
|
||||
if (!written) {
|
||||
// fleetd #247: deliberately do NOT write here. A member that starts without the
|
||||
// seed still starts — it may hit the trust dialog fleetd #149 describes and fail
|
||||
// to reach an injectable state, but that failure is visible (herdr reports it,
|
||||
// the spawn-readiness gate times out) and recoverable (retry the spawn). Writing
|
||||
// our stale copy over whatever the other writer left would be silent and, if that
|
||||
// other writer is the operator's own live session, could destroy real
|
||||
// configuration — fail toward the recoverable outcome, not the silent one.
|
||||
log.warn("seedTrustDialog: gave up seeding workspace-trust for cwd '{}' into '{}' "
|
||||
+ "after {} attempts — another writer (most plausibly the operator's own "
|
||||
+ "live Claude Code sharing this file) kept changing it faster than we "
|
||||
+ "could re-read it, so nothing was written; the member may show the "
|
||||
+ "trust dialog instead", cwd, target, MAX_TRUST_JSON_CAS_ATTEMPTS);
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.debug("cannot seed workspace-trust entry for cwd '{}' into '{}'", cwd, target, e);
|
||||
return;
|
||||
}
|
||||
// fleetd #285: the write above lands as fleetd's own OS user; under memberHerdrSocket
|
||||
// that is NOT the member's OS user, so without this the member still cannot read the
|
||||
// file it exists to seed — a silent readiness timeout with the write looking "done".
|
||||
// Deliberately OUTSIDE the swallow-all catch above: a group that fails to resolve here
|
||||
// means the seed is unreadable despite a successful write, which must fail as loudly as
|
||||
// writeCharterFile's own EnvAllowListScrub.shareWithGroup call already does.
|
||||
if (written && memberHerdrSocket) {
|
||||
EnvAllowListScrub.shareFileWithGroup(target, group);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/** Bound on {@link #seedTrustDialog}'s fleetd #247 compare-and-swap retry loop. */
|
||||
private static final int MAX_TRUST_JSON_CAS_ATTEMPTS = 5;
|
||||
|
||||
/**
|
||||
* Test-only seam for fleetd #247: invoked once per CAS attempt inside {@link #seedTrustDialog}'s
|
||||
* retry loop, after that attempt has read the target's bytes and built its replacement content,
|
||||
* but immediately before the final re-read/compare that decides whether to write. A no-op in
|
||||
* production. Package-visible (not {@code private}) so {@code ClaudeCodeLauncherTest} can install
|
||||
* a hook here that writes to the target file, deterministically simulating a writer racing
|
||||
* fleetd's own read-modify-write at the exact instant the CAS is meant to catch — the same race
|
||||
* an external process (most plausibly the operator's own live Claude Code) creates, without
|
||||
* depending on real thread scheduling to land the interleaving. A test that sets this MUST
|
||||
* restore it to the no-op default in a {@code finally} block — it is shared, static state.
|
||||
*/
|
||||
static Runnable trustJsonCasTestHook = () -> {};
|
||||
|
||||
/**
|
||||
* Parse {@code bytes} as a {@code .claude.json} tree, or hand back a fresh empty object when
|
||||
* {@code bytes} is {@code null} (no file yet) or does not parse to a JSON object — the same
|
||||
* missing-or-unreadable-is-empty fallback {@link #seedTrustDialog} always used, factored out so
|
||||
* the fleetd #247 CAS loop can call it once per attempt.
|
||||
*/
|
||||
private static ObjectNode parseTrustJsonOrEmpty(byte[] bytes) throws IOException {
|
||||
if (bytes == null) {
|
||||
return TRUST_JSON.createObjectNode();
|
||||
}
|
||||
JsonNode existing = TRUST_JSON.readTree(bytes);
|
||||
return existing instanceof ObjectNode existingObject ? existingObject : TRUST_JSON.createObjectNode();
|
||||
}
|
||||
|
||||
/**
|
||||
* Write {@code content} to {@code target} atomically: serialise to a sibling temp file in the
|
||||
* <strong>same directory</strong> as {@code target} (an atomic move is only guaranteed within
|
||||
* one filesystem — a different directory could mean a different filesystem), then
|
||||
* {@link StandardCopyOption#ATOMIC_MOVE} it into place. A reader — Claude Code itself, or
|
||||
* another {@code seedTrustDialog} call — only ever observes the fully-old file or the
|
||||
* fully-new one, never a truncated or half-written one.
|
||||
*
|
||||
* <p><b>fleetd #149 incident.</b> The original implementation used
|
||||
* {@code Files.writeString(target, content)} directly, which truncates {@code target} in place
|
||||
* before writing the replacement bytes. Combined with an ungated {@code cwd} (see
|
||||
* {@link HerdrPeerLauncher#isProvisionedWorktree}'s javadoc), a mutation-testing run hit that truncation window
|
||||
* against the operator's real {@code ~/.claude.json} and left it at 178 bytes. The gate closes
|
||||
* <em>which file</em> this can ever target; this closes <em>how</em> the target is written, so
|
||||
* that even a legitimate write against a real, live, concurrently-read {@code .claude.json}
|
||||
* cannot leave it observably empty or partial.
|
||||
*
|
||||
* <p>Preserves {@code target}'s existing POSIX permissions (Claude Code ships {@code
|
||||
* .claude.json} as {@code 0600}) when the filesystem reports them; a freshly created temp file
|
||||
* already defaults to owner-only permissions on a POSIX filesystem, so a first-ever write (no
|
||||
* existing {@code target}) is no less private without this. On a non-POSIX filesystem (e.g.
|
||||
* Windows) the permission copy is a silent no-op rather than a failure.
|
||||
*
|
||||
* <p>Package-visible (not {@code private}) so a test can drive it directly with a concurrent
|
||||
* reader thread and prove the torn-file property this method exists for — the LOCK in
|
||||
* {@link #seedTrustDialog} already fully serialises every call this launcher itself makes, so a
|
||||
* test that only ever goes through {@code seedTrustDialog}/{@code spawn()} could never observe
|
||||
* a torn file regardless of whether this method is atomic; it would be proving the lock, not
|
||||
* this method. Atomicity's actual job is protecting against a writer the lock cannot reach at
|
||||
* all — a second daemon process, or the operator's own live Claude Code — so the test for it
|
||||
* has to reach this method on its own.
|
||||
*/
|
||||
static void writeAtomically(Path target, String content) throws IOException {
|
||||
Path parent = target.getParent();
|
||||
Path tmp = Files.createTempFile(parent, target.getFileName() + ".", ".tmp");
|
||||
try {
|
||||
Files.writeString(tmp, content);
|
||||
copyPosixPermissionsIfPresent(target, tmp);
|
||||
Files.move(tmp, target, StandardCopyOption.ATOMIC_MOVE, StandardCopyOption.REPLACE_EXISTING);
|
||||
} catch (IOException e) {
|
||||
Files.deleteIfExists(tmp);
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
/** Copy {@code target}'s POSIX permissions onto {@code tmp}, or no-op where either is unsupported. */
|
||||
private static void copyPosixPermissionsIfPresent(Path target, Path tmp) {
|
||||
try {
|
||||
if (!Files.isRegularFile(target)) {
|
||||
return; // nothing to inherit from — first-ever write, temp file's own default stands
|
||||
}
|
||||
PosixFileAttributeView view = Files.getFileAttributeView(target, PosixFileAttributeView.class);
|
||||
if (view == null) {
|
||||
return; // non-POSIX filesystem — nothing this JVM can read/set here
|
||||
}
|
||||
Files.setPosixFilePermissions(tmp, Files.getPosixFilePermissions(target));
|
||||
} catch (IOException e) {
|
||||
log.debug("cannot preserve permissions of '{}' onto its replacement", target, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** {@code s}, or {@code null} when {@code s} is null/blank — the charter-presence test used above. */
|
||||
private static String nonBlank(String s) {
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
@@ -425,12 +799,83 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
* {@link OpenCodeLauncher#writeConfig} already uses for its charter file, since the process that
|
||||
* reads this file (the spawned peer) outlives this JVM call and there is no spawn-scoped teardown
|
||||
* hook to delete it synchronously.
|
||||
*
|
||||
* <p><b>fleetd #222.</b> {@code Files.createTempFile(prefix, suffix)} with no directory argument
|
||||
* resolves against {@code java.io.tmpdir} — on macOS the per-user {@code $TMPDIR} under
|
||||
* {@code /var/folders/...}, mode {@code 0700}, both resolved against FLEETD's own OS user. Under
|
||||
* {@code memberHerdrSocket:} the member pane runs as a DIFFERENT OS user, so that user cannot even
|
||||
* traverse the directory, let alone read the file — and since fleetd #220 the charter file is the
|
||||
* ONLY delivery path for {@code --append-system-prompt-file}, always, not merely the fallback it
|
||||
* used to be. A member handed a path it cannot read is not degraded, it is broken: see this
|
||||
* method's refusal branch below.
|
||||
*
|
||||
* <p><b>Measured severity (fleetd #222 real-binary check, claude 2.1.258):</b> an unreadable
|
||||
* {@code --append-system-prompt-file} is the LOUD failure, not the silent one. {@code claude}
|
||||
* checks the file before touching auth or the network — invoked with a bogus API key against a
|
||||
* {@code chmod 000} file, it printed {@code Error reading append system prompt file: EACCES:
|
||||
* permission denied, open '<path>'} and exited 1 immediately (a nonexistent path gets {@code
|
||||
* Error: Append system prompt file not found: <path>}, same exit code). So the pre-fix bug did
|
||||
* NOT leave a charter-less member silently occupying a pane and never calling {@code
|
||||
* fleet_reply} — it made the herdr pane exit immediately, which the CB-306 spawn-readiness gate
|
||||
* (this launcher's {@code spawnReadyTimeoutMs} poll) would have surfaced as "did not reach
|
||||
* injectable state", the same unexplained-timeout shape fleetd #220 already describes. Still a
|
||||
* real defect (every claude-code member under {@code memberHerdrSocket} would have failed to
|
||||
* spawn), but not the worse, undetectable failure mode.
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@code memberHerdrSocket} ABSENT (today's only live mode): byte-identical to before this
|
||||
* fix — {@code Files.createTempFile("fleetd-role-charter-", ".md")} with no directory
|
||||
* argument, i.e. still resolved against {@code java.io.tmpdir}.</li>
|
||||
* <li>{@code memberHerdrSocket} PRESENT: a fresh per-spawn directory is created under {@code
|
||||
* worktreeRoot} (never {@code java.io.tmpdir}) holding just the charter file, then shared
|
||||
* read-only with {@code worktreeGroup} via {@link EnvAllowListScrub#shareWithGroup} — the
|
||||
* SAME mechanism fleetd #213 built for the ZDOTDIR scrub and fleetd #219 reused for {@link
|
||||
* OpenCodeLauncher#writeConfig}'s {@code opencode.json} directory, reused here rather than
|
||||
* duplicated a third time. A per-spawn subdirectory (not {@code worktreeRoot} itself) is the
|
||||
* unit {@code shareWithGroup} chmods, so this never touches permissions on anything else
|
||||
* under {@code worktreeRoot}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><b>Follows fleetd #219's REFUSAL decision, not #213's degrade decision.</b> The ZDOTDIR
|
||||
* scrub is a credential CONTROL — a degraded control (the CB-596 sentinel overlay) still has
|
||||
* value, so #213 falls back rather than refusing. A charter is NOT a control, it is the member's
|
||||
* TURN CONTRACT (the rule that ends every turn with {@code fleet_reply}). A member spawned with no
|
||||
* charter is not degraded, it is broken: either claude-code exits on the unreadable
|
||||
* {@code --append-system-prompt-file} path and the spawn dies at the readiness gate (loud), or it
|
||||
* starts anyway with no charter and never calls {@code fleet_reply} — the sender silently gets
|
||||
* nothing (silent). Neither outcome is worth trading for "spawn something." So a missing {@code
|
||||
* worktreeRoot}/{@code worktreeGroup} under {@code memberHerdrSocket} refuses the spawn here,
|
||||
* naming the missing key, exactly like {@link OpenCodeLauncher#configParentDir()}.
|
||||
*
|
||||
* @throws IllegalStateException when {@code memberHerdrSocket} is configured but {@code
|
||||
* worktreeRoot} and/or {@code worktreeGroup} is not
|
||||
*/
|
||||
private static Path writeCharterFile(String charterText) {
|
||||
private Path writeCharterFile(String charterText) {
|
||||
try {
|
||||
Path file = Files.createTempFile("fleetd-role-charter-", ".md");
|
||||
if (!memberHerdrSocketConfigured()) {
|
||||
Path file = Files.createTempFile("fleetd-role-charter-", ".md");
|
||||
Files.writeString(file, charterText);
|
||||
file.toFile().deleteOnExit();
|
||||
return file;
|
||||
}
|
||||
Path parentDir = memberScrubParentDir();
|
||||
String group = memberGroup();
|
||||
if (parentDir == null || group == null) {
|
||||
throw new IllegalStateException("memberHerdrSocket is configured, so the role/reply "
|
||||
+ "charter file (mounted via --append-system-prompt-file) must be placed where "
|
||||
+ "the member's OS user can read it — worktreeRoot, shared via worktreeGroup — "
|
||||
+ "but " + (parentDir == null ? "worktreeRoot" : "worktreeGroup") + " is not "
|
||||
+ "configured. Refusing to spawn rather than hand the member a charter path it "
|
||||
+ "cannot read: that member's turn contract (the fleet_reply rule) would never "
|
||||
+ "reach it. Configure both worktreeRoot and worktreeGroup to enable claude-code "
|
||||
+ "member spawns under memberHerdrSocket.");
|
||||
}
|
||||
Path dir = Files.createTempDirectory(parentDir, "fleetd-role-charter-");
|
||||
dir.toFile().deleteOnExit();
|
||||
Path file = dir.resolve("charter.md");
|
||||
Files.writeString(file, charterText);
|
||||
file.toFile().deleteOnExit();
|
||||
EnvAllowListScrub.shareWithGroup(dir, group);
|
||||
return file;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot write role charter temp file", e);
|
||||
|
||||
@@ -3,12 +3,14 @@ package dev.ltms.fleet.member;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementCandidate;
|
||||
import dev.ltms.fleet.placement.PlacementContext;
|
||||
@@ -46,9 +48,13 @@ import java.util.stream.Collectors;
|
||||
* the single adapter that declares it. Profiles partition cleanly across adapters: the
|
||||
* constructor rejects a name claimed by two.</li>
|
||||
* <li><strong>By pane id</strong> — {@link #stop} routes to the adapter that spawned that pane
|
||||
* (recorded at spawn time). A pane the composite never spawned can use the fallback route
|
||||
* in a one-daemon fleet. With more than one herdr daemon, its owner is unknown, so stop refuses
|
||||
* the ambiguous id rather than closing a pane on an arbitrary herdr daemon.</li>
|
||||
* (recorded at spawn time). A pane the composite never spawned, or one whose record was lost
|
||||
* to a daemon restart (CB-185 blocker 1 — {@link #spawnedBy} is in-memory only), can use the
|
||||
* fallback route in a one-daemon fleet. With more than one herdr daemon, {@link #probeOwner}
|
||||
* asks each configured daemon which one actually knows the pane: exactly one match routes
|
||||
* (and caches); no match is treated as already-gone; more than one match is a genuine
|
||||
* ambiguity (pane ids are per-daemon counters, so two daemons really can both hold, say,
|
||||
* {@code w1:p1}) and stop refuses rather than closing a pane on an arbitrary herdr daemon.</li>
|
||||
* <li><strong>Fleet-wide</strong> — {@link #reapOrphanWorkers} and {@link #capabilities} fan out
|
||||
* and combine. {@link #list} is deduplicated by (owning daemon, pane id): delegates that share
|
||||
* one herdr connection report the same global agent set, but two daemons can each hold a pane
|
||||
@@ -87,6 +93,18 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
/** CB-578 stage B: credential cooldown, checked before an explicit spawn and filtered into placement. */
|
||||
private final BackendQuarantine quarantine;
|
||||
|
||||
/**
|
||||
* fleetd #201 Unit 5: credential cool-off after repeated backend errors, checked before an
|
||||
* explicit spawn and filtered into placement — a SEPARATE, shorter-lived source from
|
||||
* {@link #quarantine}. A never-{@code record}-called instance is naturally inert (its
|
||||
* {@code remainingCoolOffSeconds} always returns empty), so back-compat constructors that predate
|
||||
* this feature share one fixed instance rather than needing a {@code none()} sentinel.
|
||||
*/
|
||||
private final BackendOutagePolicy outagePolicy;
|
||||
|
||||
/** The shared inert instance back-compat constructors wire in — never {@code record}-called. */
|
||||
private static final BackendOutagePolicy NO_OUTAGE_POLICY = new BackendOutagePolicy(System::nanoTime);
|
||||
|
||||
/**
|
||||
* CB-557: the role pools an unqualified spawn draws its candidates from. A supplier that yields
|
||||
* {@code null}, and an empty pool for a role, both fall back to every configured profile — the
|
||||
@@ -125,7 +143,8 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
Map<String, FleetConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, null, BackendQuarantine.none());
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, null,
|
||||
BackendQuarantine.none(), NO_OUTAGE_POLICY);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -143,13 +162,14 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet, BackendQuarantine.none());
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet,
|
||||
BackendQuarantine.none(), NO_OUTAGE_POLICY);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with role pools and quarantine (CB-578 stage B). The full-featured
|
||||
* non-reloading form; {@link #CompositePeerLauncher(List, String, Supplier, Function, BackendQuarantine)}
|
||||
* is what {@code Fleetd.main} actually wires up.
|
||||
* Production constructor with role pools and quarantine (CB-578 stage B), no cool-off (fleetd
|
||||
* #201 Unit 5). Kept for callers that predate the cool-off feature; use the 8-arg overload below
|
||||
* to wire a real {@link BackendOutagePolicy}.
|
||||
*
|
||||
* @param quarantine required — pass {@link BackendQuarantine#none()} for a caller that does not
|
||||
* want the feature, never a defaulting overload (CB-578 stage B's own rule).
|
||||
@@ -161,17 +181,42 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet,
|
||||
BackendQuarantine quarantine) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet, quarantine,
|
||||
NO_OUTAGE_POLICY);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with role pools, quarantine (CB-578 stage B), and cool-off (fleetd
|
||||
* #201 Unit 5). The full-featured non-reloading form;
|
||||
* {@link #CompositePeerLauncher(List, String, Supplier, Function, BackendQuarantine, BackendOutagePolicy)}
|
||||
* is what {@code Fleetd.main} actually wires up.
|
||||
*
|
||||
* @param quarantine required — pass {@link BackendQuarantine#none()} for a caller that does not
|
||||
* want the feature, never a defaulting overload (CB-578 stage B's own rule).
|
||||
* @param outagePolicy required — pass a fresh, never-{@code record}-called {@link
|
||||
* BackendOutagePolicy} for a caller that does not want the feature; same
|
||||
* "explicit opt-out, never a silent default" rule as {@code quarantine}.
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, FleetConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet,
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy) {
|
||||
// LinkedHashMap, not Map.copyOf: candidates() promises definition order and the weighted
|
||||
// policy breaks exact-weight ties on it, so a salted iteration order would make placement
|
||||
// differ from one JVM run to the next.
|
||||
this(delegates, defaultProfile,
|
||||
constant(Collections.unmodifiableMap(new LinkedHashMap<>(profileConfigs))),
|
||||
constant(placementPolicy), liveCount, constant(fleet), quarantine);
|
||||
constant(placementPolicy), liveCount, constant(fleet), quarantine, outagePolicy);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor that re-reads its placement inputs per spawn (CB-559), so a config
|
||||
* reload retargets the next member without a restart.
|
||||
* reload retargets the next member without a restart. No cool-off (fleetd #201 Unit 5); use the
|
||||
* 6-arg overload below to wire a real {@link BackendOutagePolicy}.
|
||||
*
|
||||
* @param config the live configuration — read at every spawn, never captured
|
||||
* @param quarantine required — CB-578 stage B; pass {@link BackendQuarantine#none()} to opt out
|
||||
@@ -181,12 +226,31 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
Supplier<FleetConfig> config,
|
||||
Function<String, Integer> liveCount,
|
||||
BackendQuarantine quarantine) {
|
||||
this(delegates, defaultProfile, config, liveCount, quarantine, NO_OUTAGE_POLICY);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor that re-reads its placement inputs per spawn (CB-559), with cool-off
|
||||
* (fleetd #201 Unit 5). This is what {@code Fleetd.main} actually wires up.
|
||||
*
|
||||
* @param config the live configuration — read at every spawn, never captured
|
||||
* @param quarantine required — CB-578 stage B; pass {@link BackendQuarantine#none()} to opt out
|
||||
* @param outagePolicy required — fleetd #201 Unit 5; pass a fresh, never-{@code record}-called
|
||||
* {@link BackendOutagePolicy} to opt out
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Supplier<FleetConfig> config,
|
||||
Function<String, Integer> liveCount,
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy) {
|
||||
this(delegates, defaultProfile,
|
||||
() -> config.get().profiles(),
|
||||
() -> PlacementPolicies.fromName(config.get().placement()),
|
||||
liveCount,
|
||||
() -> config.get().fleet(),
|
||||
quarantine);
|
||||
quarantine,
|
||||
outagePolicy);
|
||||
}
|
||||
|
||||
/** The all-suppliers form every other constructor funnels into. */
|
||||
@@ -196,9 +260,11 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
Supplier<PlacementPolicy> placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
BackendQuarantine quarantine) {
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy) {
|
||||
this.fleet = fleet;
|
||||
this.quarantine = Objects.requireNonNull(quarantine, "quarantine");
|
||||
this.outagePolicy = Objects.requireNonNull(outagePolicy, "outagePolicy");
|
||||
if (delegates.isEmpty()) {
|
||||
throw new IllegalArgumentException("at least one peer adapter must be configured");
|
||||
}
|
||||
@@ -261,7 +327,10 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
// the charter makes explicit-profile spawns the normal path — so skipping the check
|
||||
// here would leave the cap dead config in real operation.
|
||||
HerdrPeerLauncher d = route(requestedProfile);
|
||||
// Checked in this order so exhaustion quarantine wins when both are active: quarantine
|
||||
// throws first and short-circuits before the cool-off check ever runs (fleetd #201 Unit 5).
|
||||
enforceNotQuarantined(requestedProfile);
|
||||
enforceNotCoolingOff(requestedProfile);
|
||||
enforceMaxLoad(requestedProfile);
|
||||
PeerHandle handle = d.spawn(req);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
@@ -278,7 +347,10 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
// CB-578 stage B: computed once up front — a quarantine's expiry cannot pass within one spawn
|
||||
// call, so re-deriving it per retry would only cost work, never change the answer.
|
||||
Set<String> quarantined = quarantinedProfiles(candidates);
|
||||
PlacementContext ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable, quarantined);
|
||||
// fleetd #201 Unit 5: a distinct set from quarantined — see PlacementContext.coolingOff.
|
||||
Set<String> coolingOff = coolingOffProfiles(candidates);
|
||||
PlacementContext ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable,
|
||||
quarantined, coolingOff);
|
||||
|
||||
int maxAttempts = candidates.isEmpty() ? 1 : candidates.size();
|
||||
for (int attempt = 0; attempt < maxAttempts; attempt++) {
|
||||
@@ -308,7 +380,8 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
chosen.profile(), e.getMessage());
|
||||
unreachable.add(chosen.profile());
|
||||
// Update the context for the next selection so the policy excludes this profile.
|
||||
ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable, quarantined);
|
||||
ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable,
|
||||
quarantined, coolingOff);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -374,6 +447,31 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse an explicit-profile spawn whose credential is cooling off after repeated backend errors
|
||||
* (fleetd #201 Unit 5 — {@link BackendOutagePolicy}): a SEPARATE, shorter-lived source from
|
||||
* {@link #enforceNotQuarantined}'s exhaustion quarantine. Checked after quarantine so exhaustion
|
||||
* wins when both are active — see the call site in {@link #spawn}.
|
||||
*
|
||||
* @throws PlacementException naming the profile, its credential, and the remaining cool-off
|
||||
*/
|
||||
private void enforceNotCoolingOff(String profile) {
|
||||
String credentialId = credentialIdFor(profile);
|
||||
outagePolicy.remainingCoolOffSeconds(credentialId).ifPresent(remaining -> {
|
||||
throw new PlacementException("worker profile '" + profile + "' credential '" + credentialId
|
||||
+ "' is cooling off after repeated backend errors; ~" + remaining
|
||||
+ "s remaining — refusing spawn");
|
||||
});
|
||||
}
|
||||
|
||||
/** The subset of {@code candidates} whose credential is currently cooling off (fleetd #201 Unit 5). */
|
||||
private Set<String> coolingOffProfiles(List<PlacementCandidate> candidates) {
|
||||
return candidates.stream()
|
||||
.map(PlacementCandidate::profile)
|
||||
.filter(p -> outagePolicy.remainingCoolOffSeconds(credentialIdFor(p)).isPresent())
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
private void enforceMaxLoad(String profile) {
|
||||
// Absent config, or a config whose maxLoad normalized to null (ABSENT ⇒ unlimited at load),
|
||||
// means no cap — never cap what wasn't configured. Note "non-positive ⇒ unlimited" was true
|
||||
@@ -440,12 +538,23 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
public void stop(String id) {
|
||||
HerdrPeerLauncher d = spawnedBy.get(id);
|
||||
if (d == null) {
|
||||
if (herdrDaemonCount() != 1) {
|
||||
throw new IllegalArgumentException("ambiguous paneId '" + id
|
||||
+ "': no owning herdr daemon was recorded");
|
||||
if (herdrDaemonCount() == 1) {
|
||||
log.debug("stop({}) — no recorded owner in a single-daemon fleet", id);
|
||||
d = delegates.getFirst();
|
||||
} else {
|
||||
d = probeOwner(id);
|
||||
if (d == null) {
|
||||
// No configured herdr daemon has ever heard of this pane. CB-185 blocker 1: this
|
||||
// is the normal case right after a daemon restart empties spawnedBy for a member
|
||||
// that has ALREADY been torn down since — the caller retried a stop that already
|
||||
// succeeded. Nothing to close and no owner to cache; matching the tolerance
|
||||
// HerdrPeerLauncher#stop already gives an already-gone pane (agent.close swallows
|
||||
// that as success), stop() here is a no-op rather than a refusal.
|
||||
log.debug("stop({}) — no configured herdr daemon knows this pane; "
|
||||
+ "treating as already stopped", id);
|
||||
return;
|
||||
}
|
||||
}
|
||||
log.debug("stop({}) — no recorded owner in a single-daemon fleet", id);
|
||||
d = delegates.getFirst();
|
||||
}
|
||||
// Drop the owner record only after the delegate accepted the stop. Removing it first meant a
|
||||
// delegate that threw left the pane alive with its owner forgotten, so the retry fell into
|
||||
@@ -454,6 +563,61 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
spawnedBy.remove(id);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-185 blocker 1: recover a spawnedBy cache miss by asking every distinct herdr daemon which
|
||||
* one actually knows {@code id} — the fix for "after a restart, every surviving member becomes
|
||||
* un-stoppable" (spawnedBy is in-memory only, so a restart empties it, and members intentionally
|
||||
* outlive the daemon).
|
||||
*
|
||||
* <p>Grouped by daemon identity, not by delegate, for the same reason {@link #list()} groups
|
||||
* that way: two adapters (claude-code, opencode) sharing one herdr connection would otherwise be
|
||||
* probed twice, and a pane on their shared daemon would look owned by two adapters instead of
|
||||
* one daemon.
|
||||
*
|
||||
* <p>A daemon that fails to answer {@code list()} (e.g. it is down) is treated as "does not know
|
||||
* this pane" rather than aborting the whole probe — one unreachable daemon must never make a
|
||||
* pane that a <em>different</em>, healthy daemon actually owns un-stoppable too, which would
|
||||
* resurrect the exact bug this method exists to fix.
|
||||
*
|
||||
* @return the owning delegate — cached into {@link #spawnedBy} so the next call is free — or
|
||||
* {@code null} when no daemon knows the pane
|
||||
* @throws IllegalArgumentException when more than one daemon claims the pane: pane ids are
|
||||
* per-daemon counters, so two daemons really can both hold, say, {@code w1:p1}, and there
|
||||
* is no way to tell which one the caller means
|
||||
*/
|
||||
private HerdrPeerLauncher probeOwner(String id) {
|
||||
Map<HerdrClient, HerdrPeerLauncher> byDaemon = new IdentityHashMap<>();
|
||||
for (HerdrPeerLauncher delegate : delegates) {
|
||||
byDaemon.putIfAbsent(delegate.herdr(), delegate);
|
||||
}
|
||||
List<HerdrPeerLauncher> owners = new ArrayList<>();
|
||||
for (HerdrPeerLauncher representative : byDaemon.values()) {
|
||||
List<Agent> agents;
|
||||
try {
|
||||
agents = representative.list();
|
||||
} catch (HerdrException e) {
|
||||
log.warn("stop({}) probe: a configured herdr daemon was unreachable ({}); "
|
||||
+ "treating it as not knowing this pane", id, e.getClass().getSimpleName());
|
||||
continue;
|
||||
}
|
||||
boolean knows = agents.stream().anyMatch(a -> id.equals(a.paneId()));
|
||||
if (knows) {
|
||||
owners.add(representative);
|
||||
}
|
||||
}
|
||||
if (owners.size() > 1) {
|
||||
throw new IllegalArgumentException("ambiguous paneId '" + id + "': "
|
||||
+ owners.size() + " configured herdr daemons report this pane — "
|
||||
+ "no way to tell which one the caller means");
|
||||
}
|
||||
if (owners.isEmpty()) {
|
||||
return null;
|
||||
}
|
||||
HerdrPeerLauncher owner = owners.get(0);
|
||||
spawnedBy.put(id, owner);
|
||||
return owner;
|
||||
}
|
||||
|
||||
/**
|
||||
* Count actual herdr daemons, not peer adapter kinds. Identity is intentional: separate client
|
||||
* objects may represent different daemons even if a client later implements value equality.
|
||||
|
||||
@@ -7,6 +7,9 @@ import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.attribute.GroupPrincipal;
|
||||
import java.nio.file.attribute.PosixFileAttributeView;
|
||||
import java.nio.file.attribute.PosixFilePermissions;
|
||||
import java.time.Duration;
|
||||
import java.time.Instant;
|
||||
import java.util.ArrayList;
|
||||
@@ -116,6 +119,104 @@ public final class EnvAllowListScrub {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213: as {@link #generate(Path, Set)}, plus share the generated directory with
|
||||
* {@code group} — the member OS user's group (the operator's existing {@code worktreeGroup:}
|
||||
* name, reused rather than inventing a second one) — so a member running under a different OS
|
||||
* user than fleetd's own process can still read what it needs from a directory placed outside
|
||||
* {@code java.io.tmpdir}. {@code group} null/blank ⇒ identical to {@link #generate(Path, Set)};
|
||||
* this is the single-daemon (no {@code memberHerdrSocket}) shape, where the pane is fleetd's own
|
||||
* uid and no group sharing is needed.
|
||||
*
|
||||
* @throws UncheckedIOException also when {@code group} does not resolve on this host, or a
|
||||
* group-ownership/permission call is refused — the same "fail
|
||||
* loudly rather than start unprotected" contract as above: a scrub
|
||||
* the configured member user cannot even read is not a working
|
||||
* control.
|
||||
*/
|
||||
public static Path generate(Path parentDir, Set<String> allowedNames, String group) {
|
||||
Path dir = generate(parentDir, allowedNames);
|
||||
if (group != null && !group.isBlank()) {
|
||||
shareWithGroup(dir, group);
|
||||
}
|
||||
return dir;
|
||||
}
|
||||
|
||||
/**
|
||||
* chgrp/chmod-equivalent over a freshly generated directory and the flat files already written
|
||||
* into it: owner keeps full access, {@code group} gets traverse+read on the directory ({@code
|
||||
* rwxr-x---}, so a member process — a login shell reading it via {@code ZDOTDIR}, or another
|
||||
* process simply opening a file under it — running under that group can find and read the
|
||||
* files) and read-only on each file ({@code rw-r-----}) — deliberately no group WRITE anywhere,
|
||||
* since a member never needs to add or change what fleetd generated. (For the ZDOTDIR scrub
|
||||
* specifically, this also means the scrub script's own report write inside the pane fails
|
||||
* closed rather than open — see {@code scrub.zsh}'s trailing {@code 2>/dev/null} — which {@link
|
||||
* dev.ltms.fleet.member.HerdrPeerLauncher#releaseZdotdir} already treats as "cannot be
|
||||
* confirmed to have run" rather than success.)
|
||||
*
|
||||
* <p>Package-private and named generically on purpose: fleetd #213 built this for the ZDOTDIR
|
||||
* scrub directory, and fleetd #219 reuses it verbatim for {@link
|
||||
* dev.ltms.fleet.member.OpenCodeLauncher}'s ephemeral {@code opencode.json} directory — both are
|
||||
* "a fleetd-generated directory of flat files that a different-uid member process must read but
|
||||
* never write," so the sharing mechanism is shared rather than copied a second time.
|
||||
*/
|
||||
static void shareWithGroup(Path dir, String group) {
|
||||
try {
|
||||
GroupPrincipal principal = dir.getFileSystem().getUserPrincipalLookupService()
|
||||
.lookupPrincipalByGroupName(group);
|
||||
setGroupAndPermissions(dir, principal, "rwxr-x---");
|
||||
try (Stream<Path> entries = Files.list(dir)) {
|
||||
for (Path file : entries.toList()) {
|
||||
setGroupAndPermissions(file, principal, "rw-r-----");
|
||||
}
|
||||
}
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot share generated directory " + dir + " with group '"
|
||||
+ group + "' — the group must exist, and the fleetd operator ("
|
||||
+ System.getProperty("user.name") + ") must be a member of it", e);
|
||||
} catch (UnsupportedOperationException e) {
|
||||
throw new UncheckedIOException("cannot share generated directory " + dir + " with group '"
|
||||
+ group + "' — this filesystem does not support POSIX group ownership",
|
||||
new IOException(e));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #285: share ONE file with {@code group}, read-only ({@code rw-r-----}) — the same
|
||||
* per-file mode {@link #shareWithGroup} applies to a directory's entries, and the same error
|
||||
* shapes, but without touching a parent directory. Used for a file fleetd writes into a
|
||||
* directory it does NOT own — {@code configDir}'s own traversal permissions stay the
|
||||
* operator's setup — where the directory-wide {@link #shareWithGroup} would be wrong.
|
||||
*
|
||||
* @throws UncheckedIOException when {@code group} does not resolve on this host, the
|
||||
* filesystem has no POSIX group ownership, or a
|
||||
* group-ownership/permission call is refused
|
||||
*/
|
||||
static void shareFileWithGroup(Path file, String group) {
|
||||
try {
|
||||
GroupPrincipal principal = file.getFileSystem().getUserPrincipalLookupService()
|
||||
.lookupPrincipalByGroupName(group);
|
||||
setGroupAndPermissions(file, principal, "rw-r-----");
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot share generated file " + file + " with group '"
|
||||
+ group + "' — the group must exist, and the fleetd operator ("
|
||||
+ System.getProperty("user.name") + ") must be a member of it", e);
|
||||
} catch (UnsupportedOperationException e) {
|
||||
throw new UncheckedIOException("cannot share generated file " + file + " with group '"
|
||||
+ group + "' — this filesystem does not support POSIX group ownership",
|
||||
new IOException(e));
|
||||
}
|
||||
}
|
||||
|
||||
private static void setGroupAndPermissions(Path path, GroupPrincipal group, String perms) throws IOException {
|
||||
PosixFileAttributeView view = Files.getFileAttributeView(path, PosixFileAttributeView.class);
|
||||
if (view == null) {
|
||||
throw new IOException("POSIX file attributes are not supported for " + path);
|
||||
}
|
||||
view.setGroup(group);
|
||||
Files.setPosixFilePermissions(path, PosixFilePermissions.fromString(perms));
|
||||
}
|
||||
|
||||
/** One operator-sourcing startup file: source the {@code $HOME} counterpart, change nothing else. */
|
||||
private static String homeSourcingFile(String name) {
|
||||
return """
|
||||
|
||||
@@ -3,11 +3,13 @@ package dev.ltms.fleet.member;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.Tab;
|
||||
import dev.ltms.fleet.herdr.Workspace;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.StatusRefiner;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.CharterReceipt;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
@@ -112,6 +114,14 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* daemon's own process is started the same way (a login shell sourcing the same secret store —
|
||||
* see CB-592's investigation of {@code secrets.sh}), so on a single-host deployment its env
|
||||
* mirrors what the pane's login shell is about to export.
|
||||
*
|
||||
* <p>fleetd #185 stage 2: that mirroring assumption holds only while the member pane runs under
|
||||
* the SAME OS user as the daemon. When {@code memberHerdrSocket:} is configured, member panes
|
||||
* are routed to a second herdr, and fleetd has no channel to confirm what OS user that herdr
|
||||
* runs as — it may be a different user with a different {@code $HOME} and {@code secrets.sh},
|
||||
* or the same one the daemon runs as. Either way this field's data can no longer be trusted to
|
||||
* describe what a member pane inherits. See {@link #logCredentialGap} for how that mode is
|
||||
* handled.
|
||||
*/
|
||||
private final Supplier<Set<String>> hostEnvNames;
|
||||
|
||||
@@ -145,6 +155,14 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
private final LongSupplier nowMillis; // monotonic clock (injectable for tests)
|
||||
private final Runnable sleeper; // sleep/wait hook (injectable for tests; never real-sleep in unit tests)
|
||||
|
||||
/**
|
||||
* fleetd #176 fix 2: the same UNKNOWN-refinement {@code dev.ltms.fleet.inject.StatusPoller}
|
||||
* uses, reused here for the spawn-readiness gate. Constructed once from {@link #agents} — see
|
||||
* {@link #refinedInjectable(String, Agent)} for the corroboration that keeps it from firing on
|
||||
* a dead pane's bare shell prompt.
|
||||
*/
|
||||
private final StatusRefiner statusRefiner;
|
||||
|
||||
// Per-process token mixed into each peer name so a fresh process (nameSeq back at 0) cannot
|
||||
// collide with same-profile peers that outlived a restart. See startUniquelyNamed.
|
||||
private final String nameNonce = String.format("%06x", new SecureRandom().nextInt(1 << 24));
|
||||
@@ -168,8 +186,14 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
*/
|
||||
private final ConcurrentMap<String, Path> zdotdirByPane = new ConcurrentHashMap<>();
|
||||
|
||||
/** Guards {@link #warnNonZsh} to one WARN per launcher instance, not one per spawn. */
|
||||
/**
|
||||
* Guards {@link #warnNonZsh} to one WARN per launcher instance, not one per spawn. fleetd #155:
|
||||
* only the {@code policy: deny-by-default} path still warns on a non-zsh shell — the {@code
|
||||
* policy: allow-list} path refuses the spawn instead (see {@link #applyEnvironmentAllowListPolicy}).
|
||||
*/
|
||||
private final AtomicBoolean nonZshShellWarned = new AtomicBoolean();
|
||||
/** Live config provides URI environment names that must never enter member panes. */
|
||||
private final Supplier<FleetConfig> config;
|
||||
|
||||
/**
|
||||
* @param namePrefix label prefix for this peer kind (drives naming and reap)
|
||||
@@ -238,8 +262,19 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<Set<String>> hostEnvNames) {
|
||||
this(namePrefix, agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs, nowMillis,
|
||||
sleeper, fleet, memberCredentials, hostEnvNames, null);
|
||||
}
|
||||
|
||||
/** As above, plus the live full config for secret-bearing URI environment names. */
|
||||
protected HerdrPeerLauncher(String namePrefix, AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper, Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<Set<String>> hostEnvNames) {
|
||||
Supplier<Set<String>> hostEnvNames, Supplier<FleetConfig> config) {
|
||||
this.fleet = fleet;
|
||||
this.namePrefix = namePrefix;
|
||||
this.agents = agents;
|
||||
@@ -252,6 +287,8 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
this.sleeper = sleeper;
|
||||
this.memberCredentials = memberCredentials;
|
||||
this.hostEnvNames = hostEnvNames != null ? hostEnvNames : () -> System.getenv().keySet();
|
||||
this.config = config;
|
||||
this.statusRefiner = new StatusRefiner(agents);
|
||||
}
|
||||
|
||||
// --- adapter seams -------------------------------------------------------------------------
|
||||
@@ -327,6 +364,42 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
return Files.isRegularFile(candidate) ? candidate : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code cwd} is a fleetd-provisioned git worktree — signalled the same way
|
||||
* {@code ClaudeCodeLauncher#writeIdeOverlay} already gates on: a {@code .git} that is a
|
||||
* <strong>regular file</strong> holding a {@code gitdir:} pointer, as opposed to a real
|
||||
* checkout's {@code .git} <strong>directory</strong>. {@code null}/blank never qualifies.
|
||||
*
|
||||
* <p>Shared by every write (and, since fleetd #249, every identity read) that must land only
|
||||
* in a worktree fleetd itself created for a member — never in a real checkout, an arbitrary
|
||||
* configured directory, or (see the incident below) the daemon's own fallback cwd. Package-
|
||||
* private (not {@code protected}) on purpose: {@link ClaudeCodeLauncher} and
|
||||
* {@link OpenCodeLauncher} both call it, and same-package visibility is enough — no subclass
|
||||
* outside this package needs it.
|
||||
*
|
||||
* <p><b>fleetd #149 incident.</b> {@code ClaudeCodeLauncher#seedTrustDialog} originally ran
|
||||
* unconditionally on any non-blank {@code cwd}. Most of that launcher's OWN tests spawn a
|
||||
* profile with no {@code cwd} configured, so the base class's {@code resolveCwd} falls
|
||||
* through to the real {@code user.dir} — and with no {@code configDir} either (also the
|
||||
* common case in that file's fixtures), the seed's target falls through the same way to the
|
||||
* real {@code ~/.claude.json}. Running this repo's own test suite corrupted the operator's
|
||||
* actual config file (it shrank from ~72 KB to a single seeded entry) the first time a
|
||||
* mutation happened to make the write non-additive. Gating both cwd-targeted writes on "this
|
||||
* is a worktree fleetd provisioned" — exactly the population fleetd #149 describes
|
||||
* ({@code worktree: true} always lands in a brand-new directory) — makes that class of write
|
||||
* impossible against a real checkout or an untouched fallback cwd, in production or in tests.
|
||||
*
|
||||
* <p><b>fleetd #249.</b> The same reasoning extends to a READ: {@code
|
||||
* OpenCodeSessionDiscovery#sessionIdForDirectory} keys on {@code directory}, a heuristic that
|
||||
* is only reliable when the directory is unique to this member — i.e., exactly the population
|
||||
* this gate identifies. {@link OpenCodeLauncher} uses it to withhold {@code agentSessionId()}
|
||||
* (report absence rather than a guess) and to refuse a {@code resumeSessionId} spawn that
|
||||
* cannot be resolved reliably going forward.
|
||||
*/
|
||||
static boolean isProvisionedWorktree(String cwd) {
|
||||
return cwd != null && !cwd.isBlank() && Files.isRegularFile(Path.of(cwd, ".git"));
|
||||
}
|
||||
|
||||
// --- profile surface -----------------------------------------------------------------------
|
||||
|
||||
/** The configured peer profile names (what {@code spawn(profile)} accepts). */
|
||||
@@ -624,7 +697,20 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
if (paneId == null) {
|
||||
throw new IllegalStateException("pane.split returned no pane — cannot start a peer");
|
||||
}
|
||||
Agent peer = startUniquelyNamed(cfg, argv, paneId).agent();
|
||||
Agent peer;
|
||||
try {
|
||||
peer = startUniquelyNamed(cfg, argv, paneId).agent();
|
||||
} catch (RuntimeException e) {
|
||||
// The peer never started — don't leave the pane we just created orphaned.
|
||||
// Best-effort cleanup; never let it mask the real spawn failure.
|
||||
try {
|
||||
stop(paneId);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to close orphaned pane {} after spawn error: {}",
|
||||
paneId, cleanup.getMessage());
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
log.info("{} started pane={} terminal={}", namePrefix, peer.paneId(), peer.terminalId());
|
||||
return peer;
|
||||
}
|
||||
@@ -662,6 +748,7 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
// Protocol 19 resolves the executable from the agent kind (== namePrefix here), so
|
||||
// argv[0] — the configured executable — is dropped and only the extra args are passed.
|
||||
List<String> args = argv.isEmpty() ? argv : argv.subList(1, argv.size());
|
||||
checkPaneCommandFits(cfg, argv);
|
||||
HerdrException last = null;
|
||||
for (int attempt = 0; attempt < NAME_RETRIES; attempt++) {
|
||||
long seq = nameSeq.incrementAndGet();
|
||||
@@ -677,6 +764,59 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
throw last;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #220: herdr does not exec the launch command — it TYPES it into the pane as one line,
|
||||
* and a pty line buffer holds only {@value #PANE_COMMAND_BYTE_LIMIT} bytes (BSD/macOS {@code
|
||||
* MAX_CANON}). Everything past that byte is dropped. Nothing reports it: herdr answers "agent
|
||||
* started", the backend exits on the mangled argument it was handed, the pane closes, and the
|
||||
* only symptom is {@link #waitUntilInjectableOrThrow} timing out 20 seconds later with no
|
||||
* reason. That is exactly how #214 broke every claude-code spawn — one 50-byte flag pushed a
|
||||
* 978-byte command to 1028, and the tail that got cut was {@code --autocompact 250000}.
|
||||
*
|
||||
* <p>So measure it here and refuse, loudly and immediately, rather than spawn something that
|
||||
* cannot work. The estimate is deliberately conservative: fleetd cannot see herdr's quoting, so
|
||||
* every argument is charged its own bytes plus a separator and a quote pair. An over-estimate
|
||||
* costs a clear error at a length that was already unsafe; an under-estimate would let the
|
||||
* silent truncation back in.
|
||||
*
|
||||
* @throws PeerUnreachableException when the command cannot fit — the same failure the spawn
|
||||
* would have hit anyway, named at the point it is still
|
||||
* explainable
|
||||
*/
|
||||
private void checkPaneCommandFits(FleetConfig.Profile cfg, List<String> argv) {
|
||||
int bytes = 0;
|
||||
String longest = null;
|
||||
int longestBytes = 0;
|
||||
for (String arg : argv) {
|
||||
int argBytes = arg == null ? 0 : arg.getBytes(java.nio.charset.StandardCharsets.UTF_8).length;
|
||||
bytes += argBytes + QUOTING_OVERHEAD_PER_ARG;
|
||||
if (argBytes > longestBytes) {
|
||||
longestBytes = argBytes;
|
||||
longest = arg;
|
||||
}
|
||||
}
|
||||
if (bytes <= PANE_COMMAND_BYTE_LIMIT) {
|
||||
return;
|
||||
}
|
||||
String culprit = longest == null ? "<none>"
|
||||
: longest.substring(0, Math.min(longest.length(), 60)) + (longest.length() > 60 ? "…" : "");
|
||||
throw new PeerUnreachableException(
|
||||
"launch command for profile " + cfg.profile() + " is about " + bytes + " bytes, over the "
|
||||
+ PANE_COMMAND_BYTE_LIMIT + "-byte limit of the pane line herdr types it into. "
|
||||
+ "The pty would drop the tail silently and the backend would exit on a mangled "
|
||||
+ "argument. Longest argument is " + longestBytes + " bytes: " + culprit
|
||||
+ " — move it off the command line (a file flag) or shorten it.");
|
||||
}
|
||||
|
||||
/**
|
||||
* The pty line buffer herdr types a launch command into: BSD/macOS {@code MAX_CANON}. Not a
|
||||
* fleetd choice and not configurable — see {@link #checkPaneCommandFits}.
|
||||
*/
|
||||
static final int PANE_COMMAND_BYTE_LIMIT = 1024;
|
||||
|
||||
/** Per-argument allowance for the separating space and a shell quote pair fleetd cannot see. */
|
||||
private static final int QUOTING_OVERHEAD_PER_ARG = 3;
|
||||
|
||||
/** Start the agent into {@code paneId}, waiting out the seed shell's boot with the sleeper. */
|
||||
private Agent startAwaitingShellPrompt(String name, List<String> args, String paneId) {
|
||||
HerdrException busy = null;
|
||||
@@ -786,9 +926,14 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* {@link #reapOrphanWorkers() orphan-reap} and spawn-gate-timeout paths, plus any caller that
|
||||
* passes a pane directly, keep working without an owning id.
|
||||
*
|
||||
* <p>Resolves the tab from the pane <em>before</em> closing it. An already-gone pane/tab
|
||||
* (repeated DELETE, crashed peer) is treated as success; any other failure propagates so a
|
||||
* genuinely failed teardown is not reported as done.
|
||||
* <p>Resolves the tab from the pane <em>before</em> closing it. {@code agents.close} (the pane)
|
||||
* is the one step whose failure means the teardown itself may not have happened: an already-gone
|
||||
* pane (repeated DELETE, crashed peer) is treated as success, but any other failure propagates so
|
||||
* a genuinely failed teardown is not reported as done. {@code spaces.closeTab} (fleetd #293) is
|
||||
* different — by the time it runs the pane is already closed, so it is cosmetic workspace tidying
|
||||
* rather than a real teardown failure, and a failure there is logged and never propagates, so it
|
||||
* cannot mask the two cleanups below it ({@link #releaseZdotdir}, and the caller's worktree
|
||||
* removal in {@code SessionManager.release}).
|
||||
*/
|
||||
@Override
|
||||
public void stop(String idOrPane) {
|
||||
@@ -807,7 +952,25 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
log.debug("pane.close({}) ignored — already gone: {}", paneId, e.getMessage());
|
||||
}
|
||||
if (loc != null && loc.tabPaneCount() == 1) {
|
||||
spaces.closeTab(loc.tabId());
|
||||
// fleetd #293: the pane above is already closed by this point, so a failing tab.close is
|
||||
// cosmetic workspace tidying, not a real teardown failure — it must not mask the two
|
||||
// cleanups below it (releaseZdotdir, and the caller's worktree removal). Unlike
|
||||
// agents.close above, this is not narrowed to "already gone": any failure here, whatever
|
||||
// its cause, is one we continue past, so we log it at WARN (not debug) with the tab id a
|
||||
// person can go close by hand.
|
||||
try {
|
||||
spaces.closeTab(loc.tabId());
|
||||
} catch (RuntimeException e) {
|
||||
// Caught as RuntimeException, not HerdrException, to match releaseZdotdir's own
|
||||
// guard five lines below. Today the two are the same set — HerdrCodec wraps every
|
||||
// encode/decode failure and UnixSocketHerdrClient wraps every IOException, so
|
||||
// HerdrException is all closeTab can actually throw. Narrowing to it anyway would
|
||||
// leave this step guarded against the expected failure and bare against any other,
|
||||
// which is the exact asymmetry fleetd #293 exists to remove. No behaviour change
|
||||
// today; it stops a later change inside WorkspaceControl.closeTab reopening it.
|
||||
log.warn("tab.close({}) failed — the pane is already torn down, so continuing; the "
|
||||
+ "tab may need manual cleanup: {}", loc.tabId(), e.getMessage());
|
||||
}
|
||||
} else if (loc != null) {
|
||||
log.debug("not closing tab {} — it holds {} panes (not a dedicated peer tab)",
|
||||
loc.tabId(), loc.tabPaneCount());
|
||||
@@ -835,25 +998,144 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
// --- spawn-readiness gate (CB-306) ---------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Poll {@link AgentControl#status} until the pane reports an injectable state or the configured
|
||||
* Poll {@link AgentControl#get} until the pane reports an injectable state or the configured
|
||||
* timeout elapses. On timeout, close the pane (self-reap) and throw.
|
||||
*
|
||||
* <p>fleetd #176 fix 1: {@code agents.get} was previously unguarded here, so a herdr
|
||||
* {@code *_not_found} answer — which is what happens when the backend process EXITED rather
|
||||
* than being slow — propagated as a raw {@link HerdrException} instead of the
|
||||
* {@link PeerUnreachableException} every other failure path of this gate produces, and skipped
|
||||
* teardown ({@link #stop}) entirely, leaking the pane/tab. {@link #failFastOnGoneBackend} closes
|
||||
* that gap: it stops waiting immediately (never burns the rest of the timeout), runs the same
|
||||
* teardown the timeout path below runs, and throws with a message that says the backend exited
|
||||
* rather than that the pane was slow. Any other {@link HerdrException} still propagates
|
||||
* unchanged — this gate does not interpret or recover from it, but it still closes the pane
|
||||
* it opened before handing the exception to its caller.
|
||||
*/
|
||||
private void waitUntilInjectableOrThrow(String paneId) {
|
||||
long deadline = nowMillis.getAsLong() + spawnReadyTimeoutMs;
|
||||
long start = nowMillis.getAsLong();
|
||||
long deadline = start + spawnReadyTimeoutMs;
|
||||
AgentStatus lastStatus = null;
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
if (agents.status(paneId).injectable()) {
|
||||
Agent sample;
|
||||
try {
|
||||
sample = agents.get(paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (isAlreadyGone(e)) {
|
||||
failFastOnGoneBackend(paneId, e, nowMillis.getAsLong() - start);
|
||||
}
|
||||
// This gate must not interpret an unrelated herdr error, but the caller does not
|
||||
// receive paneId when spawn throws. Close the pane here before propagating e unchanged.
|
||||
try {
|
||||
stop(paneId);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to close orphaned pane {} after readiness-gate error: {}",
|
||||
paneId, cleanup.getMessage());
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
lastStatus = sample.status();
|
||||
if (lastStatus.injectable() || refinedInjectable(paneId, sample)) {
|
||||
log.debug("peer pane={} reached injectable state", paneId);
|
||||
return;
|
||||
}
|
||||
sleeper.run();
|
||||
}
|
||||
log.warn("peer pane={} did not become injectable within {}ms — closing", paneId, spawnReadyTimeoutMs);
|
||||
// fleetd #220: read the pane BEFORE stop() closes it. Without this the gate says only that
|
||||
// it timed out, which is true of every cause — a backend that never launched, a binary that
|
||||
// rejected an argument and exited, a trust prompt, a login shell that hung. The pane holds
|
||||
// the one copy of that answer and it is destroyed a line later.
|
||||
log.warn("peer pane={} did not become injectable within {}ms (last status {}) — closing. "
|
||||
+ "Pane tail:\n{}",
|
||||
paneId, spawnReadyTimeoutMs, lastStatus, readPaneQuietly(paneId));
|
||||
stop(paneId);
|
||||
throw new PeerUnreachableException(
|
||||
"worker pane " + paneId + " did not reach injectable state within "
|
||||
+ spawnReadyTimeoutMs + "ms");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176 fix 1: the backend process exited while the gate was still waiting — herdr
|
||||
* answered {@code *_not_found} instead of ever reporting an injectable status. Fails
|
||||
* immediately (never burns the rest of {@link #spawnReadyTimeoutMs}), runs the exact same
|
||||
* teardown {@link #waitUntilInjectableOrThrow}'s timeout path runs, and throws a
|
||||
* {@link PeerUnreachableException} whose message says the process exited rather than that the
|
||||
* pane was slow.
|
||||
*
|
||||
* @throws PeerUnreachableException always — this method never returns normally
|
||||
*/
|
||||
private void failFastOnGoneBackend(String paneId, HerdrException cause, long elapsedMs) {
|
||||
String tail = readPaneQuietly(paneId); // read before stop() closes the pane
|
||||
log.warn("peer pane={} backend process exited after {}ms while waiting for injectable "
|
||||
+ "state (herdr: {}) — closing. Pane tail:\n{}",
|
||||
paneId, elapsedMs, cause.getMessage(), tail);
|
||||
stop(paneId);
|
||||
throw new PeerUnreachableException(
|
||||
"worker pane " + paneId + " backend process exited after " + elapsedMs
|
||||
+ "ms while waiting to become injectable (herdr reported: " + cause.getMessage()
|
||||
+ "). Pane tail:\n" + tail);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176 fix 2: resolve a raw {@link AgentStatus#UNKNOWN} sample into a trustworthy
|
||||
* injectable state via pane content, the same refinement {@code StatusPoller} applies during a
|
||||
* peer's working life — but corroborated, because the trap this gate is exposed to that the
|
||||
* poller is not: a pane whose backend has already exited settles at a plain shell prompt, and
|
||||
* that prompt commonly contains the same {@code ❯} glyph {@link StatusRefiner#classify} treats
|
||||
* as "idle at the Claude Code TUI prompt". Naively wiring the refiner in would turn "the backend
|
||||
* died" into "ready to inject" — strictly worse than today's timeout.
|
||||
*
|
||||
* <p>Two guards, both required:
|
||||
* <ul>
|
||||
* <li>{@link StatusRefiner#classify} is written for the Claude Code TUI only (its own javadoc
|
||||
* says so), so refinement only ever runs for the {@code claude} adapter — never for
|
||||
* {@code opencode} or any future non-Claude backend, whatever its pane looks like.</li>
|
||||
* <li>The refined status is accepted only when the <em>same</em> {@code agents.get} sample
|
||||
* ({@code sample}, taken once by the caller) still reports a non-null
|
||||
* {@link Agent#agentType()} — herdr's own "a supported backend is still detected here"
|
||||
* signal, read from the very sample the raw status came from so the two can never
|
||||
* disagree. A bare shell prompt left by an exited backend reports no {@code agentType}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Called only when {@code sample.status() == UNKNOWN}, so a healthy spawn — which never sees
|
||||
* {@code UNKNOWN} — triggers zero extra herdr calls; only a persistently-{@code unknown} pane
|
||||
* pays the one extra {@code agent.read} {@link StatusRefiner#refine} performs.
|
||||
*/
|
||||
private boolean refinedInjectable(String paneId, Agent sample) {
|
||||
if (sample.status() != AgentStatus.UNKNOWN) {
|
||||
return false;
|
||||
}
|
||||
if (!"claude".equals(namePrefix)) {
|
||||
return false; // StatusRefiner.classify reads a Claude Code TUI prompt specifically
|
||||
}
|
||||
if (sample.agentType() == null) {
|
||||
return false; // no corroborating liveness signal — could be a dead pane's bare shell
|
||||
}
|
||||
return statusRefiner.refine(paneId, AgentStatus.UNKNOWN, agents).injectable();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #220: the pane's recent output, clipped, for the readiness-gate timeout log — or a
|
||||
* short note when it cannot be read. Best-effort by construction: this runs on a path that is
|
||||
* already failing, so it must never replace the real error with one of its own.
|
||||
*/
|
||||
private String readPaneQuietly(String paneId) {
|
||||
try {
|
||||
String pane = agents.read(paneId, "recent");
|
||||
if (pane == null || pane.isBlank()) {
|
||||
return "<pane read returned nothing>";
|
||||
}
|
||||
return pane.length() <= SPAWN_FAILURE_PANE_CHARS
|
||||
? pane
|
||||
: pane.substring(pane.length() - SPAWN_FAILURE_PANE_CHARS);
|
||||
} catch (RuntimeException e) {
|
||||
return "<pane could not be read: " + e.getMessage() + ">";
|
||||
}
|
||||
}
|
||||
|
||||
/** How much of a failed spawn's pane the timeout log carries. */
|
||||
private static final int SPAWN_FAILURE_PANE_CHARS = 4000;
|
||||
|
||||
/**
|
||||
* A concrete {@link PeerHandle} wrapping herdr agent coordinates, the profile that spawned it,
|
||||
* the session identity the launch resolved (CB-547a): the bridge's logical name and the peer's
|
||||
@@ -1008,6 +1290,13 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* applied before the login shell runs and a sourced file can (and did) undo it. The control is
|
||||
* the ZDOTDIR scrub ({@link #applyEnvironmentAllowListPolicy}); {@code known}/{@code allow}
|
||||
* remain as reporting only via {@link #logCredentialGap}.
|
||||
*
|
||||
* <p>CB-633 follow-up (#192): under {@code allow-list} this method does NOT call {@link
|
||||
* #logCredentialGap} itself — at this point (called from {@link #baseEnv}, before {@link
|
||||
* #applyEnvironmentAllowListPolicy} runs) we do not yet know whether the pane's shell is zsh, so
|
||||
* we cannot yet pick correct wording. That decision, and the call, are deferred entirely to
|
||||
* {@link #applyEnvironmentAllowListPolicy}, which knows by then whether the scrub will actually
|
||||
* run.
|
||||
*/
|
||||
private void applyMemberCredentialPolicy(Map<String, String> workerEnv) {
|
||||
FleetConfig.MemberCredentials creds = memberCredentials == null ? null : memberCredentials.get();
|
||||
@@ -1016,14 +1305,21 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
}
|
||||
if (!creds.isAllowList()) {
|
||||
overlayBlockedCredentials(workerEnv, creds);
|
||||
logCredentialGap(creds, null);
|
||||
// fleetd #155: this overlay itself does not depend on the shell (it lands in the
|
||||
// pane-creation env map before any shell runs), so the spawn is never refused here —
|
||||
// only policy=allow-list's ZDOTDIR scrub needs a login shell to run at all. Still worth
|
||||
// telling the operator: the stronger post-shell control is unavailable on this shell.
|
||||
warnNonZsh(memberLoginShell());
|
||||
}
|
||||
logCredentialGap(creds);
|
||||
}
|
||||
|
||||
/** Put {@link #BLOCKED_CREDENTIAL_SENTINEL} over every blocked name in the pane-creation env map. */
|
||||
private static void overlayBlockedCredentials(Map<String, String> workerEnv,
|
||||
FleetConfig.MemberCredentials creds) {
|
||||
for (String name : creds.blockedSet()) {
|
||||
private void overlayBlockedCredentials(Map<String, String> workerEnv,
|
||||
FleetConfig.MemberCredentials creds) {
|
||||
Set<String> blocked = new java.util.TreeSet<>(creds.blockedSet());
|
||||
blocked.addAll(brokerUriEnvNames());
|
||||
for (String name : blocked) {
|
||||
workerEnv.put(name, BLOCKED_CREDENTIAL_SENTINEL);
|
||||
}
|
||||
}
|
||||
@@ -1048,6 +1344,35 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* name. It stays a one-off decision because it is a live handle to the operator's ssh-agent, not
|
||||
* a value — a member holding it can sign with every key the agent holds, so letting it ride in
|
||||
* on the generic {@code allow:} list would hand that out for an unrelated reason.
|
||||
*
|
||||
* <p>fleetd #213: which shell decides the zsh gate, and where the generated directory lives,
|
||||
* both depend on whether {@code memberHerdrSocket:} is configured — see {@link
|
||||
* #memberHerdrSocketConfigured()}'s javadoc for why fleetd's own {@code $SHELL} and {@code
|
||||
* java.io.tmpdir} describe the wrong process once member panes run under a different OS user.
|
||||
* <ul>
|
||||
* <li>{@code memberHerdrSocket} ABSENT (today's only mode): byte-identical to before this
|
||||
* fix — fleetd's own {@code $SHELL} decides zsh, and the directory is generated under
|
||||
* {@code java.io.tmpdir}.</li>
|
||||
* <li>{@code memberHerdrSocket} PRESENT: the configured {@code memberLoginShell:} decides
|
||||
* zsh instead — fleetd's own {@code $SHELL} is never consulted, since it names a
|
||||
* different user's shell, not the member's. Absent/non-zsh falls back exactly like the
|
||||
* non-zsh case below. When it IS zsh, the directory still cannot go under {@code
|
||||
* java.io.tmpdir} (mode 0700, unreadable by another uid — the exact gap fleetd #213
|
||||
* exists to close), so it is generated under {@code worktreeRoot} instead and shared
|
||||
* read-only with {@code worktreeGroup} — the same group {@link
|
||||
* dev.ltms.fleet.session.Worktrees#shareWithGroup} already uses, reused rather than
|
||||
* inventing a second group key. Either one missing means the scrub cannot be guaranteed
|
||||
* reachable by the member, which is the same "cannot guarantee the scrub runs" case as a
|
||||
* non-zsh shell — that one still gets the fallback (see fleetd #155's javadoc note below
|
||||
* on why a missing worktreeRoot/worktreeGroup is a different defect, out of that ticket's
|
||||
* scope).</li>
|
||||
* </ul>
|
||||
* fleetd #155: the one exception to "never refuse to spawn" is a non-zsh login shell, right
|
||||
* below. The operator asked for {@code policy: allow-list} specifically — its whole point is a
|
||||
* control a sourced file cannot undo — so silently degrading to the weaker overlay is exactly
|
||||
* the "silently does nothing" failure this ticket exists to remove. Every other branch in this
|
||||
* method (worktreeRoot/worktreeGroup missing) keeps the old "degrade, never refuse" behaviour;
|
||||
* that gap is real but is fleetd #213's scope, not this one.
|
||||
*/
|
||||
private Path applyEnvironmentAllowListPolicy(FleetConfig.Profile cfg, Launch launch) {
|
||||
FleetConfig.MemberCredentials creds = memberCredentials == null ? null : memberCredentials.get();
|
||||
@@ -1055,25 +1380,57 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
return null;
|
||||
}
|
||||
Set<String> allowed = derivedAllowedNames(creds, launch);
|
||||
String loginShell = resolveEnv("SHELL");
|
||||
boolean zsh = loginShell != null && (loginShell.endsWith("/zsh") || loginShell.equals("zsh"));
|
||||
boolean memberHerdrSocket = memberHerdrSocketConfigured();
|
||||
// fleetd #213 defect 1: under memberHerdrSocket the member pane runs as a DIFFERENT OS
|
||||
// user, so fleetd's own $SHELL says nothing about what that pane runs — resolveEnv("SHELL")
|
||||
// must not even be called on this path, only the explicit memberLoginShell: config can
|
||||
// answer it. With memberHerdrSocket absent, nothing here changes: fleetd's own $SHELL is
|
||||
// still the input, exactly as before this fix.
|
||||
String loginShell = memberLoginShell();
|
||||
boolean zsh = isZshShell(loginShell);
|
||||
if (!zsh) {
|
||||
// A non-zsh login shell ignores ZDOTDIR entirely: NO scrub would run, so pretending
|
||||
// otherwise would be worse than saying so. Warn loudly and fall back to the CB-596
|
||||
// sentinel overlay over the enumerated known: names — weaker (a sourced file can undo
|
||||
// it), but strictly better than nothing. Deliberately no "allowed N of M" line here: the
|
||||
// scrub this count describes does not run on this path, so printing it would tell an
|
||||
// operator that a fraction of names were blocked when the real number blocked is zero.
|
||||
// logCredentialGap's WARN (below) is the only signal for this path.
|
||||
warnNonZsh(loginShell);
|
||||
overlayBlockedCredentials(launch.env(), creds);
|
||||
logCredentialGap(creds);
|
||||
return null;
|
||||
// fleetd #155: a non-zsh login shell ignores ZDOTDIR entirely — NO scrub would run. The
|
||||
// operator explicitly asked for policy=allow-list's blocking control, so degrading to
|
||||
// the weaker CB-596 overlay and spawning anyway would be the same "control silently does
|
||||
// nothing" defect this ticket exists to close. Refuse instead — the caller (FleetMcp.spawn)
|
||||
// catches IllegalArgumentException and surfaces the message to the operator.
|
||||
throw new IllegalArgumentException(
|
||||
"memberCredentials policy=allow-list requires the member's login shell to be "
|
||||
+ "zsh, so the ZDOTDIR scrub can run after it — refusing to spawn under "
|
||||
+ "login shell '" + (loginShell == null ? "<unset>" : loginShell)
|
||||
+ "'. Configure memberLoginShell: as a zsh path (only read when "
|
||||
+ "memberHerdrSocket is set), move the member's OS account onto zsh, or "
|
||||
+ "set memberCredentials.policy: deny-by-default instead.");
|
||||
}
|
||||
Path parentDir;
|
||||
String group = null;
|
||||
if (memberHerdrSocket) {
|
||||
// fleetd #213 defect 2: java.io.tmpdir is fleetd's own per-user temp dir (mode 0700 on
|
||||
// macOS) — a member running as a different uid cannot even traverse it, let alone read
|
||||
// the generated files. worktreeRoot is the only configured location a different-uid
|
||||
// member can be given access to, and only WITH worktreeGroup to grant that access —
|
||||
// absent either, the scrub cannot be guaranteed reachable, so this falls back exactly
|
||||
// like the non-zsh case above rather than generating a directory nothing can read.
|
||||
parentDir = memberScrubParentDir();
|
||||
group = memberGroup();
|
||||
if (parentDir == null || group == null) {
|
||||
warnCannotShareScrubDirectory();
|
||||
overlayBlockedCredentials(launch.env(), creds);
|
||||
logCredentialGap(creds, null);
|
||||
return null;
|
||||
}
|
||||
} else {
|
||||
parentDir = Path.of(System.getProperty("java.io.tmpdir"));
|
||||
}
|
||||
// Only reached when the scrub is actually about to run — the count below describes that
|
||||
// scrub, so it must not be logged before this gate (see the non-zsh branch above).
|
||||
// scrub, so it must not be logged before this gate (see the non-zsh branch above). Same
|
||||
// reasoning gates logCredentialGap's wording: passing the derived `allowed` set (non-null)
|
||||
// here, and ONLY here, is what tells it the scrub will really blank an unkept name — #192.
|
||||
logAllowListCoverage(allowed);
|
||||
Path dir = EnvAllowListScrub.generate(Path.of(System.getProperty("java.io.tmpdir")), allowed);
|
||||
logCredentialGap(creds, allowed);
|
||||
Path dir = memberHerdrSocket
|
||||
? EnvAllowListScrub.generate(parentDir, allowed, group)
|
||||
: EnvAllowListScrub.generate(parentDir, allowed);
|
||||
launch.env().put("ZDOTDIR", dir.toAbsolutePath().toString());
|
||||
log.info("memberCredentials policy=allow-list: profile={} generated ZDOTDIR {} — derived "
|
||||
+ "allow-list holds {} name(s); the pane reports allowed N of M at release",
|
||||
@@ -1081,21 +1438,98 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
return dir;
|
||||
}
|
||||
|
||||
/** True when {@code shell} is a zsh login shell path or bare name — the ZDOTDIR gate. */
|
||||
private static boolean isZshShell(String shell) {
|
||||
return shell != null && (shell.endsWith("/zsh") || shell.equals("zsh"));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213: the configured {@code memberLoginShell:}, or {@code null} when unconfigured.
|
||||
* Called ONLY from the {@code memberHerdrSocket}-configured branch of {@link
|
||||
* #applyEnvironmentAllowListPolicy} — fleetd's own {@code $SHELL} is never read on that path.
|
||||
* {@link #config} being {@code null} (an older test call site, or a launcher that never
|
||||
* threaded the full config through) is treated the same as "not configured".
|
||||
*/
|
||||
private String configuredMemberLoginShell() {
|
||||
FleetConfig cfg = config == null ? null : config.get();
|
||||
return cfg == null ? null : cfg.memberLoginShell();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #155: the member pane's login shell, using the same {@code memberHerdrSocket} routing
|
||||
* {@link #applyEnvironmentAllowListPolicy} already used — fleetd's own {@code $SHELL} decides it
|
||||
* when {@code memberHerdrSocket} is absent (today's only mode); the configured {@code
|
||||
* memberLoginShell:} decides it otherwise, since fleetd's own {@code $SHELL} names a different
|
||||
* user's shell once member panes run under a different OS user. Shared by both {@link
|
||||
* #applyEnvironmentAllowListPolicy} (allow-list: refuses on non-zsh) and {@link
|
||||
* #applyMemberCredentialPolicy} (deny-by-default: warns on non-zsh) so the two policies agree on
|
||||
* what "the member's shell" means.
|
||||
*/
|
||||
private String memberLoginShell() {
|
||||
return memberHerdrSocketConfigured() ? configuredMemberLoginShell() : resolveEnv("SHELL");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213: {@code worktreeRoot}, as the ZDOTDIR scrub's parent directory under {@code
|
||||
* memberHerdrSocket}, or {@code null} when unconfigured — the same "cannot guarantee the scrub
|
||||
* runs" gap as {@link #memberGroup()} being unset (see {@link
|
||||
* #applyEnvironmentAllowListPolicy}). Deliberately no sibling-of-repo-root default here, unlike
|
||||
* {@code GitWorktrees}' own {@code worktreeRoot} resolution: that default is a convenience for
|
||||
* provisioning a worktree that will exist regardless, whereas an unconfigured value here means
|
||||
* fleetd has no operator-endorsed location to put a credential-bearing directory a different OS
|
||||
* user must reach, so falling back to the overlay is the honest answer, not a guess.
|
||||
*
|
||||
* <p>Package-private (fleetd #219) so {@link OpenCodeLauncher} can reuse the exact same
|
||||
* "different OS user, put it under worktreeRoot instead of java.io.tmpdir" resolution for its
|
||||
* own ephemeral {@code opencode.json} directory, rather than re-reading {@code config} a second
|
||||
* time with a second copy of this null/blank handling.
|
||||
*/
|
||||
Path memberScrubParentDir() {
|
||||
FleetConfig cfg = config == null ? null : config.get();
|
||||
if (cfg == null || cfg.worktreeRoot() == null || cfg.worktreeRoot().isBlank()) {
|
||||
return null;
|
||||
}
|
||||
return Path.of(cfg.worktreeRoot());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213: the configured {@code worktreeGroup:}, or {@code null} when unset/blank. Reuses
|
||||
* the group {@link dev.ltms.fleet.session.Worktrees#shareWithGroup} already establishes for
|
||||
* provisioned worktrees, rather than a second group key — see {@link
|
||||
* #applyEnvironmentAllowListPolicy}.
|
||||
*
|
||||
* <p>Package-private (fleetd #219) — reused by {@link OpenCodeLauncher} alongside {@link
|
||||
* #memberScrubParentDir()}; see that method's javadoc.
|
||||
*/
|
||||
String memberGroup() {
|
||||
FleetConfig cfg = config == null ? null : config.get();
|
||||
if (cfg == null || cfg.worktreeGroup() == null || cfg.worktreeGroup().isBlank()) {
|
||||
return null;
|
||||
}
|
||||
return cfg.worktreeGroup();
|
||||
}
|
||||
|
||||
/**
|
||||
* The full kept-name set for this spawn: the profile-derived names, unioned with {@code
|
||||
* memberCredentials.allow:} (CB-633 follow-up — previously ignored by this whole policy), the
|
||||
* ssh-agent handle when explicitly allowed, and the exact keys of THIS launch's own env map.
|
||||
*/
|
||||
private Set<String> derivedAllowedNames(FleetConfig.MemberCredentials creds, Launch launch) {
|
||||
Set<String> brokerUriEnvNames = brokerUriEnvNames();
|
||||
Set<String> allowed = new java.util.TreeSet<>(
|
||||
MemberEnvAllowList.derive(profiles.values(), creds.allowSet()));
|
||||
if (creds.sshAuthSockAllowed()) {
|
||||
MemberEnvAllowList.derive(profiles.values(), creds.allowSet(), brokerUriEnvNames));
|
||||
if (creds.sshAgentEnvInherited()) {
|
||||
allowed.add(SSH_AUTH_SOCK);
|
||||
} // blocked by default: absent from the set ⇒ blanked by the scrub like any other name
|
||||
} // omitted by default: absent from the set ⇒ blanked by the scrub like any other name
|
||||
allowed.addAll(launch.env().keySet());
|
||||
allowed.removeAll(brokerUriEnvNames);
|
||||
return allowed;
|
||||
}
|
||||
|
||||
private Set<String> brokerUriEnvNames() {
|
||||
return MemberEnvAllowList.brokerUriEnvNames(config == null ? null : config.get());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-633 follow-up: one INFO line per allow-list spawn WHOSE SCRUB ACTUALLY RUNS, so an operator
|
||||
* can read a single log line and know the scrub ran and how much of the visible environment it
|
||||
@@ -1108,10 +1542,27 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* survive {@code allowed} (including the {@code LC_*} prefix rule). Neither number is a constant:
|
||||
* both come from the actual derived set and the actual environment this spawn sees. Never logs a
|
||||
* variable NAME or VALUE — only the counts.
|
||||
*
|
||||
* <p><strong>Under {@code memberHerdrSocket} the counts describe fleetd's own process, not the
|
||||
* member's</strong> (fleetd #269 follow-up), so the message says so rather than leaving the
|
||||
* reader to infer it from this javadoc, which the operator reading the log never sees.
|
||||
*/
|
||||
private void logAllowListCoverage(Set<String> allowed) {
|
||||
Set<String> hostNames = hostEnvNames.get();
|
||||
long kept = hostNames.stream().filter(name -> MemberEnvAllowList.keeps(allowed, name)).count();
|
||||
if (memberHerdrSocketConfigured()) {
|
||||
// fleetd #269 covered the sibling line below (logCredentialGap) and stopped there.
|
||||
// This line has the same problem: read plainly, "allowed 7 of 39" is a statement about
|
||||
// the member's pane, and under memberHerdrSocket it is not -- the pane is routed to a
|
||||
// second herdr whose environment fleetd cannot inspect. The counts stay useful, so
|
||||
// this is not a WARN and not a refusal; only the claim is narrowed to what is true.
|
||||
log.info("member credentials: allowed {} of {} names in fleetd's OWN environment — "
|
||||
+ "memberHerdrSocket is configured, so member panes are routed to a "
|
||||
+ "second herdr whose environment fleetd has no channel to inspect. "
|
||||
+ "These counts describe fleetd's process, NOT the member pane's.",
|
||||
kept, hostNames.size());
|
||||
return;
|
||||
}
|
||||
log.info("member credentials: allowed {} of {}", kept, hostNames.size());
|
||||
}
|
||||
|
||||
@@ -1119,19 +1570,54 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
private static final String SSH_AUTH_SOCK = MemberEnvAllowList.SSH_AUTH_SOCK;
|
||||
|
||||
/**
|
||||
* CB-633: a non-zsh login shell means the allow-list control CANNOT run — say so once per
|
||||
* launcher instance, naming the shell, instead of failing silently.
|
||||
* fleetd #155: called ONLY from the {@code policy: deny-by-default} branch of {@link
|
||||
* #applyMemberCredentialPolicy} — under {@code policy: allow-list}, a non-zsh shell now REFUSES
|
||||
* the spawn instead (see {@link #applyEnvironmentAllowListPolicy}), since that policy's whole
|
||||
* point is a control a sourced file cannot undo. deny-by-default's own overlay does not depend
|
||||
* on the shell (it lands in the pane-creation env map before any shell runs), so this is a
|
||||
* heads-up, not a defect report: say once per launcher instance, naming the shell, that the
|
||||
* stronger allow-list control is unavailable here — never silently.
|
||||
*/
|
||||
private void warnNonZsh(String shell) {
|
||||
if (isZshShell(shell)) {
|
||||
return;
|
||||
}
|
||||
if (nonZshShellWarned.compareAndSet(false, true)) {
|
||||
log.warn("memberCredentials policy=allow-list: member login shell '{}' is NOT zsh — "
|
||||
+ "ZDOTDIR scrubbing cannot run, so members' inherited environment is "
|
||||
+ "UNPROTECTED beyond the enumerated known: fallback. Move herdr onto a "
|
||||
+ "zsh account or switch policy back to deny-by-default.",
|
||||
log.warn("memberCredentials policy=deny-by-default: member login shell '{}' is NOT zsh — "
|
||||
+ "the pane-creation credential overlay still applies here (it does not "
|
||||
+ "depend on the shell), but policy=allow-list's stronger post-shell "
|
||||
+ "ZDOTDIR scrub is unavailable on this shell.",
|
||||
shell == null ? "<unset>" : shell);
|
||||
}
|
||||
}
|
||||
|
||||
/** Guards {@link #warnCannotShareScrubDirectory} to one WARN per launcher instance. */
|
||||
private final AtomicBoolean cannotShareScrubDirWarned = new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* fleetd #213: {@code memberHerdrSocket} is configured and the member login shell IS zsh, but
|
||||
* {@code worktreeRoot} and/or {@code worktreeGroup} is missing, so the generated ZDOTDIR cannot
|
||||
* be placed anywhere fleetd can be sure the member's OS user can reach — {@code java.io.tmpdir}
|
||||
* is fleetd's own 0700 temp dir, which is unreadable if the member pane runs as a different OS
|
||||
* user, and fleetd has no channel to confirm whether it does or not. Rather than gamble on that,
|
||||
* this treats memberHerdrSocket as reason enough to require an explicitly shared location, which
|
||||
* is the exact gap this ticket exists to close. Say so once per launcher instance, instead of
|
||||
* either generating a directory that might not be readable (protection theatre) or refusing to
|
||||
* spawn (turning a degraded credential control into an outage for an opt-in feature).
|
||||
*/
|
||||
private void warnCannotShareScrubDirectory() {
|
||||
if (cannotShareScrubDirWarned.compareAndSet(false, true)) {
|
||||
log.warn("memberCredentials policy=allow-list: memberHerdrSocket is configured and the "
|
||||
+ "member login shell is zsh, but worktreeRoot and/or worktreeGroup is not "
|
||||
+ "configured — the generated ZDOTDIR cannot be placed where fleetd can be sure "
|
||||
+ "the member's OS user can read it (java.io.tmpdir is fleetd's own 0700 dir, "
|
||||
+ "unreadable if the member runs as a different OS user — fleetd has no channel "
|
||||
+ "to confirm whether it does), so the scrub cannot be guaranteed to run. Falling "
|
||||
+ "back to the CB-596 sentinel overlay. Configure both worktreeRoot and "
|
||||
+ "worktreeGroup to enable the allow-list scrub under memberHerdrSocket.");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-633 teardown half: read the pane's scrub report (the denominator report the generated
|
||||
* scrub wrote) and delete the directory. Called from {@link #stop}, which is the one funnel
|
||||
@@ -1177,18 +1663,139 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
private static final Pattern CREDENTIAL_SHAPED_NAME =
|
||||
Pattern.compile("(?i).*(TOKEN|SECRET|_KEY|APIKEY|PASSWORD|CREDENTIAL|AUTH).*");
|
||||
|
||||
/** Guards {@link #logCredentialGap} to one WARN per launcher instance, not one per spawn. */
|
||||
private final AtomicBoolean credentialGapLogged = new AtomicBoolean();
|
||||
/**
|
||||
* Guards the {@code effectiveAllowed == null} branch of {@link #logCredentialGap} — the
|
||||
* genuinely-unprotected report (deny-by-default, and the allow-list non-zsh fallback) — to one
|
||||
* WARN per launcher instance, not one per spawn.
|
||||
*
|
||||
* <p>CB-633 follow-up (#192): kept SEPARATE from {@link #allowListGapLogged} on purpose.
|
||||
* {@code memberCredentials} is a live, re-read-per-spawn supplier, so the policy can change
|
||||
* between two spawns on the same launcher. A single shared flag would let a harmless allow-list
|
||||
* INFO on spawn 1 permanently suppress the real deny-by-default WARN a later spawn deserves —
|
||||
* the report that matters most getting hidden by the report that doesn't. Two flags mean each
|
||||
* report kind fires exactly once, independent of what the other kind already logged.
|
||||
*/
|
||||
private final AtomicBoolean unprotectedGapLogged = new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* Guards the {@code effectiveAllowed != null} branch of {@link #logCredentialGap} — the
|
||||
* allow-list-scrub-covered report — to one INFO per launcher instance. See {@link
|
||||
* #unprotectedGapLogged}'s javadoc for why this is a separate flag rather than a shared one.
|
||||
*/
|
||||
private final AtomicBoolean allowListGapLogged = new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* fleetd #185 stage 2: guards {@link #warnUnknownMemberEnvironment} to one WARN per launcher
|
||||
* instance, not one per spawn — the same one-per-instance shape as {@link #unprotectedGapLogged}
|
||||
* and {@link #allowListGapLogged}, kept as its own flag for the same reason those two are split:
|
||||
* this mode is orthogonal to which of the other two branches would otherwise have fired.
|
||||
*/
|
||||
private final AtomicBoolean unknownMemberEnvironmentWarned = new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* fleetd #185 stage 2: whether {@code memberHerdrSocket:} is configured, i.e. member panes are
|
||||
* routed to a second herdr. This tests only that the config key is set — fleetd has no channel
|
||||
* to confirm what OS user that second herdr runs as, so a {@code true} result means "member
|
||||
* panes may run under a different OS user," not that they do. Re-read from the live config on
|
||||
* every call (same hot-reload shape as {@link #memberCredentials}), never cached, so a config
|
||||
* reload takes effect on the next spawn without a restart.
|
||||
*
|
||||
* <p>{@link #config} is {@code null} on any call site that never threaded the full config
|
||||
* through (every production {@code HerdrPeerLauncher} does; a handful of older tests do not) —
|
||||
* treated the same as "not configured", which is the correct, permissive default: it is exactly
|
||||
* today's single-daemon behaviour.
|
||||
*
|
||||
* <p>Package-private (fleetd #219) — {@link OpenCodeLauncher} reuses this same gate to decide
|
||||
* where its own ephemeral {@code opencode.json} directory (site 1) and its opencode session
|
||||
* discovery (site 2) may run, rather than re-deriving "is this a multi-uid fleet" a second way.
|
||||
*/
|
||||
boolean memberHerdrSocketConfigured() {
|
||||
if (config == null) {
|
||||
return false;
|
||||
}
|
||||
FleetConfig cfg = config.get();
|
||||
return cfg != null && cfg.memberHerdrSocket() != null && !cfg.memberHerdrSocket().isBlank();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #185 stage 2: the single replacement WARN for {@link #logCredentialGap}'s usual
|
||||
* conclusions when {@code memberHerdrSocket:} is configured. {@link #hostEnvNames} (and
|
||||
* everything derived from it — {@code known}/{@code allow} coverage, the allow-list scrub's
|
||||
* derived set) describes the DAEMON's own environment; under this config key member panes are
|
||||
* routed to a second herdr, and fleetd has no channel to confirm what OS user that herdr runs
|
||||
* as or to read its environment, so neither "every member pane inherits them UNBLOCKED" nor
|
||||
* "the scrub blanks them" is evidence-backed here — both would be reporting on the wrong
|
||||
* process. Logged once, names the config key, and states the honest conclusion: the gap for
|
||||
* member panes is UNKNOWN, not clean, so {@code memberCredentials} cannot be verified from this
|
||||
* daemon. The one count it does report is scoped explicitly to fleetd's own environment, never
|
||||
* presented as if it said anything about the member's — see {@link #logCredentialGap}'s javadoc
|
||||
* for why this branch exists.
|
||||
*/
|
||||
private void warnUnknownMemberEnvironment(FleetConfig.MemberCredentials creds) {
|
||||
if (!unknownMemberEnvironmentWarned.compareAndSet(false, true)) {
|
||||
return;
|
||||
}
|
||||
Set<String> covered = new HashSet<>(creds.known());
|
||||
covered.addAll(creds.allow());
|
||||
Set<String> hostNames = hostEnvNames.get();
|
||||
long gapInFleetdsOwnEnv = hostNames.stream()
|
||||
.filter(name -> CREDENTIAL_SHAPED_NAME.matcher(name).matches())
|
||||
.filter(name -> !covered.contains(name))
|
||||
.count();
|
||||
log.warn("memberCredentials gap: memberHerdrSocket is configured, so member panes are routed "
|
||||
+ "to a second herdr — fleetd has no channel to confirm what OS user that herdr "
|
||||
+ "runs as, so it cannot tell whether those panes inherit its own environment or "
|
||||
+ "a different one entirely. {} of the {} names in fleetd's OWN environment are "
|
||||
+ "credential-shaped and not on known:/allow:, but that count describes fleetd's "
|
||||
+ "process, not the member herdr's. The credential gap for member panes is "
|
||||
+ "UNKNOWN, not clean, and memberCredentials cannot be verified from here.",
|
||||
gapInFleetdsOwnEnv, hostNames.size());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-596 criterion 4: a credential-shaped host env var name on neither {@code known} nor
|
||||
* {@code allow} is not silently allowed — it is reported. {@link #hostEnvNames} enumerates the
|
||||
* daemon's own environment (see that field's javadoc for why the daemon's env is read rather
|
||||
* than the spawned pane's, which the daemon has no channel to inspect at spawn time); this logs
|
||||
* every such NAME, at WARN, at most once per launcher instance — never a value, a prefix of a
|
||||
* value, or a hash of a value, so the log itself cannot leak anything.
|
||||
* every such NAME — never a value, a prefix of a value, or a hash of a value, so the log itself
|
||||
* cannot leak anything.
|
||||
*
|
||||
* <p>CB-633 follow-up (#192): {@code effectiveAllowed} picks the wording, and it must NOT be
|
||||
* picked from {@code creds.isAllowList()} — see {@link #applyEnvironmentAllowListPolicy}'s
|
||||
* javadoc for the reasoning this mirrors. {@code null} means no scrub-derived allow-list was
|
||||
* computed for this call — true on the deny-by-default path AND on the allow-list non-zsh
|
||||
* fallback, where nothing is ever scrubbed — so the whole gap is real and gets the WARN,
|
||||
* unchanged from before this fix. Non-null means this call came from the zsh branch of {@link
|
||||
* #applyEnvironmentAllowListPolicy}, reachable ONLY after that method's own zsh gate — but
|
||||
* {@code effectiveAllowed} is a SUPERSET of {@code known ∪ allow}: {@link MemberEnvAllowList#derive}
|
||||
* also unions in every profile's {@code gitTokenEnv}/{@code gitHostEnv}/{@code tokenEnv}/
|
||||
* {@code env:} keys, and {@link #derivedAllowedNames} further unions in this very spawn's own
|
||||
* env keys — so a name can be in the gap (uncovered by {@code known}/{@code allow}) AND still be
|
||||
* kept by the derived allow-list, in which case the scrub does NOT blank it and the member DOES
|
||||
* inherit it. Lead review on #192 caught this: the first cut of this fix reported the WHOLE gap
|
||||
* as scrub-blanked without checking that, which reported a real leak as safe — the exact
|
||||
* inversion #192 exists to remove. So on this path the gap is split with {@link
|
||||
* MemberEnvAllowList#keeps}, the SAME predicate the generated scrub itself evaluates, so this
|
||||
* split cannot drift from what the scrub actually does: the names it says are kept get the WARN
|
||||
* (same severity, and same guard, as the deny-by-default case — a name genuinely reaching a
|
||||
* member unprotected is equally serious whichever path put it there), and the names it says are
|
||||
* blanked keep the INFO.
|
||||
*
|
||||
* <p>fleetd #185 stage 2: everything above assumes the member pane runs under the same OS user
|
||||
* as the daemon, so {@link #hostEnvNames} mirrors what the pane inherits — see that field's
|
||||
* javadoc. When {@code memberHerdrSocket:} is configured that assumption is false: the member
|
||||
* pane runs on a second herdr owned by a <em>different</em> user, and neither conclusion below
|
||||
* ("inherits them UNBLOCKED" / "the scrub blanks them") is backed by evidence about that user's
|
||||
* environment. So this method checks that first and, when configured, reports the honest
|
||||
* "unknown, not clean" conclusion instead — see {@link #warnUnknownMemberEnvironment}. When
|
||||
* {@code memberHerdrSocket:} is absent (the default, and the only mode this host runs) this
|
||||
* branch is never taken and every line below is unchanged.
|
||||
*/
|
||||
private void logCredentialGap(FleetConfig.MemberCredentials creds) {
|
||||
private void logCredentialGap(FleetConfig.MemberCredentials creds, Set<String> effectiveAllowed) {
|
||||
if (memberHerdrSocketConfigured()) {
|
||||
warnUnknownMemberEnvironment(creds);
|
||||
return;
|
||||
}
|
||||
Set<String> covered = new HashSet<>(creds.known());
|
||||
covered.addAll(creds.allow());
|
||||
List<String> gap = hostEnvNames.get().stream()
|
||||
@@ -1199,7 +1806,37 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
if (gap.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
if (credentialGapLogged.compareAndSet(false, true)) {
|
||||
if (effectiveAllowed == null) {
|
||||
warnGapUnprotected(gap);
|
||||
return;
|
||||
}
|
||||
List<String> keptByDerivedList = gap.stream()
|
||||
.filter(name -> MemberEnvAllowList.keeps(effectiveAllowed, name))
|
||||
.toList();
|
||||
List<String> blankedByScrub = gap.stream()
|
||||
.filter(name -> !MemberEnvAllowList.keeps(effectiveAllowed, name))
|
||||
.toList();
|
||||
if (!keptByDerivedList.isEmpty() && unprotectedGapLogged.compareAndSet(false, true)) {
|
||||
log.warn("memberCredentials gap: {} credential-shaped env var name(s) are on neither "
|
||||
+ "known: nor allow: — the derived allow-list keeps them anyway (a profile's "
|
||||
+ "gitTokenEnv/gitHostEnv/tokenEnv/env: names one, or this spawn injects it), "
|
||||
+ "so every member pane inherits them UNBLOCKED — {}. Add each to "
|
||||
+ "memberCredentials.known (or .allow if a member legitimately needs it), or "
|
||||
+ "remove it from whatever profile setting derives it in.",
|
||||
keptByDerivedList.size(), keptByDerivedList);
|
||||
}
|
||||
if (!blankedByScrub.isEmpty() && allowListGapLogged.compareAndSet(false, true)) {
|
||||
log.info("memberCredentials gap: {} credential-shaped env var name(s) are on neither "
|
||||
+ "known: nor allow: — {}. The allow-list scrub blanks them anyway (they "
|
||||
+ "are not on the derived allow-list), so no member pane keeps them; add "
|
||||
+ "each to memberCredentials.known or .allow to make that explicit.",
|
||||
blankedByScrub.size(), blankedByScrub);
|
||||
}
|
||||
}
|
||||
|
||||
/** The deny-by-default (and allow-list non-zsh fallback) WARN — unchanged byte-for-byte by #192. */
|
||||
private void warnGapUnprotected(List<String> gap) {
|
||||
if (unprotectedGapLogged.compareAndSet(false, true)) {
|
||||
log.warn("memberCredentials gap: {} credential-shaped env var name(s) are on neither "
|
||||
+ "known: nor allow: — every member pane inherits them UNBLOCKED — {}. "
|
||||
+ "Add each to memberCredentials.known (blocked by default) or .allow "
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* fleetd #111 (CB-608): a single, testable read of the {@code memberCredentials:} policy — names
|
||||
* and counts only, never a value. The daemon never holds a credential's <em>value</em> in the
|
||||
* first place (only the names configured under {@code known:}/{@code allow:}), so there is
|
||||
* nothing here to redact by construction; the point of this class is that it is the ONE place
|
||||
* that turns a policy into names-and-counts, so nothing else hand-counts a second time.
|
||||
*
|
||||
* <p>Before this class, {@link dev.ltms.fleet.Fleetd#reportMemberCredentialsGap} computed these
|
||||
* same counts inline for the startup log line, and {@code scripts/probe-member-credentials.sh}
|
||||
* carried its own hardcoded {@code NAMES} array that the live policy could grow past silently
|
||||
* (#111) — the exact "hand-maintained second copy drifts" shape #114 fixed for the tool
|
||||
* catalogue. Both now read this class: the startup log via {@link
|
||||
* dev.ltms.fleet.Fleetd#reportMemberCredentialsGap}, and a live daemon via the {@code
|
||||
* GET /member-credentials} REST endpoint ({@link dev.ltms.fleet.rest.FleetApp}), which the probe
|
||||
* script fetches instead of carrying its own list.
|
||||
*
|
||||
* @param present policy configured with at least one {@code known} name. {@code false} for an
|
||||
* absent or empty {@code memberCredentials:} block — represented honestly as "no
|
||||
* policy", never as "nothing blocked" (an empty {@link #blocked} could otherwise be
|
||||
* misread as a clean bill of health).
|
||||
* @param policy the normalized policy mode ({@link FleetConfig.MemberCredentials#policy()}), or
|
||||
* {@code null} when {@link #present} is {@code false}.
|
||||
* @param known every name the policy declares, in configured order. Names only, never a value.
|
||||
* @param allowed the subset of {@link #known} explicitly let through. Names only.
|
||||
* @param blocked {@link #known} minus {@link #allowed} — the names an actual spawn shadows. Names
|
||||
* only.
|
||||
*/
|
||||
public record MemberCredentialPolicyView(boolean present, String policy, List<String> known,
|
||||
List<String> allowed, List<String> blocked) {
|
||||
|
||||
private static final MemberCredentialPolicyView ABSENT =
|
||||
new MemberCredentialPolicyView(false, null, List.of(), List.of(), List.of());
|
||||
|
||||
public MemberCredentialPolicyView {
|
||||
known = known == null ? List.of() : List.copyOf(known);
|
||||
allowed = allowed == null ? List.of() : List.copyOf(allowed);
|
||||
blocked = blocked == null ? List.of() : List.copyOf(blocked);
|
||||
}
|
||||
|
||||
/** The honest "no policy configured" view. */
|
||||
public static MemberCredentialPolicyView absent() {
|
||||
return ABSENT;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the view straight from the live config. {@code creds} may be {@code null} (no {@code
|
||||
* memberCredentials:} block at all) — treated the same as a present-but-empty block, exactly
|
||||
* like {@link dev.ltms.fleet.Fleetd#reportMemberCredentialsGap} already did.
|
||||
*/
|
||||
public static MemberCredentialPolicyView of(FleetConfig.MemberCredentials creds) {
|
||||
if (creds == null || creds.known().isEmpty()) {
|
||||
return ABSENT;
|
||||
}
|
||||
return new MemberCredentialPolicyView(true, creds.policy(), creds.known(), creds.allow(),
|
||||
List.copyOf(creds.blockedSet()));
|
||||
}
|
||||
|
||||
public int knownCount() {
|
||||
return known.size();
|
||||
}
|
||||
|
||||
public int allowedCount() {
|
||||
return allowed.size();
|
||||
}
|
||||
|
||||
public int blockedCount() {
|
||||
return blocked.size();
|
||||
}
|
||||
}
|
||||
@@ -34,14 +34,15 @@ import java.util.TreeSet;
|
||||
*
|
||||
* <p>{@code SSH_AUTH_SOCK} is deliberately NOT here. It is a handle to the operator's ssh-agent — a
|
||||
* member holding it can sign with the operator's keys — so keeping it is a config decision
|
||||
* ({@code memberCredentials.sshAuthSock: allow}), not a derivation default.
|
||||
* ({@code memberCredentials.sshAgentEnv: inherit}), not a derivation default.
|
||||
*
|
||||
* <p><b>CB-633 follow-up:</b> the union also includes {@code memberCredentials.allow:} — the
|
||||
* operator's own explicit list. Before this, {@code policy: allow-list} silently ignored every name
|
||||
* an operator wrote under {@code allow:} unless a profile happened to carry it too, which meant
|
||||
* turning the policy on could blank credentials working members already depended on. {@code
|
||||
* SSH_AUTH_SOCK} is the one exception: even when the operator lists it under {@code allow:}, it is
|
||||
* excluded here and added back ONLY by the caller when {@code sshAuthSock: allow} is explicitly set
|
||||
* SSH_AUTH_SOCK} and configured broker URI environment names are exceptions: even when the operator
|
||||
* lists them under {@code allow:}, they are excluded here. {@code SSH_AUTH_SOCK} is added back ONLY
|
||||
* by the caller when {@code sshAgentEnv: inherit} is explicitly set
|
||||
* (see {@link #SSH_AUTH_SOCK}'s javadoc) — it is a live handle to the operator's own ssh-agent, not
|
||||
* a value, so treating it like any other allow-listed name would hand a member every key the
|
||||
* operator's agent holds the moment they typed the name under {@code allow:} for an unrelated
|
||||
@@ -52,7 +53,7 @@ public final class MemberEnvAllowList {
|
||||
/**
|
||||
* The operator's ssh-agent socket path. Deliberately excluded from {@link #derive}'s union of
|
||||
* {@code memberCredentials.allow:} — see the class javadoc's CB-633 follow-up note. Governed
|
||||
* ONLY by {@code memberCredentials.sshAuthSock}, never by appearing in {@code allow:}.
|
||||
* ONLY by {@code memberCredentials.sshAgentEnv}, never by appearing in {@code allow:}.
|
||||
*/
|
||||
public static final String SSH_AUTH_SOCK = "SSH_AUTH_SOCK";
|
||||
|
||||
@@ -106,6 +107,16 @@ public final class MemberEnvAllowList {
|
||||
* run-to-run.
|
||||
*/
|
||||
public static Set<String> derive(Collection<FleetConfig.Profile> profiles, Set<String> configuredAllow) {
|
||||
return derive(profiles, configuredAllow, Set.of());
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #derive(Collection, Set)}, while excluding names that fleetd knows carry credentials.
|
||||
* A configured broker URI contains its AMQP password inline, so it must never reach a member,
|
||||
* even when an operator put its variable name in {@code memberCredentials.allow:}.
|
||||
*/
|
||||
public static Set<String> derive(Collection<FleetConfig.Profile> profiles, Set<String> configuredAllow,
|
||||
Set<String> excludedNames) {
|
||||
Set<String> derived = new TreeSet<>(INFRASTRUCTURE_PASSTHROUGH);
|
||||
if (profiles != null) {
|
||||
for (FleetConfig.Profile p : profiles) {
|
||||
@@ -124,9 +135,37 @@ public final class MemberEnvAllowList {
|
||||
}
|
||||
}
|
||||
}
|
||||
if (excludedNames != null) {
|
||||
derived.removeAll(excludedNames);
|
||||
}
|
||||
return Set.copyOf(derived);
|
||||
}
|
||||
|
||||
/**
|
||||
* The host environment names whose values are AMQP URIs with inline passwords. Both broker
|
||||
* connections belong to fleetd, never to a member pane. Blank and absent configuration changes
|
||||
* nothing.
|
||||
*
|
||||
* <p><b>How strong this exclusion is depends on the policy, and the difference matters.</b>
|
||||
* Under {@code policy: allow-list} it is enforced by the generated ZDOTDIR scrub, which runs
|
||||
* AFTER the pane's shell has sourced the operator's chain — so a login shell that re-exports the
|
||||
* name is still blanked. Under the deny-list policy there is no scrub: the name is only removed
|
||||
* from the pre-shell env map, and a login shell that sources the operator's secret store
|
||||
* re-exports it. That is the long-standing weakness of deny-list (a sourced file can undo it),
|
||||
* not something this exclusion introduces, but it means deny-list deployments do NOT get this
|
||||
* guarantee. The same caveat applies to the non-zsh path, which has no scrub at all — see
|
||||
* {@code HerdrPeerLauncher#applyEnvironmentAllowListPolicy}.
|
||||
*/
|
||||
public static Set<String> brokerUriEnvNames(FleetConfig config) {
|
||||
if (config == null) {
|
||||
return Set.of();
|
||||
}
|
||||
Set<String> names = new TreeSet<>();
|
||||
addUriEnvIfPresent(names, config.broker());
|
||||
addUriEnvIfPresent(names, config.coordinator());
|
||||
return Set.copyOf(names);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code name} survives the scrub when {@code allowedNames} is the derived set: an exact
|
||||
* match, or an infrastructure-prefixed name ({@code LC_*}). Prefix rules live ONLY here and in
|
||||
@@ -146,4 +185,16 @@ public final class MemberEnvAllowList {
|
||||
into.add(name);
|
||||
}
|
||||
}
|
||||
|
||||
private static void addUriEnvIfPresent(Set<String> into, FleetConfig.Broker broker) {
|
||||
if (broker != null && broker.hasUriEnv()) {
|
||||
into.add(broker.uriEnv());
|
||||
}
|
||||
}
|
||||
|
||||
private static void addUriEnvIfPresent(Set<String> into, FleetConfig.Coordinator coordinator) {
|
||||
if (coordinator != null && coordinator.hasUriEnv()) {
|
||||
into.add(coordinator.uriEnv());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.CharterReceipt;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
@@ -20,6 +21,10 @@ import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.BooleanSupplier;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
@@ -71,6 +76,19 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
*/
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
|
||||
/**
|
||||
* fleetd #175: notified when {@link SessionAwareHandle} detects, on the same late-resolve
|
||||
* read that discovers the session id, that the live opencode session is running a DIFFERENT
|
||||
* model than the profile requested — opencode does not fail on an unknown {@code -m}, it
|
||||
* silently falls back to a default (potentially paid) model. Reused exactly as
|
||||
* {@code CompletionResolver}'s {@code BACKEND_EXHAUSTED} path uses it: this launcher supplies
|
||||
* only the herdr terminal id and a reason string; mapping target → session → profile →
|
||||
* credential stays entirely the sink's job (see {@code Fleetd.main}'s wiring). Defaults to
|
||||
* {@link ExhaustionSink#none()} for a caller (an older constructor, or a test not exercising
|
||||
* this) that opts out.
|
||||
*/
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so it
|
||||
* matches the legacy non-blocking spawn semantics. Config dirs are created under the JVM temp dir.
|
||||
@@ -116,12 +134,39 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs, spawnReadyPollMs,
|
||||
fleet, memberCredentials, null);
|
||||
}
|
||||
|
||||
/** Production constructor, plus the live config for URI environment exclusions. */
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs, spawnReadyPollMs,
|
||||
fleet, memberCredentials, config, ExhaustionSink.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor, plus the live config for URI environment exclusions and the fleetd
|
||||
* #175 model-mismatch quarantine sink.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet, memberCredentials);
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet, memberCredentials, config,
|
||||
exhaustionSink);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -169,6 +214,7 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
this.exhaustionSink = ExhaustionSink.none();
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -182,17 +228,84 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
Path configRoot, Path discoveryRoot,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs, nowMillis, sleeper,
|
||||
configRoot, discoveryRoot, fleet, memberCredentials, null);
|
||||
}
|
||||
|
||||
/** Full testability constructor, plus the live config for URI environment exclusions. */
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper, Path configRoot, Path discoveryRoot,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs, nowMillis, sleeper,
|
||||
configRoot, discoveryRoot, fleet, memberCredentials, config, ExhaustionSink.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the live config for URI environment exclusions and the
|
||||
* fleetd #175 model-mismatch quarantine sink — the constructor a test drives directly to
|
||||
* observe {@link ExhaustionSink#onExhausted} without going through {@code Fleetd.main}'s
|
||||
* wiring.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper, Path configRoot, Path discoveryRoot,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials);
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials, null, config);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
this.exhaustionSink = exhaustionSink == null ? ExhaustionSink.none() : exhaustionSink;
|
||||
}
|
||||
|
||||
private static Path defaultConfigRoot() {
|
||||
return Path.of(System.getProperty("java.io.tmpdir"));
|
||||
}
|
||||
|
||||
/** The default opencode storage root: {@code ~/.local/share/opencode} (the XDG data dir). */
|
||||
/**
|
||||
* The default opencode storage root: {@code ~/.local/share/opencode} (the XDG data dir) —
|
||||
* always FLEETD's OWN {@code user.home}, whichever OS user runs the daemon.
|
||||
*
|
||||
* <p><b>fleetd #219 site 2 — a decision, not a patch.</b> Under {@code memberHerdrSocket:} the
|
||||
* member pane runs as a <em>different</em> OS user, and opencode writes {@code opencode.db}
|
||||
* under <em>that</em> user's {@code $HOME}, not fleetd's. Scanning fleetd's own {@code
|
||||
* user.home} is therefore looking in the wrong place — a wrong-LOCATION failure, not a
|
||||
* wrong-PERMISSION one like site 1, and it fails quietly: {@link
|
||||
* SessionAwareHandle#agentSessionId()} would keep returning {@code null} forever, which reads
|
||||
* as "opencode does not support resume" rather than "fleetd looked in the wrong home." fleetd
|
||||
* #209 is the reason that silence is unacceptable.
|
||||
*
|
||||
* <p>Three ways to close the gap were weighed:
|
||||
* <ol>
|
||||
* <li><b>Make the member's home configurable.</b> Correct in principle, but this ticket's
|
||||
* scope is the two existing call sites, not a new config key — {@code memberHerdrSocket}
|
||||
* already carries the second herdr's socket path, not its user's home, and inventing a
|
||||
* parallel key here without also wiring it through discovery's actual callers is a
|
||||
* half-shipped feature (the exact shape CB-596/CB-611 warn against).</li>
|
||||
* <li><b>Derive it</b> (e.g. from {@code worktreeRoot}'s owner, or {@code getent passwd}).
|
||||
* Rejected: nothing in this codebase resolves a Unix username to a home directory today,
|
||||
* and guessing wrong would silently point discovery at a THIRD wrong location — worse
|
||||
* than the current gap, because it would look like it should work.</li>
|
||||
* <li><b>Declare discovery unavailable</b> under {@code memberHerdrSocket}, and say so once,
|
||||
* loudly, instead of scanning a directory that structurally cannot hold the answer.</li>
|
||||
* </ol>
|
||||
*
|
||||
* <p>Option 3 is taken — the one this ticket says to default to when unsure. {@link
|
||||
* OpenCodeLauncher#spawn} routes {@link SessionAwareHandle#agentSessionId()} through {@link
|
||||
* HerdrPeerLauncher#memberHerdrSocketConfigured()} before ever calling {@link
|
||||
* OpenCodeSessionDiscovery#sessionIdForDirectory}, so under {@code memberHerdrSocket} the
|
||||
* database at this root is never even opened, and one WARN per launcher instance names the gap
|
||||
* instead of the {@code null} return reading as "unsupported." Capability advertising is
|
||||
* unaffected: {@link #capabilities()} always includes {@code SESSION_RESUME}, since {@code
|
||||
* memberHerdrSocket} absent (today's only live mode) is unchanged by this decision.
|
||||
*/
|
||||
private static Path defaultDiscoveryRoot() {
|
||||
return Path.of(System.getProperty("user.home"), ".local", "share", "opencode");
|
||||
}
|
||||
@@ -308,7 +421,7 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
*/
|
||||
private Path writeConfig(FleetConfig.Profile cfg, String charterText, String cwd) {
|
||||
try {
|
||||
Path dir = Files.createTempDirectory(configRoot, "fleetd-opencode-");
|
||||
Path dir = Files.createTempDirectory(configParentDir(), "fleetd-opencode-");
|
||||
dir.toFile().deleteOnExit();
|
||||
|
||||
ObjectNode root = JSON.createObjectNode();
|
||||
@@ -379,6 +492,14 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
// carries operator-supplied values (URL, model id, api key), so escaping must be real.
|
||||
Files.writeString(cfgFile, JSON.writerWithDefaultPrettyPrinter().writeValueAsString(root));
|
||||
cfgFile.toFile().deleteOnExit();
|
||||
if (memberHerdrSocketConfigured()) {
|
||||
// fleetd #219: the same "different OS user" gap fleetd #213 closed for the ZDOTDIR
|
||||
// scrub — share read-only with worktreeGroup rather than leaving the directory under
|
||||
// fleetd's own 0700 java.io.tmpdir, where the member's OS user could not even
|
||||
// traverse it. memberGroup() cannot be null here: configParentDir() above already
|
||||
// refused this spawn if either worktreeRoot or worktreeGroup was missing.
|
||||
EnvAllowListScrub.shareWithGroup(dir, memberGroup());
|
||||
}
|
||||
return cfgFile;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException(
|
||||
@@ -386,6 +507,57 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #219 site 1: where {@link #writeConfig} creates its per-spawn directory.
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@code memberHerdrSocket} ABSENT (today's only mode): byte-identical to before this
|
||||
* fix — always {@link #configRoot} (defaults to {@code java.io.tmpdir}, fleetd's own
|
||||
* process).</li>
|
||||
* <li>{@code memberHerdrSocket} PRESENT: {@code java.io.tmpdir} is fleetd's own per-user temp
|
||||
* dir (mode {@code 0700} on macOS) — the member pane runs as a DIFFERENT OS user under
|
||||
* this config key and cannot even traverse it, so the directory holding {@code
|
||||
* opencode.json} (which tells the member where the bridge MCP is) and the member charter
|
||||
* would be unreadable to the very process it is written for. The directory instead goes
|
||||
* under {@code worktreeRoot}, shared read-only with {@code worktreeGroup} via {@link
|
||||
* EnvAllowListScrub#shareWithGroup} — the SAME mechanism fleetd #213 built for the ZDOTDIR
|
||||
* scrub, reused here rather than duplicated (see {@link
|
||||
* HerdrPeerLauncher#memberScrubParentDir()}).</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><b>Unlike the ZDOTDIR scrub, a missing {@code worktreeRoot}/{@code worktreeGroup} here
|
||||
* REFUSES the spawn instead of degrading.</b> The ZDOTDIR scrub is a credential CONTROL: a
|
||||
* degraded control (CB-596's sentinel overlay) is still worth having. This config file is not a
|
||||
* control — it is the ONLY way the member learns where the bridge MCP lives. Writing it
|
||||
* somewhere the member cannot read would not degrade anything; it would spawn a member that
|
||||
* occupies a pane and never becomes deliverable, since {@code fleet_send} waits ~60s on the
|
||||
* readiness gate and then fails with nothing pointing at a temp directory as the cause. Refusing
|
||||
* up front, with a message that names the missing config key, is the honest failure — an
|
||||
* undeliverable member is not a working spawn either way, so nothing is lost by refusing loudly
|
||||
* instead of failing silently later.
|
||||
*
|
||||
* @throws IllegalStateException when {@code memberHerdrSocket} is configured but {@code
|
||||
* worktreeRoot} and/or {@code worktreeGroup} is not
|
||||
*/
|
||||
private Path configParentDir() {
|
||||
if (!memberHerdrSocketConfigured()) {
|
||||
return configRoot;
|
||||
}
|
||||
Path root = memberScrubParentDir();
|
||||
String group = memberGroup();
|
||||
if (root == null || group == null) {
|
||||
throw new IllegalStateException("memberHerdrSocket is configured, so opencode's config "
|
||||
+ "directory (opencode.json + member charter) must be placed where the member's "
|
||||
+ "OS user can read it — worktreeRoot, shared via worktreeGroup — but "
|
||||
+ (root == null ? "worktreeRoot" : "worktreeGroup") + " is not configured. "
|
||||
+ "Refusing to spawn rather than write a config the member cannot read: that "
|
||||
+ "member would occupy a pane and never become deliverable, with nothing "
|
||||
+ "pointing at the real cause. Configure both worktreeRoot and worktreeGroup to "
|
||||
+ "enable opencode member spawns under memberHerdrSocket.");
|
||||
}
|
||||
return root;
|
||||
}
|
||||
|
||||
/**
|
||||
* Declare a custom OpenAI-compatible provider so the worker talks to a pinned endpoint (a local
|
||||
* vLLM, say) instead of opencode's default gateway (CB-508).
|
||||
@@ -494,11 +666,48 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
return afterScheme.contains("/") ? trimmed : trimmed + "/v1";
|
||||
}
|
||||
|
||||
/** One WARN per launcher instance for the fleetd #219 site-2 discovery-unavailable gap. */
|
||||
private final AtomicBoolean discoveryUnavailableWarned =
|
||||
new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* fleetd #267: one WARN per PROFILE (not per launcher instance — several profiles can each hit
|
||||
* this gap independently) for the model-mismatch check (fleetd #175) never getting to run
|
||||
* because the spawn was not given a fleetd-provisioned worktree (fleetd #249). Profile names
|
||||
* accumulate here for the life of this launcher instance and are never removed — the same
|
||||
* one-shot treatment {@link #discoveryUnavailableWarned} already gets, just keyed per profile
|
||||
* instead of globally.
|
||||
*/
|
||||
private final Set<String> modelCheckSkippedWarned = ConcurrentHashMap.newKeySet();
|
||||
|
||||
/** Add lazy on-disk session discovery to the base handle. */
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
String cwd = effectiveCwd(req);
|
||||
// fleetd #249: refuse rather than silently resume into unverifiable territory. opencode's
|
||||
// `-s <id>` flag itself resumes precisely — the resolved id is what fails, not the resume —
|
||||
// but resolvedSessionId() below can never confirm (or later re-report) this handle's own
|
||||
// identity without a fleetd-provisioned worktree (isProvisionedWorktree(cwd)), because the
|
||||
// directory is shared and sessionIdForDirectory's "most recently updated row" heuristic can
|
||||
// pick a sibling's session. Refusing here, before anything spawns, beats letting the member
|
||||
// start and only then discovering fleetd can never again verify who it actually is.
|
||||
if (req.resumeSessionId() != null && !req.resumeSessionId().isBlank()
|
||||
&& !isProvisionedWorktree(cwd)) {
|
||||
throw new IllegalArgumentException("resumeSessionId requires a fleetd-provisioned "
|
||||
+ "worktree for an opencode profile — without one, this member's cwd is shared "
|
||||
+ "with other sessions, so fleetd can never reliably confirm (now or later) which "
|
||||
+ "conversation it is actually running (fleetd #249). Pass fleet_spawn{worktree:"
|
||||
+ "<ticket-slug>} to resume this member.");
|
||||
}
|
||||
PeerHandle inner = super.spawn(req);
|
||||
return new SessionAwareHandle(inner, discovery, effectiveCwd(req));
|
||||
// fleetd #175: the same profile config buildLaunch resolved for this spawn (requireProfile
|
||||
// is deterministic on req.profileName(), so re-resolving here costs a map lookup, not a
|
||||
// second decision) — SessionAwareHandle needs cfg.model() to know what THIS session should
|
||||
// be running.
|
||||
FleetConfig.Profile cfg = requireProfile(req.profileName());
|
||||
return new SessionAwareHandle(inner, discovery, cwd, cfg,
|
||||
this::memberHerdrSocketConfigured, discoveryUnavailableWarned,
|
||||
modelCheckSkippedWarned, exhaustionSink);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -508,16 +717,63 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
* are untouched — only the opencode-specific identity answer is added. {@code sessionName()}
|
||||
* stays null: opencode has no display-name seam, so the logical name lives only in the bridge's
|
||||
* roster (see the SESSION_NAME capability).
|
||||
*
|
||||
* <p>fleetd #175: {@link #agentSessionId()} is also where the model-mismatch check lives (see
|
||||
* {@link #checkModelMatch()}) — it is the one method the real late-resolve path
|
||||
* ({@code SessionManager.resolveAgentSessionId}, via {@code get}/{@code rosterResolved}/
|
||||
* {@code release}) actually calls, and only while the session id is still unknown. Putting the
|
||||
* check anywhere else risks repeating PR #203's mistake: a check that runs before opencode has
|
||||
* written the row it needs, and so never fires.
|
||||
*/
|
||||
private static final class SessionAwareHandle implements PeerHandle {
|
||||
private final PeerHandle delegate;
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
private final String cwd;
|
||||
private final FleetConfig.Profile cfg;
|
||||
private final BooleanSupplier discoveryUnavailable;
|
||||
private final AtomicBoolean discoveryUnavailableWarned;
|
||||
private final Set<String> modelCheckSkippedWarned;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
/** CAS'd true the first (and only) time a model mismatch is reported for this handle. */
|
||||
private final AtomicBoolean modelMismatchReported = new AtomicBoolean();
|
||||
/**
|
||||
* The session id, once {@link OpenCodeSessionDiscovery#sessionIdForDirectory} first
|
||||
* resolves a non-null answer for this handle (fleetd #234). Sticky on purpose: {@code
|
||||
* directory} is a shared-cwd heuristic (see {@link OpenCodeSessionDiscovery}'s class
|
||||
* javadoc) that can start returning a DIFFERENT row once another session shares the same
|
||||
* directory and writes a newer one — re-deriving it on every call would let this handle's
|
||||
* identity silently drift to a sibling's session. Once resolved, this IS the answer, and
|
||||
* {@link #checkModelMatch} reads only the row this id names, never "whatever is newest in
|
||||
* the directory right now."
|
||||
*/
|
||||
private final AtomicReference<String> resolvedSessionId = new AtomicReference<>();
|
||||
/**
|
||||
* fleetd #249: whether {@link #cwd} is a fleetd-provisioned git worktree
|
||||
* ({@link HerdrPeerLauncher#isProvisionedWorktree}), computed once at spawn time since
|
||||
* {@code cwd} never changes for this handle. When {@code false} the directory is shared
|
||||
* with other sessions (the default no-worktree spawn inherits the lead's own cwd), so
|
||||
* {@link OpenCodeSessionDiscovery#sessionIdForDirectory}'s "most recently updated row for
|
||||
* this directory" heuristic can and does pick another session's row — see that class's
|
||||
* javadoc. {@link #agentSessionId()} refuses to guess in that case: it reports absent
|
||||
* rather than a possibly-foreign id.
|
||||
*/
|
||||
private final boolean worktreeProvisioned;
|
||||
|
||||
SessionAwareHandle(PeerHandle delegate, OpenCodeSessionDiscovery discovery, String cwd) {
|
||||
SessionAwareHandle(PeerHandle delegate, OpenCodeSessionDiscovery discovery, String cwd,
|
||||
FleetConfig.Profile cfg,
|
||||
BooleanSupplier discoveryUnavailable,
|
||||
AtomicBoolean discoveryUnavailableWarned,
|
||||
Set<String> modelCheckSkippedWarned,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this.delegate = delegate;
|
||||
this.discovery = discovery;
|
||||
this.cwd = cwd;
|
||||
this.cfg = cfg;
|
||||
this.discoveryUnavailable = discoveryUnavailable;
|
||||
this.discoveryUnavailableWarned = discoveryUnavailableWarned;
|
||||
this.modelCheckSkippedWarned = modelCheckSkippedWarned;
|
||||
this.exhaustionSink = exhaustionSink;
|
||||
this.worktreeProvisioned = isProvisionedWorktree(cwd);
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -542,11 +798,142 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
@Override
|
||||
public String agentSessionId() {
|
||||
// fleetd #219 site 2: under memberHerdrSocket the member pane runs as a different OS
|
||||
// user, so opencode.db lives under THAT user's $HOME, not the one discoveryRoot was
|
||||
// built from (see OpenCodeLauncher#defaultDiscoveryRoot's javadoc for the full
|
||||
// reasoning). Scanning fleetd's own $HOME under that config would only ever find "no
|
||||
// row" and read as "resume unsupported" — declare it unavailable instead, once, loudly.
|
||||
// Checked before the fleetd #249 worktree gate below: this OS-user mismatch makes
|
||||
// discovery unusable regardless of whether cwd happens to be a provisioned worktree, so
|
||||
// it earns the one-time WARN either way.
|
||||
if (discoveryUnavailable.getAsBoolean()) {
|
||||
if (discoveryUnavailableWarned.compareAndSet(false, true)) {
|
||||
log.warn("opencode session discovery unavailable: memberHerdrSocket is "
|
||||
+ "configured, so opencode's on-disk session database lives under the "
|
||||
+ "MEMBER's own $HOME, not fleetd's ({}) — agentSessionId will stay null "
|
||||
+ "for every opencode member under this config, and SESSION_RESUME "
|
||||
+ "cannot be honored (fleetd #209/#219).",
|
||||
System.getProperty("user.home"));
|
||||
}
|
||||
return null;
|
||||
}
|
||||
// fleetd #249: cwd is shared with other sessions unless fleetd itself provisioned this
|
||||
// worktree, and sessionIdForDirectory's directory-keyed heuristic cannot tell this
|
||||
// member's row apart from a sibling's in that case (measured: a three-day-old row from
|
||||
// a different profile). Refuse to guess — absent is the honest answer, and it is what
|
||||
// this codebase already returns elsewhere for absent evidence (fleetd #175's UNKNOWN).
|
||||
// This IS the ordinary, expected shape of the large majority of spawns (no worktree
|
||||
// requested), not a configuration gap — but fleetd #267 found that same shape silently
|
||||
// switches off the fleetd #175 model-mismatch check for those spawns too, since
|
||||
// checkModelMatch's only call site is right below this gate. The check cannot be moved
|
||||
// off agentSessionId()'s resolved id: the id is the only safe way to key
|
||||
// actualModelForSessionId to THIS session's own row rather than "whatever is newest in
|
||||
// the shared directory" (fleetd #234) — re-deriving a second, independent answer via
|
||||
// `directory` here would reintroduce exactly the false-positive risk #234 fixed (a
|
||||
// sibling's differently-configured model looking like THIS profile's mismatch). So the
|
||||
// model genuinely is unknowable without a provisioned worktree, and unlike the silence
|
||||
// this branch used to keep, that gap now gets the same one-time, per-profile WARN
|
||||
// treatment discoveryUnavailable already gets above — but keyed by profile, since
|
||||
// several profiles can each hit this independently.
|
||||
if (!worktreeProvisioned) {
|
||||
if (cfg.model() != null && !cfg.model().isBlank()
|
||||
&& modelCheckSkippedWarned.add(cfg.profile())) {
|
||||
log.warn("opencode model-mismatch check (fleetd #175) cannot run for profile "
|
||||
+ "'{}': it was spawned without a fleetd-provisioned worktree (fleetd "
|
||||
+ "#249), so its cwd may be shared with other sessions and the actual "
|
||||
+ "model it is running cannot be safely told apart from a sibling's — "
|
||||
+ "spawn with worktree:true to enable the check for this profile.",
|
||||
cfg.profile());
|
||||
}
|
||||
return null;
|
||||
}
|
||||
// fleetd #234: once resolved, stay resolved. Re-deriving from `directory` on every call
|
||||
// would let this handle's identity drift to a sibling session that later shares the
|
||||
// same cwd and writes a newer row — see resolvedSessionId's javadoc.
|
||||
String cached = resolvedSessionId.get();
|
||||
if (cached != null) {
|
||||
return cached;
|
||||
}
|
||||
// Lazy + retried, never a spawn-time blocker: opencode writes the session record only
|
||||
// when the session is first persisted, so null here is the correct interim answer and
|
||||
// the caller re-calls later (each call re-scans, picking up a record that has since
|
||||
// appeared).
|
||||
return discovery.sessionIdForDirectory(cwd);
|
||||
String id = discovery.sessionIdForDirectory(cwd);
|
||||
if (id != null) {
|
||||
resolvedSessionId.compareAndSet(null, id);
|
||||
}
|
||||
// fleetd #175: check on the SAME tick — while the caller (SessionManager's late-resolve
|
||||
// step) is still re-polling because the id is unknown, the row this id came from (once
|
||||
// it exists) is exactly the row that also carries the actual model. Once id resolves,
|
||||
// the caller stops calling agentSessionId() for this session, so this is naturally a
|
||||
// once-only check that happens right when the row first appears.
|
||||
checkModelMatch(id);
|
||||
return id;
|
||||
}
|
||||
|
||||
/**
|
||||
* Verify the live opencode session is running the model {@link #cfg} requested (fleetd
|
||||
* #175) and, on a real mismatch, log an ERROR and quarantine through {@link
|
||||
* #exhaustionSink}. A no-op when there is nothing to compare against — no model configured,
|
||||
* already reported once for this handle, {@code sessionId} itself is not resolved yet
|
||||
* (fleetd #234: absent evidence, not a mismatch), or the actual model is still UNKNOWN (no
|
||||
* row yet, unreadable database, or unparseable evidence). UNKNOWN must never be treated as
|
||||
* a mismatch: that is the single most important safety rule here — a false positive would
|
||||
* quarantine a perfectly working profile's credential.
|
||||
*
|
||||
* @param sessionId the id {@link #agentSessionId()} just resolved (or had cached) for THIS
|
||||
* handle — the model is read back for this exact session (fleetd #234's
|
||||
* {@link OpenCodeSessionDiscovery#actualModelForSessionId}), never
|
||||
* re-derived from {@code directory}
|
||||
*/
|
||||
private void checkModelMatch(String sessionId) {
|
||||
if (modelMismatchReported.get() || cfg.model() == null || cfg.model().isBlank()) {
|
||||
return;
|
||||
}
|
||||
if (sessionId == null || sessionId.isBlank()) {
|
||||
return; // id not resolved yet — UNKNOWN, never a mismatch (fleetd #175's rule)
|
||||
}
|
||||
OpenCodeSessionDiscovery.ActualModel actual = discovery.actualModelForSessionId(sessionId);
|
||||
if (actual == null) {
|
||||
return; // UNKNOWN evidence — never a mismatch
|
||||
}
|
||||
String[] requestedParts = splitProviderModel(cfg.model());
|
||||
String requestedProvider = requestedParts == null ? null : requestedParts[0];
|
||||
String requestedId = requestedParts == null ? cfg.model() : requestedParts[1];
|
||||
boolean idMatches = requestedId.equals(actual.id());
|
||||
// Compare the provider ONLY when BOTH sides have one. requestedProvider == null covers
|
||||
// a bare profile model with no "/" — the profile never asked for a specific provider.
|
||||
// actual.provider() == null covers opencode's model JSON having an id but no providerID
|
||||
// (a real shape parseModel accepts) — that is missing evidence, not a contradiction, and
|
||||
// acceptance rule 4 says missing evidence is UNKNOWN, never a mismatch. Narrowing the
|
||||
// provider comparison this way keeps the id comparison (the part that actually caught the
|
||||
// xf bug) fully intact — a genuine id mismatch is still caught either way (fleetd #175
|
||||
// review round 2).
|
||||
boolean providerMatches = requestedProvider == null || actual.provider() == null
|
||||
|| requestedProvider.equals(actual.provider());
|
||||
if (idMatches && providerMatches) {
|
||||
return;
|
||||
}
|
||||
if (!modelMismatchReported.compareAndSet(false, true)) {
|
||||
return; // another thread already reported this exact mismatch
|
||||
}
|
||||
String actualDisplay = actual.provider() == null
|
||||
? actual.id() : actual.provider() + "/" + actual.id();
|
||||
log.error("opencode profile '{}' requested model '{}' but the live session is actually "
|
||||
+ "running '{}' — opencode does not fail on an unknown -m, it silently "
|
||||
+ "falls back to a default model, which may be a PAID credential "
|
||||
+ "(fleetd #175); quarantining this profile's credential",
|
||||
cfg.profile(), cfg.model(), actualDisplay);
|
||||
// fleetd #234: pass our OWN profile name too. This check fires from agentSessionId(),
|
||||
// called during SessionManager.acquire() BEFORE this session is registered in
|
||||
// sessions.roster() — a roster-only sink (Fleetd's target -> session -> profile lookup)
|
||||
// finds nothing at this point and silently no-ops (defect 2). We already know exactly
|
||||
// which profile to quarantine without the roster; the sink is passed it explicitly.
|
||||
exhaustionSink.onExhausted(delegate.terminalId(),
|
||||
"opencode model mismatch: profile '" + cfg.profile() + "' requested '"
|
||||
+ cfg.model() + "' but the live session is running '" + actualDisplay
|
||||
+ "' (fleetd #175)",
|
||||
cfg.profile());
|
||||
}
|
||||
|
||||
@Override
|
||||
|
||||
@@ -2,57 +2,117 @@ package dev.ltms.fleet.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
import org.sqlite.SQLiteConfig;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.stream.Stream;
|
||||
import java.sql.Connection;
|
||||
import java.sql.PreparedStatement;
|
||||
import java.sql.ResultSet;
|
||||
import java.sql.SQLException;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
|
||||
/**
|
||||
* Resolves the opencode session id for a fleetd worker from opencode's on-disk storage — the
|
||||
* only place this adapter touches opencode's private layout, and deliberately the <em>only</em>
|
||||
* class that does.
|
||||
*
|
||||
* <p><strong>Why this is isolated behind one seam.</strong> The layout is version-coupled and not a
|
||||
* stable contract: opencode writes one JSON file per session under
|
||||
* {@code <storageRoot>/session/<projectID>/<ses_*.json>}, and each record carries a
|
||||
* {@code "version"} field (e.g. {@code "1.1.31"}), so the exact directory shape, file naming, and
|
||||
* field names can move between opencode releases. opencode also ships a headless HTTP server that
|
||||
* may supersede file scanning entirely. Everything this adapter knows about that private storage —
|
||||
* its shape, naming, and field names — lives here, so a layout change, or a switch to the HTTP
|
||||
* server, changes exactly one class and nothing in {@link OpenCodeLauncher}.
|
||||
* <p><strong>Why this is isolated behind one seam.</strong> The layout is version-coupled and not
|
||||
* a stable contract: opencode persists its session state in a SQLite database at
|
||||
* {@code <storageRoot>/opencode.db} (a {@code session} table, one row per session, keyed by id and
|
||||
* carrying a {@code directory} column). That schema can move between opencode releases exactly
|
||||
* like the JSON-file layout it replaced did (opencode migrated off a one-JSON-file-per-session
|
||||
* tree under {@code <storageRoot>/storage/session/<projectID>/ses_*.json} in January 2026 — that
|
||||
* tree is now a frozen migration artefact nothing writes, which is why this class no longer reads
|
||||
* it). opencode also ships a headless HTTP server that may supersede both of these entirely.
|
||||
* Everything this adapter knows about that private storage — its shape and column names — lives
|
||||
* here, so a layout change, or a switch to the HTTP server, changes exactly one class and nothing
|
||||
* in {@link OpenCodeLauncher}.
|
||||
*
|
||||
* <p>The determinism that makes this useful is structural, not a guess: every fleetd worker runs
|
||||
* in its own unique git worktree, so the record's {@code directory} (its project root) equals the
|
||||
* worker's cwd identifies <em>its</em> session unambiguously. We match on {@code directory} rather
|
||||
* than diffing {@code opencode session list} before/after — that races under concurrent spawns, and
|
||||
* the CLI listing does not even show the directory.
|
||||
* <p><strong>The {@code directory} match is a heuristic, not an identity — fleetd #234.</strong> A
|
||||
* worktree is opt-in: {@code fleet_spawn} only provisions one when the caller passes {@code
|
||||
* worktree:}; the default spawn inherits the lead's own cwd, which every other worker spawned the
|
||||
* same way (and every past session ever run there) shares. {@code directory} therefore does
|
||||
* <em>not</em> identify a session unambiguously in general — only in the special case of a fresh,
|
||||
* unique worktree does "most recently updated row for this directory" reliably mean "this worker's
|
||||
* own row." {@link #sessionIdForDirectory} still has to use this heuristic (the id has to come from
|
||||
* somewhere, and nothing else is available at this layer — see that method's javadoc), but a caller
|
||||
* that already holds a resolved id must never re-derive evidence about that same session via
|
||||
* {@code directory} again; see {@link #actualModelForSessionId}, which looks up by {@code id}
|
||||
* instead for exactly this reason. We match on {@code directory} rather than diffing
|
||||
* {@code opencode session list} before/after — that races under concurrent spawns, and the CLI
|
||||
* listing does not even show the directory.
|
||||
*
|
||||
* <p>All reads are best-effort and never throw: a missing or unreadable storage root, a record that
|
||||
* fails to parse, or a directory with no record yet all yield {@code null}, and the caller (the
|
||||
* session handle) treats that as "identity not resolved yet" and retries later.
|
||||
* <p>All reads are best-effort and never throw: a missing or unreadable database, a query that
|
||||
* fails, or a directory with no row yet all yield {@code null}, and the caller (the session
|
||||
* handle) treats that as "identity not resolved yet" and retries later. The database is opened
|
||||
* read-only and never written to: opencode itself may be running and writing it concurrently (WAL
|
||||
* mode), and this class must never disturb that.
|
||||
*/
|
||||
final class OpenCodeSessionDiscovery {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(OpenCodeSessionDiscovery.class);
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
|
||||
/**
|
||||
* The model opencode actually ran a session on, parsed from the {@code session.model} JSON
|
||||
* column (fleetd #175). {@code id} is never null/blank on a non-null {@code ActualModel} —
|
||||
* {@link #parseModel} returns {@code null} instead when {@code id} cannot be determined, so a
|
||||
* caller only ever sees a fully-known record or {@code null} (UNKNOWN). {@code provider} may
|
||||
* still be {@code null} on its own when the profile that requested the session named no
|
||||
* provider prefix, or opencode's JSON omitted {@code providerID}.
|
||||
*/
|
||||
record ActualModel(String provider, String id) {
|
||||
}
|
||||
|
||||
private final Path storageRoot; // e.g. ~/.local/share/opencode (injectable for tests)
|
||||
private final ObjectMapper json;
|
||||
private final Path databasePath;
|
||||
private final AtomicBoolean warnedMissingDatabase = new AtomicBoolean(false);
|
||||
|
||||
OpenCodeSessionDiscovery(Path storageRoot) {
|
||||
this.storageRoot = storageRoot;
|
||||
this.json = new ObjectMapper();
|
||||
this.databasePath = storageRoot.resolve("opencode.db");
|
||||
}
|
||||
|
||||
/**
|
||||
* The opencode session id whose record references {@code directory} (the worker's cwd), or
|
||||
* {@code null} when no record matches yet. When several records share the directory — e.g.
|
||||
* repeated spawns into the same worktree — the <em>most recently modified</em> one wins: it is
|
||||
* the session the pane most likely corresponds to.
|
||||
* A connection to {@link #databasePath} opened with SQLite's {@code SQLITE_OPEN_READONLY}
|
||||
* flag: it never creates the file, never writes, and never touches WAL or journal mode.
|
||||
* opencode may be running and writing this database concurrently, and this class must never
|
||||
* disturb it.
|
||||
*
|
||||
* <p>Never throws: a missing {@code storageRoot}, an unreadable/malformed record, or a
|
||||
* directory that has not been persisted yet all resolve to {@code null} rather than failing a
|
||||
* spawn. A fleetd worker's session record is written lazily (when the session is first
|
||||
* persisted), so {@code null} here is the normal answer right after the pane is ready, and the
|
||||
* caller retries later.
|
||||
* <p>Package-private so a test can hold the connection and prove it refuses a write. That is
|
||||
* the only way to pin this property: making the file unwritable does <em>not</em> work,
|
||||
* because SQLite silently downgrades a read-write open of an unwritable file to read-only, so
|
||||
* such a test passes whether or not the flag is set.
|
||||
*/
|
||||
Connection openReadOnly() throws SQLException {
|
||||
SQLiteConfig config = new SQLiteConfig();
|
||||
config.setReadOnly(true);
|
||||
return config.createConnection("jdbc:sqlite:" + databasePath);
|
||||
}
|
||||
|
||||
/**
|
||||
* The opencode session id whose row references {@code directory} (the worker's cwd), or
|
||||
* {@code null} when no row matches yet. When several rows share the directory — e.g. repeated
|
||||
* spawns into the same worktree, OR several workers sharing one cwd because none of them was
|
||||
* given a worktree (fleetd #234) — the row with the highest {@code time_updated} wins: it is
|
||||
* the session the pane most likely corresponds to. That "most likely" is a real caveat, not a
|
||||
* formality: when the directory is shared, this can and does pick another session's row (see
|
||||
* the class javadoc). This is the one place fleetd resolves an opencode session id at all —
|
||||
* nothing else is available at this layer to disambiguate further (no {@code opencode session
|
||||
* list} entry names the directory, and diffing before/after races under concurrent spawns) — so
|
||||
* the heuristic stays here unchanged. What must never happen is a SECOND, independent piece of
|
||||
* evidence about the same session being re-derived via {@code directory} once an id has already
|
||||
* come out of this method; see {@link #actualModelForSessionId}.
|
||||
*
|
||||
* <p>Never throws: a missing {@code opencode.db}, a locked/unreadable database, a query
|
||||
* failure, or a directory that has not been persisted yet all resolve to {@code null} rather
|
||||
* than failing a spawn. A fleetd worker's session row is written lazily (when the session is
|
||||
* first persisted), so {@code null} here is the normal answer right after the pane is ready,
|
||||
* and the caller retries later.
|
||||
*
|
||||
* @param directory the worker's cwd, as resolved for this spawn
|
||||
* @return the matching session id, or {@code null} if none is known yet
|
||||
@@ -61,62 +121,112 @@ final class OpenCodeSessionDiscovery {
|
||||
if (directory == null || directory.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
Path sessionRoot = storageRoot.resolve("session");
|
||||
if (!Files.isDirectory(sessionRoot)) {
|
||||
if (!Files.isRegularFile(databasePath)) {
|
||||
if (warnedMissingDatabase.compareAndSet(false, true)) {
|
||||
log.warn("opencode session database not found at {} — opencode's on-disk layout "
|
||||
+ "may have moved again; session discovery will keep returning null",
|
||||
databasePath);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
String best = null;
|
||||
long bestMtime = Long.MIN_VALUE;
|
||||
try (Stream<Path> projectDirs = Files.list(sessionRoot)) {
|
||||
for (Path projectDir : projectDirs.filter(Files::isDirectory).toList()) {
|
||||
try (Stream<Path> records = Files.list(projectDir)) {
|
||||
for (Path record : records.toList()) {
|
||||
String id = matchId(record, directory);
|
||||
if (id == null) {
|
||||
continue;
|
||||
}
|
||||
long mtime = lastModifiedEpochMillis(record);
|
||||
if (mtime > bestMtime) {
|
||||
bestMtime = mtime;
|
||||
best = id;
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// one project dir unreadable — skip it; another may still match
|
||||
String sql = "SELECT id FROM session WHERE directory = ? ORDER BY time_updated DESC LIMIT 1";
|
||||
try (Connection connection = openReadOnly();
|
||||
PreparedStatement statement = connection.prepareStatement(sql)) {
|
||||
statement.setString(1, directory);
|
||||
try (ResultSet rows = statement.executeQuery()) {
|
||||
if (rows.next()) {
|
||||
return rows.getString("id");
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// storage root vanished or became unreadable — "no session known yet"
|
||||
} catch (SQLException e) {
|
||||
// Locked, corrupt, or otherwise unreadable — never fatal to a spawn. Not the
|
||||
// "database moved" signal (the file exists), so this stays below WARN.
|
||||
log.debug("opencode session database unreadable at {}: {}", databasePath, e.toString());
|
||||
return null;
|
||||
}
|
||||
return best;
|
||||
log.debug("no opencode session row for directory (root={}, directory={})",
|
||||
storageRoot, directory);
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The record's session id when it references {@code directory}, else {@code null}. A record
|
||||
* that is not JSON, lacks {@code id}/{@code directory}, or points at a different directory is
|
||||
* simply not our session; a malformed one is skipped, never fatal.
|
||||
* The model opencode actually ran the session {@code sessionId} on (fleetd #175/#234), read by
|
||||
* primary-key lookup — the ONE row that id names, and no other. Deliberately keyed on
|
||||
* {@code id} rather than {@code directory}: two independent {@code WHERE directory = ? ORDER BY
|
||||
* time_updated DESC LIMIT 1} queries (one for the id, one for the model) can each pick a
|
||||
* DIFFERENT row once more than one session shares a directory (fleetd #234 — the default
|
||||
* no-worktree spawn shares the lead's cwd with every other worker and every past session ever
|
||||
* run there), silently comparing a profile's requested model against a session that is not even
|
||||
* the one whose id was returned. Keying on {@code id} instead makes that impossible: the model
|
||||
* read back is always the SAME session {@link #sessionIdForDirectory} (or a cached copy of its
|
||||
* answer) already resolved.
|
||||
*
|
||||
* <p>Kept as its own query and its own connection, independent from {@link
|
||||
* #sessionIdForDirectory}: a database whose schema predates the {@code model} column (or any
|
||||
* other read failure on this column alone) can never take id resolution down with it. That
|
||||
* would be a regression of the id-resolution feature #209 shipped; this method degrades on its
|
||||
* own.
|
||||
*
|
||||
* <p>Never throws, and every failure mode — no matching row, a missing/unreadable database, a
|
||||
* missing {@code model} column, a null/blank {@code model} value, or JSON that does not parse
|
||||
* into {@code {"id": "...", "providerID": "..."}} with a non-blank {@code id} — resolves to
|
||||
* {@code null}. That is UNKNOWN evidence, not a mismatch signal: the caller must never
|
||||
* quarantine a profile on the strength of a {@code null} here. A blank/null {@code sessionId}
|
||||
* (the id is not resolved yet) is UNKNOWN too, for the same reason — never call this with one.
|
||||
*
|
||||
* @param sessionId the session id already resolved by {@link #sessionIdForDirectory} for this
|
||||
* spawn — never re-derived from {@code directory} here
|
||||
* @return the actual model, or {@code null} when unknown
|
||||
*/
|
||||
private String matchId(Path record, String directory) {
|
||||
ActualModel actualModelForSessionId(String sessionId) {
|
||||
if (sessionId == null || sessionId.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
if (!Files.isRegularFile(databasePath)) {
|
||||
// sessionIdForDirectory already WARNs once (shared warnedMissingDatabase) for this
|
||||
// exact condition — do not double-log it here.
|
||||
return null;
|
||||
}
|
||||
String sql = "SELECT model FROM session WHERE id = ?";
|
||||
try (Connection connection = openReadOnly();
|
||||
PreparedStatement statement = connection.prepareStatement(sql)) {
|
||||
statement.setString(1, sessionId);
|
||||
try (ResultSet rows = statement.executeQuery()) {
|
||||
if (rows.next()) {
|
||||
return parseModel(rows.getString("model"));
|
||||
}
|
||||
}
|
||||
} catch (SQLException e) {
|
||||
// Locked/corrupt database, or a `model` column this schema version does not have —
|
||||
// never fatal, and never a mismatch signal. See the class doc above.
|
||||
log.debug("opencode session model unreadable at {}: {}", databasePath, e.toString());
|
||||
return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse opencode's {@code model} column — {@code {"id":"...","providerID":"..."}} — into an
|
||||
* {@link ActualModel}, or {@code null} when {@code json} is null/blank, is not valid JSON, or
|
||||
* parses without a non-blank {@code id}. {@code providerID} may be absent; that alone does not
|
||||
* make the record unknown, since a caller comparing against a profile with no provider prefix
|
||||
* never looks at it.
|
||||
*/
|
||||
private static ActualModel parseModel(String json) {
|
||||
if (json == null || json.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
JsonNode node = json.readTree(record.toFile());
|
||||
JsonNode id = node == null ? null : node.get("id");
|
||||
JsonNode dir = node == null ? null : node.get("directory");
|
||||
if (id == null || dir == null || !directory.equals(dir.asText())) {
|
||||
JsonNode node = MAPPER.readTree(json);
|
||||
String id = node.path("id").asText(null);
|
||||
if (id == null || id.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
return id.asText();
|
||||
String provider = node.path("providerID").asText(null);
|
||||
return new ActualModel(provider, id);
|
||||
} catch (IOException e) {
|
||||
log.debug("opencode session model JSON unparseable: {}", e.toString());
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** The record's last-modified epoch ms, or {@code Long.MIN_VALUE} if unreadable (never wins). */
|
||||
private static long lastModifiedEpochMillis(Path record) {
|
||||
try {
|
||||
return Files.getLastModifiedTime(record).toMillis();
|
||||
} catch (IOException e) {
|
||||
return Long.MIN_VALUE;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -22,6 +22,7 @@ import java.util.concurrent.ConcurrentSkipListMap;
|
||||
import java.util.concurrent.ExecutionException;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.TimeoutException;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
/**
|
||||
* AMQP-backed {@link ReplyInbox} (CB-307 Stage 2): genuine cross-restart durability behind the same
|
||||
@@ -93,8 +94,23 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
private final Channel channel;
|
||||
/** All channel operations (publish/declare/ack/cancel) serialize on this — a Channel is not thread-safe. */
|
||||
private final Object channelLock = new Object();
|
||||
/** target → (msgId → held delivery). Per-target map is guarded by synchronizing on itself. */
|
||||
/**
|
||||
* target → (msgId → held delivery). Per-target map is guarded by synchronizing on itself.
|
||||
*
|
||||
* <p><strong>CB-318 tombstone.</strong> The value {@link #RELEASED} is a reserved sentinel: it
|
||||
* marks a target whose {@link #release} has already run, so {@link #deliverCallback} can tell a
|
||||
* delivery landing after release() apart from a fresh target it has never seen. See both methods'
|
||||
* javadoc for why a plain {@code held.remove(target)} is not enough.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, LinkedHashMap<String, Held>> held = new ConcurrentHashMap<>();
|
||||
|
||||
/**
|
||||
* CB-318 sentinel stored in {@link #held} for a target whose {@link #release} has already run.
|
||||
* Never mutated — every read site compares it by reference ({@code ==}) before touching it as a
|
||||
* map, because it is a single object shared across every released target and calling a mutator on
|
||||
* it would corrupt state for all of them.
|
||||
*/
|
||||
private static final LinkedHashMap<String, Held> RELEASED = new LinkedHashMap<>();
|
||||
/** Targets whose queue is declared and consumer is running, mapped to their broker consumer tag. */
|
||||
private final ConcurrentHashMap<String, String> consumerTags = new ConcurrentHashMap<>();
|
||||
|
||||
@@ -201,6 +217,12 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
channel.queueDeclare(queue, true, false, false, null); // durable, non-exclusive, keep on idle
|
||||
String tag = channel.basicConsume(queue, false, deliverCallback(target), _ -> { });
|
||||
consumerTags.put(target, tag);
|
||||
// CB-318: drop a stale RELEASED tombstone from a prior ownership of this same target
|
||||
// string, so a delivery under this fresh consumer is held normally instead of being
|
||||
// nacked forever by deliverCallback's RELEASED check. Safe to do here, still under
|
||||
// channelLock: no delivery for the consumer tag just registered above can reach
|
||||
// deliverCallback before this basicConsume call returns.
|
||||
held.remove(target, RELEASED);
|
||||
log.debug("AMQP inbox owns queue {} for target {}", queue, target);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot own queue " + queue, e);
|
||||
@@ -208,18 +230,103 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Release ownership of {@code target}: cancel its consumer, then nack-with-requeue every
|
||||
* delivery still held for it instead of just dropping the local record.
|
||||
*
|
||||
* <p><strong>Cancelling a consumer does not requeue its in-flight deliveries.</strong> In AMQP,
|
||||
* a delivery that was pushed to a consumer stays unacked, attached to the still-open
|
||||
* {@link #channel}, until that channel or the connection closes — {@code basicCancel} alone does
|
||||
* neither. So before this method existed with a requeue step, it dropped {@link #held}'s entries
|
||||
* for {@code target} while the broker still considered them outstanding: never acked, never
|
||||
* nacked, never requeued, and no longer reachable by {@link #peek} — permanently invisible. This
|
||||
* is unlike {@link #handleRecovery} and {@link #close()}, whose bare {@code held.clear()} is
|
||||
* correct because each has already made the broker requeue (a real connection drop, or
|
||||
* {@code channel.close()} respectively) before clearing local state.
|
||||
*
|
||||
* <p><strong>Order: cancel first, then nack.</strong> A delivery tag stays valid for
|
||||
* {@code basicNack} on this channel regardless of whether its consumer is still attached — only
|
||||
* a channel/connection close invalidates it — so cancelling {@code target}'s consumer first does
|
||||
* not risk the tags. Doing it the other way round does: nacking a delivery with {@code requeue}
|
||||
* while its consumer is still active hands the message straight back to that <em>same</em>
|
||||
* consumer the instant a prefetch slot frees up (confirmed against a real broker — see
|
||||
* {@code AmqpReplyInboxContractTest.releaseCancelsConsumerAndRequeuesHeldDeliveryForRecovery}),
|
||||
* which races this method's own {@code held.remove(target)}: the redelivery can land after the
|
||||
* clear and leave a stale entry behind, so {@link #peek} is no longer reliably empty right after
|
||||
* {@link #release}. Cancelling first closes that consumer, so the requeued message goes back to
|
||||
* the queue for whichever consumer picks it up next (a later {@link #own}), not this one.
|
||||
*
|
||||
* <p><strong>Failure of the requeue is best-effort, not fatal.</strong> {@link #release} runs
|
||||
* during teardown ({@code Fleetd} calls it right after {@code MessageService.abandon}), and a
|
||||
* throw here would abort cleanups the caller depends on — the same argument fleetd #293 settled
|
||||
* for {@code HerdrPeerLauncher.stop()}'s tab-close step. So a failed {@code basicNack} is logged
|
||||
* at WARN, naming the target and delivery tag that leaked, and release proceeds; the delivery
|
||||
* stays unacked on the broker rather than being silently dropped, so it is still recoverable by a
|
||||
* later connection drop even though this release did not manage to requeue it immediately. A
|
||||
* failed {@code basicCancel} still throws, unchanged from before this fix — that failure means
|
||||
* the consumer may still be attached, so best-effort requeue is not attempted underneath it.
|
||||
*
|
||||
* <p><strong>CB-318: {@code held.remove(target)} alone leaves a second window open.</strong> The
|
||||
* bullet above already explains why cancelling first does not save a tag from going stale — but
|
||||
* that only accounts for a delivery landing before this method starts touching {@link #held}.
|
||||
* {@code basicCancel} stops <em>new</em> dispatches; it does not flush one already handed to the
|
||||
* consumer work pool. So a delivery can still land on that pool's thread and reach
|
||||
* {@link #deliverCallback} at any point during, or after, this method's body — and a plain
|
||||
* {@code held.remove(target)} does nothing to stop it: {@code deliverCallback}'s
|
||||
* {@code computeIfAbsent} finds the key gone and happily creates a brand-new map under it, which
|
||||
* this method — already past its {@code remove} — never looks at again. That entry then sits
|
||||
* delivered-but-unacked on {@link #channel} until the whole inbox closes: never requeued, never
|
||||
* redelivered, and {@link #peek} is never called again for a target nothing owns any more.
|
||||
*
|
||||
* <p>The fix is {@link #held}{@code .compute(target, ...)} instead of {@code remove}: it takes
|
||||
* whatever was held (to nack, same as before) and, in the same atomic step, leaves the
|
||||
* {@link #RELEASED} tombstone behind instead of an absent key. {@code computeIfAbsent} and
|
||||
* {@code compute} calls for the same key are mutually exclusive in {@link ConcurrentHashMap} —
|
||||
* whichever of this call and a concurrent {@code deliverCallback} runs first is fully visible to
|
||||
* the other, with no gap between them. So a delivery that loses the race sees a real map here and
|
||||
* gets nacked by the loop below, same as always; a delivery that wins the race (runs first) is
|
||||
* itself nacked by that same loop, once it settles into {@code held}. A delivery that arrives once
|
||||
* this method has stored {@link #RELEASED} finds it via {@code computeIfAbsent} and refuses itself
|
||||
* — see {@link #deliverCallback}. Either way nothing is silently retained forever, satisfying the
|
||||
* ticket's invariant against dropping a message. This closes the window rather than merely
|
||||
* narrowing it — correctness does not depend on how much time elapses between the swap and this
|
||||
* method returning.
|
||||
*/
|
||||
@Override
|
||||
public void release(String target) {
|
||||
synchronized (channelLock) {
|
||||
String tag = consumerTags.remove(target);
|
||||
held.remove(target); // stale delivery tags must not survive release
|
||||
if (tag == null) {
|
||||
return;
|
||||
if (tag != null) {
|
||||
try {
|
||||
channel.basicCancel(tag);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot cancel consumer for " + target, e);
|
||||
}
|
||||
}
|
||||
try {
|
||||
channel.basicCancel(tag);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot cancel consumer for " + target, e);
|
||||
AtomicReference<LinkedHashMap<String, Held>> previouslyHeld = new AtomicReference<>();
|
||||
held.compute(target, (_, v) -> {
|
||||
previouslyHeld.set(v);
|
||||
return RELEASED;
|
||||
});
|
||||
var perTarget = previouslyHeld.get();
|
||||
if (perTarget != null && perTarget != RELEASED) {
|
||||
synchronized (perTarget) {
|
||||
for (Held h : perTarget.values()) {
|
||||
try {
|
||||
channel.basicNack(h.deliveryTag(), false, true); // requeue, don't drop
|
||||
} catch (IOException | RuntimeException e) {
|
||||
// Caught broadly (not just IOException) for the same reason #293 catches
|
||||
// RuntimeException in HerdrPeerLauncher.stop(): best-effort teardown must
|
||||
// not be guarded only against the expected failure and bare against any
|
||||
// other. The message stays unacked on the broker either way — not lost,
|
||||
// just not proactively requeued — until a connection drop frees it.
|
||||
log.warn("release({}): could not requeue held delivery (msgId={}, tag={})"
|
||||
+ " back to the broker — it stays unacked until a connection"
|
||||
+ " drop frees it: {}",
|
||||
target, h.message().msgId(), h.deliveryTag(), e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -272,7 +379,7 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
if (perTarget == null || perTarget == RELEASED) {
|
||||
return List.of();
|
||||
}
|
||||
synchronized (perTarget) {
|
||||
@@ -283,7 +390,7 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
if (perTarget == null || perTarget == RELEASED) {
|
||||
return;
|
||||
}
|
||||
Held h;
|
||||
@@ -316,6 +423,20 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
String content = new String(delivery.getBody(), StandardCharsets.UTF_8);
|
||||
var perTarget = held.computeIfAbsent(target, _ -> new LinkedHashMap<>());
|
||||
if (perTarget == RELEASED) {
|
||||
// CB-318: release() already ran for this target and left the RELEASED tombstone in
|
||||
// held (see release()'s javadoc) — computeIfAbsent() is guaranteed to see it rather
|
||||
// than recreate a fresh map, because ConcurrentHashMap serializes compute/
|
||||
// computeIfAbsent calls for the same key against each other. Refuse the delivery
|
||||
// instead of holding it somewhere release() will never look at again: requeue it, the
|
||||
// same way release() nacks its own held entries, so a later owner (or a connection
|
||||
// drop) can still recover it. This does not need channelLock across a broker round
|
||||
// trip — basicNack, like the duplicate-ack case just below, does not wait for one.
|
||||
synchronized (channelLock) {
|
||||
channel.basicNack(tag, false, true);
|
||||
}
|
||||
return;
|
||||
}
|
||||
boolean duplicate;
|
||||
synchronized (perTarget) {
|
||||
if (perTarget.containsKey(msgId)) {
|
||||
|
||||
@@ -2,12 +2,14 @@ package dev.ltms.fleet.msg;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
@@ -166,15 +168,49 @@ public final class MessageService {
|
||||
private static final class Task {
|
||||
private final String ticket;
|
||||
private final String target;
|
||||
private final CompletableFuture<Reply> future = new CompletableFuture<>();
|
||||
/**
|
||||
* When this task was created (#137 fix): the tiebreaker for which of several open tasks on
|
||||
* one target gets a recovered reply in {@link #abandon} — the oldest, since it is the one
|
||||
* that has been waiting longest.
|
||||
*/
|
||||
private final long createdNanos;
|
||||
private final CompletableFuture<Reply> future = new CompletableFuture<>();
|
||||
/**
|
||||
* When {@link #future} resolved, or {@code null} while it is still pending — the clock
|
||||
* {@link #pruneTerminalTickets} measures the TTL from (#197). Deliberately a boxed
|
||||
* {@code Long} rather than a {@code long} with a sentinel: {@link System#nanoTime} may
|
||||
* legitimately return any value, zero and negatives included, so no numeric sentinel can mean
|
||||
* "not stamped yet". Stamped by a {@code whenComplete} hook registered in the constructor, so
|
||||
* every completion path stamps it — a reply, the completion fallback, a timeout, a failure,
|
||||
* or an abandon on teardown — without each of those having to remember to.
|
||||
*/
|
||||
private volatile Long completedNanos;
|
||||
private volatile Reply question;
|
||||
private volatile String turnId;
|
||||
/**
|
||||
* Set when this task's {@code fleet_ask} lapsed with no answer (fleetd #307):
|
||||
* {@link #clearAsyncQuestion} then forgets {@link #turnId} (nulls it and drops the task from
|
||||
* {@code asyncTasksByTurn}) so {@link #hasAsyncQuestion} stops reporting the target BUSY — a
|
||||
* later {@code fleet_send} to it must be accepted, not refused. But the worker's turn is
|
||||
* still genuinely live: it resumed on its own and will eventually call its real
|
||||
* {@code fleet_reply}. Losing {@link #turnId} loses {@link #askAnsweredAsyncTasks}' only
|
||||
* signal that such a reply belongs to this task, so that reply used to fall straight to the
|
||||
* inbox and strand — {@code fleet_poll} stayed {@code PENDING} forever, later force-failed by
|
||||
* {@link #abandon} with the misleading "session released before it replied". This flag is a
|
||||
* second, independent signal that survives the forgetting: {@link #askAnsweredAsyncTasks}
|
||||
* accepts it in place of a live {@link #turnId}, without ever re-adding the task to
|
||||
* {@code asyncTasksByTurn} (so the BUSY release is untouched). Cleared implicitly once
|
||||
* {@link #future} resolves — every match in {@link #askAnsweredAsyncTasks} already requires
|
||||
* {@code !future.isDone()}, so a task that recovered (or was later failed by
|
||||
* {@link #abandon}) can never match again regardless of this flag's value.
|
||||
*/
|
||||
private volatile boolean askTimedOut;
|
||||
|
||||
private Task(String ticket, String target, long createdNanos) {
|
||||
private Task(String ticket, String target, LongSupplier nowNanos) {
|
||||
this.ticket = ticket;
|
||||
this.target = target;
|
||||
this.createdNanos = createdNanos;
|
||||
this.createdNanos = nowNanos.getAsLong();
|
||||
future.whenComplete((reply, ex) -> completedNanos = nowNanos.getAsLong());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -189,6 +225,7 @@ public final class MessageService {
|
||||
}
|
||||
|
||||
private final AgentControl agents;
|
||||
private final HerdrRouter router;
|
||||
private final Injector injector;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ReplyInbox inbox;
|
||||
@@ -257,6 +294,7 @@ public final class MessageService {
|
||||
MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous, ReplyInbox inbox,
|
||||
ReplyPushLoop pushLoop, Metrics metrics, LongSupplier nowNanos) {
|
||||
this.agents = agents;
|
||||
this.router = null;
|
||||
this.injector = injector;
|
||||
this.rendezvous = rendezvous;
|
||||
this.inbox = inbox;
|
||||
@@ -265,6 +303,20 @@ public final class MessageService {
|
||||
this.nowNanos = nowNanos;
|
||||
}
|
||||
|
||||
public MessageService(HerdrRouter router, Injector injector, Rendezvous rendezvous, ReplyInbox inbox,
|
||||
ReplyPushLoop pushLoop, Metrics metrics) {
|
||||
this.agents = null;
|
||||
this.router = router;
|
||||
this.injector = injector;
|
||||
this.rendezvous = rendezvous;
|
||||
this.inbox = inbox;
|
||||
this.pushLoop = pushLoop;
|
||||
this.metrics = metrics;
|
||||
this.nowNanos = System::nanoTime;
|
||||
}
|
||||
|
||||
private AgentControl agentsFor(String target) { return router != null ? router.agentsFor(target) : agents; }
|
||||
|
||||
/** Create with an explicit {@link ReplyInbox} and no push loop. */
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous, ReplyInbox inbox) {
|
||||
this(agents, injector, rendezvous, inbox, null);
|
||||
@@ -277,7 +329,7 @@ public final class MessageService {
|
||||
|
||||
/** Current lifecycle status of a worker (the {@code GET /sessions/{id}/status} surface). */
|
||||
public AgentStatus status(String target) {
|
||||
return agents.status(target);
|
||||
return agentsFor(target).status(target);
|
||||
}
|
||||
|
||||
/** Read-only delegation fact for fleet views. */
|
||||
@@ -326,6 +378,18 @@ public final class MessageService {
|
||||
* {@link #ask} clears the ticket's question and returns it to {@code PENDING}, but {@link #send}
|
||||
* already closed the forward waiter the instant the question surfaced, so the target has
|
||||
* neither an accepted nor a queued delivery left to show for it.
|
||||
*
|
||||
* <p><strong>Deliberately still {@code question == null} only (fleetd #275).</strong> This
|
||||
* method must not also report a still-{@link Phase#ASKING} task as orphaned: the worker may
|
||||
* genuinely be waiting on a live primary that is about to (or already mid-{@link #answer})
|
||||
* answer it, and {@link dev.ltms.fleet.health.FleetHealthMonitor} would classify that as
|
||||
* {@code DELEGATION_ORPHANED} on nothing more than an active, healthy conversation. {@link
|
||||
* #abandon(String, String, boolean)}'s {@code sweepAsking} path fixes the actual reachable gap
|
||||
* (a target torn down for good while genuinely {@code ASKING}) at the point of teardown itself,
|
||||
* by completing the task's future right there — so by the time this method would ever see it,
|
||||
* {@code task.future.isDone()} is already {@code true} and it is excluded regardless of this
|
||||
* guard. Widening this check instead of that one would trade a real fix for false positives on
|
||||
* every ordinary in-flight question.
|
||||
*/
|
||||
public boolean hasOrphanedDelegation(String target) {
|
||||
if (target == null || hasAcceptedDelivery(target) || hasQueuedDelivery(target)) {
|
||||
@@ -340,21 +404,80 @@ public final class MessageService {
|
||||
}
|
||||
|
||||
/**
|
||||
* Route a worker's explicit {@code fleet_reply}: resolve an open send, or queue it in the
|
||||
* inbox if no send is currently open. Unlike the bare {@link Rendezvous#resolve}, a no-waiter
|
||||
* result is <em>not</em> a failure — the reply is held for later drain.
|
||||
* Route a worker's explicit {@code fleet_reply}: resolve an open send, complete an async ticket
|
||||
* still parked waiting on this exact turn's answer, or — only once neither applies — queue it in
|
||||
* the inbox. Unlike the bare {@link Rendezvous#resolve}, a no-waiter result is <em>not</em> a
|
||||
* failure — the reply is held for later drain.
|
||||
*
|
||||
* <p><strong>Do NOT use this for mid-turn questions.</strong> {@code fleet_ask} /
|
||||
* {@link Rendezvous#resolveQuestion} must keep today's {@code NO_WAITER} behaviour — questions
|
||||
* are interactive and must never be queued.
|
||||
*
|
||||
* @return always {@code true} — the reply either resolved a live send or was queued
|
||||
* <p><strong>Ambiguous match also falls to the inbox.</strong> {@link #askAnsweredAsyncTasks} can
|
||||
* return more than one entry — a reachable state, not a hypothetical one (see its own javadoc:
|
||||
* an {@code fleet_ask} that lapsed with no answer, fleetd #307, frees the target for a completely fresh
|
||||
* delegation, which can itself go on to ask-and-lapse before the first worker's real reply
|
||||
* arrives). Returning whichever candidate a {@code ConcurrentHashMap} iteration reaches first
|
||||
* would let a genuine reply complete the <em>wrong</em> ticket — silently handing the lead
|
||||
* something that reads like a correct answer to a delegation the worker never touched, which is
|
||||
* worse than a failure because the lead acts on it. When more than one candidate exists, guessing
|
||||
* is not safe: fall back to the inbox exactly as the zero-candidate case does, and let
|
||||
* {@link #abandon} apply the eventual recovery deterministically instead.
|
||||
*
|
||||
* <p><strong>{@code content} is required (fleetd #302).</strong> Both doors that reach this
|
||||
* method must reject a missing/blank reply the same way, so the check lives here rather than in
|
||||
* either caller: {@code FleetMcp.reply} already refuses a {@code null} content before it ever
|
||||
* calls this method (its own required-arg guard), and no test or production call site anywhere
|
||||
* in the codebase relies on replying with empty content — confirmed by searching every call site
|
||||
* of this method before adding the check, not assumed. Without this guard, a REST {@code
|
||||
* POST /sessions/{id}/reply} whose body omits {@code content} (or a client library that maps a
|
||||
* missing field to {@code ""}) used to reach {@link Rendezvous#resolve} with an empty string,
|
||||
* silently completing the lead's blocking wait with nothing — indistinguishable from a worker
|
||||
* that genuinely replied with nothing, which is worse than a loud failure because it destroys the
|
||||
* information that the reply never arrived.
|
||||
*
|
||||
* @throws IllegalArgumentException if {@code content} is {@code null} or blank — the caller must
|
||||
* report this as a client error (REST: 400 {@code bad_request}) rather than resolve
|
||||
* anything
|
||||
* @return always {@code true} — the reply resolved a live send, completed a parked ticket, or
|
||||
* was queued
|
||||
*/
|
||||
public boolean reply(String session, String content) {
|
||||
if (content == null || content.isBlank()) {
|
||||
throw new IllegalArgumentException("content is required");
|
||||
}
|
||||
if (rendezvous.resolve(session, content)) {
|
||||
count(FleetMetrics.REPLIES, "path", "rendezvous");
|
||||
return true; // a live send took it — unchanged fast path
|
||||
}
|
||||
// #137/fleetd #307: no live rendezvous waiter, but this may be the worker's real fleet_reply resuming
|
||||
// a turn that either answer() (#137) or ask() (fleetd #307) already gave up waiting on:
|
||||
// - answer()'s own bounded wait (the primary's fleet_send{turnId} call, capped well under a
|
||||
// minute) can time out and close its waiter long before the worker — now actually resuming
|
||||
// real work — finishes and replies.
|
||||
// - ask()'s own wait for the primary can time out first, with the worker resuming on its own
|
||||
// and finishing unanswered.
|
||||
// Either way that reply used to have nowhere to land but the session inbox, leaving the async
|
||||
// ticket's future unresolved forever: fleet_poll{ticket} stayed PENDING until fleet_stop's
|
||||
// abandon() forced it FAILED with a misleading "session released before it replied" reason,
|
||||
// even though the reply had, in fact, arrived. Completing the matching ticket directly here
|
||||
// means fleet_poll{ticket} sees the real reply instead.
|
||||
List<Task> candidates = askAnsweredAsyncTasks(session);
|
||||
if (candidates.size() == 1) {
|
||||
Task orphan = candidates.get(0);
|
||||
if (orphan.future.complete(new Reply(Outcome.REPLIED, content))) {
|
||||
if (orphan.turnId != null) {
|
||||
asyncTasksByTurn.remove(orphan.turnId, orphan);
|
||||
}
|
||||
count(FleetMetrics.REPLIES, "path", "async-recovered");
|
||||
return true; // the ticket itself took it — no inbox stranding at all
|
||||
}
|
||||
} else if (candidates.size() > 1) {
|
||||
List<String> tickets = candidates.stream().map(t -> t.ticket).toList();
|
||||
log.warn("reply from {} matches {} open async tickets {} — cannot tell which one it "
|
||||
+ "answers, queuing to the inbox instead of guessing", session, candidates.size(),
|
||||
tickets);
|
||||
}
|
||||
inbox.publish(session, UUID.randomUUID().toString(), content);
|
||||
// CB-640: record the stranding itself (not just the reply text) so fleet health can see a
|
||||
// worker whose replies keep missing their waiter, not only the queue depth this leaves behind.
|
||||
@@ -368,6 +491,49 @@ public final class MessageService {
|
||||
return true; // held, not lost
|
||||
}
|
||||
|
||||
/**
|
||||
* Every still-open async task on {@code target} whose worker is genuinely expected to send a
|
||||
* real {@code fleet_reply} next with nothing left registered to catch it: either its
|
||||
* {@code fleet_ask} was already answered — {@link Task#turnId} is stamped but {@link
|
||||
* Task#question} was cleared by {@link #answer} — or its {@code fleet_ask} lapsed unanswered and
|
||||
* {@link Task#askTimedOut} marks that (fleetd #307; {@link Task#turnId} is {@code null} by then, forgotten
|
||||
* so the target is not left BUSY — see {@link Task#askTimedOut}'s own javadoc). Either way the
|
||||
* task's future is not resolved yet. Empty if no such task exists, including the common case
|
||||
* where {@code target}'s worker never used {@code fleet_ask} at all (a task that was never asked
|
||||
* has both {@code turnId == null} and {@code askTimedOut == false}, so it can never match here and
|
||||
* only ever completes through the ordinary rendezvous fast path in {@link #reply}).
|
||||
*
|
||||
* <p><strong>Can return more than one entry — reachable, not just defence in depth.</strong>
|
||||
* {@link #send} refuses to open a waiter on {@code target} while {@link #hasAsyncQuestion} is
|
||||
* true, and that check matches ANY task whose {@code turnId} is still stamped in
|
||||
* {@code asyncTasksByTurn}. While a task's {@code turnId} stays stamped — {@link #answer} leaves
|
||||
* it in place ({@code clearAsyncQuestion(turnId, false)}) until {@link #finishAsyncTask} removes
|
||||
* the stamp and completes the future in the same call — no second task on the same target can
|
||||
* reach an eligible state, because {@link #send} would refuse it as BUSY first. That single-task
|
||||
* guarantee holds only for the {@code turnId}-stamped half of this method's match: an
|
||||
* {@link Task#askTimedOut} task is, by construction, no longer stamped in {@code asyncTasksByTurn}
|
||||
* (that is the whole point of forgetting {@code turnId} in {@link #clearAsyncQuestion}), so the
|
||||
* target is free the moment one ask lapses. A fresh, independent {@code sendAsync} to the same
|
||||
* target can then be dispatched, itself pause on {@code fleet_ask}, and itself time out — landing
|
||||
* a second {@code askTimedOut} task on the very target the first one is still waiting to answer
|
||||
* for. Two (or more) genuinely open tasks on one target is therefore a real, reachable state
|
||||
* today, not a hypothetical: {@link #reply} treats it as unresolvable and falls back to the
|
||||
* inbox rather than guess which task a reply belongs to (guessing wrong would hand the lead a
|
||||
* plausible-looking answer to a delegation the worker never touched — worse than a failure,
|
||||
* because the lead acts on it); {@link #abandon} instead picks the oldest deterministically (its
|
||||
* own {@code matching} list has a different, wider match — see its javadoc).
|
||||
*/
|
||||
private List<Task> askAnsweredAsyncTasks(String target) {
|
||||
List<Task> candidates = new ArrayList<>();
|
||||
for (Task task : tasks.values()) {
|
||||
if (target.equals(task.target) && task.question == null && !task.future.isDone()
|
||||
&& (task.turnId != null || task.askTimedOut)) {
|
||||
candidates.add(task);
|
||||
}
|
||||
}
|
||||
return candidates;
|
||||
}
|
||||
|
||||
/** Record a counter sample when a registry is wired; a no-op in unit tests. */
|
||||
private void count(String name, String... labels) {
|
||||
if (metrics != null) {
|
||||
@@ -409,19 +575,151 @@ public final class MessageService {
|
||||
* <p>Resolving the waiter as a failure — rather than letting it time out — also means the
|
||||
* outcome is counted, so a torn-down delegation stops being invisible to {@code /metrics}.
|
||||
*
|
||||
* @return true if a live waiter was failed
|
||||
* <p><strong>#137 defence in depth.</strong> {@link #reply} already hands a worker's real
|
||||
* {@code fleet_reply} straight to the async ticket it belongs to whenever exactly one is still
|
||||
* parked waiting for it (see {@link #askAnsweredAsyncTasks}), so by the time a session is
|
||||
* released its tasks are normally already resolved — this loop's {@code complete} calls are then
|
||||
* harmless no-ops (a {@link CompletableFuture} can only resolve once). But should some other path
|
||||
* someday strand a reply in the inbox without completing its ticket, checking
|
||||
* {@link #hasStrandedReply(String)} here — before ever writing a failure — means a torn-down
|
||||
* session whose worker in fact replied is still reported {@code REPLIED} with that reply's own
|
||||
* text, never the misleading "the worker session was released before it replied" (which also
|
||||
* means the snapshot/worktree recovery hint that follows it never prints once a reply exists).
|
||||
*
|
||||
* <p><strong>At most one task gets the recovered reply — and here, unlike {@link #reply}'s
|
||||
* {@link #askAnsweredAsyncTasks}, {@code matching.size() >= 2} alone is reachable today.</strong>
|
||||
* This method's {@code matching} filter has no {@code turnId != null} requirement, so it matches
|
||||
* any plain (never-asked) open task too — and {@link #sendAsync} does not limit a target to one
|
||||
* of those: a second {@code fleet_send{wait:false}} at a target that is still busy returns its own
|
||||
* ticket immediately and simply parks its {@link #send} behind the target's session lock for up
|
||||
* to {@link #ASYNC_TIMEOUT_MS}, exactly as {@code abandonFailsEveryPendingAsyncTicketForTheReleasedTarget}
|
||||
* already proves. Before this fix, the loop below drained the strand once and then reused that
|
||||
* same {@code Reply} for <em>every</em> task it walked past — so two open tasks really did both
|
||||
* complete {@code REPLIED} with the same text (see the pre-fix loop in commit 97f6c33's parent).
|
||||
* A stranded reply is one worker answer, so it can settle at most one open task on this target —
|
||||
* never every open task, and never a guess. When more than one task is still open here, the
|
||||
* recovered reply goes to the <em>oldest</em> (lowest {@link Task#createdNanos}) — it has been
|
||||
* waiting longest, so it is the one most likely to be what the reply actually answers. Every
|
||||
* other open task keeps the ordinary {@code WORKER_FAILED} path it would take without a stranded
|
||||
* reply at all.
|
||||
*
|
||||
* <p><strong>{@code matching.size() >= 2} together with {@code hadStrandedReply} is a different
|
||||
* question, and today it is defence in depth rather than a path this codebase's public API can
|
||||
* drive.</strong> This class has exactly two sites that ever acquire a target's entry in
|
||||
* {@code sessionLocks} — {@link #send} and {@link #answer} — and both open a {@link Rendezvous}
|
||||
* waiter for that same target as the very first thing they do after acquiring the lock, then hold
|
||||
* lock and waiter together for the rest of their critical section ({@link #send} also clears
|
||||
* {@link #strandedReplies} right there, the instant it opens its waiter — before it ever enqueues
|
||||
* delivery). So "the session lock is held" and "a live waiter is open for it" are the same fact
|
||||
* throughout this class, and {@link #reply}'s fast path always resolves a currently-open waiter
|
||||
* directly rather than stranding. The two facts this method wants therefore cannot be produced
|
||||
* side by side: while the lock is held, a real reply resolves the open waiter directly and never
|
||||
* reaches {@link #strandedReplies}; the instant the lock is free, any parked matching task's own
|
||||
* {@link #send} that is scheduled next wins it and, by opening its waiter, clears the strand again
|
||||
* before this method ever runs. There is no way to hold that lock open-but-unaccepted from outside
|
||||
* {@link #send}/{@link #answer} to freeze a window in between. Constructing both facts at once
|
||||
* through {@code sendAsync}/{@code reply}/{@code ask}/{@code answer} would need a race against
|
||||
* virtual-thread scheduling, not a deterministic sequence — so the oldest-wins code below stays as
|
||||
* defence in depth against a regression to that mechanism (e.g. clearing {@link #strandedReplies}
|
||||
* on a narrower condition than "any acceptance"), not because today's test suite exercises the
|
||||
* conjunction. {@code matching.size() >= 2} alone, without a strand, is exactly what
|
||||
* {@code abandonFailsEveryPendingAsyncTicketForTheReleasedTarget} already covers.
|
||||
*
|
||||
* <p><strong>Do not "fix" that gap with a test that reaches past this class.</strong> A test can
|
||||
* build both facts by calling {@link Rendezvous#close} itself on the waiter an accepted
|
||||
* {@link #send} is still blocked on: the send keeps the lock, no waiter is registered any more, a
|
||||
* second parked task stays open, and the next {@link #reply} then strands. That was checked, and
|
||||
* such a test does go red against the pre-fix loop. But it only goes red because it broke the
|
||||
* lock-and-waiter invariant above from outside — no caller of this class ever does that — so it
|
||||
* pins a state production cannot reach, and would read to the next person as if it could.
|
||||
*
|
||||
* @return true if a live waiter or an async task was failed (never true for one recovered as a
|
||||
* reply — see the note above)
|
||||
*/
|
||||
public boolean abandon(String target, String reason) {
|
||||
return abandon(target, reason, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #abandon(String, String)}, with control over whether a task still paused in
|
||||
* {@code fleet_ask} ({@link Phase#ASKING}) is swept too (fleetd #275).
|
||||
*
|
||||
* <p>{@code sweepAsking} must be {@code true} only when the caller has independent, certain
|
||||
* knowledge that {@code target} can never resume its turn — today that is only
|
||||
* {@code sessions.onRelease}'s teardown (an explicit {@code fleet_stop}, or the idle reaper):
|
||||
* the worker's pane is being stopped right now, so whatever it was mid-{@code fleet_ask} about
|
||||
* has no turn left to resume into. {@link dev.ltms.fleet.health.FleetHealthMonitor}'s
|
||||
* health-classification call keeps passing {@code false} (via {@link #abandon(String, String)}):
|
||||
* a GONE/NEVER_READY reading is the daemon's best guess from the live agent list, not a teardown
|
||||
* it performed itself, and {@code abandonDoesNotFailAnAsyncTicketWaitingForAnAnswer} documents
|
||||
* why an active ask must survive that guess — the primary may already be mid-{@link #answer} for
|
||||
* the very same turn, and completing it here first would preempt a real answer with a misleading
|
||||
* failure.
|
||||
*
|
||||
* <p><strong>Without {@code sweepAsking} on the release path, a target torn down while
|
||||
* genuinely {@code ASKING} was unrecoverable.</strong> {@link #resolveQuestion} had already
|
||||
* closed the forward waiter the instant the question surfaced (so the {@code waiter} branch
|
||||
* below finds nothing to fail), the {@code question == null} guard excluded the task from
|
||||
* {@code matching} (so the loop below skipped it too), and the worker's own {@code fleet_ask}
|
||||
* clears {@link Task#question} back to {@code null} only once it lapses (the reverse-rendezvous
|
||||
* window — up to {@code FleetMcp.ASK_DEFAULT_TIMEOUT_MS} / {@code FleetApp.MAX_ASK_TIMEOUT_MS},
|
||||
* 55–115s) — by which point the released session no longer appears in {@code sessions.roster()}
|
||||
* for {@link dev.ltms.fleet.health.FleetHealthMonitor} to ever re-observe, so nothing was ever
|
||||
* left to call {@link #abandon} on this target again. The ticket then sat in {@link #tasks}
|
||||
* forever: not terminal, so {@link #pruneTerminalTickets} never dropped it, and
|
||||
* {@code fleet_poll} reported it stuck at {@link Phase#PENDING} for good.
|
||||
*/
|
||||
public boolean abandon(String target, String reason, boolean sweepAsking) {
|
||||
boolean hadStrandedReply = hasStrandedReply(target);
|
||||
// CB-640: the session is gone — nothing will ever accept or deliver into it now.
|
||||
strandedReplies.remove(target);
|
||||
queuedDeliveries.remove(target);
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(target);
|
||||
boolean failed = waiter != null && !waiter.isDone() && rendezvous.resolveFailure(waiter, reason);
|
||||
boolean asyncFailed = false;
|
||||
|
||||
List<Task> matching = new ArrayList<>();
|
||||
for (Task task : tasks.values()) {
|
||||
if (target.equals(task.target) && task.question == null
|
||||
&& task.future.complete(new Reply(Outcome.WORKER_FAILED, reason))) {
|
||||
asyncFailed = true;
|
||||
if (target.equals(task.target) && (sweepAsking || task.question == null)
|
||||
&& !task.future.isDone()) {
|
||||
matching.add(task);
|
||||
}
|
||||
}
|
||||
Task recoveryTask = null;
|
||||
if (hadStrandedReply && !matching.isEmpty()) {
|
||||
recoveryTask = matching.get(0);
|
||||
for (Task candidate : matching) {
|
||||
if (candidate.createdNanos < recoveryTask.createdNanos) {
|
||||
recoveryTask = candidate;
|
||||
}
|
||||
}
|
||||
}
|
||||
Reply recovered = recoveryTask != null ? recoverStrandedReply(target) : null;
|
||||
for (Task task : matching) {
|
||||
boolean isRecovery = task == recoveryTask && recovered != null;
|
||||
Reply outcome = isRecovery ? recovered : new Reply(Outcome.WORKER_FAILED, reason);
|
||||
String turnId = task.turnId;
|
||||
if (task.future.complete(outcome)) {
|
||||
if (outcome.outcome() == Outcome.WORKER_FAILED) {
|
||||
asyncFailed = true;
|
||||
}
|
||||
if (turnId != null) {
|
||||
// #275: whether this task was swept out of ASKING or was already answered and
|
||||
// only waiting on its resumed turn's real reply (#137), nothing will ever
|
||||
// complete this turnId now — drop it from this class's own bookkeeping AND the
|
||||
// reverse-rendezvous itself, so hasAsyncQuestion(target) stops reporting a turn
|
||||
// that is actually done, and a late answer() sees it as lapsed rather than
|
||||
// resolving a question nothing is listening for any more.
|
||||
asyncTasksByTurn.remove(turnId, task);
|
||||
rendezvous.closeAsk(turnId);
|
||||
}
|
||||
} else if (isRecovery) {
|
||||
// The recovered reply was already drained out of the inbox, but this task resolved
|
||||
// through another path (e.g. a concurrent reply() or a second abandon() racing this
|
||||
// one) between us choosing it and completing it here. Put the reply back rather than
|
||||
// lose it silently — it may still belong to some other still-open task, or the next
|
||||
// caller that drains this target's inbox.
|
||||
inbox.publish(target, UUID.randomUUID().toString(), recovered.text());
|
||||
}
|
||||
}
|
||||
if (failed) {
|
||||
@@ -430,6 +728,25 @@ public final class MessageService {
|
||||
return failed || asyncFailed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Drain {@code target}'s inbox and hand its content back as a {@link Outcome#REPLIED} result
|
||||
* (#137 defence in depth for {@link #abandon}) — {@code null} if it turned out empty (the
|
||||
* stranding fact raced away, e.g. a lead's own {@code fleet_poll} on the raw session already
|
||||
* drained it first). When more than one message is queued, only the newest is the worker's actual
|
||||
* final answer ({@link #drainReplies} returns them oldest-first).
|
||||
*
|
||||
* <p>This does drain (removes the messages from the inbox) before the caller knows whether the
|
||||
* task it is recovering for will actually accept them — {@link #abandon} is the one that puts a
|
||||
* reply back if its {@code complete} call turns out to lose the race.
|
||||
*/
|
||||
private Reply recoverStrandedReply(String target) {
|
||||
var messages = drainReplies(target);
|
||||
if (messages.isEmpty()) {
|
||||
return null;
|
||||
}
|
||||
return new Reply(Outcome.REPLIED, messages.get(messages.size() - 1).content());
|
||||
}
|
||||
|
||||
/**
|
||||
* Acknowledge a specific reply by {@code msgId} for {@code target}. Removes it from the inbox
|
||||
* so that a subsequent drain or peek no longer returns it.
|
||||
@@ -529,7 +846,7 @@ public final class MessageService {
|
||||
if (task != null) {
|
||||
asyncTasksByWaiter.put(reply, task);
|
||||
}
|
||||
TurnToken token = new TurnToken(target, reply);
|
||||
TurnToken token = new TurnToken(target, reply, content);
|
||||
// The send has won the lock; the accepted-delivery hook records delegator ownership
|
||||
// here (CB-548). It runs BEFORE enqueue so a throwing hook — onAccepted is now a
|
||||
// public callback — fails the send without queuing a message that would orphan.
|
||||
@@ -608,6 +925,12 @@ public final class MessageService {
|
||||
return new AskResult(AskOutcome.ANSWERED, answer);
|
||||
} catch (TimeoutException e) {
|
||||
log.debug("fleet_ask from {} went unanswered in {}ms", workerSession, timeoutMillis);
|
||||
// fleetd #307: mark the task BEFORE clearAsyncQuestion(forgetTurn=true) below drops it out of
|
||||
// asyncTasksByTurn and nulls its turnId — that forgetting is deliberate and stays (it is
|
||||
// what keeps the target from staying BUSY forever), but it would otherwise also erase
|
||||
// askAnsweredAsyncTasks' only signal that the worker's eventual real fleet_reply still
|
||||
// belongs to this task, stranding it in the inbox with a false "never replied" verdict.
|
||||
markAskTimedOut(ticket.turnId());
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
return new AskResult(AskOutcome.TIMED_OUT, null);
|
||||
} catch (ExecutionException e) {
|
||||
@@ -655,7 +978,16 @@ public final class MessageService {
|
||||
}
|
||||
try {
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession);
|
||||
// #282: mirror send()'s registration (:802) so a SECOND fleet_ask inside this same
|
||||
// resumed turn can re-associate the async ticket with its new turnId via
|
||||
// markAsyncQuestion — without this, that second ask has no Task to attach to, and
|
||||
// markAsyncQuestion silently returns null.
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null) {
|
||||
asyncTasksByWaiter.put(reply, task);
|
||||
}
|
||||
if (!rendezvous.answerAsk(turnId, content)) {
|
||||
asyncTasksByWaiter.remove(reply);
|
||||
rendezvous.close(workerSession, reply);
|
||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||
}
|
||||
@@ -663,7 +995,21 @@ public final class MessageService {
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
Reply result = new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
finishAsyncTask(turnId, result);
|
||||
// #282: this waiter can resolve with a FRESH question rather than a terminal reply —
|
||||
// the worker chained a second fleet_ask before replying. Mirror sendAsync's own guard
|
||||
// (:1000) and leave the ticket open (markAsyncQuestion above already re-armed it under
|
||||
// the new turnId) instead of completing it here with a QUESTION "reply".
|
||||
// Measured when #282 was merged: this guard is DEFENCE IN DEPTH, not the thing
|
||||
// that makes the chained ask work. ask() calls markAsyncQuestion (:860) before
|
||||
// resolveQuestion (:861), so by the time this thread wakes, the task has already
|
||||
// moved to the new turnId and finishAsyncTask(oldTurnId, ...) finds nothing. Removing
|
||||
// this guard alone leaves the test green. Keep it anyway: it mirrors sendAsync's
|
||||
// sibling guard, and that sibling's own comment (:1017) warns the two orderings are
|
||||
// not something to rely on. Do NOT delete it as dead code without re-checking that
|
||||
// ordering, and do not treat it as the sole protection either.
|
||||
if (result.outcome() != Outcome.QUESTION) {
|
||||
finishAsyncTask(turnId, result);
|
||||
}
|
||||
return result;
|
||||
} catch (TimeoutException e) {
|
||||
// The worker resumed but hasn't replied yet — no completion fallback arms an answered
|
||||
@@ -676,6 +1022,7 @@ public final class MessageService {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + workerSession, e);
|
||||
} finally {
|
||||
asyncTasksByWaiter.remove(reply);
|
||||
rendezvous.close(workerSession, reply);
|
||||
}
|
||||
} finally {
|
||||
@@ -704,7 +1051,7 @@ public final class MessageService {
|
||||
*/
|
||||
public String sendAsync(String target, String content, Runnable onAccepted) {
|
||||
String ticket = "task-" + ticketSeq.incrementAndGet();
|
||||
Task task = new Task(ticket, target, nowNanos.getAsLong());
|
||||
Task task = new Task(ticket, target, nowNanos);
|
||||
tasks.put(ticket, task);
|
||||
if (pushLoop != null) {
|
||||
// CB-588: task.future only ever completes on a terminal phase (DONE or a failure) — a
|
||||
@@ -788,7 +1135,7 @@ public final class MessageService {
|
||||
/** Best-effort live worker status for a pending poll; never throws (a lookup error is just noise). */
|
||||
private String liveStatus(String target) {
|
||||
try {
|
||||
return agents.status(target).name().toLowerCase();
|
||||
return agentsFor(target).status(target).name().toLowerCase();
|
||||
} catch (RuntimeException e) {
|
||||
return "unknown";
|
||||
}
|
||||
@@ -804,12 +1151,25 @@ public final class MessageService {
|
||||
* (or one the reminder cap already gave up on) is pruned here but never collected there, so it
|
||||
* lingers in {@code pendingTickets} forever and rides along on every later nudge to the same lead
|
||||
* — naming a ticket {@code fleet_poll} can no longer find (CB-588 follow-up).
|
||||
*
|
||||
* <p>The TTL runs from **completion**, not from creation (#197). It used to compare against
|
||||
* {@code createdNanos}, which made the real collection window {@code TTL minus however long the
|
||||
* task ran}: a delegation that took longer than the TTL had its reply destroyed on the first
|
||||
* sweep after it landed, every time. That is the normal case here — real work runs well past ten
|
||||
* minutes — and the reply lives only in {@code future}, so pruning it discards the worker's whole
|
||||
* report with nothing to fall back on. Measuring from completion gives every ticket the same full
|
||||
* window whatever its runtime, and still bounds {@code tasks}.
|
||||
*
|
||||
* <p>A task whose future is done but whose {@code completedNanos} is not stamped yet is left
|
||||
* alone. That window is the few instructions between {@code complete()} and the constructor's
|
||||
* {@code whenComplete} hook running; the next sweep collects it.
|
||||
*/
|
||||
private void pruneTerminalTickets() {
|
||||
long cutoff = nowNanos.getAsLong() - TICKET_TTL_NANOS;
|
||||
tasks.entrySet().removeIf(e -> {
|
||||
Task t = e.getValue();
|
||||
boolean expired = t.future.isDone() && t.createdNanos < cutoff;
|
||||
Long completed = t.completedNanos;
|
||||
boolean expired = t.future.isDone() && completed != null && completed < cutoff;
|
||||
if (expired && pushLoop != null) {
|
||||
pushLoop.ticketCollected(e.getKey());
|
||||
}
|
||||
@@ -821,13 +1181,37 @@ public final class MessageService {
|
||||
private Task markAsyncQuestion(CompletableFuture<Rendezvous.Resolution> waiter, String text, String turnId) {
|
||||
Task task = waiter == null ? null : asyncTasksByWaiter.get(waiter);
|
||||
if (task != null) {
|
||||
String previousTurnId = task.turnId;
|
||||
task.question = new Reply(Outcome.QUESTION, text, turnId);
|
||||
task.turnId = turnId;
|
||||
asyncTasksByTurn.put(turnId, task);
|
||||
// #282: a second fleet_ask in the same resumed turn re-arms an already-answered task
|
||||
// (answer() re-registers it in asyncTasksByWaiter) under a FRESH turnId — drop the old
|
||||
// key so asyncTasksByTurn does not keep growing by one stale entry per chained ask.
|
||||
if (previousTurnId != null && !previousTurnId.equals(turnId)) {
|
||||
asyncTasksByTurn.remove(previousTurnId, task);
|
||||
}
|
||||
}
|
||||
return task;
|
||||
}
|
||||
|
||||
/**
|
||||
* Mark {@code turnId}'s task as having a {@code fleet_ask} that lapsed with no answer (fleetd #307), so
|
||||
* {@link #askAnsweredAsyncTasks} still recognizes the worker's eventual real {@code fleet_reply}
|
||||
* as belonging to it after {@link #clearAsyncQuestion}'s {@code forgetTurn=true} erases
|
||||
* {@link Task#turnId} — see {@link Task#askTimedOut}. Must be called before that forgetting, while
|
||||
* {@code turnId} can still resolve the task in {@code asyncTasksByTurn}; a lookup afterward would
|
||||
* find nothing. Only when it matches the task's current turn — same guard as
|
||||
* {@link #clearAsyncQuestion} — so a chained second {@code fleet_ask} (#282) that already moved
|
||||
* the task to a fresh {@code turnId} cannot mark it for a turn that is no longer its own.
|
||||
*/
|
||||
private void markAskTimedOut(String turnId) {
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null && turnId.equals(task.turnId)) {
|
||||
task.askTimedOut = true;
|
||||
}
|
||||
}
|
||||
|
||||
/** Clear an answered or lapsed question, but only when it matches the ticket's current turn. */
|
||||
private void clearAsyncQuestion(String turnId, boolean forgetTurn) {
|
||||
// CB-582: tell the push loop first — like ticketCollected, a removal for a turnId it never
|
||||
|
||||
@@ -9,8 +9,10 @@ import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collection;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
@@ -68,6 +70,12 @@ public final class ReplyPushLoop {
|
||||
static final String QUESTIONS_NUDGE_FORMAT =
|
||||
"%d workers are paused on a question — run fleet_poll(ticket=...) for each, then answer "
|
||||
+ "with fleet_send(turnId=..., content=...): %s";
|
||||
static final String BACKEND_INCIDENT_NUDGE_FORMAT =
|
||||
"Backend credential %s is cooling for %d remaining seconds; affected profiles: %s; "
|
||||
+ "affected workers: %s. Run fleet_list to see more.";
|
||||
static final String BACKEND_TARGET_UNMAPPED_NUDGE_FORMAT =
|
||||
"A backend error on worker %s could not be mapped to a credential (%s) — no cool-off "
|
||||
+ "was applied. Run fleet_list to check that worker.";
|
||||
|
||||
private final PrimaryRegistry primaryRegistry;
|
||||
private final AgentControl agents;
|
||||
@@ -93,6 +101,13 @@ public final class ReplyPushLoop {
|
||||
* how depleted an older, still-open question's count is.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, PendingQuestion> pendingQuestions = new ConcurrentHashMap<>();
|
||||
/** Backend incidents awaiting one successful delivery, keyed by incident id and owning lead. */
|
||||
private final ConcurrentHashMap<IncidentLead, PendingIncident> pendingIncidents = new ConcurrentHashMap<>();
|
||||
/** Incident/lead pairs already delivered. They make repeated incident reports one-shot. */
|
||||
private final Set<IncidentLead> deliveredIncidents = ConcurrentHashMap.newKeySet();
|
||||
private final ConcurrentHashMap<UnmappedTargetLead, PendingUnmappedTarget> pendingUnmappedTargets =
|
||||
new ConcurrentHashMap<>();
|
||||
private final Set<UnmappedTargetLead> deliveredUnmappedTargets = ConcurrentHashMap.newKeySet();
|
||||
/** CB-590: leads with an active combined reminder schedule (replies and/or tickets and/or questions). */
|
||||
private final ConcurrentHashMap<String, Boolean> activeLeads = new ConcurrentHashMap<>();
|
||||
|
||||
@@ -178,7 +193,20 @@ public final class ReplyPushLoop {
|
||||
* tracked per question, not per lead per source).
|
||||
*/
|
||||
private record PendingQuestion(String turnId, String ticket, String target, String lead,
|
||||
String question, int nudgeCount) {
|
||||
String question, int nudgeCount) {
|
||||
}
|
||||
|
||||
private record IncidentLead(String incidentId, String lead) {
|
||||
}
|
||||
|
||||
private record PendingIncident(IncidentLead key, String credential, List<String> profiles,
|
||||
List<String> targets, int remainingCoolOffSeconds, int nudgeCount) {
|
||||
}
|
||||
|
||||
private record UnmappedTargetLead(String target, String reason, String lead) {
|
||||
}
|
||||
|
||||
private record PendingUnmappedTarget(UnmappedTargetLead key, int nudgeCount) {
|
||||
}
|
||||
|
||||
/** Questions still open for {@code lead}, snapshotted fresh for one tick. */
|
||||
@@ -192,6 +220,24 @@ public final class ReplyPushLoop {
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
private List<PendingIncident> pendingIncidentsFor(String lead) {
|
||||
return pendingIncidents.values().stream().filter(i -> lead.equals(i.key().lead())).toList();
|
||||
}
|
||||
|
||||
private Set<IncidentLead> pendingIncidentKeysFor(String lead) {
|
||||
return pendingIncidentsFor(lead).stream().map(PendingIncident::key)
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
private List<PendingUnmappedTarget> pendingUnmappedTargetsFor(String lead) {
|
||||
return pendingUnmappedTargets.values().stream().filter(i -> lead.equals(i.key().lead())).toList();
|
||||
}
|
||||
|
||||
private Set<UnmappedTargetLead> pendingUnmappedTargetKeysFor(String lead) {
|
||||
return pendingUnmappedTargetsFor(lead).stream().map(PendingUnmappedTarget::key)
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* The reply-source reminder count {@link #decide} should see for {@code lead} on this tick:
|
||||
* the <em>minimum</em> nudge count among the reply targets currently pending for it (CB-598).
|
||||
@@ -236,6 +282,15 @@ public final class ReplyPushLoop {
|
||||
return min == Integer.MAX_VALUE ? 0 : min;
|
||||
}
|
||||
|
||||
private int minIncidentNudgeCountFor(String lead) {
|
||||
return pendingIncidentsFor(lead).stream().mapToInt(PendingIncident::nudgeCount).min().orElse(0);
|
||||
}
|
||||
|
||||
private int minUnmappedTargetNudgeCountFor(String lead) {
|
||||
return pendingUnmappedTargetsFor(lead).stream().mapToInt(PendingUnmappedTarget::nudgeCount)
|
||||
.min().orElse(0);
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure decision function: examine everything pending for {@code lead} — reply targets and
|
||||
* tickets alike — and return what the loop should do.
|
||||
@@ -272,17 +327,35 @@ public final class ReplyPushLoop {
|
||||
* @param questionReminderCount the lowest nudge count among questions open for this lead
|
||||
*/
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount) {
|
||||
return decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
minIncidentNudgeCountFor(lead));
|
||||
}
|
||||
|
||||
/** As above, with backend incidents as a fourth, independently bounded source. */
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount,
|
||||
int incidentReminderCount) {
|
||||
return decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
incidentReminderCount, minUnmappedTargetNudgeCountFor(lead));
|
||||
}
|
||||
|
||||
/** As above, with unmapped backend targets as a fifth, independently bounded source. */
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount,
|
||||
int incidentReminderCount, int unmappedTargetReminderCount) {
|
||||
boolean hasReplyWork = !pendingReplyTargetsFor(lead).isEmpty();
|
||||
boolean hasTicketWork = !pendingTicketIdsFor(lead).isEmpty();
|
||||
boolean hasQuestionWork = !pendingQuestionTurnIdsFor(lead).isEmpty();
|
||||
if (!hasReplyWork && !hasTicketWork && !hasQuestionWork) {
|
||||
boolean hasIncidentWork = !pendingIncidentKeysFor(lead).isEmpty();
|
||||
boolean hasUnmappedTargetWork = !pendingUnmappedTargetKeysFor(lead).isEmpty();
|
||||
if (!hasReplyWork && !hasTicketWork && !hasQuestionWork && !hasIncidentWork && !hasUnmappedTargetWork) {
|
||||
log.debug("push: nothing pending for lead {}, stopping reminder", lead);
|
||||
return Action.STOP;
|
||||
}
|
||||
boolean replyEligible = hasReplyWork && replyReminderCount < maxReminders;
|
||||
boolean ticketEligible = hasTicketWork && ticketReminderCount < maxReminders;
|
||||
boolean questionEligible = hasQuestionWork && questionReminderCount < maxReminders;
|
||||
if (!replyEligible && !ticketEligible && !questionEligible) {
|
||||
boolean incidentEligible = hasIncidentWork && incidentReminderCount < maxReminders;
|
||||
boolean unmappedTargetEligible = hasUnmappedTargetWork && unmappedTargetReminderCount < maxReminders;
|
||||
if (!replyEligible && !ticketEligible && !questionEligible && !incidentEligible && !unmappedTargetEligible) {
|
||||
log.debug("push: reminder cap ({}) reached for lead {} on every source with pending work, stopping",
|
||||
maxReminders, lead);
|
||||
countNudge("exhausted");
|
||||
@@ -394,6 +467,48 @@ public final class ReplyPushLoop {
|
||||
pendingQuestions.remove(turnId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Queue a one-shot backend credential outage notice for every distinct lead that owns an
|
||||
* affected worker. A successful injection records its {@code (incidentId, lead)} key, so a
|
||||
* repeat report never reminds that lead again.
|
||||
*/
|
||||
public void onBackendIncident(String incidentId, Collection<String> targets, String credential,
|
||||
Collection<String> profiles, int remainingCoolOffSeconds) {
|
||||
Map<String, List<String>> targetsByLead = new ConcurrentHashMap<>();
|
||||
for (String target : targets) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.warn("push: backend incident {} has no known lead for target {}", incidentId, target);
|
||||
continue;
|
||||
}
|
||||
targetsByLead.computeIfAbsent(lead.get(), _ -> new ArrayList<>()).add(target);
|
||||
}
|
||||
List<String> profileNames = profiles.stream().sorted().toList();
|
||||
for (var entry : targetsByLead.entrySet()) {
|
||||
IncidentLead key = new IncidentLead(incidentId, entry.getKey());
|
||||
if (deliveredIncidents.contains(key)) continue;
|
||||
pendingIncidents.putIfAbsent(key, new PendingIncident(key, credential, profileNames,
|
||||
entry.getValue().stream().sorted().toList(), remainingCoolOffSeconds, 0));
|
||||
startOrCoalesce(entry.getKey());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Tell the owning lead that a classified backend error could not be tied to a credential.
|
||||
* Without an owning lead, emit a warning because no control can act on the target.
|
||||
*/
|
||||
public void onBackendTargetUnmapped(String target, String reason) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.warn("push: backend target {} could not map to a credential: {}", target, reason);
|
||||
return;
|
||||
}
|
||||
UnmappedTargetLead key = new UnmappedTargetLead(target, reason, lead.get());
|
||||
if (deliveredUnmappedTargets.contains(key)) return;
|
||||
pendingUnmappedTargets.putIfAbsent(key, new PendingUnmappedTarget(key, 0));
|
||||
startOrCoalesce(lead.get());
|
||||
}
|
||||
|
||||
// --- the schedule ----------------------------------------------------------------------------
|
||||
|
||||
/** Start a reminder schedule for {@code lead}, or join the one already running. */
|
||||
@@ -425,18 +540,25 @@ public final class ReplyPushLoop {
|
||||
Set<String> repliesBefore = pendingReplyTargetsFor(lead);
|
||||
Set<String> ticketsBefore = pendingTicketIdsFor(lead);
|
||||
Set<String> questionsBefore = pendingQuestionTurnIdsFor(lead);
|
||||
Set<IncidentLead> incidentsBefore = pendingIncidentKeysFor(lead);
|
||||
Set<UnmappedTargetLead> unmappedTargetsBefore = pendingUnmappedTargetKeysFor(lead);
|
||||
int replyReminderCount = minReplyNudgeCountFor(lead);
|
||||
int ticketReminderCount = minTicketNudgeCountFor(lead);
|
||||
int questionReminderCount = minQuestionNudgeCountFor(lead);
|
||||
var action = decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount);
|
||||
int incidentReminderCount = minIncidentNudgeCountFor(lead);
|
||||
int unmappedTargetReminderCount = minUnmappedTargetNudgeCountFor(lead);
|
||||
var action = decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
incidentReminderCount, unmappedTargetReminderCount);
|
||||
switch (action) {
|
||||
case INJECT -> {
|
||||
injectNudge(lead, replyReminderCount, ticketReminderCount, questionReminderCount);
|
||||
injectNudge(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
incidentReminderCount, unmappedTargetReminderCount);
|
||||
scheduleNext(lead);
|
||||
}
|
||||
// Re-check after the configured backoff; the lead may become injectable soon.
|
||||
case WAIT_BUSY -> scheduleNext(lead);
|
||||
case STOP -> stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore);
|
||||
case STOP -> stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, incidentsBefore,
|
||||
unmappedTargetsBefore);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -480,11 +602,25 @@ public final class ReplyPushLoop {
|
||||
* decision-to-release window reclaims the schedule slot exactly like a raced-in reply or ticket.
|
||||
*/
|
||||
void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
|
||||
Set<String> questionsBefore) {
|
||||
Set<String> questionsBefore) {
|
||||
stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, pendingIncidentKeysFor(lead));
|
||||
}
|
||||
|
||||
private void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
|
||||
Set<String> questionsBefore, Set<IncidentLead> incidentsBefore) {
|
||||
stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, incidentsBefore,
|
||||
pendingUnmappedTargetKeysFor(lead));
|
||||
}
|
||||
|
||||
private void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
|
||||
Set<String> questionsBefore, Set<IncidentLead> incidentsBefore,
|
||||
Set<UnmappedTargetLead> unmappedTargetsBefore) {
|
||||
activeLeads.remove(lead);
|
||||
boolean racedIn = pendingReplyTargetsFor(lead).stream().anyMatch(t -> !repliesBefore.contains(t))
|
||||
|| pendingTicketIdsFor(lead).stream().anyMatch(t -> !ticketsBefore.contains(t))
|
||||
|| pendingQuestionTurnIdsFor(lead).stream().anyMatch(t -> !questionsBefore.contains(t));
|
||||
|| pendingQuestionTurnIdsFor(lead).stream().anyMatch(t -> !questionsBefore.contains(t))
|
||||
|| pendingIncidentKeysFor(lead).stream().anyMatch(i -> !incidentsBefore.contains(i))
|
||||
|| pendingUnmappedTargetKeysFor(lead).stream().anyMatch(i -> !unmappedTargetsBefore.contains(i));
|
||||
if (racedIn && activeLeads.putIfAbsent(lead, Boolean.TRUE) == null) {
|
||||
log.debug("push: new work for lead {} raced the reminder loop's stop — restarting", lead);
|
||||
scheduleNext(lead);
|
||||
@@ -495,18 +631,22 @@ public final class ReplyPushLoop {
|
||||
|
||||
/** Send one combined nudge covering everything currently pending for {@code lead}. */
|
||||
private void injectNudge(String lead, int replyReminderCount, int ticketReminderCount,
|
||||
int questionReminderCount) {
|
||||
int questionReminderCount, int incidentReminderCount,
|
||||
int unmappedTargetReminderCount) {
|
||||
// Re-read rather than threading it down from decide(): a reply can drain, a ticket be
|
||||
// collected, or a question be answered (or another arrive), between the decision and the
|
||||
// injection.
|
||||
Set<String> replyTargets = pendingReplyTargetsFor(lead);
|
||||
List<PendingTicket> tickets = pendingTicketsFor(lead);
|
||||
List<PendingQuestion> questions = pendingQuestionsFor(lead);
|
||||
if (replyTargets.isEmpty() && tickets.isEmpty() && questions.isEmpty()) {
|
||||
List<PendingIncident> incidents = pendingIncidentsFor(lead);
|
||||
List<PendingUnmappedTarget> unmappedTargets = pendingUnmappedTargetsFor(lead);
|
||||
if (replyTargets.isEmpty() && tickets.isEmpty() && questions.isEmpty() && incidents.isEmpty()
|
||||
&& unmappedTargets.isEmpty()) {
|
||||
log.debug("push: pending work for lead {} drained before the nudge could be sent", lead);
|
||||
return;
|
||||
}
|
||||
String nudge = formatNudge(replyTargets, tickets, questions);
|
||||
String nudge = formatNudge(replyTargets, tickets, questions, incidents, unmappedTargets);
|
||||
try {
|
||||
agents.send(lead, nudge);
|
||||
log.debug("push: nudge sent to lead {} (reply {}/{}, ticket {}/{}, question {}/{}; "
|
||||
@@ -515,6 +655,16 @@ public final class ReplyPushLoop {
|
||||
questionReminderCount + 1, maxReminders,
|
||||
replyTargets.size(), tickets.size(), questions.size());
|
||||
countNudge("delivered");
|
||||
for (PendingIncident incident : incidents) {
|
||||
if (pendingIncidents.remove(incident.key(), incident)) {
|
||||
deliveredIncidents.add(incident.key());
|
||||
}
|
||||
}
|
||||
for (PendingUnmappedTarget unmappedTarget : unmappedTargets) {
|
||||
if (pendingUnmappedTargets.remove(unmappedTarget.key(), unmappedTarget)) {
|
||||
deliveredUnmappedTargets.add(unmappedTarget.key());
|
||||
}
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("push: failed to nudge lead {} (reply {}/{}, ticket {}/{}, question {}/{}): {}",
|
||||
lead, replyReminderCount + 1, maxReminders, ticketReminderCount + 1, maxReminders,
|
||||
@@ -526,12 +676,13 @@ public final class ReplyPushLoop {
|
||||
// item already at or over the cap keeps riding along in the text (still pending, still
|
||||
// named) but its extra bumps here are inert: decide() already treats it as ineligible once
|
||||
// its count reaches maxReminders.
|
||||
bumpNudgeCounts(replyTargets, tickets, questions);
|
||||
bumpNudgeCounts(replyTargets, tickets, questions, incidents, unmappedTargets);
|
||||
}
|
||||
|
||||
/** Record that every one of these items was just named in a sent (or attempted) nudge. */
|
||||
private void bumpNudgeCounts(Set<String> replyTargets, List<PendingTicket> tickets,
|
||||
List<PendingQuestion> questions) {
|
||||
List<PendingQuestion> questions, List<PendingIncident> incidents,
|
||||
List<PendingUnmappedTarget> unmappedTargets) {
|
||||
for (String target : replyTargets) {
|
||||
pendingReplies.computeIfPresent(target, (t, e) -> new ReplyEntry(e.lead(), e.nudgeCount() + 1));
|
||||
}
|
||||
@@ -544,6 +695,14 @@ public final class ReplyPushLoop {
|
||||
new PendingQuestion(e.turnId(), e.ticket(), e.target(), e.lead(), e.question(),
|
||||
e.nudgeCount() + 1));
|
||||
}
|
||||
for (PendingIncident incident : incidents) {
|
||||
pendingIncidents.computeIfPresent(incident.key(), (id, e) -> new PendingIncident(e.key(),
|
||||
e.credential(), e.profiles(), e.targets(), e.remainingCoolOffSeconds(), e.nudgeCount() + 1));
|
||||
}
|
||||
for (PendingUnmappedTarget unmappedTarget : unmappedTargets) {
|
||||
pendingUnmappedTargets.computeIfPresent(unmappedTarget.key(), (id, e) ->
|
||||
new PendingUnmappedTarget(e.key(), e.nudgeCount() + 1));
|
||||
}
|
||||
}
|
||||
|
||||
/** Schedule the next tick on the scheduler thread pool. */
|
||||
@@ -556,7 +715,8 @@ public final class ReplyPushLoop {
|
||||
|
||||
/** Render everything pending for one lead as a single nudge line. */
|
||||
private static String formatNudge(Set<String> replyTargets, List<PendingTicket> tickets,
|
||||
List<PendingQuestion> questions) {
|
||||
List<PendingQuestion> questions, List<PendingIncident> incidents,
|
||||
List<PendingUnmappedTarget> unmappedTargets) {
|
||||
List<String> parts = new ArrayList<>();
|
||||
if (!replyTargets.isEmpty()) {
|
||||
parts.add(formatRepliesNudge(replyTargets));
|
||||
@@ -567,6 +727,12 @@ public final class ReplyPushLoop {
|
||||
if (!questions.isEmpty()) {
|
||||
parts.add(formatQuestionsNudge(questions));
|
||||
}
|
||||
if (!incidents.isEmpty()) {
|
||||
parts.addAll(incidents.stream().map(ReplyPushLoop::formatBackendIncidentNudge).toList());
|
||||
}
|
||||
if (!unmappedTargets.isEmpty()) {
|
||||
parts.addAll(unmappedTargets.stream().map(ReplyPushLoop::formatUnmappedTargetNudge).toList());
|
||||
}
|
||||
return String.join(" | ", parts);
|
||||
}
|
||||
|
||||
@@ -606,6 +772,16 @@ public final class ReplyPushLoop {
|
||||
return QUESTIONS_NUDGE_FORMAT.formatted(pending.size(), ids);
|
||||
}
|
||||
|
||||
private static String formatBackendIncidentNudge(PendingIncident incident) {
|
||||
return BACKEND_INCIDENT_NUDGE_FORMAT.formatted(incident.credential(), incident.remainingCoolOffSeconds(),
|
||||
String.join(", ", incident.profiles()), String.join(", ", incident.targets()));
|
||||
}
|
||||
|
||||
private static String formatUnmappedTargetNudge(PendingUnmappedTarget unmappedTarget) {
|
||||
return BACKEND_TARGET_UNMAPPED_NUDGE_FORMAT.formatted(unmappedTarget.key().target(),
|
||||
unmappedTarget.key().reason());
|
||||
}
|
||||
|
||||
// --- lifecycle -----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@@ -627,6 +803,10 @@ public final class ReplyPushLoop {
|
||||
pendingReplies.clear();
|
||||
pendingTickets.clear();
|
||||
pendingQuestions.clear();
|
||||
pendingIncidents.clear();
|
||||
deliveredIncidents.clear();
|
||||
pendingUnmappedTargets.clear();
|
||||
deliveredUnmappedTargets.clear();
|
||||
}
|
||||
|
||||
/** @see #stop() */
|
||||
|
||||
@@ -9,12 +9,19 @@ import java.util.concurrent.CompletableFuture;
|
||||
public final class TurnToken {
|
||||
private final String target;
|
||||
private final CompletableFuture<Rendezvous.Resolution> waiter;
|
||||
private final String injectedText;
|
||||
|
||||
public TurnToken(String target, CompletableFuture<Rendezvous.Resolution> waiter) {
|
||||
this(target, waiter, null);
|
||||
}
|
||||
|
||||
public TurnToken(String target, CompletableFuture<Rendezvous.Resolution> waiter, String injectedText) {
|
||||
this.target = target;
|
||||
this.waiter = waiter;
|
||||
this.injectedText = injectedText;
|
||||
}
|
||||
|
||||
public String target() { return target; }
|
||||
public CompletableFuture<Rendezvous.Resolution> waiter() { return waiter; }
|
||||
public String injectedText() { return injectedText; }
|
||||
}
|
||||
|
||||
@@ -0,0 +1,200 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
* Fleetd #201 / #227: correlates classified backend errors (credential outage or provider 5xx,
|
||||
* produced elsewhere by the classifier that turns raw pane text into a typed event — this class
|
||||
* knows nothing about pane text, profiles, sessions, launchers, or leads) and decides, purely from
|
||||
* counts and timing, when a credential's backend is out.
|
||||
*
|
||||
* <p>Two classified errors on the <b>same credential</b> — never the profile name, never the error
|
||||
* text, see {@code FleetConfig.Profile#effectiveCredentialId()} like {@link BackendQuarantine} —
|
||||
* <b>from two distinct targets</b> inside a 60-second window is treated as an outage: {@link
|
||||
* #record} then returns an {@link Incident} and starts a 60-second cool-off for that credential. The
|
||||
* threshold counts distinct targets, not raw events, on purpose: the classifier is a heuristic and a
|
||||
* valid member report can quote an {@code API Error:} line, so one target repeating that line twice
|
||||
* must never remove fleet capacity by itself. Two different targets independently producing a
|
||||
* classified error is far less likely to be a coincidence, and a real backend outage hits every
|
||||
* target on that credential anyway, so this loses nothing against the case being protected against.
|
||||
* This is deliberately a different store from {@link BackendQuarantine}: that one holds a
|
||||
* 1800-second exhaustion cooldown for a spent credential, and reusing it here would both use the
|
||||
* wrong duration and report "backend exhausted" for what is a short transient fault.
|
||||
*
|
||||
* <p>Correlation and cool-off live in <b>one class</b> so that crossing the threshold and setting
|
||||
* the cool-off deadline happen as a single atomic update: every {@link #record} call goes through
|
||||
* {@link ConcurrentHashMap#compute}, which serializes on the credential's map bucket, so a
|
||||
* concurrent second and third event on the same credential can never both observe "about to cross"
|
||||
* and both mint an incident.
|
||||
*
|
||||
* <p>The clock is injected ({@link LongSupplier}, conventionally {@code System::nanoTime} like
|
||||
* {@link BackendQuarantine}), never read inline, so the window and cool-off are testable without a
|
||||
* real sleep.
|
||||
*/
|
||||
public final class BackendOutagePolicy {
|
||||
|
||||
/** Distinct targets a credential needs a classified error from to declare an outage. */
|
||||
public static final int THRESHOLD = 2;
|
||||
|
||||
/** How long a credential's evidence stays fresh, inclusive of both endpoints. */
|
||||
public static final long WINDOW_NANOS = 60_000_000_000L;
|
||||
|
||||
/** How long a credential sits out once the threshold is crossed. */
|
||||
public static final long COOLOFF_NANOS = 60_000_000_000L;
|
||||
|
||||
private final ConcurrentHashMap<String, CredentialState> states = new ConcurrentHashMap<>();
|
||||
private final LongSupplier nowNanos;
|
||||
private final AtomicLong incidentSequence = new AtomicLong();
|
||||
|
||||
public BackendOutagePolicy(LongSupplier nowNanos) {
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
}
|
||||
|
||||
/**
|
||||
* Records one classified backend error for {@code credentialId} against {@code target} (e.g. a
|
||||
* session or worker id — this class never interprets it, only collects it for the incident's
|
||||
* affected-targets list, and counts distinct targets toward the threshold) with {@code reason}
|
||||
* (the classifier's free-text reason, kept per event for whoever renders the eventual notice —
|
||||
* every event's reason is kept even when the same target repeats, so {@link Incident#reasons()}
|
||||
* can be longer than {@link Incident#evidenceCount()}).
|
||||
*
|
||||
* <p>Returns a populated {@link Incident} only at the exact moment a <b>second distinct target</b>
|
||||
* is seen for this credential inside the window — never before, and never again while the
|
||||
* resulting cool-off is active. A repeat error from a target already counted does not advance the
|
||||
* threshold. While a credential is cooling off, a fresh error is ignored outright: it neither
|
||||
* extends the deadline nor produces another incident. Once the cool-off has elapsed, the next
|
||||
* error clears the old evidence and starts a brand-new window — two fresh, distinct targets are
|
||||
* required to rearm.
|
||||
*/
|
||||
public Optional<Incident> record(String credentialId, String target, String reason) {
|
||||
Objects.requireNonNull(credentialId, "credentialId");
|
||||
Objects.requireNonNull(target, "target");
|
||||
Objects.requireNonNull(reason, "reason");
|
||||
long now = nowNanos.getAsLong();
|
||||
|
||||
AtomicReference<Incident> minted = new AtomicReference<>();
|
||||
states.compute(credentialId, (_, existing) -> {
|
||||
CredentialState state = existing;
|
||||
|
||||
if (state != null && state.inCoolOff()) {
|
||||
if (now < state.coolOffUntilNanos) {
|
||||
return state; // still cooling off: no extension, no incident
|
||||
}
|
||||
state = null; // cool-off elapsed: evidence is cleared, rearm from scratch
|
||||
}
|
||||
|
||||
if (state == null || now - state.firstEventNanos > WINDOW_NANOS) {
|
||||
return CredentialState.first(now, target, reason);
|
||||
}
|
||||
|
||||
CredentialState advanced = state.withAdditionalEvidence(target, reason);
|
||||
if (advanced.evidenceCount() < THRESHOLD) {
|
||||
return advanced;
|
||||
}
|
||||
|
||||
long coolOffUntilNanos = now + COOLOFF_NANOS;
|
||||
minted.set(new Incident(
|
||||
"outage-" + credentialId + "-" + incidentSequence.incrementAndGet(),
|
||||
credentialId,
|
||||
Set.copyOf(advanced.targets),
|
||||
advanced.evidenceCount(),
|
||||
toSecondsRoundedUp(WINDOW_NANOS),
|
||||
toSecondsRoundedUp(COOLOFF_NANOS),
|
||||
List.copyOf(advanced.reasons)));
|
||||
return advanced.enteringCoolOff(coolOffUntilNanos);
|
||||
});
|
||||
|
||||
return Optional.ofNullable(minted.get());
|
||||
}
|
||||
|
||||
/** Seconds left on {@code credentialId}'s cool-off, or empty when it is not cooling off. */
|
||||
public OptionalLong remainingCoolOffSeconds(String credentialId) {
|
||||
Objects.requireNonNull(credentialId, "credentialId");
|
||||
CredentialState state = states.get(credentialId);
|
||||
if (state == null || !state.inCoolOff()) {
|
||||
return OptionalLong.empty();
|
||||
}
|
||||
long remaining = state.coolOffUntilNanos - nowNanos.getAsLong();
|
||||
return remaining > 0 ? OptionalLong.of(toSecondsRoundedUp(remaining)) : OptionalLong.empty();
|
||||
}
|
||||
|
||||
private static long toSecondsRoundedUp(long nanos) {
|
||||
return (nanos + 999_999_999L) / 1_000_000_000L;
|
||||
}
|
||||
|
||||
/**
|
||||
* One credential crossing the outage threshold. {@code remainingCoolOffSeconds} is the cool-off
|
||||
* length as observed at the moment of minting — this incident is only ever produced right as the
|
||||
* cool-off starts, so it is always the full {@link #COOLOFF_NANOS} rounded up.
|
||||
*
|
||||
* <p>{@code evidenceCount} is {@code targets.size()} — the threshold is on distinct targets, not
|
||||
* raw events — while {@code reasons} keeps every event's reason, including repeats from a target
|
||||
* already counted. The two are deliberately different lengths: a single target hammering the same
|
||||
* classified error never grows {@code evidenceCount} past 1, but each occurrence still lands in
|
||||
* {@code reasons} for whoever renders the notice.
|
||||
*/
|
||||
public record Incident(
|
||||
String id,
|
||||
String credentialId,
|
||||
Set<String> targets,
|
||||
int evidenceCount,
|
||||
long windowSeconds,
|
||||
long remainingCoolOffSeconds,
|
||||
List<String> reasons) {
|
||||
}
|
||||
|
||||
/** Evidence accumulated for one credential since its window opened, or its active cool-off. */
|
||||
private static final class CredentialState {
|
||||
final long firstEventNanos;
|
||||
final Set<String> targets;
|
||||
final List<String> reasons;
|
||||
final long coolOffUntilNanos; // 0 means "not cooling off"
|
||||
|
||||
private CredentialState(long firstEventNanos, Set<String> targets, List<String> reasons,
|
||||
long coolOffUntilNanos) {
|
||||
this.firstEventNanos = firstEventNanos;
|
||||
this.targets = targets;
|
||||
this.reasons = reasons;
|
||||
this.coolOffUntilNanos = coolOffUntilNanos;
|
||||
}
|
||||
|
||||
static CredentialState first(long nowNanos, String target, String reason) {
|
||||
Set<String> targets = new LinkedHashSet<>();
|
||||
targets.add(target);
|
||||
List<String> reasons = new ArrayList<>();
|
||||
reasons.add(reason);
|
||||
return new CredentialState(nowNanos, targets, reasons, 0L);
|
||||
}
|
||||
|
||||
CredentialState withAdditionalEvidence(String target, String reason) {
|
||||
Set<String> newTargets = new LinkedHashSet<>(targets);
|
||||
newTargets.add(target);
|
||||
List<String> newReasons = new ArrayList<>(reasons);
|
||||
newReasons.add(reason);
|
||||
return new CredentialState(firstEventNanos, newTargets, newReasons, 0L);
|
||||
}
|
||||
|
||||
CredentialState enteringCoolOff(long coolOffUntilNanos) {
|
||||
return new CredentialState(firstEventNanos, targets, reasons, coolOffUntilNanos);
|
||||
}
|
||||
|
||||
/** Distinct targets seen so far — the threshold counts this, never {@code reasons.size()}. */
|
||||
int evidenceCount() {
|
||||
return targets.size();
|
||||
}
|
||||
|
||||
boolean inCoolOff() {
|
||||
return coolOffUntilNanos > 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,55 +1,68 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Backward-compatible placement: an unqualified spawn always resolves to the configured default
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. This ignores caps and
|
||||
* reachability so that a pre-existing config behaves identically after upgrade.
|
||||
*
|
||||
* <p>Two exceptions walk past the default instead of returning it unconditionally:
|
||||
* <p>Three exceptions walk past the default instead of returning it unconditionally:
|
||||
* <ul>
|
||||
* <li>Quarantine (CB-578 stage B): a quarantined default is a credential that just refused on
|
||||
* a usage limit, not a transient capacity or reachability concern.
|
||||
* <li>Cooling off (fleetd #201 Unit 5): a credential cooling off after repeated backend errors
|
||||
* ({@code BackendOutagePolicy}) — a separate, shorter-lived source from quarantine. When a
|
||||
* profile is both quarantined and cooling off, only the quarantine reason is reported
|
||||
* (exhaustion takes priority), matching {@code CompositePeerLauncher}'s explicit-spawn order.
|
||||
* <li>Weight 0 (CB-554): {@code fixed} is still automatic selection, so a profile the operator
|
||||
* marked "never auto-select me" ({@code weight <= 0}) must be skipped here exactly as
|
||||
* {@code weighted}/{@code round-robin} skip it — an explicit {@code fleet_spawn} naming
|
||||
* the profile is unaffected, only this automatic fallback walk.
|
||||
* </ul>
|
||||
* A fleet where nothing is ever quarantined or weight-0 never exercises either path, so today's
|
||||
* behaviour is unchanged.
|
||||
* A fleet where nothing is ever quarantined, cooling off, or weight-0 never exercises any of these
|
||||
* paths, so today's behaviour is unchanged.
|
||||
*/
|
||||
final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
|
||||
@Override
|
||||
public PlacementCandidate select(PlacementContext ctx) {
|
||||
String d = ctx.defaultProfile();
|
||||
if (d != null && !d.isBlank() && !ctx.quarantined().contains(d) && !weightExcluded(ctx, d)) {
|
||||
if (d != null && !d.isBlank() && !ctx.quarantined().contains(d) && !ctx.coolingOff().contains(d)
|
||||
&& !weightExcluded(ctx, d)) {
|
||||
return new PlacementCandidate(d, null, 1.0f, null);
|
||||
}
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (!ctx.quarantined().contains(c.profile()) && !c.excluded()) {
|
||||
if (!ctx.quarantined().contains(c.profile()) && !ctx.coolingOff().contains(c.profile())
|
||||
&& !c.excluded()) {
|
||||
return new PlacementCandidate(c.profile(), null, c.weight(), c.maxLoad());
|
||||
}
|
||||
}
|
||||
if (d != null && !d.isBlank()) {
|
||||
boolean dQuarantined = ctx.quarantined().contains(d);
|
||||
// Exhaustion quarantine takes priority: reported only when quarantine is absent, so the
|
||||
// message never claims "cooling off" for a profile that is really backend-exhausted.
|
||||
boolean dCoolingOff = !dQuarantined && ctx.coolingOff().contains(d);
|
||||
boolean dWeightExcluded = weightExcluded(ctx, d);
|
||||
if (dQuarantined && dWeightExcluded) {
|
||||
throw new PlacementException("worker profile '" + d + "' is quarantined (backend "
|
||||
+ "exhausted) and has weight 0 (excluded from automatic selection), and no "
|
||||
+ "available candidate remains");
|
||||
}
|
||||
if (dWeightExcluded) {
|
||||
throw new PlacementException("worker profile '" + d + "' has weight 0 (excluded "
|
||||
+ "from automatic selection) and no available candidate remains");
|
||||
}
|
||||
if (dQuarantined) {
|
||||
throw new PlacementException("worker profile '" + d + "' is quarantined (backend "
|
||||
+ "exhausted) and no un-quarantined candidate is available");
|
||||
if (dQuarantined || dCoolingOff || dWeightExcluded) {
|
||||
List<String> reasons = new ArrayList<>();
|
||||
if (dQuarantined) {
|
||||
reasons.add("is quarantined (backend exhausted)");
|
||||
}
|
||||
if (dCoolingOff) {
|
||||
reasons.add("is cooling off after repeated backend errors");
|
||||
}
|
||||
if (dWeightExcluded) {
|
||||
reasons.add("has weight 0 (excluded from automatic selection)");
|
||||
}
|
||||
throw new PlacementException("worker profile '" + d + "' "
|
||||
+ String.join(" and ", reasons) + ", and no available candidate remains");
|
||||
}
|
||||
}
|
||||
if (!ctx.candidates().isEmpty()) {
|
||||
throw new PlacementException(
|
||||
"all worker profiles are excluded from automatic selection (quarantined or weight-0)");
|
||||
throw new PlacementException("all worker profiles are excluded from automatic "
|
||||
+ "selection (quarantined, cooling off, or weight-0)");
|
||||
}
|
||||
throw new PlacementException("no worker profiles configured");
|
||||
}
|
||||
|
||||
@@ -14,10 +14,19 @@ import java.util.function.Function;
|
||||
* @param quarantined profiles whose credential is currently quarantined (CB-578 stage B) — a
|
||||
* {@code BACKEND_EXHAUSTED} classification put it, or a profile it shares a
|
||||
* credential with, on cooldown. Filtered the same way as {@code unreachable}.
|
||||
* @param coolingOff profiles whose credential is currently cooling off after repeated backend
|
||||
* errors (fleetd #201 Unit 5 — {@code BackendOutagePolicy}), a SEPARATE,
|
||||
* shorter-lived source from {@code quarantined}: a credential outage cools off
|
||||
* even when no member was ever exhausted. Deliberately its own set rather than
|
||||
* merged into {@code quarantined} — {@link PlacementPolicyUtil} needs to tell
|
||||
* the two apart so its refusal message says "cooling off", not "exhausted",
|
||||
* when only this one is active. A profile can be in both sets at once; when it
|
||||
* is, exhaustion quarantine is reported (it takes priority).
|
||||
*/
|
||||
public record PlacementContext(String defaultProfile,
|
||||
List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount,
|
||||
Set<String> unreachable,
|
||||
Set<String> quarantined) {
|
||||
Set<String> quarantined,
|
||||
Set<String> coolingOff) {
|
||||
}
|
||||
|
||||
@@ -14,14 +14,16 @@ final class PlacementPolicyUtil {
|
||||
/**
|
||||
* Candidates that are not weight-excluded (CB-554: explicit {@code weight <= 0}, checked
|
||||
* first because it is a static config choice rather than transient state), not
|
||||
* known-unreachable, not quarantined (CB-578 stage B), and have not reached their maxLoad.
|
||||
* A {@code null} maxLoad means unlimited.
|
||||
* known-unreachable, not quarantined (CB-578 stage B), not cooling off after repeated backend
|
||||
* errors (fleetd #201 Unit 5 — a separate, shorter-lived source from quarantine), and have not
|
||||
* reached their maxLoad. A {@code null} maxLoad means unlimited.
|
||||
*/
|
||||
static List<PlacementCandidate> available(PlacementContext ctx) {
|
||||
List<PlacementCandidate> out = new ArrayList<>();
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (c.excluded() || ctx.unreachable().contains(c.profile())
|
||||
|| ctx.quarantined().contains(c.profile())) {
|
||||
|| ctx.quarantined().contains(c.profile())
|
||||
|| ctx.coolingOff().contains(c.profile())) {
|
||||
continue;
|
||||
}
|
||||
Integer cap = c.maxLoad();
|
||||
@@ -38,21 +40,27 @@ final class PlacementPolicyUtil {
|
||||
|
||||
/**
|
||||
* Build a clear exception describing why every candidate was dropped: all weight-0, all
|
||||
* quarantined, all at capacity, all unreachable, or a mix. Each candidate is counted into
|
||||
* exactly one bucket (weight-excluded takes priority) so a candidate excluded for more than
|
||||
* one reason is never double-counted.
|
||||
* quarantined, all cooling off, all at capacity, all unreachable, or a mix. Each candidate is
|
||||
* counted into exactly one bucket (weight-excluded first, then quarantined, then cooling off)
|
||||
* so a candidate excluded for more than one reason is never double-counted — a candidate that is
|
||||
* both quarantined (CB-578 stage B, backend exhausted) and cooling off (fleetd #201 Unit 5,
|
||||
* repeated backend errors) counts only as quarantined, matching {@code CompositePeerLauncher}'s
|
||||
* explicit-spawn ordering: exhaustion quarantine takes priority when both are active.
|
||||
*/
|
||||
static PlacementException emptyException(PlacementContext ctx) {
|
||||
int weightExcluded = 0;
|
||||
int atCap = 0;
|
||||
int unreachable = 0;
|
||||
int quarantined = 0;
|
||||
int coolingOff = 0;
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
Integer cap = c.maxLoad();
|
||||
if (c.excluded()) {
|
||||
weightExcluded++;
|
||||
} else if (ctx.quarantined().contains(c.profile())) {
|
||||
quarantined++;
|
||||
} else if (ctx.coolingOff().contains(c.profile())) {
|
||||
coolingOff++;
|
||||
} else if (ctx.unreachable().contains(c.profile())) {
|
||||
unreachable++;
|
||||
} else if (cap != null && ctx.liveCount().apply(c.profile()) >= cap) {
|
||||
@@ -71,6 +79,10 @@ final class PlacementPolicyUtil {
|
||||
if (quarantined == total) {
|
||||
return new PlacementException("all worker profiles are quarantined (backend exhausted)");
|
||||
}
|
||||
if (coolingOff == total) {
|
||||
return new PlacementException(
|
||||
"all worker profiles are cooling off after repeated backend errors");
|
||||
}
|
||||
if (atCap == total) {
|
||||
return new PlacementException("all worker profiles are at maxLoad");
|
||||
}
|
||||
@@ -79,7 +91,9 @@ final class PlacementPolicyUtil {
|
||||
}
|
||||
return new PlacementException("no worker profile available: " + atCap + " at maxLoad, "
|
||||
+ unreachable + " unreachable, " + quarantined + " quarantined, "
|
||||
+ coolingOff + " cooling off, "
|
||||
+ weightExcluded + " weight-0, "
|
||||
+ (total - atCap - unreachable - quarantined - weightExcluded) + " remaining");
|
||||
+ (total - atCap - unreachable - quarantined - coolingOff - weightExcluded)
|
||||
+ " remaining");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,14 +9,17 @@ import dev.ltms.fleet.auth.Principal;
|
||||
import dev.ltms.fleet.guard.GuardException;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.inject.MemberPresence;
|
||||
import dev.ltms.fleet.member.MemberCredentialPolicyView;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.session.ShuttingDownException;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.WorktreeRequest;
|
||||
@@ -32,6 +35,7 @@ import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
@@ -45,6 +49,22 @@ import java.util.stream.Collectors;
|
||||
*/
|
||||
public final class FleetApp {
|
||||
|
||||
/** The authorization action the matching route handler hands to {@link #allow}. */
|
||||
static Authz.Action routeAction(String route) {
|
||||
return switch (route) {
|
||||
case "GET /metrics" -> Authz.Action.METRICS;
|
||||
case "POST /members" -> Authz.Action.SPAWN;
|
||||
case "DELETE /members/{paneId}" -> Authz.Action.STOP;
|
||||
case "POST /sessions/{id}/message" -> Authz.Action.SEND;
|
||||
case "POST /sessions/{id}/reply" -> Authz.Action.REPLY;
|
||||
case "GET /sessions/{id}/replies" -> Authz.Action.DRAIN;
|
||||
case "POST /sessions/{id}/ask" -> Authz.Action.ASK;
|
||||
case "GET /sessions", "GET /agents", "GET /members", "GET /profiles",
|
||||
"GET /member-credentials", "GET /sessions/{id}/status", "GET /tasks/{ticket}" -> Authz.Action.READ;
|
||||
default -> throw new IllegalArgumentException("route has no authorization gate: " + route);
|
||||
};
|
||||
}
|
||||
|
||||
/** Default blocking window for a message; kept under typical HTTP idle timeouts. */
|
||||
private static final long DEFAULT_MESSAGE_TIMEOUT_MS = 25_000;
|
||||
private static final long MAX_MESSAGE_TIMEOUT_MS = 120_000;
|
||||
@@ -55,7 +75,8 @@ public final class FleetApp {
|
||||
/** Context attribute under which the resolved caller is stashed by the auth filter. */
|
||||
private static final String CALLER = "fleetd.caller";
|
||||
|
||||
private final HerdrClient herdr;
|
||||
private final HerdrClient herdr; // lead daemon
|
||||
private final HerdrClient memberHerdr; // CB-185: member daemon (same object when unconfigured)
|
||||
private final PeerLauncher workers;
|
||||
private final SessionManager sessions; // CB-301: authoritative session registry
|
||||
private final MessageService messages;
|
||||
@@ -63,6 +84,16 @@ public final class FleetApp {
|
||||
private final HttpServlet mcpServlet; // MCP Streamable-HTTP endpoint, mounted at /mcp (nullable)
|
||||
private final CallerResolver auth; // CB-501: null → authz not enforced (legacy behaviour)
|
||||
private final Metrics metrics; // CB-502: null → /metrics not exposed
|
||||
// fleetd #111: re-read per request, same hot-reload shape as every other live config read —
|
||||
// absent() (the honest "no policy configured" view) for every constructor that does not wire
|
||||
// a real one, so existing legacy call sites keep building without knowing this field exists.
|
||||
private final Supplier<MemberCredentialPolicyView> memberCredentials;
|
||||
// fleetd #297: the SAME shared sources FleetMcp.profiles/fleet_profiles reads (BackendQuarantine
|
||||
// and BackendOutagePolicy are each one instance for the whole daemon — see Fleetd wiring) so
|
||||
// GET /profiles cannot drift from fleet_profiles about which profile is quarantined/cooling off.
|
||||
// .none() (the honest "feature not wired" view) for every constructor that does not pass one.
|
||||
private final FleetMcp.QuarantineSource quarantine;
|
||||
private final FleetMcp.OutageSource outage;
|
||||
private final ObjectMapper mapper = new ObjectMapper();
|
||||
|
||||
/**
|
||||
@@ -95,7 +126,53 @@ public final class FleetApp {
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable) {
|
||||
this(herdr, herdr, workers, sessions, messages, presence, mcpServlet, auth, metrics, deliverable);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param herdr the lead daemon's client
|
||||
* @param memberHerdr the member daemon's client (CB-185); pass the same instance as
|
||||
* {@code herdr} for a single-daemon deployment — {@code healthz}/{@code
|
||||
* sessions} then make exactly one herdr call each, unchanged from before
|
||||
* the two-daemon router existed
|
||||
* @param deliverable the injector's readiness gate, shared so status reports its real result
|
||||
*/
|
||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable) {
|
||||
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics,
|
||||
deliverable, MemberCredentialPolicyView::absent);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param memberCredentials live {@code memberCredentials:} policy view (fleetd #111), re-read
|
||||
* per request for {@code GET /member-credentials}; production wiring
|
||||
* passes the same hot-reload shape as every other live config read
|
||||
*/
|
||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials) {
|
||||
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics,
|
||||
deliverable, memberCredentials, FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* @param quarantine the SAME {@link FleetMcp.QuarantineSource} instance passed to {@code
|
||||
* FleetMcp} (fleetd #297), so {@code GET /profiles} reports the identical
|
||||
* exhaustion-quarantine facts as {@code fleet_profiles} rather than a second,
|
||||
* independently-computed copy
|
||||
* @param outage the SAME {@link FleetMcp.OutageSource} instance passed to {@code FleetMcp} —
|
||||
* see {@code quarantine}; a SEPARATE check from it, never merged in
|
||||
*/
|
||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials,
|
||||
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage) {
|
||||
this.herdr = herdr;
|
||||
this.memberHerdr = memberHerdr != null ? memberHerdr : herdr;
|
||||
this.workers = workers;
|
||||
this.sessions = sessions;
|
||||
this.messages = messages;
|
||||
@@ -103,6 +180,9 @@ public final class FleetApp {
|
||||
this.mcpServlet = mcpServlet;
|
||||
this.auth = auth;
|
||||
this.metrics = metrics;
|
||||
this.memberCredentials = memberCredentials != null ? memberCredentials : MemberCredentialPolicyView::absent;
|
||||
this.quarantine = quarantine != null ? quarantine : FleetMcp.QuarantineSource.none();
|
||||
this.outage = outage != null ? outage : FleetMcp.OutageSource.none();
|
||||
}
|
||||
|
||||
/** Wire routes onto a fresh, unstarted Javalin instance. Caller starts it. */
|
||||
@@ -131,6 +211,7 @@ public final class FleetApp {
|
||||
app.get("/agents", this::agents);
|
||||
app.get("/members", this::listMembers); // CB-304: registry roster + live herdr status
|
||||
app.get("/profiles", this::profiles); // configured backend profiles
|
||||
app.get("/member-credentials", this::memberCredentials); // fleetd #111: policy names + counts, never a value
|
||||
app.post("/members", this::spawnMember); // optional ?role=&profile= or {"role":…,"profile":…}
|
||||
app.delete("/members/{paneId}", this::stopMember);
|
||||
app.post("/sessions/{id}/message", this::sendMessage); // fleet_send (primary; blocking, wait:false, or answer via turnId)
|
||||
@@ -184,36 +265,91 @@ public final class FleetApp {
|
||||
|
||||
/** Prometheus scrape endpoint (CB-502). */
|
||||
private void metrics(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.METRICS, null)) {
|
||||
if (!allow(ctx, routeAction("GET /metrics"), null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).contentType("text/plain; version=0.0.4; charset=utf-8").result(metrics.render());
|
||||
}
|
||||
|
||||
/** Liveness + herdr reachability. 200 when herdr answers ping, 503 otherwise. */
|
||||
/**
|
||||
* Liveness + herdr reachability. 200 only when BOTH daemons answer ping — 503 otherwise
|
||||
* (CB-185). With no {@code memberHerdrSocket} configured {@code memberHerdr == herdr}, so this
|
||||
* makes exactly the one {@code ping} call it always did and reports the same body; with a
|
||||
* second daemon configured, a member daemon that is down must not be masked by a healthy lead
|
||||
* daemon — every spawn goes through the member daemon and would otherwise fail silently behind
|
||||
* a green {@code /healthz}.
|
||||
*
|
||||
* <p>CB-185 blocker 2: the {@code herdr} key always carries the <em>lead</em> daemon's
|
||||
* version/protocol, unchanged, because two consumers — {@code scripts/redeploy-fleetd.sh} and
|
||||
* {@code scripts/rename-checkout.sh} — read this endpoint already (both only check the HTTP
|
||||
* status code and print the body verbatim; neither parses a specific field, so adding a key
|
||||
* alongside {@code herdr} is safe). But it is the <em>member</em> daemon's protocol that decides
|
||||
* whether a spawn works, so when a second daemon is configured its version/protocol is reported
|
||||
* too, under a separate {@code member} key — never folded into {@code herdr}, which would make a
|
||||
* mismatch invisible to whichever consumer only reads that key. If the two protocol numbers
|
||||
* differ, {@code protocolMismatch: true} calls it out explicitly rather than leaving it to be
|
||||
* spotted by comparing two numbers by eye.
|
||||
*/
|
||||
private void healthz(Context ctx) {
|
||||
JsonNode pong;
|
||||
try {
|
||||
JsonNode pong = herdr.call("ping");
|
||||
ctx.status(200).json(Map.of(
|
||||
"status", "ok",
|
||||
"herdr", Map.of(
|
||||
"version", pong.path("version").asText(""),
|
||||
"protocol", pong.path("protocol").asInt())));
|
||||
pong = herdr.call("ping");
|
||||
} catch (HerdrException e) {
|
||||
ctx.status(503).json(Map.of(
|
||||
"status", "degraded",
|
||||
"herdr", "unreachable",
|
||||
"detail", e.getMessage()));
|
||||
}
|
||||
}
|
||||
|
||||
/** Sessions view derived from herdr {@code workspace.list} (one workspace → one row). */
|
||||
private void sessions(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
return;
|
||||
}
|
||||
JsonNode result = herdr.call("workspace.list");
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
body.put("status", "ok");
|
||||
body.put("herdr", Map.of(
|
||||
"version", pong.path("version").asText(""),
|
||||
"protocol", pong.path("protocol").asInt()));
|
||||
if (memberHerdr != herdr) {
|
||||
JsonNode memberPong;
|
||||
try {
|
||||
memberPong = memberHerdr.call("ping");
|
||||
} catch (HerdrException e) {
|
||||
ctx.status(503).json(Map.of(
|
||||
"status", "degraded",
|
||||
"herdr", "member unreachable",
|
||||
"detail", e.getMessage()));
|
||||
return;
|
||||
}
|
||||
int leadProtocol = pong.path("protocol").asInt();
|
||||
int memberProtocol = memberPong.path("protocol").asInt();
|
||||
body.put("member", Map.of(
|
||||
"version", memberPong.path("version").asText(""),
|
||||
"protocol", memberProtocol));
|
||||
if (leadProtocol != memberProtocol) {
|
||||
body.put("protocolMismatch", true);
|
||||
}
|
||||
}
|
||||
ctx.status(200).json(body);
|
||||
}
|
||||
|
||||
/**
|
||||
* Sessions view derived from herdr {@code workspace.list} (one workspace → one row), merged
|
||||
* across both daemons (CB-185). With no {@code memberHerdrSocket} configured {@code
|
||||
* memberHerdr == herdr}, so this calls {@code workspace.list} exactly once, same as before the
|
||||
* router existed; with a second daemon configured, calling it twice would silently drop every
|
||||
* member workspace (they live on the member daemon only).
|
||||
*/
|
||||
private void sessions(Context ctx) {
|
||||
if (!allow(ctx, routeAction("GET /sessions"), null)) {
|
||||
return;
|
||||
}
|
||||
List<Map<String, Object>> out = new ArrayList<>();
|
||||
collectSessions(herdr, out);
|
||||
if (memberHerdr != herdr) {
|
||||
collectSessions(memberHerdr, out);
|
||||
}
|
||||
ctx.status(200).json(Map.of("sessions", out));
|
||||
}
|
||||
|
||||
private static void collectSessions(HerdrClient client, List<Map<String, Object>> out) {
|
||||
JsonNode result = client.call("workspace.list");
|
||||
for (JsonNode w : result.path("workspaces")) {
|
||||
out.add(Map.of(
|
||||
"id", w.path("workspace_id").asText(""),
|
||||
@@ -222,49 +358,103 @@ public final class FleetApp {
|
||||
"paneCount", w.path("pane_count").asInt(),
|
||||
"agentStatus", w.path("agent_status").asText("unknown")));
|
||||
}
|
||||
ctx.status(200).json(Map.of("sessions", out));
|
||||
}
|
||||
|
||||
/** Discovery: every agent herdr tracks, keyed by its Claude session UUID. */
|
||||
private void agents(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /agents"), null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).json(Map.of("agents",
|
||||
workers.list().stream().map(Agent.class::cast).map(FleetApp::view).toList()));
|
||||
try {
|
||||
ctx.status(200).json(Map.of("agents",
|
||||
workers.list().stream().map(Agent.class::cast).map(FleetApp::view).toList()));
|
||||
} catch (HerdrException e) {
|
||||
// fleetd #297: workers.list() reaches herdr — a transport failure must land in the same
|
||||
// {error, detail} envelope every other failure path here uses, not escape as a bare
|
||||
// exception and leave Javalin's default handling to respond outside the JSON contract.
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** CB-304: bridge-owned roster merged with live herdr status by paneId. */
|
||||
private void listMembers(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /members"), null)) {
|
||||
return;
|
||||
}
|
||||
// CB-519: the registry key is a host-unique id, not the pane coordinate — join on terminal.
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.map(Agent.class::cast)
|
||||
.filter(a -> a.terminalId() != null)
|
||||
.collect(Collectors.toMap(Agent::terminalId, Function.identity(), (_, b) -> b));
|
||||
List<Map<String, Object>> out = sessions.roster().stream()
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.terminalId())))
|
||||
.toList();
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
body.put("workers", out);
|
||||
// CB-586: operator visibility for the refs/wip snapshot store without shelling into the
|
||||
// repo — how many snapshot refs exist and roughly what they cost. Present only once a
|
||||
// worktree session has established the repo, so a never-snapshotted fleet reports nothing.
|
||||
sessions.wipRefs().ifPresent(st -> body.put("wipRefs",
|
||||
Map.of("count", st.count(), "costBytes", st.costBytes())));
|
||||
ctx.status(200).json(body);
|
||||
try {
|
||||
// CB-519: the registry key is a host-unique id, not the pane coordinate — join on terminal.
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.map(Agent.class::cast)
|
||||
.filter(a -> a.terminalId() != null)
|
||||
.collect(Collectors.toMap(Agent::terminalId, Function.identity(), (_, b) -> b));
|
||||
// fleetd #209: this REST roster reports agentSessionId via SessionManager.rosterView, so it
|
||||
// uses the resolving roster read (caller-driven, not a timer) rather than the plain one.
|
||||
List<Map<String, Object>> out = sessions.rosterResolved().stream()
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.terminalId())))
|
||||
.toList();
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
// fleetd #199: the endpoint became /members in the CB-634 rename but the body key stayed
|
||||
// "workers", so a caller that read "members" saw an empty fleet and reported no members at
|
||||
// all. "members" is the canonical key; "workers" stays as a deprecated alias so an existing
|
||||
// REST consumer keeps working — the out-of-band path a lead falls back to when its MCP mount
|
||||
// drops reads this endpoint. Drop the alias once nothing reads it.
|
||||
body.put("members", out);
|
||||
body.put("workers", out);
|
||||
// CB-586: operator visibility for the refs/wip snapshot store without shelling into the
|
||||
// repo — how many snapshot refs exist and roughly what they cost. Present only once a
|
||||
// worktree session has established the repo, so a never-snapshotted fleet reports nothing.
|
||||
sessions.wipRefs().ifPresent(st -> body.put("wipRefs",
|
||||
Map.of("count", st.count(), "costBytes", st.costBytes())));
|
||||
ctx.status(200).json(body);
|
||||
} catch (HerdrException e) {
|
||||
// fleetd #297: same reasoning as agents() above — this is the out-of-band roster a lead
|
||||
// falls back to when its MCP mount drops, so it must stay inside the JSON error contract
|
||||
// exactly when herdr is briefly unreachable, not escape as a bare exception.
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** The configured worker profiles and which one a no-argument spawn uses. */
|
||||
/**
|
||||
* The configured worker profiles, which one a no-argument spawn uses, and (fleetd #297) the two
|
||||
* outage states {@code fleet_profiles} already reports: {@code quarantined} (CB-578 stage B —
|
||||
* the backend reported it out of capacity) and {@code coolingOff} (fleetd #201 Unit 5 — the
|
||||
* credential threw repeated non-exhaustion backend errors). Both are read from the SAME shared
|
||||
* {@link FleetMcp.QuarantineSource}/{@link FleetMcp.OutageSource} instances {@code FleetMcp}
|
||||
* reads, never recomputed, so the two doors cannot disagree about which profile is down and why.
|
||||
* Independent checks, so a profile can appear in both maps at once; each map is present only
|
||||
* when at least one profile is in that state.
|
||||
*/
|
||||
private void profiles(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /profiles"), null)) {
|
||||
return;
|
||||
}
|
||||
// fleetd #297: ONE body builder, shared with fleet_profiles. Handing both doors the same
|
||||
// QuarantineSource/OutageSource instances stops them reading different facts; rendering
|
||||
// through the same method stops them reporting those facts differently. Both are needed.
|
||||
ctx.status(200).json(FleetMcp.profilesView(workers, quarantine, outage));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #111 (CB-608): the live {@code memberCredentials:} policy as names and counts —
|
||||
* NEVER a value. The daemon does not hold a credential's value in the first place (only the
|
||||
* name it is configured under), so there is nothing to redact here beyond what {@link
|
||||
* MemberCredentialPolicyView} already omits by construction. This is the source
|
||||
* {@code scripts/probe-member-credentials.sh} reads instead of carrying its own hardcoded
|
||||
* name list, which is exactly what let the list drift silently behind the real policy.
|
||||
*/
|
||||
private void memberCredentials(Context ctx) {
|
||||
if (!allow(ctx, routeAction("GET /member-credentials"), null)) {
|
||||
return;
|
||||
}
|
||||
MemberCredentialPolicyView view = memberCredentials.get();
|
||||
ctx.status(200).json(Map.of(
|
||||
"profiles", workers.profiles(),
|
||||
"default", workers.defaultProfile() == null ? "" : workers.defaultProfile()));
|
||||
"present", view.present(),
|
||||
"policy", view.policy() == null ? "" : view.policy(),
|
||||
"known", view.known(),
|
||||
"allowed", view.allowed(),
|
||||
"knownCount", view.knownCount(),
|
||||
"allowedCount", view.allowedCount(),
|
||||
"blockedCount", view.blockedCount()));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -273,7 +463,7 @@ public final class FleetApp {
|
||||
* the subscription boundary, 400 for an unknown profile.
|
||||
*/
|
||||
private void spawnMember(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.SPAWN, null)) {
|
||||
if (!allow(ctx, routeAction("POST /members"), null)) {
|
||||
return;
|
||||
}
|
||||
String role = ctx.queryParam("role");
|
||||
@@ -312,6 +502,11 @@ public final class FleetApp {
|
||||
ctx.status(201).json(view(member));
|
||||
} catch (GuardException e) {
|
||||
ctx.status(403).json(Map.of("error", "subscription_boundary", "detail", e.getMessage()));
|
||||
} catch (ShuttingDownException e) {
|
||||
// fleetd #308: the daemon's shutdown drain has already started — 503, not a bare 500,
|
||||
// so this reads the same as PlacementException below: valid request, refused because
|
||||
// of a transient daemon state rather than a bad argument.
|
||||
ctx.status(503).json(Map.of("error", "shutting_down", "detail", e.getMessage()));
|
||||
} catch (PlacementException e) {
|
||||
// CB-599: no candidate had capacity (maxLoad, quarantine, or all-exhausted) — a benign,
|
||||
// likely-transient refusal, distinct from "profile does not exist" below. 503: the
|
||||
@@ -321,6 +516,12 @@ public final class FleetApp {
|
||||
ctx.status(400).json(Map.of("error", "unknown_profile", "detail", e.getMessage()));
|
||||
} catch (PeerUnreachableException e) {
|
||||
ctx.status(502).json(Map.of("error", "spawn_timeout", "detail", e.getMessage()));
|
||||
} catch (HerdrException e) {
|
||||
// fleetd #304: not every herdr failure on the spawn path is a readiness timeout, so
|
||||
// PeerUnreachableException above does not cover this. Without this catch the exception
|
||||
// escapes to Javalin's default 500, while fleet_spawn reports the same failure as a
|
||||
// clean named error (FleetMcp.spawn) — the #297 one-door-guarded shape.
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -341,13 +542,29 @@ public final class FleetApp {
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
}
|
||||
|
||||
/** Tear a worker down by pane id. */
|
||||
/**
|
||||
* Tear a worker down by pane id.
|
||||
*
|
||||
* <p>fleetd #304: the {@code HerdrException} catch is not cosmetic. {@code release} deregisters
|
||||
* the session, notifies the release listener and preserves a dirty worktree <em>before</em> it
|
||||
* calls {@code launcher.stop}, so a throw from that stop arrives after the teardown the caller
|
||||
* asked for has already happened. Letting it escape gave Javalin's default 500, which tells the
|
||||
* caller to retry — and the retry finds nothing in the registry, reaches the same stop, and
|
||||
* throws again, so it can never succeed. {@code herdrError} instead answers 404 ("the pane is
|
||||
* gone, stop retrying") or 502 ("herdr is upstream and broken, a retry may help"), matching what
|
||||
* {@code fleet_stop} reports for the same failure.
|
||||
*/
|
||||
private void stopMember(Context ctx) {
|
||||
String paneId = ctx.pathParam("paneId");
|
||||
if (!allow(ctx, Authz.Action.STOP, paneId)) {
|
||||
if (!allow(ctx, routeAction("DELETE /members/{paneId}"), paneId)) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
sessions.release(paneId);
|
||||
} catch (HerdrException e) {
|
||||
herdrError(ctx, e);
|
||||
return;
|
||||
}
|
||||
sessions.release(paneId);
|
||||
ctx.status(204);
|
||||
}
|
||||
|
||||
@@ -359,7 +576,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void sendMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.SEND, id)) {
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/message"), id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
@@ -446,7 +663,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void askMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.ASK, id)) {
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/ask"), id)) {
|
||||
return;
|
||||
}
|
||||
String question;
|
||||
@@ -485,7 +702,7 @@ public final class FleetApp {
|
||||
// The rule that matters: a worker may reply only as itself. Over MCP this was already true
|
||||
// structurally (identity comes from the connection, never an argument); over REST the path
|
||||
// id was simply trusted, so this is where the invariant actually gets enforced.
|
||||
if (!allow(ctx, Authz.Action.REPLY, id)) {
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/reply"), id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
@@ -495,7 +712,19 @@ public final class FleetApp {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "body must be JSON"));
|
||||
return;
|
||||
}
|
||||
messages.reply(id, content);
|
||||
// fleetd #302: content is required. `.path("content").asText("")` above turns a missing key
|
||||
// into "" rather than throwing, so without this check an empty/blank reply used to reach
|
||||
// messages.reply(...) and silently resolve the lead's waiter — the same class of bug as the
|
||||
// sibling "content is required" guards on sendMessage/askMessage below, except this one wrote
|
||||
// a WRONG value instead of failing loudly. The check lives in MessageService.reply so both
|
||||
// this door and FleetMcp.reply inherit the same rule; this catch only translates it into the
|
||||
// {error, detail} envelope this file uses everywhere else.
|
||||
try {
|
||||
messages.reply(id, content);
|
||||
} catch (IllegalArgumentException e) {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", e.getMessage()));
|
||||
return;
|
||||
}
|
||||
ctx.status(200).json(Map.of("sessionId", id, "delivered", true));
|
||||
}
|
||||
|
||||
@@ -506,7 +735,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void drainReplies(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.DRAIN, id)) {
|
||||
if (!allow(ctx, routeAction("GET /sessions/{id}/replies"), id)) {
|
||||
return;
|
||||
}
|
||||
var replies = messages.drainReplies(id);
|
||||
@@ -524,7 +753,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void sessionStatus(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.READ, id)) {
|
||||
if (!allow(ctx, routeAction("GET /sessions/{id}/status"), id)) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
@@ -549,7 +778,7 @@ public final class FleetApp {
|
||||
|
||||
/** Poll an async (wait:false) delegation by ticket. 404 for an unknown/expired ticket. */
|
||||
private void taskStatus(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /tasks/{ticket}"), null)) {
|
||||
return;
|
||||
}
|
||||
MessageService.TaskView v = messages.poll(ctx.pathParam("ticket"));
|
||||
|
||||
@@ -13,6 +13,7 @@ import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.nio.file.attribute.PosixFilePermissions;
|
||||
import java.security.SecureRandom;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
@@ -23,6 +24,7 @@ import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.Function;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
@@ -88,28 +90,80 @@ public final class GitWorktrees implements Worktrees {
|
||||
);
|
||||
|
||||
private final String configuredRoot;
|
||||
/** OS group name for {@link #shareWithGroup} (fleetd #185 stage 3); {@code null} ⇒ feature off. */
|
||||
private final String group;
|
||||
private final Consumer<String> afterWorktreeAdded;
|
||||
/** How the initial {@code git worktree add} command runs. Package-private test seam for an
|
||||
* interrupted command after Git has made worktree state. */
|
||||
private final Function<String[], String> worktreeAddRunner;
|
||||
/** How {@link #shareWithGroup}'s processes (git config / chgrp / chmod / find) actually run.
|
||||
* Defaults to the real {@link #exec(String...)}. Package-private test seam so a unit test can
|
||||
* prove "no group configured ⇒ zero processes spawned" and inspect exactly what a configured
|
||||
* group runs, without a real second OS user or OS group on this host. */
|
||||
private final Function<String[], String> shareGroupRunner;
|
||||
private final SecureRandom random = new SecureRandom();
|
||||
private final AtomicLong seq = new AtomicLong();
|
||||
|
||||
/** Default constructor: worktree root is derived per-repo as {@code <repoRoot>/../.bridged-worktrees}. */
|
||||
public GitWorktrees() {
|
||||
this(null);
|
||||
this(null, (String) null);
|
||||
}
|
||||
|
||||
/** @param configuredRoot nullable absolute or relative path; null/blank derives a sibling of the repo root. */
|
||||
/**
|
||||
* @param configuredRoot nullable absolute or relative path; null/blank derives a sibling of
|
||||
* the repo root. No {@code worktreeGroup} configured — {@link #shareWithGroup}
|
||||
* is a no-op.
|
||||
*/
|
||||
public GitWorktrees(String configuredRoot) {
|
||||
this(configuredRoot, _ -> {});
|
||||
this(configuredRoot, (String) null);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param configuredRoot nullable absolute or relative path; null/blank derives a sibling of
|
||||
* the repo root.
|
||||
* @param group optional OS group name (fleetd #185 stage 3, {@code worktreeGroup:} in
|
||||
* config); null/blank ⇒ {@link #shareWithGroup} is a no-op.
|
||||
*/
|
||||
public GitWorktrees(String configuredRoot, String group) {
|
||||
this(configuredRoot, group, _ -> {});
|
||||
}
|
||||
|
||||
/** Test seam for changing a real worktree between its creation and its security check. */
|
||||
GitWorktrees(String configuredRoot, Consumer<String> afterWorktreeAdded) {
|
||||
this(configuredRoot, null, afterWorktreeAdded);
|
||||
}
|
||||
|
||||
/** Test seam combining a configurable {@code group} with {@link #afterWorktreeAdded}. */
|
||||
GitWorktrees(String configuredRoot, String group, Consumer<String> afterWorktreeAdded) {
|
||||
this(configuredRoot, group, afterWorktreeAdded, null, null);
|
||||
}
|
||||
|
||||
/** Test seam for changing how {@link #shareWithGroup}'s processes run. */
|
||||
GitWorktrees(String configuredRoot, String group, Consumer<String> afterWorktreeAdded,
|
||||
Function<String[], String> shareGroupRunner) {
|
||||
this(configuredRoot, group, afterWorktreeAdded, shareGroupRunner, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full test seam: also overrides how {@link #shareWithGroup}'s processes run (fleetd #185
|
||||
* stage 3), so a unit test can prove "no group configured ⇒ no process spawned" and inspect
|
||||
* exactly what commands a configured group runs, without a real second OS user/group.
|
||||
*
|
||||
* @param shareGroupRunner {@code null} ⇒ the real {@link #exec(String...)}.
|
||||
* @param worktreeAddRunner {@code null} ⇒ the real {@link #exec(String...)}.
|
||||
*/
|
||||
GitWorktrees(String configuredRoot, String group, Consumer<String> afterWorktreeAdded,
|
||||
Function<String[], String> shareGroupRunner, Function<String[], String> worktreeAddRunner) {
|
||||
this.configuredRoot = configuredRoot;
|
||||
this.group = (group == null || group.isBlank()) ? null : group;
|
||||
this.afterWorktreeAdded = afterWorktreeAdded == null ? _ -> {} : afterWorktreeAdded;
|
||||
this.shareGroupRunner = shareGroupRunner != null ? shareGroupRunner : this::exec;
|
||||
this.worktreeAddRunner = worktreeAddRunner != null ? worktreeAddRunner : this::exec;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String add(String repoRoot, String branch, String baseRef) {
|
||||
reportRemoteUrlsWithUserInfo(repoRoot);
|
||||
String base = (baseRef == null || baseRef.isBlank()) ? "HEAD" : baseRef;
|
||||
String nonce = nonce();
|
||||
Path root = resolveRoot(repoRoot);
|
||||
@@ -119,18 +173,74 @@ public final class GitWorktrees implements Worktrees {
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot create worktree root " + root + ": " + e.getMessage(), e);
|
||||
}
|
||||
shareRootWithGroup(root);
|
||||
String wt = path.toAbsolutePath().toString();
|
||||
log.info("adding worktree branch={} path={} base={}", branch, wt, base);
|
||||
removeUserInfoFromHttpsOrigin(repoRoot);
|
||||
exec("git", "-C", repoRoot, "worktree", "add", wt, "-b", branch, base);
|
||||
afterWorktreeAdded.accept(wt);
|
||||
requireCredentialFreeHttpsOrigin(wt);
|
||||
configureEnvironmentCredentialHelper(repoRoot, wt);
|
||||
configureHttpsUrlRewriteForSshOrigin(repoRoot, wt);
|
||||
isolateToolSurface(wt);
|
||||
try {
|
||||
worktreeAddRunner.apply(new String[] {"git", "-C", repoRoot, "worktree", "add", wt, "-b", branch, base});
|
||||
afterWorktreeAdded.accept(wt);
|
||||
requireCredentialFreeHttpsOrigin(wt);
|
||||
configureEnvironmentCredentialHelper(repoRoot, wt);
|
||||
configureHttpsUrlRewriteForSshOrigin(repoRoot, wt);
|
||||
isolateToolSurface(wt);
|
||||
} catch (RuntimeException e) {
|
||||
cleanupAfterAddFailure(repoRoot, wt, branch, e);
|
||||
throw e;
|
||||
}
|
||||
return wt;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code git worktree add} may have created the worktree and its branch by the time it, or any
|
||||
* later step through {@link #isolateToolSurface}, throws. This includes
|
||||
* {@link #requireCredentialFreeHttpsOrigin}, an intended security refusal, not only an IO
|
||||
* accident. A Git-reported {@code worktree add} failure usually creates nothing, but an
|
||||
* interrupted command can leave partial state. Without cleanup, {@code add()} never returns, so its caller
|
||||
* ({@code SessionManager#acquireWithWorktree}) never receives a path to register or clean up:
|
||||
* its local {@code path} stays null, the {@code if (path != null)} guard in its own catch block
|
||||
* never runs, and the worktree directory and branch leak on disk forever with nothing tracking
|
||||
* them (fleetd #274).
|
||||
*
|
||||
* <p>Reuses {@link #remove} — the same {@code git worktree remove --force} path every other
|
||||
* cleanup exit in this class already goes through — rather than a bespoke removal. It
|
||||
* additionally deletes {@code branch}: {@link #remove} alone deliberately leaves a released
|
||||
* session's branch behind (a worker's branch is expected to outlive its worktree, for PRs and
|
||||
* recovery), but a branch that never finished provisioning has no session, no PR, and nothing
|
||||
* else pointing at it, so leaving it behind would just trade one leak for a smaller one. Forced
|
||||
* (`-D`) because the branch is new and unmerged by construction. The worktree is removed first:
|
||||
* a branch checked out by a worktree cannot be deleted until the worktree that holds it is gone.
|
||||
*
|
||||
* <p>Cleanup failure must never mask {@code original} — that is the exception that explains
|
||||
* what actually went wrong — so a failure here is only logged, matching the pattern already
|
||||
* used in {@code SessionManager#acquireWithWorktree}'s own catch block.
|
||||
*/
|
||||
private void cleanupAfterAddFailure(String repoRoot, String worktreePath, String branch, RuntimeException original) {
|
||||
if (!Files.exists(Path.of(worktreePath))) {
|
||||
log.debug("provisioning failed before worktree {} existed; nothing to clean up", worktreePath);
|
||||
return;
|
||||
}
|
||||
log.warn("provisioning failed for branch={} path={}: {} — cleaning up before rethrowing",
|
||||
branch, worktreePath, original.getMessage());
|
||||
try {
|
||||
remove(repoRoot, worktreePath);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to remove leaked worktree {} after provisioning error: {}",
|
||||
worktreePath, cleanup.getMessage());
|
||||
}
|
||||
try {
|
||||
deleteBranch(repoRoot, branch);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to remove leaked branch {} after provisioning error: {}",
|
||||
branch, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void deleteBranch(String repoRoot, String branch) {
|
||||
exec("git", "-C", repoRoot, "branch", "-D", branch);
|
||||
}
|
||||
|
||||
/**
|
||||
* A linked worktree shares its primary checkout's git config. Remove HTTPS user info before
|
||||
* adding one, so a credential accidentally embedded in that config cannot reach the member.
|
||||
@@ -139,7 +249,7 @@ public final class GitWorktrees implements Worktrees {
|
||||
if (exitCode("git", "-C", repoRoot, "config", "--get", "remote.origin.url") != 0) {
|
||||
return;
|
||||
}
|
||||
String origin = exec("git", "-C", repoRoot, "config", "--get", "remote.origin.url").trim();
|
||||
String origin = execRedacted("git", "-C", repoRoot, "config", "--get", "remote.origin.url").trim();
|
||||
URI uri;
|
||||
try {
|
||||
uri = new URI(origin);
|
||||
@@ -155,6 +265,9 @@ public final class GitWorktrees implements Worktrees {
|
||||
throw new WorktreeException("origin URL has invalid HTTPS user info; cannot provision safely");
|
||||
}
|
||||
String cleanOrigin = origin.substring(0, schemeEnd) + origin.substring(userInfoEnd + 1);
|
||||
// Plain exec, not execRedacted, is correct here: this call WRITES cleanOrigin (already
|
||||
// stripped of user-info above) rather than reading a URL back from stdout, so there is
|
||||
// nothing secret left in either its argv or its stdout to redact.
|
||||
exec("git", "-C", repoRoot, "remote", "set-url", "origin", cleanOrigin);
|
||||
log.info("removed HTTPS user info from forge origin before provisioning worktree");
|
||||
}
|
||||
@@ -164,7 +277,7 @@ public final class GitWorktrees implements Worktrees {
|
||||
if (exitCode("git", "-C", worktreePath, "remote", "get-url", "--all", "origin") != 0) {
|
||||
return;
|
||||
}
|
||||
String origins = exec("git", "-C", worktreePath, "remote", "get-url", "--all", "origin");
|
||||
String origins = execRedacted("git", "-C", worktreePath, "remote", "get-url", "--all", "origin");
|
||||
for (String origin : origins.split("\\R")) {
|
||||
try {
|
||||
URI uri = new URI(origin);
|
||||
@@ -177,6 +290,99 @@ public final class GitWorktrees implements Worktrees {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Report — never refuse — every remote whose fetch or push URL carries user-info outside SSH.
|
||||
* A linked worktree shares its parent repository's git config, so a credential on ANY remote
|
||||
* (not only {@code origin}) or in a {@code pushurl} is just as readable by a member as one on
|
||||
* {@code origin}'s HTTPS fetch URL — the one case {@link #removeUserInfoFromHttpsOrigin} and
|
||||
* {@link #requireCredentialFreeHttpsOrigin} already strip and refuse. This check is additive: it
|
||||
* only logs a warning, it never mutates config and never refuses the provision.
|
||||
*
|
||||
* <p>A reporting-only check must never be able to abort a provision — PR #173 shipped one that
|
||||
* ran unguarded at the top of {@link #add}, and every git call inside it can throw ({@link #exec}
|
||||
* turns a non-zero exit or its 30-second timeout into a {@link WorktreeException}). Every git call
|
||||
* here is therefore wrapped, and on failure only the exception's <em>class</em> is logged, never
|
||||
* its message: the enumerating {@code git remote} call is not redacted, and its stderr is read
|
||||
* from the very config that may hold the URL this check exists to find.
|
||||
*/
|
||||
private void reportRemoteUrlsWithUserInfo(String repoRoot) {
|
||||
List<String> remotes;
|
||||
try {
|
||||
remotes = exec("git", "-C", repoRoot, "remote").lines()
|
||||
.map(String::trim)
|
||||
.filter(r -> !r.isBlank())
|
||||
.toList();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("could not enumerate remotes to check for credentialed URLs in {}: {}",
|
||||
Path.of(repoRoot).toAbsolutePath().normalize(), e.getClass().getName());
|
||||
return;
|
||||
}
|
||||
for (String remote : remotes) {
|
||||
reportOneRemoteUrlsWithUserInfo(repoRoot, remote);
|
||||
}
|
||||
}
|
||||
|
||||
private void reportOneRemoteUrlsWithUserInfo(String repoRoot, String remote) {
|
||||
boolean leaks = remoteUrlsLeakUserInfo(repoRoot, remote, false)
|
||||
|| remoteUrlsLeakUserInfo(repoRoot, remote, true);
|
||||
if (leaks) {
|
||||
log.warn("member worktree shares a remote URL containing user-info: remote={} repository={}; "
|
||||
+ "remove credentials from the repository's git config",
|
||||
remote, Path.of(repoRoot).toAbsolutePath().normalize());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* True when any resolved fetch (or, if {@code push}, push) URL for {@code remote} carries
|
||||
* non-empty user-info outside SSH. Never throws — a git failure here is caught, logged (its
|
||||
* class only, per the javadoc above), and treated as "nothing found", so it cannot abort or
|
||||
* otherwise affect provisioning. Uses {@link #execRedacted} because the command's stdout is
|
||||
* itself the URL this check exists to find.
|
||||
*/
|
||||
private boolean remoteUrlsLeakUserInfo(String repoRoot, String remote, boolean push) {
|
||||
try {
|
||||
String out = push
|
||||
? execRedacted("git", "-C", repoRoot, "remote", "get-url", "--push", "--all", remote)
|
||||
: execRedacted("git", "-C", repoRoot, "remote", "get-url", "--all", remote);
|
||||
return out.lines().anyMatch(url -> !url.isBlank() && urlLeaksUserInfo(url.trim()));
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("could not read the {} URL for remote {} to check for credentials: {}",
|
||||
push ? "push" : "fetch", remote, e.getClass().getName());
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* True when {@code rawUrl} parses as an absolute URI with a non-SSH-family scheme and non-empty
|
||||
* user-info. An unparsable or scheme-less URL — including the ssh scp-like shorthand
|
||||
* ({@code user@host:path}) — is not this check's concern and is treated as "no finding", the
|
||||
* same way {@link #configureHttpsUrlRewriteForSshOrigin} leaves that shorthand untouched.
|
||||
*/
|
||||
private static boolean urlLeaksUserInfo(String rawUrl) {
|
||||
URI uri;
|
||||
try {
|
||||
uri = new URI(rawUrl);
|
||||
} catch (URISyntaxException e) {
|
||||
return false;
|
||||
}
|
||||
String scheme = uri.getScheme();
|
||||
if (scheme == null || isSshLikeScheme(scheme)) {
|
||||
return false;
|
||||
}
|
||||
String userInfo = uri.getUserInfo();
|
||||
return userInfo != null && !userInfo.isEmpty();
|
||||
}
|
||||
|
||||
/**
|
||||
* SSH-family schemes deliberately excluded from {@link #urlLeaksUserInfo}: there, the user part
|
||||
* selects an account and authentication itself happens over SSH, so it is not a credential the
|
||||
* way HTTPS/HTTP user-info is.
|
||||
*/
|
||||
private static boolean isSshLikeScheme(String scheme) {
|
||||
return "ssh".equalsIgnoreCase(scheme) || "git+ssh".equalsIgnoreCase(scheme)
|
||||
|| "ssh+git".equalsIgnoreCase(scheme);
|
||||
}
|
||||
|
||||
/** Configure a per-worktree helper that supplies a token from the member environment at call time. */
|
||||
private void configureEnvironmentCredentialHelper(String repoRoot, String worktreePath) {
|
||||
exec("git", "-C", repoRoot, "config", "extensions.worktreeConfig", "true");
|
||||
@@ -234,7 +440,7 @@ public final class GitWorktrees implements Worktrees {
|
||||
if (exitCode("git", "-C", repoRoot, "config", "--get", "remote.origin.url") != 0) {
|
||||
return;
|
||||
}
|
||||
String origin = exec("git", "-C", repoRoot, "config", "--get", "remote.origin.url").trim();
|
||||
String origin = execRedacted("git", "-C", repoRoot, "config", "--get", "remote.origin.url").trim();
|
||||
URI uri;
|
||||
try {
|
||||
uri = new URI(origin);
|
||||
@@ -286,19 +492,75 @@ public final class GitWorktrees implements Worktrees {
|
||||
* neutralized copy from ever showing up as a local modification the worker might commit. A config
|
||||
* the repo does not carry is skipped silently — no stub is invented for a file the repo does not
|
||||
* have, and one missing file must never fail provisioning.
|
||||
*
|
||||
* <p>fleetd #134. The neutralization above is correct and stays unconditional — the defect was
|
||||
* that it was invisible on both sides. Neither the daemon's own log nor the worker sitting in the
|
||||
* worktree could tell a stub from the repo's real file: a real worker read a 3-byte {@code {}}
|
||||
* where the repo's {@code opencode.json} is 30+ lines, and reported — truthfully from what it
|
||||
* could see, and wrongly — that a mount key did not exist. Two fixes, aimed at two different
|
||||
* readers:
|
||||
* <ul>
|
||||
* <li>the daemon operator reads {@link #log}, so the summary below names the denominator, what
|
||||
* was neutralized, and why anything was not — the same shape {@code overlayParity} reports
|
||||
* its own copy in;
|
||||
* <li>the worker reads its own worktree, not the daemon's log, so the same fact is recorded a
|
||||
* second time in worktree-scoped git config ({@code fleet.neutralizedConfig} /
|
||||
* {@code fleet.neutralizedConfigNote}, readable with {@code git config --worktree --get-all
|
||||
* fleet.neutralizedConfig}) rather than as a file in the working tree. A working-tree file
|
||||
* would show up in {@code git status} for the worker to trip on or commit; worktree-scoped
|
||||
* config lives in {@code .git/worktrees/<nonce>/config.worktree} and can never appear there.
|
||||
* This reuses the exact mechanism {@link #configureEnvironmentCredentialHelper} and
|
||||
* {@link #configureHttpsUrlRewriteForSshOrigin} already use for other worktree-local state.
|
||||
* </ul>
|
||||
*/
|
||||
private void isolateToolSurface(String worktreePath) {
|
||||
Path root = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
List<String> neutralized = new ArrayList<>();
|
||||
List<String> skipped = new ArrayList<>();
|
||||
for (WorktreeHostileConfig cfg : WORKTREE_HOSTILE_CONFIGS) {
|
||||
neutralize(root, worktreePath, cfg);
|
||||
if (neutralize(root, worktreePath, cfg)) {
|
||||
neutralized.add(cfg.file());
|
||||
} else {
|
||||
skipped.add(cfg.file() + " absent");
|
||||
}
|
||||
}
|
||||
String detail = neutralized.isEmpty() ? String.join(", ", skipped)
|
||||
: skipped.isEmpty() ? String.join(", ", neutralized)
|
||||
: String.join(", ", neutralized) + " (" + String.join(", ", skipped) + ")";
|
||||
log.info("tool-surface isolation: neutralized {} of {} configs: {} — the worktree copy is a "
|
||||
+ "stub, not the repo's file; edit the real file in the primary checkout instead",
|
||||
neutralized.size(), WORKTREE_HOSTILE_CONFIGS.size(), detail);
|
||||
recordNeutralizedConfigForWorker(worktreePath, neutralized);
|
||||
}
|
||||
|
||||
private void neutralize(Path root, String worktreePath, WorktreeHostileConfig cfg) {
|
||||
/**
|
||||
* The worker-readable half of fleetd #134: record which files were neutralized where the worker
|
||||
* itself can read it, without a working-tree file that would show up in {@code git status}.
|
||||
* Worktree-scoped git config is per-worktree, lives under {@code .git/worktrees/<nonce>/} rather
|
||||
* than the working tree, and this repo already relies on the same mechanism (and the same
|
||||
* {@code extensions.worktreeConfig} enablement) for the credential helper and the SSH→HTTPS
|
||||
* rewrite — see {@link #configureEnvironmentCredentialHelper}.
|
||||
*/
|
||||
private void recordNeutralizedConfigForWorker(String worktreePath, List<String> neutralized) {
|
||||
if (neutralized.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
exec("git", "-C", worktreePath, "config", "extensions.worktreeConfig", "true");
|
||||
for (String file : neutralized) {
|
||||
exec("git", "-C", worktreePath, "config", "--worktree", "--add", "fleet.neutralizedConfig", file);
|
||||
}
|
||||
exec("git", "-C", worktreePath, "config", "--worktree", "fleet.neutralizedConfigNote",
|
||||
"the worktree copy of each fleet.neutralizedConfig path is a stub, not the repo's "
|
||||
+ "committed file; edit the real file from the primary checkout instead");
|
||||
}
|
||||
|
||||
/** @return true if {@code cfg} was neutralized (present, or created because {@link
|
||||
* WorktreeHostileConfig#createIfAbsent()}); false if the repo does not carry it and it was
|
||||
* left alone. */
|
||||
private boolean neutralize(Path root, String worktreePath, WorktreeHostileConfig cfg) {
|
||||
Path target = root.resolve(cfg.file());
|
||||
if (!Files.exists(target) && !cfg.createIfAbsent()) {
|
||||
log.debug("{} absent in the worktree — skipping (repo does not carry it)", cfg.file());
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
try {
|
||||
Files.writeString(target, cfg.stub());
|
||||
@@ -309,7 +571,7 @@ public final class GitWorktrees implements Worktrees {
|
||||
if (isTracked(root, cfg.file())) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", cfg.file());
|
||||
}
|
||||
log.debug("neutralized {} — worker tool surface is launcher-mounted only", cfg.file());
|
||||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -346,25 +608,46 @@ public final class GitWorktrees implements Worktrees {
|
||||
}
|
||||
Path srcRoot = Path.of(repoRoot).toAbsolutePath().normalize();
|
||||
Path dstRoot = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
List<String> copied = new ArrayList<>();
|
||||
List<String> skipped = new ArrayList<>();
|
||||
List<String> neutralized = new ArrayList<>();
|
||||
for (String rel : overlay) {
|
||||
Path src = srcRoot.resolve(rel).normalize();
|
||||
if (!Files.exists(src)) {
|
||||
log.debug("parity overlay source missing — skipping {}", rel);
|
||||
skipped.add(rel + " absent");
|
||||
continue;
|
||||
}
|
||||
Path dst = dstRoot.resolve(rel).normalize();
|
||||
try {
|
||||
Files.createDirectories(dst.getParent());
|
||||
Files.copy(src, dst, StandardCopyOption.REPLACE_EXISTING, StandardCopyOption.COPY_ATTRIBUTES);
|
||||
log.debug("copied parity overlay {}", rel);
|
||||
copied.add(rel);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot copy overlay " + rel + ": " + e.getMessage(), e);
|
||||
}
|
||||
if (isTracked(dstRoot, rel)) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", rel);
|
||||
log.debug("marked overlay --skip-worktree {}", rel);
|
||||
neutralized.add(rel);
|
||||
}
|
||||
}
|
||||
// CB-148 point 3: a bare count ("copied 1") hides which candidates were even considered — the
|
||||
// same shape of under-reporting this repo has been bitten by before. Name the denominator
|
||||
// (every configured candidate), what was actually copied, and — for anything not copied —
|
||||
// why, so a spawn's overlay outcome is legible from the log alone, no filesystem dig required.
|
||||
String detail = copied.isEmpty() ? String.join(", ", skipped)
|
||||
: skipped.isEmpty() ? String.join(", ", copied)
|
||||
: String.join(", ", copied) + " (" + String.join(", ", skipped) + ")";
|
||||
log.info("parity overlay: copied {} of {} candidates: {}", copied.size(), overlay.size(), detail);
|
||||
// CB-134: a neutralized tracked file is otherwise a silent trap — a worker edits it, git
|
||||
// ignores the change with no error, and nothing anywhere said the file could not be
|
||||
// committed from this worktree. Name every file marked --skip-worktree here, with the
|
||||
// consequence stated in the message itself, rather than adding a worktree-local marker
|
||||
// file: acceptance criterion 1 requires the worktree hold exactly the configured overlay
|
||||
// set and nothing else, so an extra marker would itself violate the fix.
|
||||
if (!neutralized.isEmpty()) {
|
||||
log.info("parity overlay marked --skip-worktree (cannot be committed from this worktree): {}",
|
||||
String.join(", ", neutralized));
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -510,6 +793,174 @@ public final class GitWorktrees implements Worktrees {
|
||||
return deleted;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>fleetd #185 stage 3. No-op — no process spawned, nothing logged — when {@link #group} is
|
||||
* null/blank. Otherwise:
|
||||
* <ol>
|
||||
* <li>{@code git -C repoRoot config core.sharedRepository group} so every future write by
|
||||
* either uid stays group-writable;</li>
|
||||
* <li>a one-time {@code chgrp}/{@code chmod g+rwX} fix-up over the worktree directory and,
|
||||
* under the repo's <em>common</em> git directory, {@code objects}, {@code refs},
|
||||
* {@code logs}, {@code worktrees} and {@code packed-refs} — with setgid
|
||||
* ({@code chmod g+s}) applied only to the directories among them, so files created later
|
||||
* inherit the group;</li>
|
||||
* <li>one INFO line naming the group and the paths touched.</li>
|
||||
* </ol>
|
||||
*
|
||||
* <p><b>Every path is skipped when it does not exist.</b> {@code .git/logs} is absent in a repo
|
||||
* with {@code core.logAllRefUpdates=false} or one that has had no ref update yet, and
|
||||
* {@code packed-refs} is absent until refs are packed. Passing a missing path to {@code chgrp}
|
||||
* exits non-zero, which would fail <em>every</em> provisioning spawn with a message blaming a
|
||||
* group that is in fact fine.
|
||||
*
|
||||
* <p><b>The git directory is resolved, not assumed.</b> {@code <repoRoot>/.git} is a
|
||||
* <em>file</em>, not a directory, when the checkout is itself a linked worktree — the very
|
||||
* thing this class creates for every member. {@code git rev-parse --git-common-dir} gives the
|
||||
* real shared store, and it may answer relatively, so it is resolved against {@code repoRoot}.
|
||||
*
|
||||
* <p><b>The fix-up re-runs on every spawn, by design.</b> {@code core.sharedRepository=group}
|
||||
* governs only what git writes <em>after</em> it is set; the walk is what covers everything
|
||||
* already on disk. It is not redundant work to optimise away — dropping it silently leaves
|
||||
* pre-existing objects unreadable to the member. It costs three walks of the object store per
|
||||
* spawn (about 3000 files in this repo, well under a second, but it grows with the repo).
|
||||
*
|
||||
* <p>This only fixes up file ownership/permissions on the operator's shared repo so a
|
||||
* different-uid member can write to it — it isolates credentials, not the repository. A member
|
||||
* in the group can still write the operator's git objects and refs.
|
||||
*
|
||||
* <p>Fails loudly: a missing group, or a {@code chgrp}/{@code chmod} refused because the
|
||||
* operator is not a member of it, becomes a {@link WorktreeException} naming the group — never
|
||||
* a silent skip that leaves a member unable to work with nothing in the log to explain why.
|
||||
*/
|
||||
@Override
|
||||
public void shareWithGroup(String repoRoot, String worktreePath) {
|
||||
if (group == null) {
|
||||
return;
|
||||
}
|
||||
List<String> touched = new ArrayList<>();
|
||||
try {
|
||||
shareGroupRunner.apply(new String[]{"git", "-C", repoRoot, "config", "core.sharedRepository", "group"});
|
||||
String commonDir = gitCommonDir(repoRoot);
|
||||
shareGroupPathIfPresent(worktreePath, true, touched);
|
||||
for (String name : List.of("objects", "refs", "logs", "worktrees")) {
|
||||
shareGroupPathIfPresent(commonDir + "/" + name, true, touched);
|
||||
}
|
||||
shareGroupPathIfPresent(commonDir + "/packed-refs", false, touched);
|
||||
} catch (WorktreeException e) {
|
||||
throw new WorktreeException("cannot share worktree with group '" + group + "': "
|
||||
+ e.getMessage() + " — the group must exist, and the fleetd operator ("
|
||||
+ System.getProperty("user.name") + ") must be a member of it", e);
|
||||
}
|
||||
log.info("worktreeGroup={} shared repoRoot={} worktreePath={} paths={}",
|
||||
group, repoRoot, worktreePath, touched);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #224: make {@code worktreeRoot} ITSELF group-traversable — established once, here,
|
||||
* where the root is created, never at a use site. {@link #shareWithGroup} shares each worktree
|
||||
* (and the repo's common git dir) with {@link #group}, but never the PARENT directory that
|
||||
* contains every worktree — and under {@code memberHerdrSocket:} the member pane runs as a
|
||||
* different OS user, which needs the execute bit on every ancestor directory to reach anything
|
||||
* underneath, no matter how carefully each child is shared. Without this, a member cannot read
|
||||
* the ephemeral {@code opencode.json} #219 places under this root, cannot reach its own
|
||||
* worktree, and cannot read #213's ZDOTDIR scrub when placed here either.
|
||||
*
|
||||
* <p>No-op — no process spawned — when {@link #group} is null/blank, so behaviour with
|
||||
* {@code worktreeGroup:} unset (today's only live mode) is unchanged. Only {@code root} itself
|
||||
* is touched (non-recursive): each child underneath is shared individually, either by
|
||||
* {@link #shareWithGroup} for a worktree or by the launcher that generates it (fleetd #213/#219)
|
||||
* for a scrub/config directory — sharing this level again would just duplicate that policy in
|
||||
* the wrong layer.
|
||||
*
|
||||
* <p>Fails loudly, naming {@code root}, its mode at the time of the attempt, and {@link #group}:
|
||||
* a member that starts and then cannot see its own checkout is worse than a refused spawn, since
|
||||
* nothing about that failure mode points at a directory's permission bits.
|
||||
*/
|
||||
private void shareRootWithGroup(Path root) {
|
||||
if (group == null) {
|
||||
return;
|
||||
}
|
||||
String mode = currentPosixMode(root);
|
||||
try {
|
||||
shareGroupRunner.apply(new String[]{"chgrp", group, root.toString()});
|
||||
shareGroupRunner.apply(new String[]{"chmod", "g+x", root.toString()});
|
||||
} catch (WorktreeException e) {
|
||||
throw new WorktreeException("cannot make worktree root " + root + " (mode " + mode
|
||||
+ ") group-traversable for group '" + group + "': " + e.getMessage()
|
||||
+ " — the group must exist, and the fleetd operator (" + System.getProperty("user.name")
|
||||
+ ") must be a member of it", e);
|
||||
}
|
||||
log.info("worktreeGroup={} made worktree root {} group-traversable (was mode {})", group, root, mode);
|
||||
}
|
||||
|
||||
/** {@code root}'s current POSIX permission string, or {@code "unknown"} on a filesystem that does
|
||||
* not support POSIX permissions — used only to name the mode in a refusal message. */
|
||||
private static String currentPosixMode(Path root) {
|
||||
try {
|
||||
return PosixFilePermissions.toString(Files.getPosixFilePermissions(root));
|
||||
} catch (IOException | UnsupportedOperationException e) {
|
||||
return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The repo's <em>common</em> git directory as an absolute path — where {@code objects},
|
||||
* {@code refs} and {@code worktrees} actually live. {@code git rev-parse --git-common-dir}
|
||||
* answers relative to {@code repoRoot} in the ordinary case ({@code .git}) and absolutely for a
|
||||
* linked worktree, so the answer is resolved against {@code repoRoot} either way. Never
|
||||
* hardcode {@code repoRoot + "/.git"}: that is a FILE when the checkout is itself a linked
|
||||
* worktree.
|
||||
*/
|
||||
private String gitCommonDir(String repoRoot) {
|
||||
String answer = shareGroupRunner.apply(
|
||||
new String[]{"git", "-C", repoRoot, "rev-parse", "--git-common-dir"});
|
||||
String trimmed = answer == null ? "" : answer.trim();
|
||||
if (trimmed.isEmpty()) {
|
||||
trimmed = ".git";
|
||||
}
|
||||
return Path.of(repoRoot).resolve(trimmed).normalize().toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@link #shareGroupPath} when {@code path} exists, recording it in {@code touched}; otherwise
|
||||
* nothing at all. A missing path is normal, not an error — see {@link #shareWithGroup}'s
|
||||
* javadoc for which ones are routinely absent and why passing them to {@code chgrp} would fail
|
||||
* every spawn.
|
||||
*/
|
||||
private void shareGroupPathIfPresent(String path, boolean recursive, List<String> touched) {
|
||||
if (!Files.exists(Path.of(path))) {
|
||||
return;
|
||||
}
|
||||
shareGroupPath(path, recursive);
|
||||
touched.add(path);
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code chgrp}/{@code chmod g+rwX} {@code path} to {@link #group}. When {@code recursive},
|
||||
* also walks the directories under {@code path} (including {@code path} itself, when it is a
|
||||
* directory) and sets setgid on each — directories only, per the javadoc on
|
||||
* {@link #shareWithGroup}.
|
||||
*/
|
||||
private void shareGroupPath(String path, boolean recursive) {
|
||||
List<String> chgrp = new ArrayList<>(List.of("chgrp"));
|
||||
if (recursive) chgrp.add("-R");
|
||||
chgrp.add(group);
|
||||
chgrp.add(path);
|
||||
shareGroupRunner.apply(chgrp.toArray(new String[0]));
|
||||
|
||||
List<String> chmod = new ArrayList<>(List.of("chmod"));
|
||||
if (recursive) chmod.add("-R");
|
||||
chmod.add("g+rwX");
|
||||
chmod.add(path);
|
||||
shareGroupRunner.apply(chmod.toArray(new String[0]));
|
||||
|
||||
if (recursive) {
|
||||
shareGroupRunner.apply(new String[]{"find", path, "-type", "d", "-exec", "chmod", "g+s", "{}", "+"});
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Every {@code refs/wip/*} ref (see {@link WipRef}). The committer date is read as a unix
|
||||
* count of seconds and converted to millis. {@code %00} (NUL) separates the fields because a
|
||||
@@ -603,6 +1054,31 @@ public final class GitWorktrees implements Worktrees {
|
||||
|
||||
/** Same as {@link #exec(String...)}, with extra environment variables set on the child process. */
|
||||
private String exec(Map<String, String> extraEnv, String... command) {
|
||||
return exec(extraEnv, false, command);
|
||||
}
|
||||
|
||||
/**
|
||||
* Same as {@link #exec(String...)}, for a command whose stdout may itself carry a credential
|
||||
* (e.g. {@code git remote get-url}, whose output is a URL). Stdout is still returned normally on
|
||||
* success — callers still get the URL to inspect — but it is suppressed from BOTH the timeout
|
||||
* message and the non-zero-exit message, so a failing call here can never copy it into a
|
||||
* {@link WorktreeException}, and from there into a caller's log.
|
||||
*/
|
||||
private String execRedacted(String... command) {
|
||||
return exec(Map.of(), true, command);
|
||||
}
|
||||
|
||||
/**
|
||||
* Shared implementation for {@link #exec(Map, String...)} and {@link #execRedacted(String...)}.
|
||||
* {@code redactOutput} suppresses captured stdout from both failure messages below.
|
||||
*
|
||||
* <p>Package-private, not {@code private}: also a test seam, the same way the
|
||||
* {@code afterWorktreeAdded} constructor parameter is. It lets a test drive the redaction
|
||||
* guarantee directly — a synthetic failing command whose stdout carries a test marker passed
|
||||
* through {@code extraEnv} rather than argv — without depending on finding a real git failure
|
||||
* mode that happens to echo a URL onto stdout before exiting non-zero.
|
||||
*/
|
||||
String exec(Map<String, String> extraEnv, boolean redactOutput, String... command) {
|
||||
String out;
|
||||
int code;
|
||||
Process p;
|
||||
@@ -624,7 +1100,8 @@ public final class GitWorktrees implements Worktrees {
|
||||
try {
|
||||
if (!p.waitFor(30, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
throw new WorktreeException("command timed out: " + String.join(" ", command) + "\n" + out);
|
||||
throw new WorktreeException("command timed out: " + String.join(" ", command)
|
||||
+ (redactOutput || out.isBlank() ? "" : "\n" + out));
|
||||
}
|
||||
code = p.exitValue();
|
||||
} catch (InterruptedException e) {
|
||||
@@ -634,7 +1111,7 @@ public final class GitWorktrees implements Worktrees {
|
||||
}
|
||||
if (code != 0) {
|
||||
throw new WorktreeException("exit " + code + " for: " + String.join(" ", command)
|
||||
+ (out.isBlank() ? "" : "\n" + out));
|
||||
+ (redactOutput || out.isBlank() ? "" : "\n" + out));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
@@ -46,7 +46,8 @@ public record MemberSession(
|
||||
String worktree,
|
||||
String branch,
|
||||
CharterReceipt charterReceipt,
|
||||
String agentSessionId) {
|
||||
String agentSessionId,
|
||||
String failureReason) {
|
||||
|
||||
/** One-shot worker lifecycle states. */
|
||||
public enum State {
|
||||
@@ -54,6 +55,7 @@ public record MemberSession(
|
||||
READY,
|
||||
BUSY,
|
||||
DONE,
|
||||
BACKEND_ERROR,
|
||||
FAILED,
|
||||
RELEASED
|
||||
}
|
||||
@@ -69,24 +71,51 @@ public record MemberSession(
|
||||
long lastActivityAtNanos, int turnCount, State state,
|
||||
String worktree, String branch) {
|
||||
this(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, null, null);
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, null, null, null);
|
||||
}
|
||||
|
||||
/** Backward-compatible shape without a backend failure reason. */
|
||||
public MemberSession(String paneId, String terminalId, String profile, MemberRole role,
|
||||
String cwd, String ownerTerminal, long spawnedAtNanos,
|
||||
long lastActivityAtNanos, int turnCount, State state,
|
||||
String worktree, String branch, CharterReceipt charterReceipt,
|
||||
String agentSessionId) {
|
||||
this(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, null);
|
||||
}
|
||||
|
||||
/** Return a copy of this session in {@code state}. */
|
||||
public MemberSession withState(State state) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId);
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
|
||||
/** Return a copy with {@code lastActivityAtNanos} updated to {@code nowNanos}. */
|
||||
public MemberSession withActivity(long nowNanos) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId);
|
||||
nowNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
|
||||
/** Return a copy with the turn count incremented and activity timestamped at {@code nowNanos}. */
|
||||
public MemberSession bumpTurn(long nowNanos) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount + 1, state, worktree, branch, charterReceipt, agentSessionId);
|
||||
nowNanos, turnCount + 1, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
|
||||
/**
|
||||
* Return a copy with {@code agentSessionId} resolved to a non-null value (fleetd #209). Some
|
||||
* adapters (opencode) cannot answer {@link dev.ltms.fleet.peer.PeerHandle#agentSessionId()} at
|
||||
* spawn time — the peer has not persisted its session record yet — so the id is discovered on
|
||||
* a later poll and swapped into the otherwise-immutable session via this wither.
|
||||
*/
|
||||
public MemberSession withAgentSessionId(String agentSessionId) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
|
||||
/** Return a copy with the durable backend failure detail. */
|
||||
public MemberSession withFailureReason(String failureReason) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -21,6 +21,7 @@ import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.LongSupplier;
|
||||
@@ -46,12 +47,23 @@ public final class SessionManager implements TurnListener {
|
||||
private final PeerLauncher launcher;
|
||||
private final Worktrees worktrees;
|
||||
private final ConcurrentHashMap<String /*paneId*/, MemberSession> registry = new ConcurrentHashMap<>();
|
||||
/**
|
||||
* fleetd #209: the live {@link PeerHandle} for every registered pane, retained solely so
|
||||
* {@link #resolveAgentSessionId} can re-poll {@link PeerHandle#agentSessionId()} after spawn.
|
||||
* The handle used to go out of scope at the end of the spawn method, so a launcher that answers
|
||||
* the id lazily (opencode — the on-disk session row is written after the pane is created) could
|
||||
* never be re-asked, and {@code fleet_list}/{@code fleet_spawn resumeSessionId} never saw it.
|
||||
* Populated on every spawn path, removed on {@link #release}.
|
||||
*/
|
||||
private final ConcurrentHashMap<String /*paneId*/, PeerHandle> handles = new ConcurrentHashMap<>();
|
||||
private final MemberPresence presence;
|
||||
private final SecureRandom nonceRandom = new SecureRandom();
|
||||
private final AtomicLong nonceSeq = new AtomicLong();
|
||||
private final LongSupplier nowNanos;
|
||||
private final int contextCap;
|
||||
private final boolean clearAfterTurn;
|
||||
/** Null in production; test seam for the interval before an idle session's conditional release. */
|
||||
private final Consumer<MemberSession> beforeIdleRelease;
|
||||
private volatile MemberLifecycle memberLifecycle = MemberLifecycle.NONE;
|
||||
/**
|
||||
* CB-586: the repo root the fleet actually works in, remembered the first time a worktree
|
||||
@@ -62,6 +74,16 @@ public final class SessionManager implements TurnListener {
|
||||
*/
|
||||
private volatile String fleetRepoRoot;
|
||||
|
||||
/**
|
||||
* fleetd #308: flips true the instant {@link #drainAll} starts, before its registry snapshot
|
||||
* is even taken — so a spawn already in flight sees the refusal as early as a plain flag can
|
||||
* make it. This alone cannot close the race completely: a caller that read {@code false} just
|
||||
* before the flip can still land in the registry after the snapshot. {@link #drainAll}'s
|
||||
* post-loop sweep is what catches that straggler; the two mechanisms are deliberately paired,
|
||||
* see {@link #drainAll}'s javadoc.
|
||||
*/
|
||||
private final AtomicBoolean draining = new AtomicBoolean(false);
|
||||
|
||||
/** CB-520: notified with a terminalId on every acquire; no-op until wired. */
|
||||
private final List<Consumer<String>> acquireListeners = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
/** CB-516: notified with a {@link ReleaseDetail} on every release; no-op until wired. */
|
||||
@@ -93,13 +115,23 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap, boolean clearAfterTurn) {
|
||||
int contextCap, boolean clearAfterTurn) {
|
||||
this(launcher, worktrees, nowNanos, contextCap, clearAfterTurn, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Package-private constructor for a deterministic reap/delivery race test. Production callers
|
||||
* use the constructor above, whose null hook adds no callback or lock to an ordinary reap.
|
||||
*/
|
||||
SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap, boolean clearAfterTurn, Consumer<MemberSession> beforeIdleRelease) {
|
||||
this.launcher = launcher;
|
||||
this.worktrees = worktrees;
|
||||
this.presence = new PresenceFleet(this);
|
||||
this.nowNanos = nowNanos;
|
||||
this.contextCap = contextCap;
|
||||
this.clearAfterTurn = clearAfterTurn;
|
||||
this.beforeIdleRelease = beforeIdleRelease;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -172,47 +204,61 @@ public final class SessionManager implements TurnListener {
|
||||
public MemberSession acquire(String profile, MemberRole role, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt,
|
||||
String sessionName, String resumeSessionId) {
|
||||
// fleetd #308: refuse before anything else runs — no slot reservation, no launcher spawn —
|
||||
// so a caller learns the daemon is going down instead of getting a session drainAll will
|
||||
// never see again. Checked here because every other acquire(...) overload delegates to
|
||||
// this one, so this is the single point every spawn path passes through.
|
||||
if (draining.get()) {
|
||||
throw new ShuttingDownException("fleetd is shutting down; refusing to spawn a session "
|
||||
+ "the shutdown drain would never see");
|
||||
}
|
||||
MemberRole memberRole = (role == null) ? MemberRole.DEV : role;
|
||||
requireResumeCapability(profile, resumeSessionId);
|
||||
// CB-619 / fleetd #123: an explicit profile bypasses placement (CompositePeerLauncher only
|
||||
// constrains an UNQUALIFIED spawn to the role's pool), so it is the one path that can ask
|
||||
// for a role with no slot to bind it to. Refuse before anything spawns. A blank profile is
|
||||
// left to placement, which already restricts an unqualified spawn to the role's pool.
|
||||
if (profile != null && !profile.isBlank()) {
|
||||
memberLifecycle.requireSlotFor(memberRole, profile);
|
||||
}
|
||||
MemberLifecycle.SlotReservation reservation = memberLifecycle.reserve(memberRole, profile);
|
||||
String launchProfile = reservation == null ? profile : reservation.profile();
|
||||
if (wt == null) {
|
||||
// CB-557: the role must ride on the SpawnRequest, not stay a local. The launcher needs it
|
||||
// to pick the profile out of that role's pool and to label the tab; a role kept only on
|
||||
// the MemberSession is recorded after the spawn it was supposed to steer.
|
||||
SpawnRequest req = new SpawnRequest(profile, requestedCwd, callerCwd, sessionName, resumeSessionId, memberRole);
|
||||
SpawnRequest req = new SpawnRequest(launchProfile, requestedCwd, callerCwd, sessionName, resumeSessionId, memberRole);
|
||||
PeerHandle handle;
|
||||
boolean bound = false;
|
||||
try {
|
||||
handle = launcher.spawn(req);
|
||||
String resolvedProfile = resolveProfile(handle, launchProfile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, requestedCwd, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberRole actualRole = acquired(memberRole, resolvedProfile, handle.terminalId(), reservation);
|
||||
bound = reservation == null || actualRole == MemberRole.ARCHITECT;
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(), handle.terminalId(), resolvedProfile, actualRole, cwd, ownerTerminal, now, now, 0,
|
||||
MemberSession.State.SPAWNING, null, null, handle.charterReceipt(), handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
handles.put(handle.id(), handle);
|
||||
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("spawn failed for profile={} role={}: {}", profile, memberRole, e.getMessage());
|
||||
if (!bound) memberLifecycle.release(reservation);
|
||||
log.warn("spawn failed for profile={} role={}: {}", launchProfile, memberRole, e.getMessage());
|
||||
throw e;
|
||||
}
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, requestedCwd, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
memberRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
now,
|
||||
0,
|
||||
MemberSession.State.SPAWNING,
|
||||
null,
|
||||
null,
|
||||
handle.charterReceipt(),
|
||||
handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
memberLifecycle.acquired(session.role(), session.profile(), session.terminalId());
|
||||
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
return acquireWithWorktree(profile, memberRole, requestedCwd, callerCwd, ownerTerminal, wt,
|
||||
sessionName, resumeSessionId);
|
||||
try {
|
||||
return acquireWithWorktree(launchProfile, memberRole, requestedCwd, callerCwd, ownerTerminal, wt,
|
||||
sessionName, resumeSessionId, reservation);
|
||||
} catch (RuntimeException e) {
|
||||
memberLifecycle.release(reservation);
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -257,6 +303,32 @@ public final class SessionManager implements TurnListener {
|
||||
*/
|
||||
private void release(String paneId, ReleaseCause cause) {
|
||||
MemberSession removed = registry.remove(paneId);
|
||||
releaseRemoved(paneId, removed, handles.remove(paneId), cause);
|
||||
}
|
||||
|
||||
/**
|
||||
* Tear a session down only while {@code expected} is still its registry value. A lifecycle
|
||||
* transition replaces the immutable record, so this prevents a reap based on an old READY or
|
||||
* DONE record from stopping a worker that delivery has made BUSY.
|
||||
*/
|
||||
private boolean releaseIfCurrent(MemberSession expected, ReleaseCause cause) {
|
||||
if (!registry.remove(expected.paneId(), expected)) {
|
||||
// A lifecycle transition replaced the record between the caller's check and this remove.
|
||||
// Log it: this race is by definition unobservable otherwise, and a reaper that silently
|
||||
// declines to reap is the hardest kind of behaviour to diagnose after the fact.
|
||||
log.debug("skipping reap of pane={}: its registry record changed after the idle check "
|
||||
+ "(most likely a delivery made it BUSY)", expected.paneId());
|
||||
return false;
|
||||
}
|
||||
releaseRemoved(expected.paneId(), expected, handles.remove(expected.paneId()), cause);
|
||||
return true;
|
||||
}
|
||||
|
||||
private void releaseRemoved(String paneId, MemberSession removed, PeerHandle removedHandle,
|
||||
ReleaseCause cause) {
|
||||
// fleetd #209: remove right alongside the registry entry so a released session's handle is
|
||||
// never leaked — but keep the local reference below, so the id can still be resolved for
|
||||
// the ReleaseDetail this teardown notifies with.
|
||||
boolean preserveWorktree = cause == ReleaseCause.SHUTDOWN;
|
||||
String snapshotRef = null;
|
||||
if (removed != null) {
|
||||
@@ -303,8 +375,11 @@ public final class SessionManager implements TurnListener {
|
||||
// too, so a failed ticket's detail can point a lead at the same tree to re-dispatch.
|
||||
// CB-584 (issue #65 criterion 5): carry agentSessionId alongside them, so a lead can
|
||||
// also resume the member's conversation, not just re-dispatch onto its files.
|
||||
notifyReleased(new ReleaseDetail(removed.terminalId(), removed.worktree(),
|
||||
removed.branch(), snapshotRef, removed.agentSessionId()));
|
||||
// fleetd #209: a late-resolving adapter (opencode) may only now have an id — resolve
|
||||
// one last time so a released member's detail carries the id it now has.
|
||||
MemberSession resolved = resolveAgentSessionId(removed, removedHandle);
|
||||
notifyReleased(new ReleaseDetail(resolved.terminalId(), resolved.worktree(),
|
||||
resolved.branch(), snapshotRef, resolved.agentSessionId()));
|
||||
}
|
||||
}
|
||||
// CB-581: the pane must always stop, even if the dirty check above threw. A session removed
|
||||
@@ -312,7 +387,20 @@ public final class SessionManager implements TurnListener {
|
||||
// slot that no longer appears in the roster and can never be reclaimed.
|
||||
launcher.stop(paneId);
|
||||
if (removed != null && !preserveWorktree && removed.worktree() != null) {
|
||||
worktrees.remove(worktrees.repoRoot(removed.cwd()), removed.worktree());
|
||||
// fleetd #283: this is the one cleanup step in this method that used to be bare. By the
|
||||
// time it runs, the registry entry, the retained handle, and the pane are all already
|
||||
// gone — so a throw here (a stale index lock, a slow filesystem, `remove`'s own 30s exec
|
||||
// timeout) must not escape release(): there is no retry path (a second stop on this
|
||||
// paneId is a no-op), and the caller would otherwise see a "failed stop" for a session
|
||||
// that is in fact fully torn down. Log and swallow, matching every sibling step above.
|
||||
try {
|
||||
worktrees.remove(worktrees.repoRoot(removed.cwd()), removed.worktree());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("failed to remove worktree {} for pane={} terminal={} after release: the "
|
||||
+ "pane is already stopped and the session already deregistered, so this is "
|
||||
+ "not retryable — the directory must be reclaimed manually: {}",
|
||||
removed.worktree(), paneId, removed.terminalId(), e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -451,7 +539,8 @@ public final class SessionManager implements TurnListener {
|
||||
|
||||
private MemberSession acquireWithWorktree(String profile, MemberRole memberRole, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt,
|
||||
String sessionName, String resumeSessionId) {
|
||||
String sessionName, String resumeSessionId,
|
||||
MemberLifecycle.SlotReservation reservation) {
|
||||
String preResolvedProfile = (profile == null || profile.isBlank())
|
||||
? launcher.defaultProfile() : profile;
|
||||
// CB-507: resolve through the launcher's CB-112 chain (requested → profile cwd → caller →
|
||||
@@ -472,27 +561,49 @@ public final class SessionManager implements TurnListener {
|
||||
try {
|
||||
path = worktrees.add(repoRoot, branch, wt.baseRef());
|
||||
worktrees.overlayParity(repoRoot, path, launcher.parityOverlay(preResolvedProfile));
|
||||
// fleetd #185 stage 3: MUST run after overlayParity, not folded into add() — overlayParity
|
||||
// copies more files into the worktree after add() returns, so sharing the group any earlier
|
||||
// leaves those overlay files operator-owned and read-only for a different-uid member.
|
||||
worktrees.shareWithGroup(repoRoot, path);
|
||||
handle = launcher.spawn(new SpawnRequest(profile, path, callerCwd, sessionName, resumeSessionId, memberRole));
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("spawn failed for profile={} role={} branch={} path={}: {}",
|
||||
preResolvedProfile, memberRole, branch, path, e.getMessage());
|
||||
if (path != null) {
|
||||
// fleetd #283: this catch covers every failure AFTER worktrees.add() returned —
|
||||
// overlayParity, shareWithGroup, launcher.spawn itself — so by this point `branch`
|
||||
// was actually created in git. #274 fixed the sibling failure INSIDE add() by having
|
||||
// GitWorktrees.cleanupAfterAddFailure delete both the worktree and the branch it
|
||||
// provisioned; this path removed only the worktree and left the branch orphaned. A
|
||||
// spawn failure here is routine (a quarantined credential, a backend refusal), so
|
||||
// every occurrence leaked a `worker/<slug>-<nonce>` branch nothing ever pointed at
|
||||
// again. Reuse the same Worktrees.deleteBranch GitWorktrees already has, rather than
|
||||
// a second copy of the git command. Best-effort and log-only, like the worktree
|
||||
// removal right above it — neither cleanup step may mask the original exception.
|
||||
try {
|
||||
worktrees.remove(repoRoot, path);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to clean up worktree {} after spawn error: {}", path, cleanup.getMessage());
|
||||
}
|
||||
try {
|
||||
worktrees.deleteBranch(repoRoot, branch);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to clean up branch {} after spawn error: {}", branch, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, path, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
// CB-619: see the no-worktree path above — bind before recording, and store the returned
|
||||
// actual role, so this session's role is never a lie about what it actually holds.
|
||||
MemberRole actualRole = acquired(memberRole, resolvedProfile, handle.terminalId(), reservation);
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
memberRole,
|
||||
actualRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
@@ -504,13 +615,25 @@ public final class SessionManager implements TurnListener {
|
||||
handle.charterReceipt(),
|
||||
handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
memberLifecycle.acquired(session.role(), session.profile(), session.terminalId());
|
||||
handles.put(handle.id(), handle);
|
||||
log.debug("acquired worktree session id={} terminal={} profile={} branch={} path={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.branch(), session.worktree());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
|
||||
private MemberRole acquired(MemberRole role, String profile, String terminal,
|
||||
MemberLifecycle.SlotReservation reservation) {
|
||||
if (reservation == null) {
|
||||
return memberLifecycle.acquired(role, profile, terminal);
|
||||
}
|
||||
if (memberLifecycle.bind(reservation, terminal)) {
|
||||
return role;
|
||||
}
|
||||
memberLifecycle.release(reservation);
|
||||
return memberLifecycle.acquired(role, profile, terminal);
|
||||
}
|
||||
|
||||
private String slug(String raw) {
|
||||
return raw == null ? "ticket" : raw.toLowerCase().replaceAll("[^a-z0-9]+", "-").replaceAll("^-+|-+$", "");
|
||||
}
|
||||
@@ -535,16 +658,73 @@ public final class SessionManager implements TurnListener {
|
||||
return launcher.defaultProfile();
|
||||
}
|
||||
|
||||
/** The session for {@code paneId}, if it is still registered and not released. */
|
||||
/**
|
||||
* The session for {@code paneId}, if it is still registered and not released. fleetd #209:
|
||||
* resolves a still-unknown {@code agentSessionId} against the retained handle before returning,
|
||||
* so {@code fleet_status} sees an id a lazy-resolving adapter has since written.
|
||||
*/
|
||||
public Optional<MemberSession> get(String paneId) {
|
||||
return Optional.ofNullable(registry.get(paneId));
|
||||
return Optional.ofNullable(registry.get(paneId)).map(this::resolveAgentSessionId);
|
||||
}
|
||||
|
||||
/** Fleet-owned roster: all registered sessions (acquired minus released). */
|
||||
/**
|
||||
* Fleet-owned roster: all registered sessions (acquired minus released). Deliberately does
|
||||
* <strong>not</strong> resolve {@code agentSessionId} (fleetd #209 follow-up) — this is the
|
||||
* roster supplier on the heartbeat and health-tick timers ({@code LeadHeartbeatLoop},
|
||||
* {@code FleetHealthMonitor} in {@code Fleetd}), on the placement/exhaustion paths, and on the
|
||||
* metrics scrape ({@code FleetMetrics}), all called far more often than any caller actually
|
||||
* reads {@code agentSessionId}. Resolving here would mean every tick opens a lazy-resolving
|
||||
* adapter's (opencode's) on-disk session store once per member whose id is still unknown — and
|
||||
* for a member whose id never appears, that cost never stops, for the life of the process. Use
|
||||
* {@link #rosterResolved()} instead wherever the id must be current.
|
||||
*/
|
||||
public List<MemberSession> roster() {
|
||||
return List.copyOf(registry.values());
|
||||
}
|
||||
|
||||
/**
|
||||
* {@link #roster()}, with each session's still-unknown {@code agentSessionId} re-resolved
|
||||
* against its retained handle (fleetd #209) — so a caller that actually reports the id (
|
||||
* {@code fleet_list}, the REST roster) sees one a lazy-resolving adapter (opencode) has since
|
||||
* written, rather than the null frozen in at spawn time. Reserved for caller-driven reads, not
|
||||
* timers: see {@link #roster()}'s javadoc for why the plain roster must stay non-resolving.
|
||||
*/
|
||||
public List<MemberSession> rosterResolved() {
|
||||
return registry.values().stream().map(this::resolveAgentSessionId).toList();
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve {@code session}'s {@code agentSessionId} if still unknown, re-polling the retained
|
||||
* {@link PeerHandle} for this pane (fleetd #209). A no-op — returning {@code session} unchanged
|
||||
* — once the id is already known, once no handle is retained for this pane (never spawned, or
|
||||
* already released), or if the handle throws while answering. A resolved id is best-effort
|
||||
* CAS-swapped into the registry via {@link #replace}; a lost race just means another caller
|
||||
* already applied the same update, so the resolved value is returned either way.
|
||||
*/
|
||||
private MemberSession resolveAgentSessionId(MemberSession session) {
|
||||
return resolveAgentSessionId(session, handles.get(session.paneId()));
|
||||
}
|
||||
|
||||
private MemberSession resolveAgentSessionId(MemberSession session, PeerHandle handle) {
|
||||
if (session.agentSessionId() != null || handle == null) {
|
||||
return session;
|
||||
}
|
||||
String resolved;
|
||||
try {
|
||||
resolved = handle.agentSessionId();
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("agentSessionId lookup failed for pane={} terminal={}: {}",
|
||||
session.paneId(), session.terminalId(), e.toString());
|
||||
return session;
|
||||
}
|
||||
if (resolved == null) {
|
||||
return session;
|
||||
}
|
||||
MemberSession updated = session.withAgentSessionId(resolved);
|
||||
replace(session, updated); // best-effort; a lost CAS just means the resolved value stands anyway
|
||||
return updated;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-304 merged roster+live view. The registry is authoritative for worktree, branch,
|
||||
* profile, owner, and state; the optional live agent supplies the herdr-reported status.
|
||||
@@ -559,6 +739,9 @@ public final class SessionManager implements TurnListener {
|
||||
m.put("profile", session.profile());
|
||||
m.put("role", session.role() == null ? "dev" : session.role().wireName());
|
||||
m.put("state", session.state().name().toLowerCase());
|
||||
if (session.failureReason() != null) {
|
||||
m.put("failureReason", session.failureReason());
|
||||
}
|
||||
if (session.worktree() != null) {
|
||||
m.put("worktree", session.worktree());
|
||||
}
|
||||
@@ -619,6 +802,36 @@ public final class SessionManager implements TurnListener {
|
||||
completeTurn(target, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* Record that a member backend failed after a delegated ticket expired. This CAS loop accepts
|
||||
* either side of the completion race: {@code BUSY -> BACKEND_ERROR} or
|
||||
* {@code DONE -> BACKEND_ERROR}. A released or unknown session is never recreated.
|
||||
*
|
||||
* @return {@code true} when the error is recorded on a live session
|
||||
*/
|
||||
public boolean onBackendError(String target, String reason) {
|
||||
while (true) {
|
||||
MemberSession current = findByTerminal(target);
|
||||
if (current == null) {
|
||||
log.warn("backend error for unknown member terminal={}", target);
|
||||
return false;
|
||||
}
|
||||
if (current.state() == MemberSession.State.RELEASED) {
|
||||
return false;
|
||||
}
|
||||
if (current.state() == MemberSession.State.BACKEND_ERROR) {
|
||||
return true;
|
||||
}
|
||||
MemberSession updated = current.withState(MemberSession.State.BACKEND_ERROR)
|
||||
.withFailureReason(reason).withActivity(nowNanos.getAsLong());
|
||||
if (replace(current, updated)) {
|
||||
log.warn("member terminal={} pane={} transitioned {} -> BACKEND_ERROR: {}",
|
||||
target, current.paneId(), current.state(), reason);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
if (!clearAfterTurn) return false;
|
||||
@@ -637,10 +850,9 @@ public final class SessionManager implements TurnListener {
|
||||
if (current == null || current.state() != MemberSession.State.BUSY) return false;
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession updated = current.withState(MemberSession.State.DONE).withActivity(now);
|
||||
if (replace(current, updated)) {
|
||||
log.debug("session transitioned terminal={} pane={} BUSY -> DONE turn={}",
|
||||
target, current.paneId(), updated.turnCount());
|
||||
}
|
||||
if (!replace(current, updated)) return false;
|
||||
log.debug("session transitioned terminal={} pane={} BUSY -> DONE turn={}",
|
||||
target, current.paneId(), updated.turnCount());
|
||||
if (contextCap > 0 && updated.turnCount() >= contextCap) {
|
||||
release(current.paneId());
|
||||
return false;
|
||||
@@ -687,14 +899,18 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
long idleNanos = now - s.lastActivityAtNanos();
|
||||
if (idleNanos > idleTtlNanos) {
|
||||
log.debug("reaping idle session terminal={} pane={}: idle {}s exceeds the {}s ttl",
|
||||
s.terminalId(), s.paneId(), TimeUnit.NANOSECONDS.toSeconds(idleNanos),
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos));
|
||||
// CB-581: one session that fails to release must not abort the whole reaping pass —
|
||||
// match drainAll's per-session try/catch so the rest of the roster still gets reaped.
|
||||
try {
|
||||
release(s.paneId());
|
||||
reaped++;
|
||||
if (beforeIdleRelease != null) {
|
||||
beforeIdleRelease.accept(s);
|
||||
}
|
||||
if (releaseIfCurrent(s, ReleaseCause.COMPLETED)) {
|
||||
log.debug("reaping idle session terminal={} pane={}: idle {}s exceeds the {}s ttl",
|
||||
s.terminalId(), s.paneId(), TimeUnit.NANOSECONDS.toSeconds(idleNanos),
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos));
|
||||
reaped++;
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("reap failed for pane={} terminal={} worktree={}; continuing with "
|
||||
+ "remaining sessions", s.paneId(), s.terminalId(), s.worktree(), e);
|
||||
@@ -705,20 +921,60 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
|
||||
/**
|
||||
* Gracefully drain all registered sessions on daemon shutdown. For each session that is
|
||||
* {@code BUSY}, poll up to {@code timeoutNanos} for it to leave {@code BUSY}, then release it
|
||||
* regardless. Non-busy sessions are released immediately. A failure releasing one session is
|
||||
* logged and does not abort the rest.
|
||||
* Gracefully drain all registered sessions on daemon shutdown. Non-busy sessions are released
|
||||
* immediately; a {@code BUSY} one is polled until it leaves {@code BUSY}, then released
|
||||
* regardless. A failure releasing one session is logged and does not abort the rest.
|
||||
*
|
||||
* <p>{@code timeoutNanos} is a budget for the WHOLE drain, not a grace period per session: the
|
||||
* deadline is taken once, before the loop. So the first BUSY session can spend all of it, and a
|
||||
* later BUSY one is then released with no wait at all. That is deliberate. This drain is only
|
||||
* one phase of shutdown — {@code Fleetd} closes the message service, the push loop, the
|
||||
* heartbeat, MCP and the router after it — and the whole sequence has to finish inside
|
||||
* launchd's exit window. A per-session grace would let N busy members drain for N * the
|
||||
* timeout, overrun that window, and get the daemon SIGKILLed part-way through; the members not
|
||||
* yet reached would then get no clean release, no preserved-worktree log, and no snapshot.
|
||||
* Cutting one turn short is the cheaper failure, and it is not silent: an abandoned BUSY
|
||||
* session is preserved, snapshotted, and logged at WARN by {@code logPreservedForShutdown}.
|
||||
*
|
||||
* <p>CB-544: this is a {@link ReleaseCause#SHUTDOWN} release — the worker's pane is stopped
|
||||
* (the process must end) but its worktree is preserved and its path logged. Shutdown is never
|
||||
* a reason to delete a worker's only copy of its uncommitted work. A session still {@code BUSY}
|
||||
* when the timeout expired is abandoned mid-turn and logged loudly so an operator can find its
|
||||
* kept worktree.
|
||||
*
|
||||
* <p>fleetd #308: {@code roster()} is a one-shot snapshot (see its javadoc), and nothing used
|
||||
* to stop a new session from registering after it was taken — {@link #acquire} stayed open for
|
||||
* as long as this drain waited on a {@code BUSY} session, up to the whole {@code timeoutNanos}
|
||||
* budget. Two things close that window, deliberately paired because neither alone is complete:
|
||||
* {@link #draining} is flipped true before the snapshot is even taken, so {@link #acquire}
|
||||
* refuses (invariant 3: loudly, via {@link ShuttingDownException}) as much of the window as a
|
||||
* plain flag can close; and the sweep below re-reads the registry once the initial snapshot has
|
||||
* fully drained and drains whatever a straggler — a caller that read the flag as {@code false}
|
||||
* a moment before it flipped — still managed to register. The sweep shares the same
|
||||
* {@code deadline} rather than getting its own: {@code timeoutNanos} is a budget for the WHOLE
|
||||
* drain (see above), and a straggler must not buy the drain more time than the flag it lost the
|
||||
* race against would have. In the ordinary case the sweep finds nothing and costs one empty
|
||||
* {@link #roster()} call.
|
||||
*/
|
||||
void drainAll(long timeoutNanos) {
|
||||
long deadline = System.nanoTime() + timeoutNanos;
|
||||
for (MemberSession s : roster()) {
|
||||
draining.set(true);
|
||||
drainSnapshot(roster(), deadline);
|
||||
List<MemberSession> stragglers = roster();
|
||||
if (!stragglers.isEmpty()) {
|
||||
log.warn("drain sweep found {} session(s) registered after the drain snapshot was "
|
||||
+ "taken (raced past the shutdown guard); draining them too", stragglers.size());
|
||||
drainSnapshot(stragglers, deadline);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Drain exactly the sessions in {@code snapshot}, waiting out a {@code BUSY} one against the
|
||||
* shared whole-drain {@code deadline} before releasing it. Shared by {@link #drainAll}'s main
|
||||
* pass and its post-loop straggler sweep (fleetd #308) so both honor the same one budget.
|
||||
*/
|
||||
private void drainSnapshot(List<MemberSession> snapshot, long deadline) {
|
||||
for (MemberSession s : snapshot) {
|
||||
try {
|
||||
if (s.state() == MemberSession.State.BUSY) {
|
||||
while (System.nanoTime() < deadline) {
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
package dev.ltms.fleet.session;
|
||||
|
||||
/**
|
||||
* Thrown by {@link SessionManager#acquire} when a spawn is requested after the daemon's shutdown
|
||||
* drain has already begun (fleetd #308).
|
||||
*
|
||||
* <p>{@link SessionManager#drainAll} snapshots the registry once and tears down exactly what is
|
||||
* in that snapshot. A session registered after the snapshot is invisible to the drain loop: its
|
||||
* pane is left running and its worktree is never preserved, and nothing else ever reclaims
|
||||
* either — the daemon's in-memory registry dies with the process. Refusing the spawn here,
|
||||
* loudly, is what stops that session from ever being created in the first place, rather than
|
||||
* silently handing the caller a session the daemon can no longer manage.
|
||||
*/
|
||||
public final class ShuttingDownException extends RuntimeException {
|
||||
public ShuttingDownException(String message) {
|
||||
super(message);
|
||||
}
|
||||
}
|
||||
@@ -11,6 +11,15 @@ public interface Worktrees {
|
||||
/** git -C <repoRoot> worktree remove --force <path>. Idempotent (already-gone tolerated). */
|
||||
void remove(String repoRoot, String worktreePath);
|
||||
|
||||
/**
|
||||
* git -C {@code repoRoot} branch -D {@code branch}. Force-deletes a branch that has no other
|
||||
* owner — used only on the failed-provisioning path (fleetd #274, #283), never on a normal
|
||||
* release: {@link SessionManager#release} deliberately leaves a released session's branch
|
||||
* behind so a lead can still recover the work, and this method must never be called from
|
||||
* that path.
|
||||
*/
|
||||
void deleteBranch(String repoRoot, String branch);
|
||||
|
||||
/**
|
||||
* True when the worktree holds uncommitted changes the bridge cannot see: tracked
|
||||
* modifications, staged files, or untracked files. {@code git status --porcelain} is the
|
||||
@@ -98,4 +107,21 @@ public interface Worktrees {
|
||||
/** CB-586: the operator-visible census of {@code refs/wip/*} in one repository. */
|
||||
record WipRefStats(int count, long costBytes) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Make {@code repoRoot}'s git store and {@code worktreePath} writable by the configured group
|
||||
* (fleetd #185 stage 3), so a member spawned as a different OS user (see
|
||||
* {@code memberHerdrSocket}) can write its own worktree, its per-worktree git metadata, and
|
||||
* its own commit objects. No-op when no group is configured.
|
||||
*
|
||||
* <p><strong>This isolates credentials, not the repository.</strong> A member in the group can
|
||||
* still write the operator's git objects and refs in the shared repo — this only fixes file
|
||||
* ownership/permissions so a different-uid member can work at all, it grants no narrower access
|
||||
* than that.
|
||||
*
|
||||
* @param repoRoot the repository whose git store ({@code .git/objects}, {@code refs},
|
||||
* {@code logs}, {@code worktrees}, {@code packed-refs}) needs sharing
|
||||
* @param worktreePath the linked worktree's own directory
|
||||
*/
|
||||
void shareWithGroup(String repoRoot, String worktreePath);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.inject.BackendErrorPatternLookup;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
|
||||
/**
|
||||
* fleetd #248 / fleetd#201 Unit 5: {@link Fleetd#backendErrorPatternLookup} is the factory that
|
||||
* replaced the local lambda {@code Fleetd.main} used to build {@code backendErrorPatterns} — one
|
||||
* of the two arguments {@code CompletionResolver} lost cleanly (0 compile errors, every test still
|
||||
* green) when this ticket's measurement dropped it alongside {@code backendErrorSink}. This class
|
||||
* proves the factory's own behaviour; {@code FleetdCompletionResolverWiringTest} proves {@code
|
||||
* main} still passes its result into {@code CompletionResolver}.
|
||||
*/
|
||||
class FleetdBackendErrorPatternLookupTest {
|
||||
|
||||
private static MemberSession session(String terminal, String profile) {
|
||||
return new MemberSession("pane-" + terminal, terminal, profile, MemberRole.DEV,
|
||||
"/cwd", null, 0L, 0L, 0, MemberSession.State.READY, null, null);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a target on a profile with a configured pattern resolves to that pattern")
|
||||
void configuredProfileResolves() {
|
||||
Map<String, Pattern> byProfile = Map.of("terra", Pattern.compile("(?i)503"));
|
||||
BackendErrorPatternLookup lookup =
|
||||
Fleetd.backendErrorPatternLookup(() -> List.of(session("term1", "terra")), byProfile);
|
||||
|
||||
assertEquals("(?i)503", lookup.patternFor("term1").pattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a target on a profile with no configured pattern resolves to null")
|
||||
void unconfiguredProfileResolvesToNull() {
|
||||
Map<String, Pattern> byProfile = Map.of("terra", Pattern.compile("x"));
|
||||
BackendErrorPatternLookup lookup =
|
||||
Fleetd.backendErrorPatternLookup(() -> List.of(session("term1", "sol")), byProfile);
|
||||
|
||||
assertNull(lookup.patternFor("term1"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an unknown target resolves to null")
|
||||
void unknownTargetResolvesToNull() {
|
||||
BackendErrorPatternLookup lookup =
|
||||
Fleetd.backendErrorPatternLookup(List::of, Map.of("terra", Pattern.compile("x")));
|
||||
|
||||
assertNull(lookup.patternFor("term_stranger"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,269 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.BackendErrorSink;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #248 / fleetd#201 Unit 5: {@link Fleetd#backendErrorSink} is the factory that replaced
|
||||
* the local lambda {@code Fleetd.main} used to build {@code backendErrorSink} — the other half of
|
||||
* the pair this ticket's measurement dropped cleanly (0 compile errors, every test still green).
|
||||
*
|
||||
* <p>Before this ticket, the closest thing to coverage was {@code
|
||||
* dev.ltms.fleet.inject.BackendOutageFlowTest}, whose own class doc said it "mirrors {@code
|
||||
* Fleetd.main}'s {@code backendErrorSink} lambda line-for-line" — a hand-copy that proves itself,
|
||||
* never that {@code main} still wires the real thing. This class exercises the actual production
|
||||
* factory instead. {@code FleetdCompletionResolverWiringTest} proves {@code main} still passes its
|
||||
* result into {@code CompletionResolver}.
|
||||
*/
|
||||
class FleetdBackendErrorSinkTest {
|
||||
|
||||
private final List<ScheduledExecutorService> schedulers = new ArrayList<>();
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
schedulers.forEach(ScheduledExecutorService::shutdownNow);
|
||||
}
|
||||
|
||||
private static FleetConfig.Profile stubWorker(String profile, String credentialId) {
|
||||
return new FleetConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, null, null,
|
||||
null, null, credentialId, null);
|
||||
}
|
||||
|
||||
private static Map<String, FleetConfig.Profile> orderedProfiles() {
|
||||
Map<String, FleetConfig.Profile> m = new LinkedHashMap<>();
|
||||
m.put("terra", stubWorker("terra", "shared-openai"));
|
||||
m.put("sol", stubWorker("sol", "shared-openai"));
|
||||
return m;
|
||||
}
|
||||
|
||||
/** Minimal recording {@code HerdrClient} for the LEAD pane — mirrors ReplyPushLoopTest's own. */
|
||||
private static final class RecordingLeadClient implements HerdrClient {
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
private final List<Object> prompts = new CopyOnWriteArrayList<>();
|
||||
volatile CountDownLatch sendLatch = new CountDownLatch(1);
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
if ("agent.get".equals(method)) {
|
||||
return MAPPER.createObjectNode().set("agent", MAPPER.createObjectNode()
|
||||
.put("terminal_id", "term_primary").put("agent_status", "idle"));
|
||||
}
|
||||
if ("agent.prompt".equals(method)) {
|
||||
prompts.add(params);
|
||||
sendLatch.countDown();
|
||||
}
|
||||
return MAPPER.createObjectNode();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
|
||||
int sendCount() {
|
||||
return prompts.size();
|
||||
}
|
||||
}
|
||||
|
||||
/** A {@link PeerLauncher} that never actually spawns — enough to construct a bare {@link SessionManager}. */
|
||||
private static final class NeverSpawnsLauncher implements PeerLauncher {
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
return Set.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilitiesFor(String profileName) {
|
||||
return Set.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
throw new UnsupportedOperationException("not reachable — this test never acquires a session");
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return Set.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
throw new UnsupportedOperationException("not reachable — this test never acquires a session");
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<?> list() {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a target with no resolvable profile logs and returns without recording an incident (never throws)")
|
||||
void unresolvableProfileDoesNotRecordOrThrow() {
|
||||
SessionManager sessions = new SessionManager(new NeverSpawnsLauncher());
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
RecordingLeadClient leadClient = new RecordingLeadClient();
|
||||
InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||
PrimaryRegistry registry = new PrimaryRegistry(null);
|
||||
ScheduledExecutorService scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
schedulers.add(scheduler);
|
||||
ReplyPushLoop pushLoop = new ReplyPushLoop(registry, new AgentControl(leadClient), inbox, scheduler, 3, 50);
|
||||
|
||||
BackendErrorSink sink = Fleetd.backendErrorSink(sessions, Map::of, outagePolicy, () -> pushLoop);
|
||||
sink.onBackendError("term_unmapped", "matched line", "503 Service Unavailable");
|
||||
|
||||
assertTrue(outagePolicy.remainingCoolOffSeconds("shared-openai").isEmpty(),
|
||||
"no credential is ever resolvable here, so nothing must be recorded");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("two distinct targets classified through the real factory start an incident and cool the credential")
|
||||
void twoDistinctTargetsStartAnIncident() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.readText("⏺ 503 Service Unavailable: upstream credential rejected\n❯ ");
|
||||
Map<String, FleetConfig.Profile> profiles = orderedProfiles();
|
||||
AtomicLong clockNanos = new AtomicLong(0L);
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(clockNanos::get);
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr),
|
||||
new WorkspaceControl(herdr), new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
profiles, "terra", _ -> "tok");
|
||||
CompositePeerLauncher workers = new CompositePeerLauncher(List.of(adapter), "terra", profiles,
|
||||
PlacementPolicies.weighted(), _ -> 0, null, BackendQuarantine.none(), outagePolicy);
|
||||
SessionManager sessions = new SessionManager(workers);
|
||||
MemberSession s1 = sessions.acquire("terra", null, null, null);
|
||||
MemberSession s2 = sessions.acquire("terra", null, null, null);
|
||||
|
||||
PrimaryRegistry registry = new PrimaryRegistry(null);
|
||||
registry.recordDelegation(s1.terminalId(), "term_primary");
|
||||
registry.recordDelegation(s2.terminalId(), "term_primary");
|
||||
InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||
inbox.own(s1.terminalId());
|
||||
inbox.own(s2.terminalId());
|
||||
RecordingLeadClient leadClient = new RecordingLeadClient();
|
||||
ScheduledExecutorService scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
schedulers.add(scheduler);
|
||||
ReplyPushLoop pushLoop = new ReplyPushLoop(registry, new AgentControl(leadClient), inbox, scheduler, 3, 50);
|
||||
AtomicReference<ReplyPushLoop> pushLoopRef = new AtomicReference<>(pushLoop);
|
||||
|
||||
// The exact object under test: Fleetd's real production factory, not a hand copy.
|
||||
BackendErrorSink sink = Fleetd.backendErrorSink(sessions, () -> profiles, outagePolicy, pushLoopRef::get);
|
||||
|
||||
sink.onBackendError(s1.terminalId(), "matched line", "503 Service Unavailable");
|
||||
assertTrue(outagePolicy.remainingCoolOffSeconds("shared-openai").isEmpty(),
|
||||
"one distinct target must not start a cool-off");
|
||||
|
||||
sink.onBackendError(s2.terminalId(), "matched line", "503 Service Unavailable");
|
||||
|
||||
assertTrue(leadClient.sendLatch.await(3, TimeUnit.SECONDS),
|
||||
"the second distinct target must cross the threshold and nudge the lead");
|
||||
var remaining = outagePolicy.remainingCoolOffSeconds("shared-openai");
|
||||
assertTrue(remaining.isPresent(), "two distinct targets must start a cool-off");
|
||||
assertEquals(1, leadClient.sendCount());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a backend-error session no longer blocks the real maxLoad spawn gate")
|
||||
void backendErrorSessionDoesNotBlockFreshSpawnAtMaxLoad() {
|
||||
SessionManager sessions = capacityLimitedSessions();
|
||||
MemberSession failed = sessions.acquire("terra", null, null, null);
|
||||
|
||||
assertTrue(sessions.onBackendError(failed.terminalId(), "backend exited"));
|
||||
|
||||
MemberSession fresh = sessions.acquire("terra", null, null, null);
|
||||
assertEquals("terra", fresh.profile(), "the real maxLoad gate grants a fresh spawn after a backend error");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a failed session no longer blocks the real maxLoad spawn gate")
|
||||
void failedSessionDoesNotBlockFreshSpawnAtMaxLoad() {
|
||||
SessionManager sessions = capacityLimitedSessions();
|
||||
MemberSession failed = sessions.acquire("terra", null, null, null);
|
||||
|
||||
sessions.onTurnFailed(failed.terminalId());
|
||||
|
||||
MemberSession fresh = sessions.acquire("terra", null, null, null);
|
||||
assertEquals("terra", fresh.profile(), "the real maxLoad gate grants a fresh spawn after a failed turn");
|
||||
}
|
||||
|
||||
private static SessionManager capacityLimitedSessions() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile profile = new FleetConfig.Profile("terra", "http://gx00.gw:8000", "coder",
|
||||
null, "FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, 1.0f, 1);
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("terra", profile);
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), profiles, "terra", _ -> "tok");
|
||||
AtomicReference<SessionManager> sessionsRef = new AtomicReference<>();
|
||||
CompositePeerLauncher workers = new CompositePeerLauncher(List.of(adapter), "terra", profiles,
|
||||
PlacementPolicies.fixed(), name -> Fleetd.liveSessionCount(sessionsRef.get().roster(), name));
|
||||
SessionManager sessions = new SessionManager(workers);
|
||||
sessionsRef.set(sessions);
|
||||
return sessions;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #248: this is the test that was actually missing. {@code Fleetd.main} builds its {@code
|
||||
* CompletionResolver} from an 8-argument constructor, and the ticket's own measurement proved two
|
||||
* ways to silently unwire it — both compiled with 0 errors and left every existing test green:
|
||||
*
|
||||
* <ul>
|
||||
* <li>replacing the worktree/branch argument (the 8th) with {@code _ -> null} — drops
|
||||
* fleetd#241's fallback-report location entirely;</li>
|
||||
* <li>replacing {@code backendErrorPatterns, backendErrorSink} (5th/6th) with {@code
|
||||
* BackendErrorPatternLookup.legacy(), BackendErrorSink.none()} — drops fleetd#201 Unit 5's
|
||||
* backend-error classification and cool-off entirely.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Neither mutation could be caught by any test that constructs its own {@code
|
||||
* CompletionResolver} (every test before this one did exactly that) or by a test of {@link
|
||||
* Fleetd#worktreeBranchLookup}, {@link Fleetd#backendErrorPatternLookup}, or {@link
|
||||
* Fleetd#backendErrorSink} in isolation (see {@code FleetdWorktreeBranchLookupTest}, {@code
|
||||
* FleetdBackendErrorPatternLookupTest}, {@code FleetdBackendErrorSinkTest}) — those prove the
|
||||
* factories work, never that {@code main} still calls them. This class is a plain source-text
|
||||
* assertion on {@code Fleetd.java} — crude, but honest about what it checks, and it turns red the
|
||||
* instant the wiring is dropped, mirroring the same fallback shape {@link
|
||||
* FleetdFleetAppConstructionTest} already uses for a different constructor argument.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||
* CompletionResolver} and never runs {@code main}.
|
||||
*/
|
||||
class FleetdCompletionResolverWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] CompletionResolver's construction call still names backendErrorPatterns and backendErrorSink")
|
||||
void backendErrorArgumentsAreStillNamedAtTheCallSite() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"exhaustionSink, backendErrorPatterns, backendErrorSink, System::nanoTime,"),
|
||||
"CompletionResolver's construction call must still pass backendErrorPatterns and "
|
||||
+ "backendErrorSink as its 5th/6th arguments. Replacing them with "
|
||||
+ "BackendErrorPatternLookup.legacy()/BackendErrorSink.none() (fleetd #248's measured "
|
||||
+ "mutation) compiles with 0 errors and leaves every behavioural test green — this "
|
||||
+ "source check is what must go red instead.");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] CompletionResolver's construction call still passes worktreeBranchLookup(sessions::roster)")
|
||||
void worktreeBranchLookupIsStillPassedAtTheCallSite() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains("worktreeBranchLookup(sessions::roster)"),
|
||||
"CompletionResolver's construction call must still pass worktreeBranchLookup(sessions::roster) "
|
||||
+ "as its 8th (last) argument. Replacing it with the inert `_ -> null` (fleetd #248's "
|
||||
+ "other measured mutation) compiles with 0 errors and leaves every behavioural test "
|
||||
+ "green — this source check is what must go red instead.");
|
||||
assertFalse(source.contains("System::nanoTime,\n _ -> null"),
|
||||
"the worktree/branch argument must never regress to the inert `_ -> null` literal");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] backendErrorPatterns is assigned from the extracted backendErrorPatternLookup(...) factory")
|
||||
void backendErrorPatternsComesFromTheFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"BackendErrorPatternLookup backendErrorPatterns = backendErrorPatternLookup(sessions::roster,"),
|
||||
"backendErrorPatterns must be assigned from Fleetd.backendErrorPatternLookup(...), not an "
|
||||
+ "inline lambda that a source check on the CompletionResolver call alone cannot see "
|
||||
+ "through");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] backendErrorSink is assigned from the extracted backendErrorSink(...) factory")
|
||||
void backendErrorSinkComesFromTheFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"BackendErrorSink backendErrorSink = backendErrorSink(sessions, () -> config.get().profiles(),"),
|
||||
"backendErrorSink must be assigned from Fleetd.backendErrorSink(...), not an inline lambda "
|
||||
+ "that a source check on the CompletionResolver call alone cannot see through");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,31 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-185: {@code ConnectionIdentity} must resolve a caller's pane on EITHER herdr daemon (a
|
||||
* lead's MCP connection resolves against the lead daemon; a member's against the member daemon).
|
||||
* Pinning {@code PaneLocator} to {@code memberHerdr} alone — the bug this guards against — leaves
|
||||
* every lead's own connection unresolvable ({@code callerTerminal == null}) the moment
|
||||
* {@code memberHerdrSocket} names a second daemon, which breaks {@code fleet_reply}/{@code
|
||||
* fleet_ask} and {@code fleet_whoami} for a lead. A unit test on {@link
|
||||
* dev.ltms.fleet.herdr.PaneLocator} alone (see {@code PaneLocatorTest}) proves the class CAN
|
||||
* search two clients, but not that {@code Fleetd.main} actually wires it that way — hence this
|
||||
* source-level assertion, the same technique {@code FleetdHerdrControlConstructionTest} uses.
|
||||
*/
|
||||
class FleetdConnectionIdentityConstructionTest {
|
||||
@Test
|
||||
void connectionIdentitySearchesBothDaemonsNotJustTheMemberOne() throws Exception {
|
||||
String source = Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
assertFalse(source.contains("new PaneLocator(memberHerdr)"),
|
||||
"PaneLocator must not be pinned to the member daemon alone — a lead's own "
|
||||
+ "connection resolves against the LEAD daemon and would never be found");
|
||||
assertTrue(source.contains("new PaneLocator(herdr, memberHerdr)"),
|
||||
"PaneLocator must search the lead daemon first, then the member daemon");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,29 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-185: {@code FleetApp} must be constructed with BOTH herdr clients (the lead's and the
|
||||
* member's), never the raw lead-only {@code herdr}. Passing only {@code herdr} — the bug this
|
||||
* guards against — makes {@code GET /healthz} green while the member daemon is down (so every
|
||||
* spawn fails invisibly) and silently drops every member workspace from {@code GET /sessions}.
|
||||
* A behavioural test on {@code FleetApp} alone (see {@code FleetAppTwoDaemonTest}) proves the
|
||||
* class merges/gates correctly when given two clients, but not that {@code Fleetd.main} actually
|
||||
* passes it two — hence this source-level assertion, mirroring
|
||||
* {@code FleetdHerdrControlConstructionTest}.
|
||||
*/
|
||||
class FleetdFleetAppConstructionTest {
|
||||
@Test
|
||||
void fleetAppIsConstructedWithBothHerdrDaemons() throws Exception {
|
||||
String source = Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
assertFalse(source.contains("new FleetApp(herdr, workers,"),
|
||||
"FleetApp must not be constructed with the lead-only herdr client");
|
||||
assertTrue(source.contains("new FleetApp(herdr, memberHerdr, workers,"),
|
||||
"FleetApp must be constructed with both the lead and the member herdr client");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
|
||||
class FleetdHerdrControlConstructionTest {
|
||||
@Test
|
||||
void fleetdDelegatesStatefulControlsToTheRouter() throws Exception {
|
||||
// AgentControl caches paneByTerminal, so the router must be its only production factory.
|
||||
String source = Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
assertFalse(source.contains("new AgentControl("));
|
||||
assertFalse(source.contains("new WorkspaceControl("));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,156 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* fleetd #176: {@link Fleetd#leadSeatLookup} is the factory {@code Fleetd.main} wires into {@code
|
||||
* FleetMcp.LeadSeatSource} so {@code fleet_list}'s {@code free} can subtract the seat(s) a
|
||||
* {@code subscription: true} profile's own live LEAD session holds on that same account —
|
||||
* {@code maxLoad} never counted the lead, only members. {@code FleetdLeadSeatWiringTest} proves
|
||||
* {@code main} still passes this factory's result in; this class proves the factory's own matching
|
||||
* logic: subscription-only, credential-matched, and counting only CURRENTLY LIVE leads.
|
||||
*/
|
||||
class FleetdLeadSeatLookupTest {
|
||||
|
||||
private static FleetConfig.Profile subscriptionProfile(String name, String credentialId) {
|
||||
return new FleetConfig.Profile(name, null, "claude-sonnet-5", null, null, null,
|
||||
"tab", "fleet", "w #{n}", null, null, null, null, null, null, null,
|
||||
null, 3, true, null, credentialId, null);
|
||||
}
|
||||
|
||||
private static FleetConfig.Profile offSubscriptionProfile(String name, String credentialId) {
|
||||
return new FleetConfig.Profile(name, "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||
null, "tab", "fleet", "w #{n}", null, null, null, null, null, null, null,
|
||||
null, 2, false, null, credentialId, null);
|
||||
}
|
||||
|
||||
private static FleetConfig.Leader leadOnProfile(String profile) {
|
||||
return new FleetConfig.Leader(profile, "lead: primary", 1, "lead:", 10, "claude", "claude-sonnet-5");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a live lead sharing the target profile's credential counts as one seat")
|
||||
void liveLeadSharingCredentialCountsAsOneSeat() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("sonnet", subscriptionProfile("sonnet", null));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("sonnet"));
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders,
|
||||
() -> Map.of("term_primary", "primary"));
|
||||
|
||||
assertEquals(1, lookup.apply("sonnet"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("no live lead names this profile ⇒ zero seats, exactly as before this ticket")
|
||||
void noLiveLeadOnTheProfileCountsAsZero() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("sonnet", subscriptionProfile("sonnet", null));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("sonnet"));
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders, Map::of);
|
||||
|
||||
assertEquals(0, lookup.apply("sonnet"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a lead entry with no `profile:` (recognise-only) contributes no seats — cannot be derived")
|
||||
void recogniseOnlyLeadWithNoProfileContributesNothing() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("sonnet", subscriptionProfile("sonnet", null));
|
||||
FleetConfig.Leader recogniseOnly = new FleetConfig.Leader(null, "lead: primary", 1, "lead:", 10,
|
||||
"claude", "claude-sonnet-5");
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", recogniseOnly);
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders,
|
||||
() -> Map.of("term_primary", "primary"));
|
||||
|
||||
assertEquals(0, lookup.apply("sonnet"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a non-subscription profile never has a lead seat subtracted, whatever the credential match")
|
||||
void nonSubscriptionProfileIsNeverAdjusted() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(
|
||||
"terra", offSubscriptionProfile("terra", "shared-openai"),
|
||||
"sonnet", subscriptionProfile("sonnet", "shared-openai"));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("sonnet"));
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders,
|
||||
() -> Map.of("term_primary", "primary"));
|
||||
|
||||
assertEquals(0, lookup.apply("terra"), "terra is not subscription:true, so it must never be adjusted");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("explicit, different credentialIds still separate two subscription profiles (post fleetd #176 "
|
||||
+ "stage 2 sentinel)")
|
||||
void differentCredentialIsNotCounted() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(
|
||||
"sonnet", subscriptionProfile("sonnet", "claude-account-a"),
|
||||
"opus", subscriptionProfile("opus", "claude-account-b"));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders,
|
||||
() -> Map.of("term_primary", "primary"));
|
||||
|
||||
assertEquals(0, lookup.apply("sonnet"), "different accounts must never be conflated into one seat count "
|
||||
+ "— an explicit credentialId on both sides must still win over the subscription sentinel, so an "
|
||||
+ "operator with two separate Claude logins on one host can keep them apart");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176 stage 2 — the exact live shape that shipped inert: a lead on subscription profile
|
||||
* {@code opus}, members on a DIFFERENTLY NAMED subscription profile {@code sonnet}, same Claude
|
||||
* login, and NEITHER profile sets {@code credentialId}. Every other test in this class puts the
|
||||
* lead on the SAME profile name as the target, which happened to keep working even with the old
|
||||
* fall-back-to-profile-name {@code effectiveCredentialId()} — this is the one that did not, and
|
||||
* its absence is what let the stage-1 fix ship without ever catching the bug it was filed for.
|
||||
*/
|
||||
@Test
|
||||
@DisplayName("[LIVE SHAPE] lead on a DIFFERENT subscription profile, same account, neither sets "
|
||||
+ "credentialId ⇒ still counts as a seat")
|
||||
void leadOnADifferentSubscriptionProfileSameAccountStillCountsAsASeat() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(
|
||||
"opus", subscriptionProfile("opus", null),
|
||||
"sonnet", subscriptionProfile("sonnet", null));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("opus"));
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders,
|
||||
() -> Map.of("term_primary", "primary"));
|
||||
|
||||
assertEquals(1, lookup.apply("sonnet"), "opus and sonnet are both subscription:true with no explicit "
|
||||
+ "credentialId, so they share one Claude login and the lead's live seat on opus must be charged "
|
||||
+ "against sonnet too — this is the live host's actual shape (fleetd #176 stage 2)");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("two live instances of the same lead count as two seats")
|
||||
void twoLiveInstancesOfTheSameLeadCountAsTwoSeats() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("sonnet", subscriptionProfile("sonnet", null));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("sonnet"));
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders,
|
||||
() -> Map.of("term_a", "primary", "term_b", "primary"));
|
||||
|
||||
assertEquals(2, lookup.apply("sonnet"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an unconfigured target profile resolves to zero, not a thrown exception")
|
||||
void unconfiguredTargetProfileIsZero() {
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(Map::of, Map.of(), Map::of);
|
||||
|
||||
assertEquals(0, lookup.apply("ghost"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("live leads are read through the supplier on every call, not snapshotted")
|
||||
void liveLeadsAreReadThroughOnEveryCall() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("sonnet", subscriptionProfile("sonnet", null));
|
||||
Map<String, FleetConfig.Leader> leaders = Map.of("primary", leadOnProfile("sonnet"));
|
||||
java.util.Map<String, String> live = new java.util.HashMap<>();
|
||||
Function<String, Integer> lookup = Fleetd.leadSeatLookup(() -> profiles, leaders, () -> live);
|
||||
|
||||
assertEquals(0, lookup.apply("sonnet"));
|
||||
live.put("term_primary", "primary");
|
||||
assertEquals(1, lookup.apply("sonnet"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,43 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #176: {@code Fleetd.main} builds its {@code FleetMcp} from a 14-argument constructor whose
|
||||
* last argument is a {@code FleetMcp.LeadSeatSource} wrapping {@link Fleetd#leadSeatLookup}. That
|
||||
* argument is exactly the kind of wiring fleetd #248 warned about: dropping it (or swapping it for
|
||||
* the inert {@code FleetMcp.LeadSeatSource.none()}) compiles with 0 errors and leaves every test
|
||||
* that builds its own {@code FleetMcp}/{@code CapacitySource} directly — every test that predates
|
||||
* this ticket — green, because none of them go through {@code main} at all.
|
||||
*
|
||||
* <p>{@link FleetdLeadSeatLookupTest} proves the factory's own matching logic; this class is the
|
||||
* plain source-text assertion that proves {@code main} still passes its result in, mirroring
|
||||
* {@code FleetdCompletionResolverWiringTest}'s approach for the same class of gap.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a
|
||||
* {@code FleetMcp} and never runs {@code main}.
|
||||
*/
|
||||
class FleetdLeadSeatWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] FleetMcp's construction call still passes a LeadSeatSource built from leadSeatLookup(...)")
|
||||
void fleetMcpConstructionStillWiresLeadSeatLookup() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains("new FleetMcp.LeadSeatSource(leadSeatLookup(() -> config.get().profiles(), "
|
||||
+ "leaders, leads))"),
|
||||
"FleetMcp's construction call must still pass a LeadSeatSource built from "
|
||||
+ "Fleetd.leadSeatLookup(...). Dropping it or swapping in "
|
||||
+ "FleetMcp.LeadSeatSource.none() (fleetd #176's would-be silent regression, the same "
|
||||
+ "shape as fleetd #248's measured mutations) compiles with 0 errors and leaves every "
|
||||
+ "existing behavioural test green — this source check is what must go red instead.");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.inject.CompletionResolver;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.function.Function;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
|
||||
/**
|
||||
* fleetd #248: {@link Fleetd#worktreeBranchLookup} is the factory that replaced the anonymous
|
||||
* lambda {@code Fleetd.main} used to build inline, as the 8th (last) argument to {@code
|
||||
* CompletionResolver}'s constructor. Before this ticket that argument was untestable wiring:
|
||||
* replacing it with {@code _ -> null} compiled clean and every existing test stayed green, because
|
||||
* every existing test builds its own {@code CompletionResolver} directly rather than going through
|
||||
* {@code main}. This class proves the factory's own behaviour; {@code
|
||||
* FleetdCompletionResolverWiringTest} proves {@code main} still passes it in.
|
||||
*/
|
||||
class FleetdWorktreeBranchLookupTest {
|
||||
|
||||
private static MemberSession session(String terminal, String worktree, String branch) {
|
||||
return new MemberSession("pane-" + terminal, terminal, "terra", MemberRole.DEV,
|
||||
"/cwd", null, 0L, 0L, 0, MemberSession.State.READY, worktree, branch);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a known target resolves to its session's worktree and branch")
|
||||
void knownTargetResolves() {
|
||||
Function<String, CompletionResolver.WorktreeBranch> lookup =
|
||||
Fleetd.worktreeBranchLookup(() -> List.of(session("term1", "/wt/worker_x", "worker/x")));
|
||||
|
||||
CompletionResolver.WorktreeBranch resolved = lookup.apply("term1");
|
||||
|
||||
assertEquals("/wt/worker_x", resolved.worktree());
|
||||
assertEquals("worker/x", resolved.branch());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an unknown target resolves to null, not a thrown exception")
|
||||
void unknownTargetResolvesToNull() {
|
||||
Function<String, CompletionResolver.WorktreeBranch> lookup =
|
||||
Fleetd.worktreeBranchLookup(() -> List.of(session("term1", "/wt/worker_x", "worker/x")));
|
||||
|
||||
assertNull(lookup.apply("term_stranger"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("the roster is read through the supplier on every call, not snapshotted")
|
||||
void rosterIsReadThroughOnEveryCall() {
|
||||
List<MemberSession> roster = new ArrayList<>();
|
||||
Function<String, CompletionResolver.WorktreeBranch> lookup = Fleetd.worktreeBranchLookup(() -> roster);
|
||||
|
||||
assertNull(lookup.apply("term_late"));
|
||||
roster.add(session("term_late", "/wt/late", "worker/late"));
|
||||
|
||||
assertEquals("/wt/late", lookup.apply("term_late").worktree());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* fleetd #184: startup must state whether members share fleetd's OS user or use a separate herdr.
|
||||
*/
|
||||
class MemberTrustModelReportTest {
|
||||
|
||||
private static FleetConfig load(Path dir, String yaml) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml);
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
private static ListAppender<ILoggingEvent> attach() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
||||
((Logger) LoggerFactory.getLogger(Fleetd.class)).detachAppender(appender);
|
||||
}
|
||||
|
||||
private static String report(FleetConfig cfg) {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
ch.qos.logback.classic.Level original = logger.getLevel();
|
||||
logger.setLevel(ch.qos.logback.classic.Level.INFO);
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportMemberTrustModel(cfg);
|
||||
} finally {
|
||||
detach(appender);
|
||||
logger.setLevel(original);
|
||||
}
|
||||
return appender.list.getFirst().getFormattedMessage();
|
||||
}
|
||||
|
||||
@Test
|
||||
void unsetMemberHerdrSocketStatesThatMembersAreNotSandboxed(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, "bind:\n host: 127.0.0.1\n port: 8765\n");
|
||||
|
||||
assertEquals("member trust model: members run as the same OS user as fleetd, not in a sandbox. "
|
||||
+ "A member can read any file this user can read, including SSH keys and credential "
|
||||
+ "stores, whatever memberCredentials says. To add a real boundary, route members to "
|
||||
+ "a second herdr under a different OS user with memberHerdrSocket.",
|
||||
report(cfg));
|
||||
}
|
||||
|
||||
@Test
|
||||
void configuredMemberHerdrSocketStatesThatFleetdCannotConfirmTheBoundary(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, "memberHerdrSocket: /tmp/member-herdr.sock\n");
|
||||
|
||||
assertEquals("member trust model: members are routed to a separate herdr through "
|
||||
+ "memberHerdrSocket. fleetd cannot see that herdr's uid, so confirm it runs "
|
||||
+ "as a different OS user before treating it as a boundary.",
|
||||
report(cfg));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import com.tngtech.archunit.base.DescribedPredicate;
|
||||
import com.tngtech.archunit.core.domain.JavaClass;
|
||||
import com.tngtech.archunit.core.domain.JavaClass.Predicates;
|
||||
import com.tngtech.archunit.core.importer.ClassFileImporter;
|
||||
import com.tngtech.archunit.core.importer.ImportOption;
|
||||
import com.tngtech.archunit.library.dependencies.SliceRule;
|
||||
import com.tngtech.archunit.library.dependencies.SlicesRuleDefinition;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
/**
|
||||
* fleetd #131 (CB-627): enforce package boundaries with an ArchUnit test instead of a
|
||||
* Maven module split.
|
||||
*
|
||||
* <p>This test fails the build the moment a NEW cycle appears between the top-level
|
||||
* {@code dev.ltms.fleet.*} packages. Today's cycles are recorded below as explicit,
|
||||
* narrow exceptions: each one ignores dependencies between exactly the two named
|
||||
* packages, in both directions, and nothing else. A cycle through any other pair of
|
||||
* packages -- or a brand new pair -- still fails this test.
|
||||
*
|
||||
* <p><b>Main code only.</b> The import excludes test classes
|
||||
* ({@link ImportOption.Predefined#DO_NOT_INCLUDE_TESTS}). Test code legitimately wires
|
||||
* across many packages for setup and mocking; that is not part of the shipped
|
||||
* architecture this rule protects. Verified: importing test classes too pulls in a much
|
||||
* larger, noisier cycle set -- {@code herdr}, {@code member}, {@code peer}, {@code
|
||||
* config}, {@code guard} and {@code placement} all show up in cycles that disappear the
|
||||
* moment test classes are excluded. Scanning off the classpath via {@code
|
||||
* importPackages(...)} (not a hardcoded {@code target/classes} path) also keeps this
|
||||
* test correct regardless of the working directory the build is invoked from.
|
||||
*
|
||||
* <p><b>No package moves here</b> -- ticket #131 is explicit that removing a cycle is
|
||||
* its own, later PR. See the comment on each exception below for which ticket step
|
||||
* removes it.
|
||||
*/
|
||||
class PackageCyclesTest {
|
||||
|
||||
@Test
|
||||
void packagesAreFreeOfCycles() {
|
||||
var classes = new ClassFileImporter()
|
||||
.withImportOption(ImportOption.Predefined.DO_NOT_INCLUDE_TESTS)
|
||||
.importPackages("dev.ltms.fleet");
|
||||
|
||||
SliceRule rule = SlicesRuleDefinition.slices()
|
||||
.matching("dev.ltms.fleet.(*)..")
|
||||
.should().beFreeOfCycles();
|
||||
|
||||
// fleetd #131 step 1: move ConnectionIdentity so authz stops depending on the
|
||||
// MCP layer. Evidence: auth/CallerResolver.java:3 imports mcp.ConnectionIdentity;
|
||||
// mcp/FleetMcp.java:3-7 imports auth.AuditLog, Authz, CallerResolver, Principal,
|
||||
// Role.
|
||||
rule = ignoreCycle(rule, "auth", "mcp");
|
||||
|
||||
// fleetd #131 step 2: PrimaryRegistry is used by loops in msg; move it, or put
|
||||
// an interface between msg and mcp. Evidence: msg/ReplyPushLoop.java:5 and
|
||||
// msg/LeadHeartbeatLoop.java:5 import mcp.PrimaryRegistry; mcp/FleetMcp.java:15-18
|
||||
// imports msg.LeadChannel, LeadMessage, MessageService, Rendezvous.
|
||||
rule = ignoreCycle(rule, "mcp", "msg");
|
||||
|
||||
// fleetd #131 -- found while implementing this test, NOT one of the ticket's
|
||||
// original three; it names its own follow-up step before removal. Evidence:
|
||||
// inject/CompletionResolver.java:4-5, inject/Injector.java:6 and
|
||||
// inject/TurnListener.java:3 import msg.Rendezvous / msg.TurnToken;
|
||||
// msg/MessageService.java:6 imports inject.Injector.
|
||||
rule = ignoreCycle(rule, "inject", "msg");
|
||||
|
||||
// fleetd #131 -- same as above, its own follow-up. Evidence:
|
||||
// metrics/FleetMetrics.java:3 imports msg.ReplyInbox; msg/MessageService.java:7-8,
|
||||
// msg/LeadHeartbeatLoop.java:6-7 and msg/ReplyPushLoop.java:6-7 import
|
||||
// metrics.FleetMetrics / metrics.Metrics.
|
||||
rule = ignoreCycle(rule, "metrics", "msg");
|
||||
|
||||
// fleetd #131 -- same as above, its own follow-up. Evidence:
|
||||
// session/SessionManager.java:7 imports msg.TurnToken;
|
||||
// msg/LeadHeartbeatLoop.java:8 imports session.MemberSession.
|
||||
rule = ignoreCycle(rule, "msg", "session");
|
||||
|
||||
rule.check(classes);
|
||||
}
|
||||
|
||||
/**
|
||||
* Accepts today's known cycle between two top-level packages, and nothing else.
|
||||
* Ignoring both directions removes exactly this pair from cycle detection; every
|
||||
* other dependency -- including any new one added later, between these same two
|
||||
* packages or any other pair -- is still checked.
|
||||
*/
|
||||
private static SliceRule ignoreCycle(SliceRule rule, String packageA, String packageB) {
|
||||
return rule
|
||||
.ignoreDependency(residesIn(packageA), residesIn(packageB))
|
||||
.ignoreDependency(residesIn(packageB), residesIn(packageA));
|
||||
}
|
||||
|
||||
private static DescribedPredicate<JavaClass> residesIn(String topLevelPackage) {
|
||||
return Predicates.resideInAPackage("dev.ltms.fleet." + topLevelPackage + "..");
|
||||
}
|
||||
}
|
||||
@@ -1,8 +1,12 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
@@ -26,19 +30,66 @@ class RequiredSecretEnvVarsTest {
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
private static ListAppender<ILoggingEvent> attach() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
||||
((Logger) LoggerFactory.getLogger(Fleetd.class)).detachAppender(appender);
|
||||
}
|
||||
|
||||
@Test
|
||||
void collectsATokenEnvPerNonSubscriptionProfile(@TempDir Path dir) throws Exception {
|
||||
void explicitlyConfiguredUnsetTokenEnvWarnsWithCurrentWording(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
tokenEnv: CB115_MISSING_TOKEN_ENV_819367
|
||||
""");
|
||||
|
||||
Map<String, List<String>> required = Fleetd.requiredSecretEnvVars(cfg);
|
||||
|
||||
assertTrue(required.containsKey("AI_GATEWAY_TOKEN"));
|
||||
assertEquals(List.of("profile 'local' tokenEnv"), required.get("AI_GATEWAY_TOKEN"));
|
||||
assertTrue(required.containsKey("CB115_MISSING_TOKEN_ENV_819367"));
|
||||
assertEquals(List.of("profile 'local' tokenEnv"), required.get("CB115_MISSING_TOKEN_ENV_819367"));
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportRequiredSecrets(cfg);
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
assertTrue(appender.list.stream().anyMatch(e ->
|
||||
e.getLevel() == ch.qos.logback.classic.Level.WARN
|
||||
&& e.getFormattedMessage().contains("startup secret CB115_MISSING_TOKEN_ENV_819367: MISSING")
|
||||
&& e.getFormattedMessage().contains("profile 'local' tokenEnv")),
|
||||
"an explicitly configured but unset tokenEnv must retain the startup warning");
|
||||
}
|
||||
|
||||
@Test
|
||||
void profileWithoutTokenEnvDoesNotRequireTheOldDefault(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
""");
|
||||
|
||||
assertTrue(Fleetd.requiredSecretEnvVars(cfg).isEmpty(),
|
||||
"an absent tokenEnv means this profile needs no token, not FLEETD_WORKER_TOKEN");
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportRequiredSecrets(cfg);
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
assertFalse(appender.list.stream().anyMatch(e -> e.getLevel() == ch.qos.logback.classic.Level.WARN),
|
||||
"a profile without tokenEnv must not produce a startup secret warning");
|
||||
}
|
||||
|
||||
@Test
|
||||
|
||||
@@ -413,4 +413,29 @@ class CallerResolverTest {
|
||||
assertThrows(IllegalArgumentException.class, () -> new CallerResolver(id, true, null));
|
||||
assertThrows(IllegalArgumentException.class, () -> new CallerResolver(id, true, " "));
|
||||
}
|
||||
@Test
|
||||
void aWorkerOnAnyLoopbackSourceAddressIsStillAWorkerNotThePrimary() {
|
||||
// fleetd #305: the escalation. ConnectionIdentity used to accept only 127.0.0.1, so a
|
||||
// worker connecting from 127.0.0.2 resolved to no terminal, and this resolver's own
|
||||
// (wider) loopback check then made it the PRIMARY — granting spawn, stop, send and drain.
|
||||
// Measured on the Linux fleet host: binding a source of 127.0.0.2 succeeds there, so the
|
||||
// path is real and not theoretical.
|
||||
CallerResolver r = new CallerResolver(workerIdentity(), false, null);
|
||||
for (String src : new String[]{"127.0.0.1", "127.0.0.2", "127.1.2.3", "::ffff:127.0.0.2"}) {
|
||||
Principal p = r.resolve(src, 55555, null);
|
||||
assertEquals(Role.WORKER, p.role(), "a worker must stay a worker from source " + src);
|
||||
assertEquals("term_a", p.terminal(), "worker terminal from source " + src);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonWorkerOnAnyLoopbackSourceAddressIsStillThePrimary() {
|
||||
// The other direction of the same fix: widening the identity check must not demote a
|
||||
// legitimate same-host primary that happens to connect from another 127.* address.
|
||||
CallerResolver r = new CallerResolver(nonWorkerIdentity(), false, null);
|
||||
for (String src : new String[]{"127.0.0.1", "127.0.0.2", "::ffff:127.0.0.1"}) {
|
||||
assertEquals(Role.PRIMARY, r.resolve(src, 55555, null).role(), "source " + src);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -386,6 +386,54 @@ class ConfigRefTest {
|
||||
assertEquals("rate limit exceeded", ref.get().profiles().get("sonnet").exhaustedPattern());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #201 Unit 5: errorPattern is compiled once into Fleetd.main's backend-error pattern
|
||||
* map at startup (see BackendErrorPatternLookup), the same way exhaustedPattern is above — a
|
||||
* reload never re-reads it either, so a changed value must be reported deferred.
|
||||
*/
|
||||
@Test
|
||||
void changingAProfilesErrorPatternIsReportedAsDeferred(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
errorPattern: "credential outage"
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
errorPattern: "provider 5xx"
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(1, out.deferred().size(), out.deferred().toString());
|
||||
assertTrue(out.deferred().getFirst().contains("sonnet"), out.deferred().toString());
|
||||
assertTrue(out.deferred().getFirst().contains("launch settings"), out.deferred().toString());
|
||||
// The snapshot still carries the new value — a restart is what makes it take effect.
|
||||
assertEquals("provider 5xx", ref.get().profiles().get("sonnet").errorPattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFixedRefHasNoFileAndRefusesToReload() {
|
||||
FleetConfig cfg = new FleetConfig(null, null, null, null, null, null,
|
||||
|
||||
@@ -6,11 +6,18 @@ import dev.ltms.fleet.peer.MemberRole;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.ParameterizedType;
|
||||
import java.lang.reflect.RecordComponent;
|
||||
import java.lang.reflect.Type;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.TreeMap;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
@@ -86,6 +93,169 @@ class FleetConfigTest {
|
||||
"unset means off — today's behaviour, unchanged");
|
||||
}
|
||||
|
||||
// ── fleetd #201 Unit 5: errorPattern ────────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void aProfileWithAnErrorPatternBindsAndHasErrorPatternIsTrue(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("error-pattern.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
ltms-local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
errorPattern: "(?i)\\\\bcredential outage\\\\b"
|
||||
""");
|
||||
|
||||
FleetConfig.Profile w = FleetConfig.load(f).profiles().get("ltms-local");
|
||||
assertTrue(w.hasErrorPattern(), "errorPattern binds and enables fleetd #201 Unit 5 classification");
|
||||
assertEquals("(?i)\\bcredential outage\\b", w.errorPattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aProfileWithNoErrorPatternLeavesItNullAndHasErrorPatternIsFalse(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("no-error-pattern.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
ltms-local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
""");
|
||||
|
||||
FleetConfig.Profile w = FleetConfig.load(f).profiles().get("ltms-local");
|
||||
assertNull(w.errorPattern(), "unset means 'use the built-in fallback' — never off");
|
||||
assertFalse(w.hasErrorPattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBlankErrorPatternNormalizesToNullJustLikeUnset(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("blank-error-pattern.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
ltms-local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
errorPattern: " "
|
||||
""");
|
||||
|
||||
FleetConfig.Profile w = FleetConfig.load(f).profiles().get("ltms-local");
|
||||
assertNull(w.errorPattern());
|
||||
assertFalse(w.hasErrorPattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aProfileWithAMalformedErrorPatternIsRejectedAtLoadNamingTheProfileAndKey(@TempDir Path dir)
|
||||
throws Exception {
|
||||
Path f = dir.resolve("malformed-error-pattern.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
ltms-local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
errorPattern: "(unterminated["
|
||||
""");
|
||||
|
||||
IllegalStateException e = assertThrows(IllegalStateException.class, () -> FleetConfig.load(f));
|
||||
assertTrue(e.getMessage().contains("ltms-local"), "the offending profile is named: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("errorPattern"), "the offending key is named: " + e.getMessage());
|
||||
}
|
||||
|
||||
// ── fleetd #273: exhaustedPattern gets the same load-time validation as its sibling errorPattern ──
|
||||
|
||||
@Test
|
||||
void aProfileWithAMalformedExhaustedPatternIsRejectedAtLoadNamingTheProfileAndKey(@TempDir Path dir)
|
||||
throws Exception {
|
||||
Path f = dir.resolve("malformed-exhausted-pattern.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
ltms-local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
exhaustedPattern: "["
|
||||
""");
|
||||
|
||||
IllegalStateException e = assertThrows(IllegalStateException.class, () -> FleetConfig.load(f));
|
||||
assertTrue(e.getMessage().contains("ltms-local"), "the offending profile is named: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("exhaustedPattern"), "the offending key is named: " + e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMalformedErrorPatternAndAMalformedExhaustedPatternAreBothReportedFromOneLoad(@TempDir Path dir)
|
||||
throws Exception {
|
||||
Path f = dir.resolve("both-malformed.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
errorPattern: "(unterminated["
|
||||
terra:
|
||||
baseUrl: http://gx01.gw:8000
|
||||
exhaustedPattern: "["
|
||||
""");
|
||||
|
||||
IllegalStateException e = assertThrows(IllegalStateException.class, () -> FleetConfig.load(f));
|
||||
assertTrue(e.getMessage().contains("sonnet"), "the errorPattern profile is named: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("errorPattern"), e.getMessage());
|
||||
assertTrue(e.getMessage().contains("terra"), "the exhaustedPattern profile is named: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("exhaustedPattern"), e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void validErrorPatternAndExhaustedPatternBothLoadFine(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("both-valid.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
ltms-local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
errorPattern: "credential outage"
|
||||
exhaustedPattern: "usage limit has been reached"
|
||||
""");
|
||||
|
||||
FleetConfig.Profile w = FleetConfig.load(f).profiles().get("ltms-local");
|
||||
assertEquals("credential outage", w.errorPattern());
|
||||
assertEquals("usage limit has been reached", w.exhaustedPattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBlankExhaustedPatternNormalizesToNullJustLikeUnset(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("blank-exhausted-pattern.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
ltms-local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
exhaustedPattern: " "
|
||||
""");
|
||||
|
||||
FleetConfig.Profile w = FleetConfig.load(f).profiles().get("ltms-local");
|
||||
assertNull(w.exhaustedPattern());
|
||||
assertFalse(w.hasExhaustedPattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aProfileWithNoExhaustedPatternLoadsFine(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("no-exhausted-pattern.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
ltms-local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
""");
|
||||
|
||||
FleetConfig.Profile w = FleetConfig.load(f).profiles().get("ltms-local");
|
||||
assertNull(w.exhaustedPattern());
|
||||
assertFalse(w.hasExhaustedPattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
void withProfileCarriesErrorPatternThrough(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("with-profile-error-pattern.yaml");
|
||||
Files.writeString(f, """
|
||||
profiles:
|
||||
ltms-local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
errorPattern: "credential outage"
|
||||
""");
|
||||
|
||||
// The compact constructor defaults `profile` to the map key via withProfile(name) — see
|
||||
// loadsFullConfig above. errorPattern must survive that same rebuild.
|
||||
FleetConfig.Profile w = FleetConfig.load(f).profiles().get("ltms-local");
|
||||
assertEquals("ltms-local", w.profile());
|
||||
assertEquals("credential outage", w.errorPattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
void appliesDefaultsForMissingSections(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("minimal.yaml");
|
||||
@@ -1036,6 +1206,35 @@ class FleetConfigTest {
|
||||
assertEquals(LeadMailbox.DEFAULT_PREFETCH, noEnv.prefetchOrDefault());
|
||||
}
|
||||
|
||||
@Test
|
||||
void absentWorktreeGroupLeavesItNull(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("no-worktree-group.yaml");
|
||||
Files.writeString(f, "bind:\n port: 8080\n");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertNull(cfg.worktreeGroup(), "no worktreeGroup: key → null → GitWorktrees.shareWithGroup is a no-op");
|
||||
}
|
||||
|
||||
@Test
|
||||
void worktreeGroupKeyParses(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("worktree-group.yaml");
|
||||
Files.writeString(f, "bind:\n port: 8080\nworktreeGroup: fleet-workers\n");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertEquals("fleet-workers", cfg.worktreeGroup());
|
||||
}
|
||||
|
||||
@Test
|
||||
void absentWorktreeGroupSurvivesTheBackCompatConstructorChain() {
|
||||
// fleetd #185 stage 3: withDefaults() (and every pre-existing call site) must not silently
|
||||
// drop a live worktreeGroup by routing through a back-compat constructor that defaults it
|
||||
// to null.
|
||||
FleetConfig cfg = new FleetConfig(null, null, null, Map.of(), null, null, null, null, null,
|
||||
null, null, null, null, null, null, null, null, null, null, null, "fleet-workers");
|
||||
assertEquals("fleet-workers", cfg.withDefaults().worktreeGroup(),
|
||||
"withDefaults() must carry a configured worktreeGroup through unchanged");
|
||||
}
|
||||
|
||||
@Test
|
||||
void absentPrimaryBlockLeavesPrimaryNull(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("no-primary.yaml");
|
||||
@@ -1258,9 +1457,18 @@ class FleetConfigTest {
|
||||
}
|
||||
|
||||
/**
|
||||
* Every optional knob the example documents must bind under the exact spelling used there.
|
||||
* Keep this list in step with {@code fleetd.example.yaml}: a rename that updates the record
|
||||
* but not the example (or vice versa) fails here instead of silently no-op'ing in production.
|
||||
* A spot check that the knobs listed below bind under the exact spelling the example uses —
|
||||
* it asserts real VALUES arrive in the record, which no name-matching guard can do.
|
||||
*
|
||||
* <p><b>This is NOT a coverage guard, and must not be read as one</b> (fleetd #113). The list
|
||||
* inside it is hand-written, so it only ever covers what someone remembered to add. Coverage
|
||||
* — "is every key the code reads documented, and does every documented key bind?" — comes
|
||||
* from {@link #everyNestedConfigKeyIsDocumentedInTheExample} and
|
||||
* {@link #everyLiveKeyInTheExampleBindsToARecordComponent}, both of which derive their key
|
||||
* set from the record tree and therefore cannot drift.
|
||||
*
|
||||
* <p>Adding a knob here is optional. Leaving one out is not a coverage gap, because the two
|
||||
* derived guards above already fail on an undocumented or unbindable key.
|
||||
*/
|
||||
@Test
|
||||
void everyOptionalKnobDocumentedInTheExampleBinds(@TempDir Path dir) throws Exception {
|
||||
@@ -1284,6 +1492,7 @@ class FleetConfigTest {
|
||||
weight: 0.5
|
||||
maxLoad: 2
|
||||
exhaustedPattern: "usage limit has been reached"
|
||||
errorPattern: "credential outage"
|
||||
placement: weighted
|
||||
lifecycle:
|
||||
idleTtlSeconds: 300
|
||||
@@ -1321,6 +1530,8 @@ class FleetConfigTest {
|
||||
assertEquals(2, w.maxLoad(), "maxLoad binds as an integer");
|
||||
assertTrue(w.hasExhaustedPattern(), "exhaustedPattern binds and enables the CB-578 stage A classification");
|
||||
assertEquals("usage limit has been reached", w.exhaustedPattern());
|
||||
assertTrue(w.hasErrorPattern(), "errorPattern binds and enables fleetd #201 Unit 5 classification");
|
||||
assertEquals("credential outage", w.errorPattern());
|
||||
|
||||
assertEquals("weighted", cfg.placement(), "placement binds at the top level");
|
||||
assertEquals(300, cfg.lifecycle().idleTtlSeconds());
|
||||
@@ -1414,6 +1625,230 @@ class FleetConfigTest {
|
||||
return p.matcher(yaml).find();
|
||||
}
|
||||
|
||||
/**
|
||||
* The nested half of {@link #everyKnownTopLevelKeyIsDocumentedInTheExample} (fleetd #113).
|
||||
*
|
||||
* <p>That guard walks {@link FleetConfig#KNOWN_TOP_LEVEL_KEYS} and anchors its regex at
|
||||
* column 0, so it sees ONLY top-level keys. Every nested key — {@code profiles.<name>.model},
|
||||
* {@code health.paneProbeIntervalSeconds} and a hundred others — is outside its scope, and
|
||||
* neither its name nor its output says so. A green run then reads as "the example documents
|
||||
* the schema" when most of the schema was never looked at.
|
||||
*
|
||||
* <p>This walks the record tree rather than a name list, so a key added to any nested record
|
||||
* is covered the moment it compiles, with no edit here. That is the point: a hand-maintained
|
||||
* second copy of a list always drifts from the thing it mirrors.
|
||||
*
|
||||
* <p><b>Scope, stated on purpose</b> (fleetd #113 criterion 3 — every check reports what it
|
||||
* did and did not look at):
|
||||
* <ul>
|
||||
* <li>It checks each key NAME appears somewhere in the example as a YAML key, live or
|
||||
* commented out. It does NOT check the key sits at the right path.</li>
|
||||
* <li>It does NOT check a documented key is read by anything. {@code paneProbeIntervalSeconds}
|
||||
* is parsed into {@link FleetConfig.Health} and used nowhere, and this guard passes it.
|
||||
* Proving a key is live code needs a call graph, which this is not.</li>
|
||||
* </ul>
|
||||
*/
|
||||
@Test
|
||||
void everyNestedConfigKeyIsDocumentedInTheExample() throws Exception {
|
||||
Path example = Path.of("fleetd.example.yaml");
|
||||
assertTrue(Files.exists(example), "fleetd.example.yaml must ship next to the pom");
|
||||
String text = Files.readString(example);
|
||||
|
||||
Map<String, String> pathByName = configKeyPaths();
|
||||
|
||||
// The denominator. An under-counting walk passes every subset check vacuously, which is
|
||||
// the exact shape fleetd #113 collects — so the walk has to prove it descended at all.
|
||||
// The floor is DERIVED, not a literal: the nested walk must find substantially more keys
|
||||
// than the top-level set the old guard used, or it has not gone below the first level.
|
||||
int topLevel = FleetConfig.KNOWN_TOP_LEVEL_KEYS.size();
|
||||
assertTrue(pathByName.size() > topLevel * 2,
|
||||
"the record walk found " + pathByName.size() + " config key(s) against "
|
||||
+ topLevel + " top-level key(s) — it has stopped descending into the "
|
||||
+ "nested records, so this guard would pass vacuously. Fix the walk "
|
||||
+ "before trusting a green run.");
|
||||
|
||||
List<String> undocumented = pathByName.entrySet().stream()
|
||||
.filter(e -> !keyDocumentedAnywhere(text, e.getKey()))
|
||||
.map(Map.Entry::getValue)
|
||||
.sorted()
|
||||
.toList();
|
||||
|
||||
assertTrue(undocumented.isEmpty(), () -> "checked " + pathByName.size()
|
||||
+ " config key(s) that FleetConfig can bind; " + undocumented.size()
|
||||
+ " appear nowhere in fleetd.example.yaml: " + undocumented
|
||||
+ " — document each one there, commented out if optional. fleetd.yaml is "
|
||||
+ "gitignored, so the example is the only committed description of the schema.");
|
||||
}
|
||||
|
||||
/**
|
||||
* The other direction: a LIVE key in the example that {@link FleetConfig} cannot bind. That is
|
||||
* a key an operator would copy into {@code fleetd.yaml} expecting it to do something, where it
|
||||
* would be silently ignored.
|
||||
*
|
||||
* <p><b>Scope, stated on purpose:</b> only live (uncommented) keys are checked. Most of the
|
||||
* example is commented-out prose, and that prose contains lines like {@code # mode: token}
|
||||
* that are indistinguishable from keys by text alone. Parsing them would produce false
|
||||
* failures, so they are deliberately out of scope — and saying so here is the point, rather
|
||||
* than letting a reader assume the whole file was validated.
|
||||
*/
|
||||
@Test
|
||||
void everyLiveKeyInTheExampleBindsToARecordComponent() throws Exception {
|
||||
Path example = Path.of("fleetd.example.yaml");
|
||||
String text = Files.readString(example);
|
||||
|
||||
List<List<String>> paths = liveKeyPaths(text);
|
||||
assertTrue(paths.size() >= 20,
|
||||
"only " + paths.size() + " live key path(s) were parsed out of the example — the "
|
||||
+ "parser is not seeing the file, so this guard would pass vacuously.");
|
||||
|
||||
List<String> unbindable = paths.stream()
|
||||
.filter(path -> !pathBinds(path))
|
||||
.map(path -> String.join(".", path))
|
||||
.distinct()
|
||||
.sorted()
|
||||
.toList();
|
||||
|
||||
assertTrue(unbindable.isEmpty(), () -> "checked " + paths.size()
|
||||
+ " live key path(s) in fleetd.example.yaml; " + unbindable.size()
|
||||
+ " bind to nothing in FleetConfig: " + unbindable
|
||||
+ " — an operator copying one of these into fleetd.yaml gets silence, not an error.");
|
||||
}
|
||||
|
||||
/**
|
||||
* Every configuration key {@link FleetConfig} can bind, at every depth, as
|
||||
* {@code name -> a dotted path to one place it appears}. Derived from the record components,
|
||||
* so it cannot drift from the code.
|
||||
*/
|
||||
private static Map<String, String> configKeyPaths() {
|
||||
Map<String, String> out = new TreeMap<>();
|
||||
collectConfigKeys(FleetConfig.class, "", new HashSet<>(), out);
|
||||
return out;
|
||||
}
|
||||
|
||||
private static void collectConfigKeys(Class<?> type, String prefix, Set<String> seen,
|
||||
Map<String, String> out) {
|
||||
if (!type.isRecord() || !seen.add(type.getName())) {
|
||||
return;
|
||||
}
|
||||
for (RecordComponent rc : type.getRecordComponents()) {
|
||||
String path = prefix.isEmpty() ? rc.getName() : prefix + "." + rc.getName();
|
||||
out.putIfAbsent(rc.getName(), path);
|
||||
Class<?> nested = rc.getType();
|
||||
if (nested.isRecord()) {
|
||||
collectConfigKeys(nested, path, seen, out);
|
||||
} else if (Map.class.isAssignableFrom(nested) || List.class.isAssignableFrom(nested)) {
|
||||
Class<?> element = elementRecord(rc);
|
||||
if (element != null) {
|
||||
String childPrefix = Map.class.isAssignableFrom(nested)
|
||||
? path + ".<name>" : path + "[]";
|
||||
collectConfigKeys(element, childPrefix, seen, out);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** The record type inside a {@code Map<String, X>} or {@code List<X>} component, else null. */
|
||||
private static Class<?> elementRecord(RecordComponent rc) {
|
||||
if (rc.getGenericType() instanceof ParameterizedType pt) {
|
||||
Type[] args = pt.getActualTypeArguments();
|
||||
if (args.length > 0 && args[args.length - 1] instanceof Class<?> c && c.isRecord()) {
|
||||
return c;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when {@code key} is documented in the example, in either of the two conventions that
|
||||
* file actually uses:
|
||||
* <ol>
|
||||
* <li>as a YAML key at any indentation, live or commented out ({@code key:}); or</li>
|
||||
* <li>in a prose block that describes a section's sub-keys, one per line, as
|
||||
* {@code # key → what it does}.</li>
|
||||
* </ol>
|
||||
*
|
||||
* <p>The second form is not decoration. {@code broker.uri} is documented ONLY that way, on
|
||||
* purpose: writing it out as a copy-pasteable {@code uri: amqp://user:pass@host} invites an
|
||||
* operator to paste a password into a file, which is the very thing {@code uriEnv} exists to
|
||||
* avoid. A guard that demanded the key form would push the file toward doing that. So this
|
||||
* encodes the convention the example really uses rather than imposing a new one.
|
||||
*/
|
||||
private static boolean keyDocumentedAnywhere(String yaml, String key) {
|
||||
String quoted = Pattern.quote(key);
|
||||
Pattern asYamlKey = Pattern.compile("(?m)^\\s*(?:#\\s*)?" + quoted + ":");
|
||||
Pattern asProseEntry = Pattern.compile("(?m)^\\s*#\\s*" + quoted + "\\s+\u2192");
|
||||
return asYamlKey.matcher(yaml).find() || asProseEntry.matcher(yaml).find();
|
||||
}
|
||||
|
||||
/** Every live (uncommented) key in {@code yaml}, as a path from the document root. */
|
||||
private static List<List<String>> liveKeyPaths(String yaml) {
|
||||
Pattern keyLine = Pattern.compile("^(\\s*)([A-Za-z][A-Za-z0-9_]*):(\\s.*)?$");
|
||||
List<String> stack = new ArrayList<>();
|
||||
List<Integer> indents = new ArrayList<>();
|
||||
List<List<String>> paths = new ArrayList<>();
|
||||
for (String line : yaml.split("\n", -1)) {
|
||||
if (line.isBlank() || line.stripLeading().startsWith("#")) {
|
||||
continue;
|
||||
}
|
||||
Matcher m = keyLine.matcher(line);
|
||||
if (!m.matches()) {
|
||||
continue;
|
||||
}
|
||||
int indent = m.group(1).length();
|
||||
while (!indents.isEmpty() && indents.get(indents.size() - 1) >= indent) {
|
||||
indents.remove(indents.size() - 1);
|
||||
stack.remove(stack.size() - 1);
|
||||
}
|
||||
indents.add(indent);
|
||||
stack.add(m.group(2));
|
||||
paths.add(List.copyOf(stack));
|
||||
}
|
||||
return paths;
|
||||
}
|
||||
|
||||
/** True when a dotted YAML path resolves to something {@link FleetConfig} can bind. */
|
||||
private static boolean pathBinds(List<String> path) {
|
||||
Class<?> type = FleetConfig.class;
|
||||
boolean nextSegmentIsAFreeFormName = false;
|
||||
for (int i = 0; i < path.size(); i++) {
|
||||
if (nextSegmentIsAFreeFormName) {
|
||||
nextSegmentIsAFreeFormName = false;
|
||||
continue;
|
||||
}
|
||||
RecordComponent rc = componentNamed(type, path.get(i));
|
||||
if (rc == null) {
|
||||
return false;
|
||||
}
|
||||
Class<?> t = rc.getType();
|
||||
if (t.isRecord()) {
|
||||
type = t;
|
||||
} else if (Map.class.isAssignableFrom(t)) {
|
||||
Class<?> element = elementRecord(rc);
|
||||
if (element == null) {
|
||||
return true; // Map<String,String>: its entries are data, not schema
|
||||
}
|
||||
type = element;
|
||||
nextSegmentIsAFreeFormName = true;
|
||||
} else {
|
||||
// A scalar or a list of scalars: nothing may legitimately nest under it.
|
||||
return i == path.size() - 1;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
private static RecordComponent componentNamed(Class<?> type, String name) {
|
||||
if (type == null || !type.isRecord()) {
|
||||
return null;
|
||||
}
|
||||
for (RecordComponent rc : type.getRecordComponents()) {
|
||||
if (rc.getName().equals(name)) {
|
||||
return rc;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
@Test
|
||||
void placementDefaultsToFixedForExistingConfigs(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("no-placement.yaml");
|
||||
@@ -1659,22 +2094,93 @@ class FleetConfigTest {
|
||||
"deny-list normalizes onto the canonical deny-by-default value");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-633: {@code SSH_AUTH_SOCK} is a decision, never a default — absent, blank, or misspelled,
|
||||
* it stays BLOCKED; only the literal "allow" (any case) passes it through. A typo like "alow"
|
||||
* failing safe here is the whole point of making it a knob.
|
||||
*/
|
||||
@Test
|
||||
void sshAuthSockDefaultsToBlockedAndOnlyExplicitAllowUnblocksIt() {
|
||||
assertTrue(new FleetConfig.MemberCredentials("allow-list", List.of(), List.of()).sshAuthSock().equals("block"),
|
||||
"absent knob blocks SSH_AUTH_SOCK");
|
||||
assertFalse(new FleetConfig.MemberCredentials("allow-list", List.of(), List.of()).sshAuthSockAllowed());
|
||||
assertFalse(new FleetConfig.MemberCredentials(null, null, null, "").sshAuthSockAllowed(),
|
||||
"blank knob blocks SSH_AUTH_SOCK");
|
||||
assertFalse(new FleetConfig.MemberCredentials(null, null, null, "alow").sshAuthSockAllowed(),
|
||||
"a misspelled value fails SAFE, not open");
|
||||
assertTrue(new FleetConfig.MemberCredentials(null, null, null, "ALLOW").sshAuthSockAllowed(),
|
||||
"the literal allow (case-insensitive) unblocks SSH_AUTH_SOCK");
|
||||
void sshAgentEnvAcceptsEveryCompatibleKeyAndValuePair(@TempDir Path dir) throws Exception {
|
||||
String[][] spellings = {
|
||||
{"sshAuthSock", "block", "omit"},
|
||||
{"sshAuthSock", "allow", "inherit"},
|
||||
{"sshAuthSock", "omit", "omit"},
|
||||
{"sshAuthSock", "inherit", "inherit"},
|
||||
{"sshAgentEnv", "block", "omit"},
|
||||
{"sshAgentEnv", "allow", "inherit"},
|
||||
{"sshAgentEnv", "omit", "omit"},
|
||||
{"sshAgentEnv", "inherit", "inherit"}
|
||||
};
|
||||
|
||||
for (int i = 0; i < spellings.length; i++) {
|
||||
Path file = dir.resolve("member-credentials-ssh-agent-" + i + ".yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
port: 8080
|
||||
memberCredentials:
|
||||
policy: allow-list
|
||||
""" + " " + spellings[i][0] + ": " + spellings[i][1] + "\n");
|
||||
|
||||
FleetConfig.MemberCredentials credentials = FleetConfig.load(file).memberCredentials();
|
||||
assertEquals(spellings[i][2], credentials.sshAgentEnv(),
|
||||
spellings[i][0] + ": " + spellings[i][1] + " must normalize correctly");
|
||||
assertEquals("inherit".equals(spellings[i][2]), credentials.sshAgentEnvInherited());
|
||||
}
|
||||
}
|
||||
|
||||
/** Protects the live {@code sshAuthSock: block} allow-list configuration during the rename. */
|
||||
@Test
|
||||
void legacySshAuthSockBlockKeepsLiveAllowListConfigOmitted(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("live-member-credentials.yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
port: 8080
|
||||
memberCredentials:
|
||||
policy: allow-list
|
||||
sshAuthSock: block
|
||||
""");
|
||||
|
||||
FleetConfig.MemberCredentials credentials = FleetConfig.load(file).memberCredentials();
|
||||
|
||||
assertEquals("omit", credentials.sshAgentEnv());
|
||||
assertFalse(credentials.sshAgentEnvInherited(), "the live config must omit SSH_AUTH_SOCK");
|
||||
}
|
||||
|
||||
@Test
|
||||
void sshAgentEnvWinsWhenBothCompatibleKeysArePresent(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("both-ssh-agent-keys.yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
port: 8080
|
||||
memberCredentials:
|
||||
policy: allow-list
|
||||
sshAuthSock: allow
|
||||
sshAgentEnv: omit
|
||||
""");
|
||||
|
||||
FleetConfig.MemberCredentials credentials = FleetConfig.load(file).memberCredentials();
|
||||
|
||||
assertEquals("omit", credentials.sshAgentEnv());
|
||||
assertFalse(credentials.sshAgentEnvInherited());
|
||||
}
|
||||
|
||||
@Test
|
||||
void sshAgentEnvDefaultsToOmitAndUnknownValuesFailClosed(@TempDir Path dir) throws Exception {
|
||||
Path absent = dir.resolve("member-credentials-ssh-agent-absent.yaml");
|
||||
Files.writeString(absent, """
|
||||
bind:
|
||||
port: 8080
|
||||
memberCredentials:
|
||||
policy: allow-list
|
||||
""");
|
||||
assertEquals("omit", FleetConfig.load(absent).memberCredentials().sshAgentEnv());
|
||||
|
||||
Path unknown = dir.resolve("member-credentials-ssh-agent-unknown.yaml");
|
||||
Files.writeString(unknown, """
|
||||
bind:
|
||||
port: 8080
|
||||
memberCredentials:
|
||||
policy: allow-list
|
||||
sshAgentEnv: inhert
|
||||
""");
|
||||
FleetConfig.MemberCredentials credentials = FleetConfig.load(unknown).memberCredentials();
|
||||
assertEquals("omit", credentials.sshAgentEnv());
|
||||
assertFalse(credentials.sshAgentEnvInherited(), "an unknown value must fail closed");
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1752,7 +2258,7 @@ class FleetConfigTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void parityOverlayDefaultsToEnvFilesOnly(@TempDir Path dir) throws Exception {
|
||||
void parityOverlayDefaultsToDotEnvOnly(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("no-overlay.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
@@ -1763,9 +2269,11 @@ class FleetConfigTest {
|
||||
""");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertEquals(List.of(".env", ".envrc"),
|
||||
assertEquals(List.of(".env"),
|
||||
cfg.profiles().get("gx10").parityOverlay(),
|
||||
"the default parity overlay is the env files; settings.local.json is no longer copied by default");
|
||||
"the default parity overlay is .env only (CB-148): .envrc is executable shell that "
|
||||
+ "direnv runs on every cd, so it is no longer copied by default; settings.local.json "
|
||||
+ "is also not copied by default");
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -1786,6 +2294,25 @@ class FleetConfigTest {
|
||||
"an operator's explicit list survives verbatim — the default only changes when unset");
|
||||
}
|
||||
|
||||
@Test
|
||||
void parityOverlayExplicitEnvrcOptInStillWorks(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("explicit-envrc-overlay.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
gx10:
|
||||
baseUrl: http://gx10.gw:8000
|
||||
parityOverlay: [".env", ".envrc"]
|
||||
""");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertEquals(List.of(".env", ".envrc"),
|
||||
cfg.profiles().get("gx10").parityOverlay(),
|
||||
"an operator can still opt into copying .envrc explicitly (CB-148); it is only "
|
||||
+ "dropped from the unset default, not removed as a capability");
|
||||
}
|
||||
|
||||
// ── CB-542: subscription:true must not smuggle an unguarded endpoint via env: ───────────────
|
||||
|
||||
@Test
|
||||
|
||||
@@ -24,7 +24,9 @@ import org.slf4j.LoggerFactory;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledFuture;
|
||||
import java.util.concurrent.ScheduledThreadPoolExecutor;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
@@ -434,4 +436,120 @@ class FleetHealthMonitorTest {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
// --- fleetd #280: a GONE/NEVER_READY guess whose fleet_ask lapses AFTER the first sweep must
|
||||
// still be swept, without ever reaching into a target that has since recovered or left the
|
||||
// roster. See FleetHealthMonitor.recheckTerminalTarget's javadoc for the full reachability chain.
|
||||
|
||||
@Test void terminalTransitionSchedulesExactlyOneDelayedRecheck() {
|
||||
ScheduledThreadPoolExecutor scheduler = new ScheduledThreadPoolExecutor(1);
|
||||
FleetHealthMonitor monitor = monitor(new FakeHerdr(),
|
||||
List.of(member("term_a", MemberSession.State.BUSY, 0, 0)), scheduler, () -> 1, 600,
|
||||
(_, _) -> { });
|
||||
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
|
||||
// Driven via reportTransition directly (not tick()), so the queue holds only the recheck.
|
||||
assertEquals(1, scheduler.getQueue().size());
|
||||
ScheduledFuture<?> scheduled = (ScheduledFuture<?>) scheduler.getQueue().peek();
|
||||
assertTrue(scheduled.getDelay(TimeUnit.SECONDS) > 100,
|
||||
"the delay must clear the worst-case fleet_ask lapse window (up to 115s)");
|
||||
|
||||
// An unchanged tick must not queue a second one (CB-580's fire-once rule extends to this).
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
assertEquals(1, scheduler.getQueue().size());
|
||||
monitor.stop();
|
||||
}
|
||||
|
||||
@Test void recheckIsANoOpOnceTheTargetHasRecovered() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
assertEquals(1, failTarget.calls.size());
|
||||
|
||||
monitor.reportTransition("term_a", HealthState.IDLE); // recovered before the recheck fired
|
||||
monitor.recheckTerminalTarget("term_a", HealthState.GONE);
|
||||
|
||||
assertEquals(1, failTarget.calls.size(), "a recovered target must not be reached into again");
|
||||
monitor.stop();
|
||||
}
|
||||
|
||||
@Test void recheckIsANoOpForATargetItNeverObserved() {
|
||||
// Mirrors "left the roster": tick() prunes states.keySet() to the current roster on release
|
||||
// (see FleetHealthMonitor.tick), so a target this monitor never recorded is the same case.
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
|
||||
monitor.recheckTerminalTarget("term_never_seen", HealthState.GONE);
|
||||
|
||||
assertEquals(0, failTarget.calls.size(), "an untracked/released target must not be reached into");
|
||||
monitor.stop();
|
||||
}
|
||||
|
||||
/**
|
||||
* The scenario from the ticket, end to end, driven through the real {@link MessageService}: a
|
||||
* target's ask is still genuinely open when health first observes GONE (sweep must skip it,
|
||||
* exactly as {@code abandonDoesNotFailAnAsyncTicketWaitingForAnAnswer} pins), the ask then lapses
|
||||
* on its own, an unchanged tick still must not refire, and only the delayed recheck sweeps the
|
||||
* now-lapsed ticket to FAILED.
|
||||
*/
|
||||
@Test void delayedRecheckSweepsATicketWhoseAskLapsedAfterGoneWasFirstObserved() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr().withAgent("worker", "term_a", "pane-term_a", "tab_a")
|
||||
.readText("$ prompt");
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
Injector injector = new Injector(agents);
|
||||
InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||
inbox.own("term_a");
|
||||
MessageService messages = new MessageService(agents, injector, rendezvous, inbox);
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents,
|
||||
() -> List.of(member("term_a", MemberSession.State.READY, 0, 0)), messages,
|
||||
new ScheduledThreadPoolExecutor(1), () -> 1, 60, 600, messages::abandon);
|
||||
|
||||
String ticket = messages.sendAsync("term_a", "task that asks");
|
||||
long deadline = System.currentTimeMillis() + 2000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the async send should have opened its waiter");
|
||||
injector.onStatus("term_a", AgentStatus.IDLE); // deliver the task
|
||||
injector.onStatus("term_a", AgentStatus.WORKING); // the worker picks it up
|
||||
|
||||
// The worker asks, with a short timeout so its own fleet_ask lapses quickly in test time.
|
||||
CompletableFuture<MessageService.AskResult> ask = CompletableFuture.supplyAsync(
|
||||
() -> messages.ask("term_a", "which config?", 200));
|
||||
awaitPhase(messages, ticket, MessageService.Phase.ASKING);
|
||||
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
assertEquals(MessageService.Phase.ASKING, messages.poll(ticket).phase(),
|
||||
"the first sweep must not fail a ticket that is still genuinely being asked");
|
||||
|
||||
// The worker's own fleet_ask now lapses on its own — task.question clears to null.
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT, ask.get(5, TimeUnit.SECONDS).outcome());
|
||||
|
||||
// An unchanged tick still must not refire (CB-580).
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket).phase());
|
||||
|
||||
// The delayed recheck scheduled for the original transition finally sweeps it.
|
||||
monitor.recheckTerminalTarget("term_a", HealthState.GONE);
|
||||
assertEquals(MessageService.Phase.FAILED, messages.poll(ticket).phase());
|
||||
|
||||
monitor.stop();
|
||||
}
|
||||
|
||||
private static MessageService.TaskView awaitPhase(MessageService messages, String ticket,
|
||||
MessageService.Phase phase) throws InterruptedException {
|
||||
long deadline = System.currentTimeMillis() + 2000;
|
||||
MessageService.TaskView view;
|
||||
do {
|
||||
view = messages.poll(ticket);
|
||||
if (view.phase() == phase) {
|
||||
return view;
|
||||
}
|
||||
Thread.sleep(5);
|
||||
} while (System.currentTimeMillis() < deadline);
|
||||
assertEquals(phase, view.phase());
|
||||
return view;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -7,6 +7,7 @@ import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
|
||||
/**
|
||||
@@ -31,26 +32,47 @@ public final class FakeHerdr implements HerdrClient {
|
||||
*/
|
||||
public final List<Call> calls = new CopyOnWriteArrayList<>();
|
||||
private boolean healthy = true;
|
||||
private String pingVersion = "0.8.0";
|
||||
private int pingProtocol = 19;
|
||||
private final List<String> extraWorkspaces = new ArrayList<>();
|
||||
private final List<String> extraAgents = new ArrayList<>();
|
||||
/** workspaceId → extra tabs that {@code tab.list} reports for it (CB-558 lead scans). */
|
||||
private final Map<String, List<String>> extraTabs = new LinkedHashMap<>();
|
||||
private int agentNameTakenFor = 0;
|
||||
private int agentPaneBusyFor = 0;
|
||||
private String agentStartErrorCode = null;
|
||||
private int workerTabPaneCount = 1;
|
||||
private String paneCloseErrorCode = null;
|
||||
private final Map<String, String> paneCloseErrorCodeFor = new ConcurrentHashMap<>();
|
||||
private String tabCloseErrorCode = null;
|
||||
private final Map<String, String> tabCloseErrorCodeFor = new ConcurrentHashMap<>();
|
||||
private String agentSendErrorCode = null;
|
||||
private boolean noPanes = false;
|
||||
private volatile String agentStatus = "idle"; // steady-state agent.get status
|
||||
private volatile String agentType = "claude"; // detected agent kind on agent.get; null = undetected
|
||||
private volatile String readText = "worker transcript tail"; // canned agent.read output
|
||||
private int pinnedStarts = 0; // how many upcoming agent.start calls report a fixed pane
|
||||
private String pinnedStartTerminal;
|
||||
private String pinnedStartPane;
|
||||
private Runnable onAgentStart; // fires the instant agent.start is called — see onAgentStart(Runnable)
|
||||
private volatile int agentGetOkCalls = Integer.MAX_VALUE; // how many agent.get calls succeed first
|
||||
private volatile String agentGetFailCode = null; // error code every agent.get call after that reports
|
||||
|
||||
public FakeHerdr healthy(boolean h) {
|
||||
this.healthy = h;
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Make {@code ping} report this version/protocol instead of the default 0.8.0/19 — CB-185
|
||||
* blocker 2's fixture for a lead and a member daemon running mismatched herdr versions.
|
||||
*/
|
||||
public FakeHerdr pingReports(String version, int protocol) {
|
||||
this.pingVersion = version;
|
||||
this.pingProtocol = protocol;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Reject the first {@code n} {@code agent.start} calls with {@code agent_name_taken}. */
|
||||
public FakeHerdr agentNameTakenTimes(int n) {
|
||||
this.agentNameTakenFor = n;
|
||||
@@ -63,24 +85,92 @@ public final class FakeHerdr implements HerdrClient {
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make every {@code agent.start} call fail with this herdr error code. */
|
||||
public FakeHerdr agentStartFailsWith(String code) {
|
||||
this.agentStartErrorCode = code;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make the worker tab (w9:t2) report this many panes in {@code tab.list} (default 1). */
|
||||
public FakeHerdr withWorkerTabPaneCount(int n) {
|
||||
this.workerTabPaneCount = n;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make {@code pane.close} fail with this herdr error code. */
|
||||
/** Make {@code pane.close} fail with this herdr error code, for every pane. */
|
||||
public FakeHerdr paneCloseFailsWith(String code) {
|
||||
this.paneCloseErrorCode = code;
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Make {@code pane.close} fail with this herdr error code, but only for the given {@code
|
||||
* pane_id} — every other pane's {@code pane.close} still succeeds. Unlike {@link
|
||||
* #paneCloseFailsWith}, which fails every call regardless of which pane it targets, this lets a
|
||||
* test reap/release several sessions at once and make exactly one of them fail to stop, so the
|
||||
* others' teardown can be asserted to proceed normally (fleetd #290).
|
||||
*/
|
||||
public FakeHerdr paneCloseFailsForPane(String paneId, String code) {
|
||||
this.paneCloseErrorCodeFor.put(paneId, code);
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make {@code tab.close} fail with this herdr error code, for every tab. */
|
||||
public FakeHerdr tabCloseFailsWith(String code) {
|
||||
this.tabCloseErrorCode = code;
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Make {@code tab.close} fail with this herdr error code, but only for the given {@code
|
||||
* tab_id} — every other tab's {@code tab.close} still succeeds. The {@code tab.close}
|
||||
* counterpart to {@link #paneCloseFailsForPane} (fleetd #290): lets a test make exactly one
|
||||
* session's tab teardown fail while proving the rest of {@code stop()} — {@code
|
||||
* releaseZdotdir}, and the caller's worktree removal — still runs (fleetd #293). Named "ForTab"
|
||||
* rather than "ForPane" (unlike its sibling) because {@code tab.close} keys on {@code tab_id},
|
||||
* not a pane id.
|
||||
*/
|
||||
public FakeHerdr tabCloseFailsForTab(String tabId, String code) {
|
||||
this.tabCloseErrorCodeFor.put(tabId, code);
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Make {@code pane.list} report no panes at all — models a second herdr daemon (CB-185) that
|
||||
* simply does not host the pane a {@link PaneLocator} is searching for.
|
||||
*/
|
||||
public FakeHerdr withNoPanes() {
|
||||
this.noPanes = true;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Set the {@code agent_status} that {@code agent.get} reports (drives the injector). */
|
||||
public FakeHerdr agentStatus(String status) {
|
||||
this.agentStatus = status;
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Set the detected agent kind ({@code "agent"} field) that {@code agent.get} reports —
|
||||
* {@code null} models herdr not (or no longer) detecting a supported backend in the pane, e.g.
|
||||
* a bare shell prompt (fleetd #176 fix 2 corroboration test).
|
||||
*/
|
||||
public FakeHerdr agentType(String type) {
|
||||
this.agentType = type;
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Make {@code agent.get} succeed normally for its first {@code okCalls} invocations, then fail
|
||||
* every call after that with {@code code} — fleetd #176 fix 1's "backend exited mid-wait"
|
||||
* fixture. {@code okCalls == 0} fails from the very first call.
|
||||
*/
|
||||
public FakeHerdr agentGetFailsWithAfter(int okCalls, String code) {
|
||||
this.agentGetOkCalls = okCalls;
|
||||
this.agentGetFailCode = code;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** The text {@code agent.read} returns (the CB-106 completion scrape). */
|
||||
public FakeHerdr readText(String text) {
|
||||
this.readText = text;
|
||||
@@ -106,6 +196,18 @@ public final class FakeHerdr implements HerdrClient {
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Run {@code hook} synchronously the instant an {@code agent.start} call reaches this fake —
|
||||
* i.e. the instant the peer PROCESS would start against a real herdr daemon. A test uses this
|
||||
* to assert something is already true at that exact point (rather than merely true once
|
||||
* {@code spawn()} returns), e.g. fleetd #149's trust-dialog seed having already been written to
|
||||
* disk before the process herdr would launch ever starts.
|
||||
*/
|
||||
public FakeHerdr onAgentStart(Runnable hook) {
|
||||
this.onAgentStart = hook;
|
||||
return this;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Seed a named agent into {@code agent.list} (e.g. an orphaned worker for CB-117 reaper tests).
|
||||
@@ -156,7 +258,8 @@ public final class FakeHerdr implements HerdrClient {
|
||||
try {
|
||||
return switch (method) {
|
||||
case "ping" -> mapper.readTree(
|
||||
"{\"type\":\"pong\",\"version\":\"0.8.0\",\"protocol\":19}");
|
||||
("{\"type\":\"pong\",\"version\":\"%s\",\"protocol\":%d}")
|
||||
.formatted(pingVersion, pingProtocol));
|
||||
case "workspace.list" -> mapper.readTree(("""
|
||||
{"type":"workspace_list","workspaces":[
|
||||
{"workspace_id":"w1","label":"dev-mgnl","focused":true,"pane_count":7,"agent_status":"unknown"},
|
||||
@@ -185,13 +288,27 @@ public final class FakeHerdr implements HerdrClient {
|
||||
}
|
||||
yield mapper.readTree("{\"type\":\"ok\"}");
|
||||
}
|
||||
case "agent.get" -> mapper.readTree(("""
|
||||
{"type":"agent_info","agent":{"terminal_id":"term_a","agent":"claude",
|
||||
case "agent.get" -> {
|
||||
if (agentGetFailCode != null) {
|
||||
long getCalls = calls.stream().filter(c -> c.method().equals("agent.get")).count();
|
||||
if (getCalls > agentGetOkCalls) {
|
||||
throw new HerdrException(
|
||||
"herdr error [" + agentGetFailCode + "]: agent target not found",
|
||||
agentGetFailCode, null);
|
||||
}
|
||||
}
|
||||
String agentField = agentType == null ? "null" : "\"" + agentType + "\"";
|
||||
yield mapper.readTree(("""
|
||||
{"type":"agent_info","agent":{"terminal_id":"term_a","agent":%s,
|
||||
"agent_status":"%s","workspace_id":"w2","tab_id":"w2:t7","pane_id":"w2:p7"}}""")
|
||||
.formatted(agentStatus));
|
||||
.formatted(agentField, agentStatus));
|
||||
}
|
||||
case "agent.read" -> mapper.readTree(mapper.writeValueAsString(
|
||||
java.util.Map.of("type", "agent_read", "read", java.util.Map.of("text", readText))));
|
||||
case "agent.start" -> {
|
||||
if (onAgentStart != null) {
|
||||
onAgentStart.run();
|
||||
}
|
||||
// Protocol 19: kind and pane_id are required — reject like the real daemon.
|
||||
java.util.Map<?, ?> p = params instanceof java.util.Map<?, ?> m ? m : java.util.Map.of();
|
||||
for (String required : new String[]{"kind", "pane_id"}) {
|
||||
@@ -201,6 +318,10 @@ public final class FakeHerdr implements HerdrClient {
|
||||
+ required + "`", "invalid_request", null);
|
||||
}
|
||||
}
|
||||
if (agentStartErrorCode != null) {
|
||||
throw new HerdrException("herdr error [" + agentStartErrorCode + "]: agent.start failed",
|
||||
agentStartErrorCode, null);
|
||||
}
|
||||
long starts = calls.stream().filter(c -> c.method().equals("agent.start")).count();
|
||||
if (starts <= agentPaneBusyFor) {
|
||||
throw new HerdrException(
|
||||
@@ -264,11 +385,23 @@ public final class FakeHerdr implements HerdrClient {
|
||||
.formatted(workerTabPaneCount,
|
||||
seeded.isEmpty() ? "" : "," + String.join(",", seeded)));
|
||||
}
|
||||
case "tab.close" -> mapper.readTree("{\"type\":\"ok\"}");
|
||||
case "tab.close" -> {
|
||||
Object tabIdParam = params instanceof Map<?, ?> m ? m.get("tab_id") : null;
|
||||
String perTabCode = tabIdParam == null ? null
|
||||
: tabCloseErrorCodeFor.get(String.valueOf(tabIdParam));
|
||||
String code = perTabCode != null ? perTabCode : tabCloseErrorCode;
|
||||
if (code != null) {
|
||||
throw new HerdrException("herdr error [" + code + "]: tab.close failed",
|
||||
code, null);
|
||||
}
|
||||
yield mapper.readTree("{\"type\":\"ok\"}");
|
||||
}
|
||||
case "pane.get" -> mapper.readTree("""
|
||||
{"type":"pane_info","pane":{"pane_id":"w9:pW","workspace_id":"w9",
|
||||
"tab_id":"w9:t2","agent_status":"idle"}}""");
|
||||
case "pane.list" -> mapper.readTree("""
|
||||
case "pane.list" -> noPanes
|
||||
? mapper.readTree("{\"type\":\"pane_list\",\"panes\":[]}")
|
||||
: mapper.readTree("""
|
||||
{"type":"pane_list","panes":[
|
||||
{"pane_id":"w2:p7","terminal_id":"term_a","workspace_id":"w2","tab_id":"w2:t7","agent":"claude"},
|
||||
{"pane_id":"w2:p9","terminal_id":"term_shell","workspace_id":"w2","tab_id":"w2:t8"}]}""");
|
||||
@@ -284,9 +417,13 @@ public final class FakeHerdr implements HerdrClient {
|
||||
"foreground_processes":[]}}""");
|
||||
}
|
||||
case "pane.close" -> {
|
||||
if (paneCloseErrorCode != null) {
|
||||
throw new HerdrException("herdr error [" + paneCloseErrorCode + "]: pane.close failed",
|
||||
paneCloseErrorCode, null);
|
||||
Object paneIdParam = params instanceof Map<?, ?> m ? m.get("pane_id") : null;
|
||||
String perPaneCode = paneIdParam == null ? null
|
||||
: paneCloseErrorCodeFor.get(String.valueOf(paneIdParam));
|
||||
String code = perPaneCode != null ? perPaneCode : paneCloseErrorCode;
|
||||
if (code != null) {
|
||||
throw new HerdrException("herdr error [" + code + "]: pane.close failed",
|
||||
code, null);
|
||||
}
|
||||
yield mapper.readTree("{\"type\":\"ok\"}");
|
||||
}
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.Map;
|
||||
import java.util.OptionalLong;
|
||||
|
||||
/**
|
||||
* Fake {@link ParentResolver} backed by an explicit pid→parent map — lets {@link PaneLocatorTest}
|
||||
* drive {@link PaneLocator}'s ancestry walk (grandchild pids, cycles) without spawning real OS
|
||||
* processes.
|
||||
*/
|
||||
final class FakeParentResolver implements ParentResolver {
|
||||
|
||||
private final Map<Long, Long> parents = new HashMap<>();
|
||||
|
||||
/** {@code pid}'s parent is {@code parentPid}. A pid with no entry here has no known parent. */
|
||||
FakeParentResolver parent(long pid, long parentPid) {
|
||||
parents.put(pid, parentPid);
|
||||
return this;
|
||||
}
|
||||
|
||||
@Override
|
||||
public OptionalLong parentOf(long pid) {
|
||||
Long parent = parents.get(pid);
|
||||
return parent == null ? OptionalLong.empty() : OptionalLong.of(parent);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,31 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertSame;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotSame;
|
||||
|
||||
class HerdrRouterTest {
|
||||
@Test
|
||||
void absentMemberClientSharesControlsForEveryTarget() {
|
||||
FakeHerdr client = new FakeHerdr();
|
||||
HerdrRouter router = new HerdrRouter(client, null, id -> id.equals("lead"));
|
||||
|
||||
assertSame(router.leadAgents(), router.memberAgents());
|
||||
assertSame(router.leadSpaces(), router.memberSpaces());
|
||||
assertSame(router.leadAgents(), router.agentsFor("lead"));
|
||||
assertSame(router.leadAgents(), router.agentsFor("member"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void separateClientsRouteLeadAndMemberTargets() {
|
||||
FakeHerdr lead = new FakeHerdr();
|
||||
FakeHerdr member = new FakeHerdr();
|
||||
HerdrRouter router = new HerdrRouter(lead, member, id -> id.equals("lead"));
|
||||
|
||||
assertNotSame(router.leadAgents(), router.memberAgents());
|
||||
assertNotSame(router.leadSpaces(), router.memberSpaces());
|
||||
assertSame(router.leadAgents(), router.agentsFor("lead"));
|
||||
assertSame(router.memberAgents(), router.agentsFor("member"));
|
||||
}
|
||||
}
|
||||
@@ -1,7 +1,11 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/** Unit tests for PID → pane resolution (the herdr half of connection-based MCP identity). */
|
||||
@@ -24,4 +28,162 @@ class PaneLocatorTest {
|
||||
assertNull(loc.terminalForPid(0));
|
||||
assertNull(loc.terminalForPid(-1));
|
||||
}
|
||||
|
||||
// --- two-daemon fallback (CB-185) -----------------------------------------
|
||||
|
||||
@Test
|
||||
void fallsBackToTheMemberClientWhenTheLeadHasNoMatch() {
|
||||
// The caller's pane lives on the member daemon only (e.g. the caller is a spawned
|
||||
// member) — the lead client reports no panes at all, so the locator must fall back.
|
||||
HerdrClient lead = new FakeHerdr().withNoPanes();
|
||||
HerdrClient member = new FakeHerdr();
|
||||
PaneLocator two = new PaneLocator(lead, member);
|
||||
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID));
|
||||
}
|
||||
|
||||
@Test
|
||||
void searchesTheLeadClientBeforeTheMemberClient() {
|
||||
// The caller's pane lives on the LEAD daemon (e.g. the caller is a peer lead) — with two
|
||||
// daemons, resolving it must not depend on the member client having a matching pane.
|
||||
HerdrClient lead = new FakeHerdr();
|
||||
HerdrClient member = new FakeHerdr().withNoPanes();
|
||||
PaneLocator two = new PaneLocator(lead, member);
|
||||
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullWhenNeitherClientHasTheMatch() {
|
||||
PaneLocator two = new PaneLocator(new FakeHerdr().withNoPanes(), new FakeHerdr().withNoPanes());
|
||||
assertNull(two.terminalForPid(FakeHerdr.WORKER_PID));
|
||||
}
|
||||
|
||||
@Test
|
||||
void collapsesToOneScanWhenLeadAndMemberAreTheSameClient() {
|
||||
// The single-daemon deployment (no memberHerdrSocket configured): the two-arg constructor
|
||||
// must behave exactly like the one-arg constructor, including making only one herdr call.
|
||||
FakeHerdr shared = new FakeHerdr();
|
||||
PaneLocator two = new PaneLocator(shared, shared);
|
||||
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID));
|
||||
long paneListCalls = shared.calls.stream().filter(c -> c.method().equals("pane.list")).count();
|
||||
assertEquals(1, paneListCalls, "same-object lead/member must scan exactly once, not twice");
|
||||
}
|
||||
|
||||
// --- ancestry walk (CB-161: grandchild pids matched no pane, resolving as primary) --------
|
||||
|
||||
@Test
|
||||
void stillResolvesAPidThatIsExactlyThePaneShellPid() {
|
||||
// Regression: a pid with no parent chain at all — no ancestry walk is needed to match it.
|
||||
OnePaneHerdr pane = new OnePaneHerdr("term_x", "pX", 5000, 6000);
|
||||
PaneLocator loc = new PaneLocator(pane, new FakeParentResolver());
|
||||
assertEquals("term_x", loc.terminalForPid(5000));
|
||||
}
|
||||
|
||||
@Test
|
||||
void stillResolvesAPidThatIsExactlyAForegroundPid() {
|
||||
// Regression: same as above, but matching via the foreground-processes list.
|
||||
OnePaneHerdr pane = new OnePaneHerdr("term_x", "pX", 5000, 6000);
|
||||
PaneLocator loc = new PaneLocator(pane, new FakeParentResolver());
|
||||
assertEquals("term_x", loc.terminalForPid(6000));
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesAGrandchildPidTwoLevelsBelowTheShellPid() {
|
||||
// The bug: a helper process a worker spawns (python3, curl, ...) is a grandchild of the
|
||||
// pane's shell — not the shell_pid and not a foreground pid directly. Before the fix,
|
||||
// paneOwnsPid only checked direct pid equality, so this pid matched no pane and the
|
||||
// caller fell through to loopback-trust as the primary.
|
||||
OnePaneHerdr pane = new OnePaneHerdr("term_x", "pX", 5000, 6000);
|
||||
FakeParentResolver parents = new FakeParentResolver()
|
||||
.parent(7002, 7001) // grandchild -> child
|
||||
.parent(7001, 5000); // child -> shell (the pane's shell_pid)
|
||||
PaneLocator loc = new PaneLocator(pane, parents);
|
||||
assertEquals("term_x", loc.terminalForPid(7002));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForAPidWhoseAncestryMatchesNoPane() {
|
||||
// Must not break the other direction: a pid that truly belongs to nothing here (e.g. the
|
||||
// real primary) must still resolve to null. Resolving everything to a worker would demote
|
||||
// the actual lead and refuse every orchestration call.
|
||||
OnePaneHerdr pane = new OnePaneHerdr("term_x", "pX", 5000, 6000);
|
||||
FakeParentResolver parents = new FakeParentResolver()
|
||||
.parent(9002, 9001)
|
||||
.parent(9001, 9000); // chain never reaches 5000 or 6000
|
||||
PaneLocator loc = new PaneLocator(pane, parents);
|
||||
assertNull(loc.terminalForPid(9002));
|
||||
}
|
||||
|
||||
@Test
|
||||
void ancestryWalkTerminatesOnACycleInsteadOfHanging() {
|
||||
// A fake (or corrupted) parent map that cycles must not hang identity resolution, which
|
||||
// runs on every MCP call. The walk must still terminate and correctly resolve to null.
|
||||
OnePaneHerdr pane = new OnePaneHerdr("term_x", "pX", 5000, 6000);
|
||||
FakeParentResolver parents = new FakeParentResolver()
|
||||
.parent(100, 101)
|
||||
.parent(101, 100); // cycle, never reaches the pane's pids
|
||||
PaneLocator loc = new PaneLocator(pane, parents);
|
||||
assertNull(loc.terminalForPid(100));
|
||||
}
|
||||
|
||||
@Test
|
||||
void ancestrySetIsComputedOnceAcrossBothClientsInTheTwoDaemonConstructor() {
|
||||
// CB-185: the two-daemon constructor searches lead then member. The ancestor set is
|
||||
// per-caller, not per-client — it must be walked once and reused, not recomputed for
|
||||
// each client searched.
|
||||
OnePaneHerdr pane = new OnePaneHerdr("term_x", "pX", 5000, 6000);
|
||||
FakeParentResolver parents = new FakeParentResolver()
|
||||
.parent(7002, 7001)
|
||||
.parent(7001, 5000);
|
||||
AtomicInteger calls = new AtomicInteger();
|
||||
ParentResolver counting = pid -> {
|
||||
calls.incrementAndGet();
|
||||
return parents.parentOf(pid);
|
||||
};
|
||||
HerdrClient noPanes = new FakeHerdr().withNoPanes();
|
||||
PaneLocator two = new PaneLocator(noPanes, pane, counting);
|
||||
assertEquals("term_x", two.terminalForPid(7002));
|
||||
assertEquals(3, calls.get(), "ancestry must be walked once (3 lookups: 7002, 7001, 5000), "
|
||||
+ "not re-walked per herdr client");
|
||||
}
|
||||
|
||||
/** Minimal single-pane {@link HerdrClient} fake, purpose-built for the ancestry tests above. */
|
||||
private static final class OnePaneHerdr implements HerdrClient {
|
||||
private final ObjectMapper mapper = new ObjectMapper();
|
||||
private final String terminalId;
|
||||
private final String paneId;
|
||||
private final long shellPid;
|
||||
private final long foregroundPid;
|
||||
|
||||
OnePaneHerdr(String terminalId, String paneId, long shellPid, long foregroundPid) {
|
||||
this.terminalId = terminalId;
|
||||
this.paneId = paneId;
|
||||
this.shellPid = shellPid;
|
||||
this.foregroundPid = foregroundPid;
|
||||
}
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
try {
|
||||
return switch (method) {
|
||||
case "pane.list" -> mapper.readTree(("""
|
||||
{"type":"pane_list","panes":[
|
||||
{"pane_id":"%s","terminal_id":"%s","workspace_id":"w1","tab_id":"w1:t1","agent":"claude"}]}""")
|
||||
.formatted(paneId, terminalId));
|
||||
case "pane.process_info" -> mapper.readTree(("""
|
||||
{"type":"pane_process_info","process_info":{"pane_id":"%s","shell_pid":%d,
|
||||
"foreground_processes":[{"pid":%d,"name":"node","argv0":"claude"}]}}""")
|
||||
.formatted(paneId, shellPid, foregroundPid));
|
||||
default -> throw new HerdrException("OnePaneHerdr has no canned response for " + method);
|
||||
};
|
||||
} catch (HerdrException e) {
|
||||
throw e;
|
||||
} catch (Exception e) {
|
||||
throw new HerdrException("OnePaneHerdr decode failed for " + method, e);
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,272 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.fleet.Fleetd;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* fleetd #201 / #227 Unit 5: proves the WIRED production path — a real {@link CompletionResolver}
|
||||
* pane-scrape classification, through the exact {@code BackendErrorSink} ordering {@code Fleetd.main}
|
||||
* builds (mirrored here line for line since the lambda itself lives inline in {@code Fleetd.main}),
|
||||
* into a real {@link BackendOutagePolicy}, a real {@link CompositePeerLauncher} spawn gate, and a
|
||||
* real {@link ReplyPushLoop} lead nudge — never by calling {@link BackendOutagePolicy#record} or the
|
||||
* sink directly, which is what the finer-grained unit tests in
|
||||
* {@code dev.ltms.fleet.member.CompositePeerLauncherTest} and {@code ReplyPushLoopTest} do for their
|
||||
* own concerns. This class exists to catch a wiring mistake those unit tests cannot see: each of them
|
||||
* hands the collaborator a value it assumes was computed correctly one layer up.
|
||||
*
|
||||
* <p>{@code fleet_list}/{@code fleet_profiles} rendering of the resulting cool-off (criteria 11/12)
|
||||
* is proven separately in {@code dev.ltms.fleet.mcp.FleetMcpTest} — {@code FleetMcp}'s view methods
|
||||
* are package-private to {@code dev.ltms.fleet.mcp} and this class needs {@code CompletionResolver}'s
|
||||
* package-private {@code InFlight}, so the two cannot share one test class. What matters for THIS
|
||||
* test is only that the same {@link BackendOutagePolicy} object the real scrape path populates is the
|
||||
* one {@code FleetMcp} reads from — proven by asserting directly against
|
||||
* {@link BackendOutagePolicy#remainingCoolOffSeconds} after the real classification below.
|
||||
*/
|
||||
class BackendOutageFlowTest {
|
||||
|
||||
private final List<ScheduledExecutorService> schedulers = new ArrayList<>();
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
schedulers.forEach(ScheduledExecutorService::shutdownNow);
|
||||
}
|
||||
|
||||
private static FleetConfig.Profile stubWorker(String profile, String credentialId) {
|
||||
return new FleetConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, null, null,
|
||||
null, null, credentialId, null);
|
||||
}
|
||||
|
||||
/** Minimal recording {@code HerdrClient} for the LEAD pane — mirrors ReplyPushLoopTest's own. */
|
||||
private static final class RecordingLeadClient implements HerdrClient {
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
private final List<Object> prompts = new CopyOnWriteArrayList<>();
|
||||
volatile CountDownLatch sendLatch = new CountDownLatch(1);
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
if ("agent.get".equals(method)) {
|
||||
return MAPPER.createObjectNode().set("agent", MAPPER.createObjectNode()
|
||||
.put("terminal_id", "term_primary").put("agent_status", "idle"));
|
||||
}
|
||||
if ("agent.prompt".equals(method)) {
|
||||
prompts.add(params);
|
||||
sendLatch.countDown();
|
||||
}
|
||||
return MAPPER.createObjectNode();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
|
||||
int sendCount() {
|
||||
return prompts.size();
|
||||
}
|
||||
|
||||
String lastText() {
|
||||
return ((Map<?, ?>) prompts.get(prompts.size() - 1)).get("text").toString();
|
||||
}
|
||||
}
|
||||
|
||||
/** Every collaborator the flow wires together, built exactly once per test. */
|
||||
private final class Flow {
|
||||
final FakeHerdr herdr = new FakeHerdr()
|
||||
.readText("⏺ 503 Service Unavailable: upstream credential rejected\n❯ ");
|
||||
final Map<String, FleetConfig.Profile> profiles = orderedProfiles();
|
||||
final AtomicLong clockNanos = new AtomicLong(0L);
|
||||
final BackendOutagePolicy outagePolicy = new BackendOutagePolicy(clockNanos::get);
|
||||
final CompositePeerLauncher workers;
|
||||
final SessionManager sessions;
|
||||
final MemberSession s1;
|
||||
final MemberSession s2;
|
||||
final PrimaryRegistry registry = new PrimaryRegistry(null);
|
||||
final InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||
final RecordingLeadClient leadClient = new RecordingLeadClient();
|
||||
final ReplyPushLoop pushLoop;
|
||||
final CompletionResolver resolver;
|
||||
final Rendezvous rendezvous = new Rendezvous();
|
||||
|
||||
Flow() {
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr),
|
||||
new WorkspaceControl(herdr), new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
profiles, "terra", _ -> "tok");
|
||||
workers = new CompositePeerLauncher(List.of(adapter), "terra", profiles,
|
||||
PlacementPolicies.weighted(), _ -> 0, null, BackendQuarantine.none(), outagePolicy);
|
||||
sessions = new SessionManager(workers);
|
||||
// Two REAL member sessions on "terra" — the roster lookup the production sink does.
|
||||
s1 = sessions.acquire("terra", null, null, null);
|
||||
s2 = sessions.acquire("terra", null, null, null);
|
||||
registry.recordDelegation(s1.terminalId(), "term_primary");
|
||||
registry.recordDelegation(s2.terminalId(), "term_primary");
|
||||
inbox.own(s1.terminalId());
|
||||
inbox.own(s2.terminalId());
|
||||
|
||||
ScheduledExecutorService scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
schedulers.add(scheduler);
|
||||
pushLoop = new ReplyPushLoop(registry, new AgentControl(leadClient), inbox, scheduler, 3, 50);
|
||||
AtomicReference<ReplyPushLoop> pushLoopRef = new AtomicReference<>(pushLoop);
|
||||
|
||||
// fleetd #248 follow-up: this used to be a 30-line hand-copy of Fleetd.main's
|
||||
// backendErrorSink lambda, with a comment promising it mirrored production "EXACTLY".
|
||||
// That promise is exactly the problem: a copy proves the copy. Editing or deleting the
|
||||
// real sink left this whole flow test green, because it never touched the real sink.
|
||||
// #248 made Fleetd.backendErrorSink public precisely so a cross-package test could
|
||||
// drive the real object, so this now calls it. Every assertion below is about
|
||||
// production code again.
|
||||
BackendErrorSink backendErrorSink =
|
||||
Fleetd.backendErrorSink(sessions, () -> profiles, outagePolicy, pushLoopRef::get);
|
||||
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(),
|
||||
ExhaustionSink.none(), patterns, backendErrorSink);
|
||||
}
|
||||
|
||||
/** Drives one real pane-scrape classification for {@code target} through {@code resolver}. */
|
||||
void classifyBackendErrorOn(String target) {
|
||||
var waiter = rendezvous.open(target);
|
||||
resolver.resolve(target, new CompletionResolver.InFlight(waiter, null));
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"a classified backend error still resolves the blocked send as a failure");
|
||||
}
|
||||
}
|
||||
|
||||
private static Map<String, FleetConfig.Profile> orderedProfiles() {
|
||||
Map<String, FleetConfig.Profile> m = new LinkedHashMap<>();
|
||||
m.put("terra", stubWorker("terra", "shared-openai"));
|
||||
m.put("sol", stubWorker("sol", "shared-openai"));
|
||||
return m;
|
||||
}
|
||||
|
||||
private long agentStartCalls(FakeHerdr herdr) {
|
||||
return herdr.calls.stream().filter(c -> c.method().equals("agent.start")).count();
|
||||
}
|
||||
|
||||
// --- criteria 4/5: two DISTINCT targets start an incident; one target never does ------------
|
||||
|
||||
@Test
|
||||
void aSingleClassifiedBackendErrorOnOneTargetNeverCoolsTheCredential() {
|
||||
Flow flow = new Flow();
|
||||
|
||||
flow.classifyBackendErrorOn(flow.s1.terminalId());
|
||||
|
||||
assertTrue(flow.outagePolicy.remainingCoolOffSeconds("shared-openai").isEmpty(),
|
||||
"one distinct target's classified error must not start a cool-off");
|
||||
assertEquals(0, flow.leadClient.sendCount(), "no incident yet, so no lead nudge");
|
||||
assertDoesNotThrow(() -> flow.workers.spawn(new SpawnRequest("terra", null, null)),
|
||||
"terra must still be spawnable after only one target's error");
|
||||
}
|
||||
|
||||
@Test
|
||||
void twoDistinctTargetsClassifiedThroughTheRealScrapeStartAnIncidentAndCoolTheCredential() throws Exception {
|
||||
Flow flow = new Flow();
|
||||
|
||||
flow.classifyBackendErrorOn(flow.s1.terminalId());
|
||||
assertTrue(flow.outagePolicy.remainingCoolOffSeconds("shared-openai").isEmpty());
|
||||
|
||||
flow.classifyBackendErrorOn(flow.s2.terminalId());
|
||||
|
||||
assertTrue(flow.leadClient.sendLatch.await(3, TimeUnit.SECONDS),
|
||||
"the second distinct target must cross the threshold and nudge the lead");
|
||||
Thread.sleep(150);
|
||||
var remaining = flow.outagePolicy.remainingCoolOffSeconds("shared-openai");
|
||||
assertTrue(remaining.isPresent(), "two distinct targets must start a cool-off");
|
||||
assertEquals(60L, remaining.getAsLong(),
|
||||
"this is the exact fact FleetMcp.capacityView/profiles reads as coolingOffForSeconds "
|
||||
+ "(see FleetMcpTest for the rendering side)");
|
||||
}
|
||||
|
||||
// --- criterion 13: the incident reaches the lead through ReplyPushLoop exactly once ----------
|
||||
|
||||
@Test
|
||||
void theIncidentSendsExactlyOneNudgeToTheLeadNamingBothCredentialAndTargets() throws Exception {
|
||||
Flow flow = new Flow();
|
||||
|
||||
flow.classifyBackendErrorOn(flow.s1.terminalId());
|
||||
flow.classifyBackendErrorOn(flow.s2.terminalId());
|
||||
|
||||
assertTrue(flow.leadClient.sendLatch.await(3, TimeUnit.SECONDS));
|
||||
Thread.sleep(200);
|
||||
assertEquals(1, flow.leadClient.sendCount(), "one incident must send exactly one nudge");
|
||||
String nudge = flow.leadClient.lastText();
|
||||
assertTrue(nudge.contains("shared-openai"), nudge);
|
||||
assertTrue(nudge.contains(flow.s1.terminalId()) && nudge.contains(flow.s2.terminalId()), nudge);
|
||||
assertTrue(nudge.contains("60"), nudge);
|
||||
}
|
||||
|
||||
// --- criterion 6: explicit spawn is refused BEFORE the adapter is ever called -----------------
|
||||
|
||||
@Test
|
||||
void explicitSpawnOnACooledCredentialIsRefusedBeforeReachingTheAdapter() throws Exception {
|
||||
Flow flow = new Flow();
|
||||
flow.classifyBackendErrorOn(flow.s1.terminalId());
|
||||
flow.classifyBackendErrorOn(flow.s2.terminalId());
|
||||
assertTrue(flow.leadClient.sendLatch.await(3, TimeUnit.SECONDS));
|
||||
|
||||
long startsBefore = agentStartCalls(flow.herdr);
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> flow.workers.spawn(new SpawnRequest("terra", null, null)));
|
||||
|
||||
assertTrue(e.getMessage().contains("cooling off"), e.getMessage());
|
||||
assertTrue(e.getMessage().contains("shared-openai"), e.getMessage());
|
||||
assertEquals(startsBefore, agentStartCalls(flow.herdr),
|
||||
"the refusal must happen before the adapter is ever called — no new agent.start");
|
||||
}
|
||||
|
||||
// --- criteria 7/9: automatic placement skips every profile sharing the cooled credential -----
|
||||
|
||||
@Test
|
||||
void automaticPlacementRefusesAndNamesCoolingOffWhenEveryCandidateSharesTheCredential() throws Exception {
|
||||
Flow flow = new Flow();
|
||||
flow.classifyBackendErrorOn(flow.s1.terminalId());
|
||||
flow.classifyBackendErrorOn(flow.s2.terminalId());
|
||||
assertTrue(flow.leadClient.sendLatch.await(3, TimeUnit.SECONDS));
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> flow.workers.spawn(new SpawnRequest(null, null, null)),
|
||||
"terra and sol both share the cooled-off credential — nothing is available");
|
||||
assertTrue(e.getMessage().contains("cooling off"), e.getMessage());
|
||||
assertThrows(PlacementException.class, () -> flow.workers.spawn(new SpawnRequest("sol", null, null)),
|
||||
"sol shares terra's credential, so it must be locked out too");
|
||||
}
|
||||
}
|
||||
@@ -4,8 +4,11 @@ import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.LoggerContext;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.TestTurnTokens;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
@@ -193,6 +196,26 @@ class CompletionResolverTest {
|
||||
assertEquals("complete report", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void suppressesABareEchoWithOnlyTuiChrome() {
|
||||
String injected = "Load the implementer skill. You own fleetd #999. ".repeat(12);
|
||||
String scrape = injected + "\nDev auto - GPT-5.6 Terra OpenAI";
|
||||
|
||||
assertTrue(CompletionResolver.echoesInjectedBrief(scrape, injected));
|
||||
}
|
||||
|
||||
@Test
|
||||
void pinsTheMaximumTuiChromeExcess() {
|
||||
String injected = "a".repeat(CompletionResolver.ECHO_MIN_CHARS);
|
||||
String underMargin = injected + "b".repeat(CompletionResolver.MAX_ECHO_EXCESS_CHARS);
|
||||
String overMargin = injected + "b".repeat(CompletionResolver.MAX_ECHO_EXCESS_CHARS + 1);
|
||||
|
||||
assertTrue(CompletionResolver.echoesInjectedBrief(underMargin, injected),
|
||||
"the configured excess itself remains an echoed brief");
|
||||
assertFalse(CompletionResolver.echoesInjectedBrief(overMargin, injected),
|
||||
"one character beyond the excess must preserve the scrape as a real report");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesSynchronouslyBeforePostTurnContextClearing() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ previous answer\n❯ ");
|
||||
@@ -499,7 +522,7 @@ class CompletionResolverTest {
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> notified.add(target + ": " + reason);
|
||||
ExhaustionSink sink = (target, reason, profile) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
@@ -520,7 +543,7 @@ class CompletionResolverTest {
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> notified.add(target + ": " + reason);
|
||||
ExhaustionSink sink = (target, reason, profile) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
@@ -566,21 +589,413 @@ class CompletionResolverTest {
|
||||
assertEquals("The usage limit has been reached.", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
// --- fleetd#164 (part 2 addendum): narrow BACKEND_ERROR pattern classification ---------
|
||||
|
||||
@Test
|
||||
void classifiesABackendErrorLineAsAFailureInsteadOfACompletedReply() {
|
||||
String block = "⏺ API Error: 400 invalid request body\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "a backend-error scrape still resolves the blocked send");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"a backend rejection is a failure, not a completed reply");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theBackendErrorReasonNamesTheMemberAndCarriesTheMatchedLine() {
|
||||
String block = "⏺ API Error: 400 invalid request body\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
String reason = waiter.getNow(null).text();
|
||||
assertTrue(reason.contains("term_a"), "the failure names the member: " + reason);
|
||||
assertTrue(reason.contains("API Error: 400 invalid request body"),
|
||||
"the failure carries the matched backend-error line: " + reason);
|
||||
}
|
||||
|
||||
@Test
|
||||
void aCaseInsensitiveApiErrorLineIsStillClassifiedAsABackendError() {
|
||||
String block = "⏺ api error: rate limited\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(), "the pattern is case-insensitive");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBackendErrorFailureStillCarriesTheRestOfTheScrape() {
|
||||
// The pattern is a heuristic: a member that forgot fleet_reply while *reporting on* a backend
|
||||
// error matches it too. Failing is still correct, but the report itself must survive — losing
|
||||
// it would be the same information-destroying defect fleetd#164 exists to fix.
|
||||
String block = "\u23fa I looked into the gateway problem.\n"
|
||||
+ "The log line was: API Error: 400 invalid request body\n"
|
||||
+ "The cause is a missing content-type header.\n\u276f ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
String reason = waiter.getNow(null).text();
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind());
|
||||
assertTrue(reason.contains("The cause is a missing content-type header."),
|
||||
"the failure carries the rest of the pane, not only the matched line: " + reason);
|
||||
}
|
||||
|
||||
// --- fleetd#211: raw-scrape fallback classification when there is no usable assistant block ---
|
||||
|
||||
@Test
|
||||
void anExhaustionLineWithNoMarkerAndLeadingChromeIsClassifiedFromTheRawScrapeAndNotifiesTheSink() {
|
||||
// No ⏺ anywhere, and the first visible line is TUI chrome (╭). lastAssistantBlock's boundary
|
||||
// scan starts at the top of the raw screen and breaks immediately, so the trimmed block is "".
|
||||
// The fix: fall back to matching the RAW scrape so this doesn't get lost as an empty scrape.
|
||||
String block = """
|
||||
╭──────────────────────────────────────╮
|
||||
The usage limit has been reached. Try again later.
|
||||
""";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "a raw-scrape match still resolves the blocked send");
|
||||
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, waiter.getNow(null).kind(),
|
||||
"classified from the raw scrape even though the trimmed block was empty");
|
||||
assertEquals(1, notified.size(),
|
||||
"the sink is the whole point of this ticket — it must be notified: " + notified);
|
||||
assertTrue(notified.get(0).startsWith("term_a: "), "the sink is told which target exhausted");
|
||||
assertTrue(notified.get(0).contains("The usage limit has been reached"),
|
||||
"the sink is told the matched reason: " + notified.get(0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBackendErrorLineWithNoMarkerAndLeadingChromeIsClassifiedFromTheRawScrape() {
|
||||
String block = """
|
||||
╭──────────────────────────────────────╮
|
||||
API Error: 400 invalid request body
|
||||
""";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "a raw-scrape backend-error match still resolves the blocked send");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"classified BACKEND_ERROR from the raw scrape even though the trimmed block was empty");
|
||||
assertTrue(waiter.getNow(null).text().contains("API Error: 400 invalid request body"),
|
||||
"the failure carries the matched line: " + waiter.getNow(null).text());
|
||||
assertTrue(waiter.getNow(null).text().contains("--- pane tail ---"),
|
||||
"fleetd#164: the failure must carry the pane, not only the matched line — with an "
|
||||
+ "empty trimmed block the raw scrape is the only copy of what the member said: "
|
||||
+ waiter.getNow(null).text());
|
||||
assertTrue(waiter.getNow(null).text().contains("╭"),
|
||||
"the carried pane is the raw scrape, chrome included: " + waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void anOrdinaryPaneWithANormalAssistantBlockIsUnaffectedByTheRawScrapeFallback() {
|
||||
// Pin: on a pane that already yields a usable block, the fallback branch is never reached —
|
||||
// same outcome, same text, sink not called — even though the raw screen around the marker
|
||||
// would itself match the configured exhausted pattern.
|
||||
String block = "The usage limit has been reached, but this is a leading TUI line above the "
|
||||
+ "marker.\n⏺ complete report\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone());
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind(),
|
||||
"unchanged: a usable assistant block never reaches the raw-scrape fallback");
|
||||
assertEquals("complete report", waiter.getNow(null).text(), "the reply text is unaffected");
|
||||
assertTrue(notified.isEmpty(), "the fallback never runs, so the sink is never called");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aGenuinelyEmptyScrapeStillFailsAsEmptyAndNeverNotifiesTheSink() {
|
||||
// The false-positive pin: no exhaustion or backend-error text anywhere on the pane (just
|
||||
// chrome, no marker) — the raw-scrape fallback must not manufacture a classification, and
|
||||
// the sink must stay untouched.
|
||||
String block = """
|
||||
╭──────────────────────────────────────╮
|
||||
│ > │
|
||||
╰──────────────────────────────────────╯
|
||||
""";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason, profile) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "an empty scrape must still resolve the send, not hang");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"no exhaustion or backend-error text anywhere ⇒ this stays the ordinary empty-scrape failure");
|
||||
assertTrue(waiter.getNow(null).text().toLowerCase().contains("empty"),
|
||||
"the failure still says the scrape was empty: " + waiter.getNow(null).text());
|
||||
assertTrue(notified.isEmpty(), "a genuinely empty pane must never quarantine a credential");
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsOffWhenNoProfileHasAPatternConfigured() {
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [terra])",
|
||||
CompletionResolver.coverage(Set.of("terra"), Set.of()));
|
||||
CompletionResolver.coverage("exhaustedPattern", Set.of("terra"), Set.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsFullWhenEveryProfileHasAPatternConfigured() {
|
||||
assertEquals("full (all profiles configured: [gx10, terra])",
|
||||
CompletionResolver.coverage(Set.of("terra", "gx10"), Set.of("terra", "gx10")));
|
||||
CompletionResolver.coverage("exhaustedPattern", Set.of("terra", "gx10"), Set.of("terra", "gx10")));
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsPartialAndNamesWhichProfilesAreConfigured() {
|
||||
assertEquals("partial (configured: [terra]; not configured: [gx10])",
|
||||
CompletionResolver.coverage(Set.of("terra", "gx10"), Set.of("terra")));
|
||||
CompletionResolver.coverage("exhaustedPattern", Set.of("terra", "gx10"), Set.of("terra")));
|
||||
}
|
||||
|
||||
/**
|
||||
* Found live on 2026-09-03, reading a real boot log rather than a test. {@code coverage} is
|
||||
* shared by two call sites — CB-578's {@code exhaustedPattern} line and fleetd#201 Unit 5's
|
||||
* {@code errorPattern} line — but its "off" branch hard-coded the word {@code exhaustedPattern}.
|
||||
* So a daemon with no {@code errorPattern} anywhere printed "no profile has an exhaustedPattern
|
||||
* configured" directly beneath a line reporting that two profiles DO have one. Both lines were
|
||||
* individually defensible and together they were nonsense, and the message sent an operator to
|
||||
* set the wrong key.
|
||||
*
|
||||
* <p>Every earlier test here passed the exhaustion case only, so none of them could see it. This
|
||||
* one pins that the message names the key the caller actually meant.
|
||||
*/
|
||||
@Test
|
||||
void coverageNamesTheConfigKeyItsCallerMeansRatherThanAlwaysSayingExhaustedPattern() {
|
||||
assertEquals("off (no profile has an errorPattern configured; profiles: [gx10, terra])",
|
||||
CompletionResolver.coverage("errorPattern", Set.of("terra", "gx10"), Set.of()));
|
||||
}
|
||||
|
||||
// --- fleetd#201 Unit 1: target-keyed backend-error pattern + typed sink ----------------------
|
||||
|
||||
@Test
|
||||
void aConfiguredBackendErrorPatternClassifiesAMatchAsAFailureAndNotifiesTheSinkOnce() {
|
||||
String block = "⏺ 503 Service Unavailable: upstream credential rejected\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "a configured-pattern match still resolves the blocked send");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind());
|
||||
assertEquals(1, notified.size(), "the sink fires exactly once for the winning classification");
|
||||
assertEquals("term_a: 503 Service Unavailable: upstream credential rejected", notified.get(0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLosingBackendErrorClassificationNeverNotifiesTheSink() {
|
||||
// The waiter was already resolved (e.g. by the worker's own fleet_reply) before this scrape
|
||||
// landed — resolveFailure loses the race and must return false, so the sink must not fire.
|
||||
String block = "⏺ 503 Service Unavailable: upstream credential rejected\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null);
|
||||
assertTrue(rendezvous.resolveCompletion(waiter, "already replied")); // fleet_reply won first
|
||||
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(notified.isEmpty(), "a classification that loses the race must not fire the sink");
|
||||
assertEquals("already replied", waiter.getNow(null).text(), "the earlier resolution stands untouched");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBackendErrorClassificationThatLosesARealRaceDuringTheScrapeNeverNotifiesTheSink() {
|
||||
// The test above pre-resolves the waiter BEFORE calling resolve(), so it is caught by
|
||||
// resolve()'s own top-of-method isDone() guard before ever reaching the win-gated
|
||||
// resolveFailure call this unit adds — it proves the outcome, not the specific gate. This
|
||||
// test forces the actual race window fleetd#201 must be safe in: a genuine fleet_reply lands
|
||||
// WHILE the herdr scrape for this turn is in flight, i.e. AFTER resolve() has already passed
|
||||
// the top guard (waiter not done yet) and committed to reading the pane, but BEFORE it
|
||||
// reaches the "if (rendezvous.resolveFailure(...))" line below. The side effect is attached
|
||||
// to the one real I/O step resolve() performs between those two points: the herdr
|
||||
// "agent.read" call itself.
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
var waiter = rendezvous.open("term_a");
|
||||
ObjectMapper mapper = new ObjectMapper();
|
||||
HerdrClient racingDuringScrape = new HerdrClient() {
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
if ("agent.read".equals(method)) {
|
||||
// The real fleet_reply that wins the race, landing mid-scrape.
|
||||
assertTrue(rendezvous.resolve("term_a", "a genuine fleet_reply landed first"),
|
||||
"the racing reply must land while the waiter is still open");
|
||||
}
|
||||
try {
|
||||
return mapper.readTree(mapper.writeValueAsString(java.util.Map.of(
|
||||
"type", "agent_read",
|
||||
"read", java.util.Map.of("text",
|
||||
"⏺ 503 Service Unavailable: upstream credential rejected\n❯ "))));
|
||||
} catch (Exception e) {
|
||||
throw new RuntimeException(e);
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
};
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(racingDuringScrape), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
|
||||
|
||||
var turn = new CompletionResolver.InFlight(waiter, null); // not done yet — passes the top guard
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(notified.isEmpty(),
|
||||
"a classification that loses a real race during the scrape must not fire the sink");
|
||||
assertEquals(Rendezvous.Kind.REPLY, waiter.getNow(null).kind(), "the genuine fleet_reply stands");
|
||||
assertEquals("a genuine fleet_reply landed first", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void exhaustionKeepsWinningOverBackendErrorEvenWhenBothPatternsMatchTheSameLine() {
|
||||
// A line matching both a configured exhausted pattern AND a configured backend-error pattern
|
||||
// must classify BACKEND_EXHAUSTED and call only the ExhaustionSink — the ordering fleetd#578
|
||||
// already relies on must not change.
|
||||
String block = "⏺ The usage limit has been reached. Try again later.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup exhausted = target -> Pattern.compile("usage limit has been reached");
|
||||
BackendErrorPatternLookup backendErrors = target -> Pattern.compile("(?i)usage limit");
|
||||
java.util.List<String> exhaustedNotified = new java.util.ArrayList<>();
|
||||
java.util.List<String> backendErrorNotified = new java.util.ArrayList<>();
|
||||
ExhaustionSink exhaustionSink = (target, reason, profile) -> exhaustedNotified.add(target);
|
||||
BackendErrorSink backendErrorSink = (target, matchedLine, reason) -> backendErrorNotified.add(target);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
exhausted, exhaustionSink, backendErrors, backendErrorSink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, waiter.getNow(null).kind(),
|
||||
"exhaustion must keep winning over backend error");
|
||||
assertEquals(1, exhaustedNotified.size(), "the exhaustion sink fires");
|
||||
assertTrue(backendErrorNotified.isEmpty(), "the backend-error sink must never fire for this turn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aConfiguredPatternAlsoClassifiesTheRawScrapeFallbackAndNotifiesTheSink() {
|
||||
// No ⏺ marker and leading TUI chrome ⇒ lastAssistantBlock() yields "", so classification must
|
||||
// fall back to the raw scrape (fleetd#211) — and it must use the configured pattern too.
|
||||
String block = """
|
||||
╭──────────────────────────────────────╮
|
||||
503 Service Unavailable: upstream credential rejected
|
||||
""";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "a raw-scrape configured-pattern match still resolves the blocked send");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"classified from the raw scrape even though the trimmed block was empty");
|
||||
assertEquals(1, notified.size(), "the sink fires exactly once from the raw-scrape path");
|
||||
assertTrue(notified.get(0).contains("503 Service Unavailable"), notified.get(0));
|
||||
}
|
||||
|
||||
// --- fleetd#201 Unit 1: classification inside the fleetd#164 MIN_TURN_NANOS floor -------------
|
||||
|
||||
@Test
|
||||
void aMatchingErrorInsideTheFloorIsTypedAndNotifiesTheSink() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ 503 Service Unavailable: upstream credential rejected\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
long[] clock = {10_000_000_000L};
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink, () -> clock[0]);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null, clock[0]); // delivered "now"
|
||||
clock[0] += CompletionResolver.MIN_TURN_NANOS - 1; // inside the floor
|
||||
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone());
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind());
|
||||
assertEquals(1, notified.size(), "a matching error inside the floor notifies the sink");
|
||||
assertTrue(notified.get(0).contains("503 Service Unavailable"), notified.get(0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonMatchInsideTheFloorStaysGenericAndNeverNotifiesTheSink() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ still starting up\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
long[] clock = {10_000_000_000L};
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink, () -> clock[0]);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null, clock[0]); // delivered "now"
|
||||
clock[0] += CompletionResolver.MIN_TURN_NANOS - 1; // inside the floor
|
||||
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone());
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"still a failure — the floor itself, not the pattern, is why");
|
||||
assertTrue(waiter.getNow(null).text().contains("too fast to be real work"),
|
||||
"a non-match inside the floor stays the generic too-fast reason: " + waiter.getNow(null).text());
|
||||
assertTrue(notified.isEmpty(), "a non-match must never notify the typed sink");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* fleetd #234. Round 3: calls the REAL production factory, {@link ExhaustionSink#forwardingTo},
|
||||
* rather than rebuilding its shape locally — a round-2 version of this test built its own copy of
|
||||
* the forwarder inline, so mutating {@code Fleetd.java}'s actual forwarder back into a lambda left
|
||||
* that copy untouched and the test kept passing while production had regressed. Calling the shared
|
||||
* factory here means the same object under test IS the object {@code Fleetd.java} builds via
|
||||
* {@code ExhaustionSink.forwardingTo(exhaustionSinkRef::get)}.
|
||||
*
|
||||
* <p>Round 4: the 3-arg overload is now {@link ExhaustionSink}'s single abstract method, so {@code
|
||||
* forwardingTo} itself is a one-line lambda and every sink below can safely be one too — there is
|
||||
* no 2-arg overload left for any of them to silently bind to instead. This test still catches a
|
||||
* regression inside {@code forwardingTo}'s body (e.g. one that calls the 2-arg default and drops
|
||||
* the hint that way) because it still goes through the shared factory rather than a rebuilt copy.
|
||||
*/
|
||||
class ExhaustionSinkForwardingHazardTest {
|
||||
|
||||
@Test
|
||||
void forwardingToDeliversTheProfileHintToWhateverSinkTheSupplierCurrentlyReturns() {
|
||||
AtomicReference<String> hintSeenByRealSink = new AtomicReference<>("NEVER CALLED");
|
||||
ExhaustionSink real = (target, reason, profileHint) -> hintSeenByRealSink.set(profileHint);
|
||||
|
||||
// Same shape Fleetd.java uses: a reference that starts at none() and is repointed later.
|
||||
AtomicReference<ExhaustionSink> ref = new AtomicReference<>(ExhaustionSink.none());
|
||||
ExhaustionSink forwarding = ExhaustionSink.forwardingTo(ref::get);
|
||||
ref.set(real);
|
||||
|
||||
forwarding.onExhausted("term_x", "model mismatch", "gx");
|
||||
|
||||
assertEquals("gx", hintSeenByRealSink.get(),
|
||||
"ExhaustionSink.forwardingTo must deliver the profile hint to the sink the supplier "
|
||||
+ "currently returns — the exact object Fleetd.java's forwarder is built from");
|
||||
}
|
||||
}
|
||||
@@ -297,6 +297,68 @@ class InjectorTest {
|
||||
assertTrue(inj.activeTargets().isEmpty(), "the wedged target is reclaimed, not polled forever");
|
||||
}
|
||||
|
||||
/** A listener whose post-turn housekeeping always starts, as SessionManager's does with clearAfterTurn on. */
|
||||
private static final class PostTurnListener implements TurnListener {
|
||||
@Override public void onTurnComplete(String target) { }
|
||||
@Override public boolean hasPostTurnAction(String target) { return true; }
|
||||
@Override public boolean onTurnCompleteWithPostAction(String target) { return true; }
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerThatWedgesInUnknownAwaitingPostTurnPickupIsReleased() {
|
||||
// fleetd #306: the post-turn phase had no way out of a sustained unknown streak, so the
|
||||
// pickup latch stayed set, the target was polled forever, and every later message to it was
|
||||
// blocked by the delivery gate — while the session still looked healthy.
|
||||
Captor cap = new Captor();
|
||||
Injector inj = new Injector(new AgentControl(herdr), new PostTurnListener());
|
||||
inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||
|
||||
inj.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
inj.onStatus(T, AgentStatus.WORKING); // turn starts
|
||||
inj.onStatus(T, AgentStatus.IDLE); // turn completes; housekeeping dispatched
|
||||
for (int i = 0; i < STALL_SAMPLES; i++) inj.onStatus(T, AgentStatus.UNKNOWN); // then wedges
|
||||
|
||||
assertTrue(inj.activeTargets().isEmpty(),
|
||||
"a target wedged awaiting post-turn pickup must be reclaimed, not polled forever");
|
||||
assertEquals(List.of(), cap.failed,
|
||||
"the delegated turn already completed — a stuck /clear must not be reported as a failed turn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerThatWedgesInUnknownAfterPickingUpTheResetIsReleased() {
|
||||
// The sibling latch. postTurnObserved is set when the reset is seen picked up (WORKING) and
|
||||
// is cleared only on a later injectable sample, so a wedge right after pickup sticks too.
|
||||
Captor cap = new Captor();
|
||||
Injector inj = new Injector(new AgentControl(herdr), new PostTurnListener());
|
||||
inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||
|
||||
inj.onStatus(T, AgentStatus.IDLE);
|
||||
inj.onStatus(T, AgentStatus.WORKING);
|
||||
inj.onStatus(T, AgentStatus.IDLE); // turn complete; reset dispatched
|
||||
inj.onStatus(T, AgentStatus.WORKING); // reset picked up -> postTurnObserved
|
||||
for (int i = 0; i < STALL_SAMPLES; i++) inj.onStatus(T, AgentStatus.UNKNOWN);
|
||||
|
||||
assertTrue(inj.activeTargets().isEmpty(), "a wedge after reset pickup must also be reclaimed");
|
||||
assertEquals(List.of(), cap.failed, "still not a turn failure");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBriefUnknownDuringPostTurnHousekeepingDoesNotDropTheLatch() {
|
||||
// The other direction: the escape must not fire on a glitch, or the queued next delegation
|
||||
// would overtake housekeeping that is still running.
|
||||
Injector inj = new Injector(new AgentControl(herdr), new PostTurnListener());
|
||||
inj.enqueue(T, "first", TestTurnTokens.inert(T));
|
||||
inj.enqueue(T, "second", TestTurnTokens.inert(T));
|
||||
|
||||
inj.onStatus(T, AgentStatus.IDLE);
|
||||
inj.onStatus(T, AgentStatus.WORKING);
|
||||
inj.onStatus(T, AgentStatus.IDLE); // first completes; reset dispatched
|
||||
for (int i = 0; i < 10; i++) inj.onStatus(T, AgentStatus.UNKNOWN); // well under the grace
|
||||
|
||||
assertFalse(inj.activeTargets().isEmpty(), "a brief glitch must not release the post-turn latch");
|
||||
assertEquals(List.of("first"), sent(), "the queued delegation must not overtake housekeeping");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aTransientUnknownGlitchNeitherFailsNorBlocksCompletion() {
|
||||
Captor cap = new Captor();
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||
import dev.ltms.fleet.msg.TestTurnTokens;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.TimeoutException;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-185: with a router split across two herdr daemons, {@link StatusPoller} must refine a raw
|
||||
* {@code UNKNOWN} status by reading the pane content from the SAME daemon the status was sampled
|
||||
* from — the lead daemon for a lead target, the member daemon for a member target. Reading the
|
||||
* wrong daemon never finds the pane, classification stays {@code UNKNOWN} forever, and the
|
||||
* status-gated {@link Injector} wedges: a queued message is never delivered.
|
||||
*
|
||||
* <p>This exercises the real production classes ({@code StatusPoller(HerdrRouter, ...)},
|
||||
* {@code Injector(HerdrRouter, ...)}) wired together, not a hand-built object graph — the earlier
|
||||
* three CB-185 bugs all passed exactly that kind of test while the real wiring stayed broken.
|
||||
*/
|
||||
class StatusPollerRoutingTest {
|
||||
|
||||
private static final String LEAD_TARGET = "term_a";
|
||||
|
||||
@Test
|
||||
void refinesALeadTargetFromTheLeadDaemonAndDelivers() throws Exception {
|
||||
// The lead daemon's pane is at a settled idle prompt; the member daemon's pane content is
|
||||
// unclassifiable garbage. A correct refiner reads the LEAD daemon and delivers.
|
||||
FakeHerdr leadHerdr = new FakeHerdr().agentStatus("unknown").readText("⏺ answer\n❯ ");
|
||||
FakeHerdr memberHerdr = new FakeHerdr().agentStatus("unknown")
|
||||
.readText("garbled ansi noise with no prompt");
|
||||
HerdrRouter router = new HerdrRouter(leadHerdr, memberHerdr, LEAD_TARGET::equals);
|
||||
Injector injector = new Injector(router, TurnListener.NOOP, _ -> true, _ -> {
|
||||
});
|
||||
StatusPoller poller = new StatusPoller(router, injector, 10);
|
||||
poller.start();
|
||||
try {
|
||||
CompletableFuture<Void> delivered =
|
||||
injector.enqueue(LEAD_TARGET, "via-poller", TestTurnTokens.inert(LEAD_TARGET));
|
||||
// Must resolve quickly: refining against the WRONG daemon (member) never classifies
|
||||
// out of UNKNOWN, so this would time out under the bug.
|
||||
delivered.get(2, TimeUnit.SECONDS);
|
||||
} finally {
|
||||
poller.stop();
|
||||
}
|
||||
assertTrue(leadHerdr.called("agent.read"), "refine must probe the LEAD daemon's pane content");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLeadTargetNeverDeliversWhenOnlyTheMemberDaemonIsClassifiable() throws Exception {
|
||||
// Inverted control: the member daemon's content WOULD classify to idle, but this is a lead
|
||||
// target — a correct implementation must not use it, so delivery must NOT happen.
|
||||
FakeHerdr leadHerdr = new FakeHerdr().agentStatus("unknown")
|
||||
.readText("garbled ansi noise with no prompt");
|
||||
FakeHerdr memberHerdr = new FakeHerdr().agentStatus("unknown").readText("⏺ answer\n❯ ");
|
||||
HerdrRouter router = new HerdrRouter(leadHerdr, memberHerdr, LEAD_TARGET::equals);
|
||||
Injector injector = new Injector(router, TurnListener.NOOP, _ -> true, _ -> {
|
||||
});
|
||||
StatusPoller poller = new StatusPoller(router, injector, 10);
|
||||
poller.start();
|
||||
try {
|
||||
CompletableFuture<Void> delivered =
|
||||
injector.enqueue(LEAD_TARGET, "via-poller", TestTurnTokens.inert(LEAD_TARGET));
|
||||
assertThrows(TimeoutException.class, () -> delivered.get(500, TimeUnit.MILLISECONDS),
|
||||
"a lead target must never be refined from the member daemon's pane content");
|
||||
} finally {
|
||||
poller.stop();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -85,4 +85,34 @@ class StatusRefinerTest {
|
||||
|
||||
assertEquals(AgentStatus.UNKNOWN, refiner.refine("term_a", AgentStatus.UNKNOWN));
|
||||
}
|
||||
|
||||
// --- refine(target, raw, control) — CB-185 per-call routing ---------------
|
||||
|
||||
@Test
|
||||
void threeArgRefineReadsThroughTheGivenControlNotTheConstructedOne() {
|
||||
// The refiner is CONSTRUCTED with one control (standing in for "the member daemon"), but
|
||||
// a call names a DIFFERENT control (standing in for "the lead daemon") — the read must go
|
||||
// to the one passed to the call, since that is the daemon the raw status came from.
|
||||
FakeHerdr constructedWith = new FakeHerdr().readText("nothing recognizable here");
|
||||
FakeHerdr passedToCall = new FakeHerdr().readText("⏺ answer\n❯ ");
|
||||
StatusRefiner refiner = new StatusRefiner(new AgentControl(constructedWith));
|
||||
|
||||
AgentStatus result = refiner.refine("term_a", AgentStatus.UNKNOWN, new AgentControl(passedToCall));
|
||||
|
||||
assertEquals(AgentStatus.IDLE, result, "must classify from the PASSED control's pane content");
|
||||
assertTrue(passedToCall.called("agent.read"));
|
||||
assertFalse(constructedWith.called("agent.read"),
|
||||
"the control fixed at construction must not be read when a call-site control is given");
|
||||
}
|
||||
|
||||
@Test
|
||||
void twoArgRefineStillReadsTheConstructedControl() {
|
||||
// The legacy 2-arg overload (single-daemon callers) must keep using the constructed
|
||||
// control — this is refine(target, raw, control) called with the field as `control`.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ answer\n❯ ");
|
||||
StatusRefiner refiner = new StatusRefiner(new AgentControl(herdr));
|
||||
|
||||
assertEquals(AgentStatus.IDLE, refiner.refine("term_a", AgentStatus.UNKNOWN));
|
||||
assertTrue(herdr.called("agent.read"));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -20,6 +20,17 @@ class ConnectionIdentityTest {
|
||||
assertEquals("term_a", with(_ -> FakeHerdr.WORKER_PID).callerTerminal("127.0.0.1", 55555));
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesWorkerFromAnyLoopbackSourceAddressNotJust127001() {
|
||||
// fleetd #305. On Linux the whole 127.0.0.0/8 is bound to lo, so a worker can connect with
|
||||
// a source address of 127.0.0.2. If identity resolution skips that address the caller has
|
||||
// no terminal, and a caller with no terminal is the primary under loopback-trust — so this
|
||||
// must resolve the worker, not null.
|
||||
assertEquals("term_a", with(_ -> FakeHerdr.WORKER_PID).callerTerminal("127.0.0.2", 55555));
|
||||
assertEquals("term_a", with(_ -> FakeHerdr.WORKER_PID).callerTerminal("127.1.2.3", 55555));
|
||||
assertEquals("term_a", with(_ -> FakeHerdr.WORKER_PID).callerTerminal("::ffff:127.0.0.2", 55555));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForOffHostCaller() {
|
||||
// A non-loopback peer can't be an on-host worker → treat as primary/unknown.
|
||||
|
||||
@@ -24,8 +24,13 @@ import io.modelcontextprotocol.spec.McpSchema;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
@@ -44,6 +49,9 @@ import static org.junit.jupiter.api.Assertions.*;
|
||||
*/
|
||||
class FleetMcpAuthzTest {
|
||||
|
||||
private static final Path MCP_SOURCE = Path.of("src/main/java/dev/ltms/fleet/mcp/FleetMcp.java");
|
||||
private static final Pattern TOOL_REGISTRATION = Pattern.compile("tool\\(\\\"(fleet_[a-z_]+)\\\"");
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
private final AgentControl agents = new AgentControl(herdr);
|
||||
private Metrics metrics;
|
||||
@@ -179,6 +187,89 @@ class FleetMcpAuthzTest {
|
||||
"no CallerResolver supplied ⇒ authorization not enforced (legacy behaviour)");
|
||||
}
|
||||
|
||||
// --- which action each tool hands the gate (fleetd #272) ------------------------------------
|
||||
|
||||
/**
|
||||
* fleetd #272: {@code fleet_poll{target}} drains a session's reply inbox, so it needs
|
||||
* {@link Authz.Action#DRAIN} -- not the {@link Authz.Action#READ} the handler passed for both
|
||||
* of its branches until this ticket.
|
||||
*
|
||||
* <p>This asserts against {@link FleetMcp#pollAction}, the method the handler itself calls, so
|
||||
* the handler holds no separate copy of the rule that this test could miss. Every other test in
|
||||
* this class checks the policy table (is a worker allowed to DRAIN?) and all of them passed for
|
||||
* the whole time the defect was live -- the table was right, the action fed to it was wrong.
|
||||
*/
|
||||
@Test
|
||||
void pollingByTargetIsADrainAndPollingByTicketIsARead() {
|
||||
assertEquals(Authz.Action.DRAIN, FleetMcp.pollAction("term_b"),
|
||||
"poll by target removes the replies — that is a drain, not an observation");
|
||||
assertEquals(Authz.Action.READ, FleetMcp.pollAction(null),
|
||||
"poll by ticket changes nothing");
|
||||
assertEquals(Authz.Action.READ, FleetMcp.pollAction(" "),
|
||||
"a blank target is an absent target");
|
||||
}
|
||||
|
||||
@Test
|
||||
void everyRegisteredToolHasItsHandlerActionPinned() {
|
||||
Set<String> registered = toolsTheServerRegisters();
|
||||
assertTrue(registered.size() >= 10,
|
||||
"scraped only " + registered.size() + " tool registrations from FleetMcp (" + registered
|
||||
+ "); the server registers eleven, so the tool(\"…\") scrape has stopped matching");
|
||||
registered.forEach(tool -> assertDoesNotThrow(() -> FleetMcp.toolAction(tool, Map.of()),
|
||||
() -> tool + " is registered but has no pinned authorization action"));
|
||||
|
||||
assertEquals(Authz.Action.SEND, FleetMcp.toolAction("fleet_send", Map.of()));
|
||||
assertEquals(Authz.Action.REPLY, FleetMcp.toolAction("fleet_reply", Map.of()));
|
||||
assertEquals(Authz.Action.ASK, FleetMcp.toolAction("fleet_ask", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_status", Map.of()));
|
||||
assertEquals(Authz.Action.DRAIN, FleetMcp.toolAction("fleet_ack", Map.of()));
|
||||
assertEquals(Authz.Action.SPAWN, FleetMcp.toolAction("fleet_spawn", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_list", Map.of()));
|
||||
assertEquals(Authz.Action.STOP, FleetMcp.toolAction("fleet_stop", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_profiles", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_whoami", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_poll", Map.of("ticket", "task")));
|
||||
assertEquals(Authz.Action.DRAIN, FleetMcp.toolAction("fleet_poll", Map.of("target", "term_b")));
|
||||
}
|
||||
|
||||
private static Set<String> toolsTheServerRegisters() {
|
||||
try {
|
||||
Matcher matcher = TOOL_REGISTRATION.matcher(Files.readString(MCP_SOURCE));
|
||||
Set<String> tools = new LinkedHashSet<>();
|
||||
while (matcher.find()) {
|
||||
tools.add(matcher.group(1));
|
||||
}
|
||||
return tools;
|
||||
} catch (Exception e) {
|
||||
throw new AssertionError("could not scrape FleetMcp tool registrations", e);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerMayNotDrainAnotherSessionsInboxByPolling() {
|
||||
FleetMcp m = mcp(true);
|
||||
|
||||
assertNotNull(m.denyFor(WORKER_A, FleetMcp.pollAction("term_b"), "term_b"),
|
||||
"a worker draining a peer's inbox would destroy replies queued for the primary");
|
||||
assertNotNull(m.denyFor(ARCH_DESIGN, FleetMcp.pollAction("term_b"), "term_b"),
|
||||
"an architect has no lifecycle rights either — same gate as fleet_ack");
|
||||
assertNull(m.denyFor(PRIMARY, FleetMcp.pollAction("term_b"), "term_b"),
|
||||
"collecting a held reply is the primary's job");
|
||||
}
|
||||
|
||||
/**
|
||||
* The tightening must not close the branch that legitimately serves non-primary callers: an
|
||||
* architect may {@code fleet_send}, so it owns tickets and must be able to poll them.
|
||||
*/
|
||||
@Test
|
||||
void pollingAnOwnTicketStaysOpenToWorkersAndArchitects() {
|
||||
FleetMcp m = mcp(true);
|
||||
|
||||
assertNull(m.denyFor(WORKER_A, FleetMcp.pollAction(null), null));
|
||||
assertNull(m.denyFor(ARCH_DESIGN, FleetMcp.pollAction(null), null),
|
||||
"an architect delegates with wait:false, so it must be able to poll its ticket");
|
||||
}
|
||||
|
||||
// --- identity reconstruction from the transport context ------------------------------------
|
||||
|
||||
@Test
|
||||
|
||||
@@ -1,10 +1,13 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import dev.ltms.fleet.auth.CallerResolver;
|
||||
import dev.ltms.fleet.auth.MemberRegistry;
|
||||
import dev.ltms.fleet.auth.Principal;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
@@ -17,6 +20,7 @@ import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.WorktreeRequest;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
@@ -27,9 +31,11 @@ import org.junit.jupiter.api.Test;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.EnumSet;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.Function;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
@@ -183,7 +189,7 @@ class FleetMcpTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void unansweredAsyncAskReturnsTheTicketToPending() throws Exception {
|
||||
void unansweredAsyncAskReturnsTheTicketToPendingThenAWorkersLateReplyStillCompletesIt() throws Exception {
|
||||
McpSchema.CallToolResult accepted = FleetMcp.sendAsync(messages, "term_a", "do it", null, Set.of());
|
||||
String ticket = textOf(accepted).substring(textOf(accepted).indexOf("ticket=") + "ticket=".length()).trim();
|
||||
|
||||
@@ -197,8 +203,15 @@ class FleetMcpTest {
|
||||
assertTrue(textOf(ask).contains("no answer"), textOf(ask));
|
||||
assertTrue(textOf(FleetMcp.poll(messages, ticket, null)).startsWith("[pending"));
|
||||
|
||||
// fleetd #307: the worker resumed on its own after the primary never answered, and its real
|
||||
// fleet_reply must complete its OWN async ticket — not strand in the inbox with
|
||||
// fleet_poll{ticket} stuck PENDING forever and later force-failed with a false "session
|
||||
// released before it replied" reason. This used to land in the inbox instead (see the old
|
||||
// assertion this replaced: messages.drainReplies("term_a").getFirst()...) — that was the bug.
|
||||
FleetMcp.reply(messages, "term_a", "finished after timeout");
|
||||
assertEquals("finished after timeout", messages.drainReplies("term_a").getFirst().content());
|
||||
assertEquals("finished after timeout", textOf(FleetMcp.poll(messages, ticket, null)));
|
||||
assertTrue(messages.drainReplies("term_a").isEmpty(),
|
||||
"the reply completed its own ticket directly and never touched the inbox");
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -320,6 +333,26 @@ class FleetMcpTest {
|
||||
assertEquals("orphan", drained.getFirst().content());
|
||||
}
|
||||
|
||||
@Test
|
||||
void replyWithBlankContentIsACleanToolErrorNotAnUncaughtException() {
|
||||
// fleetd #302: MessageService.reply now REJECTS blank content by throwing. fleet_reply's
|
||||
// handler is a bare BiFunction with no try/catch around it, so if this guard only checked
|
||||
// `== null` (as it did), a whitespace-only reply would leave the handler as an uncaught
|
||||
// IllegalArgumentException instead of a tool error the caller can read. Null and whitespace
|
||||
// are the same caller mistake and must get the same answer — the sibling fleet_send guard
|
||||
// has always used isBlank for exactly this reason.
|
||||
for (String blank : new String[] {null, "", " ", "\n\t"}) {
|
||||
McpSchema.CallToolResult res = assertDoesNotThrow(
|
||||
() -> FleetMcp.reply(messages, "term_a", blank),
|
||||
"blank content must be refused as a tool error, never thrown out of the handler");
|
||||
assertEquals(Boolean.TRUE, res.isError(), "blank content is an error result");
|
||||
assertTrue(textOf(res).contains("content is required"),
|
||||
"the error names the missing argument: " + textOf(res));
|
||||
}
|
||||
assertEquals(0, messages.drainReplies("term_a").size(),
|
||||
"a refused reply must not reach the inbox");
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgePollWithTargetDrainsReplies() {
|
||||
// A reply with no open send queues it in the inbox.
|
||||
@@ -497,6 +530,25 @@ class FleetMcpTest {
|
||||
assertTrue(out.contains("\"quarantinedForSeconds\":1800"), out);
|
||||
}
|
||||
|
||||
/** fleetd #201 Unit 5: {@code coolingOff} is a SEPARATE map from {@code quarantined}. */
|
||||
@Test
|
||||
void profilesReportsACoolingOffCredentialInASeparateMap() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
outagePolicy.record("shared-openai", "t1", "API Error: rate limited");
|
||||
outagePolicy.record("shared-openai", "t2", "API Error: rate limited"); // 2nd distinct target starts the incident
|
||||
FleetMcp.OutageSource outage = new FleetMcp.OutageSource(
|
||||
profile -> "ltms-local".equals(profile) ? "shared-openai" : null, outagePolicy);
|
||||
McpSchema.CallToolResult res = FleetMcp.profiles(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), FleetMcp.QuarantineSource.none(), outage);
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"coolingOff\""), out);
|
||||
assertTrue(out.contains("shared-openai"), out);
|
||||
assertTrue(out.contains("\"coolingOffForSeconds\":60"), out);
|
||||
assertFalse(out.contains("\"quarantined\""), "nothing is exhaustion-quarantined here: " + out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsTrackedWorkers() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
@@ -580,6 +632,60 @@ class FleetMcpTest {
|
||||
assertTrue(out.contains("\"reclaimable\":0"), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #284: {@code reclaimable} has exactly one definition, and this pins it over EVERY
|
||||
* {@link MemberSession.State} — so a state added later cannot slip through unconsidered. Both
|
||||
* views in a {@code fleet_list} response call {@link FleetMcp#reclaimable}, so they cannot
|
||||
* drift apart.
|
||||
*
|
||||
* <p>{@code BACKEND_ERROR} and {@code FAILED} are NOT reclaimable on purpose. The ticket asked
|
||||
* for them to be; that half of the ticket was wrong. Their seat is already out of
|
||||
* {@code live}, so it is already in {@code free} — counting it here too would report the same
|
||||
* seat twice.
|
||||
*/
|
||||
@Test
|
||||
void onlyReadyAndDoneSessionsAreReclaimable() {
|
||||
Set<MemberSession.State> expected = EnumSet.of(MemberSession.State.READY, MemberSession.State.DONE);
|
||||
for (MemberSession.State state : MemberSession.State.values()) {
|
||||
MemberSession session = new MemberSession("p1", "term1", "ltms-local", null,
|
||||
"/tmp", null, 0L, 0L, 0, state, null, null);
|
||||
assertEquals(expected.contains(state), FleetMcp.reclaimable(session, null),
|
||||
"state " + state + " must " + (expected.contains(state) ? "" : "not ")
|
||||
+ "count as reclaimable");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #284, the operator-visible half. {@code liveCount} here is the value the real counter
|
||||
* ({@code Fleetd.liveSessionCount}, proven against the actual spawn gate in
|
||||
* {@code FleetdBackendErrorSinkTest}) produces for this roster: 0, because both sessions are
|
||||
* terminal. What this test pins is what {@code fleet_list} says around it — the two seats show
|
||||
* up once, in {@code free}, and are NOT counted a second time as {@code reclaimable}; neither
|
||||
* is any member row; and both dead sessions are still listed so the lead can see why.
|
||||
*/
|
||||
@Test
|
||||
void terminalFailureSessionsFreeTheirSeatWithoutBeingCountedReclaimable() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
MemberSession backendError = sessions.acquire("ltms-local", null, null, null);
|
||||
MemberSession failed = sessions.acquire("ltms-local", null, null, null);
|
||||
assertTrue(sessions.onBackendError(backendError.terminalId(), "backend exited"));
|
||||
sessions.onTurnFailed(failed.terminalId());
|
||||
|
||||
String out = textOf(FleetMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
sessions, null, new FleetMcp.CapacitySource(profile -> 0, profile -> 2,
|
||||
() -> Set.of("ltms-local"), () -> 0), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), Map.of(), ""));
|
||||
|
||||
assertTrue(out.contains("\"free\":2"), "both seats are back: " + out);
|
||||
assertTrue(out.contains("\"reclaimable\":0"),
|
||||
"the freed seats must not be counted a second time as reclaimable: " + out);
|
||||
assertFalse(out.contains("\"reclaimable\":true"),
|
||||
"no member row may claim a seat the profile count says is not held: " + out);
|
||||
assertTrue(out.contains("\"state\":\"backend_error\""), "the dead session stays visible: " + out);
|
||||
assertTrue(out.contains("\"state\":\"failed\""), "the failed session stays visible: " + out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void inertCapacitySourceOmitsCapacityBlock() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
@@ -628,6 +734,123 @@ class FleetMcpTest {
|
||||
assertEquals(2, out.split("\"free\":0", -1).length - 1, out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176 stage 2 (correcting the inert stage 1): two {@code subscription: true} profiles,
|
||||
* {@code opus} and {@code sonnet}, neither setting an explicit {@code credentialId} — the exact
|
||||
* shape measured on the live Mac fleet. This is INTENDED, not a regression: a real Claude usage
|
||||
* limit on the one login behind both profiles really does take out every profile running on it,
|
||||
* the same way {@code credentialId: openai-shared} already lets two OpenAI-backed profiles share
|
||||
* one quarantine (see {@code everyProfileSharingTheQuarantinedCredentialReportsZeroFree} above).
|
||||
* The {@code credentialIdFor} function here is built the same way {@code Fleetd.main} wires it —
|
||||
* {@code profile -> profiles.get(profile).effectiveCredentialId()} — so this proves the actual
|
||||
* config-driven behaviour, not just {@code capacityView}'s arithmetic with a hand-picked string.
|
||||
*/
|
||||
@Test
|
||||
void quarantiningOneSubscriptionProfileZeroesFreeOnTheOtherSharingTheSameAccount() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(
|
||||
"opus", new FleetConfig.Profile("opus", null, "claude-opus-4", null, null, null,
|
||||
"tab", "fleet", "w #{n}", null, null, null, null, null, null, null,
|
||||
null, 3, true, null, null, null),
|
||||
"sonnet", new FleetConfig.Profile("sonnet", null, "claude-sonnet-5", null, null, null,
|
||||
"tab", "fleet", "w #{n}", null, null, null, null, null, null, null,
|
||||
null, 3, true, null, null, null));
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(20));
|
||||
quarantine.quarantine(FleetConfig.Profile.SUBSCRIPTION_CREDENTIAL_ID);
|
||||
FleetMcp.QuarantineSource source = new FleetMcp.QuarantineSource(profile -> {
|
||||
FleetConfig.Profile configured = profiles.get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, quarantine);
|
||||
|
||||
String out = textOf(FleetMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
sessions, null, new FleetMcp.CapacitySource(profile -> 0, profile -> 3,
|
||||
() -> Set.of("opus", "sonnet"), () -> 0), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
source, Map.of(), ""));
|
||||
|
||||
assertEquals(2, out.split("\"free\":0", -1).length - 1,
|
||||
"opus and sonnet share one Claude login with neither setting credentialId, so quarantining "
|
||||
+ "opus's account must also zero sonnet's free — this is intended, not a side effect: "
|
||||
+ out);
|
||||
assertEquals(2, out.split("\"credentialId\":\"" + FleetConfig.Profile.SUBSCRIPTION_CREDENTIAL_ID + "\"", -1)
|
||||
.length - 1, out);
|
||||
}
|
||||
|
||||
/** fleetd #201 Unit 5: cool-off forces {@code free:0} but never adds {@code quarantinedForSeconds}. */
|
||||
@Test
|
||||
void coolingOffProfileReportsZeroFreeButNeverQuarantinedForSeconds() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
outagePolicy.record("shared-openai", "t1", "API Error: rate limited");
|
||||
outagePolicy.record("shared-openai", "t2", "API Error: rate limited");
|
||||
FleetMcp.OutageSource outage = new FleetMcp.OutageSource(
|
||||
profile -> "terra".equals(profile) ? "shared-openai" : null, outagePolicy);
|
||||
|
||||
String out = textOf(FleetMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
sessions, null, new FleetMcp.CapacitySource(profile -> 0, profile -> 2,
|
||||
() -> Set.of("terra"), () -> 0), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), outage, Map.of(), ""));
|
||||
|
||||
assertTrue(out.contains("\"profile\":\"terra\""), out);
|
||||
assertTrue(out.contains("\"free\":0"), out);
|
||||
assertTrue(out.contains("\"credentialId\":\"shared-openai\""), out);
|
||||
assertTrue(out.contains("\"coolingOffForSeconds\":60"), out);
|
||||
assertFalse(out.contains("quarantinedForSeconds"),
|
||||
"exhaustion quarantine was never active — this key must not appear: " + out);
|
||||
}
|
||||
|
||||
/**
|
||||
* The two checks are independent — a profile can be BOTH exhaustion-quarantined AND cooling off
|
||||
* for the same credential at once, and both maps/fields report it simultaneously.
|
||||
*/
|
||||
@Test
|
||||
void bothQuarantineAndCoolingOffCanFireForTheSameProfileAtOnce() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(20));
|
||||
quarantine.quarantine("shared-openai");
|
||||
FleetMcp.QuarantineSource quarantineSource = new FleetMcp.QuarantineSource(
|
||||
profile -> "terra".equals(profile) ? "shared-openai" : null, quarantine);
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
outagePolicy.record("shared-openai", "t1", "API Error: rate limited");
|
||||
outagePolicy.record("shared-openai", "t2", "API Error: rate limited");
|
||||
FleetMcp.OutageSource outage = new FleetMcp.OutageSource(
|
||||
profile -> "terra".equals(profile) ? "shared-openai" : null, outagePolicy);
|
||||
|
||||
String out = textOf(FleetMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
sessions, null, new FleetMcp.CapacitySource(profile -> 0, profile -> 2,
|
||||
() -> Set.of("terra"), () -> 0), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
quarantineSource, outage, Map.of(), ""));
|
||||
|
||||
assertTrue(out.contains("\"free\":0"), out);
|
||||
assertTrue(out.contains("\"quarantinedForSeconds\":1200"), out);
|
||||
assertTrue(out.contains("\"coolingOffForSeconds\":60"), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listAndProfilesAgreeOnWhatIsCoolingOff() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
outagePolicy.record("shared-openai", "t1", "API Error: rate limited");
|
||||
outagePolicy.record("shared-openai", "t2", "API Error: rate limited");
|
||||
FleetMcp.OutageSource outage = new FleetMcp.OutageSource(
|
||||
profile -> "ltms-local".equals(profile) ? "shared-openai" : null, outagePolicy);
|
||||
var workers = workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
|
||||
String listOut = textOf(FleetMcp.listFleet(workers, sessions, null,
|
||||
new FleetMcp.CapacitySource(profile -> 0, profile -> 2, () -> Set.of("ltms-local"), () -> 0),
|
||||
new FleetMcp.HealthCoverageSource(() -> "off"), FleetMcp.QuarantineSource.none(), outage,
|
||||
Map.of(), ""));
|
||||
String profilesOut = textOf(FleetMcp.profiles(workers, FleetMcp.QuarantineSource.none(), outage));
|
||||
|
||||
assertTrue(listOut.contains("\"free\":0"), listOut);
|
||||
assertTrue(listOut.contains("\"credentialId\":\"shared-openai\""), listOut);
|
||||
assertTrue(profilesOut.contains("\"coolingOff\""), profilesOut);
|
||||
assertTrue(profilesOut.contains("shared-openai"), profilesOut);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listAndProfilesAgreeOnWhatIsQuarantined() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
@@ -669,6 +892,137 @@ class FleetMcpTest {
|
||||
assertFalse(out.contains("quarantinedForSeconds"), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #257: this is the exact shape measured on the Mac fleet — {@code maxLoad:3, live:2},
|
||||
* one lead sharing the subscription. The OLD formula ({@code max(0, cap - live - leadSeats)})
|
||||
* reported {@code free:0} here, disagreeing with the real spawn gate (which never read
|
||||
* {@code leadSeats} and would still grant one more spawn — see
|
||||
* {@code freeMatchesWhatTheRealPlacementGateActuallyGrants} below for that proof against the
|
||||
* actual gate). {@code free} must report {@code 1}: {@code leadSeats} is carried as a fact, not
|
||||
* subtracted.
|
||||
*/
|
||||
@Test
|
||||
void leadSeatIsReportedButNeverSubtractedFromFree() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
FleetMcp.LeadSeatSource leadSeats = new FleetMcp.LeadSeatSource(
|
||||
profile -> "sonnet".equals(profile) ? 1 : 0);
|
||||
|
||||
String out = textOf(FleetMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
sessions, null, new FleetMcp.CapacitySource(profile -> 2, profile -> 3,
|
||||
() -> Set.of("sonnet"), () -> 0), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none(), leadSeats, Map.of(), "", null));
|
||||
|
||||
assertTrue(out.contains("\"maxLoad\":3"), "maxLoad itself must be left untouched: " + out);
|
||||
assertTrue(out.contains("\"live\":2"), out);
|
||||
assertTrue(out.contains("\"free\":1"), "free is maxLoad - live only, never minus leadSeats: " + out);
|
||||
assertTrue(out.contains("\"leadSeats\":1"), "leadSeats is still reported, just not subtracted: " + out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #257: the OTHER measurement in the issue — a completely idle fleet with a lead sharing
|
||||
* the subscription. {@code maxLoad:3, live:0} must report {@code free:3}, matching what the
|
||||
* spawn gate (which has no notion of a lead's seat) would actually grant.
|
||||
*/
|
||||
@Test
|
||||
void leadSeatDoesNotLowerFreeOnAnOtherwiseIdleSubscriptionProfile() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
FleetMcp.LeadSeatSource leadSeats = new FleetMcp.LeadSeatSource(
|
||||
profile -> "sonnet".equals(profile) ? 1 : 0);
|
||||
|
||||
String out = textOf(FleetMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
sessions, null, new FleetMcp.CapacitySource(profile -> 0, profile -> 3,
|
||||
() -> Set.of("sonnet"), () -> 0), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none(), leadSeats, Map.of(), "", null));
|
||||
|
||||
assertTrue(out.contains("\"live\":0"), out);
|
||||
assertTrue(out.contains("\"free\":3"), "the real gate never subtracts the lead's seat: " + out);
|
||||
assertTrue(out.contains("\"leadSeats\":1"), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #257 — the defect, driven against the REAL placement gate, not a copy of its
|
||||
* arithmetic. {@code free} must equal exactly how many more spawns
|
||||
* {@link CompositePeerLauncher#spawn} (routed through the same {@link SessionManager} fleet_spawn
|
||||
* itself uses) will grant on this profile right now: this test reads whatever number
|
||||
* {@code fleet_list}'s real {@code listFleet} call reports, then drives that many spawns through
|
||||
* the REAL composite launcher and asserts every one succeeds, and the next one — one past what
|
||||
* fleet_list promised — is refused. A test that instead hand-computed {@code cap - live} and
|
||||
* compared it to {@code free} would pass even if both sides shared the same wrong formula (this
|
||||
* repo has been bitten by exactly that before); this one only passes when fleet_list's number and
|
||||
* the gate's real behaviour actually agree.
|
||||
*
|
||||
* <p>{@code liveCount} here is wired the same way {@code Fleetd.main} wires it in production: one
|
||||
* function, read by both the {@link CompositePeerLauncher}'s {@code maxLoad} gate and
|
||||
* {@code fleet_list}'s {@code CapacitySource}, off the SAME {@link SessionManager#roster()} — so
|
||||
* the two paths cannot silently drift on what "live" means.
|
||||
*/
|
||||
@Test
|
||||
void freeMatchesWhatTheRealPlacementGateActuallyGrants() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
FleetConfig.Profile wcfg = new FleetConfig.Profile(
|
||||
"sonnet", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN", null,
|
||||
"tab", "fleetd-workers", "worker: {profile} #{n}", null,
|
||||
null, null, null, null, null, null, null, 3);
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(wcfg.profile(), wcfg);
|
||||
ClaudeCodeLauncher delegate = new ClaudeCodeLauncher(
|
||||
new AgentControl(h), new WorkspaceControl(h), new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
profiles, wcfg.profile(), k -> "FLEETD_WORKER_TOKEN".equals(k) ? "tok" : null);
|
||||
|
||||
java.util.concurrent.atomic.AtomicReference<SessionManager> smRef =
|
||||
new java.util.concurrent.atomic.AtomicReference<>();
|
||||
Function<String, Integer> liveCount = profile -> (int) smRef.get().roster().stream()
|
||||
.filter(s -> profile.equals(s.profile())).count();
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(delegate), wcfg.profile(), profiles, PlacementPolicies.fixed(), liveCount);
|
||||
SessionManager sm = new SessionManager(composite);
|
||||
smRef.set(sm);
|
||||
|
||||
// Two members already live — the exact shape measured in fleetd #257 (maxLoad:3, live:2).
|
||||
assertFalse(FleetMcp.spawn(sm, "sonnet").isError(), "setup: first live member must spawn cleanly");
|
||||
assertFalse(FleetMcp.spawn(sm, "sonnet").isError(), "setup: second live member must spawn cleanly");
|
||||
|
||||
// A lead session shares this profile's subscription — leadSeats:1, same as the ticket.
|
||||
FleetMcp.LeadSeatSource leadSeats = new FleetMcp.LeadSeatSource(p -> "sonnet".equals(p) ? 1 : 0);
|
||||
String out = textOf(FleetMcp.listFleet(composite, sm, null,
|
||||
new FleetMcp.CapacitySource(liveCount, p -> profiles.get(p).maxLoad(), profiles::keySet, () -> 0),
|
||||
new FleetMcp.HealthCoverageSource(() -> "off"), FleetMcp.QuarantineSource.none(),
|
||||
FleetMcp.OutageSource.none(), leadSeats, Map.of(), "", null));
|
||||
int reportedFree = extractInt(out, "free");
|
||||
|
||||
for (int i = 0; i < reportedFree; i++) {
|
||||
McpSchema.CallToolResult res = FleetMcp.spawn(sm, "sonnet");
|
||||
assertFalse(res.isError(), "fleet_list promised free:" + reportedFree + "; spawn #" + (i + 1)
|
||||
+ " of that many was refused by the real gate: " + textOf(res));
|
||||
}
|
||||
McpSchema.CallToolResult overflow = FleetMcp.spawn(sm, "sonnet");
|
||||
assertTrue(overflow.isError(), "fleet_list reported free:" + reportedFree
|
||||
+ " but the real placement gate granted at least one more spawn than that: " + textOf(overflow));
|
||||
}
|
||||
|
||||
private static int extractInt(String json, String key) {
|
||||
java.util.regex.Matcher m = java.util.regex.Pattern.compile("\"" + key + "\":(-?\\d+)").matcher(json);
|
||||
assertTrue(m.find(), "no \"" + key + "\" field in: " + json);
|
||||
return Integer.parseInt(m.group(1));
|
||||
}
|
||||
|
||||
/** A profile with no lead seats reported must be byte-identical to before this ticket. */
|
||||
@Test
|
||||
void zeroLeadSeatsOmitsTheKeyAndLeavesFreeUnchanged() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
|
||||
String out = textOf(FleetMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
sessions, null, new FleetMcp.CapacitySource(profile -> 0, profile -> 2,
|
||||
() -> Set.of("terra"), () -> 0), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none(), FleetMcp.LeadSeatSource.none(),
|
||||
Map.of(), "", null));
|
||||
|
||||
assertTrue(out.contains("\"free\":2"), out);
|
||||
assertFalse(out.contains("leadSeats"), "no lead shares this profile's credential: " + out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsLeadsAndFlagsTheCallersOwnRow() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
@@ -1024,6 +1378,87 @@ class FleetMcpTest {
|
||||
assertTrue(textOf(res).contains("architect, dev, reviewer"), textOf(res));
|
||||
}
|
||||
|
||||
// ── CB-619 / fleetd #123: a spawn asking for a role its profile has no slot for must be
|
||||
// refused, never silently demoted with the roster still lying about it ───────────────────
|
||||
|
||||
/** Two profiles on one launcher, so a role's pool can name one and exclude the other. */
|
||||
private static ClaudeCodeLauncher architectCapableLauncher(FakeHerdr h) {
|
||||
FleetConfig.Profile opus = new FleetConfig.Profile(
|
||||
"opus", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN", null,
|
||||
"tab", "fleetd-workers", "worker: {profile} #{n}", null, null, null);
|
||||
FleetConfig.Profile sonnet = new FleetConfig.Profile(
|
||||
"sonnet", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN", null,
|
||||
"tab", "fleetd-workers", "worker: {profile} #{n}", null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(h), new WorkspaceControl(h),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("opus", opus, "sonnet", sonnet),
|
||||
"sonnet", _ -> "tok");
|
||||
}
|
||||
|
||||
/** {@code fleet.architects} carries only {@code opus} — {@code sonnet} has no matching slot. */
|
||||
private static MemberRegistry architectRegistry() {
|
||||
return new MemberRegistry(new FleetConfig.Fleet(Map.of(),
|
||||
Map.of("opus", new FleetConfig.Slot("opus")), Map.of(), Map.of(), null));
|
||||
}
|
||||
|
||||
/**
|
||||
* The literal defect (fleetd #123): {@code role=architect, profile=sonnet}, where
|
||||
* {@code fleet.architects} carries only {@code opus}. Drives the real path —
|
||||
* {@link FleetMcp#spawn} calls {@link SessionManager#acquire}, which must refuse before ever
|
||||
* reaching the real {@link ClaudeCodeLauncher} — never {@link MemberRegistry#bind} called
|
||||
* directly, which would walk around the gate under test.
|
||||
*/
|
||||
@Test
|
||||
void spawnRefusesAnArchitectWithNoMatchingSlotAndNeverTouchesTheLauncher() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(architectCapableLauncher(h));
|
||||
sessions.setMemberLifecycle(architectRegistry());
|
||||
|
||||
McpSchema.CallToolResult res = FleetMcp.spawn(sessions, "sonnet", "architect",
|
||||
null, null, null, null, null, null);
|
||||
|
||||
assertEquals(Boolean.TRUE, res.isError(), textOf(res));
|
||||
String msg = textOf(res);
|
||||
assertTrue(msg.contains("architect"), "names the role asked for: " + msg);
|
||||
assertTrue(msg.contains("sonnet"), "names the profile: " + msg);
|
||||
assertTrue(msg.contains("opus"), "names the pool that does carry the role: " + msg);
|
||||
assertTrue(sessions.roster().isEmpty(), "a refused spawn must register no session");
|
||||
assertTrue(h.calls.stream().noneMatch(c -> c.method().equals("agent.start")),
|
||||
"a refused spawn must never reach the launcher — no process should ever start");
|
||||
}
|
||||
|
||||
/**
|
||||
* Positive control / parity check: when the profile DOES carry a slot, the spawn succeeds, and
|
||||
* {@code GET /members} ({@link FleetMcp#listFleet}) and {@code fleet_whoami}
|
||||
* ({@link FleetMcp#whoami}) — resolved through the SAME live {@link MemberRegistry} binding via
|
||||
* a real {@link CallerResolver}, exactly as the daemon resolves a real MCP caller — must never
|
||||
* disagree about this one live member's role.
|
||||
*/
|
||||
@Test
|
||||
void rosterAndWhoamiAgreeOnceTheArchitectSlotBinds() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
// Pin the spawn onto the one pane FakeHerdr's canned pane.process_info maps to WORKER_PID,
|
||||
// so a CallerResolver can resolve THIS session's own terminal, not a fixture double.
|
||||
h.pinNextStarts(1, "term_a", "w2:p7");
|
||||
SessionManager sessions = new SessionManager(architectCapableLauncher(h));
|
||||
MemberRegistry members = architectRegistry();
|
||||
sessions.setMemberLifecycle(members);
|
||||
|
||||
McpSchema.CallToolResult spawnRes = FleetMcp.spawn(sessions, "opus", "architect",
|
||||
null, null, null, null, null, null);
|
||||
assertNotEquals(Boolean.TRUE, spawnRes.isError(), textOf(spawnRes));
|
||||
assertTrue(textOf(spawnRes).contains("\"role\":\"architect\""), textOf(spawnRes));
|
||||
|
||||
String roster = textOf(FleetMcp.listFleet(architectCapableLauncher(h), sessions, Map.of(), ""));
|
||||
assertTrue(roster.contains("\"role\":\"architect\""), "GET /members: " + roster);
|
||||
|
||||
ConnectionIdentity identity = new ConnectionIdentity(new PaneLocator(h), _ -> FakeHerdr.WORKER_PID);
|
||||
CallerResolver resolver = CallerResolver.withLeadsAndMembers(identity, false, null, Map::of, members);
|
||||
Principal caller = resolver.resolve("127.0.0.1", 42, null);
|
||||
assertTrue(caller.isArchitect(), "fleet_whoami's own resolver must agree the terminal is bound");
|
||||
String whoami = textOf(FleetMcp.whoami(caller, sessions));
|
||||
assertTrue(whoami.contains("\"role\":\"architect\""), "fleet_whoami: " + whoami);
|
||||
}
|
||||
|
||||
// ── CB-584: fleet_spawn accepts sessionName/resumeSessionId; roster shows agentSessionId ──
|
||||
|
||||
@Test
|
||||
|
||||
@@ -0,0 +1,121 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #114 (CB-609): the guard that lets {@code docs/MCP-Contract.md} name a tool at all.
|
||||
*
|
||||
* <p>That page was written in July 2026, before any MCP code existed, and then did not follow the
|
||||
* code. By August it named two tools that had never been built, omitted five that shipped, had the
|
||||
* wrong name for nearly every parameter, and still described a caller-identity rule that was a
|
||||
* privilege bug by then. Nothing failed, because nothing checked it — and {@code CLAUDE.md} sends
|
||||
* every session in the fleet to that page.
|
||||
*
|
||||
* <p>The fix was to delete the tool catalogue rather than correct it: a hand-maintained second copy
|
||||
* of the tool surface is the defect, not the particular errors it had accumulated. What survives is
|
||||
* the flows, which are shapes rather than names. But the flows still have to say {@code fleet_send}
|
||||
* somewhere to be readable, and that is exactly the sentence that rots. This test is what makes it
|
||||
* safe to write.
|
||||
*
|
||||
* <p><b>It checks source text, not behaviour.</b> It reads the Markdown and reads {@link FleetMcp}'s
|
||||
* source, and it only catches a name in the doc that the server does not register. It cannot catch a
|
||||
* flow that describes the wrong order, or a parameter name in prose — those are not name-shaped. The
|
||||
* doc's own header carries that caveat for its readers.
|
||||
*/
|
||||
class McpContractDocTest {
|
||||
|
||||
/** Tests run with the module directory as cwd, so the repo-root doc is one level up. */
|
||||
private static final Path DOC = Path.of("../docs/MCP-Contract.md");
|
||||
private static final Path MCP_SOURCE = Path.of("src/main/java/dev/ltms/fleet/mcp/FleetMcp.java");
|
||||
|
||||
private static Set<String> matches(Path file, String regex) throws Exception {
|
||||
Matcher m = Pattern.compile(regex).matcher(Files.readString(file));
|
||||
Set<String> found = new LinkedHashSet<>();
|
||||
while (m.find()) {
|
||||
found.add(m.group(1));
|
||||
}
|
||||
return found;
|
||||
}
|
||||
|
||||
/** Every {@code fleet_*} the doc mentions, in prose or in a diagram. */
|
||||
private static Set<String> toolsNamedInTheDoc() throws Exception {
|
||||
return matches(DOC, "(fleet_[a-z_]+)");
|
||||
}
|
||||
|
||||
/** Every tool {@link FleetMcp} actually registers, read from its {@code tool("…")} calls. */
|
||||
private static Set<String> toolsTheServerRegisters() throws Exception {
|
||||
return matches(MCP_SOURCE, "tool\\(\"(fleet_[a-z_]+)\"");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] every fleet_* tool named in MCP-Contract.md is one the server registers")
|
||||
void theDocNamesNoToolThatDoesNotExist() throws Exception {
|
||||
Set<String> registered = toolsTheServerRegisters();
|
||||
Set<String> named = toolsNamedInTheDoc();
|
||||
|
||||
Set<String> unknown = new LinkedHashSet<>(named);
|
||||
unknown.removeAll(registered);
|
||||
|
||||
assertTrue(unknown.isEmpty(),
|
||||
"docs/MCP-Contract.md names " + unknown + ", which FleetMcp does not register. "
|
||||
+ "Checked " + named.size() + " name(s) in the doc against " + registered.size()
|
||||
+ " registered tool(s): " + registered + ". This is the fleetd #114 defect "
|
||||
+ "recurring — the doc named fleet_read and fleet_cancel for weeks after the "
|
||||
+ "code shipped without them. Either fix the name or drop it from the page; do "
|
||||
+ "NOT weaken this test.");
|
||||
}
|
||||
|
||||
/**
|
||||
* The denominator guard. The check above passes trivially if the doc stops naming any tool at
|
||||
* all — an empty set is a subset of everything. A checker that can silently check nothing is the
|
||||
* fleetd #113 shape, so this pins that the doc really is still describing the flows, and that
|
||||
* the registration scrape really did find the server's tools.
|
||||
*/
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] the doc/server name check is not vacuous — both sides found names")
|
||||
void theCheckActuallyHasSomethingToCheck() throws Exception {
|
||||
Set<String> registered = toolsTheServerRegisters();
|
||||
Set<String> named = toolsNamedInTheDoc();
|
||||
|
||||
assertTrue(registered.size() >= 10,
|
||||
"scraped only " + registered.size() + " tool registrations from FleetMcp (" + registered
|
||||
+ "); the server registers eleven, so the tool(\"…\") scrape has stopped matching "
|
||||
+ "and the check above is now vacuous");
|
||||
assertTrue(named.size() >= 4,
|
||||
"docs/MCP-Contract.md names only " + named.size() + " fleet_* tool(s) (" + named + "). "
|
||||
+ "The flows describe delegation, clarification, detached delivery and the "
|
||||
+ "turn-done fallback, so it should name several. Too few means the page has been "
|
||||
+ "gutted and this test is guarding nothing.");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #114's actual lesson. The catalogue was deleted on purpose; a well-meaning "let me just
|
||||
* document the tools here" restores the exact second copy that drifted for a month.
|
||||
*/
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] MCP-Contract.md still says it is not the tool reference")
|
||||
void theDocStillDisclaimsBeingTheToolReference() throws Exception {
|
||||
String doc = Files.readString(DOC);
|
||||
assertTrue(doc.contains("**What this page is NOT: a tool reference.**"),
|
||||
"docs/MCP-Contract.md must keep saying it is not the tool reference. That sentence is "
|
||||
+ "the fix for fleetd #114: the page carried a hand-maintained tool catalogue that "
|
||||
+ "drifted from the code for a month while CLAUDE.md pointed every session at it.");
|
||||
assertEquals(0, countTables(doc.substring(0, doc.indexOf("## 1. Rendezvous flows"))),
|
||||
"the header of docs/MCP-Contract.md must not grow a tool/parameter table — that is the "
|
||||
+ "second copy fleetd #114 deleted");
|
||||
}
|
||||
|
||||
private static int countTables(String markdown) {
|
||||
return (int) markdown.lines().filter(l -> l.strip().startsWith("|")).count();
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -17,6 +17,7 @@ import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
@@ -346,13 +347,94 @@ class CompositePeerLauncherTest {
|
||||
|
||||
@Test
|
||||
void stopRejectsAnUnownedPaneIdWhenMultipleDaemonsCouldOwnIt() {
|
||||
// CB-185 blocker 1: genuine ambiguity — pane ids are per-daemon counters, so two daemons
|
||||
// can each really hold an agent at "w1:p1". Neither claims ownership through spawnedBy
|
||||
// (empty, as after a restart), so the probe must find BOTH and refuse rather than guess.
|
||||
FakeHerdr first = new FakeHerdr().withAgent("x", "term_x", "w1:p1", "w1:t1");
|
||||
FakeHerdr second = new FakeHerdr().withAgent("y", "term_y", "w1:p1", "w1:t1");
|
||||
PeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(claudeAdapter(new FakeHerdr()), opencodeAdapter(new FakeHerdr())), "claude");
|
||||
List.of(claudeAdapter(first), opencodeAdapter(second)), "claude");
|
||||
|
||||
IllegalArgumentException error = assertThrows(IllegalArgumentException.class,
|
||||
() -> composite.stop("w1:p1"));
|
||||
|
||||
assertEquals("ambiguous paneId 'w1:p1': no owning herdr daemon was recorded", error.getMessage());
|
||||
assertEquals("ambiguous paneId 'w1:p1': 2 configured herdr daemons report this pane — "
|
||||
+ "no way to tell which one the caller means", error.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopOnAPaneNoConfiguredDaemonKnowsIsTreatedAsAlreadyStopped() {
|
||||
// CB-185 blocker 1, the zero-owner branch: spawnedBy is empty (as after a restart) and
|
||||
// neither daemon's agent.list mentions this pane at all — it is already gone. A retried
|
||||
// stop() on an already-gone pane must succeed quietly, not refuse forever.
|
||||
FakeHerdr first = new FakeHerdr();
|
||||
FakeHerdr second = new FakeHerdr();
|
||||
PeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(claudeAdapter(first), opencodeAdapter(second)), "claude");
|
||||
|
||||
assertDoesNotThrow(() -> composite.stop("w1:p1"));
|
||||
|
||||
assertFalse(first.called("pane.close"), "no owner was found, so no delegate is told to close anything");
|
||||
assertFalse(second.called("pane.close"), "no owner was found, so no delegate is told to close anything");
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopWithEmptySpawnedByResolvesTheOwnerThroughAProbeAndSkipsTheOtherDaemon() {
|
||||
// CB-185 blocker 1, the main fix: after a restart spawnedBy is empty for every surviving
|
||||
// member. stop() must still find the one daemon that actually knows the pane and route
|
||||
// only to it — never touching the daemon that never held it.
|
||||
FakeHerdr first = new FakeHerdr().withAgent("x", "term_x", "w1:p1", "w1:t1");
|
||||
FakeHerdr second = new FakeHerdr();
|
||||
PeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(claudeAdapter(first), opencodeAdapter(second)), "claude");
|
||||
|
||||
composite.stop("w1:p1");
|
||||
|
||||
assertTrue(first.calls.stream().anyMatch(c -> c.method().equals("pane.close")
|
||||
&& "w1:p1".equals(((Map<?, ?>) c.params()).get("pane_id"))),
|
||||
"the daemon that actually knows the pane closes it");
|
||||
assertFalse(second.called("pane.close"), "the daemon that never held the pane is never touched");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aProbeSurvivesOneUnreachableDaemonAndStillFindsTheOwnerOnTheOtherOne() {
|
||||
// CB-185 blocker 1 (lead review): a daemon that is DOWN while we probe must not abort the
|
||||
// whole probe — the pane the OPERATOR actually wants stopped can live on a different,
|
||||
// healthy daemon, and that pane must not become un-stoppable because a third one is down.
|
||||
FakeHerdr down = new FakeHerdr().healthy(false);
|
||||
FakeHerdr owner = new FakeHerdr().withAgent("x", "term_x", "w1:p1", "w1:t1");
|
||||
PeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(claudeAdapter(down), opencodeAdapter(owner)), "claude");
|
||||
|
||||
assertDoesNotThrow(() -> composite.stop("w1:p1"),
|
||||
"the unreachable daemon must be skipped, not fail the whole stop");
|
||||
|
||||
assertTrue(owner.calls.stream().anyMatch(c -> c.method().equals("pane.close")
|
||||
&& "w1:p1".equals(((Map<?, ?>) c.params()).get("pane_id"))),
|
||||
"the healthy daemon that actually owns the pane still closes it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aProbedOwnerIsCachedSoARetryAfterAFailedStopNeedsNoSecondProbe() {
|
||||
// CB-185 blocker 1: the probe's whole point is to be cheap on repeat — a failed stop (e.g.
|
||||
// "pane_busy") must not force another agent.list() round trip on every retry.
|
||||
FakeHerdr first = new FakeHerdr().withAgent("x", "term_x", "w1:p1", "w1:t1")
|
||||
.paneCloseFailsWith("pane_busy");
|
||||
FakeHerdr second = new FakeHerdr();
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(claudeAdapter(first), opencodeAdapter(second)), "claude");
|
||||
|
||||
assertThrows(HerdrException.class, () -> composite.stop("w1:p1"));
|
||||
long listCallsAfterFirst = first.calls.stream().filter(c -> c.method().equals("agent.list")).count()
|
||||
+ second.calls.stream().filter(c -> c.method().equals("agent.list")).count();
|
||||
assertTrue(listCallsAfterFirst > 0, "the first stop needed a probe");
|
||||
|
||||
assertThrows(HerdrException.class, () -> composite.stop("w1:p1"),
|
||||
"still failing on the retry, but through the cached owner");
|
||||
long listCallsAfterSecond = first.calls.stream().filter(c -> c.method().equals("agent.list")).count()
|
||||
+ second.calls.stream().filter(c -> c.method().equals("agent.list")).count();
|
||||
assertEquals(listCallsAfterFirst, listCallsAfterSecond,
|
||||
"the retry is served from the cache — no additional agent.list probe");
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -830,4 +912,185 @@ class CompositePeerLauncherTest {
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("claude", null, null)));
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("gemini", null, null)));
|
||||
}
|
||||
|
||||
// ── fleetd #201 Unit 5: repeated backend errors cool a credential off — a SEPARATE, shorter- ──
|
||||
// ── lived mechanism from CB-578 stage B quarantine above, never merged with it ───────────────
|
||||
|
||||
@Test
|
||||
void explicitSpawnOntoACoolingOffProfileIsRefusedNamingTheCredentialAndRemainingTime() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"terra", stubWorker("terra", "shared-openai"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
// Two DISTINCT targets on the same credential — evidenceCount() counts distinct targets,
|
||||
// never raw events, so one target repeating a classified error can never mint an incident.
|
||||
outagePolicy.record("shared-openai", "term-1", "API Error: 500");
|
||||
assertTrue(outagePolicy.record("shared-openai", "term-2", "API Error: 500").isPresent(),
|
||||
"the second distinct target crosses the threshold and starts the cool-off");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(), outagePolicy);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("sol", null, null)));
|
||||
assertTrue(e.getMessage().contains("sol"), "message names the profile: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("shared-openai"), "message names the credential: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("cooling off"), "message says cooling off, not exhausted: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("60"), "message names roughly when it lifts: " + e.getMessage());
|
||||
assertEquals(0, adapter.spawnCount("sol"), "the cooling-off profile is never delegated to");
|
||||
}
|
||||
|
||||
/**
|
||||
* A single error from one target must never cool a credential off — the threshold is on
|
||||
* distinct targets. This proves the spawn gate really reads {@link BackendOutagePolicy}'s
|
||||
* result rather than reacting to any recorded event.
|
||||
*/
|
||||
@Test
|
||||
void explicitSpawnIsNotRefusedAfterOnlyOneBackendErrorOnOneTarget() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"terra", stubWorker("terra", "shared-openai"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
assertTrue(outagePolicy.record("shared-openai", "term-1", "API Error: 500").isEmpty(),
|
||||
"one distinct target is below THRESHOLD=2 — no incident yet");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(), outagePolicy);
|
||||
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("sol", null, null)),
|
||||
"one target's error never cools the credential off");
|
||||
assertEquals(1, adapter.spawnCount("sol"));
|
||||
}
|
||||
|
||||
/**
|
||||
* sol and terra share one OpenAI credential; cooling off because of an outage classified via
|
||||
* ONE of them must lock out the other too, exactly like CB-578 stage B quarantine does.
|
||||
*/
|
||||
@Test
|
||||
void twoProfilesSharingACredentialAreBothCoolingOffByOneOutageIncident() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"terra", stubWorker("terra", "shared-openai"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
outagePolicy.record("shared-openai", "term-1", "API Error: 500");
|
||||
outagePolicy.record("shared-openai", "term-2", "API Error: 500");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(), outagePolicy);
|
||||
|
||||
assertThrows(PlacementException.class, () -> composite.spawn(new SpawnRequest("sol", null, null)));
|
||||
assertThrows(PlacementException.class, () -> composite.spawn(new SpawnRequest("terra", null, null)),
|
||||
"terra shares sol's credential, so it must be locked out too");
|
||||
assertEquals(0, adapter.spawnCount("sol"));
|
||||
assertEquals(0, adapter.spawnCount("terra"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void placementSkipsACoolingOffProfileAndRoutesToAnotherOne() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
outagePolicy.record("shared-openai", "term-1", "API Error: 500");
|
||||
outagePolicy.record("shared-openai", "term-2", "API Error: 500");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.weighted(), _ -> 0, null, BackendQuarantine.none(), outagePolicy);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(), "sol is cooling off, so an unqualified spawn must land on b");
|
||||
assertEquals(0, adapter.spawnCount("sol"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void automaticPlacementNamesCoolingOffWhenEveryCandidateIsCoolingOff() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"terra", stubWorker("terra", "shared-openai"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
outagePolicy.record("shared-openai", "term-1", "API Error: 500");
|
||||
outagePolicy.record("shared-openai", "term-2", "API Error: 500");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.weighted(), _ -> 0, null, BackendQuarantine.none(), outagePolicy);
|
||||
|
||||
// PlacementPolicy.select throws uncaught out of spawn() when available() is empty — never
|
||||
// wrapped in PeerUnreachableException, which is only for a delegate that actually refused.
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest(null, null, null)),
|
||||
"both candidates share the cooling-off credential — nothing is available");
|
||||
assertTrue(e.getMessage().contains("cooling off"),
|
||||
"message says cooling off, not exhausted, since nothing here is quarantined: " + e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aCoolOffLiftsOnTheInjectedClockAndTheProfileBecomesSpawnableAgain() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
AtomicLong nowNanos = new AtomicLong(0L);
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(nowNanos::get);
|
||||
outagePolicy.record("shared-openai", "term-1", "API Error: 500");
|
||||
outagePolicy.record("shared-openai", "term-2", "API Error: 500");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(), outagePolicy);
|
||||
|
||||
assertThrows(PlacementException.class, () -> composite.spawn(new SpawnRequest("sol", null, null)),
|
||||
"still inside the 60s cool-off");
|
||||
|
||||
nowNanos.set(TimeUnit.SECONDS.toNanos(61));
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("sol", null, null));
|
||||
assertEquals("sol", h.profile(), "the cool-off expired on the injected clock — sol is spawnable again");
|
||||
assertEquals(1, adapter.spawnCount("sol"));
|
||||
}
|
||||
|
||||
/**
|
||||
* A credential that is BOTH exhaustion-quarantined (CB-578 stage B) and cooling off (fleetd
|
||||
* #201 Unit 5) at once — the two checks are independent and can both be true for one profile.
|
||||
* Exhaustion quarantine must win: the refusal names quarantine, never cooling off, because
|
||||
* {@code enforceNotQuarantined} is checked (and throws) before {@code enforceNotCoolingOff} ever
|
||||
* runs.
|
||||
*/
|
||||
@Test
|
||||
void quarantineWinsOverCoolingOffWhenBothAreActiveOnTheSameCredential() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"terra", stubWorker("terra", "shared-openai"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
quarantine.quarantine("shared-openai");
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
outagePolicy.record("shared-openai", "term-1", "API Error: 500");
|
||||
outagePolicy.record("shared-openai", "term-2", "API Error: 500");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, quarantine, outagePolicy);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("sol", null, null)));
|
||||
assertTrue(e.getMessage().contains("quarantined"),
|
||||
"exhaustion quarantine takes priority in the message: " + e.getMessage());
|
||||
assertFalse(e.getMessage().contains("cooling off"),
|
||||
"cooling off is never mentioned when quarantine already refused the spawn: " + e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFleetWithNoBackendErrorsEverRecordedBehavesExactlyAsBeforeCoolOffExisted() {
|
||||
// A never-record()-called BackendOutagePolicy is naturally inert — the 2/5/6-arg
|
||||
// constructors used throughout this file all wire this in implicitly (NO_OUTAGE_POLICY).
|
||||
// This test pins that an explicit fresh instance also never refuses a spawn, for any profile.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("claude", null, null)));
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("gemini", null, null)));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -114,6 +114,68 @@ class EnvAllowListScrubTest {
|
||||
assertNull(EnvAllowListScrub.readReport(dir));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #224 criterion 5 (a gap the #221 reviewer flagged): {@link EnvAllowListScrub#shareWithGroup}
|
||||
* lists one FLAT level of {@code dir} and shares every file it finds there — which does cover the
|
||||
* OPTIONAL files a caller may or may not have written before calling it ({@code member-charter.md},
|
||||
* {@code ide-rules.md} — both written by {@code OpenCodeLauncher#writeConfig}), but nothing
|
||||
* pinned that down.
|
||||
* Without this test, a future change that writes a file AFTER the sharing call, or into a
|
||||
* subdirectory, would pass every existing test while quietly leaving that file unreadable to a
|
||||
* different-uid member.
|
||||
*
|
||||
* <p>This drives {@code shareWithGroup} directly against a directory holding several flat files —
|
||||
* not only {@code opencode.json}, but also the two optional ones named above plus a third,
|
||||
* unrelated file, so the assertion is "every flat file", not "the two files someone thought of".
|
||||
*/
|
||||
@Test
|
||||
void shareWithGroupCoversEveryFlatFileIncludingTheOptionalOnes(@TempDir Path dir) throws Exception {
|
||||
String group = currentUserGroup();
|
||||
Files.writeString(dir.resolve("opencode.json"), "{}\n");
|
||||
Files.writeString(dir.resolve("member-charter.md"), "# charter\n");
|
||||
Files.writeString(dir.resolve("ide-rules.md"), "# ide rules\n");
|
||||
Files.writeString(dir.resolve("another-flat-file.txt"), "unrelated\n");
|
||||
|
||||
EnvAllowListScrub.shareWithGroup(dir, group);
|
||||
|
||||
assertEquals("rwxr-x---", java.nio.file.attribute.PosixFilePermissions.toString(
|
||||
Files.getPosixFilePermissions(dir)),
|
||||
"the directory itself must be group-traversable+readable, owner-only writable");
|
||||
for (String name : List.of("opencode.json", "member-charter.md", "ide-rules.md", "another-flat-file.txt")) {
|
||||
Path file = dir.resolve(name);
|
||||
assertEquals("rw-r-----", java.nio.file.attribute.PosixFilePermissions.toString(
|
||||
Files.getPosixFilePermissions(file)),
|
||||
name + " must be group-readable, never group-writable");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The current process's REAL primary group — resolved via {@code id -gn}, never by reading a
|
||||
* directory's owning group (fleetd #225: that reads wherever Maven happened to be started from,
|
||||
* not the process's own group, and the two diverge outside a home checkout). Skips (never fails)
|
||||
* when {@code id} is unavailable or its primary group cannot be resolved on this host.
|
||||
*/
|
||||
private static String currentUserGroup() {
|
||||
String out;
|
||||
boolean ok;
|
||||
try {
|
||||
Process p = new ProcessBuilder("id", "-gn").redirectErrorStream(true).start();
|
||||
String raw = new String(p.getInputStream().readAllBytes(), StandardCharsets.UTF_8).trim();
|
||||
out = raw;
|
||||
ok = p.waitFor(5, java.util.concurrent.TimeUnit.SECONDS) && p.exitValue() == 0 && !raw.isBlank();
|
||||
} catch (IOException e) {
|
||||
out = null;
|
||||
ok = false;
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
out = null;
|
||||
ok = false;
|
||||
}
|
||||
assumeTrue(ok, "cannot resolve this process's real primary group via `id -gn` on this host "
|
||||
+ "— skipping a POSIX-group-dependent test rather than failing it");
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* The same equality, for a shell that is INTERACTIVE but NOT a login shell — the shape herdr
|
||||
* opens on Linux.
|
||||
|
||||
+577
-19
@@ -12,20 +12,26 @@ import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
import static org.junit.jupiter.api.Assumptions.assumeTrue;
|
||||
|
||||
/**
|
||||
* CB-633: proves the allow-list scrub is actually WIRED INTO the spawn path — not merely that its
|
||||
@@ -94,18 +100,26 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
}
|
||||
|
||||
/**
|
||||
* A non-zsh shell cannot read {@code ZDOTDIR} at all. The launcher must fall back rather than
|
||||
* generate a directory nothing will ever read — a directory that would look like protection.
|
||||
* fleetd #155: a non-zsh shell cannot read {@code ZDOTDIR} at all, so under {@code
|
||||
* policy: allow-list} — the policy the operator picked specifically for a control a sourced file
|
||||
* cannot undo — the launcher must REFUSE the spawn rather than silently generate a directory
|
||||
* nothing will ever read (protection theatre) or fall back to the weaker overlay (exactly the
|
||||
* "control silently does nothing" defect this ticket exists to close). Real path: through {@link
|
||||
* HerdrPeerLauncher#spawn}, the method the daemon actually calls.
|
||||
*/
|
||||
@Test
|
||||
void aNonZshShellGeneratesNothingAndFallsBack() {
|
||||
void aNonZshShellUnderAllowListPolicyRefusesTheSpawn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/bash");
|
||||
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class,
|
||||
() -> launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV)),
|
||||
"policy=allow-list on a non-zsh shell must refuse the spawn, not silently degrade");
|
||||
|
||||
assertTrue(e.getMessage().contains("allow-list") && e.getMessage().contains("/bin/bash"),
|
||||
"the refusal must name the policy and the actual shell, got: " + e.getMessage());
|
||||
assertFalse(launcher.env.containsKey("ZDOTDIR"),
|
||||
"bash ignores ZDOTDIR; setting it would be protection theatre");
|
||||
"a refused spawn must not have generated (or wired in) a scrub directory: " + launcher.env);
|
||||
}
|
||||
|
||||
private static Supplier<FleetConfig.MemberCredentials> allowList() {
|
||||
@@ -119,6 +133,18 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
FleetConfig.MemberCredentials.POLICY_ALLOW_LIST, allow, List.of(), null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Same as {@link #allowList()} but with an operator-configured {@code known:} list, so the
|
||||
* fallback overlay (the non-zsh / no-memberLoginShell path) has something visible to shadow —
|
||||
* {@link FleetConfig.MemberCredentials#blockedSet()} is {@code known - allow}, so an empty
|
||||
* {@code known} (what {@link #allowList()} uses) blocks nothing and a fallback test would have
|
||||
* no sentinel entry to assert on.
|
||||
*/
|
||||
private static Supplier<FleetConfig.MemberCredentials> allowListWithKnown(List<String> known) {
|
||||
return () -> new FleetConfig.MemberCredentials(
|
||||
FleetConfig.MemberCredentials.POLICY_ALLOW_LIST, List.of(), known, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-633 follow-up: a name that lives ONLY in {@code memberCredentials.allow:} — no profile
|
||||
* mentions it — must survive the scrub the real spawn path generates. Calling {@code
|
||||
@@ -144,10 +170,10 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
/**
|
||||
* {@code SSH_AUTH_SOCK} is a live ssh-agent handle, not a value — it must stay blocked under
|
||||
* {@code allow-list} even when the operator lists it under {@code allow:}, because {@code
|
||||
* sshAuthSock} defaults to blocked. Governed ONLY by {@code memberCredentials.sshAuthSock}.
|
||||
* sshAgentEnv} defaults to omit. Governed ONLY by {@code memberCredentials.sshAgentEnv}.
|
||||
*/
|
||||
@Test
|
||||
void sshAuthSockStaysBlockedEvenWhenListedInMemberCredentialsAllow() {
|
||||
void sshAgentEnvStaysOmittedEvenWhenListedInMemberCredentialsAllow() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
WiringLauncher launcher = new WiringLauncher(herdr,
|
||||
allowListWithAllow(List.of("SSH_AUTH_SOCK")));
|
||||
@@ -158,7 +184,35 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
String scrub = readAll(dir.resolve(EnvAllowListScrub.SCRUB_FILE));
|
||||
assertFalse(scrub.contains("'SSH_AUTH_SOCK'"),
|
||||
"SSH_AUTH_SOCK must not be on the derived allow-list just because the operator put "
|
||||
+ "it under allow: — sshAuthSock is unset here, so it defaults to block");
|
||||
+ "it under allow: — sshAgentEnv is unset here, so it defaults to omit");
|
||||
}
|
||||
|
||||
@Test
|
||||
void brokerUriEnvStaysBlockedWhenListedInMemberCredentialsAllow() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
WiringLauncher launcher = new WiringLauncher(herdr,
|
||||
allowListWithAllow(List.of("BROKER_CONNECTION_URI")), "/bin/zsh", null,
|
||||
() -> config("BROKER_CONNECTION_URI"));
|
||||
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
Path dir = Path.of(launcher.env.get("ZDOTDIR"));
|
||||
assertFalse(readAll(dir.resolve(EnvAllowListScrub.SCRUB_FILE)).contains("'BROKER_CONNECTION_URI'"),
|
||||
"broker.uriEnv must not reach a member even when listed in memberCredentials.allow:");
|
||||
}
|
||||
|
||||
@Test
|
||||
void brokerUriEnvIsDeniedUnderTheDenyListPolicyEvenWhenAllowed() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
WiringLauncher launcher = new WiringLauncher(herdr,
|
||||
() -> new FleetConfig.MemberCredentials(null, List.of("BROKER_CONNECTION_URI"), List.of(), null),
|
||||
"/bin/bash", null,
|
||||
() -> config("BROKER_CONNECTION_URI"));
|
||||
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
assertEquals("blocked-by-fleetd-cb596-see-gitea-issue-82", launcher.env.get("BROKER_CONNECTION_URI"),
|
||||
"the deny-list overlay must deny broker.uriEnv even when allow: names it");
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -197,18 +251,26 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
}
|
||||
|
||||
/**
|
||||
* Lead-review fix: on a NON-zsh shell no scrub ever runs (bash ignores {@code ZDOTDIR}), so the
|
||||
* "allowed N of M" line — which describes what the scrub does — must not be printed there either.
|
||||
* Before this fix the line was logged BEFORE the zsh gate, so a non-zsh host printed e.g.
|
||||
* "allowed 1 of 3" while blocking nothing at all, telling an operator a control ran when it did
|
||||
* not. Real path: goes through {@link HerdrPeerLauncher#spawn}, same as the sibling test above,
|
||||
* with the shell fixed to bash so the fallback branch is the one exercised.
|
||||
* fleetd #269 follow-up: the same overclaim the WARN in {@code logCredentialGap} was fixed for,
|
||||
* in the INFO line beside it. With {@code memberHerdrSocket} configured, member panes are routed
|
||||
* to a second herdr whose environment fleetd has no channel to inspect, so the counts come from
|
||||
* fleetd's OWN environment. The bare line "member credentials: allowed 1 of 3" reads as a fact
|
||||
* about the member's pane, and there it is not one.
|
||||
*
|
||||
* <p>#269 reworded four sites and stopped at the sibling below; this pins the pair together so
|
||||
* a future edit cannot fix one and leave the other. Real path: asserted after a real {@link
|
||||
* HerdrPeerLauncher#spawn}, reading the log production actually emits.
|
||||
*/
|
||||
@Test
|
||||
void noAllowedCountLineIsEmittedOnTheNonZshFallbackPath() {
|
||||
void theAllowedCountLineSaysWhoseEnvironmentItCountedWhenMemberHerdrSocketIsSet(@TempDir Path worktreeRoot)
|
||||
throws IOException {
|
||||
String group = currentUserGroup();
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Set<String> hostEnvNames = Set.of(INJECTED, "SOME_UNRELATED_NAME", "ANOTHER_UNRELATED_NAME");
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/bash", () -> hostEnvNames);
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/bash-should-be-ignored",
|
||||
() -> hostEnvNames,
|
||||
() -> configWithMemberHerdrSocketRootAndGroup("/tmp/other-user-herdr.sock", "/bin/zsh",
|
||||
worktreeRoot.toString(), group));
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
Level original = logger.getLevel();
|
||||
@@ -223,12 +285,388 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
logger.setLevel(original);
|
||||
}
|
||||
|
||||
List<String> lines = appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList();
|
||||
String coverage = lines.stream()
|
||||
.filter(l -> l.startsWith("member credentials: allowed "))
|
||||
.findFirst()
|
||||
.orElse(null);
|
||||
assertNotNull(coverage, "the coverage line must still be logged — narrowing the claim must "
|
||||
+ "not silently delete the line: " + lines);
|
||||
assertTrue(coverage.contains("fleetd's OWN environment"),
|
||||
"the line must say whose environment it counted: " + coverage);
|
||||
assertTrue(coverage.contains("NOT the member pane's"),
|
||||
"and must say plainly that it is not the member's: " + coverage);
|
||||
// The counts themselves stay real — narrowing the claim must not turn them into constants.
|
||||
assertTrue(coverage.startsWith("member credentials: allowed 1 of 3"),
|
||||
"the real counts must survive the rewording: " + coverage);
|
||||
}
|
||||
|
||||
/**
|
||||
* Lead-review fix: on a NON-zsh shell no scrub ever runs (bash ignores {@code ZDOTDIR}), so the
|
||||
* "allowed N of M" line — which describes what the scrub does — must not be printed there either.
|
||||
* Before this fix the line was logged BEFORE the zsh gate, so a non-zsh host printed e.g.
|
||||
* "allowed 1 of 3" while blocking nothing at all, telling an operator a control ran when it did
|
||||
* not. fleetd #155: the spawn itself is now refused on this path (see {@code
|
||||
* aNonZshShellUnderAllowListPolicyRefusesTheSpawn}) rather than falling back — this test's own
|
||||
* concern still holds under the refusal: the "allowed N of M" line describes a scrub that never
|
||||
* ran here, so it must still never appear. Real path: goes through {@link
|
||||
* HerdrPeerLauncher#spawn}, same as the sibling test above, with the shell fixed to bash.
|
||||
*/
|
||||
@Test
|
||||
void noAllowedCountLineIsEmittedOnTheNonZshRefusalPath() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Set<String> hostEnvNames = Set.of(INJECTED, "SOME_UNRELATED_NAME", "ANOTHER_UNRELATED_NAME");
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/bash", () -> hostEnvNames);
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
Level original = logger.getLevel();
|
||||
logger.setLevel(Level.INFO);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV)),
|
||||
"policy=allow-list on a non-zsh shell must refuse the spawn");
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(original);
|
||||
}
|
||||
|
||||
assertFalse(appender.list.stream()
|
||||
.anyMatch(e -> e.getFormattedMessage().startsWith("member credentials: allowed ")),
|
||||
"no scrub runs on a non-zsh shell, so no 'allowed N of M' count may be printed — got: "
|
||||
+ appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #185 stage 2 pin: with {@code memberHerdrSocket:} absent (today's only mode, and the
|
||||
* default — this host runs no other), the gap detector's WARN/INFO conclusions read exactly as
|
||||
* they did before this fix. Real path: {@code FLEETD_WORKER_TOKEN} (the test profile's own
|
||||
* {@code tokenEnv}) is a name the derived allow-list keeps, so it gets the "UNBLOCKED" WARN;
|
||||
* {@code SOME_UNKNOWN_SECRET_TOKEN} is not derived from anywhere, so it gets the "scrub blanks
|
||||
* them" INFO. This is the exact shape #185 stage 2 must not touch on this path.
|
||||
*/
|
||||
@Test
|
||||
void gapConclusionsAreByteIdenticalWhenMemberHerdrSocketIsAbsent() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Set<String> hostEnvNames = Set.of("FLEETD_WORKER_TOKEN", "SOME_UNKNOWN_SECRET_TOKEN");
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/zsh", () -> hostEnvNames,
|
||||
() -> config(null));
|
||||
|
||||
List<String> messages = spawnAndCaptureLogs(launcher);
|
||||
|
||||
assertTrue(messages.contains("memberCredentials gap: 1 credential-shaped env var name(s) are on "
|
||||
+ "neither known: nor allow: — the derived allow-list keeps them anyway (a "
|
||||
+ "profile's gitTokenEnv/gitHostEnv/tokenEnv/env: names one, or this spawn "
|
||||
+ "injects it), so every member pane inherits them UNBLOCKED — [FLEETD_WORKER_TOKEN]. "
|
||||
+ "Add each to memberCredentials.known (or .allow if a member legitimately needs "
|
||||
+ "it), or remove it from whatever profile setting derives it in."),
|
||||
"expected the pre-existing UNBLOCKED WARN unchanged, got: " + messages);
|
||||
assertTrue(messages.contains("memberCredentials gap: 1 credential-shaped env var name(s) are on "
|
||||
+ "neither known: nor allow: — [SOME_UNKNOWN_SECRET_TOKEN]. The allow-list scrub "
|
||||
+ "blanks them anyway (they are not on the derived allow-list), so no member pane "
|
||||
+ "keeps them; add each to memberCredentials.known or .allow to make that explicit."),
|
||||
"expected the pre-existing 'scrub blanks them' INFO unchanged, got: " + messages);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #185 stage 2: with {@code memberHerdrSocket:} configured, member panes run under a
|
||||
* different OS user — {@link HerdrPeerLauncher#hostEnvNames} describes fleetd's own process, not
|
||||
* that user's. Neither "inherits them UNBLOCKED" nor "scrub blanks them" is evidence-backed
|
||||
* there, so neither may print; the single unknown-environment WARN must, naming the config key.
|
||||
*
|
||||
* <p>fleetd #155: {@code memberLoginShell} is now given explicitly as zsh, so the spawn reaches
|
||||
* this WARN through the (still-degrading, not refusing) missing-{@code worktreeRoot}/{@code
|
||||
* worktreeGroup} fallback rather than through the zsh gate, which now refuses instead of falling
|
||||
* back — see {@code aNonZshShellUnderAllowListPolicyRefusesTheSpawn}. This test's own concern
|
||||
* (the unknown-environment WARN) is orthogonal to which fallback reached {@code
|
||||
* logCredentialGap}, so it still holds.
|
||||
*/
|
||||
@Test
|
||||
void gapDetectorReportsUnknownInsteadOfAConclusionWhenMemberHerdrSocketIsConfigured() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Set<String> hostEnvNames = Set.of("FLEETD_WORKER_TOKEN", "SOME_UNKNOWN_SECRET_TOKEN");
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/zsh", () -> hostEnvNames,
|
||||
() -> configWithMemberHerdrSocket("/tmp/other-user-herdr.sock", "/bin/zsh"));
|
||||
|
||||
List<String> messages = spawnAndCaptureLogs(launcher);
|
||||
|
||||
assertTrue(messages.stream().anyMatch(m -> m.contains("memberHerdrSocket")
|
||||
&& m.contains("UNKNOWN") && m.contains("cannot be verified")),
|
||||
"expected the unknown-member-environment WARN naming memberHerdrSocket, got: " + messages);
|
||||
assertFalse(messages.stream().anyMatch(m -> m.contains("UNBLOCKED")),
|
||||
"the 'inherits them UNBLOCKED' conclusion must not print once the evidence is about "
|
||||
+ "the wrong (daemon's own) environment — got: " + messages);
|
||||
assertFalse(messages.stream().anyMatch(m -> m.contains("scrub blanks them")),
|
||||
"the 'scrub blanks them' conclusion must not print once the evidence is about the "
|
||||
+ "wrong (daemon's own) environment — got: " + messages);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #185 stage 2: the unknown-environment WARN is a standing fact about this launcher's
|
||||
* configuration, not per-spawn news — it must fire once per launcher instance, the same shape as
|
||||
* every other one-time WARN in this class (e.g. {@code warnNonZsh}).
|
||||
*/
|
||||
@Test
|
||||
void theUnknownEnvironmentWarnFiresOnceNotOncePerSpawn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Set<String> hostEnvNames = Set.of("FLEETD_WORKER_TOKEN", "SOME_UNKNOWN_SECRET_TOKEN");
|
||||
// fleetd #155: memberLoginShell given explicitly as zsh — see the sibling test above for why.
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/zsh", () -> hostEnvNames,
|
||||
() -> configWithMemberHerdrSocket("/tmp/other-user-herdr.sock", "/bin/zsh"));
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
Level original = logger.getLevel();
|
||||
logger.setLevel(Level.WARN);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(original);
|
||||
}
|
||||
|
||||
long count = appender.list.stream()
|
||||
.filter(e -> e.getFormattedMessage().contains("cannot be verified from here"))
|
||||
.count();
|
||||
assertEquals(1, count, "the unknown-member-environment WARN must fire once per launcher "
|
||||
+ "instance, not once per spawn — got " + count + " occurrence(s) among: "
|
||||
+ appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #184 item 5: the unknown-environment WARN must state the honest reason for the
|
||||
* UNKNOWN conclusion — fleetd has no channel to confirm what OS user the second herdr runs
|
||||
* as — and must NOT assert as fact that member panes run under a different OS user just
|
||||
* because {@code memberHerdrSocket} is configured. An operator may point it at a second herdr
|
||||
* running as the SAME user, for pane isolation alone; in that case members DO inherit fleetd's
|
||||
* environment, and asserting otherwise would tell the operator to disregard a real, known gap.
|
||||
*/
|
||||
@Test
|
||||
void unknownEnvironmentWarnStatesUncertaintyNotAnAssertedDifferentUser() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Set<String> hostEnvNames = Set.of("FLEETD_WORKER_TOKEN", "SOME_UNKNOWN_SECRET_TOKEN");
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/zsh", () -> hostEnvNames,
|
||||
() -> configWithMemberHerdrSocket("/tmp/other-user-herdr.sock", "/bin/zsh"));
|
||||
|
||||
List<String> messages = spawnAndCaptureLogs(launcher);
|
||||
|
||||
assertTrue(messages.stream().anyMatch(m -> m.contains("memberHerdrSocket")
|
||||
&& m.contains("no channel to confirm what OS user that herdr runs as")),
|
||||
"expected the WARN to name the actual uncertainty (no channel to confirm the "
|
||||
+ "herdr's uid), got: " + messages);
|
||||
assertFalse(messages.stream().anyMatch(m -> m.contains("member panes run under a different OS "
|
||||
+ "user than fleetd's own process")),
|
||||
"the WARN must not assert as fact that members run under a different OS user just "
|
||||
+ "because memberHerdrSocket is configured — got: " + messages);
|
||||
}
|
||||
|
||||
/**
|
||||
* Hard constraint: the gap detector must never log an env var VALUE, only its NAME. {@code
|
||||
* SOME_UNKNOWN_SECRET_TOKEN} resolves to a distinctive canary value through the same {@code env}
|
||||
* lookup the launcher uses elsewhere (SHELL, PATH, token resolution) — proving the value IS
|
||||
* resolvable does not mean the detector reads it, since {@link HerdrPeerLauncher#hostEnvNames}
|
||||
* (names only) is its data source, never {@code env.apply(name)} for those names.
|
||||
*/
|
||||
@Test
|
||||
void theGapDetectorNeverLogsAnEnvVarValueOnlyItsName() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
String canary = "sekrit-value-CANARY-9f3a1b7c";
|
||||
Set<String> hostEnvNames = Set.of("FLEETD_WORKER_TOKEN", "SOME_UNKNOWN_SECRET_TOKEN");
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/zsh", () -> hostEnvNames,
|
||||
() -> config(null), Map.of("SOME_UNKNOWN_SECRET_TOKEN", canary));
|
||||
|
||||
List<String> messages = spawnAndCaptureLogs(launcher);
|
||||
|
||||
assertTrue(messages.stream().anyMatch(m -> m.contains("SOME_UNKNOWN_SECRET_TOKEN")),
|
||||
"expected the credential-shaped NAME to appear in the log, got: " + messages);
|
||||
assertFalse(messages.stream().anyMatch(m -> m.contains(canary)),
|
||||
"the log must never contain an env var VALUE, only its NAME — got: " + messages);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213 defect 1, acceptance criterion 1 — updated by fleetd #155: {@code
|
||||
* memberHerdrSocket} configured and {@code memberLoginShell} configured as non-zsh must now
|
||||
* REFUSE the spawn (not fall back to the sentinel overlay — see {@code
|
||||
* aNonZshShellUnderAllowListPolicyRefusesTheSpawn}'s javadoc for why). The WiringLauncher's own
|
||||
* {@code env("SHELL")} is deliberately set to {@code /bin/zsh} — the OPPOSITE of what {@code
|
||||
* memberLoginShell} says — so a launcher that (incorrectly) fell back to fleetd's own {@code
|
||||
* $SHELL} here would wrongly pass the gate and fail this test.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketWithNonZshMemberLoginShellRefusesTheSpawn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowListWithKnown(List.of("SOME_TOKEN")),
|
||||
"/bin/zsh", null,
|
||||
() -> configWithMemberHerdrSocket("/tmp/other-user-herdr.sock", "/bin/bash"));
|
||||
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class,
|
||||
() -> launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV)),
|
||||
"a non-zsh configured memberLoginShell under policy=allow-list must refuse the spawn");
|
||||
|
||||
assertTrue(e.getMessage().contains("/bin/bash"),
|
||||
"the refusal must name the configured memberLoginShell, got: " + e.getMessage());
|
||||
assertFalse(launcher.env.containsKey("SOME_TOKEN"),
|
||||
"a refused spawn must not have touched the pane env at all: " + launcher.env);
|
||||
assertFalse(launcher.env.containsKey("ZDOTDIR"),
|
||||
"no scrub directory may be generated when the configured member login shell is not zsh");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213 defect 1, acceptance criterion 2 — updated by fleetd #155: {@code
|
||||
* memberHerdrSocket} configured and NO {@code memberLoginShell} configured must now REFUSE the
|
||||
* spawn exactly like criterion 1 above — AND fleetd's own {@code $SHELL} must never even be
|
||||
* consulted (not merely "not decisive"). The fixture's {@code env} function reports {@code
|
||||
* /bin/zsh} for {@code SHELL} — a value that would WRONGLY pass the zsh gate if the fix
|
||||
* regressed to reading it — while flagging whether it was ever asked for at all, so this test
|
||||
* fails loudly on either kind of regression.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketWithNoMemberLoginShellRefusesAndNeverConsultsFleetdsOwnShell() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AtomicBoolean shellQueried = new AtomicBoolean(false);
|
||||
Function<String, String> env = name -> {
|
||||
if ("SHELL".equals(name)) {
|
||||
shellQueried.set(true);
|
||||
return "/bin/zsh"; // would wrongly pass the zsh gate if this ever leaked through
|
||||
}
|
||||
return null;
|
||||
};
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowListWithKnown(List.of("SOME_TOKEN")), env,
|
||||
() -> configWithMemberHerdrSocket("/tmp/other-user-herdr.sock", null));
|
||||
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV)),
|
||||
"no configured memberLoginShell under policy=allow-list must refuse the spawn, same "
|
||||
+ "as an explicit non-zsh one");
|
||||
|
||||
assertFalse(shellQueried.get(), "fleetd's own $SHELL must never be consulted once "
|
||||
+ "memberHerdrSocket is configured — only memberLoginShell: may decide the gate");
|
||||
assertFalse(launcher.env.containsKey("SOME_TOKEN"),
|
||||
"a refused spawn must not have touched the pane env at all, same as a "
|
||||
+ "configured non-zsh shell");
|
||||
assertFalse(launcher.env.containsKey("ZDOTDIR"),
|
||||
"no scrub may run without a configured memberLoginShell");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213 defect 2, acceptance criterion 3: with {@code memberHerdrSocket} configured, a
|
||||
* zsh {@code memberLoginShell}, and {@code worktreeRoot}/{@code worktreeGroup} both configured,
|
||||
* the generated scrub directory must live under {@code worktreeRoot} — NEVER under {@code
|
||||
* java.io.tmpdir}, which is fleetd's own 0700 temp dir and unreadable by the member's different
|
||||
* OS user. {@code worktreeGroup} is set to the CURRENT process's own primary group so {@code
|
||||
* EnvAllowListScrub}'s group-sharing step resolves on whatever host runs this test, rather than
|
||||
* hardcoding a group name that may not exist here.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketWithZshMemberLoginShellPutsTheScrubOutsideJavaIoTmpdir(@TempDir Path worktreeRoot)
|
||||
throws IOException {
|
||||
String group = currentUserGroup();
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/bash-should-be-ignored", null,
|
||||
() -> configWithMemberHerdrSocketRootAndGroup("/tmp/other-user-herdr.sock", "/bin/zsh",
|
||||
worktreeRoot.toString(), group));
|
||||
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
String zdotdir = launcher.env.get("ZDOTDIR");
|
||||
assertNotNull(zdotdir, "a zsh memberLoginShell with worktreeRoot+worktreeGroup configured "
|
||||
+ "must still generate a ZDOTDIR");
|
||||
Path dir = Path.of(zdotdir);
|
||||
// The immediate PARENT is asserted (not merely startsWith(java.io.tmpdir)), because a
|
||||
// JUnit @TempDir is itself carved out of the JVM's java.io.tmpdir — startsWith alone would
|
||||
// pass by coincidence of the test fixture, not because the launcher used worktreeRoot.
|
||||
assertEquals(worktreeRoot.toAbsolutePath().normalize(), dir.getParent(),
|
||||
"the generated scrub directory's parent must be the configured worktreeRoot, not "
|
||||
+ "System.getProperty(\"java.io.tmpdir\") — got parent " + dir.getParent());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213, acceptance criterion 4: with {@code memberHerdrSocket} absent (today's only
|
||||
* mode), behaviour must be byte-identical to before this fix — fleetd's own {@code $SHELL}
|
||||
* still decides the gate (proven here, not merely assumed, by flagging the lookup), and the
|
||||
* scrub still lands under {@code java.io.tmpdir}.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketAbsentStillConsultsFleetdsOwnShellAndBehavesAsBefore() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AtomicBoolean shellQueried = new AtomicBoolean(false);
|
||||
Function<String, String> env = name -> {
|
||||
if ("SHELL".equals(name)) {
|
||||
shellQueried.set(true);
|
||||
return "/bin/zsh";
|
||||
}
|
||||
return null;
|
||||
};
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), env, null); // memberHerdrSocket absent
|
||||
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
assertTrue(shellQueried.get(), "with memberHerdrSocket absent, fleetd's own $SHELL must "
|
||||
+ "still decide the zsh gate, unchanged from before this fix");
|
||||
String zdotdir = launcher.env.get("ZDOTDIR");
|
||||
assertNotNull(zdotdir, "SHELL=/bin/zsh with memberHerdrSocket absent must still generate a "
|
||||
+ "ZDOTDIR, as before this fix");
|
||||
Path dir = Path.of(zdotdir);
|
||||
assertTrue(dir.startsWith(Path.of(System.getProperty("java.io.tmpdir"))),
|
||||
"with memberHerdrSocket absent the scrub directory must still be generated under "
|
||||
+ "java.io.tmpdir, unchanged from before this fix: " + dir);
|
||||
}
|
||||
|
||||
/**
|
||||
* The current process's REAL primary group — resolved via {@code id -gn}, never by reading a
|
||||
* directory's owning group (fleetd #225). Those two coincide only by accident: a directory's
|
||||
* group is whichever group happened to own the path Maven was started from — {@code staff} in
|
||||
* a home checkout, {@code wheel} under {@code /private/tmp} on macOS — and the fix-up this test
|
||||
* exercises then fails for real when the operator is not a member of that borrowed group,
|
||||
* exactly the case {@code assumeTrue(view != null, ...)} never covered (it only detects a
|
||||
* filesystem with no POSIX groups at all, not a resolvable-but-wrong one). Skips (never fails)
|
||||
* when {@code id} is unavailable or its primary group cannot be resolved on this host.
|
||||
*/
|
||||
private static String currentUserGroup() {
|
||||
String out;
|
||||
boolean ok;
|
||||
try {
|
||||
Process p = new ProcessBuilder("id", "-gn").redirectErrorStream(true).start();
|
||||
try (java.io.BufferedReader r = new java.io.BufferedReader(
|
||||
new java.io.InputStreamReader(p.getInputStream(), java.nio.charset.StandardCharsets.UTF_8))) {
|
||||
out = r.lines().collect(java.util.stream.Collectors.joining("\n")).trim();
|
||||
}
|
||||
ok = p.waitFor(5, java.util.concurrent.TimeUnit.SECONDS) && p.exitValue() == 0 && !out.isBlank();
|
||||
} catch (IOException e) {
|
||||
out = null;
|
||||
ok = false;
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
out = null;
|
||||
ok = false;
|
||||
}
|
||||
assumeTrue(ok, "cannot resolve this process's real primary group via `id -gn` on this host "
|
||||
+ "— skipping a POSIX-group-dependent test rather than failing it");
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Spawn once through the real launcher path, capturing every INFO+ line this class logs. */
|
||||
private static List<String> spawnAndCaptureLogs(HerdrPeerLauncher launcher) {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
Level original = logger.getLevel();
|
||||
logger.setLevel(Level.INFO);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(original);
|
||||
}
|
||||
return appender.list.stream().map(ILoggingEvent::getFormattedMessage).toList();
|
||||
}
|
||||
|
||||
private static String readAll(Path p) {
|
||||
try {
|
||||
return Files.readString(p);
|
||||
@@ -260,15 +698,52 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
|
||||
/** Plus an injectable {@code hostEnvNames} source, for the "allowed N of M" log line test. */
|
||||
WiringLauncher(FakeHerdr herdr, Supplier<FleetConfig.MemberCredentials> creds, String shell,
|
||||
Supplier<Set<String>> hostEnvNames) {
|
||||
Supplier<Set<String>> hostEnvNames) {
|
||||
this(herdr, creds, shell, hostEnvNames, null);
|
||||
}
|
||||
|
||||
WiringLauncher(FakeHerdr herdr, Supplier<FleetConfig.MemberCredentials> creds, String shell,
|
||||
Supplier<Set<String>> hostEnvNames, Supplier<FleetConfig> config) {
|
||||
this(herdr, creds, shell, hostEnvNames, config, Map.of());
|
||||
}
|
||||
|
||||
/**
|
||||
* Plus a host-env value map (name → value), resolved through the same {@code env} lookup
|
||||
* every adapter uses for {@code SHELL}/{@code PATH}/token resolution — fleetd #185 stage 2's
|
||||
* "the gap detector never logs a value" tests use this to prove a value that IS resolvable
|
||||
* for a credential-shaped name never reaches the log, since the detector only ever reads
|
||||
* {@code hostEnvNames} (names), never {@code env.apply(name)} (values), for those names.
|
||||
*/
|
||||
WiringLauncher(FakeHerdr herdr, Supplier<FleetConfig.MemberCredentials> creds, String shell,
|
||||
Supplier<Set<String>> hostEnvNames, Supplier<FleetConfig> config,
|
||||
Map<String, String> extraEnvValues) {
|
||||
super("test", new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of("test", profile()), "test",
|
||||
name -> "SHELL".equals(name) ? shell : null,
|
||||
0, () -> 0L, () -> { }, null, creds, hostEnvNames);
|
||||
name -> "SHELL".equals(name) ? shell
|
||||
: (extraEnvValues != null && extraEnvValues.containsKey(name))
|
||||
? extraEnvValues.get(name) : null,
|
||||
0, () -> 0L, () -> { }, null, creds, hostEnvNames, config);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full control over the {@code env} lookup, bypassing the {@code shell}/{@code
|
||||
* extraEnvValues} convenience above entirely — fleetd #213's "fleetd's own $SHELL must
|
||||
* never be consulted" tests need to OBSERVE whether {@code SHELL} was ever looked up, which
|
||||
* a plain value substitution cannot do.
|
||||
*/
|
||||
WiringLauncher(FakeHerdr herdr, Supplier<FleetConfig.MemberCredentials> creds,
|
||||
Function<String, String> env, Supplier<FleetConfig> config) {
|
||||
super("test", new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of("test", profile()), "test",
|
||||
env, 0, () -> 0L, () -> { }, null, creds, null, config);
|
||||
}
|
||||
|
||||
@Override
|
||||
protected Launch buildLaunch(FleetConfig.Profile cfg, LaunchSpec spec) {
|
||||
Map<String, String> launchEnv = baseEnv(cfg);
|
||||
launchEnv.putAll(env);
|
||||
env.clear();
|
||||
env.putAll(launchEnv);
|
||||
return new Launch(env, List.of("test"));
|
||||
}
|
||||
|
||||
@@ -278,6 +753,89 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig config(String brokerUriEnv) {
|
||||
return new FleetConfig(null, null, null, Map.of(), null, null, null, null, null,
|
||||
new FleetConfig.Broker(null, brokerUriEnv, null), null, null, null, null, null,
|
||||
null, null, null, null, null).withDefaults();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #185 stage 2: a config with {@code memberHerdrSocket:} set — member panes run on a
|
||||
* second herdr owned by a different OS user, so {@link HerdrPeerLauncher#hostEnvNames} no
|
||||
* longer describes what a member pane inherits.
|
||||
*/
|
||||
private static FleetConfig configWithMemberHerdrSocket(String memberHerdrSocket) {
|
||||
return new FleetConfig(null, null, memberHerdrSocket, Map.of(), null, null, null, null, null,
|
||||
null, null, null, null, null, null, null, null, null, null, null).withDefaults();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213: as {@link #configWithMemberHerdrSocket(String)}, plus the {@code
|
||||
* memberLoginShell:} the member's OS user actually runs — the config key {@link
|
||||
* HerdrPeerLauncher#applyEnvironmentAllowListPolicy} must consult instead of fleetd's own
|
||||
* {@code $SHELL} once {@code memberHerdrSocket} is configured.
|
||||
*/
|
||||
private static FleetConfig configWithMemberHerdrSocket(String memberHerdrSocket, String memberLoginShell) {
|
||||
return new FleetConfig(
|
||||
null, // bind
|
||||
null, // herdrSocket
|
||||
memberHerdrSocket, // memberHerdrSocket
|
||||
Map.of(), // profiles
|
||||
null, // guard
|
||||
null, // worktreeRoot
|
||||
null, // lifecycle
|
||||
null, // spawnReadyTimeoutMs
|
||||
null, // spawnReadyPollMs
|
||||
null, // broker
|
||||
null, // primary
|
||||
null, // fleet
|
||||
null, // leadHeartbeat
|
||||
null, // health
|
||||
null, // placement
|
||||
null, // auth
|
||||
null, // configReload
|
||||
null, // quarantineCooldownSeconds
|
||||
null, // memberCredentials
|
||||
null, // coordinator
|
||||
null, // worktreeGroup
|
||||
memberLoginShell // memberLoginShell
|
||||
).withDefaults();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213: as above, plus {@code worktreeRoot:}/{@code worktreeGroup:} — both required for
|
||||
* the ZDOTDIR scrub to run at all once {@code memberHerdrSocket} is configured; either missing
|
||||
* falls back to the sentinel overlay (fleetd #155 left this branch alone — it is not a shell
|
||||
* problem, so it is not this ticket's refusal).
|
||||
*/
|
||||
private static FleetConfig configWithMemberHerdrSocketRootAndGroup(String memberHerdrSocket,
|
||||
String memberLoginShell, String worktreeRoot, String worktreeGroup) {
|
||||
return new FleetConfig(
|
||||
null, // bind
|
||||
null, // herdrSocket
|
||||
memberHerdrSocket, // memberHerdrSocket
|
||||
Map.of(), // profiles
|
||||
null, // guard
|
||||
worktreeRoot, // worktreeRoot
|
||||
null, // lifecycle
|
||||
null, // spawnReadyTimeoutMs
|
||||
null, // spawnReadyPollMs
|
||||
null, // broker
|
||||
null, // primary
|
||||
null, // fleet
|
||||
null, // leadHeartbeat
|
||||
null, // health
|
||||
null, // placement
|
||||
null, // auth
|
||||
null, // configReload
|
||||
null, // quarantineCooldownSeconds
|
||||
null, // memberCredentials
|
||||
null, // coordinator
|
||||
worktreeGroup, // worktreeGroup
|
||||
memberLoginShell // memberLoginShell
|
||||
).withDefaults();
|
||||
}
|
||||
|
||||
/** The generated directory is a temp directory; make sure the test does not leave a pile. */
|
||||
@Test
|
||||
void theGeneratedDirectoryIsRemovedWhenThePaneIsStopped() {
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #111 (CB-608): {@link MemberCredentialPolicyView} is the one place that turns a {@code
|
||||
* memberCredentials:} policy into names-and-counts, so the startup log line and {@code
|
||||
* GET /member-credentials} cannot drift apart. These tests pin: the counts always match the
|
||||
* policy that produced them, an absent/empty policy is represented honestly (never as "nothing
|
||||
* blocked"), and the view carries names only — no value ever flows through it, because it is
|
||||
* built only from {@link FleetConfig.MemberCredentials}, which itself never holds a value.
|
||||
*/
|
||||
class MemberCredentialPolicyViewTest {
|
||||
|
||||
@Test
|
||||
void nullPolicyIsAbsentNotClean() {
|
||||
MemberCredentialPolicyView view = MemberCredentialPolicyView.of(null);
|
||||
|
||||
assertFalse(view.present(), "a null policy must be reported as absent");
|
||||
assertEquals(0, view.knownCount());
|
||||
assertEquals(0, view.allowedCount());
|
||||
assertEquals(0, view.blockedCount());
|
||||
assertTrue(view.known().isEmpty());
|
||||
assertTrue(view.allowed().isEmpty());
|
||||
assertTrue(view.blocked().isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void emptyKnownListIsAbsentEvenWithAPolicyModeSet() {
|
||||
// A memberCredentials: block can be present in YAML with policy: set but known: empty —
|
||||
// that must still read as "no policy configured", the same as a fully absent block,
|
||||
// because zero known names means the daemon blocks nothing either way.
|
||||
FleetConfig.MemberCredentials creds =
|
||||
new FleetConfig.MemberCredentials("deny-by-default", List.of(), List.of());
|
||||
|
||||
MemberCredentialPolicyView view = MemberCredentialPolicyView.of(creds);
|
||||
|
||||
assertFalse(view.present());
|
||||
assertEquals(0, view.knownCount());
|
||||
}
|
||||
|
||||
@Test
|
||||
void countsMatchARealPolicyExactly() {
|
||||
FleetConfig.MemberCredentials creds = new FleetConfig.MemberCredentials(
|
||||
"deny-by-default",
|
||||
List.of("AI_GATEWAY_TOKEN"),
|
||||
List.of("AI_GATEWAY_TOKEN", "GITEA_ACCESS_TOKEN", "WORKER_GITEA_TOKEN"));
|
||||
|
||||
MemberCredentialPolicyView view = MemberCredentialPolicyView.of(creds);
|
||||
|
||||
assertTrue(view.present());
|
||||
assertEquals("deny-by-default", view.policy());
|
||||
assertEquals(List.of("AI_GATEWAY_TOKEN", "GITEA_ACCESS_TOKEN", "WORKER_GITEA_TOKEN"), view.known());
|
||||
assertEquals(List.of("AI_GATEWAY_TOKEN"), view.allowed());
|
||||
assertEquals(3, view.knownCount());
|
||||
assertEquals(1, view.allowedCount());
|
||||
// known minus allowed — the two names actually shadowed on a spawn.
|
||||
assertEquals(2, view.blockedCount());
|
||||
assertTrue(view.blocked().containsAll(List.of("GITEA_ACCESS_TOKEN", "WORKER_GITEA_TOKEN")));
|
||||
}
|
||||
|
||||
@Test
|
||||
void presentPolicyThatBlocksNothingIsStillDistinctFromAbsent() {
|
||||
// known == allow => blockedCount is 0, exactly like an absent policy's blockedCount — the
|
||||
// two must still be told apart by `present`, or a reader cannot tell "policy configured,
|
||||
// nothing currently blocked" from "no policy at all".
|
||||
FleetConfig.MemberCredentials creds = new FleetConfig.MemberCredentials(
|
||||
"deny-by-default", List.of("X"), List.of("X"));
|
||||
|
||||
MemberCredentialPolicyView view = MemberCredentialPolicyView.of(creds);
|
||||
|
||||
assertTrue(view.present());
|
||||
assertEquals(1, view.knownCount());
|
||||
assertEquals(0, view.blockedCount());
|
||||
assertFalse(MemberCredentialPolicyView.absent().present());
|
||||
}
|
||||
|
||||
@Test
|
||||
void namesPassThroughUnchangedNeverAValue() {
|
||||
// The view is built only from FleetConfig.MemberCredentials, which itself carries names,
|
||||
// never values (see its javadoc) — so there is no code path here that could substitute a
|
||||
// secret's value for its name. This pins the identity: what goes into `known`/`allow` is
|
||||
// exactly what comes out, character for character.
|
||||
List<String> known = List.of("SOME_TOKEN_NAME", "ANOTHER_NAME");
|
||||
FleetConfig.MemberCredentials creds =
|
||||
new FleetConfig.MemberCredentials("deny-by-default", List.of(), known);
|
||||
|
||||
MemberCredentialPolicyView view = MemberCredentialPolicyView.of(creds);
|
||||
|
||||
assertEquals(known, view.known());
|
||||
}
|
||||
}
|
||||
@@ -109,11 +109,11 @@ class MemberEnvAllowListTest {
|
||||
/**
|
||||
* {@code SSH_AUTH_SOCK} is a live handle to the operator's ssh-agent, never a value — so it must
|
||||
* stay excluded from the derived set even when the operator lists it under {@code allow:} for an
|
||||
* unrelated reason. It is governed ONLY by {@code memberCredentials.sshAuthSock}, applied
|
||||
* unrelated reason. It is governed ONLY by {@code memberCredentials.sshAgentEnv}, applied
|
||||
* separately by the caller ({@code HerdrPeerLauncher}).
|
||||
*/
|
||||
@Test
|
||||
void sshAuthSockInMemberCredentialsAllowIsStillExcluded() {
|
||||
void sshAgentEnvInMemberCredentialsAllowIsStillExcluded() {
|
||||
Set<String> derived = MemberEnvAllowList.derive(List.of(), Set.of("SSH_AUTH_SOCK", "OTHER_NAME"));
|
||||
|
||||
assertFalse(derived.contains("SSH_AUTH_SOCK"),
|
||||
@@ -121,6 +121,34 @@ class MemberEnvAllowListTest {
|
||||
assertTrue(derived.contains("OTHER_NAME"), "other allow: names are unaffected");
|
||||
}
|
||||
|
||||
@Test
|
||||
void configuredBrokerAndCoordinatorUriEnvNamesAreExcludedEvenWhenAllowed() {
|
||||
FleetConfig config = config("BROKER_CONNECTION_URI", "COORDINATOR_CONNECTION_URI");
|
||||
Set<String> excluded = MemberEnvAllowList.brokerUriEnvNames(config);
|
||||
|
||||
Set<String> derived = MemberEnvAllowList.derive(List.of(),
|
||||
Set.of("BROKER_CONNECTION_URI", "COORDINATOR_CONNECTION_URI", "OTHER_NAME"), excluded);
|
||||
|
||||
assertFalse(derived.contains("BROKER_CONNECTION_URI"),
|
||||
"broker.uriEnv is secret-bearing and must not ride in on allow:");
|
||||
assertFalse(derived.contains("COORDINATOR_CONNECTION_URI"),
|
||||
"coordinator.uriEnv has the same inline-password shape");
|
||||
assertTrue(derived.contains("OTHER_NAME"), "unrelated allow: entries are unaffected");
|
||||
}
|
||||
|
||||
@Test
|
||||
void absentOrBlankBrokerUriEnvAddsNoExclusions() {
|
||||
assertTrue(MemberEnvAllowList.brokerUriEnvNames(config(null, null)).isEmpty());
|
||||
assertTrue(MemberEnvAllowList.brokerUriEnvNames(config(" ", "")).isEmpty());
|
||||
}
|
||||
|
||||
private static FleetConfig config(String brokerUriEnv, String coordinatorUriEnv) {
|
||||
return new FleetConfig(null, null, null, Map.of(), null, null, null, null, null,
|
||||
new FleetConfig.Broker(null, brokerUriEnv, null), null, null, null, null, null,
|
||||
null, null, null, null,
|
||||
new FleetConfig.Coordinator(null, coordinatorUriEnv, null, null)).withDefaults();
|
||||
}
|
||||
|
||||
/** {@code LC_*} categories are infrastructure by prefix; everything else needs an exact match. */
|
||||
@Test
|
||||
void keepsMatchesExactlyPlusTheLocalePrefixRule() {
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -5,68 +5,100 @@ import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.attribute.FileTime;
|
||||
import java.sql.Connection;
|
||||
import java.sql.DriverManager;
|
||||
import java.sql.PreparedStatement;
|
||||
import java.sql.SQLException;
|
||||
import java.sql.Statement;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* {@link OpenCodeSessionDiscovery} matches an opencode session record by the worker's cwd (its
|
||||
* {@code directory}) against opencode's on-disk storage. These tests populate a TEMP storage root
|
||||
* themselves — never the operator's real {@code ~/.local/share/opencode}.
|
||||
* {@link OpenCodeSessionDiscovery} matches an opencode session row by the worker's cwd (its
|
||||
* {@code directory}) against opencode's {@code opencode.db} SQLite database. These tests build a
|
||||
* SYNTHETIC database themselves, in a JUnit temp directory — never the operator's real
|
||||
* {@code ~/.local/share/opencode/opencode.db}, which a live opencode process may be writing.
|
||||
*/
|
||||
class OpenCodeSessionDiscoveryTest {
|
||||
|
||||
/**
|
||||
* Write a session record {@code {"id":..., "directory":...}} under
|
||||
* {@code <root>/session/<projectID>/<fileName>} and stamp it with a known last-modified time,
|
||||
* so "most recently modified wins" is deterministic. Static so the launcher test can reuse it.
|
||||
* Create {@code <root>/opencode.db} with a minimal {@code session} table (just the columns
|
||||
* {@link OpenCodeSessionDiscovery} reads: {@code id}, {@code directory}, {@code time_updated})
|
||||
* and insert one row. Static so {@link OpenCodeLauncherTest} can reuse it.
|
||||
*/
|
||||
static void writeRecord(Path root, String projectId, String fileName, String id,
|
||||
String directory, long lastModifiedEpochMillis) throws Exception {
|
||||
Path dir = root.resolve("session").resolve(projectId);
|
||||
Files.createDirectories(dir);
|
||||
Path file = dir.resolve(fileName);
|
||||
Files.writeString(file, "{\"id\":\"" + id + "\",\"directory\":\"" + directory
|
||||
+ "\",\"projectID\":\"" + projectId + "\",\"version\":\"1.1.31\"}");
|
||||
Files.setLastModifiedTime(file, FileTime.fromMillis(lastModifiedEpochMillis));
|
||||
static void writeRecord(Path root, String id, String directory, long timeUpdated) throws Exception {
|
||||
writeRecord(root, id, directory, timeUpdated, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Same as {@link #writeRecord(Path, String, String, long)}, plus the {@code model} column
|
||||
* fleetd #175 reads — the raw JSON opencode writes, e.g.
|
||||
* {@code {"id":"gpt-5.6-terra","providerID":"openai"}}. {@code modelJson} may be {@code null}.
|
||||
*/
|
||||
static void writeRecord(Path root, String id, String directory, long timeUpdated, String modelJson)
|
||||
throws Exception {
|
||||
Path db = root.resolve("opencode.db");
|
||||
try (Connection connection = DriverManager.getConnection("jdbc:sqlite:" + db)) {
|
||||
try (Statement statement = connection.createStatement()) {
|
||||
statement.execute("CREATE TABLE IF NOT EXISTS session ("
|
||||
+ "id TEXT PRIMARY KEY, directory TEXT, time_updated INTEGER, model TEXT)");
|
||||
}
|
||||
// Bound parameters, not string interpolation: the class under test uses a
|
||||
// PreparedStatement, and a hand-escaped INSERT here is a pattern someone copies out.
|
||||
try (PreparedStatement insert = connection.prepareStatement(
|
||||
"INSERT INTO session (id, directory, time_updated, model) VALUES (?, ?, ?, ?)")) {
|
||||
insert.setString(1, id);
|
||||
insert.setString(2, directory);
|
||||
insert.setLong(3, timeUpdated);
|
||||
insert.setString(4, modelJson);
|
||||
insert.executeUpdate();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void findsTheRecordWhoseDirectoryEqualsTheCwd(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "ses_a.json", "ses_aaa", "/w/a", 1000L);
|
||||
writeRecord(root, "p2", "ses_b.json", "ses_bbb", "/w/b", 2000L);
|
||||
void findsTheRowWhoseDirectoryEqualsTheCwd(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L);
|
||||
writeRecord(root, "ses_bbb", "/w/b", 2000L);
|
||||
|
||||
assertEquals("ses_bbb", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/b"),
|
||||
"the record whose directory equals the cwd is the one found");
|
||||
"the row whose directory equals the cwd is the one found");
|
||||
assertEquals("ses_aaa", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonMatchingDirectoryYieldsNullRatherThanAMismatch(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "ses_a.json", "ses_aaa", "/w/a", 1000L);
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L);
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/other"),
|
||||
"no record for this cwd yet → null, not a wrong session");
|
||||
"no row for this cwd yet → null, not a wrong session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void prefersTheMostRecentlyModifiedRecordWhenSeveralMatch(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "old.json", "ses_old", "/w/a", 1000L);
|
||||
writeRecord(root, "p2", "new.json", "ses_new", "/w/a", 5000L);
|
||||
void prefersTheMostRecentlyUpdatedRowWhenSeveralMatch(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_old", "/w/a", 1000L);
|
||||
writeRecord(root, "ses_new", "/w/a", 5000L);
|
||||
|
||||
assertEquals("ses_new", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"),
|
||||
"the freshest record for the cwd wins");
|
||||
"the row with the highest time_updated for the cwd wins");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMissingOrEmptyStorageRootYieldsNullWithoutThrowing(@TempDir Path root) throws Exception {
|
||||
// Missing: no session dir at all under the root.
|
||||
void aMissingDatabaseYieldsNullWithoutThrowing(@TempDir Path root) {
|
||||
// No opencode.db at all under the root.
|
||||
assertNull(new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"));
|
||||
}
|
||||
|
||||
// Present but empty: a session dir with nothing in it produces no match, not a throw.
|
||||
Path emptyRoot = root.resolve("empty");
|
||||
Files.createDirectories(emptyRoot.resolve("session"));
|
||||
assertNull(new OpenCodeSessionDiscovery(emptyRoot).sessionIdForDirectory("/w/a"));
|
||||
@Test
|
||||
void anEmptyDatabaseYieldsNullWithoutThrowing(@TempDir Path root) throws Exception {
|
||||
Path db = root.resolve("opencode.db");
|
||||
try (Connection connection = DriverManager.getConnection("jdbc:sqlite:" + db);
|
||||
Statement statement = connection.createStatement()) {
|
||||
statement.execute("CREATE TABLE session (id TEXT PRIMARY KEY, directory TEXT, "
|
||||
+ "time_updated INTEGER)");
|
||||
}
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -76,15 +108,144 @@ class OpenCodeSessionDiscoveryTest {
|
||||
assertNull(discovery.sessionIdForDirectory(" "));
|
||||
}
|
||||
|
||||
/**
|
||||
* The one line standing between fleetd and writing the operator's live {@code opencode.db} —
|
||||
* 841MB, with a running opencode writing it — is {@code config.setReadOnly(true)} in
|
||||
* {@link OpenCodeSessionDiscovery#openReadOnly()}. Delete it and every other test in this class
|
||||
* still passes, so this is the test that guards it.
|
||||
*
|
||||
* <p>It asks the connection to write, and requires a refusal. The obvious alternative — make
|
||||
* the database file unwritable and check the read still works — proves nothing: SQLite silently
|
||||
* downgrades a read-write open of an unwritable file to read-only, so that test passes either
|
||||
* way. It was tried and watched pass with the flag removed.
|
||||
*/
|
||||
@Test
|
||||
void aMalformedRecordIsSkippedRatherThanFatal(@TempDir Path root) throws Exception {
|
||||
// A record that fails to parse must not abort the scan of its siblings.
|
||||
Path dir = root.resolve("session").resolve("p1");
|
||||
Files.createDirectories(dir);
|
||||
Files.writeString(dir.resolve("broken.json"), "{not valid json");
|
||||
writeRecord(root, "p1", "good.json", "ses_good", "/w/a", 1000L);
|
||||
void theDatabaseIsOpenedReadOnly(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L);
|
||||
|
||||
assertEquals("ses_good", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"),
|
||||
"an unreadable record is skipped; a later valid one still matches");
|
||||
try (Connection connection = new OpenCodeSessionDiscovery(root).openReadOnly();
|
||||
Statement statement = connection.createStatement()) {
|
||||
SQLException refused = assertThrows(SQLException.class,
|
||||
() -> statement.executeUpdate("INSERT INTO session (id, directory, time_updated) "
|
||||
+ "VALUES ('ses_zzz', '/w/z', 1)"),
|
||||
"the connection must REFUSE a write — opencode is writing this database live");
|
||||
assertTrue(refused.getMessage().toLowerCase().contains("readonly")
|
||||
|| refused.getMessage().toLowerCase().contains("read-only"),
|
||||
"the refusal must be about read-only, not some other error: " + refused.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aCorruptDatabaseFileYieldsNullWithoutThrowing(@TempDir Path root) throws Exception {
|
||||
// A file at opencode.db that is not a SQLite database at all — the open/query must fail
|
||||
// safe, never fatal to a spawn.
|
||||
Path db = root.resolve("opencode.db");
|
||||
Files.writeString(db, "this is not a sqlite database");
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"),
|
||||
"an unreadable database resolves to null, not an exception");
|
||||
}
|
||||
|
||||
// --- fleetd #175/#234: actualModelForSessionId / the model JSON column -----------------------
|
||||
|
||||
@Test
|
||||
void parsesTheModelJsonIntoProviderAndId(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L, "{\"id\":\"gpt-5.6-terra\",\"providerID\":\"openai\"}");
|
||||
|
||||
OpenCodeSessionDiscovery.ActualModel actual =
|
||||
new OpenCodeSessionDiscovery(root).actualModelForSessionId("ses_aaa");
|
||||
|
||||
assertNotNull(actual, "a well-formed model JSON parses");
|
||||
assertEquals("gpt-5.6-terra", actual.id());
|
||||
assertEquals("openai", actual.provider());
|
||||
}
|
||||
|
||||
@Test
|
||||
void looksUpByIdEvenWhenAnotherRowInTheSameDirectoryIsNewer(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_ours", "/work/dir", 1000L, "{\"id\":\"gpt-5.6-terra\",\"providerID\":\"openai\"}");
|
||||
writeRecord(root, "ses_sibling", "/work/dir", 9000L, "{\"id\":\"deepseek-v4-flash\",\"providerID\":\"gx\"}");
|
||||
|
||||
OpenCodeSessionDiscovery discovery = new OpenCodeSessionDiscovery(root);
|
||||
assertEquals("gpt-5.6-terra", discovery.actualModelForSessionId("ses_ours").id(),
|
||||
"querying by id reads OUR row, not the directory's newest row");
|
||||
assertEquals("deepseek-v4-flash", discovery.actualModelForSessionId("ses_sibling").id(),
|
||||
"each id resolves to its own row independently of time_updated ordering");
|
||||
}
|
||||
|
||||
@Test
|
||||
void anUnknownSessionIdYieldsUnknownModelRatherThanAMismatch(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L, "{\"id\":\"x\",\"providerID\":\"y\"}");
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).actualModelForSessionId("ses_no_such_row"),
|
||||
"no row for this id yet → unknown, not a wrong model");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBlankOrNullSessionIdYieldsUnknownModel(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L, "{\"id\":\"x\",\"providerID\":\"y\"}");
|
||||
|
||||
OpenCodeSessionDiscovery discovery = new OpenCodeSessionDiscovery(root);
|
||||
assertNull(discovery.actualModelForSessionId(null));
|
||||
assertNull(discovery.actualModelForSessionId(" "));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNullModelColumnYieldsUnknownWithoutThrowing(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L, null);
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).actualModelForSessionId("ses_aaa"),
|
||||
"a row with no model value yet is unknown, not a mismatch");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unparseableModelJsonYieldsUnknownWithoutThrowing(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L, "this is not json");
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).actualModelForSessionId("ses_aaa"),
|
||||
"JSON that fails to parse resolves to unknown, never an exception");
|
||||
}
|
||||
|
||||
@Test
|
||||
void modelJsonMissingIdYieldsUnknown(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L, "{\"providerID\":\"openai\"}");
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).actualModelForSessionId("ses_aaa"),
|
||||
"no id in the JSON → unknown, since id is what a caller actually compares");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMissingDatabaseYieldsUnknownModelWithoutThrowing(@TempDir Path root) {
|
||||
assertNull(new OpenCodeSessionDiscovery(root).actualModelForSessionId("ses_aaa"));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #175 site 2: a schema that predates the {@code model} column entirely — the exact
|
||||
* shape a real opencode upgrade/downgrade could produce. This must degrade to UNKNOWN for the
|
||||
* model, and — the property that actually matters — must NOT take id resolution down with it.
|
||||
* A combined single query for both columns would fail this test; that is why
|
||||
* {@link OpenCodeSessionDiscovery#actualModelForSessionId} runs its own independent query.
|
||||
*/
|
||||
@Test
|
||||
void aMissingModelColumnYieldsUnknownButIdResolutionStillWorks(@TempDir Path root) throws Exception {
|
||||
Path db = root.resolve("opencode.db");
|
||||
try (Connection connection = DriverManager.getConnection("jdbc:sqlite:" + db)) {
|
||||
try (Statement statement = connection.createStatement()) {
|
||||
statement.execute("CREATE TABLE session (id TEXT PRIMARY KEY, directory TEXT, "
|
||||
+ "time_updated INTEGER)");
|
||||
}
|
||||
try (PreparedStatement insert = connection.prepareStatement(
|
||||
"INSERT INTO session (id, directory, time_updated) VALUES (?, ?, ?)")) {
|
||||
insert.setString(1, "ses_aaa");
|
||||
insert.setString(2, "/w/a");
|
||||
insert.setLong(3, 1000L);
|
||||
insert.executeUpdate();
|
||||
}
|
||||
}
|
||||
|
||||
OpenCodeSessionDiscovery discovery = new OpenCodeSessionDiscovery(root);
|
||||
assertEquals("ses_aaa", discovery.sessionIdForDirectory("/w/a"),
|
||||
"id resolution must survive a database with no model column at all");
|
||||
assertNull(discovery.actualModelForSessionId("ses_aaa"),
|
||||
"no model column → unknown, not a throw and not a mismatch");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -154,7 +154,7 @@ class AmqpReplyInboxContractTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void releaseCancelsConsumerAndClearsHeld() throws Exception {
|
||||
void releaseCancelsConsumerAndRequeuesHeldDeliveryForRecovery() throws Exception {
|
||||
String target = "worker-release-" + System.nanoTime();
|
||||
try (AmqpReplyInbox inbox = AmqpReplyInbox.open(uri())) {
|
||||
inbox.own(target);
|
||||
@@ -164,6 +164,22 @@ class AmqpReplyInboxContractTest {
|
||||
inbox.release(target);
|
||||
assertTrue(inbox.peek(target).isEmpty(),
|
||||
"release clears the local held snapshot");
|
||||
|
||||
// fleetd #298: release() must not just drop the local record — the broker delivery was
|
||||
// never acked, so cancelling the consumer alone leaves it unacked-but-orphaned on the
|
||||
// still-open channel unless release() nacks it back with requeue=true. Prove the message
|
||||
// is genuinely recoverable, not merely absent from peek: re-own the same target and
|
||||
// confirm the broker redelivers it to the fresh consumer.
|
||||
inbox.own(target);
|
||||
List<ReplyInbox.InboxMessage> recovered = awaitPeek(inbox, target);
|
||||
assertEquals(1, recovered.size(),
|
||||
"a reply held (but undrained) at release() time must still be recoverable — "
|
||||
+ "release() must requeue it, not silently drop it while the broker still "
|
||||
+ "considers it outstanding");
|
||||
assertEquals("m1", recovered.getFirst().msgId());
|
||||
assertEquals("release me", recovered.getFirst().content());
|
||||
|
||||
inbox.ack(target, "m1");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,264 @@
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import com.rabbitmq.client.AMQP;
|
||||
import com.rabbitmq.client.Channel;
|
||||
import com.rabbitmq.client.Connection;
|
||||
import com.rabbitmq.client.DeliverCallback;
|
||||
import com.rabbitmq.client.Delivery;
|
||||
import com.rabbitmq.client.Envelope;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.Timeout;
|
||||
|
||||
import java.lang.reflect.InvocationHandler;
|
||||
import java.lang.reflect.Proxy;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.time.Duration;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-318: a delivery landing on the consumer work-pool thread <em>after</em>
|
||||
* {@link AmqpReplyInbox#release} has already swapped the target's {@code held} entry for its
|
||||
* tombstone, but <em>before</em> {@code release()} itself returns, must be nacked-with-requeue —
|
||||
* never silently retained in a fresh map {@code release()} has already stopped looking at.
|
||||
*
|
||||
* <p><strong>This forces the actual interleaving, not a sequence of calls.</strong> {@code release()}
|
||||
* runs on its own thread and is made to block <em>inside</em> its nack loop, via a fake
|
||||
* {@link Channel} whose {@code basicNack} blocks on its first invocation. That block is only
|
||||
* reachable after {@code release()}'s {@code held.compute(...)} has already swapped in the
|
||||
* {@code RELEASED} tombstone (the compute call happens strictly before the loop that calls
|
||||
* {@code basicNack}), so observing it is direct, ordering-guaranteed proof that the tombstone is in
|
||||
* place and {@code release()} has not yet returned — still holding {@code channelLock} — when a
|
||||
* second thread fires {@code own()}'s captured {@link DeliverCallback} for a brand-new message on the
|
||||
* same target. No mocking library is on the classpath, so the fake broker is a {@link Proxy}, the
|
||||
* same pattern {@code AmqpReplyInboxRecoveryRaceTest} already uses.
|
||||
*
|
||||
* <p><strong>What this does and does not prove.</strong> It proves that a delivery whose
|
||||
* {@code computeIfAbsent} call is ordered strictly after {@code release()}'s tombstone swap — while
|
||||
* {@code release()} is still running — is nacked-with-requeue rather than silently parked forever.
|
||||
* It does not drive a real broker: {@code basicNack} here is a recorded call on a fake channel, not a
|
||||
* verified requeue-and-redeliver. That half of the contract (a nacked-with-requeue delivery really
|
||||
* does come back to a later owner) is already covered against a real broker by
|
||||
* {@code AmqpReplyInboxContractTest.releaseCancelsConsumerAndRequeuesHeldDeliveryForRecovery}, which
|
||||
* this test does not duplicate.
|
||||
*/
|
||||
class AmqpReplyInboxReleaseRaceTest {
|
||||
|
||||
@Test
|
||||
@Timeout(15)
|
||||
void deliveryArrivingWhileReleaseIsStillRunningIsNackedNotStranded() throws Exception {
|
||||
String target = "worker-release-race";
|
||||
|
||||
List<long[]> nacks = new CopyOnWriteArrayList<>(); // {deliveryTag, requeue(1/0)}
|
||||
List<Long> acks = new CopyOnWriteArrayList<>();
|
||||
AtomicReference<DeliverCallback> deliverCallback = new AtomicReference<>();
|
||||
CountDownLatch nackStarted = new CountDownLatch(1);
|
||||
CountDownLatch releaseMayFinishNack = new CountDownLatch(1);
|
||||
AtomicInteger nackCallCount = new AtomicInteger();
|
||||
|
||||
Channel consumeChannel = fakeConsumeChannel(deliverCallback, nacks, acks, nackCallCount,
|
||||
nackStarted, releaseMayFinishNack);
|
||||
Channel publishChannel = fakeInertChannel();
|
||||
Connection connection = fakeConnection(consumeChannel, publishChannel);
|
||||
|
||||
AmqpReplyInbox inbox = new AmqpReplyInbox(connection, AmqpReplyInbox.DEFAULT_PREFETCH);
|
||||
inbox.own(target);
|
||||
assertTrue(deliverCallback.get() != null, "own() must have registered a DeliverCallback");
|
||||
|
||||
// Seed one already-held delivery (m0) so release()'s nack loop has something to iterate, and
|
||||
// therefore somewhere to block, before it can return.
|
||||
deliverCallback.get().handle("ctag", delivery(1L, "m0", "first"));
|
||||
|
||||
AtomicReference<Throwable> releaseError = new AtomicReference<>();
|
||||
Thread releaseThread = new Thread(() -> {
|
||||
try {
|
||||
inbox.release(target);
|
||||
} catch (Throwable t) {
|
||||
releaseError.set(t);
|
||||
}
|
||||
}, "release-under-test");
|
||||
releaseThread.start();
|
||||
|
||||
// This latch only fires from inside the fake channel's basicNack — i.e. from inside
|
||||
// release()'s nack loop, which release()'s code only reaches AFTER held.compute(...) has
|
||||
// already swapped in RELEASED. Waiting for it is direct proof the swap has happened and
|
||||
// release() has not yet returned (it is stuck mid-loop, still holding channelLock).
|
||||
assertTrue(nackStarted.await(10, TimeUnit.SECONDS),
|
||||
"release() never reached its nack loop — it may not have started");
|
||||
|
||||
// The exact interleaving CB-318 describes: a delivery for a NEW message on the same target
|
||||
// lands on the "consumer work-pool thread" (this second thread) while release() is still
|
||||
// running. With the pre-fix code (a bare held.remove(target)) this created a brand-new map
|
||||
// under computeIfAbsent that release() — already past its remove — never looks at again.
|
||||
AtomicReference<Throwable> deliveryError = new AtomicReference<>();
|
||||
Thread deliveryThread = new Thread(() -> {
|
||||
try {
|
||||
deliverCallback.get().handle("ctag", delivery(2L, "m1", "second"));
|
||||
} catch (Throwable t) {
|
||||
deliveryError.set(t);
|
||||
}
|
||||
}, "concurrent-delivery");
|
||||
deliveryThread.start();
|
||||
|
||||
// Head start for the delivery thread to reach (and, on the fixed code, block on)
|
||||
// channelLock — release() still holds it at this point, so a correct fix cannot have
|
||||
// resolved m1's nack yet. Purely in-memory work (computeIfAbsent, a reference compare)
|
||||
// separates deliveryThread.start() from that block point, so 300ms is a large margin, not a
|
||||
// tight timing assumption.
|
||||
Thread.sleep(300);
|
||||
assertEquals(1, nackCallCount.get(),
|
||||
"the concurrent delivery must not resolve its nack before release() gives up "
|
||||
+ "channelLock — if this is 2 already, the interleaving below is not being "
|
||||
+ "tested, only a sequential call");
|
||||
|
||||
releaseMayFinishNack.countDown(); // let release() finish nacking m0 and return
|
||||
assertTrue(releaseThread.join(Duration.ofSeconds(10)), "release() did not finish");
|
||||
assertTrue(deliveryThread.join(Duration.ofSeconds(10)), "the concurrent delivery did not finish");
|
||||
|
||||
assertNull(releaseError.get(), "release() threw: " + releaseError.get());
|
||||
assertNull(deliveryError.get(), "the concurrent delivery threw: " + deliveryError.get());
|
||||
|
||||
assertEquals(2, nacks.size(),
|
||||
"both the pre-held m0 and the concurrently-arriving m1 must be nacked, got: "
|
||||
+ nacks.stream().map(n -> "[tag=" + n[0] + " requeue=" + n[1] + "]").toList());
|
||||
assertTrue(nacks.stream().allMatch(n -> n[1] == 1L),
|
||||
"invariant 1 (never drop): every nack must set requeue=true");
|
||||
assertTrue(nacks.stream().anyMatch(n -> n[0] == 1L), "m0's delivery tag must be nacked");
|
||||
assertTrue(nacks.stream().anyMatch(n -> n[0] == 2L),
|
||||
"m1 — delivered while release() was still running, after the tombstone swap — must be "
|
||||
+ "nacked, not silently retained in a map release() will never look at again");
|
||||
assertTrue(acks.isEmpty(), "invariant 1 (never drop): a held reply must never be basicAck'd");
|
||||
}
|
||||
|
||||
private static Delivery delivery(long tag, String msgId, String body) {
|
||||
Envelope envelope = new Envelope(tag, false, "", "irrelevant");
|
||||
AMQP.BasicProperties props = new AMQP.BasicProperties.Builder().messageId(msgId).build();
|
||||
return new Delivery(envelope, props, body.getBytes(StandardCharsets.UTF_8));
|
||||
}
|
||||
|
||||
/** A {@link Proxy}-backed consume {@link Channel}: blocks the FIRST {@code basicNack} call on
|
||||
* {@code releaseMayFinishNack}, after signalling {@code nackStarted} — everything else records
|
||||
* the call and returns a harmless default, matching the style already used by
|
||||
* {@code AmqpReplyInboxRecoveryRaceTest}. */
|
||||
private static Channel fakeConsumeChannel(AtomicReference<DeliverCallback> deliverCallback,
|
||||
List<long[]> nacks, List<Long> acks,
|
||||
AtomicInteger nackCallCount,
|
||||
CountDownLatch nackStarted,
|
||||
CountDownLatch releaseMayFinishNack) {
|
||||
InvocationHandler handler = (proxy, method, args) -> {
|
||||
String name = method.getName();
|
||||
if (name.equals("basicConsume")) {
|
||||
deliverCallback.set((DeliverCallback) args[2]);
|
||||
return "ctag";
|
||||
}
|
||||
if (name.equals("basicNack")) {
|
||||
long tag = (long) args[0];
|
||||
boolean requeue = (boolean) args[2];
|
||||
if (nackCallCount.incrementAndGet() == 1) {
|
||||
nackStarted.countDown();
|
||||
if (!releaseMayFinishNack.await(10, TimeUnit.SECONDS)) {
|
||||
throw new IllegalStateException("test never released the nack latch");
|
||||
}
|
||||
}
|
||||
nacks.add(new long[] {tag, requeue ? 1L : 0L});
|
||||
return null;
|
||||
}
|
||||
if (name.equals("basicAck")) {
|
||||
acks.add((long) args[0]);
|
||||
return null;
|
||||
}
|
||||
if (name.equals("equals")) {
|
||||
return proxy == args[0];
|
||||
}
|
||||
if (name.equals("hashCode")) {
|
||||
return System.identityHashCode(proxy);
|
||||
}
|
||||
if (name.equals("toString")) {
|
||||
return "FakeConsumeChannel";
|
||||
}
|
||||
return defaultValue(method.getReturnType());
|
||||
};
|
||||
return (Channel) Proxy.newProxyInstance(AmqpReplyInboxReleaseRaceTest.class.getClassLoader(),
|
||||
new Class<?>[] {Channel.class}, handler);
|
||||
}
|
||||
|
||||
/** A {@link Proxy}-backed {@link Channel} that answers every call with a harmless default — used
|
||||
* as the publish channel, which this test never actually publishes on. */
|
||||
private static Channel fakeInertChannel() {
|
||||
InvocationHandler handler = (proxy, method, args) -> {
|
||||
String name = method.getName();
|
||||
if (name.equals("equals")) {
|
||||
return proxy == args[0];
|
||||
}
|
||||
if (name.equals("hashCode")) {
|
||||
return System.identityHashCode(proxy);
|
||||
}
|
||||
if (name.equals("toString")) {
|
||||
return "FakeInertChannel";
|
||||
}
|
||||
return defaultValue(method.getReturnType());
|
||||
};
|
||||
return (Channel) Proxy.newProxyInstance(AmqpReplyInboxReleaseRaceTest.class.getClassLoader(),
|
||||
new Class<?>[] {Channel.class}, handler);
|
||||
}
|
||||
|
||||
/** A {@link Proxy}-backed {@link Connection} handing out {@code first} then {@code second} from
|
||||
* successive {@code createChannel()} calls, matching {@link AmqpReplyInbox}'s constructor. */
|
||||
private static Connection fakeConnection(Channel first, Channel second) {
|
||||
AtomicInteger calls = new AtomicInteger();
|
||||
InvocationHandler handler = (proxy, method, args) -> {
|
||||
String name = method.getName();
|
||||
if (name.equals("createChannel") && (args == null || args.length == 0)) {
|
||||
return calls.getAndIncrement() == 0 ? first : second;
|
||||
}
|
||||
if (name.equals("equals")) {
|
||||
return proxy == args[0];
|
||||
}
|
||||
if (name.equals("hashCode")) {
|
||||
return System.identityHashCode(proxy);
|
||||
}
|
||||
if (name.equals("toString")) {
|
||||
return "FakeConnection";
|
||||
}
|
||||
return defaultValue(method.getReturnType());
|
||||
};
|
||||
return (Connection) Proxy.newProxyInstance(AmqpReplyInboxReleaseRaceTest.class.getClassLoader(),
|
||||
new Class<?>[] {Connection.class}, handler);
|
||||
}
|
||||
|
||||
private static Object defaultValue(Class<?> type) {
|
||||
if (!type.isPrimitive() || type == void.class) {
|
||||
return null;
|
||||
}
|
||||
if (type == boolean.class) {
|
||||
return Boolean.FALSE;
|
||||
}
|
||||
if (type == long.class) {
|
||||
return 0L;
|
||||
}
|
||||
if (type == short.class) {
|
||||
return (short) 0;
|
||||
}
|
||||
if (type == byte.class) {
|
||||
return (byte) 0;
|
||||
}
|
||||
if (type == char.class) {
|
||||
return (char) 0;
|
||||
}
|
||||
if (type == double.class) {
|
||||
return 0.0d;
|
||||
}
|
||||
if (type == float.class) {
|
||||
return 0.0f;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
@@ -56,7 +56,11 @@ class MessageServiceTest {
|
||||
|
||||
/** Run {@code send} on a background thread; the current thread drives the worker's turn. */
|
||||
private CompletableFuture<MessageService.Reply> sendAsync() {
|
||||
return CompletableFuture.supplyAsync(() -> messages.send(T, "do the task", 5000));
|
||||
return sendAsync("do the task");
|
||||
}
|
||||
|
||||
private CompletableFuture<MessageService.Reply> sendAsync(String content) {
|
||||
return CompletableFuture.supplyAsync(() -> messages.send(T, content, 5000));
|
||||
}
|
||||
|
||||
private void awaitWaiting() throws InterruptedException {
|
||||
@@ -86,6 +90,116 @@ class MessageServiceTest {
|
||||
assertTrue(reply.completed(), "a scraped completion still counts as completed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void completionFallbackReplacesAnEchoedInjectedBriefWithNoReportOutcome() throws Exception {
|
||||
String brief = "Implement the requested change. ".repeat(20);
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync(brief);
|
||||
awaitWaiting();
|
||||
|
||||
herdr.readText("$ prompt");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
herdr.readText("⏺ " + brief + "\n❯ ");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
|
||||
MessageService.Reply reply = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.COMPLETED_UNREPLIED, reply.outcome());
|
||||
assertEquals(CompletionResolver.NO_REPORT_PREFIX + "]", reply.text(),
|
||||
"the real injector -> completion fallback path must not return the lead's brief");
|
||||
}
|
||||
|
||||
@Test
|
||||
void completionFallbackKeepsARealReportThatRestatesTheWholeBrief() throws Exception {
|
||||
String brief = "Load the implementer skill. You own fleetd #999. ".repeat(12);
|
||||
String report = brief + "\n\n## Report\n"
|
||||
+ ("I implemented the fix in GitWorktrees.java, added five tests, ran mvn clean install "
|
||||
+ "and got 1168 tests with 0 failures. Commit 321d8dc pushed. ").repeat(8);
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync(brief);
|
||||
awaitWaiting();
|
||||
|
||||
herdr.readText("$ prompt");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
herdr.readText("⏺ " + report + "\n❯ ");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
|
||||
MessageService.Reply reply = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.COMPLETED_UNREPLIED, reply.outcome());
|
||||
assertEquals(report.strip(), reply.text(), "a report that restates the whole brief must survive unchanged");
|
||||
}
|
||||
|
||||
@Test
|
||||
void completionFallbackNamesTheKnownWorktreeAndBranchForAnEchoedBrief() throws Exception {
|
||||
FakeHerdr localHerdr = new FakeHerdr();
|
||||
Rendezvous localRendezvous = new Rendezvous();
|
||||
java.util.concurrent.atomic.AtomicLong localClock = new java.util.concurrent.atomic.AtomicLong();
|
||||
CompletionResolver localCompletion = new CompletionResolver(new AgentControl(localHerdr), localRendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(),
|
||||
dev.ltms.fleet.inject.BackendErrorPatternLookup.legacy(),
|
||||
dev.ltms.fleet.inject.BackendErrorSink.none(),
|
||||
() -> localClock.addAndGet(CompletionResolver.MIN_TURN_NANOS + 1),
|
||||
target -> new CompletionResolver.WorktreeBranch("/tmp/member-worktree", "worker/cb241"));
|
||||
Injector localInjector = new Injector(new AgentControl(localHerdr), localCompletion);
|
||||
MessageService localMessages = new MessageService(new AgentControl(localHerdr), localInjector,
|
||||
localRendezvous, new InMemoryReplyInbox());
|
||||
String brief = "Implement the requested change. ".repeat(20);
|
||||
CompletableFuture<MessageService.Reply> send =
|
||||
CompletableFuture.supplyAsync(() -> localMessages.send(T, brief, 5000));
|
||||
long deadline = System.currentTimeMillis() + 2000;
|
||||
while (!localRendezvous.isWaiting(T) && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(localRendezvous.isWaiting(T));
|
||||
|
||||
localHerdr.readText("$ prompt");
|
||||
localInjector.onStatus(T, AgentStatus.IDLE);
|
||||
localInjector.onStatus(T, AgentStatus.WORKING);
|
||||
localHerdr.readText("⏺ " + brief + "\n❯ ");
|
||||
localInjector.onStatus(T, AgentStatus.IDLE);
|
||||
|
||||
assertEquals(CompletionResolver.NO_REPORT_PREFIX + " worktree=/tmp/member-worktree "
|
||||
+ "branch=worker/cb241]", send.get(5, TimeUnit.SECONDS).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void completionFallbackKeepsTheClippedMarkerWhenAnEchoedBriefIsTooLong() throws Exception {
|
||||
String brief = "a".repeat(4_001); // CompletionResolver's 4,000-character scrape cap
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync(brief);
|
||||
awaitWaiting();
|
||||
|
||||
herdr.readText("$ prompt");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
herdr.readText("⏺ " + brief + "\n❯ ");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
|
||||
assertEquals(CompletionResolver.NO_REPORT_PREFIX + "]\n"
|
||||
+ "[Pane tail clipped: member did not call fleet_reply.]",
|
||||
send.get(5, TimeUnit.SECONDS).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void backendErrorScrapeThroughMessageServiceFailsInsteadOfBecomingReplyText() throws Exception {
|
||||
// fleetd#164 (part 2 addendum): a scrape that reads cleanly but is only the backend's own
|
||||
// rejection (e.g. an HTTP 400) must reach the caller as WORKER_FAILED, not as a completed
|
||||
// reply whose text happens to be the error line.
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
|
||||
herdr.readText("$ prompt");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
herdr.readText("⏺ API Error: 400 invalid request body");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
|
||||
MessageService.Reply reply = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.WORKER_FAILED, reply.outcome(),
|
||||
"a backend rejection must use the caller's failure outcome, not a completed reply");
|
||||
assertFalse(reply.completed(), "plain backend errors are never fallback reply content");
|
||||
assertTrue(reply.text().contains("API Error: 400 invalid request body"),
|
||||
"the visible backend error is carried as the failure reason: " + reply.text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitFleetReplyResolvesAsReplied() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
@@ -666,6 +780,32 @@ class MessageServiceTest {
|
||||
assertFailedTicket(third, "agent target term_a not found");
|
||||
}
|
||||
|
||||
// --- #137 follow-up: abandon() must not guess when more than one task is open ---------------
|
||||
//
|
||||
// A test combining a genuine stranded reply (hasStrandedReply(T)==true) with two simultaneously
|
||||
// open matching tasks was attempted here and removed after investigation showed the combination
|
||||
// is not reachable through the public API today, not merely hard to time right:
|
||||
//
|
||||
// This class has exactly two call sites that ever hold a target's entry in the session-lock map
|
||||
// (send() and answer()), and both open a Rendezvous waiter for that same target as the first thing
|
||||
// they do after acquiring the lock, holding lock and waiter together for their whole critical
|
||||
// section. So "the lock is held" and "a live waiter is open" are the same fact throughout this
|
||||
// class. reply()'s fast path always resolves a currently-open waiter directly instead of
|
||||
// stranding — so a strand can only be created while NO task is accepted (lock free), and the
|
||||
// instant the lock is next taken (by any parked matching task's own send(), the moment it is
|
||||
// scheduled), that acceptance clears strandedReplies again (see send()'s CB-640 comment) before
|
||||
// abandon() can ever observe both facts together. Confirmed empirically too: an earlier version of
|
||||
// this test stranded a reply, then created an "accepted" task (awaitWaiting()) followed by a
|
||||
// "parked" one — and the accepted task's own acceptance silently cleared the strand it was
|
||||
// supposed to be racing against, so the parked task came back WORKER_FAILED instead of DONE, not
|
||||
// because the fix was missing but because the test's premise could not be constructed.
|
||||
//
|
||||
// The reachable half — matching.size() >= 2 alone, no strand — is exactly what
|
||||
// abandonFailsEveryPendingAsyncTicketForTheReleasedTarget already covers (all fail, none guess).
|
||||
// The oldest-wins code in abandon() stays as defence in depth (see its own javadoc) against a
|
||||
// regression that would make the conjunction reachable, e.g. clearing strandedReplies on a
|
||||
// narrower condition than "any acceptance" — not because this suite exercises it today.
|
||||
|
||||
@Test
|
||||
void abandonDoesNotFailAnAsyncTicketWaitingForAnAnswer() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks");
|
||||
@@ -688,6 +828,118 @@ class MessageServiceTest {
|
||||
assertEquals(MessageService.Outcome.REPLIED, answer.get(5, TimeUnit.SECONDS).outcome());
|
||||
}
|
||||
|
||||
// --- fleetd #275: a target torn down FOR GOOD while genuinely ASKING must not orphan --------
|
||||
//
|
||||
// sessions.onRelease (fleet_stop, or the idle reaper) is the one abandon() caller that knows
|
||||
// for certain the target can never resume: its pane is being stopped right now. Unlike the
|
||||
// health-classification caller above (a GONE/NEVER_READY guess, not a teardown it performed),
|
||||
// it must sweep an ASKING ticket right here — see MessageService.abandon(String, String,
|
||||
// boolean)'s javadoc for the full reachability chain this closes: without this, the forward
|
||||
// waiter is already closed by the time the question surfaces, the ASKING guard skips the task,
|
||||
// and by the time the worker's own fleet_ask lapses (~55-115s later) the released session no
|
||||
// longer appears in FleetHealthMonitor's roster for anything to ever sweep it again — leaving
|
||||
// fleet_poll{ticket} stuck PENDING forever.
|
||||
|
||||
@Test
|
||||
void abandonWithSweepAskingFailsATornDownTargetsAskingTicket() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config?", 300));
|
||||
MessageService.TaskView asking = awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
|
||||
assertTrue(messages.abandon(T, "the worker session was released before it replied", true),
|
||||
"a released target's open ask can never resume, so it must fail right here");
|
||||
|
||||
MessageService.TaskView failed = awaitTicketPhase(ticket, MessageService.Phase.FAILED);
|
||||
assertEquals("the worker session was released before it replied", failed.detail());
|
||||
|
||||
// The reverse-rendezvous ask is torn down too: the worker's still-blocked fleet_ask rides
|
||||
// out its own timeout (nothing completed its answer future), and a late answer() for the
|
||||
// same turnId must see it as lapsed rather than resolving a question nobody is waiting on.
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT, ask.get(5, TimeUnit.SECONDS).outcome());
|
||||
assertEquals(MessageService.Outcome.STALE_TURN,
|
||||
messages.answer(asking.turnId(), "config.yaml", 200).outcome());
|
||||
}
|
||||
|
||||
@Test
|
||||
void abandonWithoutSweepAskingBehavesLikeTheTwoArgOverload() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config?", 5000));
|
||||
awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
|
||||
assertFalse(messages.abandon(T, "agent target term_a not found", false),
|
||||
"sweepAsking=false must match the plain abandon(target, reason) overload");
|
||||
assertEquals(MessageService.Phase.ASKING, messages.poll(ticket).phase());
|
||||
}
|
||||
|
||||
// --- #137: a fleet_ask round-trip must not orphan the ticket's own reply -------------------
|
||||
//
|
||||
// The primary's fleet_send{turnId} answer call is itself bounded (a real MCP call, capped well
|
||||
// under a minute) — far shorter than a resumed turn can genuinely take to finish real work. These
|
||||
// drive the exact real delegation path (async send -> worker asks -> primary answers -> primary's
|
||||
// own wait gives up -> worker's real fleet_reply arrives afterwards) rather than calling a reply
|
||||
// sink directly, since the bug is specifically about which sink the resumed turn's reply reaches.
|
||||
|
||||
@Test
|
||||
void aReplyAfterAnswerTimesOutStillCompletesTheAsyncTicket() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config?", 5000));
|
||||
MessageService.TaskView asking = awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
|
||||
// The primary answers, but its own bounded wait for the worker's resumed turn is short and
|
||||
// expires before the worker (still genuinely working) gets back to it.
|
||||
MessageService.Reply answerReply = messages.answer(asking.turnId(), "config.yaml", 150);
|
||||
assertEquals("config.yaml", ask.get(5, TimeUnit.SECONDS).answer());
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_WORKING, answerReply.outcome(),
|
||||
"the primary's own bounded wait gives up before the worker finishes resuming");
|
||||
|
||||
// The worker keeps working past that window and only now calls fleet_reply.
|
||||
assertTrue(messages.reply(T, "PR opened: https://example/pulls/42"));
|
||||
|
||||
MessageService.TaskView done = awaitTicketPhase(ticket, MessageService.Phase.DONE);
|
||||
assertEquals("PR opened: https://example/pulls/42", done.reply(),
|
||||
"fleet_poll{ticket} must return the worker's real reply, not stay pending forever");
|
||||
assertEquals("reply", done.replySource());
|
||||
assertFalse(messages.hasStrandedReply(T),
|
||||
"the reply completed its own ticket directly and never touched the inbox");
|
||||
}
|
||||
|
||||
@Test
|
||||
void fleetStopAfterAnOrphanedReplyDoesNotFailTheTicket() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config?", 5000));
|
||||
MessageService.TaskView asking = awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
|
||||
MessageService.Reply answerReply = messages.answer(asking.turnId(), "config.yaml", 150);
|
||||
assertEquals("config.yaml", ask.get(5, TimeUnit.SECONDS).answer());
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_WORKING, answerReply.outcome());
|
||||
|
||||
assertTrue(messages.reply(T, "PR opened: https://example/pulls/42"));
|
||||
|
||||
// fleet_stop tears the worker's session down right after the reply landed — this must never
|
||||
// report the misleading "the worker session was released before it replied": a reply is
|
||||
// exactly what happened.
|
||||
assertFalse(messages.abandon(T, "the worker session was released before it replied"),
|
||||
"a reply already arrived, so nothing here is a genuine failure");
|
||||
|
||||
MessageService.TaskView view = awaitTicketPhase(ticket, MessageService.Phase.DONE);
|
||||
assertEquals("PR opened: https://example/pulls/42", view.reply());
|
||||
}
|
||||
|
||||
@Test
|
||||
void unansweredAsyncQuestionReturnsTheTicketToPendingAndReleasesItsTarget() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks");
|
||||
@@ -705,6 +957,76 @@ class MessageServiceTest {
|
||||
assertEquals(MessageService.Phase.DONE, awaitTicketPhase(next, MessageService.Phase.DONE).phase());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #307: a worker's {@code fleet_ask} can time out because the primary never answers —
|
||||
* distinct from {@link #aReplyAfterAnswerTimesOutStillCompletesTheAsyncTicket}, where the
|
||||
* primary DID answer and only its own bounded wait for the resumed turn expired.
|
||||
* {@code ask()}'s timeout path deliberately forgets the task's {@code turnId} (so
|
||||
* {@code hasAsyncQuestion} stops reporting the target BUSY — see
|
||||
* {@code unansweredAsyncQuestionReturnsTheTicketToPendingAndReleasesItsTarget} above), which used
|
||||
* to also erase the one signal {@code askAnsweredAsyncTasks} needed to recognize the worker's
|
||||
* eventual real {@code fleet_reply}. That reply then had nowhere to land but the inbox, and
|
||||
* {@code fleet_poll{ticket}} stayed PENDING forever — later force-failed with the false reason
|
||||
* "session released before it replied", even though the worker had, in fact, replied.
|
||||
*/
|
||||
@Test
|
||||
void aReplyAfterAnAskTimeoutStillCompletesTheAsyncTicket() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks then finishes alone");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT,
|
||||
messages.ask(T, "which config?", 200).outcome());
|
||||
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket).phase(),
|
||||
"only the question wait ended; the delegated turn may still finish");
|
||||
|
||||
// The worker keeps working past the timeout and only now calls fleet_reply — with no live
|
||||
// rendezvous waiter open (ask()'s timeout already closed it) and no new send() having
|
||||
// reopened one for this target.
|
||||
assertTrue(messages.reply(T, "PR opened: https://example/pulls/42"));
|
||||
|
||||
MessageService.TaskView done = awaitTicketPhase(ticket, MessageService.Phase.DONE);
|
||||
assertEquals("PR opened: https://example/pulls/42", done.reply(),
|
||||
"fleet_poll{ticket} must return the worker's real reply, not stay pending forever");
|
||||
assertEquals("reply", done.replySource());
|
||||
assertFalse(messages.hasStrandedReply(T),
|
||||
"the reply completed its own ticket directly and never touched the inbox");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #307's ambiguity guard: an ask timeout frees its target ({@code hasAsyncQuestion}
|
||||
* becomes false the instant it lapses — proven above), so a second, independent delegation can
|
||||
* be dispatched to the same target and itself go on to ask-and-lapse before the first worker's
|
||||
* real reply ever arrives. Two open tasks are then both eligible candidates on one target with
|
||||
* no live waiter to disambiguate them. A reply arriving now must not guess which one it answers
|
||||
* — guessing wrong would hand the lead a plausible-looking answer to a delegation the worker
|
||||
* never touched, worse than a failure because the lead acts on it — so it must fall back to the
|
||||
* inbox exactly as the zero-candidate case does.
|
||||
*/
|
||||
@Test
|
||||
void twoAskTimedOutTicketsOnOneTargetFallBackToTheInboxRatherThanGuess() throws Exception {
|
||||
String ticket1 = messages.sendAsync(T, "first task that asks");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT, messages.ask(T, "Q1?", 200).outcome());
|
||||
|
||||
String ticket2 = messages.sendAsync(T, "second task that asks");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT, messages.ask(T, "Q2?", 200).outcome());
|
||||
|
||||
assertTrue(messages.reply(T, "which task does this answer?"));
|
||||
|
||||
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket1).phase(),
|
||||
"an ambiguous reply must not guess ticket1");
|
||||
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket2).phase(),
|
||||
"an ambiguous reply must not guess ticket2");
|
||||
assertTrue(messages.hasStrandedReply(T));
|
||||
var drained = messages.drainReplies(T);
|
||||
assertEquals(1, drained.size());
|
||||
assertEquals("which task does this answer?", drained.get(0).content());
|
||||
}
|
||||
|
||||
@Test
|
||||
void asyncQuestionBelongsToTheTaskThatOwnsItsForwardWaiter() throws Exception {
|
||||
String first = messages.sendAsync(T, "first task");
|
||||
@@ -723,6 +1045,64 @@ class MessageServiceTest {
|
||||
assertEquals(MessageService.Outcome.REPLIED, answer.get(5, TimeUnit.SECONDS).outcome());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #282: a worker that chains a SECOND {@code fleet_ask} inside the same resumed turn —
|
||||
* before it ever calls {@code fleet_reply} — used to kill its own async ticket. {@code answer()}
|
||||
* opens a fresh forward waiter but (unlike {@code send()}) never registered it in
|
||||
* {@code asyncTasksByWaiter}, so the second ask's {@code markAsyncQuestion} found no {@code Task}
|
||||
* to re-associate. That waiter still resolved with the second {@code QUESTION} once the worker
|
||||
* asked again, and {@code answer()} completed the ticket's future with that QUESTION "reply"
|
||||
* unconditionally — so {@code fleet_poll} reported FAILED while the worker was still alive and
|
||||
* the primary was mid-conversation with it.
|
||||
*
|
||||
* <p>Driven entirely through {@code MessageService}'s public API (sendAsync/ask/answer/poll) —
|
||||
* never by reaching into {@link Rendezvous} or the task maps directly, so this test cannot pass
|
||||
* for a reason unrelated to the real bug.
|
||||
*/
|
||||
@Test
|
||||
void secondFleetAskInTheSameResumedTurnDoesNotKillTheAsyncTicket() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "task that asks twice");
|
||||
awaitWaiting();
|
||||
|
||||
// The worker's first fleet_ask.
|
||||
CompletableFuture<MessageService.AskResult> ask1 =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "Q1", 5000));
|
||||
MessageService.TaskView asking1 = awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
assertEquals("Q1", asking1.reply());
|
||||
|
||||
// The primary answers it — answer() resumes the turn and blocks for what comes next.
|
||||
CompletableFuture<MessageService.Reply> answer1 = CompletableFuture.supplyAsync(
|
||||
() -> messages.answer(asking1.turnId(), "a1", 5000));
|
||||
assertEquals("a1", ask1.get(5, TimeUnit.SECONDS).answer());
|
||||
|
||||
// Still in the SAME resumed turn — before replying — the worker asks again.
|
||||
CompletableFuture<MessageService.AskResult> ask2 =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "Q2", 5000));
|
||||
|
||||
// answer1's own call unblocks with the second QUESTION (documented QUESTION-chaining
|
||||
// behaviour — see FleetMcp.answer's javadoc: "Answer it by calling fleet_send again with
|
||||
// turnId=..."). The bug: this used to also kill the async ticket in the process.
|
||||
MessageService.Reply firstAnswerResult = answer1.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, firstAnswerResult.outcome());
|
||||
String turnId2 = firstAnswerResult.turnId();
|
||||
|
||||
MessageService.TaskView asking2 = awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
assertEquals("Q2", asking2.reply(),
|
||||
"the ticket must surface the SECOND question, not be dead/FAILED");
|
||||
assertEquals(turnId2, asking2.turnId());
|
||||
|
||||
// The primary answers the second question; the worker finally sends its real fleet_reply.
|
||||
CompletableFuture<MessageService.Reply> answer2 = CompletableFuture.supplyAsync(
|
||||
() -> messages.answer(turnId2, "a2", 5000));
|
||||
assertEquals("a2", ask2.get(5, TimeUnit.SECONDS).answer());
|
||||
awaitWaiting();
|
||||
assertTrue(rendezvous.resolve(T, "done"));
|
||||
assertEquals(MessageService.Outcome.REPLIED, answer2.get(5, TimeUnit.SECONDS).outcome());
|
||||
|
||||
MessageService.TaskView done = awaitTicketPhase(ticket, MessageService.Phase.DONE);
|
||||
assertEquals("done", done.reply());
|
||||
}
|
||||
|
||||
// --- CB-582: fleet_status pendingAsk() ------------------------------------------------------
|
||||
|
||||
@Test
|
||||
@@ -1057,6 +1437,74 @@ class MessageServiceTest {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* #197: the ticket TTL must run from COMPLETION, not from creation.
|
||||
*
|
||||
* <p>It used to compare the cutoff against {@code createdNanos}, so the real window to collect a
|
||||
* reply was {@code TTL minus however long the task ran}. A delegation that ran longer than the
|
||||
* TTL was already past the cutoff the moment it finished, so the very next prune destroyed its
|
||||
* reply — and the reply lives only in the task's future, so nothing could get it back. That is
|
||||
* the normal case for real work here, not an edge case: three workers in one session ran well
|
||||
* past ten minutes and two of their complete reports were lost this way.
|
||||
*
|
||||
* <p>The task below runs for longer than the whole TTL before it replies, which is exactly the
|
||||
* shape that used to lose everything. Remove the fix and this fails: {@code poll} returns
|
||||
* {@code null} because the ticket was pruned on arrival.
|
||||
*/
|
||||
@Test
|
||||
void aTaskRunningLongerThanTheTtlStillKeepsItsReport() throws Exception {
|
||||
java.util.concurrent.atomic.AtomicLong clock = new java.util.concurrent.atomic.AtomicLong(1_000_000_000L);
|
||||
try (var wiring = wireWithPushLoop(1, 50, clock::get)) {
|
||||
String slow = wiring.service().sendAsync(T, "a task that takes longer than the TTL");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
|
||||
// The worker is still working, and has been for longer than the entire TTL. Nothing may
|
||||
// be pruned yet — the ticket has not finished, so there is no report to keep or lose.
|
||||
clock.addAndGet(MessageService.TICKET_TTL_NANOS + TimeUnit.SECONDS.toNanos(30));
|
||||
|
||||
// Only now does it reply. Under the old clock this reply was born already expired.
|
||||
assertTrue(rendezvous.resolve(T, "the long report"));
|
||||
awaitTicketPhaseOn(wiring.service(), slow, MessageService.Phase.DONE);
|
||||
|
||||
// A second delegation runs pruneTerminalTickets before it returns.
|
||||
wiring.service().sendAsync(T, "an unrelated second task");
|
||||
|
||||
MessageService.TaskView view = wiring.service().poll(slow);
|
||||
assertNotNull(view, "a ticket that completed just now must survive the prune, however "
|
||||
+ "long its task ran — the TTL is the window to COLLECT the report, not the "
|
||||
+ "budget for producing it");
|
||||
assertEquals(MessageService.Phase.DONE, view.phase());
|
||||
assertEquals("the long report", view.reply(),
|
||||
"the worker's actual report must still be there, not just the ticket");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The other half of #197: the TTL must still bound {@code tasks}. Measuring from completion
|
||||
* would be a leak if a finished ticket were then kept forever, so this pins the eviction that
|
||||
* still has to happen — the same ticket, left uncollected for longer than the TTL AFTER it
|
||||
* finished, is gone.
|
||||
*/
|
||||
@Test
|
||||
void aFinishedTicketIsStillPrunedOnceTheTtlPassesSinceItFinished() throws Exception {
|
||||
java.util.concurrent.atomic.AtomicLong clock = new java.util.concurrent.atomic.AtomicLong(1_000_000_000L);
|
||||
try (var wiring = wireWithPushLoop(1, 50, clock::get)) {
|
||||
String done = wiring.service().sendAsync(T, "a quick task");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
assertTrue(rendezvous.resolve(T, "quick result"));
|
||||
awaitTicketPhaseOn(wiring.service(), done, MessageService.Phase.DONE);
|
||||
|
||||
// Nobody collected it, and the TTL has now passed since it FINISHED.
|
||||
clock.addAndGet(MessageService.TICKET_TTL_NANOS + TimeUnit.SECONDS.toNanos(1));
|
||||
wiring.service().sendAsync(T, "an unrelated second task");
|
||||
|
||||
assertNull(wiring.service().poll(done),
|
||||
"the TTL must still evict an uncollected finished ticket, or tasks grows forever");
|
||||
}
|
||||
}
|
||||
|
||||
private MessageService.TaskView awaitTicketPhaseOn(MessageService svc, String ticket,
|
||||
MessageService.Phase phase) throws Exception {
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
|
||||
@@ -43,6 +43,7 @@ class ReplyPushLoopTest {
|
||||
private static final String PRIMARY = "term_primary";
|
||||
private static final String WORKER = "term_worker";
|
||||
private static final String WORKER2 = "term_worker2";
|
||||
private static final String OTHER_PRIMARY = "term_other_primary";
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
|
||||
private PrimaryRegistry registry;
|
||||
@@ -529,6 +530,106 @@ class ReplyPushLoopTest {
|
||||
assertTrue(nudge.contains("task-1"), "the ticket must not be dropped: " + nudge);
|
||||
}
|
||||
|
||||
// --- CB-201: backend credential outage nudges -----------------------------------------------
|
||||
|
||||
@Test
|
||||
void oneBackendIncidentForTwoWorkersOfOneLeadSendsOnceAndIsOneShot() throws Exception {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
registry.recordDelegation(WORKER, PRIMARY);
|
||||
registry.recordDelegation(WORKER2, PRIMARY);
|
||||
var loop = loop(2, 100);
|
||||
|
||||
loop.onBackendIncident("incident-1", List.of(WORKER, WORKER2), "cred-a",
|
||||
List.of("terra", "codex"), 47);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS), "one lead gets one outage injection");
|
||||
Thread.sleep(200);
|
||||
assertEquals(1, rec.sendCount(), "two workers for one lead must send only one outage notice");
|
||||
String nudge = rec.sentParams().getFirst().getValue().toString();
|
||||
assertTrue(nudge.contains("cred-a"));
|
||||
assertTrue(nudge.contains("codex, terra"));
|
||||
assertTrue(nudge.contains(WORKER) && nudge.contains(WORKER2));
|
||||
assertTrue(nudge.contains("47 remaining seconds"));
|
||||
assertTrue(nudge.contains("fleet_list"));
|
||||
|
||||
loop.onBackendIncident("incident-1", List.of(WORKER, WORKER2), "cred-a",
|
||||
List.of("terra", "codex"), 47);
|
||||
Thread.sleep(250);
|
||||
assertEquals(1, rec.sendCount(), "a delivered incident id must never be sent again");
|
||||
}
|
||||
|
||||
@Test
|
||||
void oneBackendIncidentSendsOnceToEachAffectedLead() throws Exception {
|
||||
var rec = recordingClient();
|
||||
rec.sendLatch = new CountDownLatch(2);
|
||||
agents = new AgentControl(rec);
|
||||
registry.recordDelegation(WORKER, PRIMARY);
|
||||
registry.recordDelegation(WORKER2, OTHER_PRIMARY);
|
||||
var loop = loop(1, 50);
|
||||
|
||||
loop.onBackendIncident("incident-2", List.of(WORKER, WORKER2), "cred-a", List.of("terra"), 60);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS), "each affected lead gets its one notice");
|
||||
assertEquals(2, rec.sendCount());
|
||||
}
|
||||
|
||||
@Test
|
||||
void backendIncidentWaitsForBusyLeadThenCombinesWithFailedTicket() throws Exception {
|
||||
var rec = new BusyThenIdleHerdrClient(2);
|
||||
agents = new AgentControl(rec);
|
||||
var loop = loop(1, 50);
|
||||
|
||||
loop.onTicketTerminal("task-failed", WORKER, true);
|
||||
loop.onBackendIncident("incident-3", List.of(WORKER), "cred-b", List.of("terra"), 31);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS), "busy lead should receive deferred combined nudge");
|
||||
assertEquals(1, rec.sendCount(), "ticket and outage must use the same injection");
|
||||
String nudge = rec.sentParams().getFirst().getValue().toString();
|
||||
assertTrue(nudge.contains("task-failed") && nudge.contains("cred-b"),
|
||||
"the one nudge must name both the failed ticket and outage: " + nudge);
|
||||
}
|
||||
|
||||
@Test
|
||||
void failedBackendIncidentSendRetriesWithinTheExistingReminderCap() throws Exception {
|
||||
var rec = new ThrowOnceHerdrClient();
|
||||
agents = new AgentControl(rec);
|
||||
var loop = loop(2, 50);
|
||||
|
||||
loop.onBackendIncident("incident-4", List.of(WORKER), "cred-c", List.of("terra"), 20);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS), "failed first send must retry under maxReminders");
|
||||
assertEquals(2, rec.promptAttempts.get(), "one failed send and one retry use the existing cap");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unmappedBackendTargetUsesTheKnownLeadSchedule() throws Exception {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
var loop = loop(1, 50);
|
||||
|
||||
loop.onBackendTargetUnmapped(WORKER, "no credential matched");
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS));
|
||||
String nudge = ((Map<?, ?>) rec.sentParams().getFirst().getValue()).get("text").toString();
|
||||
assertEquals("A backend error on worker term_worker could not be mapped to a credential "
|
||||
+ "(no credential matched) — no cool-off was applied. "
|
||||
+ "Run fleet_list to check that worker.",
|
||||
nudge);
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopClearsPendingBackendIncidents() {
|
||||
agents = agentWithStatus("idle");
|
||||
var loop = loop(1, 100_000);
|
||||
loop.onBackendIncident("incident-stop", List.of(WORKER), "cred-stop", List.of("terra"), 60);
|
||||
|
||||
loop.stop();
|
||||
|
||||
assertEquals(ReplyPushLoop.Action.STOP, loop.decide(PRIMARY, 0, 0, 0),
|
||||
"stop must clear a pending backend incident as well as older sources");
|
||||
}
|
||||
|
||||
// --- CB-590 follow-up: per-source reminder budgets — the regression this round exists for ---
|
||||
|
||||
@Test
|
||||
@@ -928,6 +1029,31 @@ class ReplyPushLoopTest {
|
||||
return new RecordingHerdrClient();
|
||||
}
|
||||
|
||||
/** Fails its first prompt call, then records the retry. */
|
||||
private static final class ThrowOnceHerdrClient implements HerdrClient {
|
||||
private final AtomicInteger promptAttempts = new AtomicInteger();
|
||||
private final CountDownLatch sendLatch = new CountDownLatch(1);
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
if ("agent.get".equals(method)) {
|
||||
return MAPPER.createObjectNode().set("agent", MAPPER.createObjectNode()
|
||||
.put("terminal_id", PRIMARY).put("agent_status", "idle"));
|
||||
}
|
||||
if ("agent.prompt".equals(method) && promptAttempts.getAndIncrement() == 0) {
|
||||
throw new IllegalStateException("first prompt fails");
|
||||
}
|
||||
if ("agent.prompt".equals(method)) {
|
||||
sendLatch.countDown();
|
||||
}
|
||||
return MAPPER.createObjectNode();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Thread-safe fake that reports {@code working} (not injectable) for its first
|
||||
* {@code busyChecks} status calls, then {@code idle} forever after — used to prove work queued
|
||||
|
||||
@@ -0,0 +1,266 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #201 / #227: the credential-keyed outage decision itself, isolated from the classifier
|
||||
* that produces events (Unit 1) and from the lead notice / spawn wiring (Units 3 and 5). The clock
|
||||
* is a plain {@link AtomicLong} of nanos so the window and cool-off are exercised without a real
|
||||
* sleep, exactly like {@code BackendQuarantineTest}.
|
||||
*/
|
||||
class BackendOutagePolicyTest {
|
||||
|
||||
private static final long SECOND = 1_000_000_000L;
|
||||
|
||||
@Test
|
||||
void oneErrorCreatesNoIncidentAndNoCoolOff() {
|
||||
BackendOutagePolicy policy = new BackendOutagePolicy(() -> 0L);
|
||||
|
||||
Optional<BackendOutagePolicy.Incident> incident =
|
||||
policy.record("shared-openai", "session-1", "API Error: 429");
|
||||
|
||||
assertTrue(incident.isEmpty());
|
||||
assertTrue(policy.remainingCoolOffSeconds("shared-openai").isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void twoErrorsWithinTheWindowCreateExactlyOneIncidentAndA60SecondCoolOff() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendOutagePolicy policy = new BackendOutagePolicy(now::get);
|
||||
|
||||
assertTrue(policy.record("shared-openai", "session-1", "API Error: 429").isEmpty());
|
||||
now.set(30 * SECOND);
|
||||
Optional<BackendOutagePolicy.Incident> incident =
|
||||
policy.record("shared-openai", "session-2", "API Error: 500");
|
||||
|
||||
assertTrue(incident.isPresent());
|
||||
BackendOutagePolicy.Incident value = incident.get();
|
||||
assertEquals("shared-openai", value.credentialId());
|
||||
assertEquals(Set.of("session-1", "session-2"), value.targets());
|
||||
assertEquals(2, value.evidenceCount());
|
||||
assertEquals(60L, value.windowSeconds());
|
||||
assertEquals(60L, value.remainingCoolOffSeconds());
|
||||
assertEquals(OptionalLongOf60(), policy.remainingCoolOffSeconds("shared-openai"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void repeatedErrorsFromTheSameTargetNeverCreateAnIncidentHoweverManyTimesTheyRepeat() {
|
||||
// The classifier is a heuristic and a valid member report can quote an "API Error:" line, so
|
||||
// one target repeating that line inside the window must never, by itself, cost the credential
|
||||
// its capacity — only a SECOND, DISTINCT target crossing the threshold does.
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendOutagePolicy policy = new BackendOutagePolicy(now::get);
|
||||
|
||||
for (int i = 0; i < 5; i++) {
|
||||
now.set(i * 10L * SECOND);
|
||||
Optional<BackendOutagePolicy.Incident> incident =
|
||||
policy.record("shared-openai", "session-1", "API Error: 429");
|
||||
assertTrue(incident.isEmpty(),
|
||||
"the same target repeating must never create an incident on its own (repeat #" + i + ")");
|
||||
}
|
||||
assertTrue(policy.remainingCoolOffSeconds("shared-openai").isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void evidenceCountIsDistinctTargetsWhileReasonsKeepsEveryEvent() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendOutagePolicy policy = new BackendOutagePolicy(now::get);
|
||||
|
||||
assertTrue(policy.record("shared-openai", "session-1", "API Error: 429").isEmpty());
|
||||
now.set(10 * SECOND);
|
||||
// a repeat from the already-counted target: still no incident, still only 1 distinct target
|
||||
assertTrue(policy.record("shared-openai", "session-1", "API Error: 429 (again)").isEmpty());
|
||||
now.set(20 * SECOND);
|
||||
Optional<BackendOutagePolicy.Incident> incident =
|
||||
policy.record("shared-openai", "session-2", "API Error: 500");
|
||||
|
||||
assertTrue(incident.isPresent());
|
||||
BackendOutagePolicy.Incident value = incident.get();
|
||||
assertEquals(Set.of("session-1", "session-2"), value.targets());
|
||||
assertEquals(2, value.evidenceCount(), "evidenceCount is the distinct-target count");
|
||||
assertEquals(3, value.reasons().size(),
|
||||
"reasons keeps every event, including the same-target repeat that did not advance the count");
|
||||
assertEquals(java.util.List.of("API Error: 429", "API Error: 429 (again)", "API Error: 500"),
|
||||
value.reasons());
|
||||
}
|
||||
|
||||
@Test
|
||||
void twoErrorsMoreThanAMinuteApartCreateNoIncident() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendOutagePolicy policy = new BackendOutagePolicy(now::get);
|
||||
|
||||
assertTrue(policy.record("shared-openai", "session-1", "API Error: 429").isEmpty());
|
||||
now.set(60 * SECOND + 1);
|
||||
Optional<BackendOutagePolicy.Incident> incident =
|
||||
policy.record("shared-openai", "session-2", "API Error: 500");
|
||||
|
||||
assertTrue(incident.isEmpty(), "the first error's evidence must not survive past the window");
|
||||
assertTrue(policy.remainingCoolOffSeconds("shared-openai").isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void exactly60SecondsApartIsInsideTheWindow() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendOutagePolicy policy = new BackendOutagePolicy(now::get);
|
||||
|
||||
assertTrue(policy.record("shared-openai", "session-1", "API Error: 429").isEmpty());
|
||||
now.set(60 * SECOND); // exactly the boundary
|
||||
Optional<BackendOutagePolicy.Incident> incident =
|
||||
policy.record("shared-openai", "session-2", "API Error: 500");
|
||||
|
||||
assertTrue(incident.isPresent(), "60 seconds apart is inclusive of the window boundary");
|
||||
}
|
||||
|
||||
@Test
|
||||
void differentCredentialsNeverShareEvidence() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendOutagePolicy policy = new BackendOutagePolicy(now::get);
|
||||
|
||||
assertTrue(policy.record("credential-a", "session-1", "API Error: 429").isEmpty());
|
||||
now.set(10 * SECOND);
|
||||
Optional<BackendOutagePolicy.Incident> incident =
|
||||
policy.record("credential-b", "session-2", "API Error: 429");
|
||||
|
||||
assertTrue(incident.isEmpty(), "an unrelated credential must not be swept into the count");
|
||||
}
|
||||
|
||||
@Test
|
||||
void twoDifferentProfilesResolvingToTheSameCredentialShareEvidence() {
|
||||
// The caller is responsible for resolving a profile to its credential id (see
|
||||
// FleetConfig.Profile#effectiveCredentialId(), same as BackendQuarantine); this class only
|
||||
// ever sees the resolved credentialId, so two different profile-originated events landing on
|
||||
// the same credentialId must correlate.
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendOutagePolicy policy = new BackendOutagePolicy(now::get);
|
||||
|
||||
assertTrue(policy.record("shared-openai", "session-from-profile-a", "API Error: 429").isEmpty());
|
||||
now.set(5 * SECOND);
|
||||
Optional<BackendOutagePolicy.Incident> incident =
|
||||
policy.record("shared-openai", "session-from-profile-b", "API Error: 500");
|
||||
|
||||
assertTrue(incident.isPresent());
|
||||
}
|
||||
|
||||
@Test
|
||||
void errorsDuringCoolOffNeitherExtendTheDeadlineNorReturnAnotherIncident() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendOutagePolicy policy = new BackendOutagePolicy(now::get);
|
||||
|
||||
policy.record("shared-openai", "session-1", "API Error: 429");
|
||||
now.set(1 * SECOND);
|
||||
policy.record("shared-openai", "session-2", "API Error: 500"); // crosses threshold, cool-off starts at t=1s, until t=61s
|
||||
|
||||
now.set(50 * SECOND);
|
||||
Optional<BackendOutagePolicy.Incident> duringCoolOff =
|
||||
policy.record("shared-openai", "session-3", "API Error: 500");
|
||||
assertTrue(duringCoolOff.isEmpty(), "no second incident while cooling off");
|
||||
assertEquals(OptionalLongOf(11), policy.remainingCoolOffSeconds("shared-openai"),
|
||||
"the deadline must not have been pushed out by the error during cool-off");
|
||||
}
|
||||
|
||||
@Test
|
||||
void expiryClearsOldEvidenceSoTwoFreshErrorsAreRequiredToRearm() {
|
||||
// Both errors land at the SAME instant (t=0) so the window boundary (60s from t=0) and the
|
||||
// cool-off boundary (60s from the threshold crossing, also t=0) coincide exactly at t=60s.
|
||||
// That is deliberate: with WINDOW_NANOS == COOLOFF_NANOS, checking "one fresh error" at any
|
||||
// later instant would also be caught by the window naturally going stale, which would pass
|
||||
// even if the explicit evidence-clear on cool-off expiry were missing. Checking exactly AT
|
||||
// t=60s is the one instant where "now - firstEventNanos > WINDOW_NANOS" is false (60 is not
|
||||
// > 60) but the cool-off has already elapsed ("now < coolOffUntilNanos" is false at 60 >= 60)
|
||||
// — so only an explicit clear, not a side effect of the window check, keeps this from
|
||||
// instantly re-mounting the old evidence and minting a bogus incident off one fresh error.
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendOutagePolicy policy = new BackendOutagePolicy(now::get);
|
||||
|
||||
policy.record("shared-openai", "session-1", "API Error: 429");
|
||||
policy.record("shared-openai", "session-2", "API Error: 500"); // cool-off until t=60s
|
||||
|
||||
now.set(60 * SECOND); // cool-off just elapsed, exactly at the window's own boundary too
|
||||
Optional<BackendOutagePolicy.Incident> oneFreshError =
|
||||
policy.record("shared-openai", "session-4", "API Error: 500");
|
||||
assertTrue(oneFreshError.isEmpty(), "a single fresh error must not immediately re-trip");
|
||||
|
||||
now.set(61 * SECOND);
|
||||
Optional<BackendOutagePolicy.Incident> secondFreshError =
|
||||
policy.record("shared-openai", "session-5", "API Error: 500");
|
||||
assertTrue(secondFreshError.isPresent(), "two fresh errors after expiry re-arm the policy");
|
||||
}
|
||||
|
||||
@Test
|
||||
void remainingSecondsRoundUp() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendOutagePolicy policy = new BackendOutagePolicy(now::get);
|
||||
|
||||
policy.record("shared-openai", "session-1", "API Error: 429");
|
||||
now.set(1); // 1 nanosecond later, still crosses the threshold
|
||||
policy.record("shared-openai", "session-2", "API Error: 500");
|
||||
|
||||
// cool-off deadline is now(=1ns) + 60s exactly; a fraction-of-a-second-remaining check:
|
||||
now.set(1 + 59 * SECOND + 1); // 59.000000001s of the cool-off has passed -> < 1s remains
|
||||
assertEquals(OptionalLongOf(1), policy.remainingCoolOffSeconds("shared-openai"),
|
||||
"a sub-second remainder must round up to a full second, matching BackendQuarantine");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aConcurrentSecondAndThirdCallReturnOnlyOneIncidentBetweenThem() throws InterruptedException {
|
||||
for (int attempt = 0; attempt < 25; attempt++) {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendOutagePolicy policy = new BackendOutagePolicy(now::get);
|
||||
int threads = 8;
|
||||
ExecutorService pool = Executors.newFixedThreadPool(threads);
|
||||
CountDownLatch ready = new CountDownLatch(threads);
|
||||
CountDownLatch go = new CountDownLatch(1);
|
||||
AtomicInteger incidents = new AtomicInteger();
|
||||
|
||||
try {
|
||||
for (int i = 0; i < threads; i++) {
|
||||
int idx = i;
|
||||
pool.submit(() -> {
|
||||
ready.countDown();
|
||||
try {
|
||||
go.await();
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
return;
|
||||
}
|
||||
Optional<BackendOutagePolicy.Incident> incident =
|
||||
policy.record("shared-openai", "session-" + idx, "API Error: 500");
|
||||
if (incident.isPresent()) {
|
||||
incidents.incrementAndGet();
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
assertTrue(ready.await(5, TimeUnit.SECONDS));
|
||||
go.countDown();
|
||||
pool.shutdown();
|
||||
assertTrue(pool.awaitTermination(5, TimeUnit.SECONDS));
|
||||
|
||||
assertEquals(1, incidents.get(),
|
||||
"exactly one incident must be minted across every concurrent caller on attempt " + attempt);
|
||||
} finally {
|
||||
pool.shutdownNow();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private static java.util.OptionalLong OptionalLongOf60() {
|
||||
return OptionalLongOf(60);
|
||||
}
|
||||
|
||||
private static java.util.OptionalLong OptionalLongOf(long value) {
|
||||
return java.util.OptionalLong.of(value);
|
||||
}
|
||||
}
|
||||
@@ -30,7 +30,14 @@ class PlacementPolicyTest {
|
||||
private static PlacementContext ctx(List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount,
|
||||
Set<String> unreachable, Set<String> quarantined) {
|
||||
return new PlacementContext("b", candidates, liveCount, unreachable, quarantined);
|
||||
return ctx(candidates, liveCount, unreachable, quarantined, Set.of());
|
||||
}
|
||||
|
||||
private static PlacementContext ctx(List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount,
|
||||
Set<String> unreachable, Set<String> quarantined,
|
||||
Set<String> coolingOff) {
|
||||
return new PlacementContext("b", candidates, liveCount, unreachable, quarantined, coolingOff);
|
||||
}
|
||||
|
||||
private static PlacementContext ctx(List<PlacementCandidate> candidates,
|
||||
@@ -52,14 +59,14 @@ class PlacementPolicyTest {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext(null,
|
||||
List.of(PlacementCandidate.profile("a"), PlacementCandidate.profile("b")),
|
||||
noSessions(), Set.of(), Set.of());
|
||||
noSessions(), Set.of(), Set.of(), Set.of());
|
||||
assertEquals("a", policy.select(ctx).profile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void fixedThrowsWhenNoProfilesAndNoDefault() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext(null, List.of(), noSessions(), Set.of(), Set.of());
|
||||
PlacementContext ctx = new PlacementContext(null, List.of(), noSessions(), Set.of(), Set.of(), Set.of());
|
||||
assertThrows(PlacementException.class, () -> policy.select(ctx));
|
||||
}
|
||||
|
||||
@@ -68,7 +75,7 @@ class PlacementPolicyTest {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext("b",
|
||||
List.of(PlacementCandidate.profile("a"), PlacementCandidate.profile("b")),
|
||||
noSessions(), Set.of(), Set.of("b"));
|
||||
noSessions(), Set.of(), Set.of("b"), Set.of());
|
||||
assertEquals("a", policy.select(ctx).profile(),
|
||||
"the default 'b' is quarantined, so fixed falls through to the first un-quarantined candidate");
|
||||
}
|
||||
@@ -78,7 +85,7 @@ class PlacementPolicyTest {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext("b",
|
||||
List.of(PlacementCandidate.profile("a"), PlacementCandidate.profile("b")),
|
||||
noSessions(), Set.of(), Set.of("a", "b"));
|
||||
noSessions(), Set.of(), Set.of("a", "b"), Set.of());
|
||||
PlacementException e = assertThrows(PlacementException.class, () -> policy.select(ctx));
|
||||
assertTrue(e.getMessage().contains("quarantined"), e.getMessage());
|
||||
}
|
||||
@@ -211,6 +218,67 @@ class PlacementPolicyTest {
|
||||
assertTrue(e.getMessage().contains("quarantined"), e.getMessage());
|
||||
}
|
||||
|
||||
// --- fleetd #201 Unit 5: coolingOff excludes a candidate, SEPARATE from quarantined --------
|
||||
|
||||
@Test
|
||||
void roundRobinSkipsCoolingOffProfiles() {
|
||||
PlacementPolicy policy = PlacementPolicies.roundRobin();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a"),
|
||||
PlacementCandidate.profile("b"));
|
||||
PlacementContext ctx = ctx(candidates, noSessions(), Set.of(), Set.of(), Set.of("a"));
|
||||
for (int i = 0; i < 5; i++) {
|
||||
assertEquals("b", policy.select(ctx).profile(), "a is cooling off, so every pick lands on b");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedSkipsCoolingOffProfile() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 1.0f, null),
|
||||
PlacementCandidate.profile("b", 1.0f, null));
|
||||
PlacementContext ctx = ctx(candidates, noSessions(), Set.of(), Set.of(), Set.of("a"));
|
||||
for (int i = 0; i < 5; i++) {
|
||||
assertEquals("b", policy.select(ctx).profile(), "a is cooling off, so every pick lands on b");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedThrowsWhenAllCoolingOff() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a"),
|
||||
PlacementCandidate.profile("b"));
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> policy.select(ctx(candidates, noSessions(), Set.of(), Set.of(), Set.of("a", "b"))));
|
||||
assertTrue(e.getMessage().contains("cooling off"), e.getMessage());
|
||||
assertFalse(e.getMessage().contains("quarantined"),
|
||||
"nothing here is quarantined — the message must say cooling off, not exhausted: " + e.getMessage());
|
||||
}
|
||||
|
||||
/**
|
||||
* A candidate that is BOTH quarantined (CB-578 stage B) and cooling off (fleetd #201 Unit 5)
|
||||
* counts only as quarantined — exhaustion takes priority, matching
|
||||
* {@code CompositePeerLauncher}'s explicit-spawn check order.
|
||||
*/
|
||||
@Test
|
||||
void quarantinedAndCoolingOffIsNotDoubleCounted() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a"),
|
||||
PlacementCandidate.profile("b"));
|
||||
// a is BOTH quarantined and cooling off; b is quarantined only. If a were double-bucketed
|
||||
// as cooling-off instead of quarantined, quarantined would undercount to 1 of 2 candidates
|
||||
// and this would fall through to the generic "no worker profile available: N quarantined, M
|
||||
// cooling off, ..." message instead — which also happens to contain the substring
|
||||
// "quarantined", so a loose contains() check here would pass either way. Pinning the exact
|
||||
// "all ... quarantined" message is what actually proves the two are not double-counted.
|
||||
PlacementContext ctx = ctx(candidates, noSessions(), Set.of(), Set.of("a", "b"), Set.of("a"));
|
||||
PlacementException e = assertThrows(PlacementException.class, () -> policy.select(ctx));
|
||||
assertEquals("all worker profiles are quarantined (backend exhausted)", e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedThrowsWhenAllUnreachable() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
@@ -320,7 +388,7 @@ class PlacementPolicyTest {
|
||||
PlacementContext ctx = new PlacementContext("b",
|
||||
List.of(PlacementCandidate.profile("a", 1.0f, null),
|
||||
PlacementCandidate.profile("b", 0.0f, null)),
|
||||
noSessions(), Set.of(), Set.of());
|
||||
noSessions(), Set.of(), Set.of(), Set.of());
|
||||
assertEquals("a", policy.select(ctx).profile(),
|
||||
"the default 'b' has weight 0, so fixed falls through to the first available candidate");
|
||||
}
|
||||
@@ -331,7 +399,7 @@ class PlacementPolicyTest {
|
||||
PlacementContext ctx = new PlacementContext("b",
|
||||
List.of(PlacementCandidate.profile("a", 0.0f, null),
|
||||
PlacementCandidate.profile("b", 0.0f, null)),
|
||||
noSessions(), Set.of(), Set.of());
|
||||
noSessions(), Set.of(), Set.of(), Set.of());
|
||||
PlacementException e = assertThrows(PlacementException.class, () -> policy.select(ctx));
|
||||
assertTrue(e.getMessage().contains("weight 0"), e.getMessage());
|
||||
}
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package dev.ltms.fleet.rest;
|
||||
|
||||
import dev.ltms.fleet.auth.CallerResolver;
|
||||
import dev.ltms.fleet.auth.Authz;
|
||||
import dev.ltms.fleet.auth.MemberRegistry;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
@@ -25,8 +26,15 @@ import java.net.URI;
|
||||
import java.net.http.HttpClient;
|
||||
import java.net.http.HttpRequest;
|
||||
import java.net.http.HttpResponse;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
@@ -36,6 +44,10 @@ import static org.junit.jupiter.api.Assertions.*;
|
||||
*/
|
||||
class FleetAppAuthTest {
|
||||
|
||||
private static final Path REST_SOURCE = Path.of("src/main/java/dev/ltms/fleet/rest/FleetApp.java");
|
||||
private static final Pattern ROUTE_REGISTRATION =
|
||||
Pattern.compile("app\\.(get|post|delete|put|patch)\\(\\s*\"([^\"]+)\"");
|
||||
|
||||
private final HttpClient http = HttpClient.newHttpClient();
|
||||
private Javalin app;
|
||||
private Metrics metrics;
|
||||
@@ -92,6 +104,49 @@ class FleetAppAuthTest {
|
||||
return http.send(b.build(), HttpResponse.BodyHandlers.ofString());
|
||||
}
|
||||
|
||||
@Test
|
||||
void everyRegisteredRouteHasItsHandlerActionPinned() {
|
||||
Set<String> registered = routesTheServerRegisters();
|
||||
assertTrue(registered.size() >= 15,
|
||||
"scraped only " + registered.size() + " route registrations from FleetApp (" + registered
|
||||
+ "); the app.<verb>(\"…\") scrape has stopped matching");
|
||||
// Liveness must work before credentials can be checked, so this route is deliberately open.
|
||||
assertTrue(registered.remove("GET /healthz"), "GET /healthz must stay an explicit ungated exception");
|
||||
registered.forEach(route -> assertDoesNotThrow(() -> FleetApp.routeAction(route),
|
||||
() -> route + " is registered but has no pinned authorization action"));
|
||||
assertEquals(Authz.Action.METRICS, FleetApp.routeAction("GET /metrics"));
|
||||
assertEquals(Authz.Action.SPAWN, FleetApp.routeAction("POST /members"));
|
||||
assertEquals(Authz.Action.STOP, FleetApp.routeAction("DELETE /members/{paneId}"));
|
||||
assertEquals(Authz.Action.SEND, FleetApp.routeAction("POST /sessions/{id}/message"));
|
||||
assertEquals(Authz.Action.REPLY, FleetApp.routeAction("POST /sessions/{id}/reply"));
|
||||
assertEquals(Authz.Action.DRAIN, FleetApp.routeAction("GET /sessions/{id}/replies"));
|
||||
assertEquals(Authz.Action.ASK, FleetApp.routeAction("POST /sessions/{id}/ask"));
|
||||
for (String route : Set.of("GET /sessions", "GET /agents", "GET /members", "GET /profiles",
|
||||
"GET /member-credentials", "GET /sessions/{id}/status", "GET /tasks/{ticket}")) {
|
||||
assertEquals(Authz.Action.READ, FleetApp.routeAction(route), route);
|
||||
}
|
||||
assertThrows(IllegalArgumentException.class, () -> FleetApp.routeAction("GET /healthz"));
|
||||
}
|
||||
|
||||
private static Set<String> routesTheServerRegisters() {
|
||||
try {
|
||||
String source = Files.readString(REST_SOURCE).lines()
|
||||
.filter(line -> {
|
||||
String stripped = line.stripLeading();
|
||||
return !(stripped.startsWith("//") || stripped.startsWith("*") || stripped.startsWith("/*"));
|
||||
})
|
||||
.collect(Collectors.joining("\n"));
|
||||
Matcher matcher = ROUTE_REGISTRATION.matcher(source);
|
||||
Set<String> routes = new LinkedHashSet<>();
|
||||
while (matcher.find()) {
|
||||
routes.add(matcher.group(1).toUpperCase(Locale.ROOT) + " " + matcher.group(2));
|
||||
}
|
||||
return routes;
|
||||
} catch (Exception e) {
|
||||
throw new AssertionError("could not scrape FleetApp route registrations", e);
|
||||
}
|
||||
}
|
||||
|
||||
// --- loopback-trust: the caller is the primary -------------------------------------------
|
||||
|
||||
@Test
|
||||
|
||||
@@ -19,6 +19,10 @@ import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.session.Worktrees;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.member.MemberCredentialPolicyView;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
@@ -32,6 +36,7 @@ import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.Predicate;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
@@ -70,6 +75,19 @@ class FleetAppTest {
|
||||
|
||||
private int start(FakeHerdr herdr, String workerBaseUrl, Set<String> allow, String placement,
|
||||
Worktrees worktrees, Predicate<String> deliverable) {
|
||||
return start(herdr, workerBaseUrl, allow, placement, worktrees, deliverable,
|
||||
FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #297: same wiring as above, plus the two SAME shared sources {@code GET /profiles}
|
||||
* must read — lets a test prove the quarantined/coolingOff facts it reports come from a real
|
||||
* {@link dev.ltms.fleet.placement.BackendQuarantine}/{@link
|
||||
* dev.ltms.fleet.placement.BackendOutagePolicy}, exactly like {@code fleet_profiles}'s own tests.
|
||||
*/
|
||||
private int start(FakeHerdr herdr, String workerBaseUrl, Set<String> allow, String placement,
|
||||
Worktrees worktrees, Predicate<String> deliverable,
|
||||
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage) {
|
||||
FleetConfig.Profile wcfg = new FleetConfig.Profile(
|
||||
"ltms-local", workerBaseUrl, "coder", null, "FLEETD_WORKER_TOKEN", null,
|
||||
placement, "fleet", "worker: {profile} #{n}", null, null, null);
|
||||
@@ -90,8 +108,9 @@ class FleetAppTest {
|
||||
// it directly so the inbox contract holds for those endpoints.
|
||||
inbox.own("term_a");
|
||||
MessageService messages = new MessageService(agents, injector, rendezvous, inbox);
|
||||
app = new FleetApp(herdr, workers, sessions, messages, this.presence, null,
|
||||
null, null, id -> this.presence.isPresent(id) || deliverable.test(id))
|
||||
app = new FleetApp(herdr, herdr, workers, sessions, messages, this.presence, null,
|
||||
null, null, id -> this.presence.isPresent(id) || deliverable.test(id),
|
||||
MemberCredentialPolicyView::absent, quarantine, outage)
|
||||
.build().start("127.0.0.1", 0);
|
||||
return app.port();
|
||||
}
|
||||
@@ -164,6 +183,22 @@ class FleetAppTest {
|
||||
assertEquals("idle", agents.get(0).get("status").asText());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #297 gap 1: {@code workers.list()} reaches herdr, and a transport failure there must
|
||||
* land in the same {@code {error, detail}} envelope every other failure path in this file uses
|
||||
* (see {@code herdrError}), not escape as a bare exception outside the JSON contract.
|
||||
*/
|
||||
@Test
|
||||
void agentsMapsAHerdrFailureToTheJsonErrorEnvelope() throws Exception {
|
||||
FakeHerdr down = new FakeHerdr().healthy(false);
|
||||
int port = start(down, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
HttpResponse<String> res = req(port, "GET", "/agents");
|
||||
assertEquals(502, res.statusCode(), res.body());
|
||||
JsonNode body = mapper.readTree(res.body());
|
||||
assertEquals("herdr_error", body.get("error").asText());
|
||||
assertTrue(body.has("detail"), res.body());
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnWorkerLandsInOwnTabInWorkerSpaceAndInjectsBaseUrl() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
@@ -203,6 +238,42 @@ class FleetAppTest {
|
||||
JsonNode body = mapper.readTree(req(port, "GET", "/profiles").body());
|
||||
assertEquals("ltms-local", body.get("default").asText());
|
||||
assertEquals("ltms-local", body.get("profiles").get(0).asText());
|
||||
assertFalse(body.has("quarantined"), "nothing is quarantined, so the key is omitted: " + body);
|
||||
assertFalse(body.has("coolingOff"), "nothing is cooling off, so the key is omitted: " + body);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #297 gap 2: {@code GET /profiles} must report the same two outage states {@code
|
||||
* fleet_profiles} does — CB-578 stage B exhaustion quarantine and fleetd #201 Unit 5 cool-off —
|
||||
* reading the SAME shared {@link BackendQuarantine}/{@link BackendOutagePolicy} instances rather
|
||||
* than recomputing them. The two checks are independent, and this profile is deliberately put in
|
||||
* both states at once, matching {@code FleetMcpTest}'s own coverage of that overlap.
|
||||
*/
|
||||
@Test
|
||||
void profilesReportsQuarantineAndCoolingOffFromTheSameSharedSources() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
quarantine.quarantine("shared-openai");
|
||||
FleetMcp.QuarantineSource quarantineSource = new FleetMcp.QuarantineSource(
|
||||
profile -> "ltms-local".equals(profile) ? "shared-openai" : null, quarantine);
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
outagePolicy.record("shared-openai", "t1", "API Error: rate limited");
|
||||
outagePolicy.record("shared-openai", "t2", "API Error: rate limited"); // 2nd distinct target starts the incident
|
||||
FleetMcp.OutageSource outageSource = new FleetMcp.OutageSource(
|
||||
profile -> "ltms-local".equals(profile) ? "shared-openai" : null, outagePolicy);
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"), "tab", new GitWorktrees(),
|
||||
ignored -> false, quarantineSource, outageSource);
|
||||
|
||||
JsonNode body = mapper.readTree(req(port, "GET", "/profiles").body());
|
||||
assertTrue(body.has("quarantined"), body.toString());
|
||||
assertEquals("shared-openai",
|
||||
body.get("quarantined").get("ltms-local").get("credentialId").asText());
|
||||
assertEquals(1800,
|
||||
body.get("quarantined").get("ltms-local").get("quarantinedForSeconds").asLong());
|
||||
assertTrue(body.has("coolingOff"), body.toString());
|
||||
assertEquals("shared-openai",
|
||||
body.get("coolingOff").get("ltms-local").get("credentialId").asText());
|
||||
assertEquals(60, body.get("coolingOff").get("ltms-local").get("coolingOffForSeconds").asLong());
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -218,9 +289,17 @@ class FleetAppTest {
|
||||
|
||||
HttpResponse<String> res = req(port, "GET", "/members");
|
||||
assertEquals(200, res.statusCode());
|
||||
JsonNode workers = mapper.readTree(res.body()).get("workers");
|
||||
assertEquals(1, workers.size());
|
||||
JsonNode w = workers.get(0);
|
||||
JsonNode body = mapper.readTree(res.body());
|
||||
// fleetd #199: "members" is the canonical key. The endpoint is /members, so a caller that
|
||||
// reads "members" must not see an empty fleet. "workers" is kept only as a deprecated alias
|
||||
// and must carry the same rows — assert both, or the alias can silently drift.
|
||||
JsonNode members = body.get("members");
|
||||
assertNotNull(members, "GET /members must return its rows under \"members\"");
|
||||
assertEquals(1, members.size());
|
||||
JsonNode workers = body.get("workers");
|
||||
assertNotNull(workers, "the deprecated \"workers\" alias is still emitted");
|
||||
assertEquals(members, workers, "the alias must carry the same rows as \"members\"");
|
||||
JsonNode w = members.get(0);
|
||||
assertEquals(spawned.get("terminalId").asText(), w.get("sessionId").asText());
|
||||
assertEquals(paneId, w.get("paneId").asText());
|
||||
assertEquals("ltms-local", w.get("profile").asText());
|
||||
@@ -231,6 +310,23 @@ class FleetAppTest {
|
||||
"liveStatus is unknown when herdr has no matching pane");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #297 gap 1: same reasoning as {@code agentsMapsAHerdrFailureToTheJsonErrorEnvelope} —
|
||||
* {@code GET /members} is the endpoint's own comment names as "the out-of-band path a lead falls
|
||||
* back to when its MCP mount drops", so it must stay inside the {@code {error, detail}} envelope
|
||||
* exactly when herdr is briefly unreachable.
|
||||
*/
|
||||
@Test
|
||||
void membersMapsAHerdrFailureToTheJsonErrorEnvelope() throws Exception {
|
||||
FakeHerdr down = new FakeHerdr().healthy(false);
|
||||
int port = start(down, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
HttpResponse<String> res = req(port, "GET", "/members");
|
||||
assertEquals(502, res.statusCode(), res.body());
|
||||
JsonNode body = mapper.readTree(res.body());
|
||||
assertEquals("herdr_error", body.get("error").asText());
|
||||
assertTrue(body.has("detail"), res.body());
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnWithACwdParamRootsTheWorkerThere() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
@@ -319,7 +415,11 @@ class FleetAppTest {
|
||||
FakeHerdr herdr = new FakeHerdr().agentNameTakenTimes(99);
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
|
||||
assertEquals(500, req(port, "POST", "/members").statusCode());
|
||||
// fleetd #304: 502, not Javalin's default 500 — the herdr failure is named, and the body
|
||||
// carries herdr's own message, matching what fleet_spawn reports for the same failure.
|
||||
HttpResponse<String> res = req(port, "POST", "/members");
|
||||
assertEquals(502, res.statusCode());
|
||||
assertEquals("herdr_error", mapper.readTree(res.body()).get("error").asText());
|
||||
assertTrue(herdr.called("tab.create"), "a tab was created before the failed start");
|
||||
assertEquals("w9:t2", params(herdr, "tab.close").get("tab_id"), "orphaned tab must be closed");
|
||||
}
|
||||
@@ -436,6 +536,53 @@ class FleetAppTest {
|
||||
assertEquals("orphan", body.get("replies").get(0).get("content").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
void replyWithMissingContentIsRejectedAndDoesNotResolveTheWaiter() throws Exception {
|
||||
// fleetd #302: `.path("content").asText("")` used to turn a missing "content" key into an
|
||||
// empty string that reached rendezvous.resolve, silently completing the lead's blocking wait
|
||||
// with nothing. Prove the fix two ways: the bad call is rejected with 400, AND the real send
|
||||
// it would have wrongly resolved is still open afterwards — a real reply completes it.
|
||||
FakeHerdr herdr = new FakeHerdr().agentStatus("idle");
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
|
||||
var send = java.util.concurrent.CompletableFuture.supplyAsync(() -> {
|
||||
try { return postMessage(port, "{\"content\":\"review this\",\"timeoutMs\":4000}"); }
|
||||
catch (Exception e) { throw new RuntimeException(e); }
|
||||
});
|
||||
|
||||
Thread.sleep(200); // let the background send open its rendezvous waiter
|
||||
|
||||
HttpResponse<String> badReply = postJson(port, "/sessions/term_a/reply", "{}");
|
||||
assertEquals(400, badReply.statusCode());
|
||||
JsonNode err = mapper.readTree(badReply.body());
|
||||
assertEquals("bad_request", err.get("error").asText());
|
||||
assertTrue(err.has("detail"));
|
||||
|
||||
// The waiter must still be open — a real reply now completes the ORIGINAL send.
|
||||
HttpResponse<String> goodReply = postJson(port, "/sessions/term_a/reply", "{\"content\":\"LGTM ship it\"}");
|
||||
assertEquals(200, goodReply.statusCode());
|
||||
|
||||
HttpResponse<String> res = send.get(6, java.util.concurrent.TimeUnit.SECONDS);
|
||||
assertEquals(200, res.statusCode());
|
||||
assertEquals("LGTM ship it", mapper.readTree(res.body()).get("reply").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
void replyWithEmptyOrWhitespaceContentIsRejectedSameAsMissing() throws Exception {
|
||||
// fleetd #302 sibling case: present-but-blank content is treated the same as a missing key —
|
||||
// FleetMcp's own required-content guard (fleet_reply's "content is required") makes no
|
||||
// distinction between the two either, so diverging here would be a new asymmetry.
|
||||
int port = startHealthy();
|
||||
|
||||
HttpResponse<String> empty = postJson(port, "/sessions/term_a/reply", "{\"content\":\"\"}");
|
||||
assertEquals(400, empty.statusCode());
|
||||
assertEquals("bad_request", mapper.readTree(empty.body()).get("error").asText());
|
||||
|
||||
HttpResponse<String> whitespace = postJson(port, "/sessions/term_a/reply", "{\"content\":\" \"}");
|
||||
assertEquals(400, whitespace.statusCode());
|
||||
assertEquals("bad_request", mapper.readTree(whitespace.body()).get("error").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
void drainRepliesReturnsEmptyForNoReplies() throws Exception {
|
||||
int port = startHealthy();
|
||||
@@ -590,7 +737,12 @@ class FleetAppTest {
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
|
||||
// A genuine teardown failure must surface, not be reported as a successful 204.
|
||||
assertEquals(500, req(port, "DELETE", "/members/w9:pW").statusCode());
|
||||
// fleetd #304: it surfaces as a named 502 rather than Javalin's default 500. The property
|
||||
// this test guards is "not 204" and the herdr detail reaching the caller — a bare 500 gave
|
||||
// the body "Server Error" and said nothing about herdr.
|
||||
HttpResponse<String> res = req(port, "DELETE", "/members/w9:pW");
|
||||
assertEquals(502, res.statusCode());
|
||||
assertEquals("herdr_error", mapper.readTree(res.body()).get("error").asText());
|
||||
assertFalse(herdr.called("tab.close"), "tab is not removed when the pane close failed");
|
||||
}
|
||||
|
||||
@@ -603,4 +755,5 @@ class FleetAppTest {
|
||||
assertEquals(204, req(port, "DELETE", "/members/w9:pW").statusCode());
|
||||
assertTrue(herdr.called("tab.close"));
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
@@ -0,0 +1,150 @@
|
||||
package dev.ltms.fleet.rest;
|
||||
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.net.URI;
|
||||
import java.net.http.HttpClient;
|
||||
import java.net.http.HttpRequest;
|
||||
import java.net.http.HttpResponse;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-185: with a router split across two herdr daemons (lead + {@code memberHerdrSocket}),
|
||||
* {@link FleetApp#healthz} must require BOTH daemons to answer and {@link FleetApp#sessions}
|
||||
* (which the {@code GET /sessions} route calls) must merge workspaces from both — the bug this
|
||||
* guards against had {@code FleetApp} constructed with the raw lead-only client, so a down member
|
||||
* daemon was invisible behind a green {@code /healthz} (every spawn then fails) and every member
|
||||
* workspace was silently dropped from {@code GET /sessions}.
|
||||
*
|
||||
* <p>Builds the real {@link FleetApp} directly (not a hand-rolled stand-in) against only the two
|
||||
* herdr clients — the other collaborators are unused by the two routes under test here.
|
||||
*/
|
||||
class FleetAppTwoDaemonTest {
|
||||
|
||||
private final HttpClient http = HttpClient.newHttpClient();
|
||||
private Javalin app;
|
||||
|
||||
@AfterEach
|
||||
void stop() {
|
||||
if (app != null) app.stop();
|
||||
}
|
||||
|
||||
private int start(HerdrClient lead, HerdrClient member) {
|
||||
app = new FleetApp(lead, member, null, null, null, null, null, null, null, ignored -> false)
|
||||
.build().start("127.0.0.1", 0);
|
||||
return app.port();
|
||||
}
|
||||
|
||||
private HttpResponse<String> get(int port, String path) throws Exception {
|
||||
HttpRequest req = HttpRequest.newBuilder(URI.create("http://127.0.0.1:" + port + path)).GET().build();
|
||||
return http.send(req, HttpResponse.BodyHandlers.ofString());
|
||||
}
|
||||
|
||||
@Test
|
||||
void healthzIsGreenWhenBothDaemonsAnswer() throws Exception {
|
||||
int port = start(new FakeHerdr(), new FakeHerdr());
|
||||
assertEquals(200, get(port, "/healthz").statusCode());
|
||||
}
|
||||
|
||||
@Test
|
||||
void healthzIsDegradedWhenOnlyTheMemberDaemonIsDown() throws Exception {
|
||||
int port = start(new FakeHerdr(), new FakeHerdr().healthy(false));
|
||||
HttpResponse<String> res = get(port, "/healthz");
|
||||
assertEquals(503, res.statusCode(),
|
||||
"a down MEMBER daemon must not be masked by a healthy lead — every spawn goes "
|
||||
+ "through the member daemon");
|
||||
}
|
||||
|
||||
@Test
|
||||
void healthzIsDegradedWhenOnlyTheLeadDaemonIsDown() throws Exception {
|
||||
int port = start(new FakeHerdr().healthy(false), new FakeHerdr());
|
||||
assertEquals(503, get(port, "/healthz").statusCode());
|
||||
}
|
||||
|
||||
@Test
|
||||
void healthzMakesExactlyOneCallWhenLeadAndMemberAreTheSameClient() throws Exception {
|
||||
// Single-daemon deployment (no memberHerdrSocket) — must be byte-for-byte the old
|
||||
// behaviour: one ping call, 200 on success.
|
||||
FakeHerdr shared = new FakeHerdr();
|
||||
int port = start(shared, shared);
|
||||
assertEquals(200, get(port, "/healthz").statusCode());
|
||||
long pings = shared.calls.stream().filter(c -> c.method().equals("ping")).count();
|
||||
assertEquals(1, pings, "single-daemon deployment must make exactly one ping call");
|
||||
}
|
||||
|
||||
@Test
|
||||
void sessionsMergesWorkspacesFromBothDaemons() throws Exception {
|
||||
FakeHerdr lead = new FakeHerdr();
|
||||
FakeHerdr member = new FakeHerdr().withWorkspace("w9", "member-only-workspace");
|
||||
int port = start(lead, member);
|
||||
HttpResponse<String> res = get(port, "/sessions");
|
||||
assertEquals(200, res.statusCode(), res.body());
|
||||
assertTrue(res.body().contains("member-only-workspace"),
|
||||
"GET /sessions must not silently drop the member daemon's workspaces");
|
||||
}
|
||||
|
||||
@Test
|
||||
void sessionsMakesExactlyOneWorkspaceListCallWhenLeadAndMemberAreTheSameClient() throws Exception {
|
||||
FakeHerdr shared = new FakeHerdr();
|
||||
int port = start(shared, shared);
|
||||
assertEquals(200, get(port, "/sessions").statusCode());
|
||||
long calls = shared.calls.stream().filter(c -> c.method().equals("workspace.list")).count();
|
||||
assertEquals(1, calls, "single-daemon deployment must call workspace.list exactly once");
|
||||
}
|
||||
|
||||
// ── CB-185 blocker 2: /healthz must report the MEMBER daemon's protocol too ────────────────
|
||||
|
||||
@Test
|
||||
void healthzReportsBothDaemonsWhenTheirProtocolsDiffer() throws Exception {
|
||||
FakeHerdr lead = new FakeHerdr().pingReports("0.8.0", 19);
|
||||
FakeHerdr member = new FakeHerdr().pingReports("0.7.0", 18);
|
||||
int port = start(lead, member);
|
||||
|
||||
HttpResponse<String> res = get(port, "/healthz");
|
||||
|
||||
assertEquals(200, res.statusCode(), res.body());
|
||||
assertTrue(res.body().contains("\"protocol\":19"),
|
||||
"the herdr key keeps reporting the LEAD's protocol, unchanged: " + res.body());
|
||||
assertTrue(res.body().contains("\"member\""), "a separate member key is present: " + res.body());
|
||||
assertTrue(res.body().contains("\"protocol\":18"),
|
||||
"the member key reports the member daemon's own protocol: " + res.body());
|
||||
assertTrue(res.body().contains("\"protocolMismatch\":true"),
|
||||
"a differing protocol is called out explicitly, not left to be spotted by eye: " + res.body());
|
||||
}
|
||||
|
||||
@Test
|
||||
void healthzReportsBothDaemonsWithNoMismatchWhenProtocolsMatch() throws Exception {
|
||||
int port = start(new FakeHerdr(), new FakeHerdr());
|
||||
|
||||
HttpResponse<String> res = get(port, "/healthz");
|
||||
|
||||
assertEquals(200, res.statusCode(), res.body());
|
||||
assertTrue(res.body().contains("\"member\""), "the member key is present whenever a second daemon "
|
||||
+ "is configured, even when the protocols happen to agree: " + res.body());
|
||||
assertFalse(res.body().contains("protocolMismatch"),
|
||||
"matching protocols must not raise a mismatch flag: " + res.body());
|
||||
}
|
||||
|
||||
@Test
|
||||
void healthzWithOneDaemonCarriesNoMemberOrMismatchKey() throws Exception {
|
||||
// The single-daemon deployment (no memberHerdrSocket) must see no change at all beyond the
|
||||
// historical body: no "member" key, no "protocolMismatch" key. (Map.of()'s own key order is
|
||||
// JVM-salted regardless of this fix, so this checks content, not exact key order.)
|
||||
FakeHerdr shared = new FakeHerdr();
|
||||
int port = start(shared, shared);
|
||||
|
||||
HttpResponse<String> res = get(port, "/healthz");
|
||||
|
||||
assertEquals(200, res.statusCode());
|
||||
assertTrue(res.body().contains("\"status\":\"ok\""), res.body());
|
||||
assertTrue(res.body().contains("\"protocol\":19"), res.body());
|
||||
assertTrue(res.body().contains("\"version\":\"0.8.0\""), res.body());
|
||||
assertFalse(res.body().contains("\"member\""), "no second daemon configured, so no member key: " + res.body());
|
||||
assertFalse(res.body().contains("protocolMismatch"), res.body());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,147 @@
|
||||
package dev.ltms.fleet.rest;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.Locale;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Collectors;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #252: the REST surface has no supported operator entry point of its own — it is the
|
||||
* documented fallback for when the MCP mount drops (see the operator wiki's REST page), and
|
||||
* nothing has ever checked a hand-written route list against it. Proof it drifts: the ticket
|
||||
* itself was filed with 14 routes, and {@link FleetApp} registers 15 — {@code GET
|
||||
* /member-credentials} (fleetd #111) shipped hours before the ticket and was already missing from
|
||||
* its list.
|
||||
*
|
||||
* <p>This test is modelled on {@code McpContractDocTest} (fleetd #114 / CB-609), which solved the
|
||||
* same shape of problem for the MCP tool catalogue: read the source text with a regex instead of
|
||||
* trusting a maintained copy. Here the "doc" is an explicit inventory written directly in this
|
||||
* test rather than a separate Markdown file, because the operator wiki page lives in a submodule
|
||||
* a worker cannot read reliably (see {@code CLAUDE.md} → Project addendum). Keeping the expected
|
||||
* list in the test means it still fails loudly the moment {@link FleetApp} changes, which is the
|
||||
* property that actually matters; a human keeps the wiki page in sync using the failure message
|
||||
* below as the diff.
|
||||
*
|
||||
* <p><b>It checks source text, not behaviour.</b> It reads {@link FleetApp}'s source for {@code
|
||||
* app.<verb>("<path>")} registrations and does not boot a server. It cannot catch a route that is
|
||||
* registered through some other mechanism entirely (a filter, a redirect) — only ones shaped like
|
||||
* the {@code app.get/post/delete/put/patch(...)} calls every route here actually uses.
|
||||
*/
|
||||
class RestRouteInventoryTest {
|
||||
|
||||
/** Tests run with the module directory as cwd. */
|
||||
private static final Path REST_SOURCE = Path.of("src/main/java/dev/ltms/fleet/rest/FleetApp.java");
|
||||
|
||||
/**
|
||||
* The REST surface as verified against {@link FleetApp} on 2026-09-03 (fleetd #252). Update
|
||||
* this list AND the operator wiki's REST-surface entry together whenever a route is added,
|
||||
* removed, or renamed — never one without the other.
|
||||
*
|
||||
* <p>{@code /mcp} is deliberately excluded: it is a raw Jetty {@code ServletHolder} mount
|
||||
* (see {@code FleetApp.build()}, around the {@code modifyServletContextHandler} call), not an
|
||||
* {@code app.<verb>(...)} route, so it is a different registration mechanism and this test's
|
||||
* regex does not — and should not — see it. If {@code /mcp} ever moves to a Javalin route,
|
||||
* add it here explicitly rather than relying on the regex to pick it up by accident.
|
||||
*/
|
||||
private static final Set<String> EXPECTED_ROUTES = Set.of(
|
||||
"GET /healthz",
|
||||
"GET /metrics",
|
||||
"GET /sessions",
|
||||
"GET /agents",
|
||||
"GET /members",
|
||||
"GET /profiles",
|
||||
"GET /member-credentials",
|
||||
"POST /members",
|
||||
"DELETE /members/{paneId}",
|
||||
"POST /sessions/{id}/message",
|
||||
"POST /sessions/{id}/reply",
|
||||
"GET /sessions/{id}/replies",
|
||||
"POST /sessions/{id}/ask",
|
||||
"GET /sessions/{id}/status",
|
||||
"GET /tasks/{ticket}"
|
||||
);
|
||||
|
||||
private static final Pattern ROUTE_CALL =
|
||||
Pattern.compile("app\\.(get|post|delete|put|patch)\\(\\s*\"([^\"]+)\"");
|
||||
|
||||
/**
|
||||
* Drop whole-line comments before scraping. Without this the scrape reads commented-out code as
|
||||
* live: a registration disabled with {@code //} still matched, so the route stayed in the
|
||||
* inventory while the server no longer served it — a silent false PASS, measured on 2026-09-03
|
||||
* by commenting out {@code app.get("/tasks/{ticket}", ...)} and watching this test stay green.
|
||||
* Deleting the same line was caught correctly, so only the commented-out shape was blind.
|
||||
*
|
||||
* <p>Only lines whose first non-blank characters are {@code //}, {@code *} or {@code /*} are
|
||||
* dropped — deliberately NOT every {@code //} anywhere on a line, because that would also cut a
|
||||
* string literal containing {@code //} (a URL) and could silently delete a real registration
|
||||
* sharing that line. The remaining gap is a trailing comment on the same line as real code; no
|
||||
* registration in this file has that shape, and the vacuity test below would catch a scrape that
|
||||
* lost registrations wholesale.
|
||||
*/
|
||||
private static String withoutCommentLines(String source) {
|
||||
return source.lines()
|
||||
.filter(line -> {
|
||||
String t = line.stripLeading();
|
||||
return !(t.startsWith("//") || t.startsWith("*") || t.startsWith("/*"));
|
||||
})
|
||||
.collect(Collectors.joining("\n"));
|
||||
}
|
||||
|
||||
/**
|
||||
* Every {@code app.<verb>("<path>")} call in {@link FleetApp}'s source, as {@code "VERB path"}.
|
||||
* This matches inside an {@code if (...) { ... }} block just as well as a top-level statement —
|
||||
* the regex only looks for the call shape, not its surrounding control flow — which is what
|
||||
* catches {@code GET /metrics} (registered conditionally on {@code metrics != null}).
|
||||
*/
|
||||
private static Set<String> routesTheServerRegisters() throws Exception {
|
||||
Matcher m = ROUTE_CALL.matcher(withoutCommentLines(Files.readString(REST_SOURCE)));
|
||||
Set<String> found = new LinkedHashSet<>();
|
||||
while (m.find()) {
|
||||
found.add(m.group(1).toUpperCase(Locale.ROOT) + " " + m.group(2));
|
||||
}
|
||||
return found;
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] FleetApp registers exactly the documented REST route inventory")
|
||||
void theRegisteredRoutesMatchTheExpectedInventory() throws Exception {
|
||||
Set<String> actual = routesTheServerRegisters();
|
||||
|
||||
Set<String> added = new LinkedHashSet<>(actual);
|
||||
added.removeAll(EXPECTED_ROUTES);
|
||||
Set<String> removed = new LinkedHashSet<>(EXPECTED_ROUTES);
|
||||
removed.removeAll(actual);
|
||||
|
||||
assertTrue(added.isEmpty() && removed.isEmpty(),
|
||||
"FleetApp's registered REST routes no longer match this test's expected inventory. "
|
||||
+ "Added (in FleetApp, not in this test): " + added + ". "
|
||||
+ "Removed (in this test, not in FleetApp): " + removed + ". "
|
||||
+ "Update EXPECTED_ROUTES in RestRouteInventoryTest AND the operator wiki's "
|
||||
+ "REST-surface entry together — this is the fleetd #252 defect: the route "
|
||||
+ "list drifted for a month with nothing checking it. Do NOT weaken this test.");
|
||||
}
|
||||
|
||||
/**
|
||||
* The denominator guard (same shape as {@code McpContractDocTest}'s vacuity check). Pins that
|
||||
* the regex really is still finding registrations, so a scrape that silently stops matching
|
||||
* can't make the check above pass by finding nothing on both sides.
|
||||
*/
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] the route scrape is not vacuous — it found the expected count")
|
||||
void theScrapeActuallyFoundRoutes() throws Exception {
|
||||
Set<String> actual = routesTheServerRegisters();
|
||||
assertTrue(actual.size() >= EXPECTED_ROUTES.size(),
|
||||
"scraped only " + actual.size() + " route registration(s) from FleetApp (" + actual
|
||||
+ "), but this test expects at least " + EXPECTED_ROUTES.size()
|
||||
+ "; the app.<verb>(\"...\") scrape has stopped matching and the check above "
|
||||
+ "is now vacuous");
|
||||
}
|
||||
}
|
||||
@@ -17,6 +17,9 @@ public final class FakeWorktrees implements Worktrees {
|
||||
public record RemoveCall(String repoRoot, String worktreePath) {
|
||||
}
|
||||
|
||||
public record DeleteBranchCall(String repoRoot, String branch) {
|
||||
}
|
||||
|
||||
public record OverlayCall(String repoRoot, String worktreePath,
|
||||
List<String> requested, List<String> copied, List<String> skipWorktree) {
|
||||
}
|
||||
@@ -30,17 +33,30 @@ public final class FakeWorktrees implements Worktrees {
|
||||
public record PruneCall(String repoRoot, long minAgeMillis) {
|
||||
}
|
||||
|
||||
public record ShareCall(String repoRoot, String worktreePath) {
|
||||
}
|
||||
|
||||
private final List<AddCall> addCalls = new CopyOnWriteArrayList<>();
|
||||
private final List<RemoveCall> removeCalls = new CopyOnWriteArrayList<>();
|
||||
private final List<DeleteBranchCall> deleteBranchCalls = new CopyOnWriteArrayList<>();
|
||||
private final List<OverlayCall> overlayCalls = new CopyOnWriteArrayList<>();
|
||||
private final List<RepoRootCall> repoRootCalls = new CopyOnWriteArrayList<>();
|
||||
private final List<SnapshotCall> snapshotCalls = new CopyOnWriteArrayList<>();
|
||||
private final List<PruneCall> pruneCalls = new CopyOnWriteArrayList<>();
|
||||
private final List<ShareCall> shareCalls = new CopyOnWriteArrayList<>();
|
||||
/** Tags every {@code overlayParity}/{@code shareWithGroup} call in call order, so a test can
|
||||
* pin that sharing runs after the overlay copy (fleetd #185 stage 3). */
|
||||
private final List<String> overlayShareOrder = new CopyOnWriteArrayList<>();
|
||||
private final Set<String> existingPaths = ConcurrentHashMap.newKeySet();
|
||||
private final Set<String> trackedPaths = ConcurrentHashMap.newKeySet();
|
||||
/** Worktree paths that currently exist, mirroring GitWorktrees' {@code Files.exists} check for
|
||||
* the already-gone case (CB-576 review, fleetd #116). */
|
||||
private final Set<String> worktreePaths = ConcurrentHashMap.newKeySet();
|
||||
private final AtomicLong snapshotSeq = new AtomicLong();
|
||||
private volatile RuntimeException addFailure;
|
||||
private volatile RuntimeException snapshotFailure;
|
||||
private volatile RuntimeException removeFailure;
|
||||
private volatile RuntimeException overlayFailure;
|
||||
private volatile boolean dirty = false;
|
||||
private volatile String repoRoot = "/repo";
|
||||
private volatile String prefix = "/worktrees";
|
||||
@@ -88,6 +104,21 @@ public final class FakeWorktrees implements Worktrees {
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make subsequent {@link #remove} calls throw (fleetd #283: a stale index lock, a slow
|
||||
* filesystem, or {@code remove}'s own 30s exec timeout escaping the last, previously bare,
|
||||
* step of {@link SessionManager#release}). */
|
||||
public FakeWorktrees failRemove(String message) {
|
||||
this.removeFailure = new WorktreeException(message);
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make subsequent {@link #overlayParity} calls throw (simulates a post-{@code add()} spawn
|
||||
* failure — fleetd #283 defect 2 — so the {@code acquireWithWorktree} catch runs). */
|
||||
public FakeWorktrees failOverlay(String message) {
|
||||
this.overlayFailure = new WorktreeException(message);
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Configure the value returned by {@link #wipRefs}. */
|
||||
public FakeWorktrees withWipRefs(WipRefStats stats) {
|
||||
this.wipRefs = stats;
|
||||
@@ -108,21 +139,45 @@ public final class FakeWorktrees implements Worktrees {
|
||||
}
|
||||
// The branch already carries a unique nonce, so the derived path is distinct per acquire
|
||||
// without an extra counter — keep it a pure function of the branch the test can predict.
|
||||
return prefix + "/" + branch.replace('/', '_');
|
||||
String path = prefix + "/" + branch.replace('/', '_');
|
||||
worktreePaths.add(path);
|
||||
return path;
|
||||
}
|
||||
|
||||
/** Model an operator / {@code git worktree prune} removing the worktree before release. */
|
||||
public FakeWorktrees markGone(String worktreePath) {
|
||||
worktreePaths.remove(worktreePath);
|
||||
return this;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void remove(String repoRoot, String worktreePath) {
|
||||
removeCalls.add(new RemoveCall(repoRoot, worktreePath));
|
||||
if (removeFailure != null) {
|
||||
throw removeFailure;
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void deleteBranch(String repoRoot, String branch) {
|
||||
deleteBranchCalls.add(new DeleteBranchCall(repoRoot, branch));
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasUncommitted(String worktreePath) {
|
||||
// A path that was never added, or was marked gone, is reported clean, mirroring
|
||||
// GitWorktrees' already-gone guard — never an error, so teardown still completes.
|
||||
if (!worktreePaths.contains(worktreePath)) {
|
||||
return false;
|
||||
}
|
||||
return dirty;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void overlayParity(String repoRoot, String worktreePath, List<String> overlay) {
|
||||
if (overlayFailure != null) {
|
||||
throw overlayFailure;
|
||||
}
|
||||
List<String> copied = new java.util.ArrayList<>();
|
||||
List<String> skipped = new java.util.ArrayList<>();
|
||||
for (String rel : overlay) {
|
||||
@@ -136,6 +191,13 @@ public final class FakeWorktrees implements Worktrees {
|
||||
}
|
||||
overlayCalls.add(new OverlayCall(repoRoot, worktreePath, List.copyOf(overlay),
|
||||
List.copyOf(copied), List.copyOf(skipped)));
|
||||
overlayShareOrder.add("overlay:" + worktreePath);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void shareWithGroup(String repoRoot, String worktreePath) {
|
||||
shareCalls.add(new ShareCall(repoRoot, worktreePath));
|
||||
overlayShareOrder.add("share:" + worktreePath);
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -188,6 +250,14 @@ public final class FakeWorktrees implements Worktrees {
|
||||
return removeCalls.isEmpty() ? null : removeCalls.getLast();
|
||||
}
|
||||
|
||||
public List<DeleteBranchCall> deleteBranchCalls() {
|
||||
return List.copyOf(deleteBranchCalls);
|
||||
}
|
||||
|
||||
public DeleteBranchCall lastDeleteBranch() {
|
||||
return deleteBranchCalls.isEmpty() ? null : deleteBranchCalls.getLast();
|
||||
}
|
||||
|
||||
public OverlayCall lastOverlay() {
|
||||
return overlayCalls.isEmpty() ? null : overlayCalls.getLast();
|
||||
}
|
||||
@@ -203,4 +273,17 @@ public final class FakeWorktrees implements Worktrees {
|
||||
public SnapshotCall lastSnapshot() {
|
||||
return snapshotCalls.isEmpty() ? null : snapshotCalls.getLast();
|
||||
}
|
||||
|
||||
public List<ShareCall> shareCalls() {
|
||||
return List.copyOf(shareCalls);
|
||||
}
|
||||
|
||||
public ShareCall lastShare() {
|
||||
return shareCalls.isEmpty() ? null : shareCalls.getLast();
|
||||
}
|
||||
|
||||
/** Call-order tags ({@code "overlay:<path>"}/{@code "share:<path>"}) — see field javadoc. */
|
||||
public List<String> overlayShareOrder() {
|
||||
return List.copyOf(overlayShareOrder);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,16 +1,27 @@
|
||||
package dev.ltms.fleet.session;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.LoggerContext;
|
||||
import ch.qos.logback.classic.spi.IThrowableProxy;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
@@ -352,6 +363,237 @@ class GitWorktreesTest {
|
||||
assertEquals("worktree origin contains HTTPS user info; refusing provision", error.getMessage());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #274. {@code add()} creates the worktree and its branch, then runs several more steps
|
||||
* that can throw — {@code requireCredentialFreeHttpsOrigin} among them, an intended security
|
||||
* refusal, not an IO accident. Before the fix, any exception from those later steps left
|
||||
* {@code add()} never returning, so its caller never learned the path and the worktree
|
||||
* directory plus its branch leaked on disk forever with nothing tracking them.
|
||||
*
|
||||
* <p>This drives the exact same {@code afterWorktreeAdded} test seam as
|
||||
* {@link #provisioningRefusesAWorktreeWhoseOriginStillHasHttpsUserInfo} — a mutation applied
|
||||
* right after {@code git worktree add}, so the step that throws
|
||||
* ({@code requireCredentialFreeHttpsOrigin}, reached moments later inside {@code add()} itself)
|
||||
* runs strictly after the worktree and branch already exist, not downstream of {@code add()}
|
||||
* in some other caller. {@code afterWorktreeAdded} also hands back the created path, so the
|
||||
* assertions below don't have to guess the generated nonce.
|
||||
*/
|
||||
@Test
|
||||
void addCleansUpTheWorktreeAndBranchWhenAPostCreationStepThrows(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
git(repo, "remote", "add", "origin", "https://git.ltms.dev/akb/kb.git");
|
||||
String branch = "cb-274-leak";
|
||||
AtomicReference<String> createdPath = new AtomicReference<>();
|
||||
GitWorktrees worktrees = new GitWorktrees(tmp.resolve("wts").toString(), worktreePath -> {
|
||||
createdPath.set(worktreePath);
|
||||
try {
|
||||
git(Path.of(worktreePath), "remote", "set-url", "origin",
|
||||
"https://synthetic-test-token@git.ltms.dev/akb/kb.git");
|
||||
} catch (Exception e) {
|
||||
throw new RuntimeException(e);
|
||||
}
|
||||
});
|
||||
|
||||
assertThrows(WorktreeException.class, () -> worktrees.add(repo.toString(), branch, "HEAD"));
|
||||
|
||||
assertNotNull(createdPath.get(), "afterWorktreeAdded must have run with the created path");
|
||||
assertFalse(Files.exists(Path.of(createdPath.get())),
|
||||
"the worktree directory leaked after a post-creation step threw");
|
||||
String heads = forEachRef(repo, "refs/heads/" + branch);
|
||||
assertTrue(heads.isBlank(), "the branch leaked after a post-creation step threw:\n" + heads);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #309. The add runner is a narrow seam for the case where Git has created state but
|
||||
* the caller then kills the process. The runner first performs the real add in this throwaway
|
||||
* repo, then throws the same kind of exception that {@link GitWorktrees#exec} uses for a timeout.
|
||||
* This proves the failure path cleans both real Git objects without waiting for a slow checkout.
|
||||
*/
|
||||
@Test
|
||||
void addCleansUpWhenTheWorktreeAddRunnerFailsAfterCreatingState(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
String branch = "cb-309-timeout";
|
||||
WorktreeException timeout = new WorktreeException("command timed out: synthetic git worktree add");
|
||||
AtomicReference<String> createdPath = new AtomicReference<>();
|
||||
GitWorktrees worktrees = new GitWorktrees(tmp.resolve("wts").toString(), null, null, null, command -> {
|
||||
createdPath.set(command[5]);
|
||||
try {
|
||||
git(repo, "worktree", "add", command[5], "-b", command[7], command[8]);
|
||||
} catch (Exception e) {
|
||||
throw new AssertionError("test setup could not create the worktree", e);
|
||||
}
|
||||
throw timeout;
|
||||
});
|
||||
|
||||
WorktreeException thrown = assertThrows(WorktreeException.class,
|
||||
() -> worktrees.add(repo.toString(), branch, "HEAD"));
|
||||
|
||||
assertSame(timeout, thrown, "cleanup must not replace the add failure");
|
||||
assertNotNull(createdPath.get(), "the add runner must receive the worktree path");
|
||||
assertFalse(Files.exists(Path.of(createdPath.get())),
|
||||
"the worktree directory leaked after the add runner failed");
|
||||
assertFalse(refExists(repo, "refs/heads/" + branch), "the branch leaked after the add runner failed");
|
||||
}
|
||||
|
||||
/** An ordinary Git refusal must not delete the existing branch or log a cleanup warning. */
|
||||
@Test
|
||||
void addFailureBeforeCreatingAWorktreeIsQuiet(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
String branch = "already-exists";
|
||||
git(repo, "branch", branch);
|
||||
|
||||
assertThrows(WorktreeException.class, () -> new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), branch, "HEAD"));
|
||||
|
||||
assertTrue(refExists(repo, "refs/heads/" + branch), "the existing branch must remain");
|
||||
assertTrue(capturedMessages().isEmpty(), "an ordinary Git refusal logged a warning: " + capturedMessages());
|
||||
}
|
||||
|
||||
// ---- CB-189: broader remote-URL coverage — every remote, both fetch and push URLs, any
|
||||
// non-SSH scheme. Reporting only, additive to the origin/https strip-and-refuse tests above. ----
|
||||
|
||||
private Logger reportingLogger;
|
||||
private ListAppender<ILoggingEvent> reportingAppender;
|
||||
|
||||
/** {@link GitWorktrees}'s own logger, captured fresh for each test so assertions never see a
|
||||
* message left over from a previous test. */
|
||||
@BeforeEach
|
||||
void attachReportingLogCapture() {
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
reportingLogger = ctx.getLogger(GitWorktrees.class);
|
||||
reportingLogger.setLevel(Level.WARN);
|
||||
reportingAppender = new ListAppender<>();
|
||||
reportingAppender.setContext(ctx);
|
||||
reportingAppender.start();
|
||||
reportingLogger.addAppender(reportingAppender);
|
||||
}
|
||||
|
||||
@AfterEach
|
||||
void detachReportingLogCapture() {
|
||||
reportingLogger.detachAppender(reportingAppender);
|
||||
}
|
||||
|
||||
private List<String> capturedMessages() {
|
||||
return reportingAppender.list.stream().map(ILoggingEvent::getFormattedMessage).toList();
|
||||
}
|
||||
|
||||
/** Asserts {@code secret} appears in no captured message, and in no attached exception's
|
||||
* message either — the constraint is that a credential must never reach a log, however it
|
||||
* would have gotten there. */
|
||||
private void assertNoLeak(String secret) {
|
||||
for (ILoggingEvent event : reportingAppender.list) {
|
||||
assertFalse(event.getFormattedMessage().contains(secret),
|
||||
"log message leaked a credential (" + secret + "): " + event.getFormattedMessage());
|
||||
IThrowableProxy thrown = event.getThrowableProxy();
|
||||
if (thrown != null && thrown.getMessage() != null) {
|
||||
assertFalse(thrown.getMessage().contains(secret),
|
||||
"logged exception leaked a credential (" + secret + "): " + thrown.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Gap 1: only {@code origin} was ever inspected. A credential on any other remote's fetch URL
|
||||
* must now be reported. */
|
||||
@Test
|
||||
void aCredentialedUrlOnANonOriginRemoteIsReported(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
git(repo, "remote", "add", "origin", "https://git.ltms.dev/akb/kb.git");
|
||||
git(repo, "remote", "add", "upstream", "https://leaky-upstream-token@git.ltms.dev/akb/kb.git");
|
||||
|
||||
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-189-a", "HEAD");
|
||||
|
||||
List<String> messages = capturedMessages();
|
||||
assertTrue(messages.stream().anyMatch(m -> m.contains("remote=upstream")),
|
||||
"expected a report naming the leaking non-origin remote:\n" + messages);
|
||||
assertNoLeak("leaky-upstream-token");
|
||||
assertNoLeak("https://leaky-upstream-token@git.ltms.dev/akb/kb.git");
|
||||
assertNoLeak("git.ltms.dev");
|
||||
}
|
||||
|
||||
/** Gap 2: push URLs were never inspected. A credential visible only on {@code pushurl} — the
|
||||
* fetch URL for the same remote stays clean — must now be reported. */
|
||||
@Test
|
||||
void aCredentialedPushUrlIsReported(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
git(repo, "remote", "add", "origin", "https://git.ltms.dev/akb/kb.git");
|
||||
git(repo, "remote", "add", "mirror", "https://git.ltms.dev/akb/mirror.git");
|
||||
git(repo, "remote", "set-url", "--push", "mirror",
|
||||
"https://leaky-push-token@git.ltms.dev/akb/mirror.git");
|
||||
|
||||
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-189-b", "HEAD");
|
||||
|
||||
List<String> messages = capturedMessages();
|
||||
assertTrue(messages.stream().anyMatch(m -> m.contains("remote=mirror")),
|
||||
"expected a report naming the remote with the leaking pushurl:\n" + messages);
|
||||
assertNoLeak("leaky-push-token");
|
||||
assertNoLeak("https://leaky-push-token@git.ltms.dev/akb/mirror.git");
|
||||
assertNoLeak("git.ltms.dev");
|
||||
}
|
||||
|
||||
/** Gap 3: only {@code https} was handled. A plain {@code http://user:pass@…} remote — worse
|
||||
* than https, not better — must now be reported. */
|
||||
@Test
|
||||
void anHttpUrlWithCredentialsIsReported(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
git(repo, "remote", "add", "origin", "https://git.ltms.dev/akb/kb.git");
|
||||
git(repo, "remote", "add", "insecure", "http://plainuser:plainpass@git.ltms.dev/akb/kb.git");
|
||||
|
||||
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-189-c", "HEAD");
|
||||
|
||||
List<String> messages = capturedMessages();
|
||||
assertTrue(messages.stream().anyMatch(m -> m.contains("remote=insecure")),
|
||||
"expected a report for the credentialed plain-http remote:\n" + messages);
|
||||
assertNoLeak("plainuser");
|
||||
assertNoLeak("plainpass");
|
||||
assertNoLeak("plainuser:plainpass");
|
||||
assertNoLeak("git.ltms.dev");
|
||||
}
|
||||
|
||||
/** A normal {@code ssh://} remote and a credential-free {@code https://} remote must produce no
|
||||
* report at all — the check must not cry wolf on ordinary, safe configuration. */
|
||||
@Test
|
||||
void anSshRemoteAndACleanHttpsRemoteProduceNoReport(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
git(repo, "remote", "add", "origin", "ssh://git@git.ltms.dev:2224/akb/kb.git");
|
||||
git(repo, "remote", "add", "clean", "https://git.ltms.dev/akb/kb.git");
|
||||
|
||||
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-189-d", "HEAD");
|
||||
|
||||
assertTrue(reportingAppender.list.isEmpty(),
|
||||
"expected no report for an ssh remote and a credential-free https remote, got:\n"
|
||||
+ capturedMessages());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-189 review fix. Even a FAILING command that read a credential onto its stdout must never
|
||||
* let that value reach the thrown {@link WorktreeException}'s message — this is gap 4 from the
|
||||
* CB-189 issue, and the reason {@link GitWorktrees#execRedacted} exists at all. Drives the
|
||||
* shared {@code exec}/{@code execRedacted} seam directly (it is package-private for exactly this,
|
||||
* the same way the {@code afterWorktreeAdded} constructor parameter is a test seam) with a
|
||||
* synthetic, non-git command whose stdout carries a marker — passed through the environment,
|
||||
* never through argv, so the marker cannot leak via the command line that IS always printed
|
||||
* unconditionally in the exception message — and which exits non-zero. This isolates the
|
||||
* redaction guarantee itself rather than depending on a specific git failure mode that happens to
|
||||
* echo a URL onto stdout before failing: none of the git subcommands this class actually runs was
|
||||
* found to have one (a corrupted config makes {@code git config --get} fail before it ever reads
|
||||
* the target key, so its output never carries the URL either). The marker is generated per-test
|
||||
* run and injected only by the test, never a real-looking credential, so even a failing assertion
|
||||
* could not itself print a secret.
|
||||
*/
|
||||
@Test
|
||||
void execRedactedNeverCopiesFailingCommandOutputIntoTheExceptionMessage(@TempDir Path tmp) {
|
||||
String marker = "cb189-marker-" + System.nanoTime();
|
||||
GitWorktrees worktrees = new GitWorktrees(tmp.toString());
|
||||
|
||||
WorktreeException thrown = assertThrows(WorktreeException.class, () -> worktrees.exec(
|
||||
Map.of("MARKER", marker), true, "sh", "-c", "echo \"$MARKER\"; exit 7"));
|
||||
|
||||
assertNotNull(thrown.getMessage());
|
||||
assertFalse(thrown.getMessage().contains(marker),
|
||||
"a failing command's captured stdout leaked into the exception message: "
|
||||
+ thrown.getMessage());
|
||||
}
|
||||
|
||||
/** Neutralizing must not look like work in progress, or a worker would commit it into its PR. */
|
||||
@Test
|
||||
void theNeutralizedConfigIsNotAPendingLocalModification(@TempDir Path tmp) throws Exception {
|
||||
@@ -511,6 +753,103 @@ class GitWorktreesTest {
|
||||
assertEquals("", status(Path.of(wt), ".autoenv"), ".autoenv still shows as modified");
|
||||
}
|
||||
|
||||
// ---- fleetd #134: isolateToolSurface must report what it neutralized (to the daemon operator's
|
||||
// log) and record it where the worker itself can read it (worktree-scoped git config), without
|
||||
// ever showing up in the worker's own `git status`. All drive the real provisioning path,
|
||||
// GitWorktrees#add, per criterion 5 — the whole provisioned worktree is what's under test here. ----
|
||||
|
||||
private static Path initRepoWithAllThreeConfigs(Path dir) throws Exception {
|
||||
Files.createDirectories(dir);
|
||||
git(dir, "init", "-q", "-b", "main");
|
||||
git(dir, "config", "user.email", "test@example.invalid");
|
||||
git(dir, "config", "user.name", "Test");
|
||||
Files.writeString(dir.resolve(".mcp.json"), WITH_SERVERS);
|
||||
Files.writeString(dir.resolve("opencode.json"), OPENCODE_WITH_FILE_REF);
|
||||
Files.writeString(dir.resolve(".autoenv"), AUTOENV_WITH_DIRECTIVE);
|
||||
Files.writeString(dir.resolve("README.md"), "seed\n");
|
||||
git(dir, "add", ".mcp.json", "opencode.json", ".autoenv", "README.md");
|
||||
git(dir, "commit", "-q", "-m", "seed");
|
||||
return dir;
|
||||
}
|
||||
|
||||
/** Criterion 1, all three present: the summary names the denominator and every neutralized file. */
|
||||
@Test
|
||||
void isolateToolSurfaceLogsAllThreeConfigsNeutralized(@TempDir Path tmp) throws Exception {
|
||||
reportingLogger.setLevel(Level.INFO);
|
||||
Path repo = initRepoWithAllThreeConfigs(tmp.resolve("repo"));
|
||||
|
||||
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-134-log-all", "HEAD");
|
||||
|
||||
assertTrue(capturedMessages().contains(
|
||||
"tool-surface isolation: neutralized 3 of 3 configs: .mcp.json, opencode.json, "
|
||||
+ ".autoenv — the worktree copy is a stub, not the repo's file; edit the "
|
||||
+ "real file in the primary checkout instead"),
|
||||
"expected the all-neutralized summary line, got:\n" + capturedMessages());
|
||||
}
|
||||
|
||||
/** Criterion 1, two absent: the summary must still name the denominator and say why. */
|
||||
@Test
|
||||
void isolateToolSurfaceLogsAbsentConfigsWithReason(@TempDir Path tmp) throws Exception {
|
||||
reportingLogger.setLevel(Level.INFO);
|
||||
Path repo = initRepo(tmp.resolve("repo")); // only .mcp.json + README committed
|
||||
|
||||
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-134-log-partial", "HEAD");
|
||||
|
||||
assertTrue(capturedMessages().contains(
|
||||
"tool-surface isolation: neutralized 1 of 3 configs: .mcp.json (opencode.json "
|
||||
+ "absent, .autoenv absent) — the worktree copy is a stub, not the repo's "
|
||||
+ "file; edit the real file in the primary checkout instead"),
|
||||
"expected the partial summary line, got:\n" + capturedMessages());
|
||||
}
|
||||
|
||||
/**
|
||||
* Criterion 2. The daemon's own log is invisible to the worker process — it never reads fleetd's
|
||||
* stdout. This is the mechanism the worker itself can query, from inside its own worktree, to
|
||||
* learn "this file is neutralized here, the repo's real file differs" instead of trusting what it
|
||||
* just read on disk.
|
||||
*/
|
||||
@Test
|
||||
void aWorkerCanDiscoverNeutralizedConfigsFromWorktreeScopedGitConfig(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepoWithAllThreeConfigs(tmp.resolve("repo"));
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-134-discover", "HEAD");
|
||||
|
||||
String recorded = gitOutput(Path.of(wt), "config", "--worktree", "--get-all", "fleet.neutralizedConfig");
|
||||
Set<String> files = new HashSet<>();
|
||||
for (String line : recorded.split("\\R")) {
|
||||
if (!line.isBlank()) {
|
||||
files.add(line.trim());
|
||||
}
|
||||
}
|
||||
assertEquals(Set.of(".mcp.json", "opencode.json", ".autoenv"), files,
|
||||
"the worker-readable record must name every neutralized file: " + recorded);
|
||||
|
||||
String note = gitOutput(Path.of(wt), "config", "--worktree", "--get", "fleet.neutralizedConfigNote").trim();
|
||||
assertTrue(note.contains("stub"), "note must say the worktree copy is a stub: " + note);
|
||||
assertTrue(note.contains("primary checkout"),
|
||||
"note must state the consequence — where to edit the real file instead: " + note);
|
||||
}
|
||||
|
||||
/**
|
||||
* Criterion 3. Whatever fleetd#134's worker-discovery mechanism writes must never appear as
|
||||
* untracked or modified in the worker's own `git status` — a worker that sees a stray file either
|
||||
* commits it by mistake or burns a turn asking about it. This runs the full porcelain status, not
|
||||
* a single-file check, so any leftover file anywhere in the worktree would fail it.
|
||||
*/
|
||||
@Test
|
||||
void aProvisionedWorktreeHasCleanGitStatusDespiteNeutralizedConfigRecordkeeping(@TempDir Path tmp)
|
||||
throws Exception {
|
||||
Path repo = initRepoWithAllThreeConfigs(tmp.resolve("repo"));
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-134-clean-status", "HEAD");
|
||||
|
||||
assertEquals("", fullStatus(Path.of(wt)),
|
||||
"a freshly provisioned worktree must show a clean `git status --porcelain`, including "
|
||||
+ "after the worker-readable neutralized-config record was written");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-578 stage C, acceptance criterion 1. A dirty worktree — a tracked edit plus a brand-new
|
||||
* untracked file, exactly the shape lost in CB-576 — must land in {@code refs/wip/<branch>}'s
|
||||
@@ -769,4 +1108,365 @@ class GitWorktreesTest {
|
||||
assertEquals(2, stats.count(), "two snapshot refs are reported");
|
||||
assertTrue(stats.costBytes() > 0, "the cost of the snapshots is a positive byte count");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #185 stage 3: a recording {@link java.util.function.Function} test seam stands in for
|
||||
* every process {@link GitWorktrees#shareWithGroup} would run — no real second OS user/group
|
||||
* exists on this host, so these are unit tests against that seam, not a live-group integration
|
||||
* test (out of scope per the ticket).
|
||||
*/
|
||||
private static List<String> joined(String[] command) {
|
||||
return List.of(command);
|
||||
}
|
||||
|
||||
/** Remove {@code path} and anything under it. Tolerates an already-absent path. */
|
||||
private static void deleteRecursively(Path path) throws Exception {
|
||||
if (!Files.exists(path)) {
|
||||
return;
|
||||
}
|
||||
if (Files.isDirectory(path)) {
|
||||
try (java.util.stream.Stream<Path> children = Files.list(path)) {
|
||||
for (Path child : children.toList()) {
|
||||
deleteRecursively(child);
|
||||
}
|
||||
}
|
||||
}
|
||||
Files.delete(path);
|
||||
}
|
||||
|
||||
/** {@code worktreeGroup} absent ⇒ zero processes spawned and no git config written. */
|
||||
@Test
|
||||
void shareWithGroupIsNoopWhenNoGroupConfigured(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
List<List<String>> recorded = new java.util.ArrayList<>();
|
||||
java.util.function.Function<String[], String> recordingRunner = cmd -> {
|
||||
recorded.add(joined(cmd));
|
||||
return "";
|
||||
};
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString(), null, _ -> {}, recordingRunner);
|
||||
|
||||
gitWorktrees.shareWithGroup(repo.toString(), repo.resolve("some-worktree").toString());
|
||||
|
||||
assertTrue(recorded.isEmpty(), "no group configured must spawn no process at all: " + recorded);
|
||||
}
|
||||
|
||||
/** A configured group runs {@code git config core.sharedRepository group} first, then
|
||||
* chgrp/chmod/setgid over every path {@link GitWorktrees#shareWithGroup} documents. */
|
||||
@Test
|
||||
void shareWithGroupRunsConfigThenChgrpChmodSetgidPerPath(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
String repoRoot = repo.toString();
|
||||
Path worktree = Files.createDirectories(repo.resolve("some-worktree"));
|
||||
String worktreePath = worktree.toString();
|
||||
Files.createDirectories(repo.resolve(".git/worktrees"));
|
||||
List<List<String>> recorded = new java.util.ArrayList<>();
|
||||
java.util.function.Function<String[], String> recordingRunner = cmd -> {
|
||||
recorded.add(joined(cmd));
|
||||
// What real git answers for an ordinary (non-linked) checkout: relative to repoRoot.
|
||||
return List.of(cmd).contains("--git-common-dir") ? ".git\n" : "";
|
||||
};
|
||||
GitWorktrees gitWorktrees =
|
||||
new GitWorktrees(tmp.resolve("wts").toString(), "devteam", _ -> {}, recordingRunner);
|
||||
|
||||
gitWorktrees.shareWithGroup(repoRoot, worktreePath);
|
||||
|
||||
assertEquals(List.of("git", "-C", repoRoot, "config", "core.sharedRepository", "group"), recorded.get(0),
|
||||
"core.sharedRepository must be set first, so it keeps working after the one-time fix-up");
|
||||
assertTrue(recorded.contains(List.of("git", "-C", repoRoot, "rev-parse", "--git-common-dir")),
|
||||
"the git dir must be asked for, never hardcoded as <repoRoot>/.git — that is a FILE "
|
||||
+ "when the checkout is itself a linked worktree: " + recorded);
|
||||
|
||||
for (String dir : List.of(worktreePath, repoRoot + "/.git/objects", repoRoot + "/.git/refs",
|
||||
repoRoot + "/.git/logs", repoRoot + "/.git/worktrees")) {
|
||||
assertTrue(recorded.contains(List.of("chgrp", "-R", "devteam", dir)), "missing chgrp -R for " + dir);
|
||||
assertTrue(recorded.contains(List.of("chmod", "-R", "g+rwX", dir)), "missing chmod -R for " + dir);
|
||||
assertTrue(recorded.contains(List.of("find", dir, "-type", "d", "-exec", "chmod", "g+s", "{}", "+")),
|
||||
"missing setgid find pass for " + dir);
|
||||
}
|
||||
// packed-refs does not exist in a freshly-init'd repo (only git gc / pack-refs creates it) —
|
||||
// tolerated absence, so it must not appear at all: no recursive/-R treatment for a plain file.
|
||||
String packedRefs = repoRoot + "/.git/packed-refs";
|
||||
assertTrue(recorded.stream().noneMatch(c -> c.contains(packedRefs)),
|
||||
"packed-refs is absent here and must be skipped, not chgrp'd: " + recorded);
|
||||
}
|
||||
|
||||
/**
|
||||
* A path that does not exist is skipped, never handed to {@code chgrp}. {@code .git/logs} is
|
||||
* absent whenever {@code core.logAllRefUpdates} is false or no ref has been updated yet, and
|
||||
* {@code chgrp} on a missing path exits non-zero — which would fail EVERY provisioning spawn
|
||||
* with a message blaming a group that is in fact fine.
|
||||
*/
|
||||
@Test
|
||||
void shareWithGroupSkipsPathsThatDoNotExist(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
String repoRoot = repo.toString();
|
||||
deleteRecursively(repo.resolve(".git/logs"));
|
||||
assertFalse(Files.exists(repo.resolve(".git/logs")), "fixture: .git/logs must be gone");
|
||||
List<List<String>> recorded = new java.util.ArrayList<>();
|
||||
java.util.function.Function<String[], String> recordingRunner = cmd -> {
|
||||
recorded.add(joined(cmd));
|
||||
return List.of(cmd).contains("--git-common-dir") ? ".git\n" : "";
|
||||
};
|
||||
GitWorktrees gitWorktrees =
|
||||
new GitWorktrees(tmp.resolve("wts").toString(), "devteam", _ -> {}, recordingRunner);
|
||||
|
||||
gitWorktrees.shareWithGroup(repoRoot, repo.resolve("no-such-worktree").toString());
|
||||
|
||||
String logs = repoRoot + "/.git/logs";
|
||||
assertTrue(recorded.stream().noneMatch(c -> c.contains(logs)),
|
||||
"a missing .git/logs must be skipped, not chgrp'd: " + recorded);
|
||||
assertTrue(recorded.stream().noneMatch(c -> c.contains(repo.resolve("no-such-worktree").toString())),
|
||||
"a missing worktree path must be skipped too: " + recorded);
|
||||
assertTrue(recorded.contains(List.of("chgrp", "-R", "devteam", repoRoot + "/.git/objects")),
|
||||
"paths that DO exist are still shared: " + recorded);
|
||||
}
|
||||
|
||||
/**
|
||||
* The git store is located by {@code rev-parse --git-common-dir}, not by appending
|
||||
* {@code /.git}. When git answers with an absolute path — what it does for a linked worktree,
|
||||
* where {@code <repoRoot>/.git} is a file — every shared path must follow that answer.
|
||||
*/
|
||||
@Test
|
||||
void shareWithGroupFollowsAnAbsoluteGitCommonDir(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Path realGitDir = repo.resolve(".git");
|
||||
List<List<String>> recorded = new java.util.ArrayList<>();
|
||||
java.util.function.Function<String[], String> recordingRunner = cmd -> {
|
||||
recorded.add(joined(cmd));
|
||||
return List.of(cmd).contains("--git-common-dir") ? realGitDir + "\n" : "";
|
||||
};
|
||||
GitWorktrees gitWorktrees =
|
||||
new GitWorktrees(tmp.resolve("wts").toString(), "devteam", _ -> {}, recordingRunner);
|
||||
|
||||
gitWorktrees.shareWithGroup(tmp.resolve("some/linked/worktree").toString(),
|
||||
repo.resolve("wt").toString());
|
||||
|
||||
assertTrue(recorded.contains(List.of("chgrp", "-R", "devteam", realGitDir + "/objects")),
|
||||
"objects must be taken from the reported common dir, not <repoRoot>/.git: " + recorded);
|
||||
}
|
||||
|
||||
/** {@code packed-refs}, when present, is chgrp/chmod'd but never setgid'd (it is a file, not a dir). */
|
||||
@Test
|
||||
void shareWithGroupIncludesPackedRefsWhenPresent(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
String repoRoot = repo.toString();
|
||||
Path packedRefsPath = repo.resolve(".git/packed-refs");
|
||||
Files.writeString(packedRefsPath, "");
|
||||
List<List<String>> recorded = new java.util.ArrayList<>();
|
||||
java.util.function.Function<String[], String> recordingRunner = cmd -> {
|
||||
recorded.add(joined(cmd));
|
||||
return "";
|
||||
};
|
||||
GitWorktrees gitWorktrees =
|
||||
new GitWorktrees(tmp.resolve("wts").toString(), "devteam", _ -> {}, recordingRunner);
|
||||
|
||||
gitWorktrees.shareWithGroup(repoRoot, repo.resolve("some-worktree").toString());
|
||||
|
||||
String packedRefs = packedRefsPath.toString();
|
||||
assertTrue(recorded.contains(List.of("chgrp", "devteam", packedRefs)),
|
||||
"packed-refs must be chgrp'd non-recursively when present: " + recorded);
|
||||
assertTrue(recorded.contains(List.of("chmod", "g+rwX", packedRefs)),
|
||||
"packed-refs must be chmod'd non-recursively when present: " + recorded);
|
||||
assertTrue(recorded.stream().noneMatch(c -> c.contains("find") && c.contains(packedRefs)),
|
||||
"packed-refs (a file) must never get the recursive setgid pass: " + recorded);
|
||||
}
|
||||
|
||||
/** A group that does not exist (or that the operator is not a member of) fails loudly, naming it. */
|
||||
@Test
|
||||
void shareWithGroupThrowsNamingTheGroupWhenChgrpFails(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString(), "cb185-nonexistent-group-zz");
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-185-share", "HEAD");
|
||||
|
||||
WorktreeException e = assertThrows(WorktreeException.class,
|
||||
() -> gitWorktrees.shareWithGroup(repo.toString(), wt));
|
||||
assertTrue(e.getMessage().contains("cb185-nonexistent-group-zz"),
|
||||
"exception must name the missing/refused group: " + e.getMessage());
|
||||
}
|
||||
|
||||
// --- fleetd #224: worktreeRoot itself must be group-traversable, established in add() ---------
|
||||
|
||||
/**
|
||||
* {@code add} must make the worktree ROOT itself group-traversable when a group is configured —
|
||||
* established once here, where the root is created, never at a use site (never inside a
|
||||
* launcher). A recording runner stands in for chgrp/chmod, the same seam
|
||||
* {@link #shareWithGroupRunsConfigThenChgrpChmodSetgidPerPath} uses for the per-worktree share.
|
||||
*/
|
||||
@Test
|
||||
void addSharesWorktreeRootWithGroupWhenConfigured(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Path root = tmp.resolve("wts");
|
||||
List<List<String>> recorded = new java.util.ArrayList<>();
|
||||
java.util.function.Function<String[], String> recordingRunner = cmd -> {
|
||||
recorded.add(joined(cmd));
|
||||
return "";
|
||||
};
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(root.toString(), "devteam", _ -> {}, recordingRunner);
|
||||
|
||||
gitWorktrees.add(repo.toString(), "cb-224-branch", "HEAD");
|
||||
|
||||
assertTrue(recorded.contains(List.of("chgrp", "devteam", root.toString())),
|
||||
"the worktree root itself must be chgrp'd to the configured group: " + recorded);
|
||||
assertTrue(recorded.contains(List.of("chmod", "g+x", root.toString())),
|
||||
"the worktree root itself must gain group-execute so a different-uid member can "
|
||||
+ "traverse into it: " + recorded);
|
||||
}
|
||||
|
||||
/** {@code worktreeGroup} unset (today's only live mode) ⇒ {@code add} spawns no share process
|
||||
* for the root at all — behaviour must be byte-identical to before fleetd #224. */
|
||||
@Test
|
||||
void addSharesNothingForTheRootWhenNoGroupConfigured(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Path root = tmp.resolve("wts");
|
||||
List<List<String>> recorded = new java.util.ArrayList<>();
|
||||
java.util.function.Function<String[], String> recordingRunner = cmd -> {
|
||||
recorded.add(joined(cmd));
|
||||
return "";
|
||||
};
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(root.toString(), null, _ -> {}, recordingRunner);
|
||||
|
||||
gitWorktrees.add(repo.toString(), "cb-224-nogroup", "HEAD");
|
||||
|
||||
assertTrue(recorded.isEmpty(), "no group configured must spawn no share process for the "
|
||||
+ "root at all: " + recorded);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #224, acceptance criterion 2/3: when the root cannot be made group-traversable — here
|
||||
* because the configured group does not exist, the same real-failure shape
|
||||
* {@link #shareWithGroupThrowsNamingTheGroupWhenChgrpFails} drives for the per-worktree share —
|
||||
* the spawn is refused with a message naming the root, its current mode, and the group. This
|
||||
* drives the REAL {@code chgrp} (no recording runner), and the refusal happens before {@code git
|
||||
* worktree add} ever runs, so no partial worktree is left behind either.
|
||||
*/
|
||||
@Test
|
||||
void addRefusesWhenWorktreeRootCannotBeMadeGroupTraversable(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Path root = tmp.resolve("wts");
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(root.toString(), "cb224-nonexistent-group-zz");
|
||||
|
||||
WorktreeException e = assertThrows(WorktreeException.class,
|
||||
() -> gitWorktrees.add(repo.toString(), "cb-224-refuse", "HEAD"));
|
||||
|
||||
assertTrue(e.getMessage().contains(root.toString()),
|
||||
"refusal must name the worktree root: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("cb224-nonexistent-group-zz"),
|
||||
"refusal must name the missing/refused group: " + e.getMessage());
|
||||
assertTrue(Files.isDirectory(root), "the root is created before the group check runs");
|
||||
String mode = java.nio.file.attribute.PosixFilePermissions.toString(Files.getPosixFilePermissions(root));
|
||||
assertTrue(e.getMessage().contains(mode),
|
||||
"refusal must name the root's current mode (" + mode + "): " + e.getMessage());
|
||||
try (java.util.stream.Stream<Path> children = Files.list(root)) {
|
||||
assertTrue(children.findAny().isEmpty(), "no worktree must be left behind under the root: "
|
||||
+ "the refusal must happen before `git worktree add` ever runs");
|
||||
}
|
||||
}
|
||||
|
||||
// ---- fleetd #134 / #148 point 3: overlayParity must report what it did, and marking a tracked
|
||||
// file --skip-worktree must say the file can no longer be committed from this worktree. Drives
|
||||
// overlayParity directly against a real worktree (git worktree add, no GitWorktrees#add) so these
|
||||
// tests are independent of origin/credential-helper provisioning, which is not under test here. ----
|
||||
|
||||
/** A bare worktree, sibling to {@code repo}, created with plain git — the target overlayParity
|
||||
* copies into. Deliberately not {@link GitWorktrees#add}: that method does unrelated
|
||||
* provisioning (origin rewrite, .mcp.json neutralization, credential helper) that would only
|
||||
* add noise to the log assertions below. */
|
||||
private static Path bareWorktree(Path repo, Path wtDir, String branch) throws Exception {
|
||||
git(repo, "worktree", "add", "-q", wtDir.toString(), "-b", branch, "HEAD");
|
||||
return wtDir;
|
||||
}
|
||||
|
||||
/** Acceptance criterion 1: only the configured candidates land in the worktree — nothing else
|
||||
* from the source tree leaks in alongside them. */
|
||||
@Test
|
||||
void overlayParityCopiesExactlyTheConfiguredFilesAndNothingElse(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Files.writeString(repo.resolve(".env"), "A=1\n");
|
||||
Files.writeString(repo.resolve(".envrc"), "export A=1\n");
|
||||
Files.writeString(repo.resolve("not-overlaid.txt"), "must not be copied\n");
|
||||
Path wt = bareWorktree(repo, tmp.resolve("wt"), "cb134-exact");
|
||||
|
||||
new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.overlayParity(repo.toString(), wt.toString(), List.of(".env", ".envrc"));
|
||||
|
||||
assertEquals("A=1\n", Files.readString(wt.resolve(".env")));
|
||||
assertEquals("export A=1\n", Files.readString(wt.resolve(".envrc")));
|
||||
assertFalse(Files.exists(wt.resolve("not-overlaid.txt")),
|
||||
"overlayParity must copy only the configured candidates, not the whole source tree");
|
||||
}
|
||||
|
||||
/** Criterion 2: both candidates present — the summary line names both and the denominator. */
|
||||
@Test
|
||||
void overlayParityLogsBothCopiedWhenBothCandidatesArePresent(@TempDir Path tmp) throws Exception {
|
||||
reportingLogger.setLevel(Level.INFO);
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Files.writeString(repo.resolve(".env"), "A=1\n");
|
||||
Files.writeString(repo.resolve(".envrc"), "export A=1\n");
|
||||
Path wt = bareWorktree(repo, tmp.resolve("wt"), "cb134-both");
|
||||
|
||||
new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.overlayParity(repo.toString(), wt.toString(), List.of(".env", ".envrc"));
|
||||
|
||||
assertTrue(capturedMessages().contains("parity overlay: copied 2 of 2 candidates: .env, .envrc"),
|
||||
"expected the both-copied summary line, got:\n" + capturedMessages());
|
||||
}
|
||||
|
||||
/** Criterion 2: one candidate present, one absent — the summary must name the copied file, the
|
||||
* denominator, and why the other candidate was not copied. */
|
||||
@Test
|
||||
void overlayParityLogsOneCopiedOneAbsent(@TempDir Path tmp) throws Exception {
|
||||
reportingLogger.setLevel(Level.INFO);
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Files.writeString(repo.resolve(".env"), "A=1\n");
|
||||
// .envrc deliberately not created — the absent candidate.
|
||||
Path wt = bareWorktree(repo, tmp.resolve("wt"), "cb134-partial");
|
||||
|
||||
new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.overlayParity(repo.toString(), wt.toString(), List.of(".env", ".envrc"));
|
||||
|
||||
assertTrue(Files.exists(wt.resolve(".env")));
|
||||
assertFalse(Files.exists(wt.resolve(".envrc")));
|
||||
assertTrue(capturedMessages().contains("parity overlay: copied 1 of 2 candidates: .env (.envrc absent)"),
|
||||
"expected the copied/absent summary line, got:\n" + capturedMessages());
|
||||
}
|
||||
|
||||
/** Criterion 3: a tracked candidate is marked --skip-worktree, and that must be named in the log
|
||||
* with the consequence spelled out — a worker editing it afterward finds git ignoring the
|
||||
* change, silently, unless this line told it so beforehand. */
|
||||
@Test
|
||||
void overlayParityLogsSkipWorktreeConsequenceForATrackedFile(@TempDir Path tmp) throws Exception {
|
||||
reportingLogger.setLevel(Level.INFO);
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Files.writeString(repo.resolve(".env"), "A=1\n");
|
||||
git(repo, "add", ".env");
|
||||
git(repo, "commit", "-q", "-m", "track env");
|
||||
Path wt = bareWorktree(repo, tmp.resolve("wt"), "cb134-tracked");
|
||||
// Change the source after the worktree checkout, so the overlay copy actually overwrites it.
|
||||
Files.writeString(repo.resolve(".env"), "A=2\n");
|
||||
|
||||
new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.overlayParity(repo.toString(), wt.toString(), List.of(".env"));
|
||||
|
||||
assertEquals("A=2\n", Files.readString(wt.resolve(".env")));
|
||||
assertEquals("", status(wt, ".env"),
|
||||
"the skip-worktree'd file must not show as modified even though its content changed");
|
||||
assertTrue(capturedMessages().contains(
|
||||
"parity overlay marked --skip-worktree (cannot be committed from this worktree): .env"),
|
||||
"expected the skip-worktree consequence line, got:\n" + capturedMessages());
|
||||
}
|
||||
|
||||
/** Criterion 5: null and empty overlay lists return quietly — no exception, no log noise. */
|
||||
@Test
|
||||
void overlayParityWithNoCandidatesLogsNothing(@TempDir Path tmp) throws Exception {
|
||||
reportingLogger.setLevel(Level.INFO);
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Path wt = bareWorktree(repo, tmp.resolve("wt"), "cb134-empty");
|
||||
GitWorktrees worktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
|
||||
worktrees.overlayParity(repo.toString(), wt.toString(), null);
|
||||
worktrees.overlayParity(repo.toString(), wt.toString(), List.of());
|
||||
|
||||
assertTrue(reportingAppender.list.isEmpty(),
|
||||
"a null/empty overlay must log nothing, got:\n" + capturedMessages());
|
||||
}
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user