Compare commits
116 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 4b9ebda1b3 | |||
| 1e68d7ee39 | |||
| 42820fbe75 | |||
| d91ff886da | |||
| 386e760a5c | |||
| 2e349139e9 | |||
| a639969a9a | |||
| 17c3a69c57 | |||
| 49a5875586 | |||
| 634d33b50b | |||
| 1db79bcaa9 | |||
| 4507bc5a70 | |||
| d7239ed23b | |||
| dfeb9340b4 | |||
| 1513d4f260 | |||
| 4ca7d72303 | |||
| c0545d003d | |||
| 275ac0d251 | |||
| c1e06c9e12 | |||
| 204da67d66 | |||
| b091c51eee | |||
| 84d631b030 | |||
| 3f8c38fc54 | |||
| a4dbc8f8b7 | |||
| ed2fd6646a | |||
| 5441a2b321 | |||
| 384867dfa3 | |||
| f40c19ecf0 | |||
| 034e17bb32 | |||
| 68b428c484 | |||
| e20ccab1eb | |||
| d83821bbce | |||
| ba2f4d16f8 | |||
| db4c98ac60 | |||
| 8f80d267a0 | |||
| 738d34a609 | |||
| a46e4058ac | |||
| f4f5f3106e | |||
| 8d79d229ff | |||
| cb64bc8157 | |||
| ed28b51f12 | |||
| 4f9aba40e7 | |||
| dab697fae0 | |||
| bfac14108f | |||
| 7a3b2bb7ee | |||
| f188947750 | |||
| f606fccf7f | |||
| 735b6af976 | |||
| 8b4320ed24 | |||
| 351ee1ea6d | |||
| d4a51c6274 | |||
| 26f380a00b | |||
| b8182c96c2 | |||
| 3da44eed63 | |||
| 93a9ed3f83 | |||
| 0b032f5a1a | |||
| fad99c4c5e | |||
| 87871eaefb | |||
| a476a14f1c | |||
| 5eb4267a4a | |||
| 7611b69667 | |||
| cc302fe4af | |||
| 4a8a780274 | |||
| 343ce0f4c0 | |||
| cec3e191d4 | |||
| 1850a5f324 | |||
| e4eb3dbed4 | |||
| 57cd96f5e6 | |||
| 202e37e3b3 | |||
| 90253f832d | |||
| f1640f5dcc | |||
| c7903c1efe | |||
| 7d711942fe | |||
| bec87f987c | |||
| 7f8a8829f9 | |||
| af9589783e | |||
| 190436c9cf | |||
| 8ea5c2bb1f | |||
| 8335b12562 | |||
| d25c863118 | |||
| 7c34e8f4f9 | |||
| be07ed2033 | |||
| a6415f3e52 | |||
| bb6fc9e0d7 | |||
| 3366590dbe | |||
| 01adc841fa | |||
| 08771e270b | |||
| b5843ab43f | |||
| d8985719eb | |||
| 5b1e13ca3d | |||
| c89a375e5d | |||
| 37dcefa834 | |||
| 6a7342b1f0 | |||
| a5ad7c6561 | |||
| c71ac231e5 | |||
| 33720c42b3 | |||
| 3833d8e52b | |||
| b37def9238 | |||
| 8f02576df6 | |||
| 525bc1c5f4 | |||
| d59ece6dec | |||
| 32408d1e64 | |||
| 6e23bf8309 | |||
| aa4c0b84c3 | |||
| 40c593cd09 | |||
| 979adf82eb | |||
| 36870836aa | |||
| 136312fb11 | |||
| 4f28da62a3 | |||
| 708f1795ad | |||
| 599419f9e6 | |||
| ac351ee1de | |||
| 274afafde6 | |||
| 8f59019305 | |||
| b17f37a683 | |||
| dcd505286f |
@@ -0,0 +1,23 @@
|
|||||||
|
---
|
||||||
|
name: hunter
|
||||||
|
description: Sweep one assigned scope for defects and report ranked findings without changes.
|
||||||
|
---
|
||||||
|
|
||||||
|
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||||
|
|
||||||
|
You sweep the assigned package or scope for real defects. Read the full assigned scope before you
|
||||||
|
judge it. Report several ranked findings when the evidence supports them. Change nothing: do not
|
||||||
|
edit code, commit, push, or open a pull request.
|
||||||
|
|
||||||
|
You may run the build or tests to check a finding. Read the complete output and report the real
|
||||||
|
result. Do not hide failures with a pipe. State only checks you actually ran. The primary's IDE
|
||||||
|
tools are not yours. A mounted forge tool may use a blocked credential and fail by design.
|
||||||
|
|
||||||
|
Do only the assigned scope. Note anything outside it in one line and do not investigate it further.
|
||||||
|
Use `fleet_ask{question}` only when a decision belongs to the lead, such as an unclear requirement
|
||||||
|
or two defensible fixes. Do not ask about something you can decide by reading more code.
|
||||||
|
|
||||||
|
Your handoff must name the files you read, each ranked finding or `NO FINDINGS`, the checks you ran,
|
||||||
|
and any caveat for review.
|
||||||
|
|
||||||
|
The launcher provides the required bridge reply instructions for every member.
|
||||||
@@ -39,6 +39,10 @@ jobs:
|
|||||||
# that runs does so against the fake UDS herdr and fake ccs/claude stubs.
|
# that runs does so against the fake UDS herdr and fake ccs/claude stubs.
|
||||||
run: mvn -B clean install
|
run: mvn -B clean install
|
||||||
|
|
||||||
|
- name: javadoc reference lint
|
||||||
|
working-directory: fleetd
|
||||||
|
run: mvn -B -DskipTests javadoc:javadoc -Ddoclint=reference
|
||||||
|
|
||||||
# Deliberately NOT actions/upload-artifact: this Gitea instance presents as GHES, and
|
# Deliberately NOT actions/upload-artifact: this Gitea instance presents as GHES, and
|
||||||
# @actions/artifact v2+ (i.e. upload-artifact@v4) refuses to run there —
|
# @actions/artifact v2+ (i.e. upload-artifact@v4) refuses to run there —
|
||||||
# "GHESNotSupportedError ... not currently supported on GHES", which red-Xes an otherwise
|
# "GHESNotSupportedError ... not currently supported on GHES", which red-Xes an otherwise
|
||||||
@@ -54,6 +58,23 @@ jobs:
|
|||||||
done
|
done
|
||||||
exit 0
|
exit 0
|
||||||
|
|
||||||
|
# fleetd #550 — nothing ran scripts/test-redeploy-fleetd.sh in CI before this, on any platform,
|
||||||
|
# so it had run only on macOS by hand and two Linux-only bugs (this issue's items 1 and 2)
|
||||||
|
# survived undetected: shasum is a macOS-only tool (it ships with Perl; GNU coreutils, i.e. every
|
||||||
|
# mainstream Linux distro including this runner's ubuntu-latest, does not have it and ships
|
||||||
|
# sha256sum instead). The gate here is the step's own exit code, nothing else: a `run:` step in
|
||||||
|
# Gitea/GitHub Actions already fails the job on a non-zero exit with no extra scripting needed,
|
||||||
|
# so this deliberately does NOT grep the output for a `FAIL:` count. That is the #550 item-2
|
||||||
|
# lesson one level up — a suite that dies before it runs a single test prints zero FAIL lines,
|
||||||
|
# which is exactly what a clean pass also prints, so counting FAIL lines can never be the gate.
|
||||||
|
shell-tests:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: redeploy-fleetd.sh shell suite
|
||||||
|
run: bash scripts/test-redeploy-fleetd.sh
|
||||||
|
|
||||||
# CB-521 — actually run the AMQP contract test in CI, against a REAL broker. The broker is a
|
# CB-521 — actually run the AMQP contract test in CI, against a REAL broker. The broker is a
|
||||||
# RabbitMQ SERVICE CONTAINER, not Testcontainers-with-Docker: the runner image has no Docker, so
|
# RabbitMQ SERVICE CONTAINER, not Testcontainers-with-Docker: the runner image has no Docker, so
|
||||||
# AmqpReplyInboxContractTest reads AMQP_URI (set below to the service's network alias) and binds
|
# AmqpReplyInboxContractTest reads AMQP_URI (set below to the service's network alias) and binds
|
||||||
|
|||||||
@@ -95,9 +95,21 @@ below are the procedure — run them in order, every task, not only the big ones
|
|||||||
`fleet_status`, never by reading its terminal; it also reports an open question and the `turnId`
|
`fleet_status`, never by reading its terminal; it also reports an open question and the `turnId`
|
||||||
that answers it. **A worker's ask waits ~55 seconds, and no nudge makes that longer** — so never
|
that answers it. **A worker's ask waits ~55 seconds, and no nudge makes that longer** — so never
|
||||||
brief a worker to "ask me". Decide before you delegate, or give it an explicit default.
|
brief a worker to "ask me". Decide before you delegate, or give it an explicit default.
|
||||||
|
**A correction cannot reach a busy member.** A `fleet_send` to a working member is *accepted* and
|
||||||
|
returns a ticket, and is then never delivered — measured here three times in one session, and the
|
||||||
|
member was released still executing a brief that had been retracted twice. The receipt is true and
|
||||||
|
it is a fact about the *mailbox*; what you needed was a fact about the *pane*. **A push delivery
|
||||||
|
needs the recipient free at send time; a pull channel needs only that they look before acting.** So
|
||||||
|
put every correction on the **ticket**, which they can read whenever they look, and send as the
|
||||||
|
notification. That obliges you, not them: **all corrections go to the ticket, and the brief is
|
||||||
|
write-once.** The member cannot check which source is newer — it just always prefers the ticket —
|
||||||
|
so the day you revise a brief in place instead of commenting, it obeys your rule and does the wrong
|
||||||
|
thing. A *first* brief for a unit not yet running is not a correction, and may be the whole spec.
|
||||||
6. **Verify yourself.** Re-run the build and the checks. A worker cannot run your IDE tooling, any
|
6. **Verify yourself.** Re-run the build and the checks. A worker cannot run your IDE tooling, any
|
||||||
forge tools it appears to have hold a blocked credential and fail, and a piped command
|
forge MCP server it appears to have holds a blocked credential and fails every call, and a piped
|
||||||
(`… | tail`) hides failures behind a zero exit — never promote a worker's "clean" to a fact.
|
command (`… | tail`) hides failures behind a zero exit — never promote a worker's "clean" to a
|
||||||
|
fact. Its injected repo-scoped `GITEA_TOKEN` is a different credential and does work, so a worker
|
||||||
|
reporting that it opened its own PR is reporting something it really can do.
|
||||||
7. **Review — fan out.** Spawn reviewers against the diff, one per dimension or per file, with
|
7. **Review — fan out.** Spawn reviewers against the diff, one per dimension or per file, with
|
||||||
`wait:false`. Never the implementer of the scope it reviews, and brief them from the diff — not
|
`wait:false`. Never the implementer of the scope it reviews, and brief them from the diff — not
|
||||||
from the implementer's rationale, which carries its own blind spot. Dispatch each PR's reviewers
|
from the implementer's rationale, which carries its own blind spot. Dispatch each PR's reviewers
|
||||||
@@ -127,12 +139,21 @@ prefer `wait:false` + `fleet_poll` for anything non-trivial: a blocking `fleet_s
|
|||||||
**Delegating does not delegate responsibility.** Workers open PRs; you are the gate. Never delegate
|
**Delegating does not delegate responsibility.** Workers open PRs; you are the gate. Never delegate
|
||||||
the merge — and merging on a reviewer's word is delegating it by proxy.
|
the merge — and merging on a reviewer's word is delegating it by proxy.
|
||||||
|
|
||||||
|
**When a decision blocks you, consult architects — not the operator.** Spawn one or more architect
|
||||||
|
members, give them the question and the evidence you have, and act on what they agree. They are
|
||||||
|
authorized to settle it, not only to advise. If two of them still disagree after two rounds, they
|
||||||
|
return both positions and you decide. Go to the operator only for something outside the fleet's
|
||||||
|
authority: money, credentials, or a promise made to someone else. **Then write the decision on the
|
||||||
|
ticket.** Taking the operator out of the loop also removes the signal they used to get, because
|
||||||
|
that signal was the block itself — work stopped, so they found out. A ticket comment replaces it,
|
||||||
|
and it reaches them whether or not they are at a terminal when you decide.
|
||||||
|
|
||||||
| Intent | Tool |
|
| Intent | Tool |
|
||||||
|---|---|
|
|---|---|
|
||||||
| Confirm your own role | `fleet_whoami` |
|
| Confirm your own role | `fleet_whoami` |
|
||||||
| See backends available | `fleet_profiles` |
|
| See backends available | `fleet_profiles` |
|
||||||
| Start a member | `fleet_spawn{role?, profile?, cwd?, worktree?, ticket?, sessionName?, resumeSessionId?}` → `sessionId` + `paneId` |
|
| Start a member | `fleet_spawn{role?, profile?, cwd?, worktree?, ticket?, sessionName?, resumeSessionId?}` → `sessionId` + `paneId` |
|
||||||
| See the fleet | `fleet_list` → `leads` (your peers) + `members` (each carries `agentSessionId` when its backend knows one) · one peer's state: `fleet_status{sessionId}` |
|
| See the fleet | `fleet_list` → `leads` (your peers) + `members` (each carries `agentSessionId` when its backend knows one) + `loopHealth` (`RUNNING`, `STALLED`, or `STOPPED` for `statusPoller` and `sessionReaper`) · one peer's state: `fleet_status{sessionId}` |
|
||||||
| Delegate (blocking) | `fleet_send{sessionId, content}` |
|
| Delegate (blocking) | `fleet_send{sessionId, content}` |
|
||||||
| Delegate (long task) | `fleet_send{sessionId, content, wait:false}` → ticket → `fleet_poll{ticket}` |
|
| Delegate (long task) | `fleet_send{sessionId, content, wait:false}` → ticket → `fleet_poll{ticket}` |
|
||||||
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId` |
|
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId` |
|
||||||
@@ -189,20 +210,26 @@ simply complies has thrown away the reason there are two of you.
|
|||||||
3. **`fleet_ask{question}`** when a decision is genuinely the lead's (ambiguous requirement, two
|
3. **`fleet_ask{question}`** when a decision is genuinely the lead's (ambiguous requirement, two
|
||||||
defensible fixes, "bug or intended?"). It blocks and you resume the *same* turn with the answer.
|
defensible fixes, "bug or intended?"). It blocks and you resume the *same* turn with the answer.
|
||||||
Don't ask what you could decide yourself.
|
Don't ask what you could decide yourself.
|
||||||
4. **End the turn with exactly one `fleet_reply{content}`**, carrying your complete answer. This is
|
4. **Re-read the ticket before you act on anything you were told earlier**, and again before you
|
||||||
|
commit. A message reaches you only while you are free to receive it; the ticket is there whenever
|
||||||
|
you look. **If a ticket comment contradicts your brief, the ticket comment is newer and it wins.**
|
||||||
|
5. **End the turn with exactly one `fleet_reply{content}`**, carrying your complete answer. This is
|
||||||
the whole handoff. No `fleet_reply` ⇒ the sender gets nothing and the exchange stalls.
|
the whole handoff. No `fleet_reply` ⇒ the sender gets nothing and the exchange stalls.
|
||||||
Do **not** lean on the completion fallback to carry your answer for you: when you end a turn
|
Do **not** lean on the completion fallback to carry your answer for you: when you end a turn
|
||||||
without replying, the bridge scrapes your pane, and it can return only the last 4000 characters.
|
without replying, the bridge scrapes your pane, and it can return only the last 4000 characters.
|
||||||
A clipped scrape is marked as partial, but the missing text is gone — your report reaches the
|
A clipped scrape is marked as partial, but the missing text is gone — your report reaches the
|
||||||
lead with its end cut off.
|
lead with its end cut off.
|
||||||
5. **Report honestly.** State only what you actually ran and its real output, including failures,
|
6. **Report honestly.** State only what you actually ran and its real output, including failures,
|
||||||
and never claim the result of a check you had no way to run. **Measure your own tools; do not
|
and never claim the result of a check you had no way to run. **Measure your own tools; do not
|
||||||
assume them.** What you mount depends on your backend: an opencode member gets the bridge and
|
assume them.** What you mount depends on your backend: an opencode member gets the bridge and
|
||||||
nothing else, while a Claude Code member also inherits the operator's user-scope MCP servers,
|
nothing else, while a Claude Code member also inherits the operator's user-scope MCP servers,
|
||||||
which the bridge never chose for you. Two rules follow. The primary's IDE tooling is still not
|
which the bridge never chose for you. Two rules follow. The primary's IDE tooling is still not
|
||||||
yours, whatever you see. And **a mounted tool is not a working tool** — the forge server you may
|
yours, whatever you see. And **a mounted tool is not a working tool** — the forge MCP server you
|
||||||
find there holds a deliberately blocked credential and fails every call, by design.
|
may find there holds a deliberately blocked credential and fails every call, by design. That is
|
||||||
6. **Never merge.** Stage files explicitly — never `git add -A` — and leave alone anything the
|
not your only forge route, and the two must not be confused: the repo-scoped `GITEA_TOKEN` the
|
||||||
|
daemon injects into your environment does work, and using it to open your own PR is part of the
|
||||||
|
job. A blocked MCP tool is never a reason to skip that step.
|
||||||
|
7. **Never merge.** Stage files explicitly — never `git add -A` — and leave alone anything the
|
||||||
project marks as not-yours-to-commit.
|
project marks as not-yours-to-commit.
|
||||||
|
|
||||||
### Where each rule lives (don't duplicate — extend the right layer)
|
### Where each rule lives (don't duplicate — extend the right layer)
|
||||||
@@ -257,6 +284,8 @@ must obey belongs in the charter, not here.
|
|||||||
it for a multi-finding sweep hands the worker two contradictory output contracts. That has
|
it for a multi-finding sweep hands the worker two contradictory output contracts. That has
|
||||||
already cost three workers' turns: each wrote a good report to its terminal and ended the turn
|
already cost three workers' turns: each wrote a good report to its terminal and ended the turn
|
||||||
with no `fleet_reply`, and the scrape returned the tail of the brief instead.
|
with no `fleet_reply`, and the scrape returned the tail of the brief instead.
|
||||||
|
Spawn `implementer` with role `dev`, `reviewer` with role `reviewer`, and `hunter` with role
|
||||||
|
`hunter`.
|
||||||
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
||||||
`port-to-opencode` (make an OpenCode session a participant in this workspace),
|
`port-to-opencode` (make an OpenCode session a participant in this workspace),
|
||||||
`fleets-status` (report every fleet that shares one LavinMQ instance),
|
`fleets-status` (report every fleet that shares one LavinMQ instance),
|
||||||
|
|||||||
@@ -5,6 +5,21 @@
|
|||||||
here took a revert and two upstream fixes — see §7.1, which is the useful part of this document. One
|
here took a revert and two upstream fixes — see §7.1, which is the useful part of this document. One
|
||||||
risk is **accepted rather than solved**: a stream cut by any mid-response timer arrives as HTTP 200
|
risk is **accepted rather than solved**: a stream cut by any mid-response timer arrives as HTTP 200
|
||||||
with no terminator, and our third-party members cannot detect it (§7.2).
|
with no terminator, and our third-party members cannot detect it (§7.2).
|
||||||
|
|
||||||
|
> **Superseded in part — 2026-09-13.** Two claims on this page are no longer true of the live fleet.
|
||||||
|
> I measured both on this host today.
|
||||||
|
>
|
||||||
|
> 1. **The model is named `acoder` now, not `deepseek-v4-flash`.** `acoder` is a stable alias, and
|
||||||
|
> the model behind it changed on 2026-08-28: it is Qwen3.8-27B, not DeepSeek. The old name is
|
||||||
|
> still served, so nothing broke — the gateway answers it and reports `"model": "acoder"` in the
|
||||||
|
> reply, which is how you can see for yourself that it is an alias. `fleetd.yaml` moved to
|
||||||
|
> `acoder` on 2026-09-13. Do not guess behaviour from the name; ask the gateway's own manifest,
|
||||||
|
> `GET https://llm.ltms.dev/v1/deployment`, and read its `generation` field.
|
||||||
|
> 2. **`local` sits at `weight: 0`, not 100.** Only `gx` is auto-selected today.
|
||||||
|
>
|
||||||
|
> §2 and §3 below are the plan as written in August. They are the record of the migration, so they
|
||||||
|
> stay as they are. If this note stops matching `fleetd.yaml`, re-measure and rewrite the note.
|
||||||
|
|
||||||
· **Upstream:** [systems/vms wiki → LLM and MCP Gateway](https://git.ltms.dev/systems/vms/wiki/LLM-and-MCP-Gateway)
|
· **Upstream:** [systems/vms wiki → LLM and MCP Gateway](https://git.ltms.dev/systems/vms/wiki/LLM-and-MCP-Gateway)
|
||||||
· **Upstream issue:** [systems/vms#31](https://git.ltms.dev/systems/vms/issues/31)
|
· **Upstream issue:** [systems/vms#31](https://git.ltms.dev/systems/vms/issues/31)
|
||||||
|
|
||||||
@@ -380,7 +395,10 @@ one turn; this one costs the whole task and is indistinguishable from a slow wor
|
|||||||
|
|
||||||
- Token accepted on both surfaces. **Unauthenticated → 401**, so the Caddy proxy really does gate —
|
- Token accepted on both surfaces. **Unauthenticated → 401**, so the Caddy proxy really does gate —
|
||||||
the wiki's "SecurityPolicy fails open" warning is about the gateway itself, not the edge.
|
the wiki's "SecurityPolicy fails open" warning is about the gateway itself, not the edge.
|
||||||
- `/v1/models` returns exactly `["deepseek-v4-flash"]`, so trap 3 is clear.
|
- `/v1/models` returned exactly `["deepseek-v4-flash"]` **on 2026-08-15**, so trap 3 was clear
|
||||||
|
then. It returns 6 ids now — `acoder`, `qwen3.8-27b-nvfp4`, `deepseek-v4-flash` and three
|
||||||
|
embedding names — measured on this host 2026-09-13. The exact-name rule still holds; the
|
||||||
|
one-item list does not.
|
||||||
- **Reasoning survives both surfaces** — see §3b above.
|
- **Reasoning survives both surfaces** — see §3b above.
|
||||||
- The launcher's generated opencode provider block is correct, carrying a real 48-character `llmk-`
|
- The launcher's generated opencode provider block is correct, carrying a real 48-character `llmk-`
|
||||||
key rather than the `fleetd-local-noauth` placeholder.
|
key rather than the `fleetd-local-noauth` placeholder.
|
||||||
|
|||||||
@@ -560,7 +560,7 @@ placement: weighted
|
|||||||
# older keys: `leaders:`, `members:`, `leadScan:` and `defaultProfile:`.
|
# older keys: `leaders:`, `members:`, `leadScan:` and `defaultProfile:`.
|
||||||
#
|
#
|
||||||
# A member is anything a lead spawns, and every member has two INDEPENDENT attributes:
|
# A member is anything a lead spawns, and every member has two INDEPENDENT attributes:
|
||||||
# role — which contract: architect, dev or reviewer. It picks the launch charter, the role
|
# role — which contract: architect, dev, hunter or reviewer. It picks the launch charter, the role
|
||||||
# file, the playbook skill and the authz row.
|
# file, the playbook skill and the authz row.
|
||||||
# profile — which backend: one of the `profiles:` keys above (model, CLI adapter, cost).
|
# profile — which backend: one of the `profiles:` keys above (model, CLI adapter, cost).
|
||||||
# They vary on their own. A reviewer may run on the same profile as the dev whose diff it reads,
|
# They vary on their own. A reviewer may run on the same profile as the dev whose diff it reads,
|
||||||
@@ -572,13 +572,13 @@ placement: weighted
|
|||||||
#
|
#
|
||||||
# Each pool lists the profiles that role MAY run on — these are pools, not identities. That is also
|
# Each pool lists the profiles that role MAY run on — these are pools, not identities. That is also
|
||||||
# what replaced `defaultProfile:`: an unqualified spawn names a role, and that role's pool supplies
|
# what replaced `defaultProfile:`: an unqualified spawn names a role, and that role's pool supplies
|
||||||
# the candidates, in definition order. A dev and a reviewer staying anonymous is exactly compatible
|
# the candidates, in definition order. A dev, hunter and reviewer staying anonymous is exactly
|
||||||
# with being listed here; the entry key just names the entry.
|
# compatible with being listed here; the entry key just names the entry.
|
||||||
fleet:
|
fleet:
|
||||||
# Optional launch-charter text, keyed only by the singular role wire names: architect, dev,
|
# Optional launch-charter text, keyed only by the singular role wire names: architect, dev,
|
||||||
# reviewer. Changes are HOT and reach the next spawn without a daemon restart. Do not put secrets
|
# hunter, reviewer. Changes are HOT and reach the next spawn without a daemon restart. Do not put
|
||||||
# here: a later launch step writes this text to a world-readable temp file, and ${ENV} interpolation
|
# secrets here: a later launch step writes this text to a world-readable temp file, and ${ENV}
|
||||||
# is deliberately not supported.
|
# interpolation is deliberately not supported.
|
||||||
charters:
|
charters:
|
||||||
architect: |-
|
architect: |-
|
||||||
You are an architect in this fleet. You refine work before anyone builds it:
|
You are an architect in this fleet. You refine work before anyone builds it:
|
||||||
@@ -589,6 +589,9 @@ fleet:
|
|||||||
dev: |-
|
dev: |-
|
||||||
You implement the one unit you were given, and nothing else. You test it,
|
You implement the one unit you were given, and nothing else. You test it,
|
||||||
commit it, and open your own pull request. You never merge.
|
commit it, and open your own pull request. You never merge.
|
||||||
|
hunter: |-
|
||||||
|
You sweep the assigned scope for real defects. You may run the build or tests
|
||||||
|
to check a finding. You change nothing, and report several ranked findings.
|
||||||
reviewer: |-
|
reviewer: |-
|
||||||
You review the diff you were given. You report bugs, risks and missing tests.
|
You review the diff you were given. You report bugs, risks and missing tests.
|
||||||
You do not change code.
|
You do not change code.
|
||||||
@@ -666,6 +669,9 @@ fleet:
|
|||||||
developers:
|
developers:
|
||||||
gx10:
|
gx10:
|
||||||
profile: gx10
|
profile: gx10
|
||||||
|
# hunters:
|
||||||
|
# gx10:
|
||||||
|
# profile: gx10 # a hunt may run checks, but never changes code
|
||||||
# reviewers:
|
# reviewers:
|
||||||
# gx10:
|
# gx10:
|
||||||
# profile: gx10 # the same backend may serve two roles; that is the point
|
# profile: gx10 # the same backend may serve two roles; that is the point
|
||||||
|
|||||||
@@ -21,6 +21,7 @@ import dev.ltms.fleet.inject.ExhaustedPatternLookup;
|
|||||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||||
import dev.ltms.fleet.inject.LiveExhaustedPatterns;
|
import dev.ltms.fleet.inject.LiveExhaustedPatterns;
|
||||||
import dev.ltms.fleet.inject.Injector;
|
import dev.ltms.fleet.inject.Injector;
|
||||||
|
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||||
import dev.ltms.fleet.inject.StatusPoller;
|
import dev.ltms.fleet.inject.StatusPoller;
|
||||||
import dev.ltms.fleet.inject.TurnListener;
|
import dev.ltms.fleet.inject.TurnListener;
|
||||||
import dev.ltms.fleet.inject.MemberPresence;
|
import dev.ltms.fleet.inject.MemberPresence;
|
||||||
@@ -79,7 +80,9 @@ import java.util.concurrent.Executors;
|
|||||||
import java.util.concurrent.ScheduledExecutorService;
|
import java.util.concurrent.ScheduledExecutorService;
|
||||||
import java.util.concurrent.TimeUnit;
|
import java.util.concurrent.TimeUnit;
|
||||||
import java.util.concurrent.atomic.AtomicReference;
|
import java.util.concurrent.atomic.AtomicReference;
|
||||||
|
import java.util.function.BooleanSupplier;
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
import java.util.function.Predicate;
|
import java.util.function.Predicate;
|
||||||
import java.util.function.Supplier;
|
import java.util.function.Supplier;
|
||||||
import java.util.regex.Pattern;
|
import java.util.regex.Pattern;
|
||||||
@@ -270,15 +273,12 @@ public final class Fleetd {
|
|||||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||||
// would crash the daemon into a restart loop. Wait, then degrade rather than die: serving
|
// would crash the daemon into a restart loop. Wait, then degrade rather than die: serving
|
||||||
// with /healthz reporting "degraded" is strictly more useful than exiting.
|
// with /healthz reporting "degraded" is strictly more useful than exiting.
|
||||||
boolean herdrUp = awaitHerdr(herdr);
|
HerdrAwaitOutcome herdrOutcome = awaitHerdr(herdr, System::nanoTime, Fleetd::sleepHerdrPoll);
|
||||||
|
boolean herdrUp = logHerdrWaitOutcomeAndShouldReap(herdrOutcome);
|
||||||
if (herdrUp) {
|
if (herdrUp) {
|
||||||
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
||||||
// with the previous process — reap those leaked orphans now, before we start serving.
|
// with the previous process — reap those leaked orphans now, before we start serving.
|
||||||
workers.reapOrphanWorkers();
|
workers.reapOrphanWorkers();
|
||||||
} else {
|
|
||||||
log.warn("herdr did not answer within {}s — starting anyway; /healthz will report "
|
|
||||||
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
|
||||||
HERDR_WAIT_SECONDS);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
||||||
@@ -493,45 +493,16 @@ public final class Fleetd {
|
|||||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||||
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||||
MemberPresence presence = sessions.asPresence();
|
MemberPresence presence = sessions.asPresence();
|
||||||
TurnListener turnListener = new TurnListener() {
|
// fleetd #561: extracted to a static factory (see turnListener below) — the anonymous class
|
||||||
@Override
|
// this replaced had two bare, unguarded statements per callback, and nothing enforced that
|
||||||
public void onTurnComplete(String target) {
|
// the completion half went first beyond call order in the source.
|
||||||
completion.onTurnComplete(target);
|
TurnListener turnListener = turnListener(completion, sessions);
|
||||||
sessions.onTurnComplete(target);
|
|
||||||
}
|
|
||||||
|
|
||||||
@Override
|
|
||||||
public boolean hasPostTurnAction(String target) {
|
|
||||||
return sessions.hasPostTurnAction(target);
|
|
||||||
}
|
|
||||||
|
|
||||||
@Override
|
|
||||||
public boolean onTurnCompleteWithPostAction(String target) {
|
|
||||||
completion.resolveBeforePostAction(target);
|
|
||||||
return sessions.onTurnCompleteWithPostAction(target);
|
|
||||||
}
|
|
||||||
|
|
||||||
@Override
|
|
||||||
public void onDelivered(String target, dev.ltms.fleet.msg.TurnToken token) {
|
|
||||||
completion.onDelivered(target, token);
|
|
||||||
sessions.onDelivered(target, token);
|
|
||||||
}
|
|
||||||
|
|
||||||
@Override
|
|
||||||
public void onTurnFailed(String target) {
|
|
||||||
completion.onTurnFailed(target);
|
|
||||||
sessions.onTurnFailed(target);
|
|
||||||
}
|
|
||||||
|
|
||||||
@Override
|
|
||||||
public void onTurnFailed(String target, String reason) {
|
|
||||||
completion.onTurnFailed(target, reason);
|
|
||||||
sessions.onTurnFailed(target);
|
|
||||||
}
|
|
||||||
};
|
|
||||||
Predicate<String> deliverable = deliverableTo(presence, leads);
|
Predicate<String> deliverable = deliverableTo(presence, leads);
|
||||||
|
// fleetd #556: registration is wired directly to `completion`, not folded into the
|
||||||
|
// `turnListener` fan-out above — so it survives `sessions.onDelivered` (or any future
|
||||||
|
// listener) throwing, regardless of call order. See TurnRegistrar's javadoc.
|
||||||
Injector injector = new Injector(router, turnListener, deliverable,
|
Injector injector = new Injector(router, turnListener, deliverable,
|
||||||
presence::forget);
|
presence::forget, completion::register);
|
||||||
StatusPoller poller = new StatusPoller(router, injector, Injector.POLL_INTERVAL_MILLIS);
|
StatusPoller poller = new StatusPoller(router, injector, Injector.POLL_INTERVAL_MILLIS);
|
||||||
poller.start();
|
poller.start();
|
||||||
|
|
||||||
@@ -695,14 +666,12 @@ public final class Fleetd {
|
|||||||
return configured == null ? null : configured.effectiveCredentialId();
|
return configured == null ? null : configured.effectiveCredentialId();
|
||||||
}, outagePolicy);
|
}, outagePolicy);
|
||||||
|
|
||||||
|
FleetMcp.LoopHealthSource loopHealth = loopHealthSource(poller, reaper);
|
||||||
FleetMcp mcp = new FleetMcp(messages, workers, sessions, identity, presence,
|
FleetMcp mcp = new FleetMcp(messages, workers, sessions, identity, presence,
|
||||||
primaryRegistry, callers, metrics,
|
primaryRegistry, callers, FleetMcp.AuthorizationMode.ENFORCED, metrics,
|
||||||
capacitySource(config, cfg, profile -> liveCountRef.get().apply(profile)),
|
capacitySource(config, cfg, profile -> liveCountRef.get().apply(profile)),
|
||||||
new FleetMcp.HealthCoverageSource(() -> {
|
healthCoverageSource(config),
|
||||||
var health = config.get().health();
|
loopHealth,
|
||||||
return FleetHealthMonitor.coverage(health != null && health.isEnabled(),
|
|
||||||
health != null && health.notifications() != null && health.notifications().configured());
|
|
||||||
}),
|
|
||||||
quarantineSource,
|
quarantineSource,
|
||||||
leadMailbox,
|
leadMailbox,
|
||||||
outageSource,
|
outageSource,
|
||||||
@@ -794,7 +763,7 @@ public final class Fleetd {
|
|||||||
Javalin app = new FleetApp(herdr, memberHerdr, workers, sessions, messages, presence, mcp.servlet(),
|
Javalin app = new FleetApp(herdr, memberHerdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||||
callers, metrics, deliverable,
|
callers, metrics, deliverable,
|
||||||
() -> MemberCredentialPolicyView.of(config.get().memberCredentials()),
|
() -> MemberCredentialPolicyView.of(config.get().memberCredentials()),
|
||||||
quarantineSource, outageSource).build();
|
quarantineSource, outageSource, loopHealth).build();
|
||||||
app.start(cfg.bind().host(), cfg.bind().port());
|
app.start(cfg.bind().host(), cfg.bind().port());
|
||||||
log.info("fleetd listening on {}:{}, herdr socket {}",
|
log.info("fleetd listening on {}:{}, herdr socket {}",
|
||||||
cfg.bind().host(), cfg.bind().port(), socket);
|
cfg.bind().host(), cfg.bind().port(), socket);
|
||||||
@@ -1040,6 +1009,63 @@ public final class Fleetd {
|
|||||||
cfg.profiles()::keySet, System::nanoTime);
|
cfg.profiles()::keySet, System::nanoTime);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #426: package-private factory for {@code fleet_list}'s {@code healthCoverage} source,
|
||||||
|
* extracted out of {@code main} for the same reason {@link #capacitySource} and {@link
|
||||||
|
* #quarantineSource} were — and the same reason {@link #exhaustedPatternCoverageLine}/{@link
|
||||||
|
* #errorPatternCoverageLine} exist: {@link FleetHealthMonitor#coverage}'s three-branch method
|
||||||
|
* is easy to pin directly (a plain {@code (boolean, boolean) -> String} call), but that proves
|
||||||
|
* nothing about whether <em>this call site</em> pairs the right boolean with the right meaning.
|
||||||
|
* fleetd #415's measured lesson is the reason this matters here — swapping the two arguments at
|
||||||
|
* a call site like this one compiled clean and left the full suite green, because every existing
|
||||||
|
* test exercised the method in both directions without ever exercising the pairing.
|
||||||
|
*
|
||||||
|
* <p>{@code #407}'s "keep the config invalid, assert on the log line before the throw" option
|
||||||
|
* does not apply to this call site: the five reporters #407 covers all run in {@code main}
|
||||||
|
* <em>before</em> {@code cfg.validateAll()} (line ~171), so an invalid config still exercises
|
||||||
|
* them. This call site is built during {@code FleetMcp} construction, which runs only after
|
||||||
|
* {@code UnixSocketHerdrClient.connect} has already opened a real herdr socket (line ~188) —
|
||||||
|
* reaching it at all means main already performed real I/O, which the no-socket constraint on
|
||||||
|
* this ticket rules out. So the pairing is pinned by extracting it to this directly-callable
|
||||||
|
* factory instead, the same shape {@link #capacitySource}/{@link #quarantineSource} already use.
|
||||||
|
*
|
||||||
|
* <p>Reads {@code config.get().health()} live (health.notifications is a {@code SPLIT_KEYS}
|
||||||
|
* entry — see {@link ConfigRef#SPLIT_KEYS}), so a hot-reloaded notifications block changes what
|
||||||
|
* {@code fleet_list} reports without a restart, exactly like {@link #capacitySource}'s maxLoad.
|
||||||
|
*/
|
||||||
|
static FleetMcp.HealthCoverageSource healthCoverageSource(ConfigRef config) {
|
||||||
|
return new FleetMcp.HealthCoverageSource(() -> {
|
||||||
|
var health = config.get().health();
|
||||||
|
return FleetHealthMonitor.coverage(health != null && health.isEnabled(),
|
||||||
|
health != null && health.notifications() != null && health.notifications().configured());
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #562 follow-up: package-private factory for {@code fleet_list}'s and {@code
|
||||||
|
* /healthz}'s {@code loopHealth} source, extracted out of {@code main} for the same reason
|
||||||
|
* {@link #capacitySource} and {@link #healthCoverageSource} were. Before this ticket the
|
||||||
|
* {@link FleetMcp.LoopHealthSource} was built inline with a bare {@code new}, so there was
|
||||||
|
* nothing a test could call directly — measured: replacing {@code poller::health} with a
|
||||||
|
* constant {@code () -> LoopWatchdog.State.RUNNING} at the call site compiled clean and left
|
||||||
|
* the full suite green, meaning the daemon could report the {@link StatusPoller} as always
|
||||||
|
* {@code RUNNING} even while it was actually stalled. That is a false negative on the exact
|
||||||
|
* signal this ticket exists to surface, and is the mirror of a false positive muting a real
|
||||||
|
* monitoring component — worse, because there is no noise for anyone to notice and then
|
||||||
|
* silence. {@link FleetdLoopHealthSourceWiringTest} calls this factory directly and pins both
|
||||||
|
* halves separately, plus the {@code reaper == null} branch below.
|
||||||
|
*
|
||||||
|
* <p>{@code reaper} may be {@code null} — a {@link SessionReaper} is only constructed when
|
||||||
|
* {@code lifecycle.idleTtlSeconds} is configured (see the {@code reaper} local above) — and
|
||||||
|
* this factory preserves the existing behaviour of reporting {@link LoopWatchdog.State#STOPPED}
|
||||||
|
* in that case, rather than a {@code NullPointerException} on the first {@code fleet_list} or
|
||||||
|
* {@code /healthz} call.
|
||||||
|
*/
|
||||||
|
static FleetMcp.LoopHealthSource loopHealthSource(StatusPoller poller, SessionReaper reaper) {
|
||||||
|
return new FleetMcp.LoopHealthSource(poller::health,
|
||||||
|
() -> reaper == null ? LoopWatchdog.State.STOPPED : reaper.health());
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* fleetd #248: package-private factory for the member worktree/branch lookup {@link
|
* fleetd #248: package-private factory for the member worktree/branch lookup {@link
|
||||||
* CompletionResolver} uses to name a fallback report's worktree and branch (fleetd#241).
|
* CompletionResolver} uses to name a fallback report's worktree and branch (fleetd#241).
|
||||||
@@ -1067,6 +1093,157 @@ public final class Fleetd {
|
|||||||
.orElse(null);
|
.orElse(null);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #561: compose the production {@link TurnListener} from its two halves — the
|
||||||
|
* completion resolver (which resolves a blocked {@code fleet_send}'s waiter) and the session
|
||||||
|
* manager (which drives the member's lifecycle state) — hardened so a throw from either
|
||||||
|
* half's callback can never suppress the other half's callback for the same event.
|
||||||
|
*
|
||||||
|
* <p>Before this, each method below was two bare, unguarded statements: whichever ran first
|
||||||
|
* throwing meant the second one never ran at all, and nothing beyond call order in the
|
||||||
|
* source enforced "the completion half goes first". #556 fixed the identical shape for {@code
|
||||||
|
* onDelivered}'s registration by moving it off the fan-out entirely (see {@code
|
||||||
|
* TurnRegistrar}); this fixes the four remaining callbacks — {@code onTurnComplete}, {@code
|
||||||
|
* onTurnCompleteWithPostAction}, and both {@code onTurnFailed} overloads — by hardening the
|
||||||
|
* fan-out itself instead, since none of them can be pulled out of the listener the way
|
||||||
|
* registration was.
|
||||||
|
*
|
||||||
|
* <p>The invariant this composition guarantees: <b>a throwing session-listener must not
|
||||||
|
* prevent the completion resolver from being told the turn ended.</b> The completion half is
|
||||||
|
* always attempted first, and — as a bonus the resolver does not depend on — its own throw
|
||||||
|
* does not stop the session half from running either. Whatever escapes (from one half or
|
||||||
|
* both) is rethrown once both have been attempted, with a second failure recorded via {@link
|
||||||
|
* Throwable#addSuppressed} on the first, so it still reaches {@link StatusPoller}'s {@code
|
||||||
|
* catch (Throwable)} and logs at ERROR. Nothing here swallows a failure to make the two halves
|
||||||
|
* look safe.
|
||||||
|
*
|
||||||
|
* <p>{@code onTurnCompleteWithPostAction} is the one place order is not just fault-tolerance
|
||||||
|
* but a functional requirement: {@code completion.resolveBeforePostAction} must resolve the
|
||||||
|
* scrape before {@code sessions.onTurnCompleteWithPostAction}'s adapter housekeeping can erase
|
||||||
|
* the pane's rendered output (see {@link CompletionResolver#resolveBeforePostAction}). That is
|
||||||
|
* why this composition does not treat the pair symmetrically the way {@link #bothMustRun}
|
||||||
|
* does for the other three callbacks: the mirror case (completion half throws, session half's
|
||||||
|
* return value still observed) is not preserved here — once the completion half's failure
|
||||||
|
* escapes, the session half's return value is discarded, matching how the {@link Injector}
|
||||||
|
* already treats any throw from this callback as "the action did not start" (see the {@code
|
||||||
|
* started} default at its call site, {@code Injector.java} ~line 616).
|
||||||
|
*
|
||||||
|
* <p>Package-private so {@code FleetdTurnListenerCompositionTest} can build this listener
|
||||||
|
* directly from a real {@link CompletionResolver} and a fake {@link TurnListener} standing in
|
||||||
|
* for {@code sessions}, without booting the rest of {@code main}.
|
||||||
|
*/
|
||||||
|
static TurnListener turnListener(CompletionResolver completion, TurnListener sessions) {
|
||||||
|
return new TurnListener() {
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
bothMustRun(() -> completion.onTurnComplete(target), () -> sessions.onTurnComplete(target));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean hasPostTurnAction(String target) {
|
||||||
|
return sessions.hasPostTurnAction(target);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean onTurnCompleteWithPostAction(String target) {
|
||||||
|
return bothMustRunKeepingSecondResult(() -> completion.resolveBeforePostAction(target),
|
||||||
|
() -> sessions.onTurnCompleteWithPostAction(target));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onDelivered(String target, dev.ltms.fleet.msg.TurnToken token) {
|
||||||
|
// fleetd #561: left unguarded on purpose, not because registration survives a
|
||||||
|
// throw elsewhere. completion.onDelivered runs CompletionResolver.captureBaseline,
|
||||||
|
// which already wraps its scrape read in its own catch (RuntimeException) and
|
||||||
|
// fails open (baseline = null) — so this call does not realistically throw, and
|
||||||
|
// there is nothing here for bothMustRun to protect.
|
||||||
|
completion.onDelivered(target, token);
|
||||||
|
sessions.onDelivered(target, token);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onTurnFailed(String target) {
|
||||||
|
bothMustRun(() -> completion.onTurnFailed(target), () -> sessions.onTurnFailed(target));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onTurnFailed(String target, String reason) {
|
||||||
|
bothMustRun(() -> completion.onTurnFailed(target, reason), () -> sessions.onTurnFailed(target));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #561: run two listener-callback halves for one lifecycle event, guaranteeing the
|
||||||
|
* SECOND always runs even when the FIRST throws. Whatever is thrown is rethrown once both
|
||||||
|
* halves have been attempted — a second failure is attached to the first via {@link
|
||||||
|
* Throwable#addSuppressed} rather than dropped. Never swallows.
|
||||||
|
*/
|
||||||
|
private static void bothMustRun(Runnable completionHalf, Runnable sessionsHalf) {
|
||||||
|
Throwable failure = null;
|
||||||
|
try {
|
||||||
|
completionHalf.run();
|
||||||
|
} catch (Throwable t) {
|
||||||
|
failure = t;
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
sessionsHalf.run();
|
||||||
|
} catch (Throwable t) {
|
||||||
|
if (failure == null) {
|
||||||
|
failure = t;
|
||||||
|
} else {
|
||||||
|
failure.addSuppressed(t);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (failure != null) {
|
||||||
|
throwUnchecked(failure);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #561: like {@link #bothMustRun}, but for {@code onTurnCompleteWithPostAction}, whose
|
||||||
|
* session half returns the value the {@link Injector} needs. The second (session) half's
|
||||||
|
* result is what this method returns; if the first (completion) half throws, the second half
|
||||||
|
* still runs and its result is still computed here, but the throw is rethrown afterward
|
||||||
|
* regardless — so that result is discarded at the {@link Injector} call site exactly as it
|
||||||
|
* already is today when this callback throws (see {@link #turnListener}'s javadoc).
|
||||||
|
*/
|
||||||
|
private static boolean bothMustRunKeepingSecondResult(Runnable completionHalf,
|
||||||
|
BooleanSupplier sessionsHalf) {
|
||||||
|
Throwable failure = null;
|
||||||
|
try {
|
||||||
|
completionHalf.run();
|
||||||
|
} catch (Throwable t) {
|
||||||
|
failure = t;
|
||||||
|
}
|
||||||
|
boolean result = false;
|
||||||
|
try {
|
||||||
|
result = sessionsHalf.getAsBoolean();
|
||||||
|
} catch (Throwable t) {
|
||||||
|
if (failure == null) {
|
||||||
|
failure = t;
|
||||||
|
} else {
|
||||||
|
failure.addSuppressed(t);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (failure != null) {
|
||||||
|
throwUnchecked(failure);
|
||||||
|
}
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #561: rethrow a captured {@link Throwable} without a checked-exception wrapper. The
|
||||||
|
* two callback halves above never declare a checked exception (both existing production
|
||||||
|
* halves — {@code CompletionResolver} and {@code SessionManager} — only ever throw unchecked),
|
||||||
|
* so this only ever actually rethrows a {@link RuntimeException} or {@link Error}; the generic
|
||||||
|
* cast is the standard "sneaky throw" idiom, not a claim that a checked exception is expected.
|
||||||
|
*/
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
private static <T extends Throwable> void throwUnchecked(Throwable t) throws T {
|
||||||
|
throw (T) t;
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* fleetd #480: construct the {@link LeadRollover} executor only when {@code leadRollover:} is
|
* fleetd #480: construct the {@link LeadRollover} executor only when {@code leadRollover:} is
|
||||||
* present at startup — the same presence gate {@code leadHeartbeat:} uses just above this
|
* present at startup — the same presence gate {@code leadHeartbeat:} uses just above this
|
||||||
@@ -1696,12 +1873,70 @@ public final class Fleetd {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Poll herdr's {@code ping} until it answers or {@link #HERDR_WAIT_SECONDS} elapses (CB-504).
|
* How {@link #awaitHerdr} ended (fleetd #498). The old code returned a bare {@code boolean},
|
||||||
*
|
* which collapsed two different facts onto the same {@code false}: the configured wait budget
|
||||||
* @return true if herdr answered, false if it never did
|
* genuinely running out, and the waiting thread being interrupted possibly milliseconds in.
|
||||||
|
* Those need different operator messages — see {@link #logHerdrWaitOutcomeAndShouldReap} — so
|
||||||
|
* this is a third state, not a better number (the same shape fleetd #497 named). Never treat
|
||||||
|
* {@link #INTERRUPTED} as if it were {@link #DEADLINE_PASSED}: only the latter means herdr was
|
||||||
|
* actually given the full {@link #HERDR_WAIT_SECONDS} and still failed to answer.
|
||||||
*/
|
*/
|
||||||
private static boolean awaitHerdr(HerdrClient herdr) {
|
enum HerdrWaitResult {
|
||||||
long deadline = System.nanoTime() + HERDR_WAIT_SECONDS * 1_000_000_000L;
|
/** herdr answered {@code ping} before the deadline. */
|
||||||
|
ANSWERED,
|
||||||
|
/** the configured {@link #HERDR_WAIT_SECONDS} budget elapsed with no answer. */
|
||||||
|
DEADLINE_PASSED,
|
||||||
|
/**
|
||||||
|
* the waiting thread was interrupted before the budget ran out — a different event from
|
||||||
|
* {@link #DEADLINE_PASSED} and must never be reported as "did not answer within Ns".
|
||||||
|
*/
|
||||||
|
INTERRUPTED
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The outcome of one {@link #awaitHerdr} call, carrying the MEASURED elapsed wait time
|
||||||
|
* alongside {@link #result}. {@code elapsedNanos} is always measured against the {@code nanos}
|
||||||
|
* supplier passed to {@link #awaitHerdr} — never assume it equals the configured budget, the
|
||||||
|
* same defect fleetd #494 already fixed once in {@code LeadRollover}.
|
||||||
|
*/
|
||||||
|
record HerdrAwaitOutcome(HerdrWaitResult result, long elapsedNanos) {}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The real per-poll wait {@link #main} passes to {@link #awaitHerdr}: sleep
|
||||||
|
* {@link #HERDR_WAIT_POLL_MILLIS}, and on interruption re-set the thread's interrupt flag
|
||||||
|
* rather than throwing — {@link #awaitHerdr} detects an interruption by checking {@link
|
||||||
|
* Thread#isInterrupted()} right after this returns, so a poller that swallowed the flag
|
||||||
|
* instead of restoring it would make that check silently miss the interruption.
|
||||||
|
*/
|
||||||
|
private static void sleepHerdrPoll() {
|
||||||
|
try {
|
||||||
|
Thread.sleep(HERDR_WAIT_POLL_MILLIS);
|
||||||
|
} catch (InterruptedException ie) {
|
||||||
|
Thread.currentThread().interrupt();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Poll herdr's {@code ping} until it answers, the configured {@link #HERDR_WAIT_SECONDS}
|
||||||
|
* budget elapses, or the waiting thread is interrupted (CB-504, fleetd #498).
|
||||||
|
*
|
||||||
|
* <p>{@code nanos} and {@code poller} are required parameters with no defaulted overload
|
||||||
|
* (fleetd #415's shape: a defaulted overload is a silent survivor a green suite would vouch
|
||||||
|
* for) — the previous version read {@link System#nanoTime()} and called {@link Thread#sleep}
|
||||||
|
* directly, so nothing could drive it from a test. The one production call site in {@link
|
||||||
|
* #main} passes {@code System::nanoTime} and {@link #sleepHerdrPoll}.
|
||||||
|
*
|
||||||
|
* @param nanos a monotonic elapsed-time clock, e.g. {@code System::nanoTime} — never a
|
||||||
|
* wall-clock source, since only elapsed time (not a timestamp) is measured here
|
||||||
|
* @param poller called once per failed ping while the budget remains; must, on an
|
||||||
|
* {@link InterruptedException}, re-set the thread's interrupt flag rather than
|
||||||
|
* throw or swallow it — this method's interruption check reads that flag right
|
||||||
|
* after {@code poller.run()} returns
|
||||||
|
* @return the outcome and the measured elapsed wait time — see {@link HerdrAwaitOutcome}
|
||||||
|
*/
|
||||||
|
static HerdrAwaitOutcome awaitHerdr(HerdrClient herdr, LongSupplier nanos, Runnable poller) {
|
||||||
|
long start = nanos.getAsLong();
|
||||||
|
long deadline = start + HERDR_WAIT_SECONDS * 1_000_000_000L;
|
||||||
boolean waited = false;
|
boolean waited = false;
|
||||||
while (true) {
|
while (true) {
|
||||||
try {
|
try {
|
||||||
@@ -1709,25 +1944,57 @@ public final class Fleetd {
|
|||||||
if (waited) {
|
if (waited) {
|
||||||
log.info("herdr is up");
|
log.info("herdr is up");
|
||||||
}
|
}
|
||||||
return true;
|
return new HerdrAwaitOutcome(HerdrWaitResult.ANSWERED, nanos.getAsLong() - start);
|
||||||
} catch (HerdrException e) {
|
} catch (HerdrException e) {
|
||||||
if (System.nanoTime() >= deadline) {
|
if (nanos.getAsLong() >= deadline) {
|
||||||
return false;
|
return new HerdrAwaitOutcome(HerdrWaitResult.DEADLINE_PASSED, nanos.getAsLong() - start);
|
||||||
}
|
}
|
||||||
if (!waited) {
|
if (!waited) {
|
||||||
log.info("waiting up to {}s for the herdr socket…", HERDR_WAIT_SECONDS);
|
log.info("waiting up to {}s for the herdr socket…", HERDR_WAIT_SECONDS);
|
||||||
waited = true;
|
waited = true;
|
||||||
}
|
}
|
||||||
try {
|
poller.run();
|
||||||
Thread.sleep(HERDR_WAIT_POLL_MILLIS);
|
if (Thread.currentThread().isInterrupted()) {
|
||||||
} catch (InterruptedException ie) {
|
return new HerdrAwaitOutcome(HerdrWaitResult.INTERRUPTED, nanos.getAsLong() - start);
|
||||||
Thread.currentThread().interrupt();
|
|
||||||
return false;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Log the right message for {@code outcome} — never the configured {@link #HERDR_WAIT_SECONDS}
|
||||||
|
* budget alone, always the measured elapsed time next to it — and say whether {@link #main}
|
||||||
|
* should now reap orphan worker panes (fleetd #498).
|
||||||
|
*
|
||||||
|
* <p>Extracted out of {@link #main} so this decision is drivable from a test: {@link #main}
|
||||||
|
* boots the whole daemon and cannot itself be run in a unit test, but this is the exact,
|
||||||
|
* unmodified code {@link #main} calls for the decision, not a re-derivation of it.
|
||||||
|
*
|
||||||
|
* @return true only for {@link HerdrWaitResult#ANSWERED} — orphan workers are reaped only
|
||||||
|
* then, exactly as before this ticket
|
||||||
|
*/
|
||||||
|
static boolean logHerdrWaitOutcomeAndShouldReap(HerdrAwaitOutcome outcome) {
|
||||||
|
long elapsedMillis = TimeUnit.NANOSECONDS.toMillis(outcome.elapsedNanos());
|
||||||
|
if (outcome.result() == HerdrWaitResult.ANSWERED) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
if (outcome.result() == HerdrWaitResult.DEADLINE_PASSED) {
|
||||||
|
log.warn("herdr did not answer within the configured wait (configured={}s elapsed={}ms) "
|
||||||
|
+ "— starting anyway; /healthz will report degraded until it comes up. Orphaned "
|
||||||
|
+ "worker panes (if any) were NOT reaped.",
|
||||||
|
HERDR_WAIT_SECONDS, elapsedMillis);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
// HerdrWaitResult.INTERRUPTED — a different fact from DEADLINE_PASSED (fleetd #498): the
|
||||||
|
// wait was cut short, not exhausted, and must never be reported as "did not answer within
|
||||||
|
// Ns" — that claim would be false and would send an operator to debug herdr for nothing.
|
||||||
|
log.warn("herdr wait was interrupted before the configured wait ran out (configured={}s "
|
||||||
|
+ "elapsed={}ms) — starting anyway; /healthz will report degraded until it comes "
|
||||||
|
+ "up. Orphaned worker panes (if any) were NOT reaped.",
|
||||||
|
HERDR_WAIT_SECONDS, elapsedMillis);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
private Fleetd() {
|
private Fleetd() {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -221,7 +221,8 @@ public final class CallerResolver {
|
|||||||
// The config/live binding names this pane as an architect slot's own. Same
|
// The config/live binding names this pane as an architect slot's own. Same
|
||||||
// unforgeable pane mapping; the live binding, never a request argument, decides.
|
// unforgeable pane mapping; the live binding, never a request argument, decides.
|
||||||
// Check the slot role too: this defence in depth prevents a bad lifecycle bind from
|
// Check the slot role too: this defence in depth prevents a bad lifecycle bind from
|
||||||
// escalating a dev or reviewer into an architect. Checked before the worker fallback.
|
// escalating a dev, hunter or reviewer into an architect. Checked before
|
||||||
|
// the worker fallback.
|
||||||
return Principal.architect(memberSlotNames.apply(slot), c.terminal(), c.pid());
|
return Principal.architect(memberSlotNames.apply(slot), c.terminal(), c.pid());
|
||||||
}
|
}
|
||||||
return Principal.worker(c.terminal(), c.pid()); // unforgeable; never token-gated
|
return Principal.worker(c.terminal(), c.pid()); // unforgeable; never token-gated
|
||||||
@@ -244,7 +245,15 @@ public final class CallerResolver {
|
|||||||
// already names what happens if that case is handed the primary role: a worker→primary
|
// already names what happens if that case is handed the primary role: a worker→primary
|
||||||
// escalation. So an unresolved caller is refused (ANONYMOUS — the same clean, already-tested
|
// escalation. So an unresolved caller is refused (ANONYMOUS — the same clean, already-tested
|
||||||
// "authenticated as nothing" outcome used everywhere else in this method), never promoted.
|
// "authenticated as nothing" outcome used everywhere else in this method), never promoted.
|
||||||
return isLoopback(remoteAddr) && c.resolved() ? Principal.primary(c.pid()) : Principal.anonymous();
|
//
|
||||||
|
// fleetd #505: the OTHER way a real pid can wrongly reach here with a null terminal — not a
|
||||||
|
// failed lsof lookup, but a herdr error partway through PaneLocator's pane scan. c.resolved()
|
||||||
|
// says nothing about that; it only tests the lsof sentinel (by design — see
|
||||||
|
// ConnectionIdentity.Caller#resolved). c.scanComplete() is the separate signal: a scan that
|
||||||
|
// could not check every pane must not be read as "checked everywhere, no match" — the pane it
|
||||||
|
// could not check might have been the caller's own. So both must hold before this promotes.
|
||||||
|
return isLoopback(remoteAddr) && c.resolved() && c.scanComplete()
|
||||||
|
? Principal.primary(c.pid()) : Principal.anonymous();
|
||||||
}
|
}
|
||||||
|
|
||||||
private boolean presentedTokenMatches(String authorizationHeader) {
|
private boolean presentedTokenMatches(String authorizationHeader) {
|
||||||
|
|||||||
@@ -42,7 +42,8 @@ public interface MemberLifecycle {
|
|||||||
* Try to bind a newly spawned {@code terminal} into the role it was granted.
|
* Try to bind a newly spawned {@code terminal} into the role it was granted.
|
||||||
*
|
*
|
||||||
* @return the role this session actually holds: {@code role} unchanged for a role with no
|
* @return the role this session actually holds: {@code role} unchanged for a role with no
|
||||||
* live slot-binding semantics (dev, reviewer), or when the bind succeeded; a fallback
|
* live slot-binding semantics (dev, hunter, reviewer), or when the bind
|
||||||
|
* succeeded; a fallback
|
||||||
* role — never {@code role} — when a slot-bound role (architect) could not be bound.
|
* role — never {@code role} — when a slot-bound role (architect) could not be bound.
|
||||||
* Callers must record THIS value on the session, never the requested {@code role}, so
|
* Callers must record THIS value on the session, never the requested {@code role}, so
|
||||||
* a later roster read never reports a role the session does not hold (CB-619). In
|
* a later roster read never reports a role the session does not hold (CB-619). In
|
||||||
|
|||||||
@@ -20,7 +20,8 @@ import java.util.function.Supplier;
|
|||||||
*
|
*
|
||||||
* <p>Two halves, split by who owns each:
|
* <p>Two halves, split by who owns each:
|
||||||
* <ul>
|
* <ul>
|
||||||
* <li><b>slots</b> — read from {@code fleet.architects}/{@code developers}/{@code reviewers}
|
* <li><b>slots</b> — read from {@code fleet.architects}/{@code developers}/
|
||||||
|
* {@code hunters}/{@code reviewers}
|
||||||
* (see {@link #slots()}), each carrying the {@code profile} reference the spawn lifecycle
|
* (see {@link #slots()}), each carrying the {@code profile} reference the spawn lifecycle
|
||||||
* reads when it stands the slot up. <strong>Live, since fleetd #424</strong>: {@link #live}
|
* reads when it stands the slot up. <strong>Live, since fleetd #424</strong>: {@link #live}
|
||||||
* re-reads {@code fleet:} on every call, through a supplier the same shape as
|
* re-reads {@code fleet:} on every call, through a supplier the same shape as
|
||||||
@@ -322,7 +323,7 @@ public final class MemberRegistry implements MemberLifecycle {
|
|||||||
* CB-619 / fleetd #123: refuse an architect acquire before anything spawns when no configured
|
* CB-619 / fleetd #123: refuse an architect acquire before anything spawns when no configured
|
||||||
* slot carries {@code profile} — the config-gap case from the original defect report (a spawn
|
* slot carries {@code profile} — the config-gap case from the original defect report (a spawn
|
||||||
* asked for {@code role=architect, profile=sonnet}, and {@code fleet.architects} carried only
|
* asked for {@code role=architect, profile=sonnet}, and {@code fleet.architects} carried only
|
||||||
* {@code opus} and {@code sol}). A dev/reviewer acquire is always a no-op: those pools are
|
* {@code opus} and {@code sol}). A dev/hunter/reviewer acquire is always a no-op: those pools are
|
||||||
* placement candidates only (see {@code CompositePeerLauncher}), never a live identity binding,
|
* placement candidates only (see {@code CompositePeerLauncher}), never a live identity binding,
|
||||||
* so there is nothing here to refuse — an explicit profile outside the pool for those roles is a
|
* so there is nothing here to refuse — an explicit profile outside the pool for those roles is a
|
||||||
* documented operator override, not a defect.
|
* documented operator override, not a defect.
|
||||||
|
|||||||
@@ -31,7 +31,7 @@ import java.util.function.Supplier;
|
|||||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Both are
|
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Both are
|
||||||
* read through a supplier on {@code CompositePeerLauncher}, which is what makes them hot —
|
* read through a supplier on {@code CompositePeerLauncher}, which is what makes them hot —
|
||||||
* not the fact that they are config. Most of {@code fleet:} — every role pool
|
* not the fact that they are config. Most of {@code fleet:} — every role pool
|
||||||
* ({@code architects}/{@code developers}/{@code reviewers}), {@code charters}, and
|
* ({@code architects}/{@code developers}/{@code hunters}/{@code reviewers}), {@code charters}, and
|
||||||
* {@code tabLabel} — is read the same live way, through the same supplier
|
* {@code tabLabel} — is read the same live way, through the same supplier
|
||||||
* ({@code () -> config.get().fleet()}). {@code architects} in particular is hot for
|
* ({@code () -> config.get().fleet()}). {@code architects} in particular is hot for
|
||||||
* <strong>two independent consumers</strong> (fleetd #424): {@code CompositePeerLauncher}
|
* <strong>two independent consumers</strong> (fleetd #424): {@code CompositePeerLauncher}
|
||||||
@@ -208,7 +208,7 @@ import java.util.function.Supplier;
|
|||||||
* that is correctly hot and for a key nobody triaged. Three times now — {@code worktreeGroup} (#323),
|
* that is correctly hot and for a key nobody triaged. Three times now — {@code worktreeGroup} (#323),
|
||||||
* {@code primary}/{@code configReload} (#326), and {@code fleet.leaders} sitting in the escape hatch
|
* {@code primary}/{@code configReload} (#326), and {@code fleet.leaders} sitting in the escape hatch
|
||||||
* (#333) — the second kind hid among the first. A top-level coverage checker in the
|
* (#333) — the second kind hid among the first. A top-level coverage checker in the
|
||||||
* {@link ConfigRefProfileCoverageTest} shape (one level up, over {@code FleetConfig} itself rather
|
* {@code ConfigRefProfileCoverageTest} shape (one level up, over {@code FleetConfig} itself rather
|
||||||
* than {@code FleetConfig.Profile}) proves this file's four classes exhaust the record's components
|
* than {@code FleetConfig.Profile}) proves this file's four classes exhaust the record's components
|
||||||
* — see {@code ConfigRefTopLevelCoverageTest}. That test proves the record's <em>shape</em> is fully
|
* — see {@code ConfigRefTopLevelCoverageTest}. That test proves the record's <em>shape</em> is fully
|
||||||
* triaged; it does NOT prove a {@code SPLIT_KEYS}/{@code COLD_KEYS}/{@code DEFERRED_KEYS} member has
|
* triaged; it does NOT prove a {@code SPLIT_KEYS}/{@code COLD_KEYS}/{@code DEFERRED_KEYS} member has
|
||||||
@@ -605,7 +605,7 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
|||||||
+ "opened once and needs a restart; the broker URI env-var name kept out of a "
|
+ "opened once and needs a restart; the broker URI env-var name kept out of a "
|
||||||
+ "member's environment is read live on every spawn and already applied");
|
+ "member's environment is read live on every spawn and already applied");
|
||||||
}
|
}
|
||||||
// fleetd #333: unlike health/coordinator above, most of `fleet:` (developers, reviewers,
|
// fleetd #333: unlike health/coordinator above, most of `fleet:` (developers, hunters, reviewers,
|
||||||
// charters, tabLabel) is genuinely hot — ConfigRefTest.aHotChangeIsAppliedAndRead-
|
// charters, tabLabel) is genuinely hot — ConfigRefTest.aHotChangeIsAppliedAndRead-
|
||||||
// ThroughGet and aCharterChangeIsHotAndReachesTheLiveConfig prove it reaches the live config
|
// ThroughGet and aCharterChangeIsHotAndReachesTheLiveConfig prove it reaches the live config
|
||||||
// with no restart note. `architects` is hot too, and — since fleetd #424 — hot for BOTH of
|
// with no restart note. `architects` is hot too, and — since fleetd #424 — hot for BOTH of
|
||||||
@@ -630,7 +630,7 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
|||||||
+ "identity map and to auto-launch leads, and neither is rebuilt on reload, so a "
|
+ "identity map and to auto-launch leads, and neither is rebuilt on reload, so a "
|
||||||
+ "lead added, removed, or given a new tab: label needs a restart — until then it "
|
+ "lead added, removed, or given a new tab: label needs a restart — until then it "
|
||||||
+ "stays unrecognised, and a caller from its new tab resolves as a worker, not a "
|
+ "stays unrecognised, and a caller from its new tab resolves as a worker, not a "
|
||||||
+ "lead; the rest of fleet: (developers, reviewers, charters, tabLabel) is read "
|
+ "lead; the rest of fleet: (developers, hunters, reviewers, charters, tabLabel) is read "
|
||||||
+ "live through the supplier on CompositePeerLauncher, and architects is read "
|
+ "live through the supplier on CompositePeerLauncher, and architects is read "
|
||||||
+ "live through that same supplier for placement AND through a separate supplier "
|
+ "live through that same supplier for placement AND through a separate supplier "
|
||||||
+ "on MemberRegistry for spawn-time identity — both already applied");
|
+ "on MemberRegistry for spawn-time identity — both already applied");
|
||||||
|
|||||||
@@ -62,7 +62,7 @@ import java.util.regex.PatternSyntaxException;
|
|||||||
* @param fleet who the daemon may run and under which role (CB-557). One block replacing
|
* @param fleet who the daemon may run and under which role (CB-557). One block replacing
|
||||||
* the former {@code leaders:}, {@code members:}, {@code leadScan:} and
|
* the former {@code leaders:}, {@code members:}, {@code leadScan:} and
|
||||||
* {@code defaultProfile:}. Role is the containing key — {@code leaders},
|
* {@code defaultProfile:}. Role is the containing key — {@code leaders},
|
||||||
* {@code architects}, {@code developers}, {@code reviewers} — and each entry
|
* {@code architects}, {@code developers}, {@code hunters}, {@code reviewers} — and each entry
|
||||||
* names the {@code profiles:} backend it runs on. See {@link Fleet}
|
* names the {@code profiles:} backend it runs on. See {@link Fleet}
|
||||||
* @param leadHeartbeat opt-in idle-lead heartbeat (CB-551); {@code null} ⇒ off, and an upgraded
|
* @param leadHeartbeat opt-in idle-lead heartbeat (CB-551); {@code null} ⇒ off, and an upgraded
|
||||||
* daemon never nudges an idle lead on its own initiative
|
* daemon never nudges an idle lead on its own initiative
|
||||||
@@ -1200,6 +1200,7 @@ public record FleetConfig(
|
|||||||
* @param leaders panes that orchestrate rather than are orchestrated, keyed by lead name
|
* @param leaders panes that orchestrate rather than are orchestrated, keyed by lead name
|
||||||
* @param architects profiles the {@code architect} role may run on
|
* @param architects profiles the {@code architect} role may run on
|
||||||
* @param developers profiles the {@code dev} role may run on
|
* @param developers profiles the {@code dev} role may run on
|
||||||
|
* @param hunters profiles the {@code hunter} role may run on
|
||||||
* @param reviewers profiles the {@code reviewer} role may run on
|
* @param reviewers profiles the {@code reviewer} role may run on
|
||||||
* @param charters optional launch-charter text keyed by singular role wire name
|
* @param charters optional launch-charter text keyed by singular role wire name
|
||||||
* @param tabLabel template for a member tab's label; {@code {role}}, {@code {profile}},
|
* @param tabLabel template for a member tab's label; {@code {role}}, {@code {profile}},
|
||||||
@@ -1210,6 +1211,7 @@ public record FleetConfig(
|
|||||||
public record Fleet(Map<String, Leader> leaders,
|
public record Fleet(Map<String, Leader> leaders,
|
||||||
Map<String, Slot> architects,
|
Map<String, Slot> architects,
|
||||||
Map<String, Slot> developers,
|
Map<String, Slot> developers,
|
||||||
|
Map<String, Slot> hunters,
|
||||||
Map<String, Slot> reviewers,
|
Map<String, Slot> reviewers,
|
||||||
Map<String, String> charters,
|
Map<String, String> charters,
|
||||||
String tabLabel) {
|
String tabLabel) {
|
||||||
@@ -1226,6 +1228,7 @@ public record FleetConfig(
|
|||||||
leaders = unmodifiableOrEmpty(leaders);
|
leaders = unmodifiableOrEmpty(leaders);
|
||||||
architects = unmodifiableOrEmpty(architects);
|
architects = unmodifiableOrEmpty(architects);
|
||||||
developers = unmodifiableOrEmpty(developers);
|
developers = unmodifiableOrEmpty(developers);
|
||||||
|
hunters = unmodifiableOrEmpty(hunters);
|
||||||
reviewers = unmodifiableOrEmpty(reviewers);
|
reviewers = unmodifiableOrEmpty(reviewers);
|
||||||
charters = unmodifiableOrEmpty(charters);
|
charters = unmodifiableOrEmpty(charters);
|
||||||
tabLabel = (tabLabel == null || tabLabel.isBlank()) ? DEFAULT_TAB_LABEL : tabLabel;
|
tabLabel = (tabLabel == null || tabLabel.isBlank()) ? DEFAULT_TAB_LABEL : tabLabel;
|
||||||
@@ -1240,9 +1243,15 @@ public record FleetConfig(
|
|||||||
* constructor: the launcher reads {@code fleet.charters()} from the live config. Jackson
|
* constructor: the launcher reads {@code fleet.charters()} from the live config. Jackson
|
||||||
* binds the canonical constructor, so this one cannot swallow an operator's YAML.
|
* binds the canonical constructor, so this one cannot swallow an operator's YAML.
|
||||||
*/
|
*/
|
||||||
|
public Fleet(Map<String, Leader> leaders, Map<String, Slot> architects,
|
||||||
|
Map<String, Slot> developers, Map<String, Slot> reviewers,
|
||||||
|
Map<String, String> charters, String tabLabel) {
|
||||||
|
this(leaders, architects, developers, null, reviewers, charters, tabLabel);
|
||||||
|
}
|
||||||
|
|
||||||
public Fleet(Map<String, Leader> leaders, Map<String, Slot> architects,
|
public Fleet(Map<String, Leader> leaders, Map<String, Slot> architects,
|
||||||
Map<String, Slot> developers, Map<String, Slot> reviewers, String tabLabel) {
|
Map<String, Slot> developers, Map<String, Slot> reviewers, String tabLabel) {
|
||||||
this(leaders, architects, developers, reviewers, null, tabLabel);
|
this(leaders, architects, developers, null, reviewers, null, tabLabel);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -1264,6 +1273,7 @@ public record FleetConfig(
|
|||||||
return switch (role) {
|
return switch (role) {
|
||||||
case ARCHITECT -> architects;
|
case ARCHITECT -> architects;
|
||||||
case DEV -> developers;
|
case DEV -> developers;
|
||||||
|
case HUNTER -> hunters;
|
||||||
case REVIEWER -> reviewers;
|
case REVIEWER -> reviewers;
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
@@ -1670,7 +1680,7 @@ public record FleetConfig(
|
|||||||
* <p>A herdr pane runs a login shell that re-sources the operator's own secret store, so a
|
* <p>A herdr pane runs a login shell that re-sources the operator's own secret store, so a
|
||||||
* member inherits every credential the operator's shell holds — measured at 31 names on this
|
* member inherits every credential the operator's shell holds — measured at 31 names on this
|
||||||
* host, of which only one ({@code GITEA_ACCESS_TOKEN}) used to be blocked, and that block was a
|
* host, of which only one ({@code GITEA_ACCESS_TOKEN}) used to be blocked, and that block was a
|
||||||
* single name hardcoded in {@link HerdrPeerLauncher} rather than driven by config (gitea issue
|
* single name hardcoded in {@link dev.ltms.fleet.member.HerdrPeerLauncher} rather than driven by config (gitea issue
|
||||||
* #82). This record replaces that hardcoded shadow with a config-driven one.
|
* #82). This record replaces that hardcoded shadow with a config-driven one.
|
||||||
*
|
*
|
||||||
* <p><b>deny-by-default, not a deny-list.</b> A deny-list (block these specific names, let
|
* <p><b>deny-by-default, not a deny-list.</b> A deny-list (block these specific names, let
|
||||||
@@ -1872,7 +1882,7 @@ public record FleetConfig(
|
|||||||
|
|
||||||
/** The {@code fleet:} child blocks whose direct children are slot names. */
|
/** The {@code fleet:} child blocks whose direct children are slot names. */
|
||||||
private static final Set<String> FLEET_POOL_KEYS =
|
private static final Set<String> FLEET_POOL_KEYS =
|
||||||
Set.of("leaders", "architects", "developers", "reviewers");
|
Set.of("leaders", "architects", "developers", "hunters", "reviewers");
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Reject a {@code fleet:} role pool whose slot names repeat (CB-548, re-homed by CB-557).
|
* Reject a {@code fleet:} role pool whose slot names repeat (CB-548, re-homed by CB-557).
|
||||||
@@ -1882,7 +1892,7 @@ public record FleetConfig(
|
|||||||
* daemon would never know. Jackson's YAML parser does not fail on duplicate mapping keys by
|
* daemon would never know. Jackson's YAML parser does not fail on duplicate mapping keys by
|
||||||
* default, so duplicates are caught here, at parse time, before the map is built.
|
* default, so duplicates are caught here, at parse time, before the map is built.
|
||||||
*
|
*
|
||||||
* <p>Only the four pools <em>directly under the top-level {@code fleet:}</em> are considered,
|
* <p>Only the five pools <em>directly under the top-level {@code fleet:}</em> are considered,
|
||||||
* and only their direct child keys (the slot names). A nested field elsewhere, even one also
|
* and only their direct child keys (the slot names). A nested field elsewhere, even one also
|
||||||
* named {@code developers:}, is ignored, so parsing of the rest of the config is unaffected.
|
* named {@code developers:}, is ignored, so parsing of the rest of the config is unaffected.
|
||||||
*
|
*
|
||||||
@@ -2049,8 +2059,9 @@ public record FleetConfig(
|
|||||||
"defaultProfile", "a role pool under 'fleet:' — an unqualified spawn now names a role,"
|
"defaultProfile", "a role pool under 'fleet:' — an unqualified spawn now names a role,"
|
||||||
+ " and that role's pool supplies the candidate profiles",
|
+ " and that role's pool supplies the candidate profiles",
|
||||||
"architects", "'fleet.architects'",
|
"architects", "'fleet.architects'",
|
||||||
"members", "a role pool under 'fleet:' — 'fleet.architects', 'fleet.developers' or"
|
"members", "a role pool under 'fleet:' — 'fleet.architects', 'fleet.developers',"
|
||||||
+ " 'fleet.reviewers'; the role is the containing key, not a 'role:' field",
|
+ " 'fleet.hunters' or 'fleet.reviewers'; the role is the containing key, not"
|
||||||
|
+ " a 'role:' field",
|
||||||
"leaders", "'fleet.leaders'",
|
"leaders", "'fleet.leaders'",
|
||||||
"leadScan", "'fleet.leaders.<name>.tabPrefix' and '.scanIntervalSeconds' — lead"
|
"leadScan", "'fleet.leaders.<name>.tabPrefix' and '.scanIntervalSeconds' — lead"
|
||||||
+ " discovery is now configured on the lead it discovers");
|
+ " discovery is now configured on the lead it discovers");
|
||||||
|
|||||||
@@ -1,6 +1,8 @@
|
|||||||
package dev.ltms.fleet.herdr;
|
package dev.ltms.fleet.herdr;
|
||||||
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
import org.slf4j.Logger;
|
||||||
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import java.util.LinkedHashSet;
|
import java.util.LinkedHashSet;
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
@@ -36,6 +38,8 @@ import java.util.Set;
|
|||||||
*/
|
*/
|
||||||
public final class PaneLocator {
|
public final class PaneLocator {
|
||||||
|
|
||||||
|
private static final Logger log = LoggerFactory.getLogger(PaneLocator.class);
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Bound on how many ancestor generations {@link #ancestorsOf} walks. This runs on every MCP
|
* Bound on how many ancestor generations {@link #ancestorsOf} walks. This runs on every MCP
|
||||||
* call, so a cycle or a pathologically deep process tree must not hang identity resolution;
|
* call, so a cycle or a pathologically deep process tree must not hang identity resolution;
|
||||||
@@ -73,22 +77,46 @@ public final class PaneLocator {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The {@code terminal_id} of the agent pane whose process tree contains {@code pid}, or
|
* The outcome of a {@link #terminalForPid} scan: the {@code terminal_id} of the agent pane
|
||||||
* {@code null} if no agent pane on any searched daemon owns it (e.g. the caller is the
|
* whose process tree contains the pid ({@link #terminal} is {@code null} if none matched),
|
||||||
* primary, or off-host).
|
* and whether the scan that produced that answer ran to completion on every daemon searched.
|
||||||
|
*
|
||||||
|
* <p>{@link #complete} is {@code false} exactly when some {@code pane.process_info} call
|
||||||
|
* failed and, despite that, no pane was ever found to own the pid. In that case a {@code null}
|
||||||
|
* {@link #terminal} means "could not tell", not "definitely not a worker" — fleetd #505: a
|
||||||
|
* transient herdr error on the very pane that <em>does</em> own the caller's pid must not read
|
||||||
|
* as a clean negative and fall through to {@code Principal.primary}, the same way #317's
|
||||||
|
* {@code Caller.resolved()} already guards a failed lsof lookup. Callers ({@code
|
||||||
|
* ConnectionIdentity}, {@code CallerResolver}) must refuse rather than promote on an incomplete
|
||||||
|
* scan.
|
||||||
|
*
|
||||||
|
* <p>When a pane genuinely owns the pid, {@link #complete} is {@code true} regardless of
|
||||||
|
* whether some other, unrelated pane failed to answer earlier in the same scan — a positive
|
||||||
|
* match is definitive and does not need every pane to have been checked (a pane that "vanished
|
||||||
|
* mid-scan" but was never the match is still a clean, complete result).
|
||||||
*/
|
*/
|
||||||
public String terminalForPid(long pid) {
|
public record Lookup(String terminal, boolean complete) {
|
||||||
|
private static final Lookup NOT_FOUND = new Lookup(null, true);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Resolve {@code pid} to the agent pane whose process tree contains it, across every searched
|
||||||
|
* herdr daemon. See {@link Lookup} for how to read a {@code null} terminal.
|
||||||
|
*/
|
||||||
|
public Lookup terminalForPid(long pid) {
|
||||||
if (pid <= 0) {
|
if (pid <= 0) {
|
||||||
return null;
|
return Lookup.NOT_FOUND;
|
||||||
}
|
}
|
||||||
Set<Long> ancestry = ancestorsOf(pid);
|
Set<Long> ancestry = ancestorsOf(pid);
|
||||||
for (HerdrClient herdr : herdrs) {
|
boolean complete = true;
|
||||||
String terminal = terminalForPid(herdr, ancestry);
|
for (int i = 0; i < herdrs.size(); i++) {
|
||||||
if (terminal != null) {
|
Lookup outcome = scan(herdrs.get(i), i, herdrs.size(), ancestry);
|
||||||
return terminal;
|
if (outcome.terminal() != null) {
|
||||||
|
return outcome; // a definite match — no need to finish checking other clients
|
||||||
}
|
}
|
||||||
|
complete = complete && outcome.complete();
|
||||||
}
|
}
|
||||||
return null;
|
return new Lookup(null, complete);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -117,31 +145,51 @@ public final class PaneLocator {
|
|||||||
return ancestry;
|
return ancestry;
|
||||||
}
|
}
|
||||||
|
|
||||||
private static String terminalForPid(HerdrClient herdr, Set<Long> ancestry) {
|
/** Whether a pane owns one of the scanned pid's ancestors, or the check of it failed outright. */
|
||||||
|
private enum Ownership { OWNS, DOES_NOT_OWN, UNKNOWN }
|
||||||
|
|
||||||
|
private static Lookup scan(HerdrClient herdr, int clientIndex, int clientCount, Set<Long> ancestry) {
|
||||||
|
boolean complete = true;
|
||||||
for (JsonNode pane : herdr.call("pane.list", Map.of()).path("panes")) {
|
for (JsonNode pane : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||||
String paneId = pane.path("pane_id").asText(null);
|
String paneId = pane.path("pane_id").asText(null);
|
||||||
if (paneId != null && paneOwnsAnyOf(herdr, paneId, ancestry)) {
|
if (paneId == null) {
|
||||||
return pane.path("terminal_id").asText(null);
|
continue;
|
||||||
|
}
|
||||||
|
Ownership owns = paneOwnsAnyOf(herdr, clientIndex, clientCount, paneId, ancestry);
|
||||||
|
if (owns == Ownership.OWNS) {
|
||||||
|
return new Lookup(pane.path("terminal_id").asText(null), true);
|
||||||
|
}
|
||||||
|
if (owns == Ownership.UNKNOWN) {
|
||||||
|
complete = false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return null;
|
return new Lookup(null, complete);
|
||||||
}
|
}
|
||||||
|
|
||||||
private static boolean paneOwnsAnyOf(HerdrClient herdr, String paneId, Set<Long> ancestry) {
|
private static Ownership paneOwnsAnyOf(HerdrClient herdr, int clientIndex, int clientCount,
|
||||||
|
String paneId, Set<Long> ancestry) {
|
||||||
JsonNode info;
|
JsonNode info;
|
||||||
try {
|
try {
|
||||||
info = herdr.call("pane.process_info", Map.of("pane_id", paneId)).path("process_info");
|
info = herdr.call("pane.process_info", Map.of("pane_id", paneId)).path("process_info");
|
||||||
} catch (HerdrException e) {
|
} catch (HerdrException e) {
|
||||||
return false; // pane vanished mid-scan — just skip it
|
// fleetd #505: this used to be read as a clean "does not own it" (the pane vanished
|
||||||
|
// mid-scan, just skip it) — one boolean carrying two different facts. It is UNKNOWN
|
||||||
|
// now: if THIS pane is the one that owns the pid, the caller must not be told "no pane
|
||||||
|
// owns it", because that reads as a real primary and is promoted under loopback-trust.
|
||||||
|
log.warn("pane.process_info failed for pane {} on herdr client {} of {} during a "
|
||||||
|
+ "pid-owner scan — treating it as \"could not tell\", not a clean "
|
||||||
|
+ "negative (fleetd #505): {}",
|
||||||
|
paneId, clientIndex + 1, clientCount, e.getMessage());
|
||||||
|
return Ownership.UNKNOWN;
|
||||||
}
|
}
|
||||||
if (ancestry.contains(info.path("shell_pid").asLong(-1))) {
|
if (ancestry.contains(info.path("shell_pid").asLong(-1))) {
|
||||||
return true;
|
return Ownership.OWNS;
|
||||||
}
|
}
|
||||||
for (JsonNode p : info.path("foreground_processes")) {
|
for (JsonNode p : info.path("foreground_processes")) {
|
||||||
if (ancestry.contains(p.path("pid").asLong(-1))) {
|
if (ancestry.contains(p.path("pid").asLong(-1))) {
|
||||||
return true;
|
return Ownership.OWNS;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return false;
|
return Ownership.DOES_NOT_OWN;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -50,7 +50,7 @@ import java.util.regex.Pattern;
|
|||||||
* so the scrape's herdr round-trip never stalls the status poller. The captured waiter is read on the
|
* so the scrape's herdr round-trip never stalls the status poller. The captured waiter is read on the
|
||||||
* poller thread (before any next-turn delivery can overwrite it) and passed into the virtual thread.
|
* poller thread (before any next-turn delivery can overwrite it) and passed into the virtual thread.
|
||||||
*/
|
*/
|
||||||
public final class CompletionResolver implements TurnListener {
|
public final class CompletionResolver implements TurnListener, TurnRegistrar {
|
||||||
|
|
||||||
private static final Logger log = LoggerFactory.getLogger(CompletionResolver.class);
|
private static final Logger log = LoggerFactory.getLogger(CompletionResolver.class);
|
||||||
|
|
||||||
@@ -235,6 +235,26 @@ public final class CompletionResolver implements TurnListener {
|
|||||||
this.worktreeBranches = Objects.requireNonNull(worktreeBranches, "worktreeBranches");
|
this.worktreeBranches = Objects.requireNonNull(worktreeBranches, "worktreeBranches");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #556: {@link TurnRegistrar}'s structural half of what {@link #captureBaseline} used to
|
||||||
|
* do alone — record the waiter, no I/O. Called directly and unconditionally by the {@link
|
||||||
|
* Injector} as part of delivering a turn, before {@link #onDelivered} ever runs, so this
|
||||||
|
* registration cannot be skipped by a {@link TurnListener} throwing (from {@code onDelivered} or
|
||||||
|
* any other callback). {@link #captureBaseline} still performs this exact check-and-put itself
|
||||||
|
* as well — harmless and idempotent when it runs right after this — so a caller that only wires
|
||||||
|
* the {@link TurnListener} path (e.g. an existing test that never mentions {@link TurnRegistrar})
|
||||||
|
* keeps working unchanged.
|
||||||
|
*/
|
||||||
|
@Override
|
||||||
|
public void register(String target, TurnToken token) {
|
||||||
|
CompletableFuture<Rendezvous.Resolution> waiter = token.waiter();
|
||||||
|
if (waiter == null) {
|
||||||
|
inFlight.remove(target); // no send is waiting on this delivery — nothing to resolve later
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
inFlight.put(target, new InFlight(waiter, null, nowNanos.getAsLong(), token.injectedText()));
|
||||||
|
}
|
||||||
|
|
||||||
@Override
|
@Override
|
||||||
public void onDelivered(String target, TurnToken token) {
|
public void onDelivered(String target, TurnToken token) {
|
||||||
// Capture the exact waiter this turn belongs to (CB-116) and snapshot the pane's pre-turn
|
// Capture the exact waiter this turn belongs to (CB-116) and snapshot the pane's pre-turn
|
||||||
@@ -244,7 +264,14 @@ public final class CompletionResolver implements TurnListener {
|
|||||||
captureBaseline(target, token);
|
captureBaseline(target, token);
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Capture the in-flight turn: its waiter and pre-turn baseline (the testable core of {@link #onDelivered}). */
|
/**
|
||||||
|
* Capture the in-flight turn: its waiter and pre-turn baseline (the testable core of {@link
|
||||||
|
* #onDelivered}). fleetd #556: this still does its own full waiter check and {@code inFlight.put}
|
||||||
|
* — the same registration {@link #register} performs — so it keeps working standalone (as every
|
||||||
|
* test calling it directly already does) even though the {@link Injector} now also calls {@link
|
||||||
|
* #register} on its own, earlier and unconditionally. The two writes are idempotent with each
|
||||||
|
* other; only the second (this one) carries a real baseline, since only this one pays for a scrape.
|
||||||
|
*/
|
||||||
void captureBaseline(String target, TurnToken token) {
|
void captureBaseline(String target, TurnToken token) {
|
||||||
CompletableFuture<Rendezvous.Resolution> waiter = token.waiter();
|
CompletableFuture<Rendezvous.Resolution> waiter = token.waiter();
|
||||||
if (waiter == null) {
|
if (waiter == null) {
|
||||||
|
|||||||
@@ -2,6 +2,7 @@ package dev.ltms.fleet.inject;
|
|||||||
|
|
||||||
import dev.ltms.fleet.herdr.AgentControl;
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
import dev.ltms.fleet.herdr.AgentStatus;
|
import dev.ltms.fleet.herdr.AgentStatus;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrException;
|
||||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||||
import dev.ltms.fleet.msg.TurnToken;
|
import dev.ltms.fleet.msg.TurnToken;
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
@@ -11,10 +12,12 @@ import java.util.ArrayDeque;
|
|||||||
import java.util.ArrayList;
|
import java.util.ArrayList;
|
||||||
import java.util.Deque;
|
import java.util.Deque;
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
|
import java.util.Objects;
|
||||||
import java.util.Set;
|
import java.util.Set;
|
||||||
import java.util.concurrent.CompletableFuture;
|
import java.util.concurrent.CompletableFuture;
|
||||||
import java.util.concurrent.ConcurrentHashMap;
|
import java.util.concurrent.ConcurrentHashMap;
|
||||||
import java.util.function.Consumer;
|
import java.util.function.Consumer;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
import java.util.function.Predicate;
|
import java.util.function.Predicate;
|
||||||
import java.util.stream.Collectors;
|
import java.util.stream.Collectors;
|
||||||
|
|
||||||
@@ -44,6 +47,17 @@ import java.util.stream.Collectors;
|
|||||||
* turn — the pickup-grace path (a turn too fast to sample) unwedges the queue but does not fire
|
* turn — the pickup-grace path (a turn too fast to sample) unwedges the queue but does not fire
|
||||||
* completion, since without a sampled {@code working} there is no trustworthy "the worker just
|
* completion, since without a sampled {@code working} there is no trustworthy "the worker just
|
||||||
* finished the task" signal to act on.
|
* finished the task" signal to act on.
|
||||||
|
*
|
||||||
|
* <p><strong>Delivery honesty (fleetd #551).</strong> The one external send call
|
||||||
|
* ({@link AgentControl#send}) is irreversible, and its own response can still fail after the text
|
||||||
|
* has already reached the worker's pane — herdr replied (or may have), so the request was already
|
||||||
|
* processed, but the reply itself then failed to parse or carried an error. A queued entry is
|
||||||
|
* polled off the queue and marked {@code ATTEMPTED} <em>before</em> that call is made, not after,
|
||||||
|
* so a failure from the call itself can never be recorded as a confident {@code NOT_DELIVERED} for
|
||||||
|
* text that may already be sitting in the pane. {@code NOT_DELIVERED} stays reserved for the cases
|
||||||
|
* where nothing was ever attempted — the readiness grace expiring, {@link #drop}, or a herdr
|
||||||
|
* {@code *_not_found} error, which the rest of this codebase already treats as a confirmed absence
|
||||||
|
* rather than a merely inconclusive failure (see {@code StatusPoller}, {@code AgentControl}).
|
||||||
*/
|
*/
|
||||||
public final class Injector {
|
public final class Injector {
|
||||||
|
|
||||||
@@ -92,8 +106,21 @@ public final class Injector {
|
|||||||
private final AgentControl agents;
|
private final AgentControl agents;
|
||||||
private final HerdrRouter router;
|
private final HerdrRouter router;
|
||||||
private final TurnListener turnListener;
|
private final TurnListener turnListener;
|
||||||
|
/**
|
||||||
|
* fleetd #556: the Injector's own registration invariant, called directly and unconditionally —
|
||||||
|
* never folded into {@link #turnListener}'s fan-out. See {@link TurnRegistrar}.
|
||||||
|
*/
|
||||||
|
private final TurnRegistrar registrar;
|
||||||
private final Predicate<String> ready; // CB-113: a target is deliverable only when available
|
private final Predicate<String> ready; // CB-113: a target is deliverable only when available
|
||||||
private final Consumer<String> forget; // CB-114: clear a gone worker's readiness/presence
|
private final Consumer<String> forget; // CB-114: clear a gone worker's readiness/presence
|
||||||
|
/**
|
||||||
|
* Wall-clock source for the readiness-grace elapsed time logged in {@link #onStatus} (fleetd
|
||||||
|
* #501). Production constructors default this to {@code System::currentTimeMillis}; the
|
||||||
|
* package-private constructors below take it explicitly so a test can supply a stub whose
|
||||||
|
* advance does not track {@link #POLL_INTERVAL_MILLIS} — copying the shape {@code LeadRollover}
|
||||||
|
* already uses for the same purpose.
|
||||||
|
*/
|
||||||
|
private final LongSupplier nowMillis;
|
||||||
private final ConcurrentHashMap<String, Target> targets = new ConcurrentHashMap<>();
|
private final ConcurrentHashMap<String, Target> targets = new ConcurrentHashMap<>();
|
||||||
|
|
||||||
/** Delivery only; completion signalling is a no-op and every target is treated as available. */
|
/** Delivery only; completion signalling is a no-op and every target is treated as available. */
|
||||||
@@ -125,20 +152,75 @@ public final class Injector {
|
|||||||
*/
|
*/
|
||||||
public Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
public Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||||
Consumer<String> forget) {
|
Consumer<String> forget) {
|
||||||
|
// fleetd #556: no explicit registrar named here, so fall back to turnListener itself when it
|
||||||
|
// happens to also implement TurnRegistrar (true for CompletionResolver, the only production
|
||||||
|
// TurnListener that needs CB-106 registration) — every existing caller of this overload keeps
|
||||||
|
// working unchanged. A turnListener that does NOT implement TurnRegistrar (a fan-out object,
|
||||||
|
// or a bare test lambda) gets TurnRegistrar.NOOP here, same as before this ticket.
|
||||||
|
this(agents, turnListener, ready, forget,
|
||||||
|
turnListener instanceof TurnRegistrar r ? r : TurnRegistrar.NOOP);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #556: explicit-registrar overload. Use this whenever {@code turnListener} does not
|
||||||
|
* itself implement {@link TurnRegistrar} — e.g. a fan-out object that also notifies unrelated
|
||||||
|
* observers — so registration is wired directly to the one component that owns it, rather than
|
||||||
|
* relying on {@code turnListener} happening to implement both interfaces.
|
||||||
|
*/
|
||||||
|
public Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||||
|
Consumer<String> forget, TurnRegistrar registrar) {
|
||||||
|
this(agents, turnListener, ready, forget, registrar, System::currentTimeMillis);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Package-private constructor — for tests: an injectable wall-clock supplier (fleetd #501),
|
||||||
|
* with no explicit registrar (fleetd #556 fallback, same as the public 4-arg overload above).
|
||||||
|
*/
|
||||||
|
Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||||
|
Consumer<String> forget, LongSupplier nowMillis) {
|
||||||
|
this(agents, turnListener, ready, forget,
|
||||||
|
turnListener instanceof TurnRegistrar r ? r : TurnRegistrar.NOOP, nowMillis);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Full constructor — for tests: an explicit registrar and an injectable wall-clock supplier. */
|
||||||
|
Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||||
|
Consumer<String> forget, TurnRegistrar registrar, LongSupplier nowMillis) {
|
||||||
this.agents = agents;
|
this.agents = agents;
|
||||||
this.router = null;
|
this.router = null;
|
||||||
this.turnListener = turnListener;
|
this.turnListener = turnListener;
|
||||||
|
this.registrar = Objects.requireNonNull(registrar, "registrar");
|
||||||
this.ready = ready;
|
this.ready = ready;
|
||||||
this.forget = forget;
|
this.forget = forget;
|
||||||
|
this.nowMillis = nowMillis;
|
||||||
}
|
}
|
||||||
|
|
||||||
public Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
public Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
||||||
Consumer<String> forget) {
|
Consumer<String> forget) {
|
||||||
|
// fleetd #556: see the AgentControl overload above for the fallback rationale.
|
||||||
|
this(router, turnListener, ready, forget,
|
||||||
|
turnListener instanceof TurnRegistrar r ? r : TurnRegistrar.NOOP);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #556: explicit-registrar overload — production wiring ({@code Fleetd}) uses this, since
|
||||||
|
* its {@code turnListener} is a fan-out object that also notifies {@code SessionManager} and does
|
||||||
|
* not itself implement {@link TurnRegistrar}.
|
||||||
|
*/
|
||||||
|
public Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
||||||
|
Consumer<String> forget, TurnRegistrar registrar) {
|
||||||
|
this(router, turnListener, ready, forget, registrar, System::currentTimeMillis);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Full constructor — for tests: an injectable wall-clock supplier (fleetd #501). */
|
||||||
|
Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
||||||
|
Consumer<String> forget, TurnRegistrar registrar, LongSupplier nowMillis) {
|
||||||
this.agents = null;
|
this.agents = null;
|
||||||
this.router = router;
|
this.router = router;
|
||||||
this.turnListener = turnListener;
|
this.turnListener = turnListener;
|
||||||
|
this.registrar = Objects.requireNonNull(registrar, "registrar");
|
||||||
this.ready = ready;
|
this.ready = ready;
|
||||||
this.forget = forget;
|
this.forget = forget;
|
||||||
|
this.nowMillis = nowMillis;
|
||||||
}
|
}
|
||||||
|
|
||||||
private AgentControl agentsFor(String target) {
|
private AgentControl agentsFor(String target) {
|
||||||
@@ -147,8 +229,23 @@ public final class Injector {
|
|||||||
|
|
||||||
/** The result of trying to remove an undelivered message from the injector. */
|
/** The result of trying to remove an undelivered message from the injector. */
|
||||||
public enum Cancellation {
|
public enum Cancellation {
|
||||||
|
/** The message was still queued and this call removed it; the target saw nothing. */
|
||||||
CANCELLED,
|
CANCELLED,
|
||||||
|
/** The message reached the worker's pane and is confirmed delivered. */
|
||||||
DELIVERED,
|
DELIVERED,
|
||||||
|
/**
|
||||||
|
* fleetd #551: the injector called {@link AgentControl#send} for this message and that call
|
||||||
|
* threw before its outcome was known — the text may or may not have reached the pane. A
|
||||||
|
* caller reporting this to an operator must say "uncertain", not "definitely not
|
||||||
|
* delivered": reading it as a confident negative invites a resend of text that may already
|
||||||
|
* be sitting in the pane (a double delivery), which is worse than the ambiguity itself.
|
||||||
|
*/
|
||||||
|
ATTEMPTED,
|
||||||
|
/**
|
||||||
|
* The message never reached the worker's pane — nothing was ever attempted for it (the
|
||||||
|
* readiness grace expired, the target was dropped, or send failed with a herdr
|
||||||
|
* {@code *_not_found} error, which is a confirmed absence, not merely inconclusive).
|
||||||
|
*/
|
||||||
NOT_DELIVERED
|
NOT_DELIVERED
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -171,7 +268,29 @@ public final class Injector {
|
|||||||
|
|
||||||
/** A pending message and the future that completes when it has been delivered. */
|
/** A pending message and the future that completes when it has been delivered. */
|
||||||
private static final class Pending {
|
private static final class Pending {
|
||||||
enum State { QUEUED, DELIVERED, NOT_DELIVERED, CANCELLED }
|
enum State {
|
||||||
|
/** Still in the target's queue, not yet attempted. */
|
||||||
|
QUEUED,
|
||||||
|
/**
|
||||||
|
* fleetd #551: {@link AgentControl#send} has been called for this entry and its outcome
|
||||||
|
* is not yet known — recorded BEFORE the call (see {@link #onStatus}), so a Throwable
|
||||||
|
* from send() can never leave the entry either still QUEUED or falsely marked
|
||||||
|
* NOT_DELIVERED. Upgraded to DELIVERED on success; left as ATTEMPTED on an ordinary
|
||||||
|
* failure, since reaching the catch does not prove the text never reached the pane.
|
||||||
|
*/
|
||||||
|
ATTEMPTED,
|
||||||
|
/** {@code send()} returned normally: the text is confirmed to have reached the pane. */
|
||||||
|
DELIVERED,
|
||||||
|
/**
|
||||||
|
* Confirmed — not merely inconclusive — that nothing was ever sent for this entry: the
|
||||||
|
* worker never became ready ({@link #onStatus}'s readiness-grace expiry), the target was
|
||||||
|
* dropped ({@link #drop}), or send failed with a herdr {@code *_not_found} error. Never
|
||||||
|
* written for an entry whose send() outcome is unknown; see ATTEMPTED.
|
||||||
|
*/
|
||||||
|
NOT_DELIVERED,
|
||||||
|
/** Removed from the queue by {@link #cancel} before it was ever attempted. */
|
||||||
|
CANCELLED
|
||||||
|
}
|
||||||
|
|
||||||
final String target;
|
final String target;
|
||||||
final String text;
|
final String text;
|
||||||
@@ -209,6 +328,7 @@ public final class Injector {
|
|||||||
int unknownSinceTurn; // consecutive `unknown` samples while a delegation is outstanding (CB-109)
|
int unknownSinceTurn; // consecutive `unknown` samples while a delegation is outstanding (CB-109)
|
||||||
int unknownSincePostTurn; // the same, for the post-turn housekeeping phase (fleetd #306)
|
int unknownSincePostTurn; // the same, for the post-turn housekeeping phase (fleetd #306)
|
||||||
int notReadySincePoll; // consecutive injectable samples a queued message waited on the readiness gate (CB-114)
|
int notReadySincePoll; // consecutive injectable samples a queued message waited on the readiness gate (CB-114)
|
||||||
|
long notReadySinceMillis; // wall-clock time of the FIRST non-ready sample in the current notReadySincePoll streak (fleetd #501); reset alongside it
|
||||||
boolean postTurnPending; // completion observed; adapter housekeeping has not started yet
|
boolean postTurnPending; // completion observed; adapter housekeeping has not started yet
|
||||||
boolean awaitingPostTurnPickup;
|
boolean awaitingPostTurnPickup;
|
||||||
boolean postTurnObserved;
|
boolean postTurnObserved;
|
||||||
@@ -262,7 +382,14 @@ public final class Injector {
|
|||||||
}
|
}
|
||||||
|
|
||||||
private static Cancellation cancellationOf(Pending p) {
|
private static Cancellation cancellationOf(Pending p) {
|
||||||
return p.state == Pending.State.DELIVERED ? Cancellation.DELIVERED : Cancellation.NOT_DELIVERED;
|
// fleetd #551 (comment 17037): a two-way split on a now-three-way question folded ATTEMPTED
|
||||||
|
// into NOT_DELIVERED with no compiler error and no failing test — the exact defect this
|
||||||
|
// ticket exists to fix, one layer up. ATTEMPTED gets its own answer instead.
|
||||||
|
return switch (p.state) {
|
||||||
|
case DELIVERED -> Cancellation.DELIVERED;
|
||||||
|
case ATTEMPTED -> Cancellation.ATTEMPTED;
|
||||||
|
case QUEUED, NOT_DELIVERED, CANCELLED -> Cancellation.NOT_DELIVERED;
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
private static boolean isQuiescent(Target t) {
|
private static boolean isQuiescent(Target t) {
|
||||||
@@ -282,7 +409,7 @@ public final class Injector {
|
|||||||
if (t == null) return;
|
if (t == null) return;
|
||||||
|
|
||||||
Pending sent = null;
|
Pending sent = null;
|
||||||
RuntimeException sendError = null;
|
Throwable sendError = null;
|
||||||
boolean turnCompleted = false;
|
boolean turnCompleted = false;
|
||||||
boolean turnFailed = false;
|
boolean turnFailed = false;
|
||||||
boolean resubmit = false;
|
boolean resubmit = false;
|
||||||
@@ -302,6 +429,7 @@ public final class Injector {
|
|||||||
t.unknownSinceTurn = 0;
|
t.unknownSinceTurn = 0;
|
||||||
t.unknownSincePostTurn = 0;
|
t.unknownSincePostTurn = 0;
|
||||||
t.notReadySincePoll = 0;
|
t.notReadySincePoll = 0;
|
||||||
|
t.notReadySinceMillis = 0;
|
||||||
if (t.awaitingCompletion) t.turnObserved = true;
|
if (t.awaitingCompletion) t.turnObserved = true;
|
||||||
} else if (status.injectable()) { // IDLE or BLOCKED
|
} else if (status.injectable()) { // IDLE or BLOCKED
|
||||||
t.unknownSinceTurn = 0;
|
t.unknownSinceTurn = 0;
|
||||||
@@ -353,40 +481,93 @@ public final class Injector {
|
|||||||
Pending p = t.queue.peek();
|
Pending p = t.queue.peek();
|
||||||
if (p != null && ready.test(target)) {
|
if (p != null && ready.test(target)) {
|
||||||
t.notReadySincePoll = 0;
|
t.notReadySincePoll = 0;
|
||||||
|
t.notReadySinceMillis = 0;
|
||||||
|
// fleetd #551: poll and record BEFORE the irreversible send, not after.
|
||||||
|
// The entry comes off the queue and its state is set to ATTEMPTED here,
|
||||||
|
// unconditionally — so a Throwable escaping the send call below (caught or
|
||||||
|
// not) can never leave the entry QUEUED at the head of t.queue (the fleetd
|
||||||
|
// #546 hazard, since peek() alone would let the next onStatus round re-enter
|
||||||
|
// this block and send the same text again), and no path can write a
|
||||||
|
// confident DELIVERED or NOT_DELIVERED before we actually know which one
|
||||||
|
// happened.
|
||||||
|
t.queue.poll();
|
||||||
|
p.state = Pending.State.ATTEMPTED;
|
||||||
try {
|
try {
|
||||||
agentsFor(target).send(target, p.text());
|
agentsFor(target).send(target, p.text());
|
||||||
t.queue.poll();
|
|
||||||
p.state = Pending.State.DELIVERED;
|
p.state = Pending.State.DELIVERED;
|
||||||
t.awaitingPickup = true;
|
t.awaitingPickup = true;
|
||||||
t.awaitingCompletion = true;
|
t.awaitingCompletion = true;
|
||||||
t.turnObserved = false;
|
t.turnObserved = false;
|
||||||
t.injectableSincePickup = 0;
|
t.injectableSincePickup = 0;
|
||||||
sent = p;
|
sent = p;
|
||||||
} catch (RuntimeException e) {
|
} catch (Throwable e) {
|
||||||
// Delivery failed at herdr; drop the poisoned message and surface it
|
// fleetd #551: leave p.state == ATTEMPTED (recorded above, before the
|
||||||
// rather than blocking the queue behind it.
|
// call) rather than downgrading it to NOT_DELIVERED here — reaching this
|
||||||
t.queue.poll();
|
// catch does not prove the text never reached the pane. Three of the
|
||||||
p.state = Pending.State.NOT_DELIVERED;
|
// four HerdrException throw sites in HerdrCodec fire only after herdr
|
||||||
|
// has already replied (so it processed the request), and the fourth (a
|
||||||
|
// transport IOException) leaves it genuinely unknown whether herdr even
|
||||||
|
// received the bytes — see #551 comment 16867. The one exception is a
|
||||||
|
// herdr `*_not_found` error: that family is already read as "definitely
|
||||||
|
// absent, not merely inconclusive" everywhere else in this codebase
|
||||||
|
// (StatusPoller, AgentControl's own retry, WorkspaceControl,
|
||||||
|
// HerdrPeerLauncher, FleetApp, ReplyPushLoop) because it means the
|
||||||
|
// target pane/agent does not exist at all, so nothing could have been
|
||||||
|
// pasted anywhere — #551 keeps the new state consistent with that
|
||||||
|
// existing vocabulary rather than inventing a second one.
|
||||||
|
if (e instanceof HerdrException he && he.code() != null
|
||||||
|
&& he.code().endsWith("_not_found")) {
|
||||||
|
p.state = Pending.State.NOT_DELIVERED;
|
||||||
|
}
|
||||||
sent = p;
|
sent = p;
|
||||||
sendError = e;
|
sendError = e;
|
||||||
}
|
}
|
||||||
} else if (p != null && ++t.notReadySincePoll >= READINESS_GRACE_POLLS) {
|
} else if (p != null) {
|
||||||
// The worker has been idle-but-not-ready for the whole grace: its Claude
|
// fleetd #501: stamp the wall-clock time of the FIRST non-ready sample in
|
||||||
// never connected the bridge MCP (crashed during boot, or wedged on a
|
// this streak, so the expiry log below can print how long the target
|
||||||
// startup prompt). The readiness gate would hold this message forever, so
|
// actually sat non-ready — not just how many polls that took.
|
||||||
// fail every queued message and release the target (CB-114) instead of
|
if (t.notReadySincePoll == 0) {
|
||||||
// polling it indefinitely with the caller's future never completing.
|
t.notReadySinceMillis = nowMillis.getAsLong();
|
||||||
notReady = new ArrayList<>(t.queue);
|
}
|
||||||
for (Pending pending : notReady) {
|
if (++t.notReadySincePoll >= READINESS_GRACE_POLLS) {
|
||||||
pending.state = Pending.State.NOT_DELIVERED;
|
// The worker has been idle-but-not-ready for the whole grace: its Claude
|
||||||
|
// never connected the bridge MCP (crashed during boot, or wedged on a
|
||||||
|
// startup prompt). The readiness gate would hold this message forever, so
|
||||||
|
// fail every queued message and release the target (CB-114) instead of
|
||||||
|
// polling it indefinitely with the caller's future never completing.
|
||||||
|
notReady = new ArrayList<>(t.queue);
|
||||||
|
for (Pending pending : notReady) {
|
||||||
|
pending.state = Pending.State.NOT_DELIVERED;
|
||||||
|
}
|
||||||
|
// fleetd #501: t.notReadySincePoll — the loop's own counter, already in
|
||||||
|
// scope — is printed here instead of the READINESS_GRACE_POLLS constant.
|
||||||
|
// On this branch the counter has JUST reached the threshold, so the two
|
||||||
|
// agree by construction and no test can tell them apart. Printed anyway:
|
||||||
|
// it gives this line one source of truth instead of two, so a later
|
||||||
|
// change to the loop above cannot leave this message reporting a number
|
||||||
|
// the loop no longer produces.
|
||||||
|
//
|
||||||
|
// elapsedMillis is a different case: it is NOT equal-by-construction to
|
||||||
|
// the truth. notReadySincePoll only increments on a sample that reaches
|
||||||
|
// this branch (p != null, not ready) — a poll that misses that condition
|
||||||
|
// advances real time without advancing the counter — and this loop's real
|
||||||
|
// period is not guaranteed to equal POLL_INTERVAL_MILLIS (load, or a host
|
||||||
|
// sleep, can widen the real gap far past it). READINESS_GRACE_POLLS *
|
||||||
|
// POLL_INTERVAL_MILLIS / 1000 is arithmetic on two constants, not a
|
||||||
|
// measurement, so it stays here only as the labelled CONFIGURED budget,
|
||||||
|
// never presented as elapsed time.
|
||||||
|
long elapsedMillis = nowMillis.getAsLong() - t.notReadySinceMillis;
|
||||||
|
log.warn("readiness grace for {} expired after {} polls (configured={} "
|
||||||
|
+ "polls/{}s elapsed={}ms): target never became "
|
||||||
|
+ "deliverable, so failing {} queued message(s) that "
|
||||||
|
+ "never reached its pane",
|
||||||
|
target, t.notReadySincePoll, READINESS_GRACE_POLLS,
|
||||||
|
READINESS_GRACE_POLLS * POLL_INTERVAL_MILLIS / 1000, elapsedMillis,
|
||||||
|
notReady.size());
|
||||||
|
t.queue.clear();
|
||||||
|
t.notReadySincePoll = 0;
|
||||||
|
t.notReadySinceMillis = 0;
|
||||||
}
|
}
|
||||||
log.warn("readiness grace for {} expired after {} polls ({}s): target never "
|
|
||||||
+ "became deliverable, so failing {} queued message(s) that never "
|
|
||||||
+ "reached its pane",
|
|
||||||
target, READINESS_GRACE_POLLS,
|
|
||||||
READINESS_GRACE_POLLS * POLL_INTERVAL_MILLIS / 1000, notReady.size());
|
|
||||||
t.queue.clear();
|
|
||||||
t.notReadySincePoll = 0;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -434,54 +615,182 @@ public final class Injector {
|
|||||||
|
|
||||||
// Fire listeners / herdr calls after releasing the monitor so nothing runs on the poller
|
// Fire listeners / herdr calls after releasing the monitor so nothing runs on the poller
|
||||||
// thread while it holds the target lock.
|
// thread while it holds the target lock.
|
||||||
if (resubmit) {
|
//
|
||||||
try {
|
// fleetd #553: wrapped in try/finally. By this point, if `sent != null`, the delivery has
|
||||||
agentsFor(target).submit(target); // nudge a raced Enter so the pending paste submits
|
// already happened inside the monitor above — off the queue, p.state == DELIVERED, text
|
||||||
} catch (RuntimeException e) {
|
// typed into the target's pane — so `sent`'s future MUST be completed one way or another,
|
||||||
log.debug("resubmit to {} failed (will retry next poll): {}", target, e.getMessage());
|
// in every path out of this region, or the caller (a blocking fleet_send, or an async
|
||||||
|
// ticket) waits forever on a message it actually received. But the finally must not
|
||||||
|
// swallow whatever escaped: a listener that throws is a defect in THAT listener, and it
|
||||||
|
// must still reach StatusPoller's catch (Throwable) so log.error fires — converting a
|
||||||
|
// loud listener bug into a silently orphaned future would be worse than the bug itself.
|
||||||
|
//
|
||||||
|
// There are TWO futures at stake here, not one: `sent.delivered()` (the delivery future)
|
||||||
|
// and `sent.token().waiter()` (the rendezvous waiter a blocking fleet_send actually waits
|
||||||
|
// on for the worker's ANSWER). fleetd #556: the waiter is registered by `registrar.register`
|
||||||
|
// now — called directly and unconditionally, structural rather than delegated to a listener
|
||||||
|
// callback — never by turnListener.onDelivered() alone (before this ticket, that WAS the
|
||||||
|
// only path, and any TurnListener wired ahead of the one that mattered could throw and skip
|
||||||
|
// it; #553 could only make the reachable listener behave, not remove that dependency).
|
||||||
|
// Completing `sent.delivered()` without also calling `registrar.register` leaves the waiter
|
||||||
|
// unregistered forever — a hang with a success receipt, which is worse than the plain hang
|
||||||
|
// #553 was about. So the `finally` below does not just complete the future: on the path
|
||||||
|
// where the `if (sent != null)` block never ran, it does that block's whole job —
|
||||||
|
// register, then onDelivered (notification only, may throw), then complete.
|
||||||
|
//
|
||||||
|
// `sentHandled` is NOT allowed to move earlier than this: an earlier comment on #553
|
||||||
|
// proposed running `if (sent != null)` first, before turnCompleted/turnFailed, and that
|
||||||
|
// was withdrawn — `registrar.register` WRITES CompletionResolver's inFlight record for this
|
||||||
|
// turn, onTurnComplete READS it for the PREVIOUS turn, and running register first makes
|
||||||
|
// onTurnComplete resolve the NEW turn's waiter with the PREVIOUS turn's stale output
|
||||||
|
// (CB-116). The `if (sent != null)` block stays last; the `finally` is a backstop for it,
|
||||||
|
// not a replacement. fleetd #556 keeps this ordering unchanged — it only splits what used
|
||||||
|
// to be one listener call (register + notify) into two calls in the same position.
|
||||||
|
boolean sentHandled = false;
|
||||||
|
try {
|
||||||
|
if (resubmit) {
|
||||||
|
try {
|
||||||
|
agentsFor(target).submit(target); // nudge a raced Enter so the pending paste submits
|
||||||
|
} catch (Throwable e) {
|
||||||
|
// fleetd #553: widened from RuntimeException (same reasoning as #549 at :382) —
|
||||||
|
// this is the FIRST block after the monitor, so an Error escaping it used to skip
|
||||||
|
// every block below it, including the `sent` completion.
|
||||||
|
log.debug("resubmit to {} failed (will retry next poll): {}", target, e.getMessage());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
if (notReady != null) {
|
||||||
if (notReady != null) {
|
// Worker never became available: unblock every queued caller FIRST — these messages'
|
||||||
// Worker never became available: forget its (never-set) readiness, unblock every queued
|
// fate (NOT_DELIVERED, off the queue) was already decided inside the monitor above, so
|
||||||
// caller, and route the awaiting send through the same failure path as a stalled turn so
|
// fleetd #553 completes every one of them before calling forget.accept or
|
||||||
// a blocking or async waiter resolves WORKER_FAILED rather than riding out the timeout.
|
// turnListener.onTurnFailed below. Either of those is a listener/consumer callback and
|
||||||
forget.accept(target);
|
// can throw (an ordinary RuntimeException is enough — the same reasoning as the rest of
|
||||||
RuntimeException cause = new IllegalStateException(
|
// this ticket): completing the futures first means such a throw can no longer leave any
|
||||||
target + " never became available (no bridge MCP connection within the boot window)");
|
// of them permanently pending, regardless of which one throws or in which order.
|
||||||
for (Pending p : notReady) {
|
RuntimeException cause = new IllegalStateException(
|
||||||
p.delivered().completeExceptionally(cause);
|
target + " never became available (no bridge MCP connection within the boot window)");
|
||||||
|
for (Pending p : notReady) {
|
||||||
|
p.delivered().completeExceptionally(cause);
|
||||||
|
}
|
||||||
|
// route the awaiting send through the same failure path as a stalled turn so a blocking
|
||||||
|
// or async waiter resolves WORKER_FAILED rather than riding out the timeout.
|
||||||
|
forget.accept(target);
|
||||||
|
turnListener.onTurnFailed(target);
|
||||||
}
|
}
|
||||||
turnListener.onTurnFailed(target);
|
if (turnCompleted) {
|
||||||
}
|
if (startPostTurn) {
|
||||||
if (turnCompleted) {
|
// fleetd #553: the listener call is wrapped so `t.postTurnPending` (set true inside
|
||||||
if (startPostTurn) {
|
// the monitor above, before this call) is always reset. Before this wrapping, a
|
||||||
boolean started = turnListener.onTurnCompleteWithPostAction(target);
|
// RuntimeException from onTurnCompleteWithPostAction skipped the reset below,
|
||||||
synchronized (t) {
|
// permanently wedging the target: postTurnPending stayed true forever, so this
|
||||||
t.postTurnPending = false;
|
// target's delivery guard (:376) would never again pass and no further message to it
|
||||||
if (started) {
|
// would ever be delivered — a target-level lockup, not just one skipped future.
|
||||||
t.awaitingPostTurnPickup = true;
|
// `started` defaults to false so an exception is treated as "the post-turn action did
|
||||||
t.injectableSincePostTurnPickup = 0;
|
// not start" rather than falsely arming the post-turn pickup latch for an action that
|
||||||
|
// never ran.
|
||||||
|
boolean started = false;
|
||||||
|
try {
|
||||||
|
started = turnListener.onTurnCompleteWithPostAction(target);
|
||||||
|
} finally {
|
||||||
|
synchronized (t) {
|
||||||
|
t.postTurnPending = false;
|
||||||
|
if (started) {
|
||||||
|
t.awaitingPostTurnPickup = true;
|
||||||
|
t.injectableSincePostTurnPickup = 0;
|
||||||
|
}
|
||||||
|
if (t.queue.isEmpty() && !t.awaitingPostTurnPickup) {
|
||||||
|
targets.remove(target, t);
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if (t.queue.isEmpty() && !t.awaitingPostTurnPickup) {
|
} else {
|
||||||
targets.remove(target, t);
|
turnListener.onTurnComplete(target);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (turnFailed) {
|
||||||
|
turnListener.onTurnFailed(target);
|
||||||
|
}
|
||||||
|
if (sent != null) {
|
||||||
|
// fleetd #553: record the INTENT to handle before any side effect, not the fact of
|
||||||
|
// having completed it. If register()/onDelivered() throws part way through — e.g.
|
||||||
|
// after CompletionResolver's captureBaseline has already done its inFlight.put — a
|
||||||
|
// flag set only after this block would still read false, and the `finally` below
|
||||||
|
// would re-run this block's job a SECOND time. A second captureBaseline runs later,
|
||||||
|
// once onTurnComplete has already thrown and possibly scraped the pane, and can
|
||||||
|
// snapshot a pane that already absorbed this turn's output — which means the pane's
|
||||||
|
// tail never differs from that baseline again and CompletionResolver.resolve's own
|
||||||
|
// suppression at :364-369 (`return` keeping the in-flight record) drops every future
|
||||||
|
// completion for this turn, permanently. So this flag must go true FIRST.
|
||||||
|
sentHandled = true;
|
||||||
|
if (sendError != null) {
|
||||||
|
log.warn("inject to {} failed, dropped message: {}", target, sendError.getMessage());
|
||||||
|
sent.delivered().completeExceptionally(sendError);
|
||||||
|
} else {
|
||||||
|
// fleetd #556: register the waiter FIRST and unconditionally — this is the
|
||||||
|
// Injector's own invariant, kept structural rather than delegated. Even if
|
||||||
|
// turnListener.onDelivered below throws (a bug in an unrelated observer, e.g.
|
||||||
|
// SessionManager's bookkeeping, or a fan-out wired in a different order), the
|
||||||
|
// waiter is already registered and resolvable — a throwing TurnListener can no
|
||||||
|
// longer take this invariant down with it.
|
||||||
|
registrar.register(target, sent.token());
|
||||||
|
// Notification only, from here down: baseline the pane's pre-turn content so a
|
||||||
|
// misattributed completion (no new output) can't resolve this send with the
|
||||||
|
// previous turn's stale answer (CB-115). Allowed to throw and to fail (the herdr
|
||||||
|
// scrape inside already fails open with baseline == null on its own).
|
||||||
|
turnListener.onDelivered(target, sent.token());
|
||||||
|
sent.delivered().complete(null);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} finally {
|
||||||
|
// Backstop (fleetd #553, extended #556): if an earlier block in this try threw before the
|
||||||
|
// `if (sent != null)` block above ran, `sentHandled` is still false here, and this does
|
||||||
|
// that block's WHOLE job — not just the future completion. Skipping registrar.register()
|
||||||
|
// here would leave sent.token().waiter() never registered with CompletionResolver, so a
|
||||||
|
// blocking fleet_send would be told its message was delivered and then wait out its full
|
||||||
|
// timeout for an answer that can never resolve — worse than the plain hang, because it now
|
||||||
|
// looks like success. This block runs only when sendError == null: nothing was delivered
|
||||||
|
// on the error path, so there is nothing to register.
|
||||||
|
//
|
||||||
|
// fleetd #553 (lead review, comment 16916): `sentHandled` guards ONLY the register/notify
|
||||||
|
// re-call below, never the future completion outside this inner try. One boolean cannot
|
||||||
|
// carry both meanings — "the turn was handled" and "the future has been dealt with" —
|
||||||
|
// because they come apart exactly when onDelivered throws PART WAY THROUGH: sentHandled
|
||||||
|
// is already true (set before the call, correctly — see the `if (sent != null)` block
|
||||||
|
// above), so a guard on the outer `if` here would skip this whole recovery, including the
|
||||||
|
// completion, and leave sent.delivered() pending forever even though the message really
|
||||||
|
// was typed into the pane and taken off the queue. So `sent != null` alone gates whether
|
||||||
|
// this target has anything to finish; `!sentHandled` gates only the register/notify
|
||||||
|
// re-call inside. CompletableFuture.complete/completeExceptionally are idempotent — on the
|
||||||
|
// ordinary path (sentHandled == true, no throw) the `if (sent != null)` block above has
|
||||||
|
// already completed this future, so the calls below are a no-op returning false.
|
||||||
|
//
|
||||||
|
// This recovery is wrapped in its own try/catch(Throwable) that swallows only ITS OWN
|
||||||
|
// throwable and logs at WARN — the future is still completed either way — while the
|
||||||
|
// ORIGINAL throwable from the try block above is left alone to keep unwinding out of
|
||||||
|
// this method to StatusPoller's catch (Throwable), so a listener bug stays loud.
|
||||||
|
if (sent != null) {
|
||||||
|
try {
|
||||||
|
if (!sentHandled && sendError == null) {
|
||||||
|
// fleetd #556: register first here too, same as the ordinary path above — this
|
||||||
|
// recovery path is reached only when an EARLIER block (resubmit/turnCompleted/
|
||||||
|
// turnFailed) threw before the ordinary path ever ran, so nothing has
|
||||||
|
// registered this turn yet.
|
||||||
|
registrar.register(target, sent.token());
|
||||||
|
turnListener.onDelivered(target, sent.token());
|
||||||
|
}
|
||||||
|
} catch (Throwable recoveryError) {
|
||||||
|
log.warn("fleetd #553 backstop: onDelivered failed for {} while recovering from "
|
||||||
|
+ "an earlier listener failure; completing its delivery future "
|
||||||
|
+ "anyway: {}",
|
||||||
|
target, recoveryError.getMessage());
|
||||||
|
} finally {
|
||||||
|
// No-op (returns false) on the ordinary path, where the `if (sent != null)` block
|
||||||
|
// above already completed this future — see the comment above this block.
|
||||||
|
if (sendError != null) {
|
||||||
|
sent.delivered().completeExceptionally(sendError);
|
||||||
|
} else {
|
||||||
|
sent.delivered().complete(null);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
} else {
|
|
||||||
turnListener.onTurnComplete(target);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (turnFailed) {
|
|
||||||
turnListener.onTurnFailed(target);
|
|
||||||
}
|
|
||||||
if (sent != null) {
|
|
||||||
if (sendError != null) {
|
|
||||||
log.warn("inject to {} failed, dropped message: {}", target, sendError.getMessage());
|
|
||||||
sent.delivered().completeExceptionally(sendError);
|
|
||||||
} else {
|
|
||||||
// Baseline the pane's pre-turn content so a misattributed completion (no new output)
|
|
||||||
// can't resolve this send with the previous turn's stale answer (CB-115).
|
|
||||||
turnListener.onDelivered(target, sent.token());
|
|
||||||
sent.delivered().complete(null);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,91 @@
|
|||||||
|
package dev.ltms.fleet.inject;
|
||||||
|
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A progress watchdog for a singleton background loop (fleetd #544): {@link StatusPoller} and
|
||||||
|
* {@code dev.ltms.fleet.session.SessionReaper} each hold one. It tracks the monotonic timestamp
|
||||||
|
* of the loop's last <em>completed</em> round and classifies health from that timestamp alone —
|
||||||
|
* never from thread liveness. That choice is deliberate: measured on {@code main} at
|
||||||
|
* {@code 7611b69}, {@code UnixSocketHerdrClient.call()} reads with no deadline anywhere in
|
||||||
|
* {@code herdr/}, so a herdrd that accepts a connection and never answers parks the calling
|
||||||
|
* loop's thread forever — the thread stays alive and {@code Thread.isAlive()} keeps reporting
|
||||||
|
* {@code true} the whole time. A last-round timestamp that stops advancing catches that parked
|
||||||
|
* case exactly the same way it catches a thread that died outright: either way, nothing marks a
|
||||||
|
* new round complete.
|
||||||
|
*
|
||||||
|
* <p>Lives in {@code inject} (rather than {@code health}, which would be the more obvious home
|
||||||
|
* for a health-reporting fact) because {@code health} already depends on {@code session}
|
||||||
|
* ({@code HealthSnapshot}/{@code FleetHealthMonitor} read {@code MemberSession}), and
|
||||||
|
* {@code session} already depends on {@code inject} ({@code SessionManager} implements
|
||||||
|
* {@code TurnListener} and extends {@code MemberPresence}). Putting this class in {@code health}
|
||||||
|
* would have made {@code SessionReaper} (in {@code session}) depend back on {@code health},
|
||||||
|
* closing a cycle through {@code health -> session -> health} — {@code PackageCyclesTest} catches
|
||||||
|
* exactly this. {@code inject} has no dependency on {@code health}, so {@code session} depending
|
||||||
|
* on {@code inject} for this stays a one-way edge, same direction it already depends in.
|
||||||
|
*
|
||||||
|
* <p>Three states, not two (the fleetd #512 shape — one flag carrying two conditions that need
|
||||||
|
* opposite handling). {@code stop()} also makes the loop go quiet, exactly like a crash or a
|
||||||
|
* hang does, so a caller must say which one happened: {@link #markStoppedByCaller()} records
|
||||||
|
* "this halt was on purpose," and {@link #state()} reports {@link State#STOPPED} for it
|
||||||
|
* regardless of how stale the last round looks — an intentional stop is never an alarm.
|
||||||
|
*/
|
||||||
|
public final class LoopWatchdog {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* {@code RUNNING} — the loop is going and its last round finished recently.
|
||||||
|
* {@code STALLED} — the loop was never stopped on purpose, and its last-completed-round
|
||||||
|
* timestamp is older than the threshold. Covers both a dead loop and a parked one.
|
||||||
|
* {@code STOPPED} — {@link #markStoppedByCaller()} was called. Not an alarm.
|
||||||
|
*/
|
||||||
|
public enum State { RUNNING, STALLED, STOPPED }
|
||||||
|
|
||||||
|
private final LongSupplier nowNanos;
|
||||||
|
private final long staleAfterNanos;
|
||||||
|
private volatile long lastRoundNanos;
|
||||||
|
private volatile boolean stoppedByCaller;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @param nowNanos a monotonic elapsed-time clock, e.g. {@code System::nanoTime} live, a
|
||||||
|
* controllable stub in tests.
|
||||||
|
* @param staleAfterNanos how long a last-completed-round timestamp may age before {@link #state()}
|
||||||
|
* reports {@link State#STALLED}. Derive this from the loop's own poll
|
||||||
|
* interval with generous headroom — see the caller's own derivation comment.
|
||||||
|
*/
|
||||||
|
public LoopWatchdog(LongSupplier nowNanos, long staleAfterNanos) {
|
||||||
|
this.nowNanos = nowNanos;
|
||||||
|
this.staleAfterNanos = staleAfterNanos;
|
||||||
|
this.lastRoundNanos = nowNanos.getAsLong();
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Call once per completed round, from inside the loop. */
|
||||||
|
public void recordRoundComplete() {
|
||||||
|
lastRoundNanos = nowNanos.getAsLong();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Call from {@code start()} (and any future restart): clears a previous stop's mark and
|
||||||
|
* resets the clock, so the loop gets a fresh grace period before its first round completes
|
||||||
|
* rather than being judged against a timestamp left over from before it was (re)started.
|
||||||
|
*/
|
||||||
|
public void reset() {
|
||||||
|
stoppedByCaller = false;
|
||||||
|
lastRoundNanos = nowNanos.getAsLong();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Call from {@code stop()}: marks the halt as intentional, so {@link #state()} reports
|
||||||
|
* {@link State#STOPPED} rather than {@link State#STALLED} no matter how stale the last round is.
|
||||||
|
*/
|
||||||
|
public void markStoppedByCaller() {
|
||||||
|
stoppedByCaller = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The current fact about this loop, per the three-state contract above. */
|
||||||
|
public State state() {
|
||||||
|
if (stoppedByCaller) {
|
||||||
|
return State.STOPPED;
|
||||||
|
}
|
||||||
|
return (nowNanos.getAsLong() - lastRoundNanos >= staleAfterNanos) ? State.STALLED : State.RUNNING;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -8,6 +8,8 @@ import org.slf4j.Logger;
|
|||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import java.util.Set;
|
import java.util.Set;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Drives the {@link Injector} by sampling each active worker's {@code agent_status} and
|
* Drives the {@link Injector} by sampling each active worker's {@code agent_status} and
|
||||||
@@ -22,11 +24,25 @@ public final class StatusPoller {
|
|||||||
|
|
||||||
private static final Logger log = LoggerFactory.getLogger(StatusPoller.class);
|
private static final Logger log = LoggerFactory.getLogger(StatusPoller.class);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* How many multiples of {@code intervalMillis} the last-completed round may age before
|
||||||
|
* {@link #health()} reports {@link LoopWatchdog.State#STALLED} (fleetd #544). A round this
|
||||||
|
* loop runs is normally a small fraction of one interval — it only takes longer when a herdr
|
||||||
|
* call is genuinely wedged (herdr's {@code UnixSocketHerdrClient} has no read deadline, so a
|
||||||
|
* herdrd that accepts and never answers parks the thread forever) — so a wide multiplier
|
||||||
|
* avoids false alarms from an ordinarily slow round. At the production interval
|
||||||
|
* ({@link Injector#POLL_INTERVAL_MILLIS} = 250ms) this puts the threshold at 10s: long enough
|
||||||
|
* that a transient slow poll never trips it, short enough that a stuck poller is visible well
|
||||||
|
* before anything else downstream (delivery, `/healthz` consumers) would show a symptom.
|
||||||
|
*/
|
||||||
|
private static final long STALE_AFTER_INTERVAL_MULTIPLIER = 40;
|
||||||
|
|
||||||
private final AgentControl agents;
|
private final AgentControl agents;
|
||||||
private final HerdrRouter router;
|
private final HerdrRouter router;
|
||||||
private final Injector injector;
|
private final Injector injector;
|
||||||
private final StatusRefiner refiner;
|
private final StatusRefiner refiner;
|
||||||
private final long intervalMillis;
|
private final long intervalMillis;
|
||||||
|
private final LoopWatchdog watchdog;
|
||||||
private volatile boolean running;
|
private volatile boolean running;
|
||||||
private Thread thread;
|
private Thread thread;
|
||||||
|
|
||||||
@@ -36,14 +52,26 @@ public final class StatusPoller {
|
|||||||
|
|
||||||
public StatusPoller(AgentControl agents, Injector injector, StatusRefiner refiner,
|
public StatusPoller(AgentControl agents, Injector injector, StatusRefiner refiner,
|
||||||
long intervalMillis) {
|
long intervalMillis) {
|
||||||
|
this(agents, injector, refiner, intervalMillis, System::nanoTime);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Full constructor — for tests: an injectable monotonic clock (fleetd #544). */
|
||||||
|
StatusPoller(AgentControl agents, Injector injector, StatusRefiner refiner,
|
||||||
|
long intervalMillis, LongSupplier nowNanos) {
|
||||||
this.agents = agents;
|
this.agents = agents;
|
||||||
this.router = null;
|
this.router = null;
|
||||||
this.injector = injector;
|
this.injector = injector;
|
||||||
this.refiner = refiner;
|
this.refiner = refiner;
|
||||||
this.intervalMillis = intervalMillis;
|
this.intervalMillis = intervalMillis;
|
||||||
|
this.watchdog = new LoopWatchdog(nowNanos, staleAfterNanos(intervalMillis));
|
||||||
}
|
}
|
||||||
|
|
||||||
public StatusPoller(HerdrRouter router, Injector injector, long intervalMillis) {
|
public StatusPoller(HerdrRouter router, Injector injector, long intervalMillis) {
|
||||||
|
this(router, injector, intervalMillis, System::nanoTime);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Full constructor — for tests: an injectable monotonic clock (fleetd #544). */
|
||||||
|
StatusPoller(HerdrRouter router, Injector injector, long intervalMillis, LongSupplier nowNanos) {
|
||||||
this.agents = null;
|
this.agents = null;
|
||||||
this.router = router;
|
this.router = router;
|
||||||
this.injector = injector;
|
this.injector = injector;
|
||||||
@@ -53,45 +81,70 @@ public final class StatusPoller {
|
|||||||
// against the LEAD daemon even though this field points at the member one.
|
// against the LEAD daemon even though this field points at the member one.
|
||||||
this.refiner = new StatusRefiner(router.memberAgents());
|
this.refiner = new StatusRefiner(router.memberAgents());
|
||||||
this.intervalMillis = intervalMillis;
|
this.intervalMillis = intervalMillis;
|
||||||
|
this.watchdog = new LoopWatchdog(nowNanos, staleAfterNanos(intervalMillis));
|
||||||
|
}
|
||||||
|
|
||||||
|
private static long staleAfterNanos(long intervalMillis) {
|
||||||
|
return TimeUnit.MILLISECONDS.toNanos(Math.max(intervalMillis, 1)) * STALE_AFTER_INTERVAL_MULTIPLIER;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* This loop's progress fact (fleetd #544): {@code RUNNING}, {@code STALLED} (dead or parked —
|
||||||
|
* indistinguishable from outside, and never observed on purpose), or {@code STOPPED}
|
||||||
|
* ({@link #stop()} was called). See {@link LoopWatchdog} for why staleness — not thread
|
||||||
|
* liveness — is the signal.
|
||||||
|
*/
|
||||||
|
public LoopWatchdog.State health() {
|
||||||
|
return watchdog.state();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Start the polling loop on a virtual thread. Idempotent. */
|
/** Start the polling loop on a virtual thread. Idempotent. */
|
||||||
public synchronized void start() {
|
public synchronized void start() {
|
||||||
if (running) return;
|
if (running) return;
|
||||||
running = true;
|
running = true;
|
||||||
|
watchdog.reset(); // fleetd #544: a fresh grace period, not last run's stale mark/timestamp.
|
||||||
thread = Thread.ofVirtual().name("status-poller").start(this::loop);
|
thread = Thread.ofVirtual().name("status-poller").start(this::loop);
|
||||||
log.info("status poller started (interval {}ms)", intervalMillis);
|
log.info("status poller started (interval {}ms)", intervalMillis);
|
||||||
}
|
}
|
||||||
|
|
||||||
private void loop() {
|
private void loop() {
|
||||||
while (running) {
|
try {
|
||||||
Set<String> active = injector.activeTargets();
|
while (running) {
|
||||||
for (String target : active) {
|
Set<String> active = injector.activeTargets();
|
||||||
if (!running) return;
|
for (String target : active) {
|
||||||
try {
|
if (!running) return;
|
||||||
// herdr's agent_status can misreport a settled worker as `unknown`; refine it
|
try {
|
||||||
// against the pane content before it drives delivery/completion (CB-115).
|
// herdr's agent_status can misreport a settled worker as `unknown`; refine it
|
||||||
// CB-185: refine THROUGH the same control the raw status came from — a router
|
// against the pane content before it drives delivery/completion (CB-115).
|
||||||
// splits lead/member targets across two herdr daemons, and reading a lead's pane
|
// CB-185: refine THROUGH the same control the raw status came from — a router
|
||||||
// through the (fixed) member refiner never finds it, wedging that lead at UNKNOWN.
|
// splits lead/member targets across two herdr daemons, and reading a lead's pane
|
||||||
AgentControl control = router != null ? router.agentsFor(target) : agents;
|
// through the (fixed) member refiner never finds it, wedging that lead at UNKNOWN.
|
||||||
AgentStatus status = refiner.refine(target, control.status(target), control);
|
AgentControl control = router != null ? router.agentsFor(target) : agents;
|
||||||
injector.onStatus(target, status);
|
AgentStatus status = refiner.refine(target, control.status(target), control);
|
||||||
} catch (HerdrException e) {
|
injector.onStatus(target, status);
|
||||||
// The worker's agent is gone — stop trying and unblock its waiters.
|
} catch (HerdrException e) {
|
||||||
if (e.code() != null && e.code().endsWith("_not_found")) {
|
// The worker's agent is gone — stop trying and unblock its waiters.
|
||||||
log.debug("target {} gone; dropping its queue", target);
|
if (e.code() != null && e.code().endsWith("_not_found")) {
|
||||||
injector.drop(target, e);
|
log.debug("target {} gone; dropping its queue", target);
|
||||||
} else {
|
injector.drop(target, e);
|
||||||
log.debug("status poll for {} failed (will retry): {}", target, e.getMessage());
|
} else {
|
||||||
|
log.debug("status poll for {} failed (will retry): {}", target, e.getMessage());
|
||||||
|
}
|
||||||
|
} catch (Throwable e) {
|
||||||
|
log.error("unexpected failure polling {}; skipping this round", target, e);
|
||||||
}
|
}
|
||||||
} catch (RuntimeException e) {
|
|
||||||
// Never let one target's unexpected error (e.g. an odd agent.get shape) kill
|
|
||||||
// the single poller thread and stall injection for every worker.
|
|
||||||
log.warn("unexpected error polling {}; skipping this round", target, e);
|
|
||||||
}
|
}
|
||||||
|
// fleetd #544: a round is "complete" only once every active target has been polled —
|
||||||
|
// a target parked mid-for-loop (control.status(target) never returning) means this
|
||||||
|
// line is never reached, so the watchdog goes stale exactly like a dead loop would.
|
||||||
|
watchdog.recordRoundComplete();
|
||||||
|
sleep();
|
||||||
}
|
}
|
||||||
sleep();
|
} finally {
|
||||||
|
if (running) {
|
||||||
|
log.error("status poller loop exited unexpectedly; it can be restarted");
|
||||||
|
}
|
||||||
|
running = false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -107,6 +160,7 @@ public final class StatusPoller {
|
|||||||
/** Stop the polling loop. Idempotent. */
|
/** Stop the polling loop. Idempotent. */
|
||||||
public synchronized void stop() {
|
public synchronized void stop() {
|
||||||
running = false;
|
running = false;
|
||||||
|
watchdog.markStoppedByCaller(); // fleetd #544: this halt is on purpose — health() must say STOPPED, not STALLED.
|
||||||
if (thread != null) thread.interrupt();
|
if (thread != null) thread.interrupt();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,29 @@
|
|||||||
|
package dev.ltms.fleet.inject;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.msg.TurnToken;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #556: the {@link Injector}'s own invariant — every delivered turn has a registered
|
||||||
|
* waiter — kept structural rather than delegated to a {@link TurnListener} callback that can
|
||||||
|
* throw. The {@link Injector} calls {@link #register} directly and unconditionally as part of
|
||||||
|
* delivering a turn, before it ever calls {@code turnListener.onDelivered}. A {@link TurnListener}
|
||||||
|
* that throws from every callback (a bug in an unrelated observer, e.g. session bookkeeping, or a
|
||||||
|
* fan-out object wired in the wrong order) can therefore never leave a delivered turn unregistered
|
||||||
|
* — the shape #553 could not close, since that fix could only make the reachable listener behave,
|
||||||
|
* never remove the Injector's dependency on a listener behaving at all.
|
||||||
|
*
|
||||||
|
* <p>Deliberately narrow: registration only, no I/O, nothing that scrapes a pane or can reasonably
|
||||||
|
* fail. The CB-115 staleness baseline (a herdr round-trip, allowed to fail) stays a
|
||||||
|
* {@link TurnListener#onDelivered} concern — fired by the {@link Injector} only after this call has
|
||||||
|
* already run, so its own failure cannot un-register anything.
|
||||||
|
*/
|
||||||
|
@FunctionalInterface
|
||||||
|
public interface TurnRegistrar {
|
||||||
|
|
||||||
|
/** Record {@code token}'s waiter as the turn currently in flight for {@code target}. */
|
||||||
|
void register(String target, TurnToken token);
|
||||||
|
|
||||||
|
/** No-op registrar for callers that don't need CB-106 rendezvous tracking. */
|
||||||
|
TurnRegistrar NOOP = (_, _) -> {
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -33,9 +33,10 @@ public final class ConnectionIdentity {
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* The caller resolved from the connection: its worker {@code terminal} (or {@code null} for the
|
* The caller resolved from the connection: its worker {@code terminal} (or {@code null} for the
|
||||||
* primary / an off-host client) and its {@code pid} (or {@code -1} if not resolvable).
|
* primary / an off-host client), its {@code pid} (or {@code -1} if not resolvable), and whether
|
||||||
|
* the pane scan behind {@code terminal} ran to completion ({@link #scanComplete}).
|
||||||
*/
|
*/
|
||||||
public record Caller(String terminal, long pid) {
|
public record Caller(String terminal, long pid, boolean scanComplete) {
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Whether the OS peer-PID lookup actually succeeded — {@code false} means {@code pid} is
|
* Whether the OS peer-PID lookup actually succeeded — {@code false} means {@code pid} is
|
||||||
@@ -51,6 +52,10 @@ public final class ConnectionIdentity {
|
|||||||
* {@link ConnectionIdentity#isLoopback} is centralised rather than left for each caller to
|
* {@link ConnectionIdentity#isLoopback} is centralised rather than left for each caller to
|
||||||
* reimplement: a raw {@code pid > 0} check duplicated at every call site is precisely the
|
* reimplement: a raw {@code pid > 0} check duplicated at every call site is precisely the
|
||||||
* "one rule, two copies" shape that let #305 drift.
|
* "one rule, two copies" shape that let #305 drift.
|
||||||
|
*
|
||||||
|
* <p>This method is deliberately NOT widened for fleetd #505's failure (a herdr error
|
||||||
|
* during the pane scan, not a failed lsof lookup) — it still tests only the sentinel it is
|
||||||
|
* named for. #505 is a different axis, carried separately in {@link #scanComplete}.
|
||||||
*/
|
*/
|
||||||
public boolean resolved() {
|
public boolean resolved() {
|
||||||
return pid > 0;
|
return pid > 0;
|
||||||
@@ -60,10 +65,11 @@ public final class ConnectionIdentity {
|
|||||||
/** Resolve the caller's terminal and PID from one peer-PID lookup. */
|
/** Resolve the caller's terminal and PID from one peer-PID lookup. */
|
||||||
public Caller resolve(String remoteAddr, int remotePort) {
|
public Caller resolve(String remoteAddr, int remotePort) {
|
||||||
if (!isLoopback(remoteAddr)) {
|
if (!isLoopback(remoteAddr)) {
|
||||||
return new Caller(null, -1); // only same-host callers can be workers
|
return new Caller(null, -1, true); // only same-host callers can be workers
|
||||||
}
|
}
|
||||||
long pid = pids.pidForLocalPort(remotePort);
|
long pid = pids.pidForLocalPort(remotePort);
|
||||||
return new Caller(panes.terminalForPid(pid), pid);
|
PaneLocator.Lookup lookup = panes.terminalForPid(pid);
|
||||||
|
return new Caller(lookup.terminal(), pid, lookup.complete());
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ import dev.ltms.fleet.msg.Rendezvous;
|
|||||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||||
|
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||||
import dev.ltms.fleet.placement.PlacementException;
|
import dev.ltms.fleet.placement.PlacementException;
|
||||||
import dev.ltms.fleet.session.SessionManager;
|
import dev.ltms.fleet.session.SessionManager;
|
||||||
import dev.ltms.fleet.session.MemberSession;
|
import dev.ltms.fleet.session.MemberSession;
|
||||||
@@ -93,10 +94,22 @@ public final class FleetMcp {
|
|||||||
|
|
||||||
private final HttpServletStreamableServerTransportProvider transport;
|
private final HttpServletStreamableServerTransportProvider transport;
|
||||||
private final McpSyncServer server;
|
private final McpSyncServer server;
|
||||||
private final CallerResolver authz; // CB-501: null → authorization not enforced (legacy)
|
/**
|
||||||
|
* fleetd #518: whether {@link #denyFor} enforces the CB-505 policy table at all. Replaces the
|
||||||
|
* old {@code CallerResolver authz} field, whose null-ness used to decide BOTH this AND which
|
||||||
|
* principal-resolution code path {@link #contextExtractor} ran — reaching "authorization off"
|
||||||
|
* by simply not passing a {@link CallerResolver} also meant the resolved {@link Principal}
|
||||||
|
* came from a second, separately-maintained heuristic ({@code legacyPrincipal}, now deleted)
|
||||||
|
* that nothing ever exercised. There is now exactly one resolution path ({@code callers},
|
||||||
|
* required and non-null below) and a separate, explicitly-chosen {@link AuthorizationMode}
|
||||||
|
* for this flag — so a caller can turn enforcement off without silently swapping in a second,
|
||||||
|
* untested identity heuristic.
|
||||||
|
*/
|
||||||
|
private final boolean authorizationEnforced;
|
||||||
private final Metrics metrics; // CB-502: null → auth failures not counted
|
private final Metrics metrics; // CB-502: null → auth failures not counted
|
||||||
private final CapacitySource capacity;
|
private final CapacitySource capacity;
|
||||||
private final HealthCoverageSource healthCoverage;
|
private final HealthCoverageSource healthCoverage;
|
||||||
|
private final LoopHealthSource loopHealth;
|
||||||
private final QuarantineSource quarantine;
|
private final QuarantineSource quarantine;
|
||||||
/** fleetd #201 Unit 5: SEPARATE from {@link #quarantine} — see {@link OutageSource}'s doc. */
|
/** fleetd #201 Unit 5: SEPARATE from {@link #quarantine} — see {@link OutageSource}'s doc. */
|
||||||
private final OutageSource outage;
|
private final OutageSource outage;
|
||||||
@@ -125,6 +138,15 @@ public final class FleetMcp {
|
|||||||
/** Coverage is supplied by the health wiring, not inferred from a missing dependency. */
|
/** Coverage is supplied by the health wiring, not inferred from a missing dependency. */
|
||||||
public record HealthCoverageSource(Supplier<String> value) { }
|
public record HealthCoverageSource(Supplier<String> value) { }
|
||||||
|
|
||||||
|
/** Progress states for fleetd's singleton background loops, read by {@code fleet_list} and {@code /healthz}. */
|
||||||
|
public record LoopHealthSource(Supplier<LoopWatchdog.State> statusPoller,
|
||||||
|
Supplier<LoopWatchdog.State> sessionReaper) {
|
||||||
|
/** Inert source for callers that do not wire the background loops. */
|
||||||
|
public static LoopHealthSource none() {
|
||||||
|
return new LoopHealthSource(() -> LoopWatchdog.State.STOPPED, () -> LoopWatchdog.State.STOPPED);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* CB-578 stage B quarantine facts used by {@code fleet_profiles}: a profile → credential id
|
* CB-578 stage B quarantine facts used by {@code fleet_profiles}: a profile → credential id
|
||||||
* lookup, plus the shared {@link BackendQuarantine} to read remaining cooldowns off.
|
* lookup, plus the shared {@link BackendQuarantine} to read remaining cooldowns off.
|
||||||
@@ -250,6 +272,17 @@ public final class FleetMcp {
|
|||||||
public static CoordinationSource none() { return new CoordinationSource(null, List.of()); }
|
public static CoordinationSource none() { return new CoordinationSource(null, List.of()); }
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #518: whether {@link #denyFor} enforces the CB-505 policy table. A required
|
||||||
|
* constructor parameter with no default, so "authorization is off" can only be reached by a
|
||||||
|
* caller explicitly saying so — never by omitting a {@link CallerResolver} the way the old
|
||||||
|
* {@code callers == null} idiom allowed. {@code callers} itself is required either way: even
|
||||||
|
* under {@link #UNENFORCED}, the one real {@link CallerResolver} still resolves every caller's
|
||||||
|
* {@link Principal} (so {@code markSpawnedMemberPresent}/{@code recordPrimarySingleton} see a
|
||||||
|
* real identity), and {@link #denyFor} is the only thing that changes.
|
||||||
|
*/
|
||||||
|
public enum AuthorizationMode { ENFORCED, UNENFORCED }
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The only constructor (fleetd #480 Unit C correction round). Every field below used to have
|
* The only constructor (fleetd #480 Unit C correction round). Every field below used to have
|
||||||
* its own defaulting overload — {@code leadChannel}/{@code outage}/{@code leadSeats}/
|
* its own defaulting overload — {@code leadChannel}/{@code outage}/{@code leadSeats}/
|
||||||
@@ -268,10 +301,16 @@ public final class FleetMcp {
|
|||||||
* {@link OutageSource#none()}, {@link LeadSeatSource#none()}, {@code List.of()} are all still
|
* {@link OutageSource#none()}, {@link LeadSeatSource#none()}, {@code List.of()} are all still
|
||||||
* perfectly fine values, just never an implicit default reached by omission.
|
* perfectly fine values, just never an implicit default reached by omission.
|
||||||
*
|
*
|
||||||
* @param callers resolves each call's {@link Principal}; {@code null} disables
|
* @param callers resolves each call's {@link Principal}. Required, never {@code null} —
|
||||||
* authorization. This surface needs its own enforcement: {@code /mcp} is a
|
* fleetd #518: use {@link AuthorizationMode#UNENFORCED} to disable
|
||||||
* raw servlet on Jetty's context handler and never passes through
|
* enforcement, not a missing resolver. This surface needs its own
|
||||||
* Javalin's {@code before} filter, so the REST guard does not cover it.
|
* enforcement: {@code /mcp} is a raw servlet on Jetty's context handler and
|
||||||
|
* never passes through Javalin's {@code before} filter, so the REST guard
|
||||||
|
* does not cover it.
|
||||||
|
* @param authorizationMode fleetd #518: whether {@link #denyFor} enforces the CB-505 policy
|
||||||
|
* table ({@link AuthorizationMode#ENFORCED}) or leaves the gate open
|
||||||
|
* ({@link AuthorizationMode#UNENFORCED}, for the pre-CB-513 test suite that
|
||||||
|
* does not exercise authorization). Required, with no default.
|
||||||
* @param metrics registry for auth-failure counting; may be {@code null}
|
* @param metrics registry for auth-failure counting; may be {@code null}
|
||||||
* @param quarantine CB-578 stage B facts for {@code fleet_profiles}; pass
|
* @param quarantine CB-578 stage B facts for {@code fleet_profiles}; pass
|
||||||
* {@link QuarantineSource#none()} for a caller that does not want the
|
* {@link QuarantineSource#none()} for a caller that does not want the
|
||||||
@@ -300,10 +339,25 @@ public final class FleetMcp {
|
|||||||
* instead of throwing. See {@link #handover}.
|
* instead of throwing. See {@link #handover}.
|
||||||
*/
|
*/
|
||||||
public FleetMcp(MessageService messages, PeerLauncher workers, SessionManager sessions,
|
public FleetMcp(MessageService messages, PeerLauncher workers, SessionManager sessions,
|
||||||
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||||
CallerResolver callers, Metrics metrics, CapacitySource capacity, HealthCoverageSource healthCoverage,
|
CallerResolver callers, AuthorizationMode authorizationMode, Metrics metrics,
|
||||||
QuarantineSource quarantine, LeadChannel leadChannel, OutageSource outage,
|
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||||
LeadSeatSource leadSeats, List<String> peers, LeadRollover leadRollover) {
|
QuarantineSource quarantine, LeadChannel leadChannel, OutageSource outage,
|
||||||
|
LeadSeatSource leadSeats, List<String> peers, LeadRollover leadRollover) {
|
||||||
|
this(messages, workers, sessions, identity, presence, primaryRegistry, callers, authorizationMode, metrics,
|
||||||
|
capacity, healthCoverage, LoopHealthSource.none(), quarantine, leadChannel, outage, leadSeats,
|
||||||
|
peers, leadRollover);
|
||||||
|
}
|
||||||
|
|
||||||
|
public FleetMcp(MessageService messages, PeerLauncher workers, SessionManager sessions,
|
||||||
|
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||||
|
CallerResolver callers, AuthorizationMode authorizationMode, Metrics metrics,
|
||||||
|
CapacitySource capacity, HealthCoverageSource healthCoverage, LoopHealthSource loopHealth,
|
||||||
|
QuarantineSource quarantine, LeadChannel leadChannel, OutageSource outage,
|
||||||
|
LeadSeatSource leadSeats, List<String> peers, LeadRollover leadRollover) {
|
||||||
|
Objects.requireNonNull(callers, "callers");
|
||||||
|
this.authorizationEnforced = Objects.requireNonNull(authorizationMode, "authorizationMode")
|
||||||
|
== AuthorizationMode.ENFORCED;
|
||||||
this.leadChannel = leadChannel;
|
this.leadChannel = leadChannel;
|
||||||
this.peers = peers == null ? List.of() : List.copyOf(peers);
|
this.peers = peers == null ? List.of() : List.copyOf(peers);
|
||||||
this.capacity = capacity;
|
this.capacity = capacity;
|
||||||
@@ -311,6 +365,7 @@ public final class FleetMcp {
|
|||||||
this.outage = Objects.requireNonNull(outage, "outage");
|
this.outage = Objects.requireNonNull(outage, "outage");
|
||||||
this.leadSeats = Objects.requireNonNull(leadSeats, "leadSeats");
|
this.leadSeats = Objects.requireNonNull(leadSeats, "leadSeats");
|
||||||
this.healthCoverage = healthCoverage;
|
this.healthCoverage = healthCoverage;
|
||||||
|
this.loopHealth = Objects.requireNonNull(loopHealth, "loopHealth");
|
||||||
this.leadRollover = leadRollover;
|
this.leadRollover = leadRollover;
|
||||||
McpJsonMapper json = new JacksonMcpJsonMapperSupplier().get();
|
McpJsonMapper json = new JacksonMcpJsonMapperSupplier().get();
|
||||||
this.transport = HttpServletStreamableServerTransportProvider.builder()
|
this.transport = HttpServletStreamableServerTransportProvider.builder()
|
||||||
@@ -322,11 +377,11 @@ public final class FleetMcp {
|
|||||||
// (CB-113) — its MCP initialize is the reliable "the agent is up" signal.
|
// (CB-113) — its MCP initialize is the reliable "the agent is up" signal.
|
||||||
.contextExtractor(req -> {
|
.contextExtractor(req -> {
|
||||||
// One resolution per call, shared with the REST surface via CallerResolver so
|
// One resolution per call, shared with the REST surface via CallerResolver so
|
||||||
// the two paths cannot drift on who a caller is.
|
// the two paths cannot drift on who a caller is. fleetd #518: callers is
|
||||||
Principal p = callers != null
|
// required (never null) so there is no second, untested resolution path to
|
||||||
? callers.resolve(req.getRemoteAddr(), req.getRemotePort(),
|
// fall back to here — AuthorizationMode governs enforcement, not identity.
|
||||||
req.getHeader("Authorization"))
|
Principal p = callers.resolve(req.getRemoteAddr(), req.getRemotePort(),
|
||||||
: legacyPrincipal(identity, req.getRemoteAddr(), req.getRemotePort());
|
req.getHeader("Authorization"));
|
||||||
// CB-532: guard on the ROLE, not on the terminal being null. This excludes a
|
// CB-532: guard on the ROLE, not on the terminal being null. This excludes a
|
||||||
// lead, which carries its pane too, while including every spawned member role.
|
// lead, which carries its pane too, while including every spawned member role.
|
||||||
// Enrolling a lead would count it as an available member in the roster.
|
// Enrolling a lead would count it as an available member in the roster.
|
||||||
@@ -442,8 +497,8 @@ public final class FleetMcp {
|
|||||||
(exchange, _) -> {
|
(exchange, _) -> {
|
||||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_list", Map.of()), null);
|
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_list", Map.of()), null);
|
||||||
if (denied != null) return denied;
|
if (denied != null) return denied;
|
||||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
return listFleet(workers, sessions, messages, capacity, healthCoverage, loopHealth, quarantine, outage,
|
||||||
leadSeats, callers == null ? Map.of() : callers.leads(),
|
leadSeats, callers.leads(),
|
||||||
callerTerminal(exchange),
|
callerTerminal(exchange),
|
||||||
new CoordinationSource(leadChannel, peers),
|
new CoordinationSource(leadChannel, peers),
|
||||||
coordinatorVisibleTo(principal(exchange)));
|
coordinatorVisibleTo(principal(exchange)));
|
||||||
@@ -522,21 +577,9 @@ public final class FleetMcp {
|
|||||||
.toolCall(fleetWhoami, whoamiHandler)
|
.toolCall(fleetWhoami, whoamiHandler)
|
||||||
.toolCall(fleetHandover, handoverHandler)
|
.toolCall(fleetHandover, handoverHandler)
|
||||||
.build();
|
.build();
|
||||||
this.authz = callers;
|
|
||||||
this.metrics = metrics;
|
this.metrics = metrics;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Pre-CB-501 identity: worker if the connection maps to a pane, otherwise the primary. Used
|
|
||||||
* only by the legacy constructor, where authorization is not enforced anyway.
|
|
||||||
*/
|
|
||||||
private static Principal legacyPrincipal(ConnectionIdentity identity, String addr, int port) {
|
|
||||||
ConnectionIdentity.Caller c = identity.resolve(addr, port);
|
|
||||||
return c.terminal() != null
|
|
||||||
? Principal.worker(c.terminal(), c.pid())
|
|
||||||
: Principal.primary(c.pid());
|
|
||||||
}
|
|
||||||
|
|
||||||
/** The caller reconstructed from the transport context. */
|
/** The caller reconstructed from the transport context. */
|
||||||
private static Principal principal(McpSyncServerExchange exchange) {
|
private static Principal principal(McpSyncServerExchange exchange) {
|
||||||
return principalFrom(exchange.transportContext().get(CALLER_ROLE),
|
return principalFrom(exchange.transportContext().get(CALLER_ROLE),
|
||||||
@@ -591,8 +634,8 @@ public final class FleetMcp {
|
|||||||
McpSchema.CallToolResult denyFor(Principal caller, Authz.Action action, String target) {
|
McpSchema.CallToolResult denyFor(Principal caller, Authz.Action action, String target) {
|
||||||
// The enforcement switch lives HERE rather than in the exchange-facing wrapper: any future
|
// The enforcement switch lives HERE rather than in the exchange-facing wrapper: any future
|
||||||
// tool that calls this directly must not be able to skip the gate by accident.
|
// tool that calls this directly must not be able to skip the gate by accident.
|
||||||
if (authz == null) {
|
if (!authorizationEnforced) {
|
||||||
return null; // legacy constructor: authorization not enforced
|
return null; // AuthorizationMode.UNENFORCED: authorization not enforced (fleetd #518)
|
||||||
}
|
}
|
||||||
if (Authz.permits(caller, action, target)) {
|
if (Authz.permits(caller, action, target)) {
|
||||||
if (action != Authz.Action.READ) {
|
if (action != Authz.Action.READ) {
|
||||||
@@ -785,6 +828,12 @@ public final class FleetMcp {
|
|||||||
+ "answered (turnId stale)");
|
+ "answered (turnId stale)");
|
||||||
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, BUSY -> text("[no reply within " + timeout + "ms — worker "
|
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, BUSY -> text("[no reply within " + timeout + "ms — worker "
|
||||||
+ r.outcome().name().toLowerCase().replace("timed_out_", "") + "; retry or poll status]");
|
+ r.outcome().name().toLowerCase().replace("timed_out_", "") + "; retry or poll status]");
|
||||||
|
// fleetd #571: delivery is unknown here — agent.prompt pastes and submits in one call,
|
||||||
|
// so the message may already be sitting in the pane. Do not invite a blind retry the way
|
||||||
|
// the case above does; a resend on this route can double-deliver the same brief.
|
||||||
|
case TIMED_OUT_UNCONFIRMED -> text("[no reply within " + timeout + "ms — delivery unconfirmed; "
|
||||||
|
+ "the message may already have reached the worker, so a retry risks sending it "
|
||||||
|
+ "twice — poll status before resending]");
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1526,25 +1575,42 @@ public final class FleetMcp {
|
|||||||
* @param selfTerm the calling pane's terminal id, or blank for a caller with no pane
|
* @param selfTerm the calling pane's terminal id, or blank for a caller with no pane
|
||||||
*/
|
*/
|
||||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions,
|
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions,
|
||||||
Map<String, String> leads, String selfTerm) {
|
Map<String, String> leads, String selfTerm) {
|
||||||
return listFleet(workers, sessions, null, CapacitySource.none(), new HealthCoverageSource(() -> "off"),
|
return listFleet(workers, sessions, null, CapacitySource.none(), new HealthCoverageSource(() -> "off"),
|
||||||
QuarantineSource.none(), leads, selfTerm);
|
LoopHealthSource.none(), QuarantineSource.none(), leads, selfTerm);
|
||||||
|
}
|
||||||
|
|
||||||
|
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||||
|
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||||
|
QuarantineSource quarantine, Map<String, String> leads, String selfTerm) {
|
||||||
|
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, leads, selfTerm,
|
||||||
|
CoordinationSource.none());
|
||||||
}
|
}
|
||||||
|
|
||||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||||
QuarantineSource quarantine, Map<String, String> leads, String selfTerm) {
|
LoopHealthSource loopHealth, QuarantineSource quarantine,
|
||||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, leads, selfTerm,
|
Map<String, String> leads, String selfTerm) {
|
||||||
|
return listFleet(workers, sessions, messages, capacity, healthCoverage, loopHealth, quarantine, leads, selfTerm,
|
||||||
CoordinationSource.none());
|
CoordinationSource.none());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||||
|
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||||
|
LoopHealthSource loopHealth, QuarantineSource quarantine,
|
||||||
|
Map<String, String> leads, String selfTerm,
|
||||||
|
CoordinationSource coordination) {
|
||||||
|
return listFleet(workers, sessions, messages, capacity, healthCoverage, loopHealth, quarantine,
|
||||||
|
OutageSource.none(), LeadSeatSource.none(), leads, selfTerm, coordination, false);
|
||||||
|
}
|
||||||
|
|
||||||
/** As above, plus fleetd #201 Unit 5 cool-off facts (see {@link OutageSource}). */
|
/** As above, plus fleetd #201 Unit 5 cool-off facts (see {@link OutageSource}). */
|
||||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||||
QuarantineSource quarantine, OutageSource outage,
|
QuarantineSource quarantine, OutageSource outage,
|
||||||
Map<String, String> leads, String selfTerm) {
|
Map<String, String> leads, String selfTerm) {
|
||||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, outage,
|
||||||
LeadSeatSource.none(), leads, selfTerm, CoordinationSource.none());
|
LeadSeatSource.none(), leads, selfTerm, CoordinationSource.none(), false);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -1560,8 +1626,8 @@ public final class FleetMcp {
|
|||||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||||
QuarantineSource quarantine, Map<String, String> leads, String selfTerm,
|
QuarantineSource quarantine, Map<String, String> leads, String selfTerm,
|
||||||
CoordinationSource coordination) {
|
CoordinationSource coordination) {
|
||||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, OutageSource.none(),
|
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, OutageSource.none(),
|
||||||
LeadSeatSource.none(), leads, selfTerm, coordination);
|
LeadSeatSource.none(), leads, selfTerm, coordination, false);
|
||||||
}
|
}
|
||||||
|
|
||||||
/** As above, plus fleetd #201 Unit 5 cool-off facts (see {@link OutageSource}). */
|
/** As above, plus fleetd #201 Unit 5 cool-off facts (see {@link OutageSource}). */
|
||||||
@@ -1569,8 +1635,8 @@ public final class FleetMcp {
|
|||||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||||
QuarantineSource quarantine, OutageSource outage,
|
QuarantineSource quarantine, OutageSource outage,
|
||||||
Map<String, String> leads, String selfTerm, CoordinationSource coordination) {
|
Map<String, String> leads, String selfTerm, CoordinationSource coordination) {
|
||||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, outage,
|
||||||
LeadSeatSource.none(), leads, selfTerm, coordination);
|
LeadSeatSource.none(), leads, selfTerm, coordination, false);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -1591,7 +1657,7 @@ public final class FleetMcp {
|
|||||||
QuarantineSource quarantine, OutageSource outage,
|
QuarantineSource quarantine, OutageSource outage,
|
||||||
LeadSeatSource leadSeats, Map<String, String> leads, String selfTerm,
|
LeadSeatSource leadSeats, Map<String, String> leads, String selfTerm,
|
||||||
CoordinationSource coordination) {
|
CoordinationSource coordination) {
|
||||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, outage,
|
||||||
leadSeats, leads, selfTerm, coordination, false);
|
leadSeats, leads, selfTerm, coordination, false);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1611,8 +1677,18 @@ public final class FleetMcp {
|
|||||||
* an explicit {@code true}
|
* an explicit {@code true}
|
||||||
*/
|
*/
|
||||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||||
QuarantineSource quarantine, OutageSource outage,
|
QuarantineSource quarantine, OutageSource outage,
|
||||||
|
LeadSeatSource leadSeats, Map<String, String> leads, String selfTerm,
|
||||||
|
CoordinationSource coordination, boolean callerIsPrimary) {
|
||||||
|
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine,
|
||||||
|
outage, leadSeats, leads, selfTerm, coordination, callerIsPrimary);
|
||||||
|
}
|
||||||
|
|
||||||
|
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||||
|
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||||
|
LoopHealthSource loopHealth,
|
||||||
|
QuarantineSource quarantine, OutageSource outage,
|
||||||
LeadSeatSource leadSeats, Map<String, String> leads, String selfTerm,
|
LeadSeatSource leadSeats, Map<String, String> leads, String selfTerm,
|
||||||
CoordinationSource coordination, boolean callerIsPrimary) {
|
CoordinationSource coordination, boolean callerIsPrimary) {
|
||||||
try {
|
try {
|
||||||
@@ -1636,6 +1712,9 @@ public final class FleetMcp {
|
|||||||
Map<String, Object> result = new LinkedHashMap<>();
|
Map<String, Object> result = new LinkedHashMap<>();
|
||||||
result.put("leads", leadRows); result.put("members", out);
|
result.put("leads", leadRows); result.put("members", out);
|
||||||
result.put("healthCoverage", healthCoverage.value().get());
|
result.put("healthCoverage", healthCoverage.value().get());
|
||||||
|
result.put("loopHealth", Map.of(
|
||||||
|
"statusPoller", loopHealth.statusPoller().get().name(),
|
||||||
|
"sessionReaper", loopHealth.sessionReaper().get().name()));
|
||||||
// fleetd #439: coordinator/coordinatorView is lead-to-lead coordination state and must
|
// fleetd #439: coordinator/coordinatorView is lead-to-lead coordination state and must
|
||||||
// never reach a worker or an architect -- gate BEFORE assembling it, not after, so the
|
// never reach a worker or an architect -- gate BEFORE assembling it, not after, so the
|
||||||
// key is absent rather than present-and-empty.
|
// key is absent rather than present-and-empty.
|
||||||
@@ -2055,8 +2134,9 @@ public final class FleetMcp {
|
|||||||
return tool(FleetTool.SPAWN.wireName(),
|
return tool(FleetTool.SPAWN.wireName(),
|
||||||
"Spawn a new off-subscription member session. A member has two independent attributes: "
|
"Spawn a new off-subscription member session. A member has two independent attributes: "
|
||||||
+ "role (what it is for) and profile (which backend it runs on). Pass role to pick "
|
+ "role (what it is for) and profile (which backend it runs on). Pass role to pick "
|
||||||
+ "the contract — 'dev' implements a unit and opens its own PR, 'reviewer' reviews a "
|
+ "the contract — 'dev' implements a unit and opens its own PR, 'hunter' sweeps a "
|
||||||
+ "diff it did not write, 'architect' refines a ticket before anyone builds it; omit "
|
+ "scope without changing it, 'reviewer' reviews a diff it did not write, 'architect' "
|
||||||
|
+ "refines a ticket before anyone builds it; omit "
|
||||||
+ "it for 'dev'. Pass profile (from fleet_profiles) to pick the backend, or omit it "
|
+ "it for 'dev'. Pass profile (from fleet_profiles) to pick the backend, or omit it "
|
||||||
+ "for the default. The two are independent: a reviewer may run on the same profile "
|
+ "for the default. The two are independent: a reviewer may run on the same profile "
|
||||||
+ "as the dev it reviews. The member opens your current directory by default; pass "
|
+ "as the dev it reviews. The member opens your current directory by default; pass "
|
||||||
@@ -2073,7 +2153,7 @@ public final class FleetMcp {
|
|||||||
+ "one. Returns the member's sessionId (use with fleet_send) and paneId (use with "
|
+ "one. Returns the member's sessionId (use with fleet_send) and paneId (use with "
|
||||||
+ "fleet_stop).",
|
+ "fleet_stop).",
|
||||||
objectSchema(Map.of(
|
objectSchema(Map.of(
|
||||||
"role", stringProp("What the member is for: architect, dev or reviewer (default dev)"),
|
"role", stringProp("What the member is for: architect, dev, hunter, or reviewer (default dev)"),
|
||||||
"profile", stringProp("Which backend to run it on (omit for the default profile)"),
|
"profile", stringProp("Which backend to run it on (omit for the default profile)"),
|
||||||
"cwd", stringProp("Working directory for the member (omit to inherit yours)"),
|
"cwd", stringProp("Working directory for the member (omit to inherit yours)"),
|
||||||
"worktree", Map.of("type", "string", "description", "'true' or a ticket slug — requests an isolated git worktree"),
|
"worktree", Map.of("type", "string", "description", "'true' or a ticket slug — requests an isolated git worktree"),
|
||||||
@@ -2115,9 +2195,10 @@ public final class FleetMcp {
|
|||||||
+ "cannot reliably re-identify: some backends (e.g. opencode) resolve it from the "
|
+ "cannot reliably re-identify: some backends (e.g. opencode) resolve it from the "
|
||||||
+ "member's working directory, which only uniquely identifies a member when it "
|
+ "member's working directory, which only uniquely identifies a member when it "
|
||||||
+ "was spawned into its own fleetd-provisioned worktree (worktree:true/<slug>); a "
|
+ "was spawned into its own fleetd-provisioned worktree (worktree:true/<slug>); a "
|
||||||
+ "member spawned without one shares its directory with others and never reports "
|
+ "member spawned without one shares its directory with others and never reports "
|
||||||
+ "an id, however long it runs (fleetd #249). An empty 'members' "
|
+ "an id, however long it runs (fleetd #249). An empty 'members' "
|
||||||
+ "means no members are spawned; it says nothing about peers. When capacity "
|
+ "means no members are spawned; it says nothing about peers. 'loopHealth' reports "
|
||||||
|
+ "the RUNNING, STALLED, or STOPPED state of statusPoller and sessionReaper. When capacity "
|
||||||
+ "facts are configured, a 'capacity' row per profile reports 'free' — the "
|
+ "facts are configured, a 'capacity' row per profile reports 'free' — the "
|
||||||
+ "slots a fresh fleet_spawn on that profile will actually be granted right "
|
+ "slots a fresh fleet_spawn on that profile will actually be granted right "
|
||||||
+ "now (max(0, maxLoad - live)), the same check the spawn gate itself runs. A "
|
+ "now (max(0, maxLoad - live)), the same check the spawn gate itself runs. A "
|
||||||
|
|||||||
@@ -246,7 +246,7 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
|||||||
* neither. So before this method existed with a requeue step, it dropped {@link #held}'s entries
|
* neither. So before this method existed with a requeue step, it dropped {@link #held}'s entries
|
||||||
* for {@code target} while the broker still considered them outstanding: never acked, never
|
* for {@code target} while the broker still considered them outstanding: never acked, never
|
||||||
* nacked, never requeued, and no longer reachable by {@link #peek} — permanently invisible. This
|
* nacked, never requeued, and no longer reachable by {@link #peek} — permanently invisible. This
|
||||||
* is unlike {@link #handleRecovery} and {@link #close()}, whose bare {@code held.clear()} is
|
* is unlike {@link RecoveryListener#handleRecovery(Recoverable)} and {@link #close()}, whose bare {@code held.clear()} is
|
||||||
* correct because each has already made the broker requeue (a real connection drop, or
|
* correct because each has already made the broker requeue (a real connection drop, or
|
||||||
* {@code channel.close()} respectively) before clearing local state.
|
* {@code channel.close()} respectively) before clearing local state.
|
||||||
*
|
*
|
||||||
|
|||||||
@@ -92,8 +92,31 @@ public final class MessageService {
|
|||||||
QUESTION,
|
QUESTION,
|
||||||
/** Timed out after the message was delivered — the worker is still working. */
|
/** Timed out after the message was delivered — the worker is still working. */
|
||||||
TIMED_OUT_WORKING,
|
TIMED_OUT_WORKING,
|
||||||
/** Timed out before delivery — the message is still queued for the worker. */
|
/**
|
||||||
|
* Timed out with no confirmed delivery, and the target saw nothing — the message will not
|
||||||
|
* arrive later, so a caller may resend. {@link #send} reaches this outcome through {@link
|
||||||
|
* Injector#cancel} reporting one of two routes: {@link Injector.Cancellation#CANCELLED}
|
||||||
|
* means the message was still queued and this call removed it; {@link
|
||||||
|
* Injector.Cancellation#NOT_DELIVERED} means nothing was ever sent — the queue was cleared
|
||||||
|
* because the target never became ready or was abandoned, or the injector's call to the
|
||||||
|
* target's terminal ({@link dev.ltms.fleet.herdr.AgentControl#send}) failed with a herdr
|
||||||
|
* error that this codebase already treats as a confirmed absence. A third route,
|
||||||
|
* {@link Injector.Cancellation#ATTEMPTED}, used to be folded into this same outcome
|
||||||
|
* (fleetd #571) — it no longer is; see {@link #TIMED_OUT_UNCONFIRMED}.
|
||||||
|
*/
|
||||||
TIMED_OUT_QUEUED,
|
TIMED_OUT_QUEUED,
|
||||||
|
/**
|
||||||
|
* Timed out with delivery unknown. {@link #send} reaches this outcome when {@link
|
||||||
|
* Injector#cancel} reports {@link Injector.Cancellation#ATTEMPTED} (fleetd #551): the call
|
||||||
|
* to the target's terminal ({@link dev.ltms.fleet.herdr.AgentControl#send}) was made, but
|
||||||
|
* this caller never observed whether it reached the pane. {@code agent.prompt} pastes
|
||||||
|
* <em>and submits</em> in one call, so the target may already hold a complete, submitted
|
||||||
|
* turn and be working on it right now — the same reality as {@link #TIMED_OUT_WORKING},
|
||||||
|
* just not confirmed. The message may or may not have arrived. Treat this as neither a
|
||||||
|
* confirmed delivery nor a confirmed absence: a caller that resends on this outcome risks a
|
||||||
|
* double delivery — the same brief typed into the pane twice (fleetd #571).
|
||||||
|
*/
|
||||||
|
TIMED_OUT_UNCONFIRMED,
|
||||||
/** Another send to this session was in flight for the whole window. */
|
/** Another send to this session was in flight for the whole window. */
|
||||||
BUSY,
|
BUSY,
|
||||||
/**
|
/**
|
||||||
@@ -299,12 +322,21 @@ public final class MessageService {
|
|||||||
*/
|
*/
|
||||||
private final ConcurrentHashMap<String, Boolean> strandedReplies = new ConcurrentHashMap<>();
|
private final ConcurrentHashMap<String, Boolean> strandedReplies = new ConcurrentHashMap<>();
|
||||||
/**
|
/**
|
||||||
* Targets whose last send timed out with {@link Outcome#TIMED_OUT_QUEUED} (CB-640) — the
|
* Targets whose last send timed out with no confirmed delivery (CB-640) — {@link #send} called
|
||||||
* message never reached the {@link Injector} delivery window before the caller's deadline, so
|
* {@link Injector#cancel} and got back something other than {@code DELIVERED}. That covers
|
||||||
* it is still sitting in the injector's own per-target queue. Set where {@link #send} already
|
* three histories, not one: {@link Injector.Cancellation#CANCELLED} — the message was still
|
||||||
* computes {@code wasDelivered} for that outcome; no new queue is kept here, only the fact.
|
* queued and {@code cancel} removed it right there; {@link Injector.Cancellation#NOT_DELIVERED}
|
||||||
* Cleared the same way as {@link #strandedReplies}: the next accepted delivery for the target
|
* — nothing was ever sent, because the target never became ready, was torn down, or the call to
|
||||||
* ({@link #send} opening a fresh waiter) or a teardown ({@link #abandon}).
|
* its terminal failed with a herdr error this codebase already treats as a confirmed absence; or
|
||||||
|
* {@link Injector.Cancellation#ATTEMPTED} (fleetd #551) — the call to the target's terminal was
|
||||||
|
* made and its outcome is unknown, so the target may already hold a complete, submitted turn.
|
||||||
|
* Only the first two mean the message will not arrive later and the target saw nothing; on the
|
||||||
|
* third it may already have arrived in full — and the caller sees a different outcome for it
|
||||||
|
* ({@link Outcome#TIMED_OUT_UNCONFIRMED}, fleetd #571) than for the first two ({@link
|
||||||
|
* Outcome#TIMED_OUT_QUEUED}). Set where {@link #send} already computes {@code wasDelivered} for
|
||||||
|
* that outcome; no queue is kept here, only the fact that the send ended with no confirmed
|
||||||
|
* delivery. Cleared the same way as {@link #strandedReplies}: the next accepted delivery for the
|
||||||
|
* target ({@link #send} opening a fresh waiter) or a teardown ({@link #abandon}).
|
||||||
*/
|
*/
|
||||||
private final ConcurrentHashMap<String, Boolean> queuedDeliveries = new ConcurrentHashMap<>();
|
private final ConcurrentHashMap<String, Boolean> queuedDeliveries = new ConcurrentHashMap<>();
|
||||||
private final AtomicLong ticketSeq = new AtomicLong();
|
private final AtomicLong ticketSeq = new AtomicLong();
|
||||||
@@ -391,12 +423,22 @@ public final class MessageService {
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* Read-only delegation fact for fleet views (CB-640): {@code target}'s last send timed out
|
* Read-only delegation fact for fleet views (CB-640): {@code target}'s last send timed out
|
||||||
* before the {@link Injector} ever delivered it — the caller saw
|
* with no confirmed delivery — the caller saw {@link Outcome#TIMED_OUT_QUEUED} or {@link
|
||||||
* {@link Outcome#TIMED_OUT_QUEUED} (see the {@code TimeoutException} branch of {@link #send}),
|
* Outcome#TIMED_OUT_UNCONFIRMED} (fleetd #571; see the {@code TimeoutException} branch of
|
||||||
* and the message is still sitting in the injector's per-target queue waiting for the worker
|
* {@link #send}). Despite the method's name, this is not proof that a message is sitting in a
|
||||||
* to go idle. Distinct from {@link Outcome#TIMED_OUT_WORKING}, where delivery already happened
|
* queue: {@link Injector#cancel} reports this outcome through three routes. {@link
|
||||||
* and only the reply is outstanding. Cleared the next time this target's delivery is accepted
|
* Injector.Cancellation#CANCELLED} means the message was still queued and got removed right
|
||||||
* or the target is abandoned — see {@link #queuedDeliveries}.
|
* there. {@link Injector.Cancellation#NOT_DELIVERED} means nothing was ever sent — the target
|
||||||
|
* never became ready, was torn down, or the call to its terminal failed with a herdr error this
|
||||||
|
* codebase already treats as a confirmed absence. Only these two routes mean the message will
|
||||||
|
* not arrive later, and both report {@code TIMED_OUT_QUEUED}. {@link
|
||||||
|
* Injector.Cancellation#ATTEMPTED} (fleetd #551) means the call to the target's terminal was
|
||||||
|
* made and its outcome is unknown: {@code agent.prompt} pastes <em>and submits</em> in one
|
||||||
|
* call, so on this route the target may already hold a complete, submitted turn and be
|
||||||
|
* working on it right now — it does NOT follow that the target saw nothing, and this route
|
||||||
|
* reports {@code TIMED_OUT_UNCONFIRMED} instead. Distinct from {@link Outcome#TIMED_OUT_WORKING},
|
||||||
|
* where delivery already happened and only the reply is outstanding. Cleared the next time this
|
||||||
|
* target's delivery is accepted or the target is abandoned — see {@link #queuedDeliveries}.
|
||||||
*/
|
*/
|
||||||
public boolean hasQueuedDelivery(String target) {
|
public boolean hasQueuedDelivery(String target) {
|
||||||
return target != null && queuedDeliveries.containsKey(target);
|
return target != null && queuedDeliveries.containsKey(target);
|
||||||
@@ -416,15 +458,15 @@ public final class MessageService {
|
|||||||
/**
|
/**
|
||||||
* Read-only delegation fact for fleet views (CB-640): an async ticket is still
|
* Read-only delegation fact for fleet views (CB-640): an async ticket is still
|
||||||
* {@link Phase#PENDING} against {@code target}, yet nothing is actually in flight for it — no
|
* {@link Phase#PENDING} against {@code target}, yet nothing is actually in flight for it — no
|
||||||
* open rendezvous waiter ({@link #hasAcceptedDelivery}) and no message still sitting in the
|
* open rendezvous waiter ({@link #hasAcceptedDelivery}) and no record of a send that ended with
|
||||||
* injector's queue ({@link #hasQueuedDelivery}). A healthy PENDING ticket can briefly look this
|
* no confirmed delivery ({@link #hasQueuedDelivery}). A healthy PENDING ticket can briefly look
|
||||||
* way while its virtual thread has not yet been scheduled or is blocked on the session lock
|
* this way while its virtual thread has not yet been scheduled or is blocked on the session
|
||||||
* behind another send to the same target, so this is a snapshot fact for the health classifier
|
* lock behind another send to the same target, so this is a snapshot fact for the health
|
||||||
* to weigh across ticks, not proof on its own that the ticket is stuck. It also genuinely
|
* classifier to weigh across ticks, not proof on its own that the ticket is stuck. It also
|
||||||
* persists — not just as a passing race — once an async {@code fleet_ask} lapses unanswered:
|
* genuinely persists — not just as a passing race — once an async {@code fleet_ask} lapses
|
||||||
* {@link #ask} clears the ticket's question and returns it to {@code PENDING}, but {@link #send}
|
* unanswered: {@link #ask} clears the ticket's question and returns it to {@code PENDING}, but
|
||||||
* already closed the forward waiter the instant the question surfaced, so the target has
|
* {@link #send} already closed the forward waiter the instant the question surfaced, so the
|
||||||
* neither an accepted nor a queued delivery left to show for it.
|
* target has neither an accepted nor a queued delivery left to show for it.
|
||||||
*
|
*
|
||||||
* <p><strong>Deliberately still {@code question == null} only (fleetd #275).</strong> This
|
* <p><strong>Deliberately still {@code question == null} only (fleetd #275).</strong> This
|
||||||
* method must not also report a still-{@link Phase#ASKING} task as orphaned: the worker may
|
* method must not also report a still-{@link Phase#ASKING} task as orphaned: the worker may
|
||||||
@@ -611,7 +653,7 @@ public final class MessageService {
|
|||||||
return switch (o) {
|
return switch (o) {
|
||||||
case REPLIED -> "replied";
|
case REPLIED -> "replied";
|
||||||
case COMPLETED_UNREPLIED -> "completion_fallback";
|
case COMPLETED_UNREPLIED -> "completion_fallback";
|
||||||
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, BUSY -> "timeout";
|
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, TIMED_OUT_UNCONFIRMED, BUSY -> "timeout";
|
||||||
case WORKER_FAILED -> "failed";
|
case WORKER_FAILED -> "failed";
|
||||||
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
||||||
case STALE_TURN, QUESTION -> null; // not a completed delegation
|
case STALE_TURN, QUESTION -> null; // not a completed delegation
|
||||||
@@ -713,7 +755,7 @@ public final class MessageService {
|
|||||||
* failure.
|
* failure.
|
||||||
*
|
*
|
||||||
* <p><strong>Without {@code sweepAsking} on the release path, a target torn down while
|
* <p><strong>Without {@code sweepAsking} on the release path, a target torn down while
|
||||||
* genuinely {@code ASKING} was unrecoverable.</strong> {@link #resolveQuestion} had already
|
* genuinely {@code ASKING} was unrecoverable.</strong> {@link Rendezvous#resolveQuestion(String, String, String)} had already
|
||||||
* closed the forward waiter the instant the question surfaced (so the {@code waiter} branch
|
* closed the forward waiter the instant the question surfaced (so the {@code waiter} branch
|
||||||
* below finds nothing to fail), the {@code question == null} guard excluded the task from
|
* below finds nothing to fail), the {@code question == null} guard excluded the task from
|
||||||
* {@code matching} (so the loop below skipped it too), and the worker's own {@code fleet_ask}
|
* {@code matching} (so the loop below skipped it too), and the worker's own {@code fleet_ask}
|
||||||
@@ -940,6 +982,7 @@ public final class MessageService {
|
|||||||
} catch (TimeoutException e) {
|
} catch (TimeoutException e) {
|
||||||
boolean wasDelivered = delivery.completion().isDone()
|
boolean wasDelivered = delivery.completion().isDone()
|
||||||
&& !delivery.completion().isCompletedExceptionally();
|
&& !delivery.completion().isCompletedExceptionally();
|
||||||
|
Injector.Cancellation cancellation = null;
|
||||||
if (!wasDelivered) {
|
if (!wasDelivered) {
|
||||||
if (timeoutCancellationRaceHookForTest != null) {
|
if (timeoutCancellationRaceHookForTest != null) {
|
||||||
// Test-only (fleetd #345): see the field's own javadoc.
|
// Test-only (fleetd #345): see the field's own javadoc.
|
||||||
@@ -947,16 +990,27 @@ public final class MessageService {
|
|||||||
}
|
}
|
||||||
// The target monitor makes cancellation atomic with onStatus picking this
|
// The target monitor makes cancellation atomic with onStatus picking this
|
||||||
// Pending up. If pickup won, report TIMED_OUT_WORKING because the text landed.
|
// Pending up. If pickup won, report TIMED_OUT_WORKING because the text landed.
|
||||||
wasDelivered = injector.cancel(delivery) == Injector.Cancellation.DELIVERED;
|
cancellation = injector.cancel(delivery);
|
||||||
|
wasDelivered = cancellation == Injector.Cancellation.DELIVERED;
|
||||||
}
|
}
|
||||||
log.debug("send to {} timed out (delivered={})", target, wasDelivered);
|
log.debug("send to {} timed out (delivered={})", target, wasDelivered);
|
||||||
if (!wasDelivered) {
|
Outcome outcome;
|
||||||
// CB-640: record that delivery did not happen for fleet health (see
|
if (wasDelivered) {
|
||||||
// queuedDeliveries). The exact Pending was cancelled, so it cannot arrive later.
|
outcome = Outcome.TIMED_OUT_WORKING;
|
||||||
|
} else if (cancellation == Injector.Cancellation.ATTEMPTED) {
|
||||||
|
// fleetd #571: the call to the target's terminal was made and its outcome is
|
||||||
|
// unknown — the message may already have arrived in full, so this must not
|
||||||
|
// be reported as TIMED_OUT_QUEUED, which promises it never will.
|
||||||
|
outcome = Outcome.TIMED_OUT_UNCONFIRMED;
|
||||||
|
} else {
|
||||||
|
// CB-640: record that delivery is not confirmed, for fleet health (see
|
||||||
|
// queuedDeliveries). cancellation is CANCELLED (this call removed a
|
||||||
|
// still-queued Pending) or NOT_DELIVERED (an earlier attempt already failed
|
||||||
|
// with a confirmed absence) — both mean the target saw nothing.
|
||||||
queuedDeliveries.put(target, Boolean.TRUE);
|
queuedDeliveries.put(target, Boolean.TRUE);
|
||||||
|
outcome = Outcome.TIMED_OUT_QUEUED;
|
||||||
}
|
}
|
||||||
return recorded(new Reply(
|
return recorded(new Reply(outcome, null));
|
||||||
wasDelivered ? Outcome.TIMED_OUT_WORKING : Outcome.TIMED_OUT_QUEUED, null));
|
|
||||||
} catch (ExecutionException e) {
|
} catch (ExecutionException e) {
|
||||||
Throwable cause = e.getCause();
|
Throwable cause = e.getCause();
|
||||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||||
@@ -1096,21 +1150,33 @@ public final class MessageService {
|
|||||||
}
|
}
|
||||||
try {
|
try {
|
||||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession);
|
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession);
|
||||||
// #282: mirror send()'s registration (:802) so a SECOND fleet_ask inside this same
|
// fleetd #575: this try used to open below, AFTER the Task lookup/registration and the
|
||||||
// resumed turn can re-associate the async ticket with its new turnId via
|
// STALE_TURN early return that follows it — so that return was covered only by a
|
||||||
// markAsyncQuestion — without this, that second ask has no Task to attach to, and
|
// hand-rolled copy of the finally's own cleanup pair, not the finally itself. Widening the
|
||||||
// markAsyncQuestion silently returns null.
|
// try up to wrap the registration closes the gap structurally: every exit from here on,
|
||||||
Task task = asyncTasksByTurn.get(turnId);
|
// STALE_TURN included, now runs through the one finally below exactly once, and the
|
||||||
if (task != null) {
|
// duplicated pair is gone. This did not fix a live leak — see the ticket: neither
|
||||||
asyncTasksByWaiter.put(reply, task);
|
// rendezvous.answerAsk nor clearAsyncQuestion(turnId, false) can throw, so nothing ever
|
||||||
}
|
// actually left through the old gap uncovered — but #572 found this exact drift (one
|
||||||
if (!rendezvous.answerAsk(turnId, content)) {
|
// finally asserted, an identical sibling not) on this same file, and the hand-rolled copy
|
||||||
asyncTasksByWaiter.remove(reply);
|
// was the wrong shape to keep regardless of whether it was ever exercised.
|
||||||
rendezvous.close(workerSession, reply);
|
|
||||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
|
||||||
}
|
|
||||||
clearAsyncQuestion(turnId, false);
|
|
||||||
try {
|
try {
|
||||||
|
// #282: mirror send()'s registration (:802) so a SECOND fleet_ask inside this same
|
||||||
|
// resumed turn can re-associate the async ticket with its new turnId via
|
||||||
|
// markAsyncQuestion — without this, that second ask has no Task to attach to, and
|
||||||
|
// markAsyncQuestion silently returns null.
|
||||||
|
Task task = asyncTasksByTurn.get(turnId);
|
||||||
|
if (task != null) {
|
||||||
|
asyncTasksByWaiter.put(reply, task);
|
||||||
|
}
|
||||||
|
if (answerAskLapseRaceHookForTest != null) {
|
||||||
|
// Test-only (fleetd #575): see the field's own javadoc.
|
||||||
|
answerAskLapseRaceHookForTest.run();
|
||||||
|
}
|
||||||
|
if (!rendezvous.answerAsk(turnId, content)) {
|
||||||
|
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||||
|
}
|
||||||
|
clearAsyncQuestion(turnId, false);
|
||||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||||
Reply result = new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
Reply result = new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||||
// #282: this waiter can resolve with a FRESH question rather than a terminal reply —
|
// #282: this waiter can resolve with a FRESH question rather than a terminal reply —
|
||||||
@@ -1610,6 +1676,30 @@ public final class MessageService {
|
|||||||
this.askTimeoutRaceHookForTest = hook;
|
this.askTimeoutRaceHookForTest = hook;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Null in production; test seam for fleetd #575 — invoked from {@link #answer}, right after this
|
||||||
|
* call's own {@code Task} registration and right before its {@code rendezvous.answerAsk(turnId,
|
||||||
|
* content)} call. A test installs this to complete the SAME turnId's ask directly via {@link
|
||||||
|
* Rendezvous#answerAsk} from inside that exact window, deterministically reproducing what a
|
||||||
|
* second, concurrent {@code answer()} call racing to unblock the same ask can otherwise only
|
||||||
|
* win by timing luck: this call's own {@code askSession(turnId)} lookup at the top already saw
|
||||||
|
* the ask as open, but by the time it reaches {@code rendezvous.answerAsk} here, the other call
|
||||||
|
* already completed it (or the worker's own {@code ask()} teardown already closed it) — so this
|
||||||
|
* call must see {@code false} and return {@link Outcome#STALE_TURN}, exactly the "lapsed between
|
||||||
|
* the lookup and the unblock" case named at that call site. Proves the #575 fix (widening this
|
||||||
|
* method's try so a single finally covers this exit) does not change that outcome and still
|
||||||
|
* cleans this call's own {@code reply} up exactly once.
|
||||||
|
*/
|
||||||
|
private volatile Runnable answerAskLapseRaceHookForTest;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Test-only (fleetd #575): install {@link #answerAskLapseRaceHookForTest}. Package-private so the
|
||||||
|
* test, in the same package, can reach it without widening any production API.
|
||||||
|
*/
|
||||||
|
void setAnswerAskLapseRaceHookForTest(Runnable hook) {
|
||||||
|
this.answerAskLapseRaceHookForTest = hook;
|
||||||
|
}
|
||||||
|
|
||||||
/** A new send must not open a waiter while an async ticket owns this worker's paused turn. */
|
/** A new send must not open a waiter while an async ticket owns this worker's paused turn. */
|
||||||
private boolean hasAsyncQuestion(String target) {
|
private boolean hasAsyncQuestion(String target) {
|
||||||
return asyncTasksByTurn.values().stream().anyMatch(task -> target.equals(task.target));
|
return asyncTasksByTurn.values().stream().anyMatch(task -> target.equals(task.target));
|
||||||
|
|||||||
@@ -44,7 +44,7 @@ import java.util.stream.Collectors;
|
|||||||
* still holds an unacked message ({@link #pendingReplies}), tickets not yet collected
|
* still holds an unacked message ({@link #pendingReplies}), tickets not yet collected
|
||||||
* ({@link #pendingTickets}), and open questions not yet answered or lapsed
|
* ({@link #pendingTickets}), and open questions not yet answered or lapsed
|
||||||
* ({@link #pendingQuestions}) — and sends at most one combined nudge per tick
|
* ({@link #pendingQuestions}) — and sends at most one combined nudge per tick
|
||||||
* ({@link #injectNudge(String, int, int, int)}). Work that arrives while the lead is busy is
|
* ({@link #injectNudge(String, int, int, int, int, int)}). Work that arrives while the lead is busy is
|
||||||
* never lost: it is re-read fresh on every tick until the lead is injectable or its own reminder
|
* never lost: it is re-read fresh on every tick until the lead is injectable or its own reminder
|
||||||
* cap ({@link #maxReminders}) is reached — each source spends from its own budget, so one source
|
* cap ({@link #maxReminders}) is reached — each source spends from its own budget, so one source
|
||||||
* exhausting its cap does not stop nudges about the others (post-CB-590 regression fix; see
|
* exhausting its cap does not stop nudges about the others (post-CB-590 regression fix; see
|
||||||
|
|||||||
@@ -29,8 +29,8 @@ public enum MemberRole {
|
|||||||
* <p>Reads the repo and writes analysis. Never commits code and never opens a pull request —
|
* <p>Reads the repo and writes analysis. Never commits code and never opens a pull request —
|
||||||
* an architect that starts implementing has stopped doing the job that makes it useful.
|
* an architect that starts implementing has stopped doing the job that makes it useful.
|
||||||
*
|
*
|
||||||
* <p>Architects are the one member kind declared in config, because a lead addresses the same
|
* <p>Architects are the one member kind with live slot binding, because a lead addresses the
|
||||||
* slots across many tickets and needs a stable name for them.
|
* same slots across many tickets and needs a stable name for them.
|
||||||
*/
|
*/
|
||||||
ARCHITECT,
|
ARCHITECT,
|
||||||
|
|
||||||
@@ -43,6 +43,14 @@ public enum MemberRole {
|
|||||||
*/
|
*/
|
||||||
DEV,
|
DEV,
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Sweeps an assigned package for defects and reports several ranked findings.
|
||||||
|
*
|
||||||
|
* <p>Never changes code, commits, or opens a pull request. A hunt gathers evidence, which can
|
||||||
|
* include running the build, but leaves every fix to a later implementation unit.
|
||||||
|
*/
|
||||||
|
HUNTER,
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Reviews a diff it did not write and reports one structured finding.
|
* Reviews a diff it did not write and reports one structured finding.
|
||||||
*
|
*
|
||||||
@@ -59,7 +67,7 @@ public enum MemberRole {
|
|||||||
|
|
||||||
/**
|
/**
|
||||||
* The {@code fleet:} block that holds this role's pool — {@code architects},
|
* The {@code fleet:} block that holds this role's pool — {@code architects},
|
||||||
* {@code developers}, {@code reviewers}.
|
* {@code developers}, {@code hunters}, {@code reviewers}.
|
||||||
*
|
*
|
||||||
* <p>Plural, and not always the wire name: the pool of things a {@code dev} may run on reads
|
* <p>Plural, and not always the wire name: the pool of things a {@code dev} may run on reads
|
||||||
* naturally as {@code developers:}. The wire name stays the singular {@code dev}, because that
|
* naturally as {@code developers:}. The wire name stays the singular {@code dev}, because that
|
||||||
@@ -69,6 +77,7 @@ public enum MemberRole {
|
|||||||
return switch (this) {
|
return switch (this) {
|
||||||
case ARCHITECT -> "architects";
|
case ARCHITECT -> "architects";
|
||||||
case DEV -> "developers";
|
case DEV -> "developers";
|
||||||
|
case HUNTER -> "hunters";
|
||||||
case REVIEWER -> "reviewers";
|
case REVIEWER -> "reviewers";
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -94,6 +94,7 @@ public final class FleetApp {
|
|||||||
// .none() (the honest "feature not wired" view) for every constructor that does not pass one.
|
// .none() (the honest "feature not wired" view) for every constructor that does not pass one.
|
||||||
private final FleetMcp.QuarantineSource quarantine;
|
private final FleetMcp.QuarantineSource quarantine;
|
||||||
private final FleetMcp.OutageSource outage;
|
private final FleetMcp.OutageSource outage;
|
||||||
|
private final FleetMcp.LoopHealthSource loopHealth;
|
||||||
private final ObjectMapper mapper = new ObjectMapper();
|
private final ObjectMapper mapper = new ObjectMapper();
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -155,7 +156,8 @@ public final class FleetApp {
|
|||||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials) {
|
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials) {
|
||||||
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics,
|
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics,
|
||||||
deliverable, memberCredentials, FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none());
|
deliverable, memberCredentials, FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none(),
|
||||||
|
FleetMcp.LoopHealthSource.none());
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -169,8 +171,18 @@ public final class FleetApp {
|
|||||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||||
MessageService messages, MemberPresence presence,
|
MessageService messages, MemberPresence presence,
|
||||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials,
|
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials,
|
||||||
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage) {
|
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage) {
|
||||||
|
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics, deliverable,
|
||||||
|
memberCredentials, quarantine, outage, FleetMcp.LoopHealthSource.none());
|
||||||
|
}
|
||||||
|
|
||||||
|
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||||
|
MessageService messages, MemberPresence presence,
|
||||||
|
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||||
|
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials,
|
||||||
|
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage,
|
||||||
|
FleetMcp.LoopHealthSource loopHealth) {
|
||||||
this.herdr = herdr;
|
this.herdr = herdr;
|
||||||
this.memberHerdr = memberHerdr != null ? memberHerdr : herdr;
|
this.memberHerdr = memberHerdr != null ? memberHerdr : herdr;
|
||||||
this.workers = workers;
|
this.workers = workers;
|
||||||
@@ -183,6 +195,7 @@ public final class FleetApp {
|
|||||||
this.memberCredentials = memberCredentials != null ? memberCredentials : MemberCredentialPolicyView::absent;
|
this.memberCredentials = memberCredentials != null ? memberCredentials : MemberCredentialPolicyView::absent;
|
||||||
this.quarantine = quarantine != null ? quarantine : FleetMcp.QuarantineSource.none();
|
this.quarantine = quarantine != null ? quarantine : FleetMcp.QuarantineSource.none();
|
||||||
this.outage = outage != null ? outage : FleetMcp.OutageSource.none();
|
this.outage = outage != null ? outage : FleetMcp.OutageSource.none();
|
||||||
|
this.loopHealth = loopHealth != null ? loopHealth : FleetMcp.LoopHealthSource.none();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Wire routes onto a fresh, unstarted Javalin instance. Caller starts it. */
|
/** Wire routes onto a fresh, unstarted Javalin instance. Caller starts it. */
|
||||||
@@ -291,15 +304,19 @@ public final class FleetApp {
|
|||||||
* spotted by comparing two numbers by eye.
|
* spotted by comparing two numbers by eye.
|
||||||
*/
|
*/
|
||||||
private void healthz(Context ctx) {
|
private void healthz(Context ctx) {
|
||||||
|
HealthzResponse response = healthzResponse(herdr, memberHerdr, loopHealth);
|
||||||
|
ctx.status(response.status()).json(response.body());
|
||||||
|
}
|
||||||
|
|
||||||
|
record HealthzResponse(int status, Map<String, Object> body) { }
|
||||||
|
|
||||||
|
static HealthzResponse healthzResponse(HerdrClient herdr, HerdrClient memberHerdr,
|
||||||
|
FleetMcp.LoopHealthSource loopHealth) {
|
||||||
JsonNode pong;
|
JsonNode pong;
|
||||||
try {
|
try {
|
||||||
pong = herdr.call("ping");
|
pong = herdr.call("ping");
|
||||||
} catch (HerdrException e) {
|
} catch (HerdrException e) {
|
||||||
ctx.status(503).json(Map.of(
|
return degradedResponse("unreachable", e.getMessage(), loopHealth);
|
||||||
"status", "degraded",
|
|
||||||
"herdr", "unreachable",
|
|
||||||
"detail", e.getMessage()));
|
|
||||||
return;
|
|
||||||
}
|
}
|
||||||
Map<String, Object> body = new LinkedHashMap<>();
|
Map<String, Object> body = new LinkedHashMap<>();
|
||||||
body.put("status", "ok");
|
body.put("status", "ok");
|
||||||
@@ -311,11 +328,11 @@ public final class FleetApp {
|
|||||||
try {
|
try {
|
||||||
memberPong = memberHerdr.call("ping");
|
memberPong = memberHerdr.call("ping");
|
||||||
} catch (HerdrException e) {
|
} catch (HerdrException e) {
|
||||||
ctx.status(503).json(Map.of(
|
return new HealthzResponse(503, Map.of(
|
||||||
"status", "degraded",
|
"status", "degraded",
|
||||||
"herdr", "member unreachable",
|
"herdr", "member unreachable",
|
||||||
"detail", e.getMessage()));
|
"detail", e.getMessage(),
|
||||||
return;
|
"loopHealth", loopHealthView(loopHealth)));
|
||||||
}
|
}
|
||||||
int leadProtocol = pong.path("protocol").asInt();
|
int leadProtocol = pong.path("protocol").asInt();
|
||||||
int memberProtocol = memberPong.path("protocol").asInt();
|
int memberProtocol = memberPong.path("protocol").asInt();
|
||||||
@@ -326,7 +343,20 @@ public final class FleetApp {
|
|||||||
body.put("protocolMismatch", true);
|
body.put("protocolMismatch", true);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
ctx.status(200).json(body);
|
body.put("loopHealth", loopHealthView(loopHealth));
|
||||||
|
return new HealthzResponse(200, body);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Map<String, String> loopHealthView(FleetMcp.LoopHealthSource loopHealth) {
|
||||||
|
return Map.of("statusPoller", loopHealth.statusPoller().get().name(),
|
||||||
|
"sessionReaper", loopHealth.sessionReaper().get().name());
|
||||||
|
}
|
||||||
|
|
||||||
|
private static HealthzResponse degradedResponse(String herdr, String detail,
|
||||||
|
FleetMcp.LoopHealthSource loopHealth) {
|
||||||
|
return new HealthzResponse(503, Map.of(
|
||||||
|
"status", "degraded", "herdr", herdr, "detail", detail,
|
||||||
|
"loopHealth", loopHealthView(loopHealth)));
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -640,15 +670,31 @@ public final class FleetApp {
|
|||||||
}
|
}
|
||||||
default -> ctx.status(202).json(Map.of(
|
default -> ctx.status(202).json(Map.of(
|
||||||
"sessionId", id,
|
"sessionId", id,
|
||||||
|
// fleetd #571 (ticket comment 17126): no `default` here on purpose. This switch
|
||||||
|
// is an expression, so the compiler already demands every Outcome constant have
|
||||||
|
// an arm — adding an 11th constant to Outcome is a compile error here, not a
|
||||||
|
// silent fall-through. That is exactly the bug this ticket exists to fix:
|
||||||
|
// `default -> "done"` used to sit here and would have told a REST caller the
|
||||||
|
// delegation completed for TIMED_OUT_UNCONFIRMED, the one outcome where delivery
|
||||||
|
// is unknown. REPLIED, COMPLETED_UNREPLIED, QUESTION and STALE_TURN can never
|
||||||
|
// actually reach this inner switch — the outer switch above always dispatches
|
||||||
|
// them first — but they still need an arm to keep this switch exhaustive.
|
||||||
"status", switch (reply.outcome()) {
|
"status", switch (reply.outcome()) {
|
||||||
case TIMED_OUT_WORKING -> "working";
|
case TIMED_OUT_WORKING -> "working";
|
||||||
case TIMED_OUT_QUEUED -> "queued";
|
case TIMED_OUT_QUEUED -> "queued";
|
||||||
|
// Delivery here is unknown, not merely still queued — see
|
||||||
|
// Outcome#TIMED_OUT_UNCONFIRMED's own javadoc.
|
||||||
|
case TIMED_OUT_UNCONFIRMED -> "unconfirmed";
|
||||||
case BUSY -> "busy";
|
case BUSY -> "busy";
|
||||||
case WORKER_FAILED -> "failed";
|
case WORKER_FAILED -> "failed";
|
||||||
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
||||||
default -> "done"; // unreachable (terminal outcomes handled above)
|
case REPLIED, COMPLETED_UNREPLIED, QUESTION, STALE_TURN -> "done"; // unreachable
|
||||||
},
|
},
|
||||||
"detail", (reply.outcome() == MessageService.Outcome.WORKER_FAILED
|
"detail", reply.outcome() == MessageService.Outcome.TIMED_OUT_UNCONFIRMED
|
||||||
|
? "no reply within " + timeout + "ms; delivery is unconfirmed — the "
|
||||||
|
+ "message may already have reached the worker, so a resend "
|
||||||
|
+ "risks sending it twice; poll status first"
|
||||||
|
: (reply.outcome() == MessageService.Outcome.WORKER_FAILED
|
||||||
|| reply.outcome() == MessageService.Outcome.BACKEND_EXHAUSTED)
|
|| reply.outcome() == MessageService.Outcome.BACKEND_EXHAUSTED)
|
||||||
&& reply.text() != null
|
&& reply.text() != null
|
||||||
? reply.text()
|
? reply.text()
|
||||||
|
|||||||
@@ -302,9 +302,10 @@ public final class SessionManager implements TurnListener {
|
|||||||
* with no copy and no error. Do NOT fuse these back together; the cost of an orphaned worktree
|
* with no copy and no error. Do NOT fuse these back together; the cost of an orphaned worktree
|
||||||
* is a logged path an operator can reclaim, the cost of a deleted one is unrecoverable work.
|
* is a logged path an operator can reclaim, the cost of a deleted one is unrecoverable work.
|
||||||
*/
|
*/
|
||||||
private void release(String paneId, ReleaseCause cause) {
|
private MemberSession release(String paneId, ReleaseCause cause) {
|
||||||
MemberSession removed = registry.remove(paneId);
|
MemberSession removed = registry.remove(paneId);
|
||||||
releaseRemoved(paneId, removed, handles.remove(paneId), cause);
|
releaseRemoved(paneId, removed, handles.remove(paneId), cause);
|
||||||
|
return removed;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -1064,16 +1065,62 @@ public final class SessionManager implements TurnListener {
|
|||||||
* drain (see above), and a straggler must not buy the drain more time than the flag it lost the
|
* drain (see above), and a straggler must not buy the drain more time than the flag it lost the
|
||||||
* race against would have. In the ordinary case the sweep finds nothing and costs one empty
|
* race against would have. In the ordinary case the sweep finds nothing and costs one empty
|
||||||
* {@link #roster()} call.
|
* {@link #roster()} call.
|
||||||
|
*
|
||||||
|
* <p>fleetd #512: a drain that releases every session cleanly used to log nothing at all — the
|
||||||
|
* only log calls in this method and {@link #drainSnapshot} sit on abnormal paths, so "nothing
|
||||||
|
* logged" was indistinguishable from "died on the first session". The {@code log.info} at the
|
||||||
|
* end below is a positive assertion that the drain actually finished, on the normal path,
|
||||||
|
* every time — including the all-zero case, which is a common and legitimate outcome (no
|
||||||
|
* members were live) and must still produce the line. Both {@link #drainSnapshot} passes (the
|
||||||
|
* main snapshot and the straggler sweep) are folded into the one line: a caller reading two
|
||||||
|
* lines could not tell a two-pass drain from two separate drains.
|
||||||
|
*
|
||||||
|
* <p><strong>Non-goal: this line must never move into a {@code finally} block, and this method
|
||||||
|
* must never grow one around it.</strong> "Every time" above means every time the drain
|
||||||
|
* <em>finishes</em>, not every time this method exits. The absence of the line is the signal
|
||||||
|
* that the drain died, so a {@code finally} would destroy the signal and print confident
|
||||||
|
* partial counts in the same edit — the line would appear after a drain that threw, carrying
|
||||||
|
* whatever {@code tally} it had reached. Both halves of the value are lost at once. The line
|
||||||
|
* has to be the last statement of the successful path and reachable only from it.
|
||||||
|
*
|
||||||
|
* <p>This is written down because it is the obvious review comment ("shouldn't we always log
|
||||||
|
* the drain result?"), it sounds like thoroughness, and the paragraph above reads as an
|
||||||
|
* invitation to it. Raised by the fleet01 lead on 2026-09-12, from their 2026-09-10 incident:
|
||||||
|
* we only know that drain died on that host because it <em>threw</em>, and a
|
||||||
|
* {@code NoClassDefFoundError} reached the JVM's uncaught handler. A drain that hung on one
|
||||||
|
* session, or returned early on a condition rather than an exception, would leave no stack
|
||||||
|
* trace, no {@code ERROR} token and no priority — only a missing line. That makes the loud
|
||||||
|
* variant the one we have seen and the quiet variants the ones this line exists to catch.
|
||||||
|
*
|
||||||
|
* <p>Related: {@code released} and {@code abandoned} are counted incrementally inside {@link
|
||||||
|
* #drainSnapshot}'s loop and folded with {@link DrainTally#plus}, rather than derived from a
|
||||||
|
* collection read at the end, for the same reason. If a partial report is ever wanted it must
|
||||||
|
* be a different line with a different verb. One line must not serve both, or a reader cannot
|
||||||
|
* tell a finished drain from an interrupted one by its wording.
|
||||||
*/
|
*/
|
||||||
void drainAll(long timeoutNanos) {
|
void drainAll(long timeoutNanos) {
|
||||||
long deadline = System.nanoTime() + timeoutNanos;
|
long deadline = System.nanoTime() + timeoutNanos;
|
||||||
draining.set(true);
|
draining.set(true);
|
||||||
drainSnapshot(roster(), deadline);
|
DrainTally tally = drainSnapshot(roster(), deadline);
|
||||||
List<MemberSession> stragglers = roster();
|
List<MemberSession> stragglers = roster();
|
||||||
if (!stragglers.isEmpty()) {
|
if (!stragglers.isEmpty()) {
|
||||||
log.warn("drain sweep found {} session(s) registered after the drain snapshot was "
|
log.warn("drain sweep found {} session(s) registered after the drain snapshot was "
|
||||||
+ "taken (raced past the shutdown guard); draining them too", stragglers.size());
|
+ "taken (raced past the shutdown guard); draining them too", stragglers.size());
|
||||||
drainSnapshot(stragglers, deadline);
|
tally = tally.plus(drainSnapshot(stragglers, deadline));
|
||||||
|
}
|
||||||
|
log.info("drain complete: released={} abandoned={} (still BUSY at the shutdown deadline)",
|
||||||
|
tally.released(), tally.abandoned());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Running count for one {@link #drainAll} invocation, folded across both {@link #drainSnapshot}
|
||||||
|
* passes (fleetd #512). {@code abandoned} counts sessions that were still {@code BUSY} at the
|
||||||
|
* moment they were released — i.e. the whole-drain deadline passed before they left {@code BUSY}
|
||||||
|
* on their own (see {@link #drainSnapshot}) — a subset of {@code released}, not additional to it.
|
||||||
|
*/
|
||||||
|
private record DrainTally(int released, int abandoned) {
|
||||||
|
private DrainTally plus(DrainTally other) {
|
||||||
|
return new DrainTally(released + other.released, abandoned + other.abandoned);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1081,8 +1128,12 @@ public final class SessionManager implements TurnListener {
|
|||||||
* Drain exactly the sessions in {@code snapshot}, waiting out a {@code BUSY} one against the
|
* Drain exactly the sessions in {@code snapshot}, waiting out a {@code BUSY} one against the
|
||||||
* shared whole-drain {@code deadline} before releasing it. Shared by {@link #drainAll}'s main
|
* shared whole-drain {@code deadline} before releasing it. Shared by {@link #drainAll}'s main
|
||||||
* pass and its post-loop straggler sweep (fleetd #308) so both honor the same one budget.
|
* pass and its post-loop straggler sweep (fleetd #308) so both honor the same one budget.
|
||||||
|
* Returns how many sessions this pass released, and how many of those were still {@code BUSY}
|
||||||
|
* (abandoned mid-turn) at the moment of release.
|
||||||
*/
|
*/
|
||||||
private void drainSnapshot(List<MemberSession> snapshot, long deadline) {
|
private DrainTally drainSnapshot(List<MemberSession> snapshot, long deadline) {
|
||||||
|
int released = 0;
|
||||||
|
int abandoned = 0;
|
||||||
for (MemberSession s : snapshot) {
|
for (MemberSession s : snapshot) {
|
||||||
try {
|
try {
|
||||||
if (s.state() == MemberSession.State.BUSY) {
|
if (s.state() == MemberSession.State.BUSY) {
|
||||||
@@ -1100,11 +1151,16 @@ public final class SessionManager implements TurnListener {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
release(s.paneId(), ReleaseCause.SHUTDOWN);
|
MemberSession removed = release(s.paneId(), ReleaseCause.SHUTDOWN);
|
||||||
|
released++;
|
||||||
|
if (removed != null && removed.state() == MemberSession.State.BUSY) {
|
||||||
|
abandoned++;
|
||||||
|
}
|
||||||
} catch (RuntimeException e) {
|
} catch (RuntimeException e) {
|
||||||
log.warn("drain failed for pane={}; continuing with remaining sessions", s.paneId(), e);
|
log.warn("drain failed for pane={}; continuing with remaining sessions", s.paneId(), e);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
return new DrainTally(released, abandoned);
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
@@ -1,9 +1,11 @@
|
|||||||
package dev.ltms.fleet.session;
|
package dev.ltms.fleet.session;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||||
import org.slf4j.Logger;
|
import org.slf4j.Logger;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import java.util.concurrent.TimeUnit;
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Periodic virtual-thread reaper that tears down {@code READY}/{@code DONE} sessions which have
|
* Periodic virtual-thread reaper that tears down {@code READY}/{@code DONE} sessions which have
|
||||||
@@ -14,6 +16,16 @@ public final class SessionReaper {
|
|||||||
|
|
||||||
private static final Logger log = LoggerFactory.getLogger(SessionReaper.class);
|
private static final Logger log = LoggerFactory.getLogger(SessionReaper.class);
|
||||||
private static final long DEFAULT_INTERVAL_MILLIS = 5000;
|
private static final long DEFAULT_INTERVAL_MILLIS = 5000;
|
||||||
|
/**
|
||||||
|
* How many multiples of {@code intervalMillis} the last-completed round may age before
|
||||||
|
* {@link #health()} reports {@link LoopWatchdog.State#STALLED} (fleetd #544). One round here
|
||||||
|
* is {@code sessions.reapIdle} plus the (rare, best-effort) WIP sweep — both normally finish
|
||||||
|
* in a small fraction of one interval. At the default 5s interval this puts the threshold at
|
||||||
|
* 60s: generous enough that an occasional slow git call in the WIP sweep never trips it, short
|
||||||
|
* enough that a genuinely wedged reap round (mirroring the herdr-read-with-no-deadline hang
|
||||||
|
* {@code StatusPoller} guards against — see fleetd #544) is caught inside about a minute.
|
||||||
|
*/
|
||||||
|
private static final long STALE_AFTER_INTERVAL_MULTIPLIER = 12;
|
||||||
/** CB-586: the refs/wip age floor — never sweep a snapshot younger than 24h (the CB-586 rule). */
|
/** CB-586: the refs/wip age floor — never sweep a snapshot younger than 24h (the CB-586 rule). */
|
||||||
private static final long WIP_MIN_AGE_MILLIS = TimeUnit.HOURS.toMillis(24);
|
private static final long WIP_MIN_AGE_MILLIS = TimeUnit.HOURS.toMillis(24);
|
||||||
/**
|
/**
|
||||||
@@ -25,6 +37,7 @@ public final class SessionReaper {
|
|||||||
private final SessionManager sessions;
|
private final SessionManager sessions;
|
||||||
private final long idleTtlNanos;
|
private final long idleTtlNanos;
|
||||||
private final long intervalMillis;
|
private final long intervalMillis;
|
||||||
|
private final LoopWatchdog watchdog;
|
||||||
private volatile boolean running;
|
private volatile boolean running;
|
||||||
private Thread thread;
|
private Thread thread;
|
||||||
/**
|
/**
|
||||||
@@ -45,29 +58,61 @@ public final class SessionReaper {
|
|||||||
|
|
||||||
/** Construct a reaper with an explicit polling interval (useful for tests). */
|
/** Construct a reaper with an explicit polling interval (useful for tests). */
|
||||||
public SessionReaper(SessionManager sessions, long idleTtlSeconds, long intervalMillis) {
|
public SessionReaper(SessionManager sessions, long idleTtlSeconds, long intervalMillis) {
|
||||||
|
this(sessions, idleTtlSeconds, intervalMillis, System::nanoTime);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Full constructor — for tests: an injectable monotonic clock (fleetd #544). */
|
||||||
|
SessionReaper(SessionManager sessions, long idleTtlSeconds, long intervalMillis, LongSupplier nowNanos) {
|
||||||
this.sessions = sessions;
|
this.sessions = sessions;
|
||||||
this.idleTtlNanos = TimeUnit.SECONDS.toNanos(idleTtlSeconds);
|
this.idleTtlNanos = TimeUnit.SECONDS.toNanos(idleTtlSeconds);
|
||||||
this.intervalMillis = intervalMillis;
|
this.intervalMillis = intervalMillis;
|
||||||
|
this.watchdog = new LoopWatchdog(nowNanos, staleAfterNanos(intervalMillis));
|
||||||
|
}
|
||||||
|
|
||||||
|
private static long staleAfterNanos(long intervalMillis) {
|
||||||
|
return TimeUnit.MILLISECONDS.toNanos(Math.max(intervalMillis, 1)) * STALE_AFTER_INTERVAL_MULTIPLIER;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* This loop's progress fact (fleetd #544): {@code RUNNING}, {@code STALLED} (dead or parked —
|
||||||
|
* indistinguishable from outside, and never observed on purpose), or {@code STOPPED}
|
||||||
|
* ({@link #stop()} was called). See {@link LoopWatchdog} for why staleness — not thread
|
||||||
|
* liveness — is the signal.
|
||||||
|
*/
|
||||||
|
public LoopWatchdog.State health() {
|
||||||
|
return watchdog.state();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Start the reaper loop on a virtual thread. Idempotent. */
|
/** Start the reaper loop on a virtual thread. Idempotent. */
|
||||||
public synchronized void start() {
|
public synchronized void start() {
|
||||||
if (running) return;
|
if (running) return;
|
||||||
running = true;
|
running = true;
|
||||||
|
watchdog.reset(); // fleetd #544: a fresh grace period, not last run's stale mark/timestamp.
|
||||||
thread = Thread.ofVirtual().name("session-reaper").start(this::loop);
|
thread = Thread.ofVirtual().name("session-reaper").start(this::loop);
|
||||||
log.info("session reaper started (idle ttl {}s, interval {}ms)",
|
log.info("session reaper started (idle ttl {}s, interval {}ms)",
|
||||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos), intervalMillis);
|
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos), intervalMillis);
|
||||||
}
|
}
|
||||||
|
|
||||||
private void loop() {
|
private void loop() {
|
||||||
while (running) {
|
try {
|
||||||
try {
|
while (running) {
|
||||||
sessions.reapIdle(idleTtlNanos);
|
try {
|
||||||
} catch (RuntimeException e) {
|
sessions.reapIdle(idleTtlNanos);
|
||||||
log.warn("session reaper iteration failed; continuing", e);
|
} catch (Throwable e) {
|
||||||
|
log.error("session reaper iteration failed; continuing", e);
|
||||||
|
}
|
||||||
|
maybeSweepWipRefs();
|
||||||
|
// fleetd #544: a round is "complete" only after reapIdle AND the WIP-sweep gate have
|
||||||
|
// both returned — a hang in either (e.g. a wedged git call) means this line is never
|
||||||
|
// reached, so the watchdog goes stale exactly like a dead loop would.
|
||||||
|
watchdog.recordRoundComplete();
|
||||||
|
sleep();
|
||||||
}
|
}
|
||||||
maybeSweepWipRefs();
|
} finally {
|
||||||
sleep();
|
if (running) {
|
||||||
|
log.error("session reaper loop exited unexpectedly; it can be restarted");
|
||||||
|
}
|
||||||
|
running = false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -90,8 +135,8 @@ public final class SessionReaper {
|
|||||||
log.info("refs/wip retention sweep deleted {} snapshot ref(s) older than 24h whose "
|
log.info("refs/wip retention sweep deleted {} snapshot ref(s) older than 24h whose "
|
||||||
+ "content was already reachable from main", deleted);
|
+ "content was already reachable from main", deleted);
|
||||||
}
|
}
|
||||||
} catch (RuntimeException e) {
|
} catch (Throwable e) {
|
||||||
log.warn("refs/wip retention sweep failed; continuing", e);
|
log.error("refs/wip retention sweep failed; continuing", e);
|
||||||
}
|
}
|
||||||
// Set even when the sweep threw, so a broken repo is retried on the slow cadence rather
|
// Set even when the sweep threw, so a broken repo is retried on the slow cadence rather
|
||||||
// than hammering git on every 5-second iteration.
|
// than hammering git on every 5-second iteration.
|
||||||
@@ -111,6 +156,7 @@ public final class SessionReaper {
|
|||||||
/** Stop the reaper loop. Idempotent. */
|
/** Stop the reaper loop. Idempotent. */
|
||||||
public synchronized void stop() {
|
public synchronized void stop() {
|
||||||
running = false;
|
running = false;
|
||||||
|
watchdog.markStoppedByCaller(); // fleetd #544: this halt is on purpose — health() must say STOPPED, not STALLED.
|
||||||
if (thread != null) thread.interrupt();
|
if (thread != null) thread.interrupt();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,192 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import ch.qos.logback.classic.Level;
|
||||||
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrException;
|
||||||
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.util.concurrent.atomic.AtomicBoolean;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
import static org.junit.jupiter.api.Assertions.fail;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #498: {@code Fleetd.awaitHerdr} used to return a bare {@code boolean}, collapsing "the
|
||||||
|
* configured wait budget genuinely ran out" and "the waiting thread was interrupted, possibly
|
||||||
|
* milliseconds in" onto the same {@code false} — and the caller's log line printed only the
|
||||||
|
* configured budget, never how long the wait actually ran. This class covers both halves of the
|
||||||
|
* fix:
|
||||||
|
* <ul>
|
||||||
|
* <li>the seam — {@link Fleetd#awaitHerdr} itself, driven with an injected clock and a stub
|
||||||
|
* {@link HerdrClient}, one test per {@link Fleetd.HerdrWaitResult};</li>
|
||||||
|
* <li>the call site — {@link Fleetd#logHerdrWaitOutcomeAndShouldReap}, the exact decision {@code
|
||||||
|
* main} calls (extracted here because {@code main} itself boots the whole daemon and cannot
|
||||||
|
* be driven from a unit test), pinning the three distinct log messages it emits.</li>
|
||||||
|
* </ul>
|
||||||
|
* Every expected message below is a plain literal, not built from {@code HERDR_WAIT_SECONDS} or
|
||||||
|
* any other production constant — a test that derives its expectation the way the code does
|
||||||
|
* cannot see a change to either (fleetd #496's identical trap).
|
||||||
|
*/
|
||||||
|
class FleetdAwaitHerdrTest {
|
||||||
|
|
||||||
|
// ---- the seam: Fleetd.awaitHerdr ----------------------------------------------------------
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void answeredReturnsImmediatelyWithZeroElapsedAndNeverPolls() {
|
||||||
|
HerdrStub herdr = new HerdrStub(0); // succeeds on the very first call
|
||||||
|
LongSupplier clock = fixedClock(1_000L);
|
||||||
|
AtomicBoolean polled = new AtomicBoolean(false);
|
||||||
|
Runnable poller = () -> polled.set(true);
|
||||||
|
|
||||||
|
Fleetd.HerdrAwaitOutcome outcome = Fleetd.awaitHerdr(herdr, clock, poller);
|
||||||
|
|
||||||
|
assertEquals(Fleetd.HerdrWaitResult.ANSWERED, outcome.result());
|
||||||
|
assertEquals(0L, outcome.elapsedNanos(), "a fixed clock must measure zero elapsed time");
|
||||||
|
assertFalse(polled.get(), "herdr answering on the first try must never poll");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void deadlinePassedIsMeasuredNotAssumed() {
|
||||||
|
HerdrStub herdr = new HerdrStub(-1); // never succeeds
|
||||||
|
// call order inside awaitHerdr: start, then per failed attempt: deadline-check, elapsed-calc
|
||||||
|
ScriptedClock clock = new ScriptedClock(0L, 30_500_000_000L, 30_500_000_000L);
|
||||||
|
Runnable poller = () -> fail("the deadline was already exceeded on the first attempt — must not poll");
|
||||||
|
|
||||||
|
Fleetd.HerdrAwaitOutcome outcome = Fleetd.awaitHerdr(herdr, clock, poller);
|
||||||
|
|
||||||
|
assertEquals(Fleetd.HerdrWaitResult.DEADLINE_PASSED, outcome.result());
|
||||||
|
assertEquals(30_500_000_000L, outcome.elapsedNanos(),
|
||||||
|
"elapsed must be the MEASURED clock delta, not the configured budget");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void interruptedIsDistinctFromDeadlinePassedAndPreservesTheInterruptFlag() {
|
||||||
|
HerdrStub herdr = new HerdrStub(-1); // never succeeds
|
||||||
|
// start=0, deadline-check returns 500ms (well under the 30s budget) -> not deadline-passed,
|
||||||
|
// then the poller interrupts, and the elapsed-calc call returns 750ms.
|
||||||
|
ScriptedClock clock = new ScriptedClock(0L, 500_000_000L, 750_000_000L);
|
||||||
|
Runnable poller = () -> Thread.currentThread().interrupt();
|
||||||
|
|
||||||
|
try {
|
||||||
|
Fleetd.HerdrAwaitOutcome outcome = Fleetd.awaitHerdr(herdr, clock, poller);
|
||||||
|
|
||||||
|
assertEquals(Fleetd.HerdrWaitResult.INTERRUPTED, outcome.result());
|
||||||
|
assertEquals(750_000_000L, outcome.elapsedNanos(),
|
||||||
|
"elapsed must be measured even when the wait ends via interruption, not the deadline");
|
||||||
|
assertTrue(Thread.currentThread().isInterrupted(),
|
||||||
|
"the interrupt flag the old code re-set must still be set on return");
|
||||||
|
} finally {
|
||||||
|
Thread.interrupted(); // clear it so it cannot leak into another test on this thread
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- the call site: Fleetd.logHerdrWaitOutcomeAndShouldReap -------------------------------
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void answeredLogsNothingAndSaysReap() {
|
||||||
|
try (CapturedLog log = attach()) {
|
||||||
|
boolean shouldReap = Fleetd.logHerdrWaitOutcomeAndShouldReap(
|
||||||
|
new Fleetd.HerdrAwaitOutcome(Fleetd.HerdrWaitResult.ANSWERED, 0L));
|
||||||
|
|
||||||
|
assertTrue(shouldReap, "only ANSWERED should tell main to reap orphan workers");
|
||||||
|
assertEquals(0, log.events().size(), "the answered path logs nothing itself");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void deadlinePassedLogsConfiguredAndMeasuredElapsedTogether() {
|
||||||
|
try (CapturedLog log = attach()) {
|
||||||
|
boolean shouldReap = Fleetd.logHerdrWaitOutcomeAndShouldReap(
|
||||||
|
new Fleetd.HerdrAwaitOutcome(Fleetd.HerdrWaitResult.DEADLINE_PASSED, 30_500_000_000L));
|
||||||
|
|
||||||
|
assertFalse(shouldReap, "a deadline-passed wait must not tell main to reap");
|
||||||
|
assertEquals(1, log.events().size());
|
||||||
|
ILoggingEvent event = log.events().getFirst();
|
||||||
|
assertEquals(Level.WARN, event.getLevel());
|
||||||
|
assertEquals("herdr did not answer within the configured wait (configured=30s "
|
||||||
|
+ "elapsed=30500ms) — starting anyway; /healthz will report degraded until it "
|
||||||
|
+ "comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||||
|
event.getFormattedMessage());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void interruptedLogsItsOwnMessageAndNeverClaimsTheBudgetElapsed() {
|
||||||
|
try (CapturedLog log = attach()) {
|
||||||
|
// 3ms: the ticket's own example of "a few milliseconds in", not the 30s budget.
|
||||||
|
boolean shouldReap = Fleetd.logHerdrWaitOutcomeAndShouldReap(
|
||||||
|
new Fleetd.HerdrAwaitOutcome(Fleetd.HerdrWaitResult.INTERRUPTED, 3_000_000L));
|
||||||
|
|
||||||
|
assertFalse(shouldReap, "an interrupted wait must not tell main to reap");
|
||||||
|
assertEquals(1, log.events().size());
|
||||||
|
ILoggingEvent event = log.events().getFirst();
|
||||||
|
assertEquals(Level.WARN, event.getLevel());
|
||||||
|
String message = event.getFormattedMessage();
|
||||||
|
assertEquals("herdr wait was interrupted before the configured wait ran out "
|
||||||
|
+ "(configured=30s elapsed=3ms) — starting anyway; /healthz will report "
|
||||||
|
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||||
|
message);
|
||||||
|
assertFalse(message.contains("did not answer"),
|
||||||
|
"an interrupted wait must not be reported as if herdr failed to answer within the budget");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- fixtures --------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/** Always returns the same value, i.e. a clock that measures zero elapsed time. */
|
||||||
|
private static LongSupplier fixedClock(long value) {
|
||||||
|
return () -> value;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Returns each value in order, then repeats the last one for any call beyond the list. */
|
||||||
|
private static final class ScriptedClock implements LongSupplier {
|
||||||
|
private final long[] values;
|
||||||
|
private int index;
|
||||||
|
|
||||||
|
ScriptedClock(long... values) {
|
||||||
|
this.values = values;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public long getAsLong() {
|
||||||
|
long v = values[Math.min(index, values.length - 1)];
|
||||||
|
if (index < values.length - 1) {
|
||||||
|
index++;
|
||||||
|
}
|
||||||
|
return v;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Fails {@code failuresBeforeSuccess} times, then succeeds forever; {@code -1} never succeeds. */
|
||||||
|
private static final class HerdrStub implements HerdrClient {
|
||||||
|
private final int failuresBeforeSuccess;
|
||||||
|
private int calls;
|
||||||
|
|
||||||
|
HerdrStub(int failuresBeforeSuccess) {
|
||||||
|
this.failuresBeforeSuccess = failuresBeforeSuccess;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public JsonNode call(String method, Object params) throws HerdrException {
|
||||||
|
calls++;
|
||||||
|
if (failuresBeforeSuccess < 0 || calls <= failuresBeforeSuccess) {
|
||||||
|
throw new HerdrException("herdr not up yet");
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static CapturedLog attach() {
|
||||||
|
return CapturedLog.at(Fleetd.class, Level.DEBUG);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,160 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.mcp.FleetMcp;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
|
|
||||||
|
import java.nio.file.Files;
|
||||||
|
import java.nio.file.Path;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #426: {@code FleetHealthMonitor.coverage} had zero references anywhere in the test
|
||||||
|
* tree — not the method, not either output string, not the field it populates. {@code
|
||||||
|
* FleetHealthMonitorCoverageTest} (package {@code dev.ltms.fleet.health}) pins the three-branch
|
||||||
|
* method itself; that is the easy half.
|
||||||
|
*
|
||||||
|
* <p>The half that actually matters is this one: {@link Fleetd#healthCoverageSource} is the exact
|
||||||
|
* call site {@code Fleetd.main} wires into {@code FleetMcp}'s constructor, and it is what feeds
|
||||||
|
* {@code fleet_list}'s {@code healthCoverage} field (see {@code FleetMcp#listFleet}'s {@code
|
||||||
|
* result.put("healthCoverage", healthCoverage.value().get())}). Measured precedent on fleetd #423
|
||||||
|
* (for #415): swapping two arguments at a call site like this one — recreating #415's defect with
|
||||||
|
* the keys exchanged — compiled with 0 errors and ran the ENTIRE suite (1506 tests) green. A
|
||||||
|
* thoroughly-tested method proves nothing about whether the call site pairs its arguments correctly;
|
||||||
|
* only a test that drives the call site itself can catch that.
|
||||||
|
*
|
||||||
|
* <p><strong>Why this is not driven through a real {@code Fleetd.main} the way #407 drives its five
|
||||||
|
* reporters</strong> (keep the config invalid, assert on the log line emitted before {@code
|
||||||
|
* cfg.validateAll()} throws): all five of #407's reporters run in {@code main} before {@code
|
||||||
|
* validateAll()} (line ~171). {@link Fleetd#healthCoverageSource} is built during {@code FleetMcp}
|
||||||
|
* construction, which happens only after {@code UnixSocketHerdrClient.connect} has already opened a
|
||||||
|
* real herdr socket (line ~188) and after {@code sessions}/{@code workers} are constructed. Reaching
|
||||||
|
* this call site by actually running {@code main} would require a real socket connect — banned by
|
||||||
|
* this ticket's hard constraints — so #407's option 1 does not apply here. Instead {@link
|
||||||
|
* Fleetd#healthCoverageSource} is extracted to a directly-callable package-private factory, the same
|
||||||
|
* shape {@link Fleetd#capacitySource} and {@link Fleetd#quarantineSource} already use for the same
|
||||||
|
* reason (see {@code FleetdCapacitySourceWiringTest}, the direct precedent this test follows).
|
||||||
|
*
|
||||||
|
* <p>Uses a real {@link FleetConfig#load} + {@link ConfigRef} (no socket, no port bind, no spawn,
|
||||||
|
* nothing written outside {@code @TempDir}) so the fixture goes through the actual YAML parser and
|
||||||
|
* {@code FleetConfig.Health}/{@code Notifications} records, not a hand-built stand-in that could
|
||||||
|
* silently drift from what the parser actually produces.
|
||||||
|
*/
|
||||||
|
class FleetdHealthCoverageSourceWiringTest {
|
||||||
|
|
||||||
|
private static final String BASE = """
|
||||||
|
bind:
|
||||||
|
host: 127.0.0.1
|
||||||
|
port: 8765
|
||||||
|
herdrSocket: ~/.config/herdr/herdr.sock
|
||||||
|
""";
|
||||||
|
|
||||||
|
private static final String HEALTH_DISABLED_WITH_WEBHOOK = BASE + """
|
||||||
|
health:
|
||||||
|
enabled: false
|
||||||
|
notifications:
|
||||||
|
mode: webhook
|
||||||
|
""";
|
||||||
|
|
||||||
|
private static final String HEALTH_DETECTION_ONLY = BASE + """
|
||||||
|
health:
|
||||||
|
enabled: true
|
||||||
|
""";
|
||||||
|
|
||||||
|
private static final String HEALTH_FULL = BASE + """
|
||||||
|
health:
|
||||||
|
enabled: true
|
||||||
|
notifications:
|
||||||
|
mode: webhook
|
||||||
|
""";
|
||||||
|
|
||||||
|
private static final String HEALTH_ABSENT = BASE;
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("enabled: false reports off, even with a webhook configured")
|
||||||
|
void disabledHealthReportsOff(@TempDir Path dir) throws Exception {
|
||||||
|
FleetMcp.HealthCoverageSource source = sourceFor(dir, HEALTH_DISABLED_WITH_WEBHOOK);
|
||||||
|
|
||||||
|
assertEquals("off", source.value().get(),
|
||||||
|
"health.enabled: false must report 'off' regardless of notifications — flipping "
|
||||||
|
+ "the 'enabled' argument at the HealthCoverageSource call site would report "
|
||||||
|
+ "'full' here instead");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("enabled with no notifications reports detection-only")
|
||||||
|
void enabledWithoutNotificationsReportsDetectionOnly(@TempDir Path dir) throws Exception {
|
||||||
|
FleetMcp.HealthCoverageSource source = sourceFor(dir, HEALTH_DETECTION_ONLY);
|
||||||
|
|
||||||
|
assertEquals("detection-only", source.value().get());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("enabled with a webhook configured reports full")
|
||||||
|
void enabledWithNotificationsReportsFull(@TempDir Path dir) throws Exception {
|
||||||
|
FleetMcp.HealthCoverageSource source = sourceFor(dir, HEALTH_FULL);
|
||||||
|
|
||||||
|
assertEquals("full", source.value().get(),
|
||||||
|
"health.enabled: true with notifications.mode: webhook must report 'full' — "
|
||||||
|
+ "swapping 'full' and 'detection-only' at the call site, or breaking the "
|
||||||
|
+ "enabled/notificationConfigured argument pairing, would report "
|
||||||
|
+ "'detection-only' here instead");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("an absent health: block reports off")
|
||||||
|
void absentHealthBlockReportsOff(@TempDir Path dir) throws Exception {
|
||||||
|
FleetMcp.HealthCoverageSource source = sourceFor(dir, HEALTH_ABSENT);
|
||||||
|
|
||||||
|
assertEquals("off", source.value().get());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #426, the live-wiring half: {@code health.notifications} is a {@code
|
||||||
|
* ConfigRef.SPLIT_KEYS} entry, and {@link Fleetd#healthCoverageSource} reads {@code
|
||||||
|
* config.get().health()} live (not the frozen startup {@code cfg}) — exactly like {@link
|
||||||
|
* Fleetd#capacitySource}'s {@code maxLoad} ({@code FleetdCapacitySourceWiringTest}'s {@code
|
||||||
|
* reloadedMaxLoadStillChangesWhatFleetListReports}). A hot-reloaded notifications block must
|
||||||
|
* change what {@code fleet_list} reports without a restart; a fix that froze the whole source
|
||||||
|
* against the startup snapshot would silently break that and every other test above would stay
|
||||||
|
* green, since none of them reload.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
@DisplayName("a hot notifications reload still changes what fleet_list reports")
|
||||||
|
void reloadedNotificationsStillChangeWhatFleetListReports(@TempDir Path dir) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, HEALTH_DETECTION_ONLY);
|
||||||
|
FleetConfig cfg = FleetConfig.load(file);
|
||||||
|
ConfigRef config = new ConfigRef(file, cfg);
|
||||||
|
|
||||||
|
FleetMcp.HealthCoverageSource source = Fleetd.healthCoverageSource(config);
|
||||||
|
assertEquals("detection-only", source.value().get(),
|
||||||
|
"sanity: detection-only before any reload");
|
||||||
|
|
||||||
|
Files.writeString(file, HEALTH_FULL);
|
||||||
|
assertTrue(config.reload().applied());
|
||||||
|
// The live snapshot now carries the webhook — proves the reload really happened and this
|
||||||
|
// test is not accidentally passing because nothing changed.
|
||||||
|
assertTrue(config.get().health().notifications() != null
|
||||||
|
&& config.get().health().notifications().configured(),
|
||||||
|
"sanity: the reloaded config really carries a configured webhook");
|
||||||
|
|
||||||
|
assertEquals("full", source.value().get(),
|
||||||
|
"the SAME HealthCoverageSource instance must reflect a reloaded notifications "
|
||||||
|
+ "block without a restart — health.notifications is read live off "
|
||||||
|
+ "config.get(), exactly like capacitySource's maxLoad");
|
||||||
|
}
|
||||||
|
|
||||||
|
private static FleetMcp.HealthCoverageSource sourceFor(Path dir, String yaml) throws Exception {
|
||||||
|
Path file = dir.resolve("fleetd.yaml");
|
||||||
|
Files.writeString(file, yaml);
|
||||||
|
FleetConfig cfg = FleetConfig.load(file);
|
||||||
|
ConfigRef config = new ConfigRef(file, cfg);
|
||||||
|
return Fleetd.healthCoverageSource(config);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,15 +1,12 @@
|
|||||||
package dev.ltms.fleet;
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
import ch.qos.logback.classic.Level;
|
import ch.qos.logback.classic.Level;
|
||||||
import ch.qos.logback.classic.Logger;
|
|
||||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
import ch.qos.logback.core.read.ListAppender;
|
|
||||||
import dev.ltms.fleet.config.FleetConfig;
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
import dev.ltms.fleet.msg.LeadMailbox;
|
import dev.ltms.fleet.msg.LeadMailbox;
|
||||||
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
import org.slf4j.LoggerFactory;
|
|
||||||
|
|
||||||
import java.util.List;
|
|
||||||
import java.util.Map;
|
import java.util.Map;
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.*;
|
import static org.junit.jupiter.api.Assertions.*;
|
||||||
@@ -52,29 +49,22 @@ class FleetdLeadMailboxSelectionTest {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
private static ListAppender<ILoggingEvent> captureFleetdLogs() {
|
private static String joined(CapturedLog captured, Level level) {
|
||||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
return captured.events().stream().filter(e -> e.getLevel() == level)
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.start();
|
|
||||||
logger.addAppender(appender);
|
|
||||||
return appender;
|
|
||||||
}
|
|
||||||
|
|
||||||
private static String joined(ListAppender<ILoggingEvent> appender, Level level) {
|
|
||||||
return appender.list.stream().filter(e -> e.getLevel() == level)
|
|
||||||
.map(ILoggingEvent::getFormattedMessage).reduce("", (a, b) -> a + "\n" + b);
|
.map(ILoggingEvent::getFormattedMessage).reduce("", (a, b) -> a + "\n" + b);
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void noCoordinatorBlockLeavesTheFeatureOffSilently() {
|
void noCoordinatorBlockLeavesTheFeatureOffSilently() {
|
||||||
var appender = captureFleetdLogs();
|
try (var captured = CapturedLog.of(Fleetd.class)) {
|
||||||
var opener = new RecordingOpener();
|
var opener = new RecordingOpener();
|
||||||
|
|
||||||
assertNull(Fleetd.openLeadMailbox(null, Map.of(), opener));
|
assertNull(Fleetd.openLeadMailbox(null, Map.of(), opener));
|
||||||
|
|
||||||
assertNull(opener.offeredUri, "nothing configured means nothing is opened");
|
assertNull(opener.offeredUri, "nothing configured means nothing is opened");
|
||||||
assertEquals("", joined(appender, Level.WARN),
|
assertEquals("", joined(captured, Level.WARN),
|
||||||
"an opt-in feature nobody asked for must not warn on every boot");
|
"an opt-in feature nobody asked for must not warn on every boot");
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -114,31 +104,33 @@ class FleetdLeadMailboxSelectionTest {
|
|||||||
|
|
||||||
@Test
|
@Test
|
||||||
void warnsAndStaysOffWhenSelfIdIsMissing() {
|
void warnsAndStaysOffWhenSelfIdIsMissing() {
|
||||||
var appender = captureFleetdLogs();
|
try (var captured = CapturedLog.of(Fleetd.class)) {
|
||||||
var opener = new RecordingOpener();
|
var opener = new RecordingOpener();
|
||||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, null, null, null);
|
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, null, null, null);
|
||||||
|
|
||||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener));
|
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener));
|
||||||
|
|
||||||
assertNull(opener.offeredUri, "a mailbox with no owning coord-id has no queue to declare");
|
assertNull(opener.offeredUri, "a mailbox with no owning coord-id has no queue to declare");
|
||||||
String warns = joined(appender, Level.WARN);
|
String warns = joined(captured, Level.WARN);
|
||||||
assertTrue(warns.contains("coordinator.selfId"), () -> "say which key is missing: " + warns);
|
assertTrue(warns.contains("coordinator.selfId"), () -> "say which key is missing: " + warns);
|
||||||
assertFalse(warns.contains(SECRET), () -> "the URI's password must never be logged: " + warns);
|
assertFalse(warns.contains(SECRET), () -> "the URI's password must never be logged: " + warns);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void warnsAndStaysOffWhenTheBrokerIsUnreachableAtBoot() {
|
void warnsAndStaysOffWhenTheBrokerIsUnreachableAtBoot() {
|
||||||
var appender = captureFleetdLogs();
|
try (var captured = CapturedLog.of(Fleetd.class)) {
|
||||||
var opener = new RecordingOpener();
|
var opener = new RecordingOpener();
|
||||||
opener.unreachable = true;
|
opener.unreachable = true;
|
||||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null, null);
|
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null, null);
|
||||||
|
|
||||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener),
|
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener),
|
||||||
"a down coordination broker turns the feature off; it must never take the daemon down");
|
"a down coordination broker turns the feature off; it must never take the daemon down");
|
||||||
|
|
||||||
String warns = joined(appender, Level.WARN);
|
String warns = joined(captured, Level.WARN);
|
||||||
assertTrue(warns.contains("coord.example"), () -> "name the host that failed: " + warns);
|
assertTrue(warns.contains("coord.example"), () -> "name the host that failed: " + warns);
|
||||||
assertFalse(warns.contains(SECRET), () -> "with credentials stripped: " + warns);
|
assertFalse(warns.contains(SECRET), () -> "with credentials stripped: " + warns);
|
||||||
assertTrue(warns.contains("Connection refused"), () -> "and the real reason: " + warns);
|
assertTrue(warns.contains("Connection refused"), () -> "and the real reason: " + warns);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,174 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.inject.Injector;
|
||||||
|
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||||
|
import dev.ltms.fleet.inject.StatusPoller;
|
||||||
|
import dev.ltms.fleet.mcp.FleetMcp;
|
||||||
|
import dev.ltms.fleet.peer.Capability;
|
||||||
|
import dev.ltms.fleet.peer.PeerHandle;
|
||||||
|
import dev.ltms.fleet.peer.PeerLauncher;
|
||||||
|
import dev.ltms.fleet.peer.SpawnRequest;
|
||||||
|
import dev.ltms.fleet.placement.PlacementDecision;
|
||||||
|
import dev.ltms.fleet.session.SessionManager;
|
||||||
|
import dev.ltms.fleet.session.SessionReaper;
|
||||||
|
import org.junit.jupiter.api.DisplayName;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Set;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #562 follow-up (issue comment "HOLD on PR #579"): {@code Fleetd.main}'s {@code loopHealth}
|
||||||
|
* local used to be a bare {@code new FleetMcp.LoopHealthSource(poller::health, ...)} built inline,
|
||||||
|
* with nothing a test could call directly. Measured on that shape: replacing {@code
|
||||||
|
* poller::health} with a constant {@code () -> LoopWatchdog.State.RUNNING} at the call site
|
||||||
|
* compiled with 0 errors and left all 1771 existing tests green — the daemon could be changed to
|
||||||
|
* always report the {@link StatusPoller} as {@code RUNNING}, so the watchdog could never fire and
|
||||||
|
* a stalled poller would be invisible, while every test stayed green. That is exactly the false
|
||||||
|
* negative this ticket exists to prevent.
|
||||||
|
*
|
||||||
|
* <p>The five tests PR #579 added ({@code FleetMcpTest}, {@code FleetAppTest}) all build their own
|
||||||
|
* {@link FleetMcp.LoopHealthSource} directly with fixed lambdas — they prove the seam ({@code
|
||||||
|
* LoopHealthSource} reports what it is given) and nothing about what {@code Fleetd.main} actually
|
||||||
|
* gives it. This is the same hand-built-vs-config-wired shape as fleetd #561/#248/#426.
|
||||||
|
*
|
||||||
|
* <p>The fix extracts the inline {@code new} into {@link Fleetd#loopHealthSource}, a package-private
|
||||||
|
* factory in the same style as {@link Fleetd#capacitySource} and {@link Fleetd#healthCoverageSource}
|
||||||
|
* — which is exactly what makes it directly callable here. This test calls that factory with real
|
||||||
|
* {@link StatusPoller}/{@link SessionReaper} instances (never started, so no herdr or git I/O
|
||||||
|
* happens) and pins each half separately, plus the {@code reaper == null} branch: one invariant
|
||||||
|
* wired at three places needs three assertions, not one combined check whose non-zero total could
|
||||||
|
* hide a gap at any single place.
|
||||||
|
*/
|
||||||
|
class FleetdLoopHealthSourceWiringTest {
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("the statusPoller half reports the real poller's health, not a hardcoded state")
|
||||||
|
void statusPollerHalfReflectsThePollersRealHealth() {
|
||||||
|
// Stopped without ever being started — stop() still marks the watchdog STOPPED. A poller
|
||||||
|
// that has never reported RUNNING is the discriminating case: if Fleetd.loopHealthSource
|
||||||
|
// ever hardcoded RUNNING (the exact mutation this test exists to catch), this would fail.
|
||||||
|
StatusPoller stoppedPoller = freshPoller();
|
||||||
|
stoppedPoller.stop();
|
||||||
|
SessionReaper unusedReaper = freshReaper(); // present only to satisfy the signature
|
||||||
|
|
||||||
|
FleetMcp.LoopHealthSource source = Fleetd.loopHealthSource(stoppedPoller, unusedReaper);
|
||||||
|
|
||||||
|
assertEquals(LoopWatchdog.State.STOPPED, source.statusPoller().get(),
|
||||||
|
"the statusPoller supplier must delegate to the real poller's health() — "
|
||||||
|
+ "replacing poller::health with a constant () -> RUNNING at the "
|
||||||
|
+ "Fleetd.loopHealthSource call site must fail this assertion");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("the sessionReaper half reports the real reaper's health, not a hardcoded state")
|
||||||
|
void sessionReaperHalfReflectsTheReapersRealHealth() {
|
||||||
|
StatusPoller unusedPoller = freshPoller(); // present only to satisfy the signature
|
||||||
|
SessionReaper stoppedReaper = freshReaper();
|
||||||
|
stoppedReaper.stop();
|
||||||
|
|
||||||
|
FleetMcp.LoopHealthSource source = Fleetd.loopHealthSource(unusedPoller, stoppedReaper);
|
||||||
|
|
||||||
|
assertEquals(LoopWatchdog.State.STOPPED, source.sessionReaper().get(),
|
||||||
|
"the sessionReaper supplier must delegate to the real reaper's health() — "
|
||||||
|
+ "replacing reaper.health() with a constant at the "
|
||||||
|
+ "Fleetd.loopHealthSource call site must fail this assertion");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
@DisplayName("a null reaper (idle ttl not configured) still reports STOPPED, not a crash")
|
||||||
|
void nullReaperStillReportsStopped() {
|
||||||
|
// SessionReaper is only constructed when lifecycle.idleTtlSeconds is configured (see the
|
||||||
|
// `reaper` local in Fleetd.main) — a real deployment routinely passes null here. That null
|
||||||
|
// check is real behaviour, not a simplification to delete: it must keep reporting STOPPED
|
||||||
|
// rather than throwing a NullPointerException on the first fleet_list/healthz call.
|
||||||
|
StatusPoller runningPoller = freshPoller();
|
||||||
|
|
||||||
|
FleetMcp.LoopHealthSource source = Fleetd.loopHealthSource(runningPoller, null);
|
||||||
|
|
||||||
|
assertEquals(LoopWatchdog.State.STOPPED, source.sessionReaper().get(),
|
||||||
|
"reaper == null must still report STOPPED, exactly like an intentionally-stopped "
|
||||||
|
+ "reaper would — do not delete this null check to simplify the wiring");
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Never started, so no herdr call is ever made; freshly constructed reports RUNNING. */
|
||||||
|
private static StatusPoller freshPoller() {
|
||||||
|
AgentControl agents = new AgentControl(new FakeHerdr());
|
||||||
|
return new StatusPoller(agents, new Injector(agents), 1000);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Never started, so no git/session I/O is ever made; freshly constructed reports RUNNING. */
|
||||||
|
private static SessionReaper freshReaper() {
|
||||||
|
return new SessionReaper(new SessionManager(new NeverSpawnsLauncher()), 60, 1000);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Same minimal shape as {@code FleetdBackendErrorSinkTest.NeverSpawnsLauncher} — every method
|
||||||
|
* throws or returns an empty/no-op value, since a {@link SessionReaper} that is only ever
|
||||||
|
* constructed and then stopped (never started) never calls any of them.
|
||||||
|
*/
|
||||||
|
private static final class NeverSpawnsLauncher implements PeerLauncher {
|
||||||
|
@Override
|
||||||
|
public Set<Capability> capabilities() {
|
||||||
|
return Set.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Set<Capability> capabilitiesFor(String profileName) {
|
||||||
|
return Set.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public PeerHandle spawn(SpawnRequest req) {
|
||||||
|
throw new UnsupportedOperationException("not reachable — this test never acquires a session");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public PeerHandle spawn(SpawnRequest req, PlacementDecision decision) {
|
||||||
|
throw new UnsupportedOperationException("not reachable — this test never acquires a session");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public Set<String> profiles() {
|
||||||
|
return Set.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public String defaultProfile() {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public String effectiveCwd(SpawnRequest req) {
|
||||||
|
throw new UnsupportedOperationException("not reachable — this test never acquires a session");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<String> parityOverlay(String profileName) {
|
||||||
|
return List.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public List<?> list() {
|
||||||
|
return List.of();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public int reapOrphanWorkers() {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void stop(String id) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean clearContext(String id) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,15 +1,13 @@
|
|||||||
package dev.ltms.fleet;
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
import ch.qos.logback.classic.Level;
|
import ch.qos.logback.classic.Level;
|
||||||
import ch.qos.logback.classic.Logger;
|
|
||||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
import ch.qos.logback.core.read.ListAppender;
|
|
||||||
import dev.ltms.fleet.config.FleetConfig;
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
import dev.ltms.fleet.msg.AmqpReplyInbox;
|
import dev.ltms.fleet.msg.AmqpReplyInbox;
|
||||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||||
import dev.ltms.fleet.msg.ReplyInbox;
|
import dev.ltms.fleet.msg.ReplyInbox;
|
||||||
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
import org.slf4j.LoggerFactory;
|
|
||||||
|
|
||||||
import java.net.ServerSocket;
|
import java.net.ServerSocket;
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
@@ -50,43 +48,34 @@ class FleetdReplyInboxSelectionTest {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
private static ListAppender<ILoggingEvent> attach() {
|
private static CapturedLog attach() {
|
||||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
|
||||||
// logback-test.xml pins dev.ltms.fleet to WARN; raise it so INFO selection lines are captured.
|
// logback-test.xml pins dev.ltms.fleet to WARN; raise it so INFO selection lines are captured.
|
||||||
logger.setLevel(Level.INFO);
|
return CapturedLog.at(Fleetd.class, Level.INFO);
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.start();
|
|
||||||
logger.addAppender(appender);
|
|
||||||
return appender;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
private static void assertNoLogContains(List<ILoggingEvent> events, String secret) {
|
||||||
((Logger) LoggerFactory.getLogger(Fleetd.class)).detachAppender(appender);
|
assertTrue(events.stream().noneMatch(e -> e.getFormattedMessage().contains(secret)),
|
||||||
}
|
|
||||||
|
|
||||||
private static void assertNoLogContains(ListAppender<ILoggingEvent> appender, String secret) {
|
|
||||||
assertTrue(appender.list.stream().noneMatch(e -> e.getFormattedMessage().contains(secret)),
|
|
||||||
"no log line may contain the resolved URI's password");
|
"no log line may contain the resolved URI's password");
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void uriEnvSetAndPresentSelectsAmqpWithTheResolvedUri() {
|
void uriEnvSetAndPresentSelectsAmqpWithTheResolvedUri() {
|
||||||
FleetConfig.Broker broker = new FleetConfig.Broker(null, "LAVINMQ_URI", null);
|
FleetConfig.Broker broker = new FleetConfig.Broker(null, "LAVINMQ_URI", null);
|
||||||
recording(broker, Map.of("LAVINMQ_URI", RESOLVED_URI), false, (opener, appender, inbox) -> {
|
recording(broker, Map.of("LAVINMQ_URI", RESOLVED_URI), false, (opener, events, inbox) -> {
|
||||||
assertEquals(RESOLVED_URI, opener.offeredUri,
|
assertEquals(RESOLVED_URI, opener.offeredUri,
|
||||||
"the daemon must connect with the value resolved from uriEnv — selection, not just parse");
|
"the daemon must connect with the value resolved from uriEnv — selection, not just parse");
|
||||||
assertEquals(opener.inbox, inbox, "the AMQP opener's inbox is what is selected");
|
assertEquals(opener.inbox, inbox, "the AMQP opener's inbox is what is selected");
|
||||||
assertNoLogContains(appender, SECRET);
|
assertNoLogContains(events, SECRET);
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void uriEnvSetButVariableMissingFallsBackToInMemoryAndWarns() {
|
void uriEnvSetButVariableMissingFallsBackToInMemoryAndWarns() {
|
||||||
FleetConfig.Broker broker = new FleetConfig.Broker(null, "LAVINMQ_URI", null);
|
FleetConfig.Broker broker = new FleetConfig.Broker(null, "LAVINMQ_URI", null);
|
||||||
recording(broker, Map.of(), false, (opener, appender, inbox) -> {
|
recording(broker, Map.of(), false, (opener, events, inbox) -> {
|
||||||
assertInstanceOf(InMemoryReplyInbox.class, inbox);
|
assertInstanceOf(InMemoryReplyInbox.class, inbox);
|
||||||
assertNull(opener.offeredUri, "AMQP must never be attempted when the variable is missing");
|
assertNull(opener.offeredUri, "AMQP must never be attempted when the variable is missing");
|
||||||
assertTrue(hasWarnContaining(appender, "LAVINMQ_URI") && hasWarnContaining(appender, "DISABLED"),
|
assertTrue(hasWarnContaining(events, "LAVINMQ_URI") && hasWarnContaining(events, "DISABLED"),
|
||||||
"a missing uriEnv variable must warn loudly, not fail silently");
|
"a missing uriEnv variable must warn loudly, not fail silently");
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
@@ -94,10 +83,10 @@ class FleetdReplyInboxSelectionTest {
|
|||||||
@Test
|
@Test
|
||||||
void uriEnvSetButVariableBlankFallsBackToInMemoryAndWarns() {
|
void uriEnvSetButVariableBlankFallsBackToInMemoryAndWarns() {
|
||||||
FleetConfig.Broker broker = new FleetConfig.Broker("amqp://user:lame@old:5672/", "LAVINMQ_URI", null);
|
FleetConfig.Broker broker = new FleetConfig.Broker("amqp://user:lame@old:5672/", "LAVINMQ_URI", null);
|
||||||
recording(broker, Map.of("LAVINMQ_URI", " "), false, (opener, appender, inbox) -> {
|
recording(broker, Map.of("LAVINMQ_URI", " "), false, (opener, events, inbox) -> {
|
||||||
assertInstanceOf(InMemoryReplyInbox.class, inbox);
|
assertInstanceOf(InMemoryReplyInbox.class, inbox);
|
||||||
assertNull(opener.offeredUri, "a blank env value must not select AMQP, not even via the literal uri");
|
assertNull(opener.offeredUri, "a blank env value must not select AMQP, not even via the literal uri");
|
||||||
assertTrue(hasWarnContaining(appender, "LAVINMQ_URI"),
|
assertTrue(hasWarnContaining(events, "LAVINMQ_URI"),
|
||||||
"a blank uriEnv value must warn, and must not fall back to the literal uri");
|
"a blank uriEnv value must warn, and must not fall back to the literal uri");
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
@@ -106,22 +95,22 @@ class FleetdReplyInboxSelectionTest {
|
|||||||
void bothUriAndUriEnvSetUriEnvWinsDeterministically() {
|
void bothUriAndUriEnvSetUriEnvWinsDeterministically() {
|
||||||
FleetConfig.Broker broker
|
FleetConfig.Broker broker
|
||||||
= new FleetConfig.Broker("amqp://user:oldpw@old.example:5672/", "LAVINMQ_URI", null);
|
= new FleetConfig.Broker("amqp://user:oldpw@old.example:5672/", "LAVINMQ_URI", null);
|
||||||
recording(broker, Map.of("LAVINMQ_URI", RESOLVED_URI), false, (opener, appender, inbox) -> {
|
recording(broker, Map.of("LAVINMQ_URI", RESOLVED_URI), false, (opener, events, inbox) -> {
|
||||||
assertEquals(RESOLVED_URI, opener.offeredUri,
|
assertEquals(RESOLVED_URI, opener.offeredUri,
|
||||||
"uriEnv must win over uri, deterministically, every run");
|
"uriEnv must win over uri, deterministically, every run");
|
||||||
assertTrue(appender.list.stream().anyMatch(e -> e.getFormattedMessage().contains("broker.uri is ignored")),
|
assertTrue(events.stream().anyMatch(e -> e.getFormattedMessage().contains("broker.uri is ignored")),
|
||||||
"must log that the literal uri is ignored when uriEnv is set");
|
"must log that the literal uri is ignored when uriEnv is set");
|
||||||
assertNoLogContains(appender, SECRET);
|
assertNoLogContains(events, SECRET);
|
||||||
assertNoLogContains(appender, "oldpw");
|
assertNoLogContains(events, "oldpw");
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void unreachableBrokerStartsDaemonWithInMemoryInboxAndLoudWarning() {
|
void unreachableBrokerStartsDaemonWithInMemoryInboxAndLoudWarning() {
|
||||||
FleetConfig.Broker broker = new FleetConfig.Broker(null, "LAVINMQ_URI", null);
|
FleetConfig.Broker broker = new FleetConfig.Broker(null, "LAVINMQ_URI", null);
|
||||||
recording(broker, Map.of("LAVINMQ_URI", RESOLVED_URI), true, (opener, appender, inbox) -> {
|
recording(broker, Map.of("LAVINMQ_URI", RESOLVED_URI), true, (opener, events, inbox) -> {
|
||||||
assertInstanceOf(InMemoryReplyInbox.class, inbox, "an unreachable broker must NOT stop the daemon");
|
assertInstanceOf(InMemoryReplyInbox.class, inbox, "an unreachable broker must NOT stop the daemon");
|
||||||
String warn = appender.list.stream()
|
String warn = events.stream()
|
||||||
.filter(e -> e.getLevel() == Level.WARN)
|
.filter(e -> e.getLevel() == Level.WARN)
|
||||||
.map(ILoggingEvent::getFormattedMessage)
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
.reduce("", (a, b) -> a + "\n" + b)
|
.reduce("", (a, b) -> a + "\n" + b)
|
||||||
@@ -129,16 +118,16 @@ class FleetdReplyInboxSelectionTest {
|
|||||||
assertTrue(warn.contains("durable") && warn.contains("soft-state"),
|
assertTrue(warn.contains("durable") && warn.contains("soft-state"),
|
||||||
"the warning must say exactly what was lost: durable delivery off, replies soft-state");
|
"the warning must say exactly what was lost: durable delivery off, replies soft-state");
|
||||||
assertTrue(!warn.contains(SECRET), "the failing URI must be logged with credentials stripped");
|
assertTrue(!warn.contains(SECRET), "the failing URI must be logged with credentials stripped");
|
||||||
assertNoLogContains(appender, SECRET);
|
assertNoLogContains(events, SECRET);
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void noBrokerConfiguredStaysQuietInMemory() {
|
void noBrokerConfiguredStaysQuietInMemory() {
|
||||||
FleetConfig.Broker broker = null;
|
FleetConfig.Broker broker = null;
|
||||||
recording(broker, Map.of(), false, (opener, appender, inbox) -> {
|
recording(broker, Map.of(), false, (opener, events, inbox) -> {
|
||||||
assertInstanceOf(InMemoryReplyInbox.class, inbox);
|
assertInstanceOf(InMemoryReplyInbox.class, inbox);
|
||||||
assertTrue(appender.list.stream().noneMatch(e -> e.getLevel() == Level.WARN),
|
assertTrue(events.stream().noneMatch(e -> e.getLevel() == Level.WARN),
|
||||||
"no broker configured must keep the existing QUIET in-memory path — no warning");
|
"no broker configured must keep the existing QUIET in-memory path — no warning");
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
@@ -152,21 +141,18 @@ class FleetdReplyInboxSelectionTest {
|
|||||||
closedPort = s.getLocalPort();
|
closedPort = s.getLocalPort();
|
||||||
}
|
}
|
||||||
FleetConfig.Broker broker = new FleetConfig.Broker(null, "LAVINMQ_URI", null);
|
FleetConfig.Broker broker = new FleetConfig.Broker(null, "LAVINMQ_URI", null);
|
||||||
ListAppender<ILoggingEvent> appender = attach();
|
try (CapturedLog log = attach()) {
|
||||||
try {
|
|
||||||
ReplyInbox inbox = Fleetd.selectReplyInbox(
|
ReplyInbox inbox = Fleetd.selectReplyInbox(
|
||||||
broker, Map.of("LAVINMQ_URI", "amqp://user:" + SECRET + "@127.0.0.1:" + closedPort + "/vh"),
|
broker, Map.of("LAVINMQ_URI", "amqp://user:" + SECRET + "@127.0.0.1:" + closedPort + "/vh"),
|
||||||
AmqpReplyInbox::open);
|
AmqpReplyInbox::open);
|
||||||
assertInstanceOf(InMemoryReplyInbox.class, inbox,
|
assertInstanceOf(InMemoryReplyInbox.class, inbox,
|
||||||
"a genuinely unreachable broker (real AmqpReplyInbox::open) must fall back to in-memory");
|
"a genuinely unreachable broker (real AmqpReplyInbox::open) must fall back to in-memory");
|
||||||
} finally {
|
assertNoLogContains(log.events(), SECRET);
|
||||||
detach(appender);
|
|
||||||
}
|
}
|
||||||
assertNoLogContains(appender, SECRET);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
private boolean hasWarnContaining(ListAppender<ILoggingEvent> appender, String fragment) {
|
private boolean hasWarnContaining(List<ILoggingEvent> events, String fragment) {
|
||||||
return appender.list.stream().anyMatch(e ->
|
return events.stream().anyMatch(e ->
|
||||||
e.getLevel() == Level.WARN && e.getFormattedMessage().contains(fragment));
|
e.getLevel() == Level.WARN && e.getFormattedMessage().contains(fragment));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -174,18 +160,14 @@ class FleetdReplyInboxSelectionTest {
|
|||||||
private void recording(FleetConfig.Broker broker, Map<String, String> env, boolean unreachable, Check check) {
|
private void recording(FleetConfig.Broker broker, Map<String, String> env, boolean unreachable, Check check) {
|
||||||
RecordingAmqp opener = new RecordingAmqp();
|
RecordingAmqp opener = new RecordingAmqp();
|
||||||
opener.unreachable = unreachable;
|
opener.unreachable = unreachable;
|
||||||
ListAppender<ILoggingEvent> appender = attach();
|
try (CapturedLog log = attach()) {
|
||||||
ReplyInbox inbox;
|
ReplyInbox inbox = Fleetd.selectReplyInbox(broker, env, opener);
|
||||||
try {
|
check.run(opener, log.events(), inbox);
|
||||||
inbox = Fleetd.selectReplyInbox(broker, env, opener);
|
|
||||||
} finally {
|
|
||||||
detach(appender);
|
|
||||||
}
|
}
|
||||||
check.run(opener, appender, inbox);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
@FunctionalInterface
|
@FunctionalInterface
|
||||||
private interface Check {
|
private interface Check {
|
||||||
void run(RecordingAmqp opener, ListAppender<ILoggingEvent> appender, ReplyInbox inbox);
|
void run(RecordingAmqp opener, List<ILoggingEvent> events, ReplyInbox inbox);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,286 @@
|
|||||||
|
package dev.ltms.fleet;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.inject.CompletionResolver;
|
||||||
|
import dev.ltms.fleet.inject.ExhaustedPatternLookup;
|
||||||
|
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||||
|
import dev.ltms.fleet.inject.TurnListener;
|
||||||
|
import dev.ltms.fleet.msg.Rendezvous;
|
||||||
|
import dev.ltms.fleet.msg.TurnToken;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.util.LinkedHashSet;
|
||||||
|
import java.util.Set;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #561: {@code Fleetd.turnListener} composes the completion resolver and the session
|
||||||
|
* manager into one {@link TurnListener}. Each of the four callbacks below used to be two bare,
|
||||||
|
* unguarded statements (completion first, then sessions) — a throw from the session half used to
|
||||||
|
* skip nothing <em>after</em> it (there was nothing after it), but nothing enforced that the
|
||||||
|
* completion half had to come first either, beyond call order in the source. {@code onDelivered}
|
||||||
|
* had the identical shape and was fixed by #556 (moving its registration off the fan-out
|
||||||
|
* entirely); these four callbacks cannot be fixed that way, so the fan-out itself is hardened
|
||||||
|
* instead (see {@link Fleetd#turnListener} and {@code Fleetd.bothMustRun}/{@code
|
||||||
|
* bothMustRunKeepingSecondResult}).
|
||||||
|
*
|
||||||
|
* <p>The invariant under test: <b>a throwing session-listener must not prevent the completion
|
||||||
|
* resolver from being told the turn ended.</b> Each test below builds the exact production
|
||||||
|
* composition ({@link Fleetd#turnListener}) from a real {@link CompletionResolver} and a fake
|
||||||
|
* {@code sessions} half that throws, then asserts the completion half's effect (the captured
|
||||||
|
* {@link Rendezvous} waiter resolving) happened anyway — never by inspecting call order directly.
|
||||||
|
*/
|
||||||
|
class FleetdTurnListenerCompositionTest {
|
||||||
|
|
||||||
|
/** Records which callbacks ran and can be told to throw from a chosen one. */
|
||||||
|
private static final class RecordingSessions implements TurnListener {
|
||||||
|
final Set<String> called = new LinkedHashSet<>();
|
||||||
|
private final Set<String> throwing;
|
||||||
|
|
||||||
|
RecordingSessions(String... throwingMethods) {
|
||||||
|
this.throwing = Set.of(throwingMethods);
|
||||||
|
}
|
||||||
|
|
||||||
|
private void maybeThrow(String method) {
|
||||||
|
called.add(method);
|
||||||
|
if (throwing.contains(method)) {
|
||||||
|
throw new IllegalStateException("boom: sessions." + method);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
maybeThrow("onTurnComplete");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean onTurnCompleteWithPostAction(String target) {
|
||||||
|
maybeThrow("onTurnCompleteWithPostAction");
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onTurnFailed(String target) {
|
||||||
|
maybeThrow("onTurnFailed");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onTurnFailed(String target, String reason) {
|
||||||
|
maybeThrow("onTurnFailedWithReason");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static CompletionResolver newResolver(FakeHerdr herdr, Rendezvous rendezvous) {
|
||||||
|
return new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||||
|
ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A resolver with a controllable clock, and the clock itself, so a test can register a turn
|
||||||
|
* "delivered" at time 0 and then jump the clock past {@link CompletionResolver#MIN_TURN_NANOS}
|
||||||
|
* before resolving it — otherwise {@code onTurnComplete}/{@code onTurnCompleteWithPostAction}
|
||||||
|
* resolve within microseconds of registering in-test, well inside the fleetd#164 floor, and
|
||||||
|
* get classified as a too-fast crash rather than a real completion. Mirrors the {@code
|
||||||
|
* LongSupplier} clock-injection pattern fleetd#164's own tests use.
|
||||||
|
*/
|
||||||
|
private static CompletionResolver newResolverPastTheFloor(FakeHerdr herdr, Rendezvous rendezvous,
|
||||||
|
AtomicLong clock) {
|
||||||
|
LongSupplier nowNanos = clock::get;
|
||||||
|
return new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||||
|
ExhaustedPatternLookup.none(), ExhaustionSink.none(), nowNanos);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void onTurnCompleteResolvesTheWaiterEvenWhenTheSessionHalfThrows() throws Exception {
|
||||||
|
FakeHerdr herdr = new FakeHerdr().readText("⏺ real answer\n❯ ");
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
AtomicLong clock = new AtomicLong(0);
|
||||||
|
CompletionResolver completion = newResolverPastTheFloor(herdr, rendezvous, clock);
|
||||||
|
var waiter = rendezvous.open("term_a");
|
||||||
|
completion.register("term_a", new TurnToken("term_a", waiter)); // delivered at clock=0
|
||||||
|
clock.set(CompletionResolver.MIN_TURN_NANOS + 1_000_000); // past the too-fast floor
|
||||||
|
|
||||||
|
RecordingSessions sessions = new RecordingSessions("onTurnComplete");
|
||||||
|
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||||
|
|
||||||
|
IllegalStateException thrown = assertThrows(IllegalStateException.class,
|
||||||
|
() -> composed.onTurnComplete("term_a"),
|
||||||
|
"the session half's throw must still escape the composed listener");
|
||||||
|
assertEquals("boom: sessions.onTurnComplete", thrown.getMessage());
|
||||||
|
assertTrue(sessions.called.contains("onTurnComplete"), "the session half must have run");
|
||||||
|
|
||||||
|
// onTurnComplete resolves off-thread (a virtual thread) — wait on the waiter itself,
|
||||||
|
// exactly like MessageServiceTest.completionFallbackIsNeverQueued does.
|
||||||
|
Rendezvous.Resolution resolution = waiter.get(5, TimeUnit.SECONDS);
|
||||||
|
assertEquals(Rendezvous.Kind.COMPLETION, resolution.kind(),
|
||||||
|
"the completion resolver must still have resolved the send, despite the session "
|
||||||
|
+ "half throwing");
|
||||||
|
assertEquals("real answer", resolution.text());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void onTurnFailedResolvesTheWaiterAsFailedEvenWhenTheSessionHalfThrows() throws Exception {
|
||||||
|
FakeHerdr herdr = new FakeHerdr().readText("⏺ crash context\n❯ ");
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver completion = newResolver(herdr, rendezvous);
|
||||||
|
var waiter = rendezvous.open("term_b");
|
||||||
|
completion.register("term_b", new TurnToken("term_b", waiter));
|
||||||
|
|
||||||
|
RecordingSessions sessions = new RecordingSessions("onTurnFailed");
|
||||||
|
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||||
|
|
||||||
|
IllegalStateException thrown = assertThrows(IllegalStateException.class,
|
||||||
|
() -> composed.onTurnFailed("term_b"),
|
||||||
|
"the session half's throw must still escape the composed listener");
|
||||||
|
assertEquals("boom: sessions.onTurnFailed", thrown.getMessage());
|
||||||
|
assertTrue(sessions.called.contains("onTurnFailed"), "the session half must have run");
|
||||||
|
|
||||||
|
Rendezvous.Resolution resolution = waiter.get(5, TimeUnit.SECONDS);
|
||||||
|
assertEquals(Rendezvous.Kind.FAILED, resolution.kind(),
|
||||||
|
"the completion resolver must still have failed the send, despite the session "
|
||||||
|
+ "half throwing");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void onTurnFailedWithReasonResolvesTheWaiterEvenWhenTheSessionHalfThrows() throws Exception {
|
||||||
|
FakeHerdr herdr = new FakeHerdr();
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver completion = newResolver(herdr, rendezvous);
|
||||||
|
var waiter = rendezvous.open("term_c");
|
||||||
|
completion.register("term_c", new TurnToken("term_c", waiter));
|
||||||
|
|
||||||
|
// fleetd #561: the composed listener calls sessions.onTurnFailed(target) — the ONE-arg
|
||||||
|
// overload — for this reason-carrying event too (matching the pre-existing production
|
||||||
|
// behaviour: SessionManager never overrides the two-arg overload either), so the
|
||||||
|
// throwing key here is "onTurnFailed", not a distinct "...WithReason" one.
|
||||||
|
RecordingSessions sessions = new RecordingSessions("onTurnFailed");
|
||||||
|
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||||
|
|
||||||
|
IllegalStateException thrown = assertThrows(IllegalStateException.class,
|
||||||
|
() -> composed.onTurnFailed("term_c", "worker unreachable"),
|
||||||
|
"the session half's throw must still escape the composed listener");
|
||||||
|
assertEquals("boom: sessions.onTurnFailed", thrown.getMessage());
|
||||||
|
assertTrue(sessions.called.contains("onTurnFailed"), "the session half must have run");
|
||||||
|
|
||||||
|
Rendezvous.Resolution resolution = waiter.get(5, TimeUnit.SECONDS);
|
||||||
|
assertEquals(Rendezvous.Kind.FAILED, resolution.kind(),
|
||||||
|
"the completion resolver must still have failed the send, despite the session "
|
||||||
|
+ "half throwing");
|
||||||
|
assertEquals("worker unreachable", resolution.text(),
|
||||||
|
"the explicit reason must still reach the resolved send");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void onTurnCompleteWithPostActionResolvesTheWaiterEvenWhenTheSessionHalfThrows() {
|
||||||
|
FakeHerdr herdr = new FakeHerdr().readText("⏺ answer before reset\n❯ ");
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
AtomicLong clock = new AtomicLong(0);
|
||||||
|
CompletionResolver completion = newResolverPastTheFloor(herdr, rendezvous, clock);
|
||||||
|
var waiter = rendezvous.open("term_d");
|
||||||
|
completion.register("term_d", new TurnToken("term_d", waiter)); // delivered at clock=0
|
||||||
|
clock.set(CompletionResolver.MIN_TURN_NANOS + 1_000_000); // past the too-fast floor
|
||||||
|
|
||||||
|
RecordingSessions sessions = new RecordingSessions("onTurnCompleteWithPostAction");
|
||||||
|
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||||
|
|
||||||
|
IllegalStateException thrown = assertThrows(IllegalStateException.class,
|
||||||
|
() -> composed.onTurnCompleteWithPostAction("term_d"),
|
||||||
|
"the session half's throw must still escape the composed listener");
|
||||||
|
assertEquals("boom: sessions.onTurnCompleteWithPostAction", thrown.getMessage());
|
||||||
|
assertTrue(sessions.called.contains("onTurnCompleteWithPostAction"), "the session half must have run");
|
||||||
|
|
||||||
|
// resolveBeforePostAction is synchronous by design (it must run before the context reset
|
||||||
|
// can erase the pane) — the waiter is already resolved by the time the throw propagates.
|
||||||
|
assertTrue(waiter.isDone(), "resolveBeforePostAction is synchronous — the send must "
|
||||||
|
+ "already be resolved once the composed call returns (by throwing)");
|
||||||
|
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind());
|
||||||
|
assertEquals("answer before reset", waiter.getNow(null).text());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The mirror case (comment 17041's acceptance item 2): the completion half throws, and the
|
||||||
|
* session half still ran. The production {@link CompletionResolver} is deliberately defensive
|
||||||
|
* (its scrape reads are wrapped in {@code catch (RuntimeException)}, by the same fail-open
|
||||||
|
* design {@link CompletionResolver#captureBaseline} documents) so it essentially never throws
|
||||||
|
* synchronously in normal operation — forcing it to do so needs a genuinely-real trigger, not
|
||||||
|
* a fabricated one. {@code ConcurrentHashMap.get(null)} is that trigger: passing a {@code null}
|
||||||
|
* target makes {@code onTurnComplete}'s {@code inFlight.get(target)} throw a
|
||||||
|
* {@link NullPointerException} before it ever starts its resolving thread — a real code path,
|
||||||
|
* not a contrived one.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void sessionHalfStillRunsWhenTheCompletionHalfThrowsSynchronously() {
|
||||||
|
FakeHerdr herdr = new FakeHerdr();
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver completion = newResolver(herdr, rendezvous);
|
||||||
|
|
||||||
|
RecordingSessions sessions = new RecordingSessions(); // throws from nothing
|
||||||
|
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||||
|
|
||||||
|
assertThrows(NullPointerException.class, () -> composed.onTurnComplete(null),
|
||||||
|
"ConcurrentHashMap.get(null) inside CompletionResolver.onTurnComplete must still "
|
||||||
|
+ "escape the composed listener");
|
||||||
|
assertTrue(sessions.called.contains("onTurnComplete"),
|
||||||
|
"the session half must still have run even though the completion half threw first");
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The mirror case above only exercises {@code onTurnComplete}/{@code bothMustRun}. {@code
|
||||||
|
* onTurnCompleteWithPostAction} is composed through the OTHER helper,
|
||||||
|
* {@code bothMustRunKeepingSecondResult}, and nothing previously asserted that its session half
|
||||||
|
* still runs when its completion half throws — that helper could be reverted to the pre-#561
|
||||||
|
* broken shape (run the session half only if the completion half did not throw) and the suite
|
||||||
|
* would still stay green. Uses the same real, non-fabricated trigger as the test above: a
|
||||||
|
* {@code null} target makes {@code resolveBeforePostAction}'s {@code inFlight.get(target)}
|
||||||
|
* throw a {@link NullPointerException} before {@code resolve} is ever entered.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void sessionHalfStillRunsWhenTheCompletionHalfThrowsSynchronouslyForPostAction() {
|
||||||
|
FakeHerdr herdr = new FakeHerdr();
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver completion = newResolver(herdr, rendezvous);
|
||||||
|
|
||||||
|
RecordingSessions sessions = new RecordingSessions(); // throws from nothing
|
||||||
|
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||||
|
|
||||||
|
assertThrows(NullPointerException.class, () -> composed.onTurnCompleteWithPostAction(null),
|
||||||
|
"ConcurrentHashMap.get(null) inside CompletionResolver.resolveBeforePostAction must "
|
||||||
|
+ "still escape the composed listener");
|
||||||
|
assertTrue(sessions.called.contains("onTurnCompleteWithPostAction"),
|
||||||
|
"the session half must still have run even though the completion half threw first");
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* {@code bothMustRun} must not just run both halves — it must not DROP a second failure when
|
||||||
|
* both halves throw. Forces the completion half to throw (the same real {@code
|
||||||
|
* inFlight.get(null)} NullPointerException trigger used above) while the session half throws a
|
||||||
|
* distinct {@link IllegalStateException}, and asserts the completion half's throwable is what
|
||||||
|
* escapes while the session half's throwable survives as a suppressed exception rather than
|
||||||
|
* being silently discarded.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void bothFailuresEscapeWhenBothHalvesThrowDistinctExceptions() {
|
||||||
|
FakeHerdr herdr = new FakeHerdr();
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver completion = newResolver(herdr, rendezvous);
|
||||||
|
|
||||||
|
RecordingSessions sessions = new RecordingSessions("onTurnComplete");
|
||||||
|
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||||
|
|
||||||
|
NullPointerException thrown = assertThrows(NullPointerException.class,
|
||||||
|
() -> composed.onTurnComplete(null),
|
||||||
|
"the completion half's throw (NPE from inFlight.get(null)) must be what escapes");
|
||||||
|
assertTrue(sessions.called.contains("onTurnComplete"), "the session half must still have run");
|
||||||
|
assertEquals(1, thrown.getSuppressed().length,
|
||||||
|
"the session half's distinct failure must be recorded as suppressed, not dropped");
|
||||||
|
assertEquals(IllegalStateException.class, thrown.getSuppressed()[0].getClass());
|
||||||
|
assertEquals("boom: sessions.onTurnComplete", thrown.getSuppressed()[0].getMessage());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,15 +1,12 @@
|
|||||||
package dev.ltms.fleet.auth;
|
package dev.ltms.fleet.auth;
|
||||||
|
|
||||||
import ch.qos.logback.classic.Level;
|
import ch.qos.logback.classic.Level;
|
||||||
import ch.qos.logback.classic.LoggerContext;
|
|
||||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
|
||||||
import ch.qos.logback.core.read.ListAppender;
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
import org.junit.jupiter.api.AfterEach;
|
import org.junit.jupiter.api.AfterEach;
|
||||||
import org.junit.jupiter.api.BeforeEach;
|
import org.junit.jupiter.api.BeforeEach;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
import org.slf4j.LoggerFactory;
|
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.*;
|
import static org.junit.jupiter.api.Assertions.*;
|
||||||
|
|
||||||
@@ -24,28 +21,21 @@ import static org.junit.jupiter.api.Assertions.*;
|
|||||||
class AuditLogTest {
|
class AuditLogTest {
|
||||||
|
|
||||||
private final ObjectMapper mapper = new ObjectMapper();
|
private final ObjectMapper mapper = new ObjectMapper();
|
||||||
private ListAppender<ILoggingEvent> appender;
|
private CapturedLog auditLog;
|
||||||
private ch.qos.logback.classic.Logger auditLogger;
|
|
||||||
|
|
||||||
@BeforeEach
|
@BeforeEach
|
||||||
void attach() {
|
void attach() {
|
||||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
auditLog = CapturedLog.at("audit", Level.INFO);
|
||||||
auditLogger = ctx.getLogger("audit");
|
|
||||||
appender = new ListAppender<>();
|
|
||||||
appender.setContext(ctx);
|
|
||||||
appender.start();
|
|
||||||
auditLogger.addAppender(appender);
|
|
||||||
auditLogger.setLevel(Level.INFO);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
@AfterEach
|
@AfterEach
|
||||||
void detach() {
|
void detach() {
|
||||||
auditLogger.detachAppender(appender);
|
auditLog.close();
|
||||||
}
|
}
|
||||||
|
|
||||||
private JsonNode onlyRecord() throws Exception {
|
private JsonNode onlyRecord() throws Exception {
|
||||||
assertEquals(1, appender.list.size(), "exactly one audit line expected");
|
assertEquals(1, auditLog.events().size(), "exactly one audit line expected");
|
||||||
String line = appender.list.getFirst().getFormattedMessage();
|
String line = auditLog.events().getFirst().getFormattedMessage();
|
||||||
return mapper.readTree(line); // throws if the line is not valid JSON
|
return mapper.readTree(line); // throws if the line is not valid JSON
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -186,6 +186,43 @@ class CallerResolverTest {
|
|||||||
assertEquals(Role.PRIMARY, r.resolve("127.0.0.1", 99, "BEARER s3cret").role());
|
assertEquals(Role.PRIMARY, r.resolve("127.0.0.1", 99, "BEARER s3cret").role());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ── fleetd #505: a herdr error DURING THE SCAN must not be conflated with "not a worker" ──────
|
||||||
|
// #317 (above) covers a failed lsof lookup. This is the other input to the same decision: the
|
||||||
|
// lsof lookup succeeds (a real pid), but PaneLocator's own pane scan hits a herdr error on the
|
||||||
|
// pane that owns that pid — so c.resolved() is true and c.terminal() is null, exactly like a
|
||||||
|
// real primary. c.scanComplete() is what tells them apart.
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The discriminating case named in the ticket: the error must land on the pane that DOES own
|
||||||
|
* the caller's pid, or the test proves nothing (any other pane's failure is invisible to the
|
||||||
|
* scan's outcome, since a match found elsewhere is definitive regardless).
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void aHerdrErrorOnTheOwningPaneDuringTheScanIsRefusedNotPromotedToPrimary() {
|
||||||
|
FakeHerdr failing = new FakeHerdr().processInfoFailsForPane("w2:p7", "transient");
|
||||||
|
ConnectionIdentity incomplete = new ConnectionIdentity(new PaneLocator(failing), _ -> FakeHerdr.WORKER_PID);
|
||||||
|
|
||||||
|
Principal p = new CallerResolver(incomplete).resolve("127.0.0.1", 55555, null);
|
||||||
|
|
||||||
|
assertEquals(Role.ANONYMOUS, p.role(),
|
||||||
|
"an incomplete pane scan must never be read as a clean negative and promoted to primary");
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The companion invariant: a herdr error on a DIFFERENT, non-owning pane must not turn every
|
||||||
|
* mid-scan teardown into a refusal — the real match is still found and resolves as a worker.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void aHerdrErrorOnANonOwningPaneStillResolvesTheRealWorker() {
|
||||||
|
FakeHerdr vanishedElsewhere = new FakeHerdr().processInfoFailsForPane("w2:p9", "pane_not_found");
|
||||||
|
ConnectionIdentity id = new ConnectionIdentity(new PaneLocator(vanishedElsewhere), _ -> FakeHerdr.WORKER_PID);
|
||||||
|
|
||||||
|
Principal p = new CallerResolver(id).resolve("127.0.0.1", 55555, null);
|
||||||
|
|
||||||
|
assertEquals(Role.WORKER, p.role());
|
||||||
|
assertEquals("term_a", p.terminal());
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void aNonLoopbackCallerIsNeverThePrimaryUnderLoopbackTrust() {
|
void aNonLoopbackCallerIsNeverThePrimaryUnderLoopbackTrust() {
|
||||||
// Defence in depth: startup already refuses this pairing (validateAuthExposure), but if a
|
// Defence in depth: startup already refuses this pairing (validateAuthExposure), but if a
|
||||||
|
|||||||
@@ -345,7 +345,7 @@ class FleetConfigTest {
|
|||||||
IllegalStateException unknownError = assertThrows(IllegalStateException.class,
|
IllegalStateException unknownError = assertThrows(IllegalStateException.class,
|
||||||
() -> FleetConfig.load(unknown).validateCharters());
|
() -> FleetConfig.load(unknown).validateCharters());
|
||||||
assertTrue(unknownError.getMessage().contains("architetc"));
|
assertTrue(unknownError.getMessage().contains("architetc"));
|
||||||
assertTrue(unknownError.getMessage().contains("[architect, dev, reviewer]"));
|
assertTrue(unknownError.getMessage().contains("[architect, dev, hunter, reviewer]"));
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -812,13 +812,17 @@ class FleetConfigTest {
|
|||||||
reviewers:
|
reviewers:
|
||||||
b:
|
b:
|
||||||
profile: sonnet
|
profile: sonnet
|
||||||
|
hunters:
|
||||||
|
c:
|
||||||
|
profile: sonnet
|
||||||
""");
|
""");
|
||||||
|
|
||||||
FleetConfig cfg = FleetConfig.load(f);
|
FleetConfig cfg = FleetConfig.load(f);
|
||||||
assertEquals(List.of("sonnet"), cfg.fleet().profilesFor(MemberRole.DEV));
|
assertEquals(List.of("sonnet"), cfg.fleet().profilesFor(MemberRole.DEV));
|
||||||
|
assertEquals(List.of("sonnet"), cfg.fleet().profilesFor(MemberRole.HUNTER));
|
||||||
assertEquals(List.of("sonnet"), cfg.fleet().profilesFor(MemberRole.REVIEWER));
|
assertEquals(List.of("sonnet"), cfg.fleet().profilesFor(MemberRole.REVIEWER));
|
||||||
assertTrue(cfg.fleet().profilesFor(MemberRole.ARCHITECT).isEmpty());
|
assertTrue(cfg.fleet().profilesFor(MemberRole.ARCHITECT).isEmpty());
|
||||||
assertEquals(List.of(MemberRole.DEV, MemberRole.REVIEWER), cfg.fleet().rolesConfigured());
|
assertEquals(List.of(MemberRole.DEV, MemberRole.HUNTER, MemberRole.REVIEWER), cfg.fleet().rolesConfigured());
|
||||||
}
|
}
|
||||||
|
|
||||||
/** The case the two axes exist for: one backend, two roles, and neither is a duplicate. */
|
/** The case the two axes exist for: one backend, two roles, and neither is a duplicate. */
|
||||||
|
|||||||
@@ -0,0 +1,40 @@
|
|||||||
|
package dev.ltms.fleet.health;
|
||||||
|
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #426: {@link FleetHealthMonitor#coverage} had zero references anywhere in the test tree —
|
||||||
|
* not the method, not either output string, not the field it populates. This pins the method's own
|
||||||
|
* three branches directly.
|
||||||
|
*
|
||||||
|
* <p>This is the easy half. It proves the method words each combination correctly, but it proves
|
||||||
|
* nothing about whether {@code Fleetd.java}'s two call sites pass the right argument in the right
|
||||||
|
* position — see {@code FleetdHealthCoverageSourceWiringTest} (package {@code dev.ltms.fleet}) for
|
||||||
|
* the half that actually guards the call site, following the measured fleetd #415 lesson that a
|
||||||
|
* thoroughly-tested method and an untested argument pairing at its call site are different risks.
|
||||||
|
*
|
||||||
|
* <p><strong>The three output strings are load-bearing and must not change here.</strong> {@code
|
||||||
|
* "detection-only"} is read live off a running daemon's {@code fleet_list} today (measured
|
||||||
|
* 2026-09-12) — this test intentionally asserts the exact literal strings so a future edit to the
|
||||||
|
* wording trips it here first.
|
||||||
|
*/
|
||||||
|
class FleetHealthMonitorCoverageTest {
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void disabledIsOffRegardlessOfNotificationConfig() {
|
||||||
|
assertEquals("off", FleetHealthMonitor.coverage(false, false));
|
||||||
|
assertEquals("off", FleetHealthMonitor.coverage(false, true));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void enabledWithoutNotificationsIsDetectionOnly() {
|
||||||
|
assertEquals("detection-only", FleetHealthMonitor.coverage(true, false));
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void enabledWithNotificationsIsFull() {
|
||||||
|
assertEquals("full", FleetHealthMonitor.coverage(true, true));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -44,6 +44,7 @@ public final class FakeHerdr implements HerdrClient {
|
|||||||
private int workerTabPaneCount = 1;
|
private int workerTabPaneCount = 1;
|
||||||
private String paneCloseErrorCode = null;
|
private String paneCloseErrorCode = null;
|
||||||
private final Map<String, String> paneCloseErrorCodeFor = new ConcurrentHashMap<>();
|
private final Map<String, String> paneCloseErrorCodeFor = new ConcurrentHashMap<>();
|
||||||
|
private final Map<String, String> processInfoErrorCodeFor = new ConcurrentHashMap<>();
|
||||||
private String tabCloseErrorCode = null;
|
private String tabCloseErrorCode = null;
|
||||||
private final Map<String, String> tabCloseErrorCodeFor = new ConcurrentHashMap<>();
|
private final Map<String, String> tabCloseErrorCodeFor = new ConcurrentHashMap<>();
|
||||||
private String agentSendErrorCode = null;
|
private String agentSendErrorCode = null;
|
||||||
@@ -144,6 +145,18 @@ public final class FakeHerdr implements HerdrClient {
|
|||||||
return this;
|
return this;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Make {@code pane.process_info} fail with this herdr error code, but only for the given
|
||||||
|
* {@code pane_id} — every other pane's {@code pane.process_info} still succeeds. Models a
|
||||||
|
* transient herdr failure partway through a {@link PaneLocator} pid→pane scan (fleetd #505):
|
||||||
|
* the scan must be able to tell "this pane does not own the pid" apart from "the scan could
|
||||||
|
* not check this pane at all", instead of collapsing both into one {@code false}.
|
||||||
|
*/
|
||||||
|
public FakeHerdr processInfoFailsForPane(String paneId, String code) {
|
||||||
|
this.processInfoErrorCodeFor.put(paneId, code);
|
||||||
|
return this;
|
||||||
|
}
|
||||||
|
|
||||||
/** Set the {@code agent_status} that {@code agent.get} reports (drives the injector). */
|
/** Set the {@code agent_status} that {@code agent.get} reports (drives the injector). */
|
||||||
public FakeHerdr agentStatus(String status) {
|
public FakeHerdr agentStatus(String status) {
|
||||||
this.agentStatus = status;
|
this.agentStatus = status;
|
||||||
@@ -407,6 +420,13 @@ public final class FakeHerdr implements HerdrClient {
|
|||||||
{"pane_id":"w2:p9","terminal_id":"term_shell","workspace_id":"w2","tab_id":"w2:t8"}]}""");
|
{"pane_id":"w2:p9","terminal_id":"term_shell","workspace_id":"w2","tab_id":"w2:t8"}]}""");
|
||||||
case "pane.process_info" -> {
|
case "pane.process_info" -> {
|
||||||
Object paneId = params instanceof java.util.Map<?, ?> m ? m.get("pane_id") : null;
|
Object paneId = params instanceof java.util.Map<?, ?> m ? m.get("pane_id") : null;
|
||||||
|
String failCode = paneId == null ? null
|
||||||
|
: processInfoErrorCodeFor.get(String.valueOf(paneId));
|
||||||
|
if (failCode != null) {
|
||||||
|
throw new HerdrException(
|
||||||
|
"herdr error [" + failCode + "]: pane.process_info failed",
|
||||||
|
failCode, null);
|
||||||
|
}
|
||||||
yield "w2:p7".equals(paneId)
|
yield "w2:p7".equals(paneId)
|
||||||
? mapper.readTree(("""
|
? mapper.readTree(("""
|
||||||
{"type":"pane_process_info","process_info":{"pane_id":"w2:p7","shell_pid":%d,
|
{"type":"pane_process_info","process_info":{"pane_id":"w2:p7","shell_pid":%d,
|
||||||
|
|||||||
@@ -37,7 +37,7 @@ class PaneLocatorContractTest {
|
|||||||
.path("pane").path("terminal_id").asText(null);
|
.path("pane").path("terminal_id").asText(null);
|
||||||
assertNotNull(terminalId, "seed pane should carry a terminal_id");
|
assertNotNull(terminalId, "seed pane should carry a terminal_id");
|
||||||
|
|
||||||
assertEquals(terminalId, new PaneLocator(herdr).terminalForPid(shellPid),
|
assertEquals(terminalId, new PaneLocator(herdr).terminalForPid(shellPid).terminal(),
|
||||||
"a real PID must resolve back to its own pane's terminal_id");
|
"a real PID must resolve back to its own pane's terminal_id");
|
||||||
} finally {
|
} finally {
|
||||||
spaces.closeTab(tab.tab().tabId());
|
spaces.closeTab(tab.tab().tabId());
|
||||||
|
|||||||
@@ -15,18 +15,20 @@ class PaneLocatorTest {
|
|||||||
|
|
||||||
@Test
|
@Test
|
||||||
void resolvesTerminalForAForegroundPid() {
|
void resolvesTerminalForAForegroundPid() {
|
||||||
assertEquals("term_a", loc.terminalForPid(FakeHerdr.WORKER_PID));
|
assertEquals("term_a", loc.terminalForPid(FakeHerdr.WORKER_PID).terminal());
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void nullForAPidInNoPane() {
|
void nullForAPidInNoPane() {
|
||||||
assertNull(loc.terminalForPid(999_999));
|
PaneLocator.Lookup outcome = loc.terminalForPid(999_999);
|
||||||
|
assertNull(outcome.terminal());
|
||||||
|
assertTrue(outcome.complete(), "a full, error-free scan that finds no match is complete");
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void nullForNonPositivePid() {
|
void nullForNonPositivePid() {
|
||||||
assertNull(loc.terminalForPid(0));
|
assertNull(loc.terminalForPid(0).terminal());
|
||||||
assertNull(loc.terminalForPid(-1));
|
assertNull(loc.terminalForPid(-1).terminal());
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- two-daemon fallback (CB-185) -----------------------------------------
|
// --- two-daemon fallback (CB-185) -----------------------------------------
|
||||||
@@ -38,7 +40,7 @@ class PaneLocatorTest {
|
|||||||
HerdrClient lead = new FakeHerdr().withNoPanes();
|
HerdrClient lead = new FakeHerdr().withNoPanes();
|
||||||
HerdrClient member = new FakeHerdr();
|
HerdrClient member = new FakeHerdr();
|
||||||
PaneLocator two = new PaneLocator(lead, member);
|
PaneLocator two = new PaneLocator(lead, member);
|
||||||
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID));
|
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID).terminal());
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -48,13 +50,13 @@ class PaneLocatorTest {
|
|||||||
HerdrClient lead = new FakeHerdr();
|
HerdrClient lead = new FakeHerdr();
|
||||||
HerdrClient member = new FakeHerdr().withNoPanes();
|
HerdrClient member = new FakeHerdr().withNoPanes();
|
||||||
PaneLocator two = new PaneLocator(lead, member);
|
PaneLocator two = new PaneLocator(lead, member);
|
||||||
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID));
|
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID).terminal());
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void nullWhenNeitherClientHasTheMatch() {
|
void nullWhenNeitherClientHasTheMatch() {
|
||||||
PaneLocator two = new PaneLocator(new FakeHerdr().withNoPanes(), new FakeHerdr().withNoPanes());
|
PaneLocator two = new PaneLocator(new FakeHerdr().withNoPanes(), new FakeHerdr().withNoPanes());
|
||||||
assertNull(two.terminalForPid(FakeHerdr.WORKER_PID));
|
assertNull(two.terminalForPid(FakeHerdr.WORKER_PID).terminal());
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -63,7 +65,7 @@ class PaneLocatorTest {
|
|||||||
// must behave exactly like the one-arg constructor, including making only one herdr call.
|
// must behave exactly like the one-arg constructor, including making only one herdr call.
|
||||||
FakeHerdr shared = new FakeHerdr();
|
FakeHerdr shared = new FakeHerdr();
|
||||||
PaneLocator two = new PaneLocator(shared, shared);
|
PaneLocator two = new PaneLocator(shared, shared);
|
||||||
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID));
|
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID).terminal());
|
||||||
long paneListCalls = shared.calls.stream().filter(c -> c.method().equals("pane.list")).count();
|
long paneListCalls = shared.calls.stream().filter(c -> c.method().equals("pane.list")).count();
|
||||||
assertEquals(1, paneListCalls, "same-object lead/member must scan exactly once, not twice");
|
assertEquals(1, paneListCalls, "same-object lead/member must scan exactly once, not twice");
|
||||||
}
|
}
|
||||||
@@ -75,7 +77,7 @@ class PaneLocatorTest {
|
|||||||
// Regression: a pid with no parent chain at all — no ancestry walk is needed to match it.
|
// Regression: a pid with no parent chain at all — no ancestry walk is needed to match it.
|
||||||
OnePaneHerdr pane = new OnePaneHerdr("term_x", "pX", 5000, 6000);
|
OnePaneHerdr pane = new OnePaneHerdr("term_x", "pX", 5000, 6000);
|
||||||
PaneLocator loc = new PaneLocator(pane, new FakeParentResolver());
|
PaneLocator loc = new PaneLocator(pane, new FakeParentResolver());
|
||||||
assertEquals("term_x", loc.terminalForPid(5000));
|
assertEquals("term_x", loc.terminalForPid(5000).terminal());
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -83,7 +85,7 @@ class PaneLocatorTest {
|
|||||||
// Regression: same as above, but matching via the foreground-processes list.
|
// Regression: same as above, but matching via the foreground-processes list.
|
||||||
OnePaneHerdr pane = new OnePaneHerdr("term_x", "pX", 5000, 6000);
|
OnePaneHerdr pane = new OnePaneHerdr("term_x", "pX", 5000, 6000);
|
||||||
PaneLocator loc = new PaneLocator(pane, new FakeParentResolver());
|
PaneLocator loc = new PaneLocator(pane, new FakeParentResolver());
|
||||||
assertEquals("term_x", loc.terminalForPid(6000));
|
assertEquals("term_x", loc.terminalForPid(6000).terminal());
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -97,7 +99,7 @@ class PaneLocatorTest {
|
|||||||
.parent(7002, 7001) // grandchild -> child
|
.parent(7002, 7001) // grandchild -> child
|
||||||
.parent(7001, 5000); // child -> shell (the pane's shell_pid)
|
.parent(7001, 5000); // child -> shell (the pane's shell_pid)
|
||||||
PaneLocator loc = new PaneLocator(pane, parents);
|
PaneLocator loc = new PaneLocator(pane, parents);
|
||||||
assertEquals("term_x", loc.terminalForPid(7002));
|
assertEquals("term_x", loc.terminalForPid(7002).terminal());
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -110,7 +112,7 @@ class PaneLocatorTest {
|
|||||||
.parent(9002, 9001)
|
.parent(9002, 9001)
|
||||||
.parent(9001, 9000); // chain never reaches 5000 or 6000
|
.parent(9001, 9000); // chain never reaches 5000 or 6000
|
||||||
PaneLocator loc = new PaneLocator(pane, parents);
|
PaneLocator loc = new PaneLocator(pane, parents);
|
||||||
assertNull(loc.terminalForPid(9002));
|
assertNull(loc.terminalForPid(9002).terminal());
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -122,7 +124,7 @@ class PaneLocatorTest {
|
|||||||
.parent(100, 101)
|
.parent(100, 101)
|
||||||
.parent(101, 100); // cycle, never reaches the pane's pids
|
.parent(101, 100); // cycle, never reaches the pane's pids
|
||||||
PaneLocator loc = new PaneLocator(pane, parents);
|
PaneLocator loc = new PaneLocator(pane, parents);
|
||||||
assertNull(loc.terminalForPid(100));
|
assertNull(loc.terminalForPid(100).terminal());
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -141,11 +143,69 @@ class PaneLocatorTest {
|
|||||||
};
|
};
|
||||||
HerdrClient noPanes = new FakeHerdr().withNoPanes();
|
HerdrClient noPanes = new FakeHerdr().withNoPanes();
|
||||||
PaneLocator two = new PaneLocator(noPanes, pane, counting);
|
PaneLocator two = new PaneLocator(noPanes, pane, counting);
|
||||||
assertEquals("term_x", two.terminalForPid(7002));
|
assertEquals("term_x", two.terminalForPid(7002).terminal());
|
||||||
assertEquals(3, calls.get(), "ancestry must be walked once (3 lookups: 7002, 7001, 5000), "
|
assertEquals(3, calls.get(), "ancestry must be walked once (3 lookups: 7002, 7001, 5000), "
|
||||||
+ "not re-walked per herdr client");
|
+ "not re-walked per herdr client");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- fleetd #505: a herdr error during the scan must not read as a clean negative ---------
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void anErrorOnThePaneThatOwnsThePidMakesTheScanIncompleteNotAClearNegative() {
|
||||||
|
// The discriminating case: pane.process_info fails for exactly the pane that DOES own the
|
||||||
|
// caller's pid ("w2:p7", term_a). Before the fix, that failure was swallowed into a plain
|
||||||
|
// "does not own it" and the scan finished with a clean-looking null — indistinguishable
|
||||||
|
// from a real primary. It must now report incomplete, not a definite null.
|
||||||
|
FakeHerdr herdr = new FakeHerdr().processInfoFailsForPane("w2:p7", "transient");
|
||||||
|
PaneLocator loc = new PaneLocator(herdr);
|
||||||
|
|
||||||
|
PaneLocator.Lookup outcome = loc.terminalForPid(FakeHerdr.WORKER_PID);
|
||||||
|
|
||||||
|
assertNull(outcome.terminal(), "the failing pane's ownership could not be confirmed");
|
||||||
|
assertFalse(outcome.complete(),
|
||||||
|
"a scan that could not check the owning pane must not report as complete");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aVanishedPaneThatIsNotTheMatchLeavesAnOtherwiseSuccessfulScanComplete() {
|
||||||
|
// The companion invariant: a DIFFERENT pane (not the caller's own) failing mid-scan must
|
||||||
|
// not turn every mid-scan teardown into a refusal — the real match is still found, and the
|
||||||
|
// scan is still reported complete.
|
||||||
|
FakeHerdr herdr = new FakeHerdr().processInfoFailsForPane("w2:p9", "pane_not_found");
|
||||||
|
PaneLocator loc = new PaneLocator(herdr);
|
||||||
|
|
||||||
|
PaneLocator.Lookup outcome = loc.terminalForPid(FakeHerdr.WORKER_PID);
|
||||||
|
|
||||||
|
assertEquals("term_a", outcome.terminal());
|
||||||
|
assertTrue(outcome.complete(), "a positive match elsewhere in the scan is definitive");
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- fleetd #509: the completeness fold across clients must not collapse to "last wins" ----
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void anEarlierClientsErrorSurvivesALaterClientsCleanNegative() {
|
||||||
|
// terminalForPid folds each client's Lookup.complete() with
|
||||||
|
// complete = complete && outcome.complete();
|
||||||
|
// (PaneLocator.java:117). With a SINGLE client, a fold that keeps only the last outcome
|
||||||
|
// (dropping the "complete &&" prefix) agrees with the real fold — which is why 14 of the
|
||||||
|
// 15 pre-existing tests never catch that mutation: none of them vary the number of clients.
|
||||||
|
// Here the LEAD client errors on exactly the pane that would have owned the pid (so its
|
||||||
|
// scan is incomplete AND finds no match), and the MEMBER client cleanly reports no panes
|
||||||
|
// at all (a complete, negative scan). The real fold ANDs the two into false. A fold that
|
||||||
|
// just keeps the last client's outcome would read this as a clean true — the earlier
|
||||||
|
// error is erased, and CallerResolver.java:254 would read scanComplete() as true and
|
||||||
|
// promote an unverified caller to the primary.
|
||||||
|
HerdrClient lead = new FakeHerdr().processInfoFailsForPane("w2:p7", "transient");
|
||||||
|
HerdrClient member = new FakeHerdr().withNoPanes();
|
||||||
|
PaneLocator two = new PaneLocator(lead, member);
|
||||||
|
|
||||||
|
PaneLocator.Lookup outcome = two.terminalForPid(FakeHerdr.WORKER_PID);
|
||||||
|
|
||||||
|
assertNull(outcome.terminal(), "the pane that could have owned the pid was never checked");
|
||||||
|
assertFalse(outcome.complete(),
|
||||||
|
"an earlier client's error must survive a later client's clean negative");
|
||||||
|
}
|
||||||
|
|
||||||
/** Minimal single-pane {@link HerdrClient} fake, purpose-built for the ancestry tests above. */
|
/** Minimal single-pane {@link HerdrClient} fake, purpose-built for the ancestry tests above. */
|
||||||
private static final class OnePaneHerdr implements HerdrClient {
|
private static final class OnePaneHerdr implements HerdrClient {
|
||||||
private final ObjectMapper mapper = new ObjectMapper();
|
private final ObjectMapper mapper = new ObjectMapper();
|
||||||
|
|||||||
@@ -1,9 +1,7 @@
|
|||||||
package dev.ltms.fleet.inject;
|
package dev.ltms.fleet.inject;
|
||||||
|
|
||||||
import ch.qos.logback.classic.Level;
|
import ch.qos.logback.classic.Level;
|
||||||
import ch.qos.logback.classic.LoggerContext;
|
|
||||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
import ch.qos.logback.core.read.ListAppender;
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||||
import dev.ltms.fleet.herdr.AgentControl;
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
@@ -12,8 +10,8 @@ import dev.ltms.fleet.herdr.HerdrClient;
|
|||||||
import dev.ltms.fleet.msg.Rendezvous;
|
import dev.ltms.fleet.msg.Rendezvous;
|
||||||
import dev.ltms.fleet.msg.TestTurnTokens;
|
import dev.ltms.fleet.msg.TestTurnTokens;
|
||||||
import dev.ltms.fleet.msg.TurnToken;
|
import dev.ltms.fleet.msg.TurnToken;
|
||||||
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
import org.slf4j.LoggerFactory;
|
|
||||||
|
|
||||||
import java.util.Set;
|
import java.util.Set;
|
||||||
import java.util.regex.Pattern;
|
import java.util.regex.Pattern;
|
||||||
@@ -21,6 +19,7 @@ import java.util.regex.Pattern;
|
|||||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
/** Unit behaviour of the CB-106 completion resolver in isolation from the injector. */
|
/** Unit behaviour of the CB-106 completion resolver in isolation from the injector. */
|
||||||
@@ -470,15 +469,7 @@ class CompletionResolverTest {
|
|||||||
// CB-564: this used to be a bare DEBUG "failed send to X via turn-stall fallback" — a symptom
|
// CB-564: this used to be a bare DEBUG "failed send to X via turn-stall fallback" — a symptom
|
||||||
// with no cause, and below the level anyone watching for member health would see. A fail that
|
// with no cause, and below the level anyone watching for member health would see. A fail that
|
||||||
// resolves a caller's blocked send is at least WARN and must carry the reason.
|
// resolves a caller's blocked send is at least WARN and must carry the reason.
|
||||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
try (CapturedLog log = CapturedLog.at(CompletionResolver.class, Level.WARN)) {
|
||||||
ch.qos.logback.classic.Logger resolverLog =
|
|
||||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(CompletionResolver.class);
|
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.setContext(ctx);
|
|
||||||
appender.start();
|
|
||||||
resolverLog.addAppender(appender);
|
|
||||||
resolverLog.setLevel(Level.WARN);
|
|
||||||
try {
|
|
||||||
FakeHerdr herdr = new FakeHerdr().readText("stuck on an error screen");
|
FakeHerdr herdr = new FakeHerdr().readText("stuck on an error screen");
|
||||||
Rendezvous rendezvous = new Rendezvous();
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||||
@@ -486,7 +477,7 @@ class CompletionResolverTest {
|
|||||||
|
|
||||||
resolver.fail("term_a", null);
|
resolver.fail("term_a", null);
|
||||||
|
|
||||||
String warn = appender.list.stream()
|
String warn = log.events().stream()
|
||||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||||
.map(ILoggingEvent::getFormattedMessage)
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
.findFirst()
|
.findFirst()
|
||||||
@@ -494,8 +485,6 @@ class CompletionResolverTest {
|
|||||||
assertTrue(warn.contains("term_a"), "the log names the target: " + warn);
|
assertTrue(warn.contains("term_a"), "the log names the target: " + warn);
|
||||||
assertTrue(warn.contains("stuck on an error screen"), "the log carries the reason: " + warn);
|
assertTrue(warn.contains("stuck on an error screen"), "the log carries the reason: " + warn);
|
||||||
assertTrue(waiter.isDone());
|
assertTrue(waiter.isDone());
|
||||||
} finally {
|
|
||||||
resolverLog.detachAppender(appender);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -531,6 +520,254 @@ class CompletionResolverTest {
|
|||||||
assertTrue(rendezvous.isWaiting("term_a"), "turn N+1 is still awaiting its own resolution");
|
assertTrue(rendezvous.isWaiting("term_a"), "turn N+1 is still awaiting its own resolution");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- fleetd #556 (comment 16908): name the invariant the two-arg remove(target, turn) idiom
|
||||||
|
// exists to maintain — a superseded turn's terminal handling must not evict its successor's
|
||||||
|
// registration — and test THAT, not the idiom's literal spelling. Three entry points reach a
|
||||||
|
// conditional remove: resolve()'s plain completion path, resolve()'s echoed-brief/noReportMessage
|
||||||
|
// sub-path, and fail(). This is a non-goal-to-break for fleetd #556's redesign: flattening any of
|
||||||
|
// these removes to the one-arg form must turn one of these three red. -------------------------
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aSupersededTurnsPlainCompletionMustNotEvictItsSuccessorsRegistration() {
|
||||||
|
// Turn A is overtaken by turn B on the same target (a rapid back-to-back send): B's own
|
||||||
|
// delivery overwrites the map entry A's delivery installed. A's late completion fallback
|
||||||
|
// then finally fires and reaches resolve()'s ordinary (non-echoed) completion branch — the
|
||||||
|
// two-arg remove(target, turnA) there must be a no-op once the map holds B, not a blind
|
||||||
|
// evict.
|
||||||
|
FakeHerdr herdr = new FakeHerdr().readText("⏺ A's pre-turn pane\n❯ ");
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||||
|
ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||||
|
|
||||||
|
var waiterA = rendezvous.open("term_a");
|
||||||
|
var turnA = new CompletionResolver.InFlight(waiterA, null); // baseline null: fails open, never suppressed
|
||||||
|
rendezvous.close("term_a", waiterA); // A's own send's finally, freeing the session for B
|
||||||
|
var waiterB = rendezvous.open("term_a");
|
||||||
|
resolver.captureBaseline("term_a", new TurnToken("term_a", waiterB)); // the injector registering B
|
||||||
|
|
||||||
|
herdr.readText("⏺ A's late, genuine answer\n❯ "); // what the pane shows when A's fallback fires
|
||||||
|
resolver.resolve("term_a", turnA);
|
||||||
|
|
||||||
|
assertTrue(waiterA.isDone(), "A's own resolve must still complete A's waiter normally");
|
||||||
|
assertEquals(Rendezvous.Kind.COMPLETION, waiterA.getNow(null).kind());
|
||||||
|
|
||||||
|
CompletionResolver.InFlight afterA = resolver.inFlight("term_a");
|
||||||
|
assertNotNull(afterA, "B's registration must still be present — a one-arg remove(target) "
|
||||||
|
+ "here would evict it even though the map no longer holds turnA");
|
||||||
|
assertEquals(waiterB, afterA.waiter(), "the surviving entry must still be B's, untouched by A's "
|
||||||
|
+ "terminal handling");
|
||||||
|
assertTrue(rendezvous.resolve("term_a", "B replied"),
|
||||||
|
"B must still resolve normally through the ordinary reply path");
|
||||||
|
assertEquals(Rendezvous.Kind.REPLY, waiterB.getNow(null).kind());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aSupersededTurnsEchoedNoReportSubPathMustNotEvictItsSuccessorsRegistration() {
|
||||||
|
// Same invariant, driven down resolve()'s OTHER branch: a pane that merely echoes the
|
||||||
|
// injected brief back (no real report) resolves through noReportMessage(), not the plain
|
||||||
|
// tail — comment 16908 calls this "a sub-path of the first, not a separate method", reaching
|
||||||
|
// the same terminal remove() call site from different logic above it.
|
||||||
|
String echoedBrief = "y".repeat(450); // >= CompletionResolver.ECHO_MIN_CHARS normalised chars
|
||||||
|
FakeHerdr herdr = new FakeHerdr().readText("⏺ " + echoedBrief + "\n❯ ");
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||||
|
ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||||
|
|
||||||
|
var waiterA = rendezvous.open("term_a");
|
||||||
|
// Back-dated deliveredAtNanos (like the 2-arg InFlight convenience ctor) so MIN_TURN_NANOS
|
||||||
|
// never fires; injectedText == the pane's own content so echoesInjectedBrief() is true.
|
||||||
|
var turnA = new CompletionResolver.InFlight(waiterA, null, Long.MIN_VALUE / 2, echoedBrief);
|
||||||
|
rendezvous.close("term_a", waiterA);
|
||||||
|
var waiterB = rendezvous.open("term_a");
|
||||||
|
resolver.captureBaseline("term_a", new TurnToken("term_a", waiterB));
|
||||||
|
|
||||||
|
resolver.resolve("term_a", turnA);
|
||||||
|
|
||||||
|
assertTrue(waiterA.isDone(), "A's own resolve must still complete A's waiter via the echoed "
|
||||||
|
+ "no-report branch");
|
||||||
|
assertTrue(waiterA.getNow(null).text().contains(CompletionResolver.NO_REPORT_PREFIX),
|
||||||
|
"sanity: A must actually have gone down the noReportMessage sub-path, not the plain one");
|
||||||
|
|
||||||
|
CompletionResolver.InFlight afterA = resolver.inFlight("term_a");
|
||||||
|
assertNotNull(afterA, "B's registration must still be present after A's echoed-brief resolve");
|
||||||
|
assertEquals(waiterB, afterA.waiter());
|
||||||
|
assertTrue(rendezvous.resolve("term_a", "B replied"),
|
||||||
|
"B must still resolve normally through the ordinary reply path");
|
||||||
|
assertEquals(Rendezvous.Kind.REPLY, waiterB.getNow(null).kind());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aSupersededTurnsFailMustNotEvictItsSuccessorsRegistration() {
|
||||||
|
// The fail() entry point (CB-109 wedge / CB-110 drop): turn A is failed while turn B has
|
||||||
|
// already superseded it in the map. fail()'s own two-arg remove(target, turnA) must be a
|
||||||
|
// no-op once the map holds B.
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver resolver = new CompletionResolver(new AgentControl(new FakeHerdr()), rendezvous,
|
||||||
|
ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||||
|
|
||||||
|
var waiterA = rendezvous.open("term_a");
|
||||||
|
var turnA = new CompletionResolver.InFlight(waiterA, null);
|
||||||
|
rendezvous.close("term_a", waiterA);
|
||||||
|
var waiterB = rendezvous.open("term_a");
|
||||||
|
resolver.captureBaseline("term_a", new TurnToken("term_a", waiterB));
|
||||||
|
|
||||||
|
resolver.fail("term_a", turnA, "A wedged in an unknown state");
|
||||||
|
|
||||||
|
assertTrue(waiterA.isDone(), "A's own fail must still complete A's waiter");
|
||||||
|
assertEquals(Rendezvous.Kind.FAILED, waiterA.getNow(null).kind());
|
||||||
|
|
||||||
|
CompletionResolver.InFlight afterA = resolver.inFlight("term_a");
|
||||||
|
assertNotNull(afterA, "B's registration must still be present — a one-arg remove(target) in "
|
||||||
|
+ "fail() would evict it even though the map no longer holds turnA");
|
||||||
|
assertEquals(waiterB, afterA.waiter());
|
||||||
|
assertTrue(rendezvous.resolve("term_a", "B replied"),
|
||||||
|
"B must still resolve normally through the ordinary reply path");
|
||||||
|
assertEquals(Rendezvous.Kind.REPLY, waiterB.getNow(null).kind());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aSupersededDoneTurnMustNotEvictItsSuccessorsRegistration() {
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver resolver = new CompletionResolver(new AgentControl(new FakeHerdr()), rendezvous,
|
||||||
|
ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||||
|
|
||||||
|
var waiterA = rendezvous.open("term_a");
|
||||||
|
var turnA = new CompletionResolver.InFlight(waiterA, null);
|
||||||
|
assertTrue(rendezvous.resolve("term_a", "A replied"));
|
||||||
|
rendezvous.close("term_a", waiterA);
|
||||||
|
var waiterB = rendezvous.open("term_a");
|
||||||
|
resolver.captureBaseline("term_a", new TurnToken("term_a", waiterB));
|
||||||
|
|
||||||
|
resolver.resolve("term_a", turnA);
|
||||||
|
|
||||||
|
assertSuccessorRegistrationSurvives(resolver, rendezvous, waiterB,
|
||||||
|
"a done turn must not evict B from resolve()'s early return");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aSupersededExhaustedTurnMustNotEvictItsSuccessorsRegistration() {
|
||||||
|
FakeHerdr herdr = new FakeHerdr().readText("⏺ usage limit has been reached\n❯ ");
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||||
|
target -> Pattern.compile("usage limit has been reached"), ExhaustionSink.none());
|
||||||
|
|
||||||
|
var waiterA = rendezvous.open("term_a");
|
||||||
|
var turnA = new CompletionResolver.InFlight(waiterA, null);
|
||||||
|
rendezvous.close("term_a", waiterA);
|
||||||
|
var waiterB = rendezvous.open("term_a");
|
||||||
|
resolver.captureBaseline("term_a", new TurnToken("term_a", waiterB));
|
||||||
|
|
||||||
|
resolver.resolve("term_a", turnA);
|
||||||
|
|
||||||
|
assertSuccessorRegistrationSurvives(resolver, rendezvous, waiterB,
|
||||||
|
"an exhausted turn must not evict B from resolve()'s exhausted branch");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aSupersededBackendErrorTurnMustNotEvictItsSuccessorsRegistration() {
|
||||||
|
FakeHerdr herdr = new FakeHerdr().readText("⏺ API Error: 400 invalid request body\n❯ ");
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||||
|
ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||||
|
|
||||||
|
var waiterA = rendezvous.open("term_a");
|
||||||
|
var turnA = new CompletionResolver.InFlight(waiterA, null);
|
||||||
|
rendezvous.close("term_a", waiterA);
|
||||||
|
var waiterB = rendezvous.open("term_a");
|
||||||
|
resolver.captureBaseline("term_a", new TurnToken("term_a", waiterB));
|
||||||
|
|
||||||
|
resolver.resolve("term_a", turnA);
|
||||||
|
|
||||||
|
assertSuccessorRegistrationSurvives(resolver, rendezvous, waiterB,
|
||||||
|
"a backend-error turn must not evict B from resolve()'s error branch");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aSupersededRawExhaustedTurnMustNotEvictItsSuccessorsRegistration() {
|
||||||
|
FakeHerdr herdr = new FakeHerdr().readText("╭────\nusage limit has been reached");
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||||
|
target -> Pattern.compile("usage limit has been reached"), ExhaustionSink.none());
|
||||||
|
|
||||||
|
var waiterA = rendezvous.open("term_a");
|
||||||
|
var turnA = new CompletionResolver.InFlight(waiterA, null);
|
||||||
|
rendezvous.close("term_a", waiterA);
|
||||||
|
var waiterB = rendezvous.open("term_a");
|
||||||
|
resolver.captureBaseline("term_a", new TurnToken("term_a", waiterB));
|
||||||
|
|
||||||
|
resolver.resolve("term_a", turnA);
|
||||||
|
|
||||||
|
assertSuccessorRegistrationSurvives(resolver, rendezvous, waiterB,
|
||||||
|
"a raw exhausted turn must not evict B from the raw-scrape exhausted branch");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aSupersededRawBackendErrorTurnMustNotEvictItsSuccessorsRegistration() {
|
||||||
|
FakeHerdr herdr = new FakeHerdr().readText("╭────\nAPI Error: 400 invalid request body");
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||||
|
ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||||
|
|
||||||
|
var waiterA = rendezvous.open("term_a");
|
||||||
|
var turnA = new CompletionResolver.InFlight(waiterA, null);
|
||||||
|
rendezvous.close("term_a", waiterA);
|
||||||
|
var waiterB = rendezvous.open("term_a");
|
||||||
|
resolver.captureBaseline("term_a", new TurnToken("term_a", waiterB));
|
||||||
|
|
||||||
|
resolver.resolve("term_a", turnA);
|
||||||
|
|
||||||
|
assertSuccessorRegistrationSurvives(resolver, rendezvous, waiterB,
|
||||||
|
"a raw backend-error turn must not evict B from the raw-scrape error branch");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aSupersededDoneFailedTurnMustNotEvictItsSuccessorsRegistration() {
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver resolver = new CompletionResolver(new AgentControl(new FakeHerdr()), rendezvous,
|
||||||
|
ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||||
|
|
||||||
|
var waiterA = rendezvous.open("term_a");
|
||||||
|
var turnA = new CompletionResolver.InFlight(waiterA, null);
|
||||||
|
assertTrue(rendezvous.resolve("term_a", "A replied"));
|
||||||
|
rendezvous.close("term_a", waiterA);
|
||||||
|
var waiterB = rendezvous.open("term_a");
|
||||||
|
resolver.captureBaseline("term_a", new TurnToken("term_a", waiterB));
|
||||||
|
|
||||||
|
resolver.fail("term_a", turnA);
|
||||||
|
|
||||||
|
assertSuccessorRegistrationSurvives(resolver, rendezvous, waiterB,
|
||||||
|
"a done failed turn must not evict B from fail()'s early return");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aSupersededTooFastBackendErrorTurnMustNotEvictItsSuccessorsRegistration() {
|
||||||
|
FakeHerdr herdr = new FakeHerdr().readText("⏺ API Error: 400 invalid request body\n❯ ");
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
long[] clock = {10_000_000_000L};
|
||||||
|
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||||
|
ExhaustedPatternLookup.none(), ExhaustionSink.none(), () -> clock[0]);
|
||||||
|
|
||||||
|
var waiterA = rendezvous.open("term_a");
|
||||||
|
var turnA = new CompletionResolver.InFlight(waiterA, null, clock[0]);
|
||||||
|
rendezvous.close("term_a", waiterA);
|
||||||
|
var waiterB = rendezvous.open("term_a");
|
||||||
|
resolver.captureBaseline("term_a", new TurnToken("term_a", waiterB));
|
||||||
|
clock[0] += CompletionResolver.MIN_TURN_NANOS - 1;
|
||||||
|
|
||||||
|
resolver.resolve("term_a", turnA);
|
||||||
|
|
||||||
|
assertSuccessorRegistrationSurvives(resolver, rendezvous, waiterB,
|
||||||
|
"a too-fast backend-error turn must not evict B from failTooFast()");
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void assertSuccessorRegistrationSurvives(CompletionResolver resolver, Rendezvous rendezvous,
|
||||||
|
Object waiterB, String message) {
|
||||||
|
CompletionResolver.InFlight afterA = resolver.inFlight("term_a");
|
||||||
|
assertNotNull(afterA, message + " — a one-arg remove(target) would remove B");
|
||||||
|
assertEquals(waiterB, afterA.waiter(), message + " — the surviving record must belong to B");
|
||||||
|
assertTrue(rendezvous.resolve("term_a", "B replied"), message + " — B must still resolve normally");
|
||||||
|
}
|
||||||
|
|
||||||
// --- CB-578 stage A: backend-exhausted classification ---------------------------------
|
// --- CB-578 stage A: backend-exhausted classification ---------------------------------
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
|
|||||||
@@ -1,16 +1,18 @@
|
|||||||
package dev.ltms.fleet.inject;
|
package dev.ltms.fleet.inject;
|
||||||
|
|
||||||
import ch.qos.logback.classic.Level;
|
import ch.qos.logback.classic.Level;
|
||||||
import ch.qos.logback.classic.LoggerContext;
|
|
||||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
import ch.qos.logback.core.read.ListAppender;
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
import dev.ltms.fleet.herdr.AgentControl;
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
import dev.ltms.fleet.herdr.AgentStatus;
|
import dev.ltms.fleet.herdr.AgentStatus;
|
||||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
import dev.ltms.fleet.herdr.HerdrException;
|
import dev.ltms.fleet.herdr.HerdrException;
|
||||||
|
import dev.ltms.fleet.msg.Rendezvous;
|
||||||
import dev.ltms.fleet.msg.TestTurnTokens;
|
import dev.ltms.fleet.msg.TestTurnTokens;
|
||||||
|
import dev.ltms.fleet.msg.TurnToken;
|
||||||
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
import org.slf4j.LoggerFactory;
|
|
||||||
|
|
||||||
import java.util.ArrayList;
|
import java.util.ArrayList;
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
@@ -19,6 +21,8 @@ import java.util.Set;
|
|||||||
import java.util.concurrent.CompletableFuture;
|
import java.util.concurrent.CompletableFuture;
|
||||||
import java.util.concurrent.ExecutionException;
|
import java.util.concurrent.ExecutionException;
|
||||||
import java.util.concurrent.TimeUnit;
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.concurrent.atomic.AtomicInteger;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.*;
|
import static org.junit.jupiter.api.Assertions.*;
|
||||||
|
|
||||||
@@ -491,23 +495,15 @@ class InjectorTest {
|
|||||||
void readinessGraceExpiryIsLogged() {
|
void readinessGraceExpiryIsLogged() {
|
||||||
// CB-562: the grace-expiry path used to clear the queue silently, so a message that never
|
// CB-562: the grace-expiry path used to clear the queue silently, so a message that never
|
||||||
// reached the worker's pane surfaced elsewhere as an unrelated turn-stall failure. Assert the
|
// reached the worker's pane surfaced elsewhere as an unrelated turn-stall failure. Assert the
|
||||||
// expiry now names the real cause. (ListAppender capture pattern mirrors AuditLogTest.)
|
// expiry now names the real cause. (CapturedLog pattern mirrors AuditLogTest.)
|
||||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
try (CapturedLog log = CapturedLog.at(Injector.class, Level.WARN)) {
|
||||||
ch.qos.logback.classic.Logger injectorLog =
|
|
||||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(Injector.class);
|
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.setContext(ctx);
|
|
||||||
appender.start();
|
|
||||||
injectorLog.addAppender(appender);
|
|
||||||
injectorLog.setLevel(Level.WARN);
|
|
||||||
try {
|
|
||||||
Injector inj = new Injector(new AgentControl(herdr), TurnListener.NOOP, _ -> false, _ -> {
|
Injector inj = new Injector(new AgentControl(herdr), TurnListener.NOOP, _ -> false, _ -> {
|
||||||
});
|
});
|
||||||
inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
for (int i = 0; i < READINESS_SAMPLES; i++) inj.onStatus(T, AgentStatus.IDLE);
|
for (int i = 0; i < READINESS_SAMPLES; i++) inj.onStatus(T, AgentStatus.IDLE);
|
||||||
|
|
||||||
String warn = appender.list.stream()
|
String warn = log.events().stream()
|
||||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||||
.map(ILoggingEvent::getFormattedMessage)
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
.findFirst()
|
.findFirst()
|
||||||
@@ -516,8 +512,77 @@ class InjectorTest {
|
|||||||
assertTrue(warn.contains("never reached"), "the log names the real cause: " + warn);
|
assertTrue(warn.contains("never reached"), "the log names the real cause: " + warn);
|
||||||
assertTrue(warn.contains("1 queued message"),
|
assertTrue(warn.contains("1 queued message"),
|
||||||
"the log carries the failed message count: " + warn);
|
"the log carries the failed message count: " + warn);
|
||||||
} finally {
|
}
|
||||||
injectorLog.detachAppender(appender);
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void readinessGraceExpiryLogsTheMeasuredPollCountNextToTheConfiguredBudget() {
|
||||||
|
// fleetd #501, defect 1: READINESS_GRACE_POLLS (240) used to be printed twice — once as
|
||||||
|
// "after {} polls" and once inside the parenthesised budget — even though the loop's own
|
||||||
|
// counter (Target.notReadySincePoll) was in scope at the same call site. On THIS branch the
|
||||||
|
// counter has just reached the threshold, so it equals the constant by construction and this
|
||||||
|
// test cannot tell the two apart — it only pins that the message still carries both a poll
|
||||||
|
// count and a labelled configured budget, using literal numbers (240, 60), never
|
||||||
|
// READINESS_GRACE_POLLS or POLL_INTERVAL_MILLIS, so the assertion can't silently track a
|
||||||
|
// constant change instead of catching a real regression.
|
||||||
|
try (CapturedLog log = CapturedLog.at(Injector.class, Level.WARN)) {
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), TurnListener.NOOP, _ -> false, _ -> {
|
||||||
|
});
|
||||||
|
inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
for (int i = 0; i < READINESS_SAMPLES; i++) inj.onStatus(T, AgentStatus.IDLE);
|
||||||
|
|
||||||
|
String warn = log.events().stream()
|
||||||
|
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||||
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
|
.findFirst()
|
||||||
|
.orElse("no grace-expiry WARN logged");
|
||||||
|
assertTrue(warn.contains("after 240 polls"),
|
||||||
|
"must print the measured poll count as a plain number: " + warn);
|
||||||
|
assertTrue(warn.contains("configured=240 polls/60s"),
|
||||||
|
"must print the configured budget, clearly labelled: " + warn);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void readinessGraceExpiryLogsTheMeasuredElapsedTimeNotArithmeticOnConstants() {
|
||||||
|
// fleetd #501, defect 2: the old line computed "({}s)" as READINESS_GRACE_POLLS *
|
||||||
|
// POLL_INTERVAL_MILLIS / 1000 — arithmetic on two constants, never a measurement, and wrong
|
||||||
|
// in the direction that says everything ran on schedule. This stub clock returns two FIXED
|
||||||
|
// values (1_000ms at the first non-ready sample, 318_412ms at the poll that trips the grace)
|
||||||
|
// whose difference — 317_412ms — does NOT equal 240 * POLL_INTERVAL_MILLIS (=60_000ms).
|
||||||
|
// Asserting on that literal, non-derived number is what makes this test able to fail if the
|
||||||
|
// production code goes back to printing the constant-arithmetic value instead of the
|
||||||
|
// injected clock's measurement.
|
||||||
|
long[] readings = {1_000L, 318_412L};
|
||||||
|
AtomicInteger call = new AtomicInteger(0);
|
||||||
|
LongSupplier stubClock = () -> {
|
||||||
|
int i = call.getAndIncrement();
|
||||||
|
if (i >= readings.length) {
|
||||||
|
throw new AssertionError("nowMillis read more times than this fixture expects (" + i
|
||||||
|
+ "); the readiness-not-ready branch should read the clock exactly twice — "
|
||||||
|
+ "once to stamp the first non-ready sample, once at grace expiry");
|
||||||
|
}
|
||||||
|
return readings[i];
|
||||||
|
};
|
||||||
|
|
||||||
|
try (CapturedLog log = CapturedLog.at(Injector.class, Level.WARN)) {
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), TurnListener.NOOP, _ -> false, _ -> {
|
||||||
|
}, stubClock);
|
||||||
|
inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
for (int i = 0; i < READINESS_SAMPLES; i++) inj.onStatus(T, AgentStatus.IDLE);
|
||||||
|
|
||||||
|
String warn = log.events().stream()
|
||||||
|
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||||
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
|
.findFirst()
|
||||||
|
.orElse("no grace-expiry WARN logged");
|
||||||
|
assertTrue(warn.contains("elapsed=317412ms"), "must print the MEASURED elapsed time from "
|
||||||
|
+ "the injected clock (318412 - 1000 = 317412), not an arithmetic value: " + warn);
|
||||||
|
assertFalse(warn.contains("elapsed=60000ms"), "must not print "
|
||||||
|
+ "READINESS_GRACE_POLLS * POLL_INTERVAL_MILLIS (240 * 250 = 60000ms) as if it "
|
||||||
|
+ "were the measured elapsed time: " + warn);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -554,21 +619,13 @@ class InjectorTest {
|
|||||||
// CB-564: a vanished worker used to drop its queue with no log at all — the only trace was
|
// CB-564: a vanished worker used to drop its queue with no log at all — the only trace was
|
||||||
// whatever failed downstream (e.g. a caller's send timing out with no clue why). Assert the
|
// whatever failed downstream (e.g. a caller's send timing out with no clue why). Assert the
|
||||||
// drop itself now names the cause and the number of messages it failed.
|
// drop itself now names the cause and the number of messages it failed.
|
||||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
try (CapturedLog log = CapturedLog.at(Injector.class, Level.WARN)) {
|
||||||
ch.qos.logback.classic.Logger injectorLog =
|
|
||||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(Injector.class);
|
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.setContext(ctx);
|
|
||||||
appender.start();
|
|
||||||
injectorLog.addAppender(appender);
|
|
||||||
injectorLog.setLevel(Level.WARN);
|
|
||||||
try {
|
|
||||||
Injector inj = new Injector(new AgentControl(herdr), TurnListener.NOOP, _ -> true, _ -> {
|
Injector inj = new Injector(new AgentControl(herdr), TurnListener.NOOP, _ -> true, _ -> {
|
||||||
});
|
});
|
||||||
inj.enqueue(T, "orphan", TestTurnTokens.inert(T));
|
inj.enqueue(T, "orphan", TestTurnTokens.inert(T));
|
||||||
inj.drop(T, new HerdrException("worker gone", "pane_not_found", null));
|
inj.drop(T, new HerdrException("worker gone", "pane_not_found", null));
|
||||||
|
|
||||||
String warn = appender.list.stream()
|
String warn = log.events().stream()
|
||||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||||
.map(ILoggingEvent::getFormattedMessage)
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
.findFirst()
|
.findFirst()
|
||||||
@@ -576,8 +633,6 @@ class InjectorTest {
|
|||||||
assertTrue(warn.contains(T), "the log names the target terminal: " + warn);
|
assertTrue(warn.contains(T), "the log names the target terminal: " + warn);
|
||||||
assertTrue(warn.contains("1 message"), "the log carries the failed message count: " + warn);
|
assertTrue(warn.contains("1 message"), "the log carries the failed message count: " + warn);
|
||||||
assertTrue(warn.contains("worker gone"), "the log carries the real cause: " + warn);
|
assertTrue(warn.contains("worker gone"), "the log carries the real cause: " + warn);
|
||||||
} finally {
|
|
||||||
injectorLog.detachAppender(appender);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -613,4 +668,783 @@ class InjectorTest {
|
|||||||
ExecutionException ex = assertThrows(ExecutionException.class, f::get);
|
ExecutionException ex = assertThrows(ExecutionException.class, f::get);
|
||||||
assertInstanceOf(HerdrException.class, ex.getCause());
|
assertInstanceOf(HerdrException.class, ex.getCause());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A {@link HerdrClient} that throws a non-{@link RuntimeException} {@link Error} from {@code
|
||||||
|
* agent.prompt} instead of delegating — the fleetd #546 case: a stray non-RuntimeException
|
||||||
|
* throwable (e.g. a {@code NoClassDefFoundError}, fleetd #413) escaping the send seam at
|
||||||
|
* {@code Injector.java:383}. Records every call it sees itself, including the ones it throws
|
||||||
|
* for, since the delegate's own recording is never reached for {@code agent.prompt} — so a test
|
||||||
|
* can assert on exactly what this fake actually received.
|
||||||
|
*/
|
||||||
|
private static final class ErrorOnPrompt implements HerdrClient {
|
||||||
|
private final FakeHerdr delegate;
|
||||||
|
private final List<FakeHerdr.Call> calls = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||||
|
|
||||||
|
private ErrorOnPrompt(FakeHerdr delegate) {
|
||||||
|
this.delegate = delegate;
|
||||||
|
}
|
||||||
|
|
||||||
|
List<FakeHerdr.Call> calls() {
|
||||||
|
return calls;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public JsonNode call(String method, Object params) {
|
||||||
|
calls.add(new FakeHerdr.Call(method, params));
|
||||||
|
if (method.equals("agent.prompt")) {
|
||||||
|
throw new AssertionError("simulated non-RuntimeException send failure (fleetd #546)");
|
||||||
|
}
|
||||||
|
return delegate.call(method, params);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
delegate.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Mirrors StatusPoller's own per-target {@code catch (Throwable)} (fleetd #538 / PR #543): the
|
||||||
|
* production polling loop already swallows whatever escapes one target's round and comes back
|
||||||
|
* for the next one. A unit test that calls {@code onStatus} directly (bypassing StatusPoller)
|
||||||
|
* needs the same survival so it can observe what a SECOND round does, regardless of whether
|
||||||
|
* fleetd #546's fix is present.
|
||||||
|
*/
|
||||||
|
private static void pollOnceSurviving(Injector inj, String target, AgentStatus status) {
|
||||||
|
try {
|
||||||
|
inj.onStatus(target, status);
|
||||||
|
} catch (Throwable ignored) {
|
||||||
|
// matches StatusPoller.loop's own catch (Throwable) added by PR #543
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void anErrorFromSendRemovesTheMessageAndMarksItAttempted() {
|
||||||
|
// fleetd #546, acceptance test 1: an Error (not a RuntimeException) escaping the send seam
|
||||||
|
// must still be caught and the message dropped from the queue, never left QUEUED at the
|
||||||
|
// head. fleetd #551 updates what it is marked: an AssertionError from the send call itself
|
||||||
|
// proves nothing about whether the text reached the pane, so it is recorded as ATTEMPTED
|
||||||
|
// (uncertain), not a confident NOT_DELIVERED — see anErrorFromSendMarksItNotDeliveredOnlyForANotFoundCode
|
||||||
|
// below for the one case that still gets NOT_DELIVERED.
|
||||||
|
ErrorOnPrompt throwing = new ErrorOnPrompt(new FakeHerdr());
|
||||||
|
Injector inj = new Injector(new AgentControl(throwing));
|
||||||
|
Injector.Delivery delivery = inj.enqueue(T, "brief", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
assertDoesNotThrow(() -> inj.onStatus(T, AgentStatus.IDLE),
|
||||||
|
"fleetd #546: an Error from the send seam must be caught inside onStatus, not "
|
||||||
|
+ "escape it");
|
||||||
|
|
||||||
|
assertTrue(delivery.completion().isCompletedExceptionally(),
|
||||||
|
"the delivery's future must surface the send failure");
|
||||||
|
assertEquals(Injector.Cancellation.ATTEMPTED, inj.cancel(delivery),
|
||||||
|
"fleetd #551: the message must be dropped and marked ATTEMPTED (not a confident "
|
||||||
|
+ "NOT_DELIVERED), and never left QUEUED at the head of the queue");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void anErrorFromSendDoesNotRedeliverOnASecondRound() {
|
||||||
|
// fleetd #546, acceptance test 2 — the test this ticket exists for. Before the fix, an
|
||||||
|
// Error at the send seam left the message QUEUED (Injector.java:378 peeks, not polls), so a
|
||||||
|
// second onStatus round re-entered the same try and sent the same text again: the member's
|
||||||
|
// pane got the same brief typed into it twice.
|
||||||
|
ErrorOnPrompt throwing = new ErrorOnPrompt(new FakeHerdr());
|
||||||
|
Injector inj = new Injector(new AgentControl(throwing));
|
||||||
|
inj.enqueue(T, "brief", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
pollOnceSurviving(inj, T, AgentStatus.IDLE);
|
||||||
|
pollOnceSurviving(inj, T, AgentStatus.IDLE);
|
||||||
|
|
||||||
|
long promptCalls = throwing.calls().stream().filter(c -> c.method().equals("agent.prompt")).count();
|
||||||
|
assertEquals(1, promptCalls, "fleetd #546: the poisoned text must be sent exactly once — a "
|
||||||
|
+ "second onStatus round must not re-enter send for the same message");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aHerdrExceptionFromSendStillSurfacesButNowReportsAttempted() {
|
||||||
|
// fleetd #546, acceptance test 3 (control): HerdrException extends RuntimeException, so it
|
||||||
|
// was already caught before #546's widening. The send-failure/surface behaviour is
|
||||||
|
// unchanged by #551 — still dropped, still surfaced to the caller — but fleetd #551
|
||||||
|
// deliberately changes WHAT it is marked: "send_failed" is not a herdr `*_not_found` code,
|
||||||
|
// so this is exactly the response-half case #551 exists to fix. See
|
||||||
|
// aHerdrExceptionAfterThePasteIsNeverRecordedAsConfidentlyNotDelivered for the acceptance
|
||||||
|
// test this ticket was filed for.
|
||||||
|
FakeHerdr failing = new FakeHerdr().agentSendFailsWith("send_failed");
|
||||||
|
Injector inj = new Injector(new AgentControl(failing));
|
||||||
|
Injector.Delivery delivery = inj.enqueue(T, "boom", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE);
|
||||||
|
|
||||||
|
assertTrue(delivery.completion().isCompletedExceptionally(),
|
||||||
|
"a HerdrException at the send seam must still surface to the caller, unchanged by "
|
||||||
|
+ "fleetd #546's widening");
|
||||||
|
assertEquals(Injector.Cancellation.ATTEMPTED, inj.cancel(delivery),
|
||||||
|
"fleetd #551: a non-*_not_found HerdrException must now report ATTEMPTED, not a "
|
||||||
|
+ "confident NOT_DELIVERED");
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- fleetd #551: the injector records the delivery attempt BEFORE the irreversible send, not
|
||||||
|
// after, so a failure in the send's response window can never be written down as a confident
|
||||||
|
// NOT_DELIVERED for text that may already be sitting in the worker's pane. ---
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aHerdrExceptionAfterThePasteIsNeverRecordedAsConfidentlyNotDelivered() {
|
||||||
|
// fleetd #551, the ticket's acceptance test 1, and comment 17037's added test ("a test
|
||||||
|
// pinning that a caller cannot receive a confident NOT_DELIVERED for a message that reached
|
||||||
|
// the paste"): a HerdrException thrown from the response half of the send seam — herdr
|
||||||
|
// replied, with an error, so it definitely processed the request — must not leave a record
|
||||||
|
// claiming the text was never delivered. Red before the fix: on main at ba2f4d1 this
|
||||||
|
// asserted (and got) Injector.Cancellation.NOT_DELIVERED.
|
||||||
|
FakeHerdr failing = new FakeHerdr().agentSendFailsWith("send_failed");
|
||||||
|
Injector inj = new Injector(new AgentControl(failing));
|
||||||
|
Injector.Delivery delivery = inj.enqueue(T, "brief", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE);
|
||||||
|
|
||||||
|
assertTrue(delivery.completion().isCompletedExceptionally(),
|
||||||
|
"the caller must still see the send failure");
|
||||||
|
assertEquals(Injector.Cancellation.ATTEMPTED, inj.cancel(delivery),
|
||||||
|
"fleetd #551: a HerdrException from the response half of the send seam must not be "
|
||||||
|
+ "recorded as a confident NOT_DELIVERED — the text may already be sitting "
|
||||||
|
+ "in the pane");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void ordinarySuccessStillReportsDeliveredExactlyOnce() {
|
||||||
|
// fleetd #551, the ticket's acceptance test 2: the record-before-send reorder must not
|
||||||
|
// change the ordinary success path — exactly one send, and the caller-facing Cancellation
|
||||||
|
// for it is still DELIVERED, never left at the new ATTEMPTED value.
|
||||||
|
Injector.Delivery delivery = injector.enqueue(T, "hello", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
injector.onStatus(T, AgentStatus.IDLE);
|
||||||
|
|
||||||
|
assertEquals(List.of("hello"), sent(), "the text must be sent exactly once");
|
||||||
|
assertTrue(delivery.completion().isDone() && !delivery.completion().isCompletedExceptionally(),
|
||||||
|
"the ordinary success path must still complete normally");
|
||||||
|
assertEquals(Injector.Cancellation.DELIVERED, injector.cancel(delivery),
|
||||||
|
"fleetd #551: the ordinary success path must still report DELIVERED, unaffected by "
|
||||||
|
+ "the record-before-send reorder");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void transportDownStillSurfacesTheErrorAndDropsTheEntry() {
|
||||||
|
// fleetd #551, the ticket's acceptance test 3: the ordinary transport-down path (herdr
|
||||||
|
// unreachable — a plain HerdrException with no error code) must be unaffected by the #551
|
||||||
|
// reorder — the failure still surfaces to the caller, and the entry is not left sitting in
|
||||||
|
// the queue for a second onStatus round to resend.
|
||||||
|
FakeHerdr unreachable = new FakeHerdr().healthy(false);
|
||||||
|
Injector inj = new Injector(new AgentControl(unreachable));
|
||||||
|
Injector.Delivery delivery = inj.enqueue(T, "brief", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE);
|
||||||
|
|
||||||
|
assertTrue(delivery.completion().isCompletedExceptionally(),
|
||||||
|
"a transport-down failure must still surface to the caller");
|
||||||
|
assertTrue(inj.activeTargets().isEmpty(),
|
||||||
|
"the entry must not be left queued for a second round to resend");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aNotFoundCodeFromSendStillReportsAConfidentNotDelivered() {
|
||||||
|
// fleetd #551 (comment 17037, point 3): a herdr `*_not_found` error means the target
|
||||||
|
// pane/agent does not exist at all, so nothing could have been pasted anywhere — the
|
||||||
|
// codebase already treats this family as a confirmed absence everywhere else (StatusPoller,
|
||||||
|
// AgentControl's own retry). #551's new ATTEMPTED state must not swallow this case: it stays
|
||||||
|
// a confident NOT_DELIVERED.
|
||||||
|
FakeHerdr failing = new FakeHerdr().agentSendFailsWith("agent_not_found");
|
||||||
|
Injector inj = new Injector(new AgentControl(failing));
|
||||||
|
Injector.Delivery delivery = inj.enqueue(T, "brief", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE);
|
||||||
|
|
||||||
|
assertTrue(delivery.completion().isCompletedExceptionally(),
|
||||||
|
"an agent_not_found failure must still surface to the caller");
|
||||||
|
assertEquals(Injector.Cancellation.NOT_DELIVERED, inj.cancel(delivery),
|
||||||
|
"fleetd #551: a herdr *_not_found error is a confirmed absence, not merely "
|
||||||
|
+ "inconclusive — it must stay NOT_DELIVERED, not the new ATTEMPTED");
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- fleetd #553: a throwable from any listener callback in onStatus must not skip the
|
||||||
|
// delivered-future completion. The whole post-monitor region is now wrapped in try/finally. ---
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void theOriginalThrowableFromAListenerStillEscapesOnStatus() {
|
||||||
|
// fleetd #553's control: the entire point of the finally backstop is that it completes a
|
||||||
|
// future WITHOUT swallowing whatever escaped. A fix built the wrong way (e.g. catching
|
||||||
|
// Throwable in the finally, or wrapping the region in try/catch instead of try/finally)
|
||||||
|
// could make every other test in this group pass while still converting a loud listener
|
||||||
|
// bug into a silent one — which the ticket calls a worse outcome than the bug itself. This
|
||||||
|
// is deliberately its own test, not folded into another one's assertion.
|
||||||
|
RuntimeException boom = new RuntimeException("fleetd #553 control");
|
||||||
|
TurnListener throwing = new TurnListener() {
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
throw boom;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), throwing);
|
||||||
|
inj.enqueue(T, "only", TestTurnTokens.inert(T));
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE); // deliver
|
||||||
|
inj.onStatus(T, AgentStatus.WORKING); // pickup
|
||||||
|
|
||||||
|
RuntimeException thrown = assertThrows(RuntimeException.class,
|
||||||
|
() -> inj.onStatus(T, AgentStatus.IDLE),
|
||||||
|
"onStatus must still propagate the listener's own throwable, unmodified");
|
||||||
|
assertSame(boom, thrown, "must be the EXACT throwable, not a wrapper or a different instance");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aRuntimeExceptionFromOnTurnCompleteStillCompletesTheNextDelivery() {
|
||||||
|
// fleetd #553, acceptance test 1: onTurnComplete fires for "first"'s completed turn INSIDE
|
||||||
|
// the same onStatus call that then peeks and delivers "second" — the turnCompleted check
|
||||||
|
// (Injector.java, inside the released-pickup branch) clears awaitingCompletion just before
|
||||||
|
// the delivery guard right below it is evaluated, so both happen in one round when there is
|
||||||
|
// no post-turn action. Before fleetd #553, a RuntimeException thrown here unwound straight
|
||||||
|
// out of onStatus and skipped the `if (sent != null)` block, leaving "second"'s future
|
||||||
|
// pending forever even though it was already off the queue, marked DELIVERED, and typed
|
||||||
|
// into the pane.
|
||||||
|
RuntimeException boom = new RuntimeException("boom from onTurnComplete");
|
||||||
|
TurnListener throwing = new TurnListener() {
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
throw boom;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), throwing);
|
||||||
|
inj.enqueue(T, "first", TestTurnTokens.inert(T));
|
||||||
|
Injector.Delivery second = inj.enqueue(T, "second", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE); // delivers "first"
|
||||||
|
inj.onStatus(T, AgentStatus.WORKING); // picked up
|
||||||
|
|
||||||
|
RuntimeException thrown = assertThrows(RuntimeException.class,
|
||||||
|
() -> inj.onStatus(T, AgentStatus.IDLE),
|
||||||
|
"the throwable from onTurnComplete must still escape onStatus");
|
||||||
|
assertSame(boom, thrown);
|
||||||
|
|
||||||
|
assertEquals(List.of("first", "second"), sent(),
|
||||||
|
"fleetd #553: \"second\" must still be sent even though onTurnComplete threw for "
|
||||||
|
+ "\"first\"'s completion");
|
||||||
|
assertTrue(second.completion().isDone() && !second.completion().isCompletedExceptionally(),
|
||||||
|
"fleetd #553: \"second\"'s delivered future must still complete normally despite "
|
||||||
|
+ "the throw");
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final int READINESS_GRACE_SAMPLES = 240; // Injector.READINESS_GRACE_POLLS is private
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aRuntimeExceptionFromForgetStillCompletesTheReadinessFailureFutures() {
|
||||||
|
// fleetd #553: forget.accept fires as part of the CB-114 readiness-grace path (production
|
||||||
|
// wires it to presence::forget). Before fleetd #553, that block called forget.accept
|
||||||
|
// BEFORE completing the queued messages' futures, so a RuntimeException from forget left
|
||||||
|
// every one of them pending forever even though they were already marked NOT_DELIVERED and
|
||||||
|
// dropped from the queue. fleetd #553 completes those futures first, so forget.accept
|
||||||
|
// throwing afterward can no longer un-complete them.
|
||||||
|
RuntimeException boom = new RuntimeException("boom from forget");
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), TurnListener.NOOP, _ -> false, _ -> {
|
||||||
|
throw boom;
|
||||||
|
});
|
||||||
|
Injector.Delivery delivery = inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
// Build the not-ready streak up to (but not past) the grace threshold.
|
||||||
|
for (int i = 0; i < READINESS_GRACE_SAMPLES - 1; i++) inj.onStatus(T, AgentStatus.IDLE);
|
||||||
|
assertFalse(delivery.completion().isDone(), "must not be resolved before the grace expires");
|
||||||
|
|
||||||
|
RuntimeException thrown = assertThrows(RuntimeException.class,
|
||||||
|
() -> inj.onStatus(T, AgentStatus.IDLE),
|
||||||
|
"the throwable from forget.accept must still escape onStatus");
|
||||||
|
assertSame(boom, thrown);
|
||||||
|
|
||||||
|
assertTrue(delivery.completion().isCompletedExceptionally(),
|
||||||
|
"fleetd #553: the queued message's future must still be completed even though "
|
||||||
|
+ "forget.accept threw");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aRuntimeExceptionFromOnTurnFailedStillLeavesTheReadinessFailureFuturesCompleted() {
|
||||||
|
// fleetd #553, acceptance test: same readiness-grace path as the forget test above, but the
|
||||||
|
// throw comes from onTurnFailed instead. NOTE: onTurnFailed is the LAST statement in this
|
||||||
|
// block (both before and after this ticket), so the queued messages' futures are already
|
||||||
|
// completed by the time it runs, regardless of this fix — this test cannot be made to fail
|
||||||
|
// against the pre-fleetd-553 code the way the onTurnComplete/forget/postAction tests can; I
|
||||||
|
// could not find a construction where onTurnFailed's own throw is what stands between a
|
||||||
|
// future and its completion (flagged to the lead via fleet_ask; no reply arrived before the
|
||||||
|
// ~55s window closed, so recorded here instead). It still pins a real invariant this ticket
|
||||||
|
// cares about — an ordinary RuntimeException from this callback must not un-complete a
|
||||||
|
// future that was already decided — so it stays as a regression lock, not a bug-fix proof.
|
||||||
|
RuntimeException boom = new RuntimeException("boom from onTurnFailed");
|
||||||
|
TurnListener throwing = new TurnListener() {
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onTurnFailed(String target) {
|
||||||
|
throw boom;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), throwing, _ -> false, _ -> {
|
||||||
|
});
|
||||||
|
Injector.Delivery delivery = inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
for (int i = 0; i < READINESS_GRACE_SAMPLES - 1; i++) inj.onStatus(T, AgentStatus.IDLE);
|
||||||
|
|
||||||
|
RuntimeException thrown = assertThrows(RuntimeException.class,
|
||||||
|
() -> inj.onStatus(T, AgentStatus.IDLE),
|
||||||
|
"the throwable from onTurnFailed must still escape onStatus");
|
||||||
|
assertSame(boom, thrown);
|
||||||
|
|
||||||
|
assertTrue(delivery.completion().isCompletedExceptionally(),
|
||||||
|
"the queued message's future must remain completed despite onTurnFailed throwing");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aRuntimeExceptionFromOnTurnCompleteWithPostActionStillUnwedgesTheTarget() {
|
||||||
|
// fleetd #553, acceptance test: before this ticket, a RuntimeException from
|
||||||
|
// onTurnCompleteWithPostAction skipped the `t.postTurnPending = false` reset that used to
|
||||||
|
// sit right after it (with nothing guarding it), permanently wedging the target —
|
||||||
|
// postTurnPending stayed true forever, so the delivery guard never passed again and
|
||||||
|
// "second" (already queued) was never delivered, no matter how many further onStatus
|
||||||
|
// rounds ran. fleetd #553 wraps that call so the reset always runs.
|
||||||
|
RuntimeException boom = new RuntimeException("boom from onTurnCompleteWithPostAction");
|
||||||
|
class ThrowingPostTurn implements TurnListener {
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean hasPostTurnAction(String target) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean onTurnCompleteWithPostAction(String target) {
|
||||||
|
throw boom;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), new ThrowingPostTurn());
|
||||||
|
inj.enqueue(T, "first", TestTurnTokens.inert(T));
|
||||||
|
Injector.Delivery second = inj.enqueue(T, "second", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE); // delivers "first"
|
||||||
|
inj.onStatus(T, AgentStatus.WORKING); // picked up
|
||||||
|
|
||||||
|
RuntimeException thrown = assertThrows(RuntimeException.class,
|
||||||
|
() -> inj.onStatus(T, AgentStatus.IDLE), // "first"'s turn completes; post-action throws
|
||||||
|
"the throwable from onTurnCompleteWithPostAction must still escape onStatus");
|
||||||
|
assertSame(boom, thrown);
|
||||||
|
assertEquals(List.of("first"), sent(), "\"second\" must not be sent in the SAME round as the throw");
|
||||||
|
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE); // a later, unrelated round
|
||||||
|
assertEquals(List.of("first", "second"), sent(),
|
||||||
|
"fleetd #553: the target must not stay wedged — \"second\" must still be delivered "
|
||||||
|
+ "once postTurnPending is reset despite the earlier throw");
|
||||||
|
assertTrue(second.completion().isDone() && !second.completion().isCompletedExceptionally());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A {@link HerdrClient} that throws a non-{@link RuntimeException} {@link Error} from {@code
|
||||||
|
* agent.send_keys} instead of delegating — the fleetd #553 case for the resubmit nudge: its own
|
||||||
|
* catch (Injector.java, the first block after the monitor) was {@code RuntimeException}-only,
|
||||||
|
* the same one-class-too-narrow shape #546 fixed at the send seam. Records every call it sees
|
||||||
|
* itself, mirroring {@code ErrorOnPrompt} above, since the delegate's own recording is never
|
||||||
|
* reached for {@code agent.send_keys}.
|
||||||
|
*/
|
||||||
|
private static final class ErrorOnSendKeys implements HerdrClient {
|
||||||
|
private final FakeHerdr delegate;
|
||||||
|
private final List<FakeHerdr.Call> calls = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||||
|
|
||||||
|
private ErrorOnSendKeys(FakeHerdr delegate) {
|
||||||
|
this.delegate = delegate;
|
||||||
|
}
|
||||||
|
|
||||||
|
List<FakeHerdr.Call> calls() {
|
||||||
|
return calls;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public JsonNode call(String method, Object params) {
|
||||||
|
calls.add(new FakeHerdr.Call(method, params));
|
||||||
|
if (method.equals("agent.send_keys")) {
|
||||||
|
throw new AssertionError("simulated non-RuntimeException resubmit failure (fleetd #553)");
|
||||||
|
}
|
||||||
|
return delegate.call(method, params);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
delegate.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void anErrorFromTheResubmitNudgeDoesNotPreventTheSentCompletion() {
|
||||||
|
// fleetd #553, acceptance test: the resubmit nudge's own catch was RuntimeException-only —
|
||||||
|
// the same one-class-too-narrow shape #546 fixed at the send seam (now :382). It sits FIRST
|
||||||
|
// after the monitor, so before this ticket an Error escaping it skipped every block below in
|
||||||
|
// THAT round, including any `sent` completion that round might otherwise produce. Widened to
|
||||||
|
// Throwable (matching #549's own widening) so it can no longer escape onStatus at all — the
|
||||||
|
// delivery that already completed at send time, and the worker's later pickup, are both
|
||||||
|
// unaffected by the Error in between.
|
||||||
|
ErrorOnSendKeys throwing = new ErrorOnSendKeys(new FakeHerdr());
|
||||||
|
Injector inj = new Injector(new AgentControl(throwing));
|
||||||
|
Injector.Delivery delivery = inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE); // delivers "task"; awaiting pickup
|
||||||
|
assertTrue(delivery.completion().isDone() && !delivery.completion().isCompletedExceptionally(),
|
||||||
|
"the delivery completes at send time, before any resubmit nudge is attempted");
|
||||||
|
|
||||||
|
assertDoesNotThrow(() -> inj.onStatus(T, AgentStatus.IDLE), // still idle -> resubmit; Error thrown+caught
|
||||||
|
"fleetd #553: an Error from the resubmit nudge must be caught inside onStatus, not "
|
||||||
|
+ "escape it");
|
||||||
|
|
||||||
|
inj.onStatus(T, AgentStatus.WORKING); // the worker's real pickup must still resolve cleanly
|
||||||
|
long promptCalls = throwing.calls().stream().filter(c -> c.method().equals("agent.prompt")).count();
|
||||||
|
assertEquals(1, promptCalls,
|
||||||
|
"sent exactly once, unaffected by the resubmit Error in between");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void ordinarySuccessStillCompletesExactlyOnceAfterOnDelivered() {
|
||||||
|
// Acceptance: the ordinary success path is unchanged by fleetd #553 — the future completes
|
||||||
|
// normally exactly once, and onDelivered still runs before it (CB-115's pane baseline).
|
||||||
|
List<String> events = new ArrayList<>();
|
||||||
|
TurnListener listener = new TurnListener() {
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onDelivered(String target, dev.ltms.fleet.msg.TurnToken token) {
|
||||||
|
events.add("onDelivered");
|
||||||
|
}
|
||||||
|
};
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), listener);
|
||||||
|
CompletableFuture<Void> f = inj.enqueue(T, "hello", TestTurnTokens.inert(T)).completion();
|
||||||
|
f.whenComplete((v, ex) -> events.add("completed"));
|
||||||
|
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE);
|
||||||
|
|
||||||
|
assertEquals(List.of("onDelivered", "completed"), events,
|
||||||
|
"onDelivered must still run before the future completes, unchanged by fleetd #553");
|
||||||
|
assertTrue(f.isDone() && !f.isCompletedExceptionally());
|
||||||
|
}
|
||||||
|
|
||||||
|
// Acceptance: the ordinary failure path (a HerdrException at the send seam still completes the
|
||||||
|
// future exceptionally with that exception) is already covered, unchanged, by the existing
|
||||||
|
// aHerdrExceptionFromSendStillProducesNotDeliveredUnchanged test above (fleetd #546) — its
|
||||||
|
// codepath is untouched by fleetd #553's try/finally, since sendError != null is set before the
|
||||||
|
// try block begins and that branch never threw to begin with.
|
||||||
|
|
||||||
|
// --- fleetd #553, sharpened acceptance (ticket comments 16870/16884/16890/16903): completing
|
||||||
|
// sent.delivered() is only HALF the job. See the two tests below. ---
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aRuntimeExceptionFromOnTurnCompleteStillRegistersTheRendezvousWaiterForTheNextDelivery() {
|
||||||
|
// fleetd #553's real invariant, not just the completion of sent.delivered(). There are TWO
|
||||||
|
// futures at stake: sent.delivered() (the delivery future) and sent.token().waiter() (the
|
||||||
|
// rendezvous waiter a blocking fleet_send actually waits on for the worker's ANSWER). The
|
||||||
|
// waiter is registered only by turnListener.onDelivered() — production wires this to
|
||||||
|
// CompletionResolver.captureBaseline, which does the inFlight.put() — and that call lives
|
||||||
|
// inside the very `if (sent != null)` block a plain "complete the future" backstop does not
|
||||||
|
// reach. A finally that only completes sent.delivered() converts a hang into a HANG WITH A
|
||||||
|
// SUCCESS RECEIPT: the caller is told the send landed, then waits out its full timeout for
|
||||||
|
// an answer that can never resolve, because CompletionResolver.resolve (:301-308) finds no
|
||||||
|
// inFlight entry and returns silently. This test is red on that half-fix, and green only
|
||||||
|
// once the finally also runs onDelivered() when the normal block never got the chance.
|
||||||
|
//
|
||||||
|
// The scenario: onTurnComplete throws while completing "first"'s turn, in the SAME onStatus
|
||||||
|
// round that (per Injector.java :361-391) then peeks and delivers "second" — the normal
|
||||||
|
// case, not a corner, since the :365 assignment is what lets the :376 delivery guard pass.
|
||||||
|
// turnListener.onTurnComplete runs BEFORE the `if (sent != null)` block, so the throw here
|
||||||
|
// means onDelivered for "second" is never reached on the normal path — only the finally
|
||||||
|
// backstop can register it.
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver resolver = new CompletionResolver(new AgentControl(new FakeHerdr()), rendezvous,
|
||||||
|
ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||||
|
RuntimeException boom = new RuntimeException("boom from onTurnComplete");
|
||||||
|
TurnListener throwingOnTurnCompleteOnly = new TurnListener() {
|
||||||
|
@Override
|
||||||
|
public void onDelivered(String target, TurnToken token) {
|
||||||
|
resolver.onDelivered(target, token); // the real registration, exactly as production wires it
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
throw boom; // "first"'s completion callback, thrown before "second" is delivered
|
||||||
|
}
|
||||||
|
};
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), throwingOnTurnCompleteOnly);
|
||||||
|
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.open("session-a");
|
||||||
|
TurnToken secondToken = new TurnToken(T, waiter);
|
||||||
|
inj.enqueue(T, "first", TestTurnTokens.inert(T));
|
||||||
|
Injector.Delivery second = inj.enqueue(T, "second", secondToken);
|
||||||
|
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE); // delivers "first"
|
||||||
|
inj.onStatus(T, AgentStatus.WORKING); // picked up
|
||||||
|
|
||||||
|
RuntimeException thrown = assertThrows(RuntimeException.class,
|
||||||
|
() -> inj.onStatus(T, AgentStatus.IDLE), // "first" completes (throws); "second" delivered
|
||||||
|
"the original throwable from onTurnComplete must still escape onStatus");
|
||||||
|
assertSame(boom, thrown, "must be the EXACT throwable, not a wrapper or a different instance");
|
||||||
|
|
||||||
|
assertEquals(List.of("first", "second"), sent(),
|
||||||
|
"\"second\" must still be sent even though onTurnComplete threw for \"first\"");
|
||||||
|
assertTrue(second.completion().isDone() && !second.completion().isCompletedExceptionally(),
|
||||||
|
"\"second\"'s delivery future must still complete despite the throw");
|
||||||
|
|
||||||
|
CompletionResolver.InFlight inFlight = resolver.inFlight(T);
|
||||||
|
assertNotNull(inFlight,
|
||||||
|
"fleetd #553: the finally must register the new turn's waiter (via onDelivered), not "
|
||||||
|
+ "just complete its delivery future — otherwise a blocking fleet_send is told "
|
||||||
|
+ "its message landed and then waits out the full timeout for an answer that "
|
||||||
|
+ "can never resolve");
|
||||||
|
assertSame(waiter, inFlight.waiter(),
|
||||||
|
"the registered waiter must be exactly \"second\"'s waiter, not some other one");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void anOnDeliveredThrowAfterItsOwnRegistrationDoesNotRunASecondTime() {
|
||||||
|
// fleetd #553 comment 16903: `sentHandled` must be set to true BEFORE onDelivered() runs,
|
||||||
|
// not after. Setting it after would mean a throw from onDelivered() PART WAY THROUGH — e.g.
|
||||||
|
// after CompletionResolver's own captureBaseline has already done its inFlight.put() — still
|
||||||
|
// leaves the flag false, so the finally backstop (seeing "not handled") calls onDelivered() a
|
||||||
|
// SECOND time. That second captureBaseline runs later, over a pane that may already have
|
||||||
|
// absorbed this turn's output, and CompletionResolver.resolve's own baseline-equals-tail
|
||||||
|
// suppression (:364-369, which deliberately keeps the in-flight record on a match) can then
|
||||||
|
// drop every future completion for the turn, permanently. This assertion is red on a
|
||||||
|
// `sentHandled = true` placed AFTER the onDelivered() call (onDelivered runs twice) and green
|
||||||
|
// on the correct placement (runs exactly once) — regardless of what onDelivered itself did.
|
||||||
|
AtomicInteger onDeliveredCalls = new AtomicInteger();
|
||||||
|
RuntimeException boom = new RuntimeException("boom from onDelivered, after its own registration ran");
|
||||||
|
TurnListener listener = new TurnListener() {
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onDelivered(String target, TurnToken token) {
|
||||||
|
onDeliveredCalls.incrementAndGet(); // stand-in for CompletionResolver's inFlight.put
|
||||||
|
throw boom; // then fail, as if a LATER step inside onDelivered blew up
|
||||||
|
}
|
||||||
|
};
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), listener);
|
||||||
|
inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
RuntimeException thrown = assertThrows(RuntimeException.class,
|
||||||
|
() -> inj.onStatus(T, AgentStatus.IDLE), // delivers "task"; onDelivered throws
|
||||||
|
"the original throwable from onDelivered must still escape onStatus");
|
||||||
|
assertSame(boom, thrown);
|
||||||
|
|
||||||
|
assertEquals(1, onDeliveredCalls.get(),
|
||||||
|
"fleetd #553: onDelivered must be called exactly once — a `sentHandled` flag set "
|
||||||
|
+ "AFTER the call (rather than before) would leave it false here and the "
|
||||||
|
+ "finally backstop would call onDelivered a second time");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void anOnDeliveredThrowOnTheNormalPathStillCompletesTheDeliveryFuture() {
|
||||||
|
// fleetd #553, lead review of PR #557 (ticket comment 16916): `sentHandled` must guard ONLY
|
||||||
|
// the onDelivered RE-CALL in the finally, never the future completion alongside it. Gating
|
||||||
|
// BOTH behind `!sentHandled` (the shape this test is red against) misses the one path the
|
||||||
|
// flag's own correct placement creates: `sentHandled` is set to true FIRST, before the
|
||||||
|
// onDelivered() call, inside the `if (sent != null)` block above (see the previous test —
|
||||||
|
// that placement is right and must not change). So when onDelivered() itself throws on that
|
||||||
|
// NORMAL path, `sentHandled` already reads true by the time control reaches the finally, and
|
||||||
|
// a `sent != null && !sentHandled` guard around the WHOLE recovery — completion included —
|
||||||
|
// skips it entirely. The message was typed into the target's pane and taken off the queue
|
||||||
|
// inside the monitor, same as any other delivery, so its future is left pending forever: the
|
||||||
|
// exact defect this ticket exists to close, just reached from a different throwing call.
|
||||||
|
//
|
||||||
|
// The fix splits the one flag's two jobs: `!sentHandled` keeps gating only the onDelivered
|
||||||
|
// call (so the exactly-once guarantee in the test above still holds — CompletableFuture.
|
||||||
|
// complete/completeExceptionally are idempotent, so completing unconditionally here is a
|
||||||
|
// no-op on the ordinary path, where the `if (sent != null)` block already completed it.
|
||||||
|
RuntimeException boom = new RuntimeException("boom from onDelivered on the normal path");
|
||||||
|
TurnListener listener = new TurnListener() {
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onDelivered(String target, TurnToken token) {
|
||||||
|
throw boom;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), listener);
|
||||||
|
CompletableFuture<Void> delivered = inj.enqueue(T, "task", TestTurnTokens.inert(T)).completion();
|
||||||
|
|
||||||
|
RuntimeException thrown = assertThrows(RuntimeException.class,
|
||||||
|
() -> inj.onStatus(T, AgentStatus.IDLE), // delivers "task"; onDelivered throws
|
||||||
|
"the original throwable from onDelivered must still escape onStatus");
|
||||||
|
assertSame(boom, thrown);
|
||||||
|
|
||||||
|
assertTrue(delivered.isDone(),
|
||||||
|
"fleetd #553: \"task\" was actually delivered — typed into the pane and taken off "
|
||||||
|
+ "the queue inside the monitor — so its delivery future must be completed on "
|
||||||
|
+ "every path out of onStatus, including the one where onDelivered itself is "
|
||||||
|
+ "what threw. Leaving it pending here is a hang, not a fix");
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- fleetd #556: registration is the Injector's own invariant now, structural rather than
|
||||||
|
// delegated to a TurnListener callback that can throw. #553 could only make the ONE reachable
|
||||||
|
// listener behave; this closes the shape itself. ---
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aTurnListenerThatThrowsFromEveryCallbackStillLeavesTheDeliveredTurnRegisteredAndResolvable() {
|
||||||
|
// fleetd #556 acceptance criterion 1 (the ticket's own, restated by comment 16966 as the
|
||||||
|
// replacement for #561's order-dependent test): a TurnListener that throws from EVERY
|
||||||
|
// callback it implements — nothing about it is safe to lean on — must still leave a
|
||||||
|
// delivered turn registered, with its waiter still resolvable through the ordinary path.
|
||||||
|
// Before this ticket, registration lived INSIDE onDelivered, one of the callbacks that
|
||||||
|
// throws here — this test is red against that shape, because the only listener callback
|
||||||
|
// this specific delivery scenario exercises (a first delivery, no previous turn to
|
||||||
|
// complete) is exactly the one that throws.
|
||||||
|
RuntimeException boom = new RuntimeException("fleetd #556: throws from every callback");
|
||||||
|
TurnListener throwsEverywhere = new TurnListener() {
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
throw boom;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean hasPostTurnAction(String target) {
|
||||||
|
throw boom;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public boolean onTurnCompleteWithPostAction(String target) {
|
||||||
|
throw boom;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onTurnFailed(String target) {
|
||||||
|
throw boom;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onTurnFailed(String target, String reason) {
|
||||||
|
throw boom;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void onDelivered(String target, TurnToken token) {
|
||||||
|
throw boom;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver completion = new CompletionResolver(new AgentControl(new FakeHerdr()), rendezvous,
|
||||||
|
ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||||
|
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), throwsEverywhere, _ -> true, _ -> {
|
||||||
|
}, completion::register);
|
||||||
|
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.open(T);
|
||||||
|
inj.enqueue(T, "brief", new TurnToken(T, waiter));
|
||||||
|
|
||||||
|
RuntimeException thrown = assertThrows(RuntimeException.class,
|
||||||
|
() -> inj.onStatus(T, AgentStatus.IDLE), // delivers "brief"; onDelivered throws
|
||||||
|
"the throwing listener's own bug must still escape onStatus — a listener bug stays "
|
||||||
|
+ "loud (fleetd #553's own guarantee, unchanged by this ticket)");
|
||||||
|
assertSame(boom, thrown);
|
||||||
|
|
||||||
|
CompletionResolver.InFlight inFlight = completion.inFlight(T);
|
||||||
|
assertNotNull(inFlight,
|
||||||
|
"fleetd #556: the delivered turn must be registered even though the ONLY "
|
||||||
|
+ "TurnListener wired throws from every single callback, including "
|
||||||
|
+ "onDelivered — registration must not depend on that call succeeding");
|
||||||
|
assertSame(waiter, inFlight.waiter(),
|
||||||
|
"the registered entry must carry the exact waiter this turn's send opened");
|
||||||
|
|
||||||
|
assertFalse(waiter.isDone(), "sanity: nothing has resolved this waiter yet");
|
||||||
|
assertTrue(rendezvous.resolve(T, "hello"),
|
||||||
|
"the waiter registrar.register captured must be the real, live rendezvous waiter — "
|
||||||
|
+ "still resolvable through the ordinary fleet_reply path, unaffected by the "
|
||||||
|
+ "throwing listener");
|
||||||
|
assertEquals(Rendezvous.Kind.REPLY, waiter.getNow(null).kind());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void theNewTurnsRegistrationRunsAfterThePreviousTurnsCompletionIsRead() {
|
||||||
|
// fleetd #556 acceptance criterion 2 (CB-116 guard, ticket body + comment 16966): the
|
||||||
|
// ordering constraint from fleetd #553 must survive this redesign — onTurnComplete reads
|
||||||
|
// CompletionResolver's inFlight entry for the PREVIOUS turn, so the new turn's
|
||||||
|
// registrar.register() call must not run before that read, or a redesign trades this
|
||||||
|
// ticket's defect for the CB-116 cross-turn stale reply. Pinned here as a call-ORDER
|
||||||
|
// assertion so a future redesign that hoists registrar.register() earlier (e.g. "for
|
||||||
|
// simplicity", ahead of the turnCompleted block) goes red immediately, before it can ever
|
||||||
|
// reach a real cross-turn scrape race.
|
||||||
|
List<String> events = new ArrayList<>();
|
||||||
|
TurnListener listener = new TurnListener() {
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
events.add("onTurnComplete:" + target);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
TurnRegistrar registrar = (target, token) -> events.add("register:" + target);
|
||||||
|
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), listener, _ -> true, _ -> {
|
||||||
|
}, registrar);
|
||||||
|
inj.enqueue(T, "first", TestTurnTokens.inert(T));
|
||||||
|
inj.enqueue(T, "second", TestTurnTokens.inert(T));
|
||||||
|
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE); // delivers "first" -> register:term_a (not asserted below)
|
||||||
|
inj.onStatus(T, AgentStatus.WORKING); // picked up
|
||||||
|
events.clear();
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE); // "first" completes AND "second" delivers, same cycle
|
||||||
|
|
||||||
|
assertEquals(List.of("onTurnComplete:" + T, "register:" + T), events,
|
||||||
|
"onTurnComplete for the previous turn (\"first\") must run before register for the "
|
||||||
|
+ "new turn (\"second\") within the same onStatus cycle — reversing this order "
|
||||||
|
+ "is the CB-116 cross-turn stale reply fleetd #553 fixed, and this redesign "
|
||||||
|
+ "must not reopen it");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aRuntimeExceptionFromOnTurnCompleteStillLeavesTheNextDeliveryRegisteredOnTheRecoveryPath() {
|
||||||
|
// fleetd #556 rework (PR #566, comment 17009): there are TWO registrar.register call sites
|
||||||
|
// in the delivery method — the ordinary path inside `if (sent != null)`, and the fleetd
|
||||||
|
// #553 `finally` backstop, reached only when an earlier block (here, onTurnComplete for the
|
||||||
|
// PREVIOUS turn) throws before the ordinary path ever runs. The existing #553 regression
|
||||||
|
// test for this exact scenario
|
||||||
|
// (aRuntimeExceptionFromOnTurnCompleteStillCompletesTheNextDelivery, above) asserts only
|
||||||
|
// that "second"'s DELIVERED FUTURE completes — never that "second" is REGISTERED with
|
||||||
|
// CompletionResolver — so a redesign that dropped registrar.register() from the backstop
|
||||||
|
// passed every existing test while reopening this ticket's own defect on the one path
|
||||||
|
// fleetd #553 exists for. This test closes that gap: on the recovery path, the backstop is
|
||||||
|
// the ONLY thing that registers "second", so its waiter must still be resolvable afterward.
|
||||||
|
RuntimeException boom = new RuntimeException("fleetd #556 rework: boom from onTurnComplete");
|
||||||
|
TurnListener throwing = new TurnListener() {
|
||||||
|
@Override
|
||||||
|
public void onTurnComplete(String target) {
|
||||||
|
throw boom;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
Rendezvous rendezvous = new Rendezvous();
|
||||||
|
CompletionResolver completion = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||||
|
ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||||
|
Injector inj = new Injector(new AgentControl(herdr), throwing, _ -> true, _ -> {
|
||||||
|
}, completion::register);
|
||||||
|
|
||||||
|
inj.enqueue(T, "first", TestTurnTokens.inert(T));
|
||||||
|
CompletableFuture<Rendezvous.Resolution> secondWaiter = rendezvous.open(T);
|
||||||
|
inj.enqueue(T, "second", new TurnToken(T, secondWaiter));
|
||||||
|
|
||||||
|
inj.onStatus(T, AgentStatus.IDLE); // delivers "first" (inert token — nothing to register)
|
||||||
|
inj.onStatus(T, AgentStatus.WORKING); // picked up
|
||||||
|
|
||||||
|
RuntimeException thrown = assertThrows(RuntimeException.class,
|
||||||
|
// "first" completes (onTurnComplete throws) before "second"'s ordinary delivery
|
||||||
|
// path ever runs; only the finally backstop is left to register "second".
|
||||||
|
() -> inj.onStatus(T, AgentStatus.IDLE),
|
||||||
|
"the throwable from onTurnComplete must still escape onStatus");
|
||||||
|
assertSame(boom, thrown);
|
||||||
|
|
||||||
|
CompletionResolver.InFlight inFlight = completion.inFlight(T);
|
||||||
|
assertNotNull(inFlight,
|
||||||
|
"fleetd #556: \"second\" must be registered by the fleetd #553 finally backstop "
|
||||||
|
+ "even though onTurnComplete threw for \"first\"'s completion before the "
|
||||||
|
+ "ordinary registration path ever ran for \"second\"");
|
||||||
|
assertSame(secondWaiter, inFlight.waiter(),
|
||||||
|
"the registered entry must carry \"second\"'s own waiter, not some other value");
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,61 @@
|
|||||||
|
package dev.ltms.fleet.inject;
|
||||||
|
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
|
||||||
|
/** fleetd #544: the three-state contract in isolation from either loop that composes it. */
|
||||||
|
class LoopWatchdogTest {
|
||||||
|
|
||||||
|
private static final long STALE_AFTER_NANOS = 1000;
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void reportsRunningBeforeTheThresholdElapses() {
|
||||||
|
AtomicLong clock = new AtomicLong(0);
|
||||||
|
LoopWatchdog watchdog = new LoopWatchdog(clock::get, STALE_AFTER_NANOS);
|
||||||
|
clock.set(STALE_AFTER_NANOS - 1);
|
||||||
|
assertEquals(LoopWatchdog.State.RUNNING, watchdog.state());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void reportsStalledOnceTheLastRoundAgesPastTheThreshold() {
|
||||||
|
AtomicLong clock = new AtomicLong(0);
|
||||||
|
LoopWatchdog watchdog = new LoopWatchdog(clock::get, STALE_AFTER_NANOS);
|
||||||
|
clock.set(STALE_AFTER_NANOS);
|
||||||
|
assertEquals(LoopWatchdog.State.STALLED, watchdog.state());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void recordRoundCompleteResetsTheStaleClock() {
|
||||||
|
AtomicLong clock = new AtomicLong(0);
|
||||||
|
LoopWatchdog watchdog = new LoopWatchdog(clock::get, STALE_AFTER_NANOS);
|
||||||
|
clock.set(STALE_AFTER_NANOS - 1);
|
||||||
|
watchdog.recordRoundComplete(); // last round is now "now" again
|
||||||
|
clock.set(2 * STALE_AFTER_NANOS - 2);
|
||||||
|
assertEquals(LoopWatchdog.State.RUNNING, watchdog.state(),
|
||||||
|
"a fresh round completion must push the staleness deadline forward");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void markStoppedByCallerReportsStoppedRegardlessOfStaleness() {
|
||||||
|
AtomicLong clock = new AtomicLong(0);
|
||||||
|
LoopWatchdog watchdog = new LoopWatchdog(clock::get, STALE_AFTER_NANOS);
|
||||||
|
watchdog.markStoppedByCaller();
|
||||||
|
clock.set(STALE_AFTER_NANOS * 1000);
|
||||||
|
assertEquals(LoopWatchdog.State.STOPPED, watchdog.state(),
|
||||||
|
"an intentional stop must never be reported as STALLED");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void resetClearsAPreviousStopAndTheStaleClock() {
|
||||||
|
AtomicLong clock = new AtomicLong(0);
|
||||||
|
LoopWatchdog watchdog = new LoopWatchdog(clock::get, STALE_AFTER_NANOS);
|
||||||
|
watchdog.markStoppedByCaller();
|
||||||
|
clock.set(STALE_AFTER_NANOS * 1000);
|
||||||
|
watchdog.reset();
|
||||||
|
assertEquals(LoopWatchdog.State.RUNNING, watchdog.state(),
|
||||||
|
"reset() must clear both the stop mark and the stale timestamp");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,99 @@
|
|||||||
|
package dev.ltms.fleet.inject;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.msg.TestTurnTokens;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.concurrent.CompletableFuture;
|
||||||
|
import java.util.concurrent.CountDownLatch;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.concurrent.atomic.AtomicBoolean;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotSame;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
class StatusPollerResilienceTest {
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void anErrorForOneTargetDoesNotStopPollingTheNextTarget() throws Exception {
|
||||||
|
FakeHerdr fake = new FakeHerdr().withAgent("worker", "term_b", "w2:p8", "w2:t8");
|
||||||
|
CountDownLatch errorThrown = new CountDownLatch(1);
|
||||||
|
AgentControl agents = new AgentControl(new ErrorOnceForFirstTarget(fake, errorThrown));
|
||||||
|
Injector injector = new Injector(agents);
|
||||||
|
StatusPoller poller = new StatusPoller(agents, injector, 1);
|
||||||
|
poller.start();
|
||||||
|
try {
|
||||||
|
injector.enqueue("term_a", "first", TestTurnTokens.inert("term_a"));
|
||||||
|
assertTrue(errorThrown.await(2, TimeUnit.SECONDS),
|
||||||
|
"the first target must throw its test Error");
|
||||||
|
|
||||||
|
CompletableFuture<Void> delivered =
|
||||||
|
injector.enqueue("term_b", "second", TestTurnTokens.inert("term_b")).completion();
|
||||||
|
delivered.get(2, TimeUnit.SECONDS);
|
||||||
|
} finally {
|
||||||
|
poller.stop();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void anAbnormalExitClearsRunningSoStartCreatesANewLoop() throws Exception {
|
||||||
|
StatusPoller poller = new StatusPoller(new AgentControl(new FakeHerdr()), new Injector(new AgentControl(new FakeHerdr())), -1);
|
||||||
|
poller.start();
|
||||||
|
Thread first = threadOf(poller);
|
||||||
|
first.join(2000);
|
||||||
|
assertFalse(runningOf(poller), "an abnormal loop exit must clear running");
|
||||||
|
|
||||||
|
poller.start();
|
||||||
|
Thread restarted = threadOf(poller);
|
||||||
|
try {
|
||||||
|
assertNotSame(first, restarted, "start() must create a new loop after an abnormal exit");
|
||||||
|
restarted.join(2000);
|
||||||
|
} finally {
|
||||||
|
poller.stop();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Thread threadOf(StatusPoller poller) throws ReflectiveOperationException {
|
||||||
|
Field field = StatusPoller.class.getDeclaredField("thread");
|
||||||
|
field.setAccessible(true);
|
||||||
|
return (Thread) field.get(poller);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static boolean runningOf(StatusPoller poller) throws ReflectiveOperationException {
|
||||||
|
Field field = StatusPoller.class.getDeclaredField("running");
|
||||||
|
field.setAccessible(true);
|
||||||
|
return field.getBoolean(poller);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class ErrorOnceForFirstTarget implements HerdrClient {
|
||||||
|
private final FakeHerdr delegate;
|
||||||
|
private final CountDownLatch errorThrown;
|
||||||
|
private final AtomicBoolean first = new AtomicBoolean(true);
|
||||||
|
|
||||||
|
private ErrorOnceForFirstTarget(FakeHerdr delegate, CountDownLatch errorThrown) {
|
||||||
|
this.delegate = delegate;
|
||||||
|
this.errorThrown = errorThrown;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public JsonNode call(String method, Object params) {
|
||||||
|
if (method.equals("agent.get") && params instanceof Map<?, ?> map
|
||||||
|
&& "w2:p7".equals(map.get("target")) && first.compareAndSet(true, false)) {
|
||||||
|
errorThrown.countDown();
|
||||||
|
throw new AssertionError("test Error from the first poll target");
|
||||||
|
}
|
||||||
|
return delegate.call(method, params);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
delegate.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,172 @@
|
|||||||
|
package dev.ltms.fleet.inject;
|
||||||
|
|
||||||
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
|
import dev.ltms.fleet.msg.TestTurnTokens;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.util.concurrent.CountDownLatch;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #544: {@link StatusPoller#health()} — the progress watchdog, not a liveness check. Every
|
||||||
|
* test drives {@link LoopWatchdog}'s clock explicitly (never real elapsed time), so the threshold
|
||||||
|
* crossing is deterministic rather than a race against the poll interval.
|
||||||
|
*/
|
||||||
|
class StatusPollerWatchdogTest {
|
||||||
|
|
||||||
|
private static final long INTERVAL_MILLIS = 50;
|
||||||
|
// Mirrors StatusPoller.staleAfterNanos(50): 50ms * the 40x multiplier = 2000ms.
|
||||||
|
private static final long STALE_AFTER_NANOS = TimeUnit.MILLISECONDS.toNanos(INTERVAL_MILLIS) * 40;
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aLoopParkedInAHerdrCallThatNeverReturnsReportsStalled() throws Exception {
|
||||||
|
CountDownLatch entered = new CountDownLatch(1);
|
||||||
|
AgentControl agents = new AgentControl(new BlocksOnAgentGet(entered));
|
||||||
|
Injector injector = new Injector(agents);
|
||||||
|
AtomicLong clock = new AtomicLong(0);
|
||||||
|
StatusPoller poller = new StatusPoller(agents, injector, new StatusRefiner(agents),
|
||||||
|
INTERVAL_MILLIS, clock::get);
|
||||||
|
poller.start();
|
||||||
|
try {
|
||||||
|
injector.enqueue("w1:p1", "hello", TestTurnTokens.inert("w1:p1"));
|
||||||
|
assertTrue(entered.await(2, TimeUnit.SECONDS),
|
||||||
|
"the poller must have entered the blocking herdr call");
|
||||||
|
|
||||||
|
assertEquals(LoopWatchdog.State.RUNNING, poller.health(),
|
||||||
|
"freshly parked, before the threshold elapses, must still read RUNNING");
|
||||||
|
|
||||||
|
clock.set(STALE_AFTER_NANOS + 1);
|
||||||
|
assertEquals(LoopWatchdog.State.STALLED, poller.health(),
|
||||||
|
"a loop parked mid-round past the staleness threshold must report STALLED");
|
||||||
|
} finally {
|
||||||
|
poller.stop();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aLoopThatDiedReportsStalled() throws Exception {
|
||||||
|
AtomicLong clock = new AtomicLong(0);
|
||||||
|
// intervalMillis=-1: Thread.sleep(-1) throws IllegalArgumentException, uncaught by loop(),
|
||||||
|
// which kills the carrier thread — the same "died" shape StatusPollerResilienceTest uses.
|
||||||
|
StatusPoller poller = new StatusPoller(new AgentControl(new IdleHerdr()), new Injector(new AgentControl(new IdleHerdr())),
|
||||||
|
new StatusRefiner(new AgentControl(new IdleHerdr())), -1, clock::get);
|
||||||
|
poller.start();
|
||||||
|
Thread thread = threadOf(poller);
|
||||||
|
thread.join(2000);
|
||||||
|
assertFalse(thread.isAlive(), "the loop must have died from Thread.sleep(-1)");
|
||||||
|
|
||||||
|
// staleAfterNanos(-1) = TimeUnit.MILLISECONDS.toNanos(max(-1,1)) * 40 = 40ms.
|
||||||
|
clock.set(TimeUnit.MILLISECONDS.toNanos(1) * 40 + 1);
|
||||||
|
assertEquals(LoopWatchdog.State.STALLED, poller.health(),
|
||||||
|
"a dead loop's last-completed round goes stale and must report STALLED");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aLoopStoppedOnPurposeReportsStoppedNotStalled() throws Exception {
|
||||||
|
AtomicLong clock = new AtomicLong(0);
|
||||||
|
StatusPoller poller = new StatusPoller(new AgentControl(new IdleHerdr()), new Injector(new AgentControl(new IdleHerdr())),
|
||||||
|
new StatusRefiner(new AgentControl(new IdleHerdr())), INTERVAL_MILLIS, clock::get);
|
||||||
|
poller.start();
|
||||||
|
poller.stop();
|
||||||
|
|
||||||
|
// Advance the clock far past any staleness threshold — an intentional stop must never
|
||||||
|
// be reported as STALLED no matter how old the last round looks (fleetd #512 shape).
|
||||||
|
clock.set(STALE_AFTER_NANOS * 100);
|
||||||
|
assertEquals(LoopWatchdog.State.STOPPED, poller.health(),
|
||||||
|
"stop() must report STOPPED, never STALLED, however stale the last round looks");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aRestartedLoopReportsRunningAgainNotStoppedForever() throws Exception {
|
||||||
|
// fleetd #544 review (issue comment #16944): stoppedByCaller is sticky, and reset() —
|
||||||
|
// called only from start() — is the sole thing that clears it. start() is documented
|
||||||
|
// idempotent and loop()'s own error line says "it can be restarted", so stop() followed
|
||||||
|
// by start() is an anticipated path. Without the reset() call in start(), health() would
|
||||||
|
// report STOPPED forever after a restart even though the loop is genuinely running again.
|
||||||
|
AtomicLong clock = new AtomicLong(0);
|
||||||
|
StatusPoller poller = new StatusPoller(new AgentControl(new IdleHerdr()), new Injector(new AgentControl(new IdleHerdr())),
|
||||||
|
new StatusRefiner(new AgentControl(new IdleHerdr())), INTERVAL_MILLIS, clock::get);
|
||||||
|
poller.start();
|
||||||
|
poller.stop();
|
||||||
|
poller.start();
|
||||||
|
|
||||||
|
assertEquals(LoopWatchdog.State.RUNNING, poller.health(),
|
||||||
|
"an intentional stop must not outlive the restart that follows it");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aHealthyLoopReportsRunning() throws Exception {
|
||||||
|
// Real elapsed time on purpose, unlike the other tests here: a clock frozen at 0 would read
|
||||||
|
// RUNNING even if recordRoundComplete() were never wired into loop() at all, since the
|
||||||
|
// constructor's own initial timestamp already satisfies "not stale yet". Running for several
|
||||||
|
// multiples of this instance's own threshold (5ms interval * 40 = 200ms) and still reading
|
||||||
|
// RUNNING instead proves the loop's repeated round completions are the thing keeping the
|
||||||
|
// staleness deadline pushed forward.
|
||||||
|
long fastIntervalMillis = 5;
|
||||||
|
StatusPoller poller = new StatusPoller(new AgentControl(new IdleHerdr()), new Injector(new AgentControl(new IdleHerdr())),
|
||||||
|
new StatusRefiner(new AgentControl(new IdleHerdr())), fastIntervalMillis, System::nanoTime);
|
||||||
|
poller.start();
|
||||||
|
try {
|
||||||
|
Thread.sleep(800);
|
||||||
|
assertEquals(LoopWatchdog.State.RUNNING, poller.health(),
|
||||||
|
"a healthy loop's repeated round completions must keep pushing the staleness deadline forward");
|
||||||
|
} finally {
|
||||||
|
poller.stop();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Thread threadOf(StatusPoller poller) throws ReflectiveOperationException {
|
||||||
|
Field field = StatusPoller.class.getDeclaredField("thread");
|
||||||
|
field.setAccessible(true);
|
||||||
|
return (Thread) field.get(poller);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Never has any active target, so the loop just idles and completes empty rounds. */
|
||||||
|
private static final class IdleHerdr implements HerdrClient {
|
||||||
|
@Override
|
||||||
|
public JsonNode call(String method, Object params) {
|
||||||
|
throw new AssertionError("unexpected herdr call: " + method);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class BlocksOnAgentGet implements HerdrClient {
|
||||||
|
private final CountDownLatch entered;
|
||||||
|
|
||||||
|
private BlocksOnAgentGet(CountDownLatch entered) {
|
||||||
|
this.entered = entered;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public JsonNode call(String method, Object params) {
|
||||||
|
if ("agent.get".equals(method)) {
|
||||||
|
entered.countDown();
|
||||||
|
try {
|
||||||
|
// Blocks forever, mirroring UnixSocketHerdrClient.readLine()'s deadline-free
|
||||||
|
// read (fleetd #544) — a test double standing in for a parked socket, not a
|
||||||
|
// real one.
|
||||||
|
new CountDownLatch(1).await();
|
||||||
|
} catch (InterruptedException e) {
|
||||||
|
Thread.currentThread().interrupt();
|
||||||
|
}
|
||||||
|
throw new AssertionError("unreachable: only stop()'s interrupt reaches here");
|
||||||
|
}
|
||||||
|
throw new AssertionError("unexpected herdr call: " + method);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,23 +1,22 @@
|
|||||||
package dev.ltms.fleet.lead;
|
package dev.ltms.fleet.lead;
|
||||||
|
|
||||||
import ch.qos.logback.classic.Level;
|
import ch.qos.logback.classic.Level;
|
||||||
import ch.qos.logback.classic.Logger;
|
|
||||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
import ch.qos.logback.core.read.ListAppender;
|
|
||||||
import com.fasterxml.jackson.databind.JsonNode;
|
import com.fasterxml.jackson.databind.JsonNode;
|
||||||
import dev.ltms.fleet.config.FleetConfig;
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
import dev.ltms.fleet.herdr.AgentControl;
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
import dev.ltms.fleet.herdr.HerdrClient;
|
import dev.ltms.fleet.herdr.HerdrClient;
|
||||||
import dev.ltms.fleet.herdr.HerdrException;
|
import dev.ltms.fleet.herdr.HerdrException;
|
||||||
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
import org.junit.jupiter.api.DisplayName;
|
import org.junit.jupiter.api.DisplayName;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
import org.junit.jupiter.api.io.TempDir;
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
import org.slf4j.LoggerFactory;
|
|
||||||
|
|
||||||
import java.io.IOException;
|
import java.io.IOException;
|
||||||
import java.nio.file.Files;
|
import java.nio.file.Files;
|
||||||
import java.nio.file.Path;
|
import java.nio.file.Path;
|
||||||
|
import java.util.List;
|
||||||
import java.util.concurrent.atomic.AtomicLong;
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
import java.util.function.Function;
|
import java.util.function.Function;
|
||||||
import java.util.function.LongSupplier;
|
import java.util.function.LongSupplier;
|
||||||
@@ -862,25 +861,16 @@ class LeadRolloverTest {
|
|||||||
|
|
||||||
// ---- fleetd #494: the log lines must print MEASURED values, never the configured budget --
|
// ---- fleetd #494: the log lines must print MEASURED values, never the configured budget --
|
||||||
|
|
||||||
private static ListAppender<ILoggingEvent> attachLog() {
|
private static CapturedLog attachLog() {
|
||||||
Logger logger = (Logger) LoggerFactory.getLogger(LeadRollover.class);
|
return CapturedLog.at(LeadRollover.class, Level.DEBUG);
|
||||||
logger.setLevel(Level.DEBUG);
|
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.start();
|
|
||||||
logger.addAppender(appender);
|
|
||||||
return appender;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
private static void detachLog(ListAppender<ILoggingEvent> appender) {
|
private static ILoggingEvent lastEventContaining(List<ILoggingEvent> events, String substring) {
|
||||||
((Logger) LoggerFactory.getLogger(LeadRollover.class)).detachAppender(appender);
|
return events.stream()
|
||||||
}
|
|
||||||
|
|
||||||
private static ILoggingEvent lastEventContaining(ListAppender<ILoggingEvent> events, String substring) {
|
|
||||||
return events.list.stream()
|
|
||||||
.filter(e -> e.getFormattedMessage().contains(substring))
|
.filter(e -> e.getFormattedMessage().contains(substring))
|
||||||
.reduce((_, b) -> b)
|
.reduce((_, b) -> b)
|
||||||
.orElseThrow(() -> new AssertionError("no log event contained \"" + substring
|
.orElseThrow(() -> new AssertionError("no log event contained \"" + substring
|
||||||
+ "\"; got: " + events.list.stream().map(ILoggingEvent::getFormattedMessage).toList()));
|
+ "\"; got: " + events.stream().map(ILoggingEvent::getFormattedMessage).toList()));
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -914,15 +904,14 @@ class LeadRolloverTest {
|
|||||||
AtomicLong clock = new AtomicLong(1_000);
|
AtomicLong clock = new AtomicLong(1_000);
|
||||||
LeadRollover rollover = newRollover(flipsAfterClear, config, () -> clock.addAndGet(500));
|
LeadRollover rollover = newRollover(flipsAfterClear, config, () -> clock.addAndGet(500));
|
||||||
|
|
||||||
ListAppender<ILoggingEvent> events = attachLog();
|
try (CapturedLog log = attachLog()) {
|
||||||
try {
|
|
||||||
LeadRollover.PendingRollover pending = rollover.open(LEAD, "context is full");
|
LeadRollover.PendingRollover pending = rollover.open(LEAD, "context is full");
|
||||||
LeadRollover.RollDecision decision = rollover.confirm(LEAD, pending.token(), true);
|
LeadRollover.RollDecision decision = rollover.confirm(LEAD, pending.token(), true);
|
||||||
|
|
||||||
assertTrue(decision.accepted(), "every synchronous gate passes; the refusal is logged "
|
assertTrue(decision.accepted(), "every synchronous gate passes; the refusal is logged "
|
||||||
+ "only, deep inside the deferred continuation");
|
+ "only, deep inside the deferred continuation");
|
||||||
|
|
||||||
ILoggingEvent event = lastEventContaining(events, "NOT sending bootstrapText");
|
ILoggingEvent event = lastEventContaining(log.events(), "NOT sending bootstrapText");
|
||||||
assertEquals(Level.WARN, event.getLevel());
|
assertEquals(Level.WARN, event.getLevel());
|
||||||
String message = event.getFormattedMessage();
|
String message = event.getFormattedMessage();
|
||||||
assertTrue(message.contains("configured=1s"), "must label the configured budget: " + message);
|
assertTrue(message.contains("configured=1s"), "must label the configured budget: " + message);
|
||||||
@@ -934,8 +923,6 @@ class LeadRolloverTest {
|
|||||||
+ message);
|
+ message);
|
||||||
assertFalse(message.contains("within 1s"), "must not present the configured budget as if "
|
assertFalse(message.contains("within 1s"), "must not present the configured budget as if "
|
||||||
+ "it were the measured wait duration: " + message);
|
+ "it were the measured wait duration: " + message);
|
||||||
} finally {
|
|
||||||
detachLog(events);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -954,15 +941,14 @@ class LeadRolloverTest {
|
|||||||
AtomicLong clock = new AtomicLong(1_000);
|
AtomicLong clock = new AtomicLong(1_000);
|
||||||
LeadRollover rollover = newRollover(herdr, config, () -> clock.addAndGet(500));
|
LeadRollover rollover = newRollover(herdr, config, () -> clock.addAndGet(500));
|
||||||
|
|
||||||
ListAppender<ILoggingEvent> events = attachLog();
|
try (CapturedLog log = attachLog()) {
|
||||||
try {
|
|
||||||
LeadRollover.PendingRollover pending = rollover.open(LEAD, "context is full");
|
LeadRollover.PendingRollover pending = rollover.open(LEAD, "context is full");
|
||||||
LeadRollover.RollDecision decision = rollover.confirm(LEAD, pending.token(), true);
|
LeadRollover.RollDecision decision = rollover.confirm(LEAD, pending.token(), true);
|
||||||
|
|
||||||
assertTrue(decision.accepted(), "expected approval; got: " + decision.reason()
|
assertTrue(decision.accepted(), "expected approval; got: " + decision.reason()
|
||||||
+ " / " + decision.detail());
|
+ " / " + decision.detail());
|
||||||
|
|
||||||
ILoggingEvent event = lastEventContaining(events, "releasing rather than wedging the roll");
|
ILoggingEvent event = lastEventContaining(log.events(), "releasing rather than wedging the roll");
|
||||||
assertEquals(Level.WARN, event.getLevel(), "the grace-limit release must be WARN, not "
|
assertEquals(Level.WARN, event.getLevel(), "the grace-limit release must be WARN, not "
|
||||||
+ "INFO — it is exactly the case that reported false success in the real incident "
|
+ "INFO — it is exactly the case that reported false success in the real incident "
|
||||||
+ "this fix comes from (a roll that 'succeeded' after 438ms of a 20s budget)");
|
+ "this fix comes from (a roll that 'succeeded' after 438ms of a 20s budget)");
|
||||||
@@ -982,8 +968,6 @@ class LeadRolloverTest {
|
|||||||
assertTrue(message.contains("after 8 consecutive IDLE/DONE polls (7 of those were nudged)"),
|
assertTrue(message.contains("after 8 consecutive IDLE/DONE polls (7 of those were nudged)"),
|
||||||
"must print the measured poll count and nudge count as plain numbers, not the "
|
"must print the measured poll count and nudge count as plain numbers, not the "
|
||||||
+ "PICKUP_GRACE_POLLS constant standing in for either: " + message);
|
+ "PICKUP_GRACE_POLLS constant standing in for either: " + message);
|
||||||
} finally {
|
|
||||||
detachLog(events);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1000,15 +984,14 @@ class LeadRolloverTest {
|
|||||||
AtomicLong clock = new AtomicLong(1_000);
|
AtomicLong clock = new AtomicLong(1_000);
|
||||||
LeadRollover rollover = newRollover(herdr, config, () -> clock.addAndGet(500));
|
LeadRollover rollover = newRollover(herdr, config, () -> clock.addAndGet(500));
|
||||||
|
|
||||||
ListAppender<ILoggingEvent> events = attachLog();
|
try (CapturedLog log = attachLog()) {
|
||||||
try {
|
|
||||||
LeadRollover.PendingRollover pending = rollover.open(LEAD, "context is full");
|
LeadRollover.PendingRollover pending = rollover.open(LEAD, "context is full");
|
||||||
LeadRollover.RollDecision decision = rollover.confirm(LEAD, pending.token(), true);
|
LeadRollover.RollDecision decision = rollover.confirm(LEAD, pending.token(), true);
|
||||||
|
|
||||||
assertTrue(decision.accepted(), "expected approval; got: " + decision.reason()
|
assertTrue(decision.accepted(), "expected approval; got: " + decision.reason()
|
||||||
+ " / " + decision.detail());
|
+ " / " + decision.detail());
|
||||||
|
|
||||||
ILoggingEvent event = lastEventContaining(events, "lead-rollover: rolled");
|
ILoggingEvent event = lastEventContaining(log.events(), "lead-rollover: rolled");
|
||||||
assertEquals(Level.INFO, event.getLevel());
|
assertEquals(Level.INFO, event.getLevel());
|
||||||
String message = event.getFormattedMessage();
|
String message = event.getFormattedMessage();
|
||||||
// fleetd #494 follow-up: waitUntilAtTurnBoundary now also reads the injected clock one
|
// fleetd #494 follow-up: waitUntilAtTurnBoundary now also reads the injected clock one
|
||||||
@@ -1018,8 +1001,6 @@ class LeadRolloverTest {
|
|||||||
assertTrue(message.contains("elapsedMs=7000"), "must print the MEASURED elapsed time for "
|
assertTrue(message.contains("elapsedMs=7000"), "must print the MEASURED elapsed time for "
|
||||||
+ "the whole roll — with this fixture's advancing clock, the full roll (turn-settle "
|
+ "the whole roll — with this fixture's advancing clock, the full roll (turn-settle "
|
||||||
+ "wait + /clear wait + bootstrapText) took 7000ms: " + message);
|
+ "wait + /clear wait + bootstrapText) took 7000ms: " + message);
|
||||||
} finally {
|
|
||||||
detachLog(events);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1040,8 +1021,7 @@ class LeadRolloverTest {
|
|||||||
AtomicLong clock = new AtomicLong(1_000);
|
AtomicLong clock = new AtomicLong(1_000);
|
||||||
LeadRollover rollover = newRollover(herdr, config, () -> clock.addAndGet(500));
|
LeadRollover rollover = newRollover(herdr, config, () -> clock.addAndGet(500));
|
||||||
|
|
||||||
ListAppender<ILoggingEvent> events = attachLog();
|
try (CapturedLog log = attachLog()) {
|
||||||
try {
|
|
||||||
LeadRollover.PendingRollover pending = rollover.open(LEAD, "context is full");
|
LeadRollover.PendingRollover pending = rollover.open(LEAD, "context is full");
|
||||||
LeadRollover.RollDecision decision = rollover.confirm(LEAD, pending.token(), true);
|
LeadRollover.RollDecision decision = rollover.confirm(LEAD, pending.token(), true);
|
||||||
|
|
||||||
@@ -1049,7 +1029,7 @@ class LeadRolloverTest {
|
|||||||
+ "only inside the deferred continuation, which this test's synchronous runner "
|
+ "only inside the deferred continuation, which this test's synchronous runner "
|
||||||
+ "has already run to completion by the time confirm() returns");
|
+ "has already run to completion by the time confirm() returns");
|
||||||
|
|
||||||
ILoggingEvent event = lastEventContaining(events, "refusing to send /clear at all");
|
ILoggingEvent event = lastEventContaining(log.events(), "refusing to send /clear at all");
|
||||||
assertEquals(Level.WARN, event.getLevel());
|
assertEquals(Level.WARN, event.getLevel());
|
||||||
String message = event.getFormattedMessage();
|
String message = event.getFormattedMessage();
|
||||||
assertTrue(message.contains("configured=1s"), "must label the configured budget: " + message);
|
assertTrue(message.contains("configured=1s"), "must label the configured budget: " + message);
|
||||||
@@ -1058,8 +1038,6 @@ class LeadRolloverTest {
|
|||||||
+ "1s(=1000ms) configured budget: " + message);
|
+ "1s(=1000ms) configured budget: " + message);
|
||||||
assertFalse(message.contains("within 1s"), "must not present the configured budget as if "
|
assertFalse(message.contains("within 1s"), "must not present the configured budget as if "
|
||||||
+ "it were the measured wait duration: " + message);
|
+ "it were the measured wait duration: " + message);
|
||||||
} finally {
|
|
||||||
detachLog(events);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -57,6 +57,23 @@ class ConnectionIdentityTest {
|
|||||||
// must read as "resolved" — the distinction #317 turns on.
|
// must read as "resolved" — the distinction #317 turns on.
|
||||||
ConnectionIdentity.Caller c = with(_ -> 999_999).resolve("127.0.0.1", 55555);
|
ConnectionIdentity.Caller c = with(_ -> 999_999).resolve("127.0.0.1", 55555);
|
||||||
assertTrue(c.resolved());
|
assertTrue(c.resolved());
|
||||||
|
assertTrue(c.scanComplete(), "no herdr error happened, so the scan is complete");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void scanIsIncompleteWhenHerdrErrorsOnThePaneThatOwnsThePid() {
|
||||||
|
// fleetd #505: a transient herdr error on exactly the pane that DOES own the caller's pid
|
||||||
|
// must be visible as an incomplete scan, distinct from a real primary (resolved(), null
|
||||||
|
// terminal, complete scan). Both have pid > 0 and a null terminal — scanComplete is the
|
||||||
|
// only thing that tells them apart.
|
||||||
|
FakeHerdr failing = new FakeHerdr().processInfoFailsForPane("w2:p7", "transient");
|
||||||
|
ConnectionIdentity id = new ConnectionIdentity(new PaneLocator(failing), _ -> FakeHerdr.WORKER_PID);
|
||||||
|
|
||||||
|
ConnectionIdentity.Caller c = id.resolve("127.0.0.1", 55555);
|
||||||
|
|
||||||
|
assertTrue(c.resolved(), "the pid itself resolved fine — this is not #317's failure");
|
||||||
|
assertNull(c.terminal(), "the owning pane could not be confirmed");
|
||||||
|
assertFalse(c.scanComplete(), "the scan could not check the pane that owns this pid");
|
||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
|
|||||||
@@ -78,10 +78,15 @@ class FleetMcpAuthzTest {
|
|||||||
// fleetd #480 correction round: FleetMcp has one constructor now (no defaulting
|
// fleetd #480 correction round: FleetMcp has one constructor now (no defaulting
|
||||||
// overloads — see its javadoc), so every feature this test does not exercise is passed
|
// overloads — see its javadoc), so every feature this test does not exercise is passed
|
||||||
// its explicit "off" value here rather than being omitted.
|
// its explicit "off" value here rather than being omitted.
|
||||||
|
//
|
||||||
|
// fleetd #518: callers is now required (never null) either way — the resolver that used
|
||||||
|
// to be omitted to reach "legacy" is now always real, and AuthorizationMode is the
|
||||||
|
// separate, explicit choice that governs enforcement.
|
||||||
mcp = new FleetMcp(messages, workers, sessions, identity, sessions.asPresence(),
|
mcp = new FleetMcp(messages, workers, sessions, identity, sessions.asPresence(),
|
||||||
new PrimaryRegistry(null),
|
new PrimaryRegistry(null),
|
||||||
enforce ? CallerResolver.withLeadsAndMembers(identity, false, null,
|
CallerResolver.withLeadsAndMembers(identity, false, null,
|
||||||
Map::of, new MemberRegistry(null)) : null,
|
Map::of, new MemberRegistry(null)),
|
||||||
|
enforce ? FleetMcp.AuthorizationMode.ENFORCED : FleetMcp.AuthorizationMode.UNENFORCED,
|
||||||
metrics, FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
metrics, FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||||
FleetMcp.QuarantineSource.none(), null, FleetMcp.OutageSource.none(),
|
FleetMcp.QuarantineSource.none(), null, FleetMcp.OutageSource.none(),
|
||||||
FleetMcp.LeadSeatSource.none(), List.of(), null);
|
FleetMcp.LeadSeatSource.none(), List.of(), null);
|
||||||
@@ -187,7 +192,31 @@ class FleetMcpAuthzTest {
|
|||||||
// The 22 pre-existing FleetMcpTest cases rely on no authorization being enforced.
|
// The 22 pre-existing FleetMcpTest cases rely on no authorization being enforced.
|
||||||
FleetMcp m = mcp(false);
|
FleetMcp m = mcp(false);
|
||||||
assertNull(m.denyFor(ANON, Authz.Action.SPAWN, null),
|
assertNull(m.denyFor(ANON, Authz.Action.SPAWN, null),
|
||||||
"no CallerResolver supplied ⇒ authorization not enforced (legacy behaviour)");
|
"AuthorizationMode.UNENFORCED chosen explicitly ⇒ authorization not enforced "
|
||||||
|
+ "(legacy behaviour) — fleetd #518 replaced the old callers == null idiom");
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #509 was originally proven against {@code FleetMcp.legacyPrincipal} — a second,
|
||||||
|
* separately-maintained principal-resolution heuristic that only ran when {@code callers} was
|
||||||
|
* omitted (null). fleetd #518 deleted that whole heuristic: {@code callers} is now required
|
||||||
|
* and non-null under every {@link FleetMcp.AuthorizationMode}, so the ONE real
|
||||||
|
* {@link CallerResolver} resolves every caller, enforced or not, and #509's property (a
|
||||||
|
* non-loopback / unresolved caller must never earn the primary's authority) is exactly what
|
||||||
|
* {@code CallerResolverTest.aNonLoopbackCallerIsNeverThePrimaryUnderLoopbackTrust} already
|
||||||
|
* proves on that one real path. There is no longer a second heuristic here to test.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void anUnresolvedNonLoopbackCallerIsAnonymousUnderTheOneRealResolver() {
|
||||||
|
ConnectionIdentity identity = new ConnectionIdentity(new PaneLocator(herdr), _ -> 999_999);
|
||||||
|
CallerResolver resolver = CallerResolver.withLeadsAndMembers(identity, false, null,
|
||||||
|
Map::of, new MemberRegistry(null));
|
||||||
|
// A non-loopback address never even reaches the pane scan — resolve() short-circuits it
|
||||||
|
// to Caller(null, -1, true), the same "no terminal" shape a genuine primary's connection
|
||||||
|
// produces on loopback. The real resolver must not conflate the two.
|
||||||
|
Principal p = resolver.resolve("8.8.8.8", 1234, null);
|
||||||
|
assertEquals(Principal.anonymous(), p,
|
||||||
|
"an unresolved, non-loopback caller must earn no authority, not the primary's");
|
||||||
}
|
}
|
||||||
|
|
||||||
// --- fleetd #439: who may see fleet_list's coordinator row ----------------------------------
|
// --- fleetd #439: who may see fleet_list's coordinator row ----------------------------------
|
||||||
|
|||||||
@@ -0,0 +1,147 @@
|
|||||||
|
package dev.ltms.fleet.mcp;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.auth.CallerResolver;
|
||||||
|
import dev.ltms.fleet.auth.MemberRegistry;
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.PaneLocator;
|
||||||
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
|
import dev.ltms.fleet.inject.Injector;
|
||||||
|
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||||
|
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||||
|
import dev.ltms.fleet.msg.MessageService;
|
||||||
|
import dev.ltms.fleet.msg.Rendezvous;
|
||||||
|
import dev.ltms.fleet.session.FakeWorktrees;
|
||||||
|
import dev.ltms.fleet.session.SessionManager;
|
||||||
|
import io.modelcontextprotocol.client.McpClient;
|
||||||
|
import io.modelcontextprotocol.client.McpSyncClient;
|
||||||
|
import io.modelcontextprotocol.client.transport.HttpClientStreamableHttpTransport;
|
||||||
|
import io.modelcontextprotocol.spec.McpClientTransport;
|
||||||
|
import io.modelcontextprotocol.spec.McpSchema;
|
||||||
|
import org.eclipse.jetty.server.Server;
|
||||||
|
import org.eclipse.jetty.server.ServerConnector;
|
||||||
|
import org.eclipse.jetty.servlet.ServletContextHandler;
|
||||||
|
import org.eclipse.jetty.servlet.ServletHolder;
|
||||||
|
import org.junit.jupiter.api.AfterEach;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.net.http.HttpRequest;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Set;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #518 — Part 2: drive the {@code contextExtractor} closure for real.
|
||||||
|
*
|
||||||
|
* <p>{@code FleetMcp.deny()}/{@code denyFor()} has a full policy table of tests
|
||||||
|
* ({@code FleetMcpAuthzTest}), and {@code CallerResolver.resolve()} has its own full suite
|
||||||
|
* ({@code CallerResolverTest}). Neither one ever exercises the closure that WIRES them together
|
||||||
|
* inside {@code FleetMcp}'s constructor: it is built once, handed to the MCP SDK's transport, and
|
||||||
|
* only ever runs when a real MCP client makes a real HTTP request. Every existing test either
|
||||||
|
* calls {@code denyFor(Principal, ...)} with a hand-built {@link dev.ltms.fleet.auth.Principal}
|
||||||
|
* (never asking who the transport would actually have resolved) or drives a static handler method
|
||||||
|
* directly. A mutation that swapped the whole resolution decision for an unconditional fallback —
|
||||||
|
* bypassing {@link CallerResolver} entirely — passed the full suite, including every
|
||||||
|
* {@code FleetMcpAuthzTest} case, because none of them go through the transport at all.
|
||||||
|
*
|
||||||
|
* <p>This test boots the real {@code HttpServletStreamableServerTransportProvider} on a real
|
||||||
|
* Jetty server, drives it with a real MCP client over HTTP, and checks a result that only the
|
||||||
|
* real {@link CallerResolver} can produce: token-mode inspects the {@code Authorization} header
|
||||||
|
* and grants {@code PRIMARY} only for the right bearer token. The connection never resolves to a
|
||||||
|
* worker pane (the fake peer-pid lookup always misses), so the ONLY way {@code fleet_whoami} can
|
||||||
|
* come back as {@code primary} is if the closure actually called {@code callers.resolve(...)} and
|
||||||
|
* read that header — a behaviour the deleted {@code legacyPrincipal} heuristic never had at all.
|
||||||
|
*/
|
||||||
|
class FleetMcpContextExtractorTest {
|
||||||
|
|
||||||
|
private static final String TOKEN = "s3cret-mcp-token";
|
||||||
|
|
||||||
|
private final FakeHerdr herdr = new FakeHerdr();
|
||||||
|
private final AgentControl agents = new AgentControl(herdr);
|
||||||
|
private FleetMcp mcp;
|
||||||
|
private Server server;
|
||||||
|
|
||||||
|
@AfterEach
|
||||||
|
void tearDown() throws Exception {
|
||||||
|
if (server != null) {
|
||||||
|
server.stop();
|
||||||
|
}
|
||||||
|
if (mcp != null) {
|
||||||
|
mcp.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aRealMcpRequestIsResolvedByTheRealCallerResolverNotAFallback() throws Exception {
|
||||||
|
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||||
|
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN", null,
|
||||||
|
"tab", "fleetd-workers", "worker: {profile} #{n}", null, null, null);
|
||||||
|
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(agents, new WorkspaceControl(herdr),
|
||||||
|
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||||
|
_ -> "tok");
|
||||||
|
SessionManager sessions = new SessionManager(workers, new FakeWorktrees());
|
||||||
|
MessageService messages = new MessageService(agents, new Injector(agents), new Rendezvous(),
|
||||||
|
new InMemoryReplyInbox());
|
||||||
|
// The peer-pid lookup always misses (-1), so no connection here is ever resolved to a
|
||||||
|
// worker pane — every call falls through to CallerResolver's token check, the one branch
|
||||||
|
// that is unreachable through the deleted legacy heuristic.
|
||||||
|
ConnectionIdentity identity = new ConnectionIdentity(new PaneLocator(herdr), _ -> -1);
|
||||||
|
CallerResolver callers = CallerResolver.withLeadsAndMembers(identity, true, TOKEN,
|
||||||
|
Map::of, new MemberRegistry(null));
|
||||||
|
|
||||||
|
mcp = new FleetMcp(messages, workers, sessions, identity, sessions.asPresence(),
|
||||||
|
new PrimaryRegistry(null), callers, FleetMcp.AuthorizationMode.ENFORCED,
|
||||||
|
null, FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||||
|
FleetMcp.QuarantineSource.none(), null, FleetMcp.OutageSource.none(),
|
||||||
|
FleetMcp.LeadSeatSource.none(), List.of(), null);
|
||||||
|
|
||||||
|
ServletContextHandler handler = new ServletContextHandler();
|
||||||
|
handler.setContextPath("/");
|
||||||
|
handler.addServlet(new ServletHolder(mcp.servlet()), "/mcp");
|
||||||
|
server = new Server(0);
|
||||||
|
server.setHandler(handler);
|
||||||
|
server.start();
|
||||||
|
String baseUrl = "http://127.0.0.1:"
|
||||||
|
+ ((ServerConnector) server.getConnectors()[0]).getLocalPort();
|
||||||
|
|
||||||
|
// The right bearer token: the real CallerResolver grants PRIMARY, which fleet_whoami's
|
||||||
|
// READ gate lets through.
|
||||||
|
McpSchema.CallToolResult authorized = callWhoami(baseUrl, "Bearer " + TOKEN);
|
||||||
|
assertFalse(authorized.isError(), "a valid bearer token must resolve as PRIMARY and pass "
|
||||||
|
+ "fleet_whoami's READ gate: " + textOf(authorized));
|
||||||
|
assertTrue(textOf(authorized).contains("\"role\":\"primary\""),
|
||||||
|
"fleet_whoami must report the role the real CallerResolver resolved over this "
|
||||||
|
+ "connection, not a fallback: " + textOf(authorized));
|
||||||
|
|
||||||
|
// No credential at all, over the SAME wiring: the real resolver refuses it as ANONYMOUS.
|
||||||
|
// legacyPrincipal never looked at the Authorization header, so it could not have told
|
||||||
|
// these two calls apart at all -- this is the assertion the deleted mutation would fail.
|
||||||
|
McpSchema.CallToolResult unauthorized = callWhoami(baseUrl, null);
|
||||||
|
assertTrue(unauthorized.isError(), "no credential must be refused, not silently let "
|
||||||
|
+ "through: " + textOf(unauthorized));
|
||||||
|
}
|
||||||
|
|
||||||
|
private static McpSchema.CallToolResult callWhoami(String baseUrl, String authorizationHeader) {
|
||||||
|
HttpRequest.Builder requestTemplate = HttpRequest.newBuilder();
|
||||||
|
if (authorizationHeader != null) {
|
||||||
|
requestTemplate.header("Authorization", authorizationHeader);
|
||||||
|
}
|
||||||
|
McpClientTransport transport = HttpClientStreamableHttpTransport.builder(baseUrl)
|
||||||
|
.endpoint("/mcp")
|
||||||
|
.requestBuilder(requestTemplate)
|
||||||
|
.build();
|
||||||
|
try (McpSyncClient client = McpClient.sync(transport).build()) {
|
||||||
|
client.initialize();
|
||||||
|
return client.callTool(McpSchema.CallToolRequest.builder("fleet_whoami").arguments(Map.of()).build());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static String textOf(McpSchema.CallToolResult r) {
|
||||||
|
return ((McpSchema.TextContent) r.content().getFirst()).text();
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -89,6 +89,7 @@ class FleetMcpHandoverTest {
|
|||||||
mcp = new FleetMcp(messages, workers, sessions, identity, sessions.asPresence(),
|
mcp = new FleetMcp(messages, workers, sessions, identity, sessions.asPresence(),
|
||||||
new PrimaryRegistry(null),
|
new PrimaryRegistry(null),
|
||||||
CallerResolver.withLeadsAndMembers(identity, false, null, Map::of, new MemberRegistry(null)),
|
CallerResolver.withLeadsAndMembers(identity, false, null, Map::of, new MemberRegistry(null)),
|
||||||
|
FleetMcp.AuthorizationMode.ENFORCED,
|
||||||
null, FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
null, FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||||
FleetMcp.QuarantineSource.none(), null, FleetMcp.OutageSource.none(),
|
FleetMcp.QuarantineSource.none(), null, FleetMcp.OutageSource.none(),
|
||||||
FleetMcp.LeadSeatSource.none(), List.of(), leadRollover);
|
FleetMcp.LeadSeatSource.none(), List.of(), leadRollover);
|
||||||
|
|||||||
@@ -7,7 +7,9 @@ import dev.ltms.fleet.auth.Role;
|
|||||||
import dev.ltms.fleet.config.FleetConfig;
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
import dev.ltms.fleet.herdr.AgentControl;
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.AgentStatus;
|
||||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||||
import dev.ltms.fleet.herdr.PaneLocator;
|
import dev.ltms.fleet.herdr.PaneLocator;
|
||||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
import dev.ltms.fleet.inject.Injector;
|
import dev.ltms.fleet.inject.Injector;
|
||||||
@@ -57,9 +59,10 @@ class FleetMcpTest {
|
|||||||
|
|
||||||
private final FakeHerdr herdr = new FakeHerdr();
|
private final FakeHerdr herdr = new FakeHerdr();
|
||||||
private final AgentControl agents = new AgentControl(herdr);
|
private final AgentControl agents = new AgentControl(herdr);
|
||||||
|
private final Injector injector = new Injector(agents);
|
||||||
private final Rendezvous rendezvous = new Rendezvous();
|
private final Rendezvous rendezvous = new Rendezvous();
|
||||||
private final InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
private final InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||||
private final MessageService messages = new MessageService(agents, new Injector(agents), rendezvous, inbox);
|
private final MessageService messages = new MessageService(agents, injector, rendezvous, inbox);
|
||||||
|
|
||||||
@BeforeEach
|
@BeforeEach
|
||||||
void setUp() {
|
void setUp() {
|
||||||
@@ -298,6 +301,35 @@ class FleetMcpTest {
|
|||||||
assertTrue(textOf(res).contains("no reply"), "got: " + textOf(res));
|
assertTrue(textOf(res).contains("no reply"), "got: " + textOf(res));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #571 (ticket CORRECTION 5): {@code formatReply}'s {@code TIMED_OUT_UNCONFIRMED} arm is
|
||||||
|
* the one message whose whole job is to stop a caller retrying a delivery that may already have
|
||||||
|
* arrived. Pin that its wording is actually distinct from the queued/working arm's retry
|
||||||
|
* invitation — a mutation that swapped this arm's text for that one still passed every other
|
||||||
|
* test in this suite, because nothing asserted the specific wording.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void sendTimesOutWithAnUnconfirmedNoteNotARetryInvitation() throws Exception {
|
||||||
|
herdr.agentSendFailsWith("send_failed");
|
||||||
|
CompletableFuture<McpSchema.CallToolResult> send = CompletableFuture.supplyAsync(
|
||||||
|
() -> FleetMcp.send(messages, T, "hi", 150L, null, Set.of()));
|
||||||
|
long deadline = System.currentTimeMillis() + 2000;
|
||||||
|
while (!rendezvous.isWaiting(T) && System.currentTimeMillis() < deadline) {
|
||||||
|
//noinspection BusyWait
|
||||||
|
Thread.sleep(5);
|
||||||
|
}
|
||||||
|
assertTrue(rendezvous.isWaiting(T), "send should have opened its rendezvous waiter");
|
||||||
|
injector.onStatus(T, AgentStatus.IDLE); // triggers the failing delivery attempt -> ATTEMPTED
|
||||||
|
|
||||||
|
McpSchema.CallToolResult res = send.get(5, TimeUnit.SECONDS);
|
||||||
|
String text = textOf(res);
|
||||||
|
assertNotEquals(Boolean.TRUE, res.isError(), "a timeout is informational, not a tool error");
|
||||||
|
assertTrue(text.contains("delivery unconfirmed"), "got: " + text);
|
||||||
|
assertFalse(text.contains("retry or poll status"),
|
||||||
|
"an unconfirmed delivery must not carry the queued/working arm's retry invitation — "
|
||||||
|
+ "a resend here can double-deliver the same brief: got " + text);
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void sendRejectsMissingArgs() {
|
void sendRejectsMissingArgs() {
|
||||||
assertTrue(FleetMcp.send(messages, null, "hi", null, null, Set.of()).isError());
|
assertTrue(FleetMcp.send(messages, null, "hi", null, null, Set.of()).isError());
|
||||||
@@ -1053,6 +1085,36 @@ class FleetMcpTest {
|
|||||||
assertFalse(out.contains("quarantinedForSeconds"), out);
|
assertFalse(out.contains("quarantinedForSeconds"), out);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void loopHealthReportsStalledStatusPoller() {
|
||||||
|
String out = loopHealth(LoopWatchdog.State.STALLED, LoopWatchdog.State.RUNNING);
|
||||||
|
assertTrue(out.contains("\"statusPoller\":\"STALLED\""),
|
||||||
|
"fleet_list must report a stalled StatusPoller: " + out);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void loopHealthReportsStoppedSessionReaperAsStopped() {
|
||||||
|
String out = loopHealth(LoopWatchdog.State.RUNNING, LoopWatchdog.State.STOPPED);
|
||||||
|
assertTrue(out.contains("\"sessionReaper\":\"STOPPED\""),
|
||||||
|
"fleet_list must report a deliberately stopped SessionReaper as STOPPED, not an alarm: " + out);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void loopHealthReportsRunningStatusPoller() {
|
||||||
|
String out = loopHealth(LoopWatchdog.State.RUNNING, LoopWatchdog.State.STOPPED);
|
||||||
|
assertTrue(out.contains("\"statusPoller\":\"RUNNING\""),
|
||||||
|
"fleet_list must report a running StatusPoller: " + out);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static String loopHealth(LoopWatchdog.State statusPoller, LoopWatchdog.State sessionReaper) {
|
||||||
|
FakeHerdr herdr = new FakeHerdr();
|
||||||
|
return textOf(FleetMcp.listFleet(workerService(herdr, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||||
|
new SessionManager(workerService(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"))), null,
|
||||||
|
FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||||
|
new FleetMcp.LoopHealthSource(() -> statusPoller, () -> sessionReaper),
|
||||||
|
FleetMcp.QuarantineSource.none(), Map.of(), ""));
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void capacityIncludesConfiguredProfileWithoutMembers() {
|
void capacityIncludesConfiguredProfileWithoutMembers() {
|
||||||
FakeHerdr h = new FakeHerdr();
|
FakeHerdr h = new FakeHerdr();
|
||||||
@@ -1875,7 +1937,7 @@ class FleetMcpTest {
|
|||||||
null, null, null, null, null, null);
|
null, null, null, null, null, null);
|
||||||
|
|
||||||
assertEquals(Boolean.TRUE, res.isError());
|
assertEquals(Boolean.TRUE, res.isError());
|
||||||
assertTrue(textOf(res).contains("architect, dev, reviewer"), textOf(res));
|
assertTrue(textOf(res).contains("architect, dev, hunter, reviewer"), textOf(res));
|
||||||
}
|
}
|
||||||
|
|
||||||
// ── CB-619 / fleetd #123: a spawn asking for a role its profile has no slot for must be
|
// ── CB-619 / fleetd #123: a spawn asking for a role its profile has no slot for must be
|
||||||
|
|||||||
@@ -556,6 +556,31 @@ class ClaudeCodeLauncherTest {
|
|||||||
"no --agent flag when the role has no agent-definition file");
|
"no --agent flag when the role has no agent-definition file");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void hunterRoleUsesItsAgentFileAndStopsUsingItWhenRemoved(@TempDir Path cwd) throws Exception {
|
||||||
|
FakeHerdr herdr = new FakeHerdr();
|
||||||
|
Path agentFile = Files.createDirectories(cwd.resolve(".claude/agents")).resolve("hunter.md");
|
||||||
|
Files.writeString(agentFile, "---\nname: hunter\n---\nSweep for defects.");
|
||||||
|
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||||
|
"sonnet", "http://gx00.gw:8000", null, null, "FLEETD_WORKER_TOKEN",
|
||||||
|
List.of("claude"), "tab", "fleetd-workers", "w #{n}", null, null, null);
|
||||||
|
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||||
|
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||||
|
|
||||||
|
svc.spawn(new SpawnRequest("sonnet", cwd.toString(), null, null, null, MemberRole.HUNTER));
|
||||||
|
|
||||||
|
List<String> args = spawnedArgs(herdr);
|
||||||
|
int flag = args.indexOf("--agent");
|
||||||
|
assertTrue(flag >= 0, "the hunter role reaches its agent-definition file: " + args);
|
||||||
|
assertEquals("hunter", args.get(flag + 1));
|
||||||
|
|
||||||
|
Files.delete(agentFile);
|
||||||
|
svc.spawn(new SpawnRequest("sonnet", cwd.toString(), null, null, null, MemberRole.HUNTER));
|
||||||
|
|
||||||
|
assertFalse(spawnedArgs(herdr).contains("--agent"),
|
||||||
|
"the hunter role no longer gets an agent when its file is removed");
|
||||||
|
}
|
||||||
|
|
||||||
private ClaudeCodeLauncher multiProfile(FakeHerdr herdr) {
|
private ClaudeCodeLauncher multiProfile(FakeHerdr herdr) {
|
||||||
FleetConfig.Profile gx10 = new FleetConfig.Profile("gx10", "http://gx10.gw:8000", "coder",
|
FleetConfig.Profile gx10 = new FleetConfig.Profile("gx10", "http://gx10.gw:8000", "coder",
|
||||||
null, "FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers", "w #{n}", null, null, null);
|
null, "FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers", "w #{n}", null, null, null);
|
||||||
|
|||||||
@@ -1,17 +1,17 @@
|
|||||||
package dev.ltms.fleet.msg;
|
package dev.ltms.fleet.msg;
|
||||||
|
|
||||||
import ch.qos.logback.classic.Level;
|
import ch.qos.logback.classic.Level;
|
||||||
import ch.qos.logback.classic.Logger;
|
|
||||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
import ch.qos.logback.core.read.ListAppender;
|
|
||||||
import com.rabbitmq.client.Channel;
|
import com.rabbitmq.client.Channel;
|
||||||
import com.rabbitmq.client.ConnectionFactory;
|
import com.rabbitmq.client.ConnectionFactory;
|
||||||
import com.rabbitmq.client.impl.DefaultExceptionHandler;
|
import com.rabbitmq.client.impl.DefaultExceptionHandler;
|
||||||
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import java.io.IOException;
|
import java.io.IOException;
|
||||||
import java.lang.reflect.Proxy;
|
import java.lang.reflect.Proxy;
|
||||||
|
import java.util.List;
|
||||||
import java.util.concurrent.atomic.AtomicInteger;
|
import java.util.concurrent.atomic.AtomicInteger;
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
@@ -32,21 +32,17 @@ class AmqpConnectionFailureLoggerTest {
|
|||||||
assertEquals(AmqpConnectionFailureLogger.REPLY_INBOX, inboxHandler.connectionName());
|
assertEquals(AmqpConnectionFailureLogger.REPLY_INBOX, inboxHandler.connectionName());
|
||||||
assertEquals(AmqpConnectionFailureLogger.LEAD_MAILBOX, mailboxHandler.connectionName());
|
assertEquals(AmqpConnectionFailureLogger.LEAD_MAILBOX, mailboxHandler.connectionName());
|
||||||
|
|
||||||
ListAppender<ILoggingEvent> inboxEvents = attach(AmqpReplyInbox.class);
|
|
||||||
ListAppender<ILoggingEvent> mailboxEvents = attach(LeadMailbox.class);
|
|
||||||
IllegalStateException inboxFailure = new IllegalStateException("inbox failure");
|
IllegalStateException inboxFailure = new IllegalStateException("inbox failure");
|
||||||
IllegalStateException mailboxFailure = new IllegalStateException("mailbox failure");
|
IllegalStateException mailboxFailure = new IllegalStateException("mailbox failure");
|
||||||
try {
|
try (CapturedLog inboxLog = attach(AmqpReplyInbox.class);
|
||||||
|
CapturedLog mailboxLog = attach(LeadMailbox.class)) {
|
||||||
inboxHandler.handleUnexpectedConnectionDriverException(null, inboxFailure);
|
inboxHandler.handleUnexpectedConnectionDriverException(null, inboxFailure);
|
||||||
mailboxHandler.handleConnectionRecoveryException(null, mailboxFailure);
|
mailboxHandler.handleConnectionRecoveryException(null, mailboxFailure);
|
||||||
|
|
||||||
assertError(inboxEvents, "AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred",
|
assertError(inboxLog.events(), "AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred",
|
||||||
inboxFailure, "inbox failure line");
|
inboxFailure, "inbox failure line");
|
||||||
assertError(mailboxEvents, "AMQP connection fleetd-lead-mailbox: Caught an exception during connection recovery!",
|
assertError(mailboxLog.events(), "AMQP connection fleetd-lead-mailbox: Caught an exception during connection recovery!",
|
||||||
mailboxFailure, "mailbox recovery line");
|
mailboxFailure, "mailbox recovery line");
|
||||||
} finally {
|
|
||||||
detach(AmqpReplyInbox.class, inboxEvents);
|
|
||||||
detach(LeadMailbox.class, mailboxEvents);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -54,17 +50,14 @@ class AmqpConnectionFailureLoggerTest {
|
|||||||
void connectionResetKeepsForgivingHandlerWarningSemantics() {
|
void connectionResetKeepsForgivingHandlerWarningSemantics() {
|
||||||
AmqpConnectionFailureLogger handler = new AmqpConnectionFailureLogger(
|
AmqpConnectionFailureLogger handler = new AmqpConnectionFailureLogger(
|
||||||
AmqpConnectionFailureLogger.REPLY_INBOX, LoggerFactory.getLogger(AmqpReplyInbox.class));
|
AmqpConnectionFailureLogger.REPLY_INBOX, LoggerFactory.getLogger(AmqpReplyInbox.class));
|
||||||
ListAppender<ILoggingEvent> events = attach(AmqpReplyInbox.class);
|
try (CapturedLog log = attach(AmqpReplyInbox.class)) {
|
||||||
try {
|
|
||||||
handler.handleUnexpectedConnectionDriverException(null, new IOException("Connection reset"));
|
handler.handleUnexpectedConnectionDriverException(null, new IOException("Connection reset"));
|
||||||
assertEquals(1, events.list.size(), "the handler must still log a reset");
|
assertEquals(1, log.events().size(), "the handler must still log a reset");
|
||||||
ILoggingEvent event = events.list.getFirst();
|
ILoggingEvent event = log.events().getFirst();
|
||||||
assertEquals(Level.WARN, event.getLevel(), "ForgivingExceptionHandler logs connection resets at WARN");
|
assertEquals(Level.WARN, event.getLevel(), "ForgivingExceptionHandler logs connection resets at WARN");
|
||||||
assertEquals("AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred "
|
assertEquals("AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred "
|
||||||
+ "(Exception message: Connection reset)", event.getFormattedMessage());
|
+ "(Exception message: Connection reset)", event.getFormattedMessage());
|
||||||
assertTrue(event.getThrowableProxy() == null, "ForgivingExceptionHandler does not attach a reset stack trace");
|
assertTrue(event.getThrowableProxy() == null, "ForgivingExceptionHandler does not attach a reset stack trace");
|
||||||
} finally {
|
|
||||||
detach(AmqpReplyInbox.class, events);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -103,17 +96,8 @@ class AmqpConnectionFailureLoggerTest {
|
|||||||
"all exception-handling methods must remain inherited from DefaultExceptionHandler");
|
"all exception-handling methods must remain inherited from DefaultExceptionHandler");
|
||||||
}
|
}
|
||||||
|
|
||||||
private static ListAppender<ILoggingEvent> attach(Class<?> owner) {
|
private static CapturedLog attach(Class<?> owner) {
|
||||||
Logger logger = (Logger) LoggerFactory.getLogger(owner);
|
return CapturedLog.at(owner, Level.DEBUG);
|
||||||
logger.setLevel(Level.DEBUG);
|
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.start();
|
|
||||||
logger.addAppender(appender);
|
|
||||||
return appender;
|
|
||||||
}
|
|
||||||
|
|
||||||
private static void detach(Class<?> owner, ListAppender<ILoggingEvent> appender) {
|
|
||||||
((Logger) LoggerFactory.getLogger(owner)).detachAppender(appender);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
private static AmqpConnectionFailureLogger installedStrictHandler(ConnectionFactory factory, String connection) {
|
private static AmqpConnectionFailureLogger installedStrictHandler(ConnectionFactory factory, String connection) {
|
||||||
@@ -123,9 +107,9 @@ class AmqpConnectionFailureLoggerTest {
|
|||||||
return assertInstanceOf(AmqpConnectionFailureLogger.class, factory.getExceptionHandler());
|
return assertInstanceOf(AmqpConnectionFailureLogger.class, factory.getExceptionHandler());
|
||||||
}
|
}
|
||||||
|
|
||||||
private static void assertError(ListAppender<ILoggingEvent> events, String message, Throwable cause, String name) {
|
private static void assertError(List<ILoggingEvent> events, String message, Throwable cause, String name) {
|
||||||
assertEquals(1, events.list.size(), name);
|
assertEquals(1, events.size(), name);
|
||||||
ILoggingEvent event = events.list.getFirst();
|
ILoggingEvent event = events.getFirst();
|
||||||
assertEquals(Level.ERROR, event.getLevel(), name);
|
assertEquals(Level.ERROR, event.getLevel(), name);
|
||||||
assertEquals(message, event.getFormattedMessage(), name);
|
assertEquals(message, event.getFormattedMessage(), name);
|
||||||
assertEquals(cause.toString(), event.getThrowableProxy().getClassName() + ": "
|
assertEquals(cause.toString(), event.getThrowableProxy().getClassName() + ": "
|
||||||
|
|||||||
@@ -8,7 +8,10 @@ import org.junit.jupiter.api.Test;
|
|||||||
import org.junit.jupiter.api.Timeout;
|
import org.junit.jupiter.api.Timeout;
|
||||||
|
|
||||||
import java.lang.reflect.InvocationHandler;
|
import java.lang.reflect.InvocationHandler;
|
||||||
|
import java.lang.reflect.Method;
|
||||||
import java.lang.reflect.Proxy;
|
import java.lang.reflect.Proxy;
|
||||||
|
import java.io.IOException;
|
||||||
|
import java.util.Map;
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
import java.util.concurrent.CopyOnWriteArrayList;
|
import java.util.concurrent.CopyOnWriteArrayList;
|
||||||
import java.util.concurrent.CountDownLatch;
|
import java.util.concurrent.CountDownLatch;
|
||||||
@@ -20,6 +23,7 @@ import java.util.concurrent.atomic.AtomicReference;
|
|||||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* CB-528 follow-up: {@link AmqpReplyInbox#failPendingPublishesOnRecovery()} must not fail a publish
|
* CB-528 follow-up: {@link AmqpReplyInbox#failPendingPublishesOnRecovery()} must not fail a publish
|
||||||
@@ -197,11 +201,78 @@ class AmqpReplyInboxRecoveryRaceTest {
|
|||||||
+ elapsedMillis.get() + "ms");
|
+ elapsedMillis.get() + "ms");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void publishIOExceptionRemovesThePendingMessageId() throws Exception {
|
||||||
|
AtomicLong seqCounter = new AtomicLong();
|
||||||
|
Channel failing = fakeChannel(seqCounter, new CopyOnWriteArrayList<>(), new CopyOnWriteArrayList<>(),
|
||||||
|
new AtomicReference<>(), new AtomicReference<>(), true);
|
||||||
|
AmqpReplyInbox inbox = new AmqpReplyInbox(fakeConnection(failing, failing), AmqpReplyInbox.DEFAULT_PREFETCH);
|
||||||
|
|
||||||
|
try {
|
||||||
|
org.junit.jupiter.api.Assertions.assertThrows(IllegalStateException.class,
|
||||||
|
() -> inbox.publish("worker", "catch", "body"));
|
||||||
|
assertEquals(0, pendingByMsgId(inbox).size(), "publish IOException must remove its msgId entry");
|
||||||
|
} finally {
|
||||||
|
inbox.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void interruptedPublishRemovesThePendingMessageIdInFinally() throws Exception {
|
||||||
|
InboxFixture fixture = new InboxFixture();
|
||||||
|
Thread publish = fixture.startPublish("finally");
|
||||||
|
fixture.awaitPublished("finally");
|
||||||
|
publish.interrupt();
|
||||||
|
publish.join(5_000);
|
||||||
|
|
||||||
|
assertEquals(0, pendingByMsgId(fixture.inbox).size(), "publish finally must remove its msgId entry");
|
||||||
|
fixture.inbox.close();
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void confirmResolutionRemovesThePendingMessageId() throws Exception {
|
||||||
|
AmqpReplyInbox inbox = new InboxFixture().inbox;
|
||||||
|
try {
|
||||||
|
seedPending(inbox, 1, "confirm");
|
||||||
|
invoke(inbox, "resolveConfirm", new Class<?>[] {long.class, boolean.class, boolean.class}, 1L, false, true);
|
||||||
|
assertEquals(0, pendingByMsgId(inbox).size(), "confirm resolution must remove its msgId entry");
|
||||||
|
} finally {
|
||||||
|
inbox.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void recoverySweepRemovesThePendingMessageId() throws Exception {
|
||||||
|
AmqpReplyInbox inbox = new InboxFixture().inbox;
|
||||||
|
try {
|
||||||
|
seedPending(inbox, 1, "recovery");
|
||||||
|
inbox.failPendingPublishesOnRecovery();
|
||||||
|
assertEquals(0, pendingByMsgId(inbox).size(), "recovery sweep must remove its msgId entry");
|
||||||
|
} finally {
|
||||||
|
inbox.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void closeRemovesThePendingMessageId() throws Exception {
|
||||||
|
AmqpReplyInbox inbox = new InboxFixture().inbox;
|
||||||
|
seedPending(inbox, 1, "close");
|
||||||
|
inbox.close();
|
||||||
|
|
||||||
|
assertEquals(0, pendingByMsgId(inbox).size(), "close must remove its msgId entry");
|
||||||
|
}
|
||||||
|
|
||||||
/** A {@link Proxy}-backed {@link Channel}: only the calls {@link AmqpReplyInbox} actually makes
|
/** A {@link Proxy}-backed {@link Channel}: only the calls {@link AmqpReplyInbox} actually makes
|
||||||
* are meaningfully implemented; everything else returns a harmless default. */
|
* are meaningfully implemented; everything else returns a harmless default. */
|
||||||
private static Channel fakeChannel(AtomicLong seqCounter, List<Long> seqOrder, List<String> msgIdOrder,
|
private static Channel fakeChannel(AtomicLong seqCounter, List<Long> seqOrder, List<String> msgIdOrder,
|
||||||
AtomicReference<ConfirmCallback> ackCallback,
|
AtomicReference<ConfirmCallback> ackCallback,
|
||||||
AtomicReference<ConfirmCallback> nackCallback) {
|
AtomicReference<ConfirmCallback> nackCallback) {
|
||||||
|
return fakeChannel(seqCounter, seqOrder, msgIdOrder, ackCallback, nackCallback, false);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Channel fakeChannel(AtomicLong seqCounter, List<Long> seqOrder, List<String> msgIdOrder,
|
||||||
|
AtomicReference<ConfirmCallback> ackCallback,
|
||||||
|
AtomicReference<ConfirmCallback> nackCallback, boolean failPublish) {
|
||||||
InvocationHandler handler = (proxy, method, args) -> {
|
InvocationHandler handler = (proxy, method, args) -> {
|
||||||
String name = method.getName();
|
String name = method.getName();
|
||||||
if (name.equals("getNextPublishSeqNo")) {
|
if (name.equals("getNextPublishSeqNo")) {
|
||||||
@@ -210,6 +281,9 @@ class AmqpReplyInboxRecoveryRaceTest {
|
|||||||
return value;
|
return value;
|
||||||
}
|
}
|
||||||
if (name.equals("basicPublish")) {
|
if (name.equals("basicPublish")) {
|
||||||
|
if (failPublish) {
|
||||||
|
throw new IOException("test publish failure");
|
||||||
|
}
|
||||||
AMQP.BasicProperties props = (AMQP.BasicProperties) args[3];
|
AMQP.BasicProperties props = (AMQP.BasicProperties) args[3];
|
||||||
msgIdOrder.add(props.getMessageId());
|
msgIdOrder.add(props.getMessageId());
|
||||||
return null;
|
return null;
|
||||||
@@ -285,4 +359,60 @@ class AmqpReplyInboxRecoveryRaceTest {
|
|||||||
}
|
}
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
private static Map<String, Object> pendingByMsgId(AmqpReplyInbox inbox) throws Exception {
|
||||||
|
var field = AmqpReplyInbox.class.getDeclaredField("pendingByMsgId");
|
||||||
|
field.setAccessible(true);
|
||||||
|
return (Map<String, Object>) field.get(inbox);
|
||||||
|
}
|
||||||
|
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
private static void seedPending(AmqpReplyInbox inbox, long seq, String msgId) throws Exception {
|
||||||
|
Class<?> pendingType = Class.forName(AmqpReplyInbox.class.getName() + "$Pending");
|
||||||
|
var constructor = pendingType.getDeclaredConstructor(String.class);
|
||||||
|
constructor.setAccessible(true);
|
||||||
|
Object pending = constructor.newInstance(msgId);
|
||||||
|
var seqField = AmqpReplyInbox.class.getDeclaredField("pendingBySeq");
|
||||||
|
seqField.setAccessible(true);
|
||||||
|
((Map<Long, Object>) seqField.get(inbox)).put(seq, pending);
|
||||||
|
pendingByMsgId(inbox).put(msgId, pending);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void invoke(AmqpReplyInbox inbox, String name, Class<?>[] types, Object... args) throws Exception {
|
||||||
|
Method method = AmqpReplyInbox.class.getDeclaredMethod(name, types);
|
||||||
|
method.setAccessible(true);
|
||||||
|
method.invoke(inbox, args);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static final class InboxFixture {
|
||||||
|
final AtomicLong seqCounter = new AtomicLong();
|
||||||
|
final List<Long> seqOrder = new CopyOnWriteArrayList<>();
|
||||||
|
final List<String> msgIdOrder = new CopyOnWriteArrayList<>();
|
||||||
|
final AtomicReference<ConfirmCallback> ackCallback = new AtomicReference<>();
|
||||||
|
final AtomicReference<ConfirmCallback> nackCallback = new AtomicReference<>();
|
||||||
|
final AmqpReplyInbox inbox = new AmqpReplyInbox(
|
||||||
|
fakeConnection(fakeChannel(seqCounter, seqOrder, msgIdOrder, ackCallback, nackCallback),
|
||||||
|
fakeChannel(seqCounter, seqOrder, msgIdOrder, ackCallback, nackCallback)),
|
||||||
|
AmqpReplyInbox.DEFAULT_PREFETCH);
|
||||||
|
|
||||||
|
Thread startPublish(String msgId) {
|
||||||
|
Thread thread = Thread.ofVirtual().start(() -> {
|
||||||
|
try {
|
||||||
|
inbox.publish("worker", msgId, "body");
|
||||||
|
} catch (IllegalStateException ignored) {
|
||||||
|
// Interrupting the confirm wait is the path under test.
|
||||||
|
}
|
||||||
|
});
|
||||||
|
return thread;
|
||||||
|
}
|
||||||
|
|
||||||
|
void awaitPublished(String msgId) throws InterruptedException {
|
||||||
|
long deadline = System.nanoTime() + TimeUnit.SECONDS.toNanos(5);
|
||||||
|
while (!msgIdOrder.contains(msgId) && System.nanoTime() < deadline) {
|
||||||
|
Thread.sleep(10);
|
||||||
|
}
|
||||||
|
assertTrue(msgIdOrder.contains(msgId), "publish did not register " + msgId);
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -12,13 +12,18 @@ import org.testcontainers.junit.jupiter.Testcontainers;
|
|||||||
import org.testcontainers.utility.DockerImageName;
|
import org.testcontainers.utility.DockerImageName;
|
||||||
|
|
||||||
import java.io.IOException;
|
import java.io.IOException;
|
||||||
|
import java.lang.reflect.InvocationHandler;
|
||||||
|
import java.lang.reflect.Method;
|
||||||
|
import java.lang.reflect.Proxy;
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
import java.util.concurrent.TimeUnit;
|
import java.util.concurrent.TimeUnit;
|
||||||
import java.util.concurrent.atomic.AtomicLong;
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
import static org.junit.jupiter.api.Assertions.assertInstanceOf;
|
import static org.junit.jupiter.api.Assertions.assertInstanceOf;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
@@ -207,6 +212,31 @@ class LeadMailboxTest {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* {@link LeadMailbox#inspect} opens a third channel after the mailbox's consume and publish
|
||||||
|
* channels. Limit this connection to three channels, then require a replacement channel after
|
||||||
|
* the successful inspect. If inspect leaves its probe open, the broker refuses that replacement.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void inspectClosesItsSuccessfulProbeChannel() throws Exception {
|
||||||
|
var factory = LeadMailbox.connectionFactory(uri());
|
||||||
|
factory.setRequestedChannelMax(3);
|
||||||
|
Connection connection = factory.newConnection();
|
||||||
|
try (LeadMailbox mailbox = new LeadMailbox(connection, coordId("lead-inspect-probe-close"))) {
|
||||||
|
LeadChannel.MailboxState state = mailbox.inspect(mailbox.selfCoordId());
|
||||||
|
assertTrue(state.exists(), "the owned mailbox must be found before checking the probe channel");
|
||||||
|
|
||||||
|
Channel replacement = connection.createChannel();
|
||||||
|
assertNotNull(replacement,
|
||||||
|
"inspect must close its successful probe channel; the replacement channel was null");
|
||||||
|
try {
|
||||||
|
assertTrue(replacement.isOpen(), "the replacement channel must be open after inspect returns");
|
||||||
|
} finally {
|
||||||
|
replacement.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* fleetd #440: {@code heldDurable()} must be derived from what {@link LeadMailbox#own} actually
|
* fleetd #440: {@code heldDurable()} must be derived from what {@link LeadMailbox#own} actually
|
||||||
* did against the real broker — a durable queue declare plus a manual-ack consumer — not a
|
* did against the real broker — a durable queue declare plus a manual-ack consumer — not a
|
||||||
@@ -347,6 +377,72 @@ class LeadMailboxTest {
|
|||||||
() -> "expected AlreadyClosedException, got: " + thrown);
|
() -> "expected AlreadyClosedException, got: " + thrown);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void publishIOExceptionRemovesThePendingMessageId() throws Exception {
|
||||||
|
LeadMailbox mailbox = newMailbox(true);
|
||||||
|
try {
|
||||||
|
assertThrows(IllegalStateException.class,
|
||||||
|
() -> mailbox.publish("target", new LeadMessage("catch", "from", "target", "body")));
|
||||||
|
assertEquals(0, pendingByMsgId(mailbox).size(), "publish IOException must remove its msgId entry");
|
||||||
|
} finally {
|
||||||
|
mailbox.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void interruptedPublishRemovesThePendingMessageIdInFinally() throws Exception {
|
||||||
|
LeadMailbox mailbox = newMailbox(false);
|
||||||
|
Thread publish = Thread.ofVirtual().start(() -> {
|
||||||
|
try {
|
||||||
|
mailbox.publish("target", new LeadMessage("finally", "from", "target", "body"));
|
||||||
|
} catch (IllegalStateException ignored) {
|
||||||
|
// Interrupting the confirm wait is the path under test.
|
||||||
|
}
|
||||||
|
});
|
||||||
|
awaitPending(mailbox, "finally");
|
||||||
|
publish.interrupt();
|
||||||
|
publish.join(5_000);
|
||||||
|
|
||||||
|
try {
|
||||||
|
assertEquals(0, pendingByMsgId(mailbox).size(), "publish finally must remove its msgId entry");
|
||||||
|
} finally {
|
||||||
|
mailbox.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void confirmResolutionRemovesThePendingMessageId() throws Exception {
|
||||||
|
LeadMailbox mailbox = newMailbox(false);
|
||||||
|
try {
|
||||||
|
seedPending(mailbox, 1, "confirm");
|
||||||
|
invoke(mailbox, "resolveConfirm", new Class<?>[] {long.class, boolean.class, boolean.class}, 1L, false, true);
|
||||||
|
assertEquals(0, pendingByMsgId(mailbox).size(), "confirm resolution must remove its msgId entry");
|
||||||
|
} finally {
|
||||||
|
mailbox.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void recoverySweepRemovesThePendingMessageId() throws Exception {
|
||||||
|
LeadMailbox mailbox = newMailbox(false);
|
||||||
|
try {
|
||||||
|
seedPending(mailbox, 1, "recovery");
|
||||||
|
mailbox.failPendingPublishesOnRecovery();
|
||||||
|
assertEquals(0, pendingByMsgId(mailbox).size(), "recovery sweep must remove its msgId entry");
|
||||||
|
} finally {
|
||||||
|
mailbox.close();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void closeRemovesThePendingMessageId() throws Exception {
|
||||||
|
LeadMailbox mailbox = newMailbox(false);
|
||||||
|
seedPending(mailbox, 1, "close");
|
||||||
|
mailbox.close();
|
||||||
|
|
||||||
|
assertEquals(0, pendingByMsgId(mailbox).size(), "close must remove its msgId entry");
|
||||||
|
}
|
||||||
|
|
||||||
/** Poll peek until at least one message is held, or ~10s elapse (broker delivery is async). */
|
/** Poll peek until at least one message is held, or ~10s elapse (broker delivery is async). */
|
||||||
@SuppressWarnings("BusyWait")
|
@SuppressWarnings("BusyWait")
|
||||||
private static List<LeadMessage> awaitPeek(LeadMailbox inbox) throws InterruptedException {
|
private static List<LeadMessage> awaitPeek(LeadMailbox inbox) throws InterruptedException {
|
||||||
@@ -372,4 +468,108 @@ class LeadMailboxTest {
|
|||||||
}
|
}
|
||||||
return state;
|
return state;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private static LeadMailbox newMailbox(boolean failPublish) {
|
||||||
|
AtomicLong sequence = new AtomicLong();
|
||||||
|
Channel consume = fakeChannel(sequence, false);
|
||||||
|
Channel publish = fakeChannel(sequence, failPublish);
|
||||||
|
return new LeadMailbox(fakeConnection(consume, publish), "self");
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Channel fakeChannel(AtomicLong sequence, boolean failPublish) {
|
||||||
|
InvocationHandler handler = (proxy, method, args) -> {
|
||||||
|
if (method.getName().equals("getNextPublishSeqNo")) {
|
||||||
|
return sequence.incrementAndGet();
|
||||||
|
}
|
||||||
|
if (method.getName().equals("basicPublish") && failPublish) {
|
||||||
|
throw new IOException("test publish failure");
|
||||||
|
}
|
||||||
|
if (method.getName().equals("equals")) {
|
||||||
|
return proxy == args[0];
|
||||||
|
}
|
||||||
|
if (method.getName().equals("hashCode")) {
|
||||||
|
return System.identityHashCode(proxy);
|
||||||
|
}
|
||||||
|
return defaultValue(method.getReturnType());
|
||||||
|
};
|
||||||
|
return (Channel) Proxy.newProxyInstance(LeadMailboxTest.class.getClassLoader(), new Class<?>[] {Channel.class}, handler);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Connection fakeConnection(Channel first, Channel second) {
|
||||||
|
AtomicLong calls = new AtomicLong();
|
||||||
|
InvocationHandler handler = (proxy, method, args) -> {
|
||||||
|
if (method.getName().equals("createChannel") && (args == null || args.length == 0)) {
|
||||||
|
return calls.getAndIncrement() == 0 ? first : second;
|
||||||
|
}
|
||||||
|
if (method.getName().equals("equals")) {
|
||||||
|
return proxy == args[0];
|
||||||
|
}
|
||||||
|
if (method.getName().equals("hashCode")) {
|
||||||
|
return System.identityHashCode(proxy);
|
||||||
|
}
|
||||||
|
return defaultValue(method.getReturnType());
|
||||||
|
};
|
||||||
|
return (Connection) Proxy.newProxyInstance(LeadMailboxTest.class.getClassLoader(), new Class<?>[] {Connection.class}, handler);
|
||||||
|
}
|
||||||
|
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
private static Map<String, Object> pendingByMsgId(LeadMailbox mailbox) throws Exception {
|
||||||
|
var field = LeadMailbox.class.getDeclaredField("pendingByMsgId");
|
||||||
|
field.setAccessible(true);
|
||||||
|
return (Map<String, Object>) field.get(mailbox);
|
||||||
|
}
|
||||||
|
|
||||||
|
@SuppressWarnings("unchecked")
|
||||||
|
private static void seedPending(LeadMailbox mailbox, long seq, String msgId) throws Exception {
|
||||||
|
Class<?> pendingType = Class.forName(LeadMailbox.class.getName() + "$Pending");
|
||||||
|
var constructor = pendingType.getDeclaredConstructor(String.class);
|
||||||
|
constructor.setAccessible(true);
|
||||||
|
Object pending = constructor.newInstance(msgId);
|
||||||
|
var seqField = LeadMailbox.class.getDeclaredField("pendingBySeq");
|
||||||
|
seqField.setAccessible(true);
|
||||||
|
((Map<Long, Object>) seqField.get(mailbox)).put(seq, pending);
|
||||||
|
pendingByMsgId(mailbox).put(msgId, pending);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void invoke(LeadMailbox mailbox, String name, Class<?>[] types, Object... args) throws Exception {
|
||||||
|
Method method = LeadMailbox.class.getDeclaredMethod(name, types);
|
||||||
|
method.setAccessible(true);
|
||||||
|
method.invoke(mailbox, args);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void awaitPending(LeadMailbox mailbox, String msgId) throws Exception {
|
||||||
|
long deadline = System.nanoTime() + TimeUnit.SECONDS.toNanos(5);
|
||||||
|
while (!pendingByMsgId(mailbox).containsKey(msgId) && System.nanoTime() < deadline) {
|
||||||
|
Thread.sleep(10);
|
||||||
|
}
|
||||||
|
assertTrue(pendingByMsgId(mailbox).containsKey(msgId), "publish did not register " + msgId);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Object defaultValue(Class<?> type) {
|
||||||
|
if (!type.isPrimitive() || type == void.class) {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
if (type == boolean.class) {
|
||||||
|
return Boolean.FALSE;
|
||||||
|
}
|
||||||
|
if (type == long.class) {
|
||||||
|
return 0L;
|
||||||
|
}
|
||||||
|
if (type == short.class) {
|
||||||
|
return (short) 0;
|
||||||
|
}
|
||||||
|
if (type == byte.class) {
|
||||||
|
return (byte) 0;
|
||||||
|
}
|
||||||
|
if (type == char.class) {
|
||||||
|
return (char) 0;
|
||||||
|
}
|
||||||
|
if (type == double.class) {
|
||||||
|
return 0.0d;
|
||||||
|
}
|
||||||
|
if (type == float.class) {
|
||||||
|
return 0.0f;
|
||||||
|
}
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -22,6 +22,7 @@ import java.util.concurrent.TimeUnit;
|
|||||||
|
|
||||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
||||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||||
@@ -505,6 +506,52 @@ class MessageServiceTest {
|
|||||||
"an answer to a turn that never existed (or already lapsed) is stale, not a hang");
|
"an answer to a turn that never existed (or already lapsed) is stale, not a hang");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #575. {@code answer()}'s STALE_TURN return for "the ask lapsed between the lookup and
|
||||||
|
* the unblock" ({@code rendezvous.answerAsk(turnId, content)} returning {@code false} even though
|
||||||
|
* this call's own {@code rendezvous.askSession(turnId)} check at the top saw the ask as open) used
|
||||||
|
* to be covered only by a hand-rolled copy of the cleanup pair its own finally already runs, not
|
||||||
|
* the finally itself — a structural gap, closed by widening the try up to cover the registration
|
||||||
|
* above it. This test pins the exact race with {@link
|
||||||
|
* MessageService#setAnswerAskLapseRaceHookForTest}, which fires right before this call's own
|
||||||
|
* {@code rendezvous.answerAsk} call and completes the same turnId's ask directly — reproducing
|
||||||
|
* what a second, concurrent {@code answer()} winning that race could otherwise only do by timing
|
||||||
|
* luck. Proves the fix changed no behaviour on this path: it still returns {@code STALE_TURN},
|
||||||
|
* and this call's own forward waiter is still closed exactly once (not left open, and not closed
|
||||||
|
* twice — there is now only one cleanup site left to run).
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void answerLosingTheRaceToAnAlreadyAnsweredAskStillReturnsStaleTurnAndCleansUpOnce() throws Exception {
|
||||||
|
String ticket = messages.sendAsync(T, "long task");
|
||||||
|
awaitWaiting();
|
||||||
|
injectDelivery();
|
||||||
|
|
||||||
|
CompletableFuture<MessageService.AskResult> ask =
|
||||||
|
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config?", 5000));
|
||||||
|
MessageService.TaskView asking = awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||||
|
String turnId = asking.turnId();
|
||||||
|
assertNotNull(turnId, "an ASKING view carries the turnId to answer on");
|
||||||
|
|
||||||
|
messages.setAnswerAskLapseRaceHookForTest(() -> rendezvous.answerAsk(turnId, "raced in first"));
|
||||||
|
try {
|
||||||
|
MessageService.Reply r = messages.answer(turnId, "too late", 500);
|
||||||
|
assertEquals(MessageService.Outcome.STALE_TURN, r.outcome(),
|
||||||
|
"an ask already answered by the race must be seen as lapsed, not double-delivered");
|
||||||
|
} finally {
|
||||||
|
messages.setAnswerAskLapseRaceHookForTest(null);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Cleanup ran exactly once: the forward waiter THIS call opened is closed, not leaked.
|
||||||
|
assertNull(rendezvous.currentWaiter(T),
|
||||||
|
"the forward waiter this answer() call opened must be closed after a STALE_TURN return");
|
||||||
|
|
||||||
|
// The worker's own ask() call, unblocked by the hook's direct answerAsk, still completes
|
||||||
|
// normally — the race this test simulates does not strand it.
|
||||||
|
MessageService.AskResult a = ask.get(5, TimeUnit.SECONDS);
|
||||||
|
assertEquals(MessageService.AskOutcome.ANSWERED, a.outcome());
|
||||||
|
assertEquals("raced in first", a.answer());
|
||||||
|
}
|
||||||
|
|
||||||
// --- timeout, answer, poll, and lock-contention edges ----------------------------------
|
// --- timeout, answer, poll, and lock-contention edges ----------------------------------
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -554,6 +601,30 @@ class MessageServiceTest {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #571 (the acceptance test the ticket was filed for). The worker is idle so the injector
|
||||||
|
* attempts delivery, but the {@code agent.prompt} call itself fails with a herdr error that is
|
||||||
|
* not a confirmed absence (not a {@code *_not_found} code) — {@link Injector} marks the Pending
|
||||||
|
* {@code ATTEMPTED} (fleetd #551), meaning the call was made and whether it reached the pane is
|
||||||
|
* unknown. Before this fix, {@code send}'s {@code TimeoutException} branch collapsed
|
||||||
|
* {@code ATTEMPTED} into {@code TIMED_OUT_QUEUED} — a promise that the message will never arrive,
|
||||||
|
* which may already be false: {@code agent.prompt} pastes and submits in one call.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void sendTimesOutWithAttemptedDeliveryReportsUnconfirmedNotQueued() throws Exception {
|
||||||
|
herdr.agentSendFailsWith("send_failed");
|
||||||
|
CompletableFuture<MessageService.Reply> send =
|
||||||
|
CompletableFuture.supplyAsync(() -> messages.send(T, "brief", 150));
|
||||||
|
awaitWaiting();
|
||||||
|
injector.onStatus(T, AgentStatus.IDLE); // triggers the failing delivery attempt → ATTEMPTED
|
||||||
|
|
||||||
|
MessageService.Reply r = send.get(5, TimeUnit.SECONDS);
|
||||||
|
assertEquals(MessageService.Outcome.TIMED_OUT_UNCONFIRMED, r.outcome(),
|
||||||
|
"an ATTEMPTED delivery must not collapse into TIMED_OUT_QUEUED — the message may "
|
||||||
|
+ "already have arrived in full, and TIMED_OUT_QUEUED promises it never will");
|
||||||
|
assertNull(r.text());
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void answerTimesOutWhenTheResumedWorkerNeverReplies() throws Exception {
|
void answerTimesOutWhenTheResumedWorkerNeverReplies() throws Exception {
|
||||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||||
@@ -579,6 +650,183 @@ class MessageServiceTest {
|
|||||||
assertEquals("config.yaml", a.answer());
|
assertEquals("config.yaml", a.answer());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- fleetd #572: answer() must release the session lock on EVERY exit, not just return it -----
|
||||||
|
//
|
||||||
|
// answer()'s session lock is released in an outer `finally` (MessageService.java:1217-1219) that
|
||||||
|
// wraps its whole body. Removing that one line survives the entire suite: coverage is not the
|
||||||
|
// gap, assertion is — every existing answer() test checks the RETURN VALUE, never that the lock
|
||||||
|
// it took is reacquirable afterward. If it is not, the session is wedged forever with no
|
||||||
|
// exception and no log line. Each test below drives answer() through one of its four exits
|
||||||
|
// (normal reply, TIMED_OUT_WORKING, ExecutionException, InterruptedException) and then proves
|
||||||
|
// reacquisition the only way that actually proves it: a bounded follow-up `send` on the SAME
|
||||||
|
// session must not come back BUSY. A `send` only reports BUSY when `tryLock` itself timed out
|
||||||
|
// (MessageService.java:926-928) — every other early return in `send` still passes through its own
|
||||||
|
// lock-acquired `try`, so a non-BUSY probe result is specifically evidence the lock was free.
|
||||||
|
|
||||||
|
/** The REPLIED exit (the happy path) — the resumed worker's real {@code fleet_reply} arrives. */
|
||||||
|
@Test
|
||||||
|
void answerReleasesTheSessionLockAfterANormalReply() throws Exception {
|
||||||
|
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||||
|
awaitWaiting();
|
||||||
|
injector.onStatus(T, AgentStatus.IDLE);
|
||||||
|
injector.onStatus(T, AgentStatus.WORKING);
|
||||||
|
|
||||||
|
CompletableFuture<MessageService.AskResult> ask =
|
||||||
|
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||||
|
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||||
|
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||||
|
|
||||||
|
CompletableFuture<MessageService.Reply> answer =
|
||||||
|
CompletableFuture.supplyAsync(() -> messages.answer(q.turnId(), "config.yaml", 5000));
|
||||||
|
ask.get(5, TimeUnit.SECONDS); // worker resumed with the answer
|
||||||
|
|
||||||
|
awaitWaiting(); // the answering call has (re)opened its own forward waiter
|
||||||
|
assertTrue(rendezvous.resolve(T, "done"), "the worker's final reply resolves the answering send");
|
||||||
|
MessageService.Reply done = answer.get(5, TimeUnit.SECONDS);
|
||||||
|
assertEquals(MessageService.Outcome.REPLIED, done.outcome());
|
||||||
|
|
||||||
|
MessageService.Reply probe = messages.send(T, "probe after normal reply", 300);
|
||||||
|
assertNotEquals(MessageService.Outcome.BUSY, probe.outcome(),
|
||||||
|
"the session lock must be released after a normal REPLIED answer(), or this bounded "
|
||||||
|
+ "follow-up send would come back BUSY instead of timing out on its own work");
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The {@code TIMED_OUT_WORKING} exit. Same setup as {@link #answerTimesOutWhenTheResumedWorkerNeverReplies}
|
||||||
|
* (which only checks {@code answer()}'s return value — exactly the assertion the fleetd #572
|
||||||
|
* mutation survives), plus the reacquisition proof that test does not make.
|
||||||
|
*
|
||||||
|
* <p>{@code answer()} MUST run on its own thread here, not the test's main thread: {@code lock}
|
||||||
|
* is a {@link java.util.concurrent.locks.ReentrantLock}, so a probe {@code send} issued by the
|
||||||
|
* SAME thread that (under the mutation) still "holds" it would reenter for free and report a
|
||||||
|
* false pass — reentrancy, not release. Measured while writing this test: with {@code answer()}
|
||||||
|
* called inline, this test stayed green under the mutation while its three siblings correctly
|
||||||
|
* went red.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void answerReleasesTheSessionLockAfterATimedOutWorkingReturn() throws Exception {
|
||||||
|
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||||
|
awaitWaiting();
|
||||||
|
injector.onStatus(T, AgentStatus.IDLE);
|
||||||
|
injector.onStatus(T, AgentStatus.WORKING);
|
||||||
|
|
||||||
|
CompletableFuture<MessageService.AskResult> ask =
|
||||||
|
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||||
|
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||||
|
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||||
|
|
||||||
|
// The primary answers, unblocking the worker; the worker never sends its follow-up
|
||||||
|
// fleet_reply, so the answering call rides out its short window as still-working.
|
||||||
|
CompletableFuture<MessageService.Reply> answer =
|
||||||
|
CompletableFuture.supplyAsync(() -> messages.answer(q.turnId(), "config.yaml", 200));
|
||||||
|
MessageService.Reply answered = answer.get(5, TimeUnit.SECONDS);
|
||||||
|
assertEquals(MessageService.Outcome.TIMED_OUT_WORKING, answered.outcome(),
|
||||||
|
"an answered worker that never replies times out as still working");
|
||||||
|
ask.get(5, TimeUnit.SECONDS); // drain: the worker resumed with the answer
|
||||||
|
|
||||||
|
MessageService.Reply probe = messages.send(T, "probe after timed-out-working", 300);
|
||||||
|
assertNotEquals(MessageService.Outcome.BUSY, probe.outcome(),
|
||||||
|
"the session lock must be released after a TIMED_OUT_WORKING answer(), or this bounded "
|
||||||
|
+ "follow-up send would come back BUSY instead of timing out on its own work");
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The {@code ExecutionException} exit. Forces it directly on answer()'s own reopened forward
|
||||||
|
* waiter — {@code rendezvous.currentWaiter(T)} is the exact {@code CompletableFuture} its
|
||||||
|
* {@code reply.get(...)} is blocked on — rather than trying to make a real worker fail, since the
|
||||||
|
* failure mode under test is in answer()'s own wait, not in how it got triggered.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void answerReleasesTheSessionLockWhenTheReplyFutureFailsExceptionally() throws Exception {
|
||||||
|
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||||
|
awaitWaiting();
|
||||||
|
injector.onStatus(T, AgentStatus.IDLE);
|
||||||
|
injector.onStatus(T, AgentStatus.WORKING);
|
||||||
|
|
||||||
|
CompletableFuture<MessageService.AskResult> ask =
|
||||||
|
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||||
|
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||||
|
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||||
|
String turnId = q.turnId();
|
||||||
|
|
||||||
|
java.util.concurrent.atomic.AtomicReference<Throwable> caught =
|
||||||
|
new java.util.concurrent.atomic.AtomicReference<>();
|
||||||
|
Thread answerer = new Thread(() -> {
|
||||||
|
try {
|
||||||
|
messages.answer(turnId, "config.yaml", 5000);
|
||||||
|
caught.set(new AssertionError("expected answer() to throw"));
|
||||||
|
} catch (Throwable t) {
|
||||||
|
caught.set(t);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
answerer.start();
|
||||||
|
awaitWaiting(); // answer() re-opened its forward waiter and is about to block on it
|
||||||
|
|
||||||
|
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(T);
|
||||||
|
assertNotNull(waiter, "answer() should have registered a forward waiter for the worker session");
|
||||||
|
waiter.completeExceptionally(new RuntimeException("boom"));
|
||||||
|
|
||||||
|
answerer.join(5000);
|
||||||
|
assertFalse(answerer.isAlive(), "answer() should have left by throwing once its reply future failed");
|
||||||
|
Throwable thrown = caught.get();
|
||||||
|
assertNotNull(thrown, "answer() should have thrown");
|
||||||
|
assertTrue(thrown instanceof RuntimeException, "a RuntimeException cause is rethrown as-is: " + thrown);
|
||||||
|
assertEquals("boom", thrown.getMessage());
|
||||||
|
|
||||||
|
ask.get(5, TimeUnit.SECONDS); // drain: the worker resumed with the answer
|
||||||
|
|
||||||
|
MessageService.Reply probe = messages.send(T, "probe after exceptional failure", 300);
|
||||||
|
assertNotEquals(MessageService.Outcome.BUSY, probe.outcome(),
|
||||||
|
"the session lock must be released when answer()'s reply future fails exceptionally, "
|
||||||
|
+ "or this bounded follow-up send would come back BUSY instead of timing out on its own work");
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The {@code InterruptedException} exit. Same shape as the {@code ExecutionException} case above,
|
||||||
|
* but the thread blocked in {@code reply.get(...)} is interrupted instead of the future failing.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void answerReleasesTheSessionLockWhenInterrupted() throws Exception {
|
||||||
|
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||||
|
awaitWaiting();
|
||||||
|
injector.onStatus(T, AgentStatus.IDLE);
|
||||||
|
injector.onStatus(T, AgentStatus.WORKING);
|
||||||
|
|
||||||
|
CompletableFuture<MessageService.AskResult> ask =
|
||||||
|
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||||
|
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||||
|
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||||
|
String turnId = q.turnId();
|
||||||
|
|
||||||
|
java.util.concurrent.atomic.AtomicReference<Throwable> caught =
|
||||||
|
new java.util.concurrent.atomic.AtomicReference<>();
|
||||||
|
Thread answerer = new Thread(() -> {
|
||||||
|
try {
|
||||||
|
messages.answer(turnId, "config.yaml", 5000);
|
||||||
|
caught.set(new AssertionError("expected answer() to throw"));
|
||||||
|
} catch (Throwable t) {
|
||||||
|
caught.set(t);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
answerer.start();
|
||||||
|
awaitWaiting(); // answer() re-opened its forward waiter and is about to block on it
|
||||||
|
|
||||||
|
answerer.interrupt();
|
||||||
|
answerer.join(5000);
|
||||||
|
assertFalse(answerer.isAlive(), "answer() should have left by throwing once its thread was interrupted");
|
||||||
|
Throwable thrown = caught.get();
|
||||||
|
assertNotNull(thrown, "answer() should have thrown");
|
||||||
|
assertTrue(thrown instanceof IllegalStateException,
|
||||||
|
"an interrupted wait is wrapped in IllegalStateException: " + thrown);
|
||||||
|
|
||||||
|
ask.get(5, TimeUnit.SECONDS); // drain: the worker resumed with the answer
|
||||||
|
|
||||||
|
MessageService.Reply probe = messages.send(T, "probe after interruption", 300);
|
||||||
|
assertNotEquals(MessageService.Outcome.BUSY, probe.outcome(),
|
||||||
|
"the session lock must be released when answer()'s wait is interrupted, or this bounded "
|
||||||
|
+ "follow-up send would come back BUSY instead of timing out on its own work");
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void pollReturnsNullForAnUnknownTicket() {
|
void pollReturnsNullForAnUnknownTicket() {
|
||||||
assertNull(messages.poll("task-999999"), "a ticket that was never minted is unknown");
|
assertNull(messages.poll("task-999999"), "a ticket that was never minted is unknown");
|
||||||
|
|||||||
@@ -10,12 +10,13 @@ import static org.junit.jupiter.api.Assertions.assertTrue;
|
|||||||
class MemberRoleTest {
|
class MemberRoleTest {
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void theThreeRolesAreArchitectDevAndReviewer() {
|
void theFourRolesAreArchitectDevHunterAndReviewer() {
|
||||||
assertEquals(3, MemberRole.values().length,
|
assertEquals(4, MemberRole.values().length,
|
||||||
"a new role changes the charter, the role file, the skill and the authz row — "
|
"a new role changes the charter, the role file, the skill and the authz row — "
|
||||||
+ "adding one is a deliberate act, so this count is meant to fail first");
|
+ "adding one is a deliberate act, so this count is meant to fail first");
|
||||||
assertEquals("architect", MemberRole.ARCHITECT.wireName());
|
assertEquals("architect", MemberRole.ARCHITECT.wireName());
|
||||||
assertEquals("dev", MemberRole.DEV.wireName());
|
assertEquals("dev", MemberRole.DEV.wireName());
|
||||||
|
assertEquals("hunter", MemberRole.HUNTER.wireName());
|
||||||
assertEquals("reviewer", MemberRole.REVIEWER.wireName());
|
assertEquals("reviewer", MemberRole.REVIEWER.wireName());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -30,6 +31,7 @@ class MemberRoleTest {
|
|||||||
void parseIsCaseInsensitiveAndTrimsSurroundingSpace() {
|
void parseIsCaseInsensitiveAndTrimsSurroundingSpace() {
|
||||||
assertSame(MemberRole.ARCHITECT, MemberRole.parse("Architect"));
|
assertSame(MemberRole.ARCHITECT, MemberRole.parse("Architect"));
|
||||||
assertSame(MemberRole.DEV, MemberRole.parse(" DEV "));
|
assertSame(MemberRole.DEV, MemberRole.parse(" DEV "));
|
||||||
|
assertSame(MemberRole.HUNTER, MemberRole.parse("HuNtEr"));
|
||||||
assertSame(MemberRole.REVIEWER, MemberRole.parse("ReViEwEr"));
|
assertSame(MemberRole.REVIEWER, MemberRole.parse("ReViEwEr"));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -38,7 +40,7 @@ class MemberRoleTest {
|
|||||||
IllegalArgumentException e =
|
IllegalArgumentException e =
|
||||||
assertThrows(IllegalArgumentException.class, () -> MemberRole.parse("archtiect"));
|
assertThrows(IllegalArgumentException.class, () -> MemberRole.parse("archtiect"));
|
||||||
assertTrue(e.getMessage().contains("archtiect"), e.getMessage());
|
assertTrue(e.getMessage().contains("archtiect"), e.getMessage());
|
||||||
assertTrue(e.getMessage().contains("architect, dev, reviewer"),
|
assertTrue(e.getMessage().contains("architect, dev, hunter, reviewer"),
|
||||||
"a typo in config should be fixable from the message alone: " + e.getMessage());
|
"a typo in config should be fixable from the message alone: " + e.getMessage());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -8,6 +8,7 @@ import dev.ltms.fleet.herdr.AgentControl;
|
|||||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
import dev.ltms.fleet.inject.Injector;
|
import dev.ltms.fleet.inject.Injector;
|
||||||
|
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||||
import dev.ltms.fleet.inject.StatusPoller;
|
import dev.ltms.fleet.inject.StatusPoller;
|
||||||
import dev.ltms.fleet.inject.MemberPresence;
|
import dev.ltms.fleet.inject.MemberPresence;
|
||||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||||
@@ -165,6 +166,29 @@ class FleetAppTest {
|
|||||||
assertEquals("degraded", mapper.readTree(res.body()).get("status").asText());
|
assertEquals("degraded", mapper.readTree(res.body()).get("status").asText());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void healthzKeepsOkStatusAndReportsLoopHealthInItsBody() {
|
||||||
|
FleetMcp.LoopHealthSource loops = new FleetMcp.LoopHealthSource(
|
||||||
|
() -> LoopWatchdog.State.RUNNING, () -> LoopWatchdog.State.STOPPED);
|
||||||
|
|
||||||
|
FleetApp.HealthzResponse ok = FleetApp.healthzResponse(new FakeHerdr(), new FakeHerdr(), loops);
|
||||||
|
assertEquals(200, ok.status(), "a healthy herdr must keep /healthz at 200 regardless of loop states");
|
||||||
|
assertEquals(Map.of("statusPoller", "RUNNING", "sessionReaper", "STOPPED"), ok.body().get("loopHealth"),
|
||||||
|
"the /healthz body must report each loop state without making STOPPED an alarm");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void healthzKeepsDegradedStatusAndReportsLoopHealthInItsBody() {
|
||||||
|
FleetMcp.LoopHealthSource loops = new FleetMcp.LoopHealthSource(
|
||||||
|
() -> LoopWatchdog.State.RUNNING, () -> LoopWatchdog.State.STOPPED);
|
||||||
|
FleetApp.HealthzResponse degraded = FleetApp.healthzResponse(new FakeHerdr().healthy(false),
|
||||||
|
new FakeHerdr(), loops);
|
||||||
|
assertEquals(503, degraded.status(),
|
||||||
|
"an unreachable herdr must keep /healthz at 503 regardless of loop states");
|
||||||
|
assertEquals(Map.of("statusPoller", "RUNNING", "sessionReaper", "STOPPED"),
|
||||||
|
degraded.body().get("loopHealth"), "the degraded /healthz body must retain loop states");
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void sessionsMapsWorkspaceList() throws Exception {
|
void sessionsMapsWorkspaceList() throws Exception {
|
||||||
int port = startHealthy();
|
int port = startHealthy();
|
||||||
@@ -623,6 +647,28 @@ class FleetAppTest {
|
|||||||
assertTrue(herdr.called("agent.prompt"), "message was injected");
|
assertTrue(herdr.called("agent.prompt"), "message was injected");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #571: the worker is idle, so the poller attempts delivery, but the {@code agent.prompt}
|
||||||
|
* call itself fails with a herdr error that is not a confirmed absence — {@link
|
||||||
|
* dev.ltms.fleet.inject.Injector} marks this {@code ATTEMPTED}, meaning the call was made and
|
||||||
|
* whether it reached the pane is unknown. {@code writeReply}'s default arm must map this to its
|
||||||
|
* own {@code "unconfirmed"} status, not silently fall through to {@code "done"} (which would
|
||||||
|
* claim the delegation completed) nor collapse into {@code "queued"} (which would claim the
|
||||||
|
* message will never arrive, when it may already be sitting in the pane).
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void messageTimesOutUnconfirmedWhenDeliveryAttemptFails() throws Exception {
|
||||||
|
FakeHerdr herdr = new FakeHerdr().agentStatus("idle").agentSendFailsWith("send_failed");
|
||||||
|
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||||
|
|
||||||
|
HttpResponse<String> res = postMessage(port, "{\"content\":\"hi\",\"timeoutMs\":250}");
|
||||||
|
assertEquals(202, res.statusCode());
|
||||||
|
JsonNode body = mapper.readTree(res.body());
|
||||||
|
assertEquals("unconfirmed", body.get("status").asText(),
|
||||||
|
"an ATTEMPTED delivery must report its own status, not \"queued\" or \"done\"");
|
||||||
|
assertTrue(herdr.called("agent.prompt"), "delivery must have been attempted");
|
||||||
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
void messageRejectsBlankContent() throws Exception {
|
void messageRejectsBlankContent() throws Exception {
|
||||||
int port = startHealthy();
|
int port = startHealthy();
|
||||||
|
|||||||
@@ -1,16 +1,13 @@
|
|||||||
package dev.ltms.fleet.session;
|
package dev.ltms.fleet.session;
|
||||||
|
|
||||||
import ch.qos.logback.classic.Level;
|
import ch.qos.logback.classic.Level;
|
||||||
import ch.qos.logback.classic.Logger;
|
|
||||||
import ch.qos.logback.classic.LoggerContext;
|
|
||||||
import ch.qos.logback.classic.spi.IThrowableProxy;
|
import ch.qos.logback.classic.spi.IThrowableProxy;
|
||||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
import ch.qos.logback.core.read.ListAppender;
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
import org.junit.jupiter.api.AfterEach;
|
import org.junit.jupiter.api.AfterEach;
|
||||||
import org.junit.jupiter.api.BeforeEach;
|
import org.junit.jupiter.api.BeforeEach;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
import org.junit.jupiter.api.io.TempDir;
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
import org.slf4j.LoggerFactory;
|
|
||||||
|
|
||||||
import java.io.IOException;
|
import java.io.IOException;
|
||||||
import java.nio.charset.StandardCharsets;
|
import java.nio.charset.StandardCharsets;
|
||||||
@@ -489,36 +486,33 @@ class GitWorktreesTest {
|
|||||||
// ---- CB-189: broader remote-URL coverage — every remote, both fetch and push URLs, any
|
// ---- CB-189: broader remote-URL coverage — every remote, both fetch and push URLs, any
|
||||||
// non-SSH scheme. Reporting only, additive to the origin/https strip-and-refuse tests above. ----
|
// non-SSH scheme. Reporting only, additive to the origin/https strip-and-refuse tests above. ----
|
||||||
|
|
||||||
private Logger reportingLogger;
|
private CapturedLog reportingLog;
|
||||||
private ListAppender<ILoggingEvent> reportingAppender;
|
|
||||||
|
|
||||||
/** {@link GitWorktrees}'s own logger, captured fresh for each test so assertions never see a
|
/** {@link GitWorktrees}'s own logger, captured fresh for each test so assertions never see a
|
||||||
* message left over from a previous test. */
|
* message left over from a previous test. fleetd #529: {@link CapturedLog#close} restores the
|
||||||
|
* level it captured here (whatever it truly was before this test, not just WARN), so a test
|
||||||
|
* below that further lowers the level to INFO for its own assertion (via {@link
|
||||||
|
* #reportingLog}'s {@link CapturedLog#setLevel}) can never leak that INFO pin past its own
|
||||||
|
* {@code @AfterEach} — every test's window is self-contained. */
|
||||||
@BeforeEach
|
@BeforeEach
|
||||||
void attachReportingLogCapture() {
|
void attachReportingLogCapture() {
|
||||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
reportingLog = CapturedLog.at(GitWorktrees.class, Level.WARN);
|
||||||
reportingLogger = ctx.getLogger(GitWorktrees.class);
|
|
||||||
reportingLogger.setLevel(Level.WARN);
|
|
||||||
reportingAppender = new ListAppender<>();
|
|
||||||
reportingAppender.setContext(ctx);
|
|
||||||
reportingAppender.start();
|
|
||||||
reportingLogger.addAppender(reportingAppender);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
@AfterEach
|
@AfterEach
|
||||||
void detachReportingLogCapture() {
|
void detachReportingLogCapture() {
|
||||||
reportingLogger.detachAppender(reportingAppender);
|
reportingLog.close();
|
||||||
}
|
}
|
||||||
|
|
||||||
private List<String> capturedMessages() {
|
private List<String> capturedMessages() {
|
||||||
return reportingAppender.list.stream().map(ILoggingEvent::getFormattedMessage).toList();
|
return reportingLog.events().stream().map(ILoggingEvent::getFormattedMessage).toList();
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Asserts {@code secret} appears in no captured message, and in no attached exception's
|
/** Asserts {@code secret} appears in no captured message, and in no attached exception's
|
||||||
* message either — the constraint is that a credential must never reach a log, however it
|
* message either — the constraint is that a credential must never reach a log, however it
|
||||||
* would have gotten there. */
|
* would have gotten there. */
|
||||||
private void assertNoLeak(String secret) {
|
private void assertNoLeak(String secret) {
|
||||||
for (ILoggingEvent event : reportingAppender.list) {
|
for (ILoggingEvent event : reportingLog.events()) {
|
||||||
assertFalse(event.getFormattedMessage().contains(secret),
|
assertFalse(event.getFormattedMessage().contains(secret),
|
||||||
"log message leaked a credential (" + secret + "): " + event.getFormattedMessage());
|
"log message leaked a credential (" + secret + "): " + event.getFormattedMessage());
|
||||||
IThrowableProxy thrown = event.getThrowableProxy();
|
IThrowableProxy thrown = event.getThrowableProxy();
|
||||||
@@ -596,7 +590,7 @@ class GitWorktreesTest {
|
|||||||
|
|
||||||
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-189-d", "HEAD");
|
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-189-d", "HEAD");
|
||||||
|
|
||||||
assertTrue(reportingAppender.list.isEmpty(),
|
assertTrue(reportingLog.events().isEmpty(),
|
||||||
"expected no report for an ssh remote and a credential-free https remote, got:\n"
|
"expected no report for an ssh remote and a credential-free https remote, got:\n"
|
||||||
+ capturedMessages());
|
+ capturedMessages());
|
||||||
}
|
}
|
||||||
@@ -812,7 +806,7 @@ class GitWorktreesTest {
|
|||||||
/** Criterion 1, all three present: the summary names the denominator and every neutralized file. */
|
/** Criterion 1, all three present: the summary names the denominator and every neutralized file. */
|
||||||
@Test
|
@Test
|
||||||
void isolateToolSurfaceLogsAllThreeConfigsNeutralized(@TempDir Path tmp) throws Exception {
|
void isolateToolSurfaceLogsAllThreeConfigsNeutralized(@TempDir Path tmp) throws Exception {
|
||||||
reportingLogger.setLevel(Level.INFO);
|
reportingLog.setLevel(Level.INFO);
|
||||||
Path repo = initRepoWithAllThreeConfigs(tmp.resolve("repo"));
|
Path repo = initRepoWithAllThreeConfigs(tmp.resolve("repo"));
|
||||||
|
|
||||||
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-134-log-all", "HEAD");
|
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-134-log-all", "HEAD");
|
||||||
@@ -827,7 +821,7 @@ class GitWorktreesTest {
|
|||||||
/** Criterion 1, two absent: the summary must still name the denominator and say why. */
|
/** Criterion 1, two absent: the summary must still name the denominator and say why. */
|
||||||
@Test
|
@Test
|
||||||
void isolateToolSurfaceLogsAbsentConfigsWithReason(@TempDir Path tmp) throws Exception {
|
void isolateToolSurfaceLogsAbsentConfigsWithReason(@TempDir Path tmp) throws Exception {
|
||||||
reportingLogger.setLevel(Level.INFO);
|
reportingLog.setLevel(Level.INFO);
|
||||||
Path repo = initRepo(tmp.resolve("repo")); // only .mcp.json + README committed
|
Path repo = initRepo(tmp.resolve("repo")); // only .mcp.json + README committed
|
||||||
|
|
||||||
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-134-log-partial", "HEAD");
|
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-134-log-partial", "HEAD");
|
||||||
@@ -1435,7 +1429,7 @@ class GitWorktreesTest {
|
|||||||
/** Criterion 2: both candidates present — the summary line names both and the denominator. */
|
/** Criterion 2: both candidates present — the summary line names both and the denominator. */
|
||||||
@Test
|
@Test
|
||||||
void overlayParityLogsBothCopiedWhenBothCandidatesArePresent(@TempDir Path tmp) throws Exception {
|
void overlayParityLogsBothCopiedWhenBothCandidatesArePresent(@TempDir Path tmp) throws Exception {
|
||||||
reportingLogger.setLevel(Level.INFO);
|
reportingLog.setLevel(Level.INFO);
|
||||||
Path repo = initRepo(tmp.resolve("repo"));
|
Path repo = initRepo(tmp.resolve("repo"));
|
||||||
Files.writeString(repo.resolve(".env"), "A=1\n");
|
Files.writeString(repo.resolve(".env"), "A=1\n");
|
||||||
Files.writeString(repo.resolve(".envrc"), "export A=1\n");
|
Files.writeString(repo.resolve(".envrc"), "export A=1\n");
|
||||||
@@ -1452,7 +1446,7 @@ class GitWorktreesTest {
|
|||||||
* denominator, and why the other candidate was not copied. */
|
* denominator, and why the other candidate was not copied. */
|
||||||
@Test
|
@Test
|
||||||
void overlayParityLogsOneCopiedOneAbsent(@TempDir Path tmp) throws Exception {
|
void overlayParityLogsOneCopiedOneAbsent(@TempDir Path tmp) throws Exception {
|
||||||
reportingLogger.setLevel(Level.INFO);
|
reportingLog.setLevel(Level.INFO);
|
||||||
Path repo = initRepo(tmp.resolve("repo"));
|
Path repo = initRepo(tmp.resolve("repo"));
|
||||||
Files.writeString(repo.resolve(".env"), "A=1\n");
|
Files.writeString(repo.resolve(".env"), "A=1\n");
|
||||||
// .envrc deliberately not created — the absent candidate.
|
// .envrc deliberately not created — the absent candidate.
|
||||||
@@ -1472,7 +1466,7 @@ class GitWorktreesTest {
|
|||||||
* change, silently, unless this line told it so beforehand. */
|
* change, silently, unless this line told it so beforehand. */
|
||||||
@Test
|
@Test
|
||||||
void overlayParityLogsSkipWorktreeConsequenceForATrackedFile(@TempDir Path tmp) throws Exception {
|
void overlayParityLogsSkipWorktreeConsequenceForATrackedFile(@TempDir Path tmp) throws Exception {
|
||||||
reportingLogger.setLevel(Level.INFO);
|
reportingLog.setLevel(Level.INFO);
|
||||||
Path repo = initRepo(tmp.resolve("repo"));
|
Path repo = initRepo(tmp.resolve("repo"));
|
||||||
Files.writeString(repo.resolve(".env"), "A=1\n");
|
Files.writeString(repo.resolve(".env"), "A=1\n");
|
||||||
git(repo, "add", ".env");
|
git(repo, "add", ".env");
|
||||||
@@ -1495,7 +1489,7 @@ class GitWorktreesTest {
|
|||||||
/** Criterion 5: null and empty overlay lists return quietly — no exception, no log noise. */
|
/** Criterion 5: null and empty overlay lists return quietly — no exception, no log noise. */
|
||||||
@Test
|
@Test
|
||||||
void overlayParityWithNoCandidatesLogsNothing(@TempDir Path tmp) throws Exception {
|
void overlayParityWithNoCandidatesLogsNothing(@TempDir Path tmp) throws Exception {
|
||||||
reportingLogger.setLevel(Level.INFO);
|
reportingLog.setLevel(Level.INFO);
|
||||||
Path repo = initRepo(tmp.resolve("repo"));
|
Path repo = initRepo(tmp.resolve("repo"));
|
||||||
Path wt = bareWorktree(repo, tmp.resolve("wt"), "cb134-empty");
|
Path wt = bareWorktree(repo, tmp.resolve("wt"), "cb134-empty");
|
||||||
GitWorktrees worktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
GitWorktrees worktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||||
@@ -1503,7 +1497,7 @@ class GitWorktreesTest {
|
|||||||
worktrees.overlayParity(repo.toString(), wt.toString(), null);
|
worktrees.overlayParity(repo.toString(), wt.toString(), null);
|
||||||
worktrees.overlayParity(repo.toString(), wt.toString(), List.of());
|
worktrees.overlayParity(repo.toString(), wt.toString(), List.of());
|
||||||
|
|
||||||
assertTrue(reportingAppender.list.isEmpty(),
|
assertTrue(reportingLog.events().isEmpty(),
|
||||||
"a null/empty overlay must log nothing, got:\n" + capturedMessages());
|
"a null/empty overlay must log nothing, got:\n" + capturedMessages());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1683,7 +1677,7 @@ class GitWorktreesTest {
|
|||||||
* denominator, what was seeded, and what was kept because the repo already had it. */
|
* denominator, what was seeded, and what was kept because the repo already had it. */
|
||||||
@Test
|
@Test
|
||||||
void seedSkillsLogsSeededAndKept(@TempDir Path tmp) throws Exception {
|
void seedSkillsLogsSeededAndKept(@TempDir Path tmp) throws Exception {
|
||||||
reportingLogger.setLevel(Level.INFO);
|
reportingLog.setLevel(Level.INFO);
|
||||||
Path repo = tmp.resolve("repo");
|
Path repo = tmp.resolve("repo");
|
||||||
Files.createDirectories(repo);
|
Files.createDirectories(repo);
|
||||||
git(repo, "init", "-q", "-b", "main");
|
git(repo, "init", "-q", "-b", "main");
|
||||||
|
|||||||
@@ -1,9 +1,7 @@
|
|||||||
package dev.ltms.fleet.session;
|
package dev.ltms.fleet.session;
|
||||||
|
|
||||||
import ch.qos.logback.classic.Level;
|
import ch.qos.logback.classic.Level;
|
||||||
import ch.qos.logback.classic.LoggerContext;
|
|
||||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
import ch.qos.logback.core.read.ListAppender;
|
|
||||||
import dev.ltms.fleet.auth.MemberRegistry;
|
import dev.ltms.fleet.auth.MemberRegistry;
|
||||||
import dev.ltms.fleet.auth.MemberLifecycle;
|
import dev.ltms.fleet.auth.MemberLifecycle;
|
||||||
import dev.ltms.fleet.config.ConfigRef;
|
import dev.ltms.fleet.config.ConfigRef;
|
||||||
@@ -25,7 +23,13 @@ import dev.ltms.fleet.peer.SpawnRequest;
|
|||||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||||
import dev.ltms.fleet.placement.PlacementDecision;
|
import dev.ltms.fleet.placement.PlacementDecision;
|
||||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||||
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
|
import org.junit.jupiter.api.AfterAll;
|
||||||
|
import org.junit.jupiter.api.BeforeAll;
|
||||||
|
import org.junit.jupiter.api.MethodOrderer;
|
||||||
|
import org.junit.jupiter.api.Order;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.TestMethodOrder;
|
||||||
import org.junit.jupiter.api.io.TempDir;
|
import org.junit.jupiter.api.io.TempDir;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
@@ -50,9 +54,46 @@ import static org.junit.jupiter.api.Assertions.*;
|
|||||||
* CB-301 / CB-303 acceptance tests for the authoritative session registry, one-shot lifecycle FSM,
|
* CB-301 / CB-303 acceptance tests for the authoritative session registry, one-shot lifecycle FSM,
|
||||||
* and configurable lifecycle limits (idle TTL, context cap, drain).
|
* and configurable lifecycle limits (idle TTL, context cap, drain).
|
||||||
* No live herdr — everything runs against the same {@link FakeHerdr} the rest of the project uses.
|
* No live herdr — everything runs against the same {@link FakeHerdr} the rest of the project uses.
|
||||||
|
*
|
||||||
|
* <p>fleetd #525: only {@link #onTurnFailedIsLoggedAtWarnWithThePriorState} (explicitly
|
||||||
|
* {@link Order#value() @Order(1)}) and the proving test right after it
|
||||||
|
* ({@link #sharedSessionManagerLoggerLevelIsRestoredAfterOnTurnFailedPinsWarn}, {@code @Order(2)})
|
||||||
|
* care about method order — every other test here has no {@code @Order} and so runs after both of
|
||||||
|
* these (JUnit 5's {@link MethodOrderer.OrderAnnotation} gives an unannotated method the lowest
|
||||||
|
* priority), in whatever relative order it already ran in.
|
||||||
*/
|
*/
|
||||||
|
@TestMethodOrder(MethodOrderer.OrderAnnotation.class)
|
||||||
class SessionManagerTest {
|
class SessionManagerTest {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #525: the level {@link SessionManager}'s logger had when this class started, captured
|
||||||
|
* before any test here — including the leak this ticket fixes — can touch it. {@code
|
||||||
|
* pinSessionManagerLoggerToAKnownBaseline} then forces a distinctive, known value (DEBUG) so
|
||||||
|
* {@link #sharedSessionManagerLoggerLevelIsRestoredAfterOnTurnFailedPinsWarn} can tell "the
|
||||||
|
* level came back to what it was" apart from "the level happens to already be WARN because
|
||||||
|
* some earlier test class in this JVM fork (surefire reuses forks by default) left it there" —
|
||||||
|
* a real risk, since {@code ch.qos.logback.classic.Logger} instances are cached per class and
|
||||||
|
* shared across the whole JVM, and this exact logger is also touched by
|
||||||
|
* {@code WorktreeSessionManagerTest#releasePreservesDirtyWorktreeAndLogsWarn}, which has the
|
||||||
|
* same unfixed leak (reported, not fixed — out of this ticket's scope).
|
||||||
|
*/
|
||||||
|
private static Level sessionManagerLevelBeforeThisClass;
|
||||||
|
|
||||||
|
@BeforeAll
|
||||||
|
static void pinSessionManagerLoggerToAKnownBaseline() {
|
||||||
|
ch.qos.logback.classic.Logger sessionLog =
|
||||||
|
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||||
|
sessionManagerLevelBeforeThisClass = sessionLog.getLevel();
|
||||||
|
sessionLog.setLevel(Level.DEBUG);
|
||||||
|
}
|
||||||
|
|
||||||
|
@AfterAll
|
||||||
|
static void restoreSessionManagerLoggerLevel() {
|
||||||
|
ch.qos.logback.classic.Logger sessionLog =
|
||||||
|
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||||
|
sessionLog.setLevel(sessionManagerLevelBeforeThisClass);
|
||||||
|
}
|
||||||
|
|
||||||
private SessionManager sessionManager(FakeHerdr herdr) {
|
private SessionManager sessionManager(FakeHerdr herdr) {
|
||||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||||
@@ -81,6 +122,9 @@ class SessionManagerTest {
|
|||||||
return new SessionManager(workers, worktrees, clock);
|
return new SessionManager(workers, worktrees, clock);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// fleetd #529: CapturedLog moved to dev.ltms.fleet.testing.CapturedLog (imported above) so
|
||||||
|
// every test file shares one implementation instead of each hand-rolling its own capture.
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* CB-581: a {@link Worktrees} test double whose {@code hasUncommitted} and {@code remove} can
|
* CB-581: a {@link Worktrees} test double whose {@code hasUncommitted} and {@code remove} can
|
||||||
* be told to throw, so {@link SessionManager#release} can be exercised against exactly the
|
* be told to throw, so {@link SessionManager#release} can be exercised against exactly the
|
||||||
@@ -466,24 +510,15 @@ class SessionManagerTest {
|
|||||||
|
|
||||||
@Test
|
@Test
|
||||||
void backendErrorForUnknownTargetIsWarnedAndDoesNotCreateASession() {
|
void backendErrorForUnknownTargetIsWarnedAndDoesNotCreateASession() {
|
||||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
try (CapturedLog log = CapturedLog.of(SessionManager.class)) {
|
||||||
ch.qos.logback.classic.Logger sessionLog =
|
|
||||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.setContext(ctx);
|
|
||||||
appender.start();
|
|
||||||
sessionLog.addAppender(appender);
|
|
||||||
try {
|
|
||||||
SessionManager sessions = sessionManager(new FakeHerdr());
|
SessionManager sessions = sessionManager(new FakeHerdr());
|
||||||
|
|
||||||
assertFalse(sessions.onBackendError("term_missing", "backend exited"));
|
assertFalse(sessions.onBackendError("term_missing", "backend exited"));
|
||||||
|
|
||||||
assertTrue(sessions.roster().isEmpty(), "unknown target must not create a session");
|
assertTrue(sessions.roster().isEmpty(), "unknown target must not create a session");
|
||||||
assertTrue(appender.list.stream().anyMatch(e -> e.getLevel().equals(Level.WARN)
|
assertTrue(log.events().stream().anyMatch(e -> e.getLevel().equals(Level.WARN)
|
||||||
&& e.getFormattedMessage().contains("term_missing")),
|
&& e.getFormattedMessage().contains("term_missing")),
|
||||||
"unknown target is logged at WARN");
|
"unknown target is logged at WARN");
|
||||||
} finally {
|
|
||||||
sessionLog.detachAppender(appender);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -516,19 +551,12 @@ class SessionManagerTest {
|
|||||||
}
|
}
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
|
@Order(1)
|
||||||
void onTurnFailedIsLoggedAtWarnWithThePriorState() {
|
void onTurnFailedIsLoggedAtWarnWithThePriorState() {
|
||||||
// CB-564: this transition used to be a bare DEBUG "session marked failed" — a symptom with no
|
// CB-564: this transition used to be a bare DEBUG "session marked failed" — a symptom with no
|
||||||
// cause. A member that can no longer be delegated to must be at least WARN, and should name
|
// cause. A member that can no longer be delegated to must be at least WARN, and should name
|
||||||
// what stage it failed at (here: BUSY, i.e. a turn was in flight and never resolved).
|
// what stage it failed at (here: BUSY, i.e. a turn was in flight and never resolved).
|
||||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
try (CapturedLog log = CapturedLog.at(SessionManager.class, Level.WARN)) {
|
||||||
ch.qos.logback.classic.Logger sessionLog =
|
|
||||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.setContext(ctx);
|
|
||||||
appender.start();
|
|
||||||
sessionLog.addAppender(appender);
|
|
||||||
sessionLog.setLevel(Level.WARN);
|
|
||||||
try {
|
|
||||||
FakeHerdr herdr = new FakeHerdr();
|
FakeHerdr herdr = new FakeHerdr();
|
||||||
SessionManager sessions = sessionManager(herdr);
|
SessionManager sessions = sessionManager(herdr);
|
||||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||||
@@ -538,18 +566,37 @@ class SessionManagerTest {
|
|||||||
|
|
||||||
sessions.onTurnFailed(terminal);
|
sessions.onTurnFailed(terminal);
|
||||||
|
|
||||||
String warn = appender.list.stream()
|
String warn = log.events().stream()
|
||||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||||
.map(ILoggingEvent::getFormattedMessage)
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
.findFirst()
|
.findFirst()
|
||||||
.orElse("no turn-failed WARN logged");
|
.orElse("no turn-failed WARN logged");
|
||||||
assertTrue(warn.contains(terminal), "the log names the member: " + warn);
|
assertTrue(warn.contains(terminal), "the log names the member: " + warn);
|
||||||
assertTrue(warn.contains("BUSY"), "the log names the stage it failed at: " + warn);
|
assertTrue(warn.contains("BUSY"), "the log names the stage it failed at: " + warn);
|
||||||
} finally {
|
|
||||||
sessionLog.detachAppender(appender);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #525: proves the leak in {@link #onTurnFailedIsLoggedAtWarnWithThePriorState} above
|
||||||
|
* (which runs immediately before this, via {@code @Order}) is closed. That test pins the
|
||||||
|
* shared {@link SessionManager} logger to WARN through a {@link CapturedLog}; if {@link
|
||||||
|
* CapturedLog#close} only detached the appender — the original bug, before this ticket's fix —
|
||||||
|
* the level would still read WARN here instead of the {@code DEBUG} baseline this class's
|
||||||
|
* {@code @BeforeAll} set. Runs at {@code @Order(2)}, guaranteed after {@code @Order(1)} and
|
||||||
|
* before every other (unannotated) test in this class.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
@Order(2)
|
||||||
|
void sharedSessionManagerLoggerLevelIsRestoredAfterOnTurnFailedPinsWarn() {
|
||||||
|
ch.qos.logback.classic.Logger sessionLog =
|
||||||
|
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||||
|
assertEquals(Level.DEBUG, sessionLog.getLevel(),
|
||||||
|
"onTurnFailedIsLoggedAtWarnWithThePriorState pins the shared SessionManager logger "
|
||||||
|
+ "to WARN; its cleanup must restore the level it captured (DEBUG, set by "
|
||||||
|
+ "this class's @BeforeAll) rather than leaving WARN pinned for every test "
|
||||||
|
+ "that runs after it");
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* fleetd #226: a contended slot is refused through the real {@link SessionManager#acquire}
|
* fleetd #226: a contended slot is refused through the real {@link SessionManager#acquire}
|
||||||
* path before the real launcher can hand an architect charter to a process.
|
* path before the real launcher can hand an architect charter to a process.
|
||||||
@@ -576,28 +623,18 @@ class SessionManagerTest {
|
|||||||
SessionManager sessions = sessionManager(herdr);
|
SessionManager sessions = sessionManager(herdr);
|
||||||
MemberRegistry members = architectRegistry();
|
MemberRegistry members = architectRegistry();
|
||||||
sessions.setMemberLifecycle(bindFailureAfterReservation(members));
|
sessions.setMemberLifecycle(bindFailureAfterReservation(members));
|
||||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
try (CapturedLog log = CapturedLog.at(MemberRegistry.class, Level.WARN)) {
|
||||||
ch.qos.logback.classic.Logger registryLog = (ch.qos.logback.classic.Logger)
|
|
||||||
LoggerFactory.getLogger(MemberRegistry.class);
|
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.setContext(ctx);
|
|
||||||
appender.start();
|
|
||||||
registryLog.addAppender(appender);
|
|
||||||
registryLog.setLevel(Level.WARN);
|
|
||||||
try {
|
|
||||||
MemberSession session = sessions.acquire("ltms-local", MemberRole.ARCHITECT, null,
|
MemberSession session = sessions.acquire("ltms-local", MemberRole.ARCHITECT, null,
|
||||||
"/caller", "term_primary", null);
|
"/caller", "term_primary", null);
|
||||||
|
|
||||||
assertEquals(MemberRole.DEV, session.role(), "a failed reservation bind must use the fallback");
|
assertEquals(MemberRole.DEV, session.role(), "a failed reservation bind must use the fallback");
|
||||||
String warn = appender.list.stream()
|
String warn = log.events().stream()
|
||||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||||
.map(ILoggingEvent::getFormattedMessage)
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
.findFirst()
|
.findFirst()
|
||||||
.orElse("no slot-exhaustion WARN logged");
|
.orElse("no slot-exhaustion WARN logged");
|
||||||
assertTrue(warn.contains("ltms-local"), "the WARN names the profile: " + warn);
|
assertTrue(warn.contains("ltms-local"), "the WARN names the profile: " + warn);
|
||||||
assertTrue(warn.contains(session.terminalId()), "the WARN names the terminal: " + warn);
|
assertTrue(warn.contains(session.terminalId()), "the WARN names the terminal: " + warn);
|
||||||
} finally {
|
|
||||||
registryLog.detachAppender(appender);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -902,6 +939,82 @@ class SessionManagerTest {
|
|||||||
.count();
|
.count();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #512: a drain that releases every session cleanly used to log nothing at all — the
|
||||||
|
* two log calls in {@code drainAll}/{@code drainSnapshot} both sit on abnormal paths, so
|
||||||
|
* "clean drain" and "died on the first session" were indistinguishable. This asserts the new
|
||||||
|
* {@code log.info} line fires on the ordinary, nothing-went-wrong path, and that its numbers
|
||||||
|
* are the real counts (two released, zero abandoned) rather than just a non-empty string.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void drainAllLogsACompletionLineWithTheRealCountsOnACleanDrain() {
|
||||||
|
FakeHerdr herdr = new FakeHerdr();
|
||||||
|
SessionManager sessions = sessionManager(herdr);
|
||||||
|
MemberSession first = sessions.acquire("ltms-local", "/one", "/caller", "ownerOne");
|
||||||
|
MemberSession second = sessions.acquire("ltms-local", "/two", "/caller", "ownerTwo");
|
||||||
|
sessions.asPresence().markPresent(first.terminalId());
|
||||||
|
sessions.asPresence().markPresent(second.terminalId());
|
||||||
|
// Both stay READY — neither is delivered a turn, so neither is BUSY and the drain below
|
||||||
|
// has nothing abnormal to hit.
|
||||||
|
|
||||||
|
// Pin INFO explicitly: fleetd #525 made CapturedLog itself restore the level it pins, but
|
||||||
|
// this pin stays anyway as belt-and-braces — a later change to the sweep must not be able
|
||||||
|
// to make this INFO assertion vacuous again by leaving some other test's WARN pin in place.
|
||||||
|
try (CapturedLog log = CapturedLog.at(SessionManager.class, Level.INFO)) {
|
||||||
|
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||||
|
|
||||||
|
assertTrue(sessions.roster().isEmpty(), "precondition: the drain actually ran");
|
||||||
|
String info = log.events().stream()
|
||||||
|
.filter(e -> e.getLevel().equals(Level.INFO))
|
||||||
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
|
.filter(m -> m.contains("drain complete"))
|
||||||
|
.findFirst()
|
||||||
|
.orElse("no drain-complete INFO logged");
|
||||||
|
assertTrue(info.contains("released=2"),
|
||||||
|
"both released sessions must be counted: " + info);
|
||||||
|
assertTrue(info.contains("abandoned=0"),
|
||||||
|
"neither session was BUSY, so nothing was abandoned mid-turn: " + info);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #512: the same completion line must also report a non-zero abandoned count when a
|
||||||
|
* session is still {@code BUSY} once the whole-drain deadline passes — the case the ticket
|
||||||
|
* calls out as the one a script needs to be able to see. Reuses the same BUSY/READY mix as
|
||||||
|
* {@link #drainAllReleasesBusyAndReadySessionsAndWaitsForBusy}, which already forces the busy
|
||||||
|
* session to spin until the real-time deadline expires (its state never leaves BUSY on its
|
||||||
|
* own), and adds the log assertion that test does not make.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void drainAllLogsANonZeroAbandonedCountForASessionStillBusyAtTheDeadline() {
|
||||||
|
FakeHerdr herdr = new FakeHerdr();
|
||||||
|
SessionManager sessions = sessionManager(herdr);
|
||||||
|
MemberSession ready = sessions.acquire("ltms-local", "/ready", "/caller", "ownerR");
|
||||||
|
MemberSession busy = sessions.acquire("ltms-local", "/busy", "/caller", "ownerB");
|
||||||
|
sessions.asPresence().markPresent(ready.terminalId());
|
||||||
|
sessions.asPresence().markPresent(busy.terminalId());
|
||||||
|
sessions.onDelivered(busy.terminalId(), TestTurnTokens.inert(busy.terminalId()));
|
||||||
|
// busy never leaves BUSY — no completion is delivered — so the drain below must spin the
|
||||||
|
// full timeout and then release it anyway, counting it abandoned.
|
||||||
|
|
||||||
|
// Pin INFO explicitly — see the comment in drainAllLogsACompletionLineWithTheRealCountsOnACleanDrain.
|
||||||
|
try (CapturedLog log = CapturedLog.at(SessionManager.class, Level.INFO)) {
|
||||||
|
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||||
|
|
||||||
|
assertTrue(sessions.roster().isEmpty(), "precondition: the drain actually ran");
|
||||||
|
String info = log.events().stream()
|
||||||
|
.filter(e -> e.getLevel().equals(Level.INFO))
|
||||||
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
|
.filter(m -> m.contains("drain complete"))
|
||||||
|
.findFirst()
|
||||||
|
.orElse("no drain-complete INFO logged");
|
||||||
|
assertTrue(info.contains("released=2"),
|
||||||
|
"both the ready and the busy session are released: " + info);
|
||||||
|
assertTrue(info.contains("abandoned=1"),
|
||||||
|
"the busy session hit the deadline still BUSY and must be counted: " + info);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// --- fleetd #308: a spawn accepted while the shutdown drain is running must not orphan ---
|
// --- fleetd #308: a spawn accepted while the shutdown drain is running must not orphan ---
|
||||||
|
|
||||||
@Test
|
@Test
|
||||||
@@ -1202,21 +1315,13 @@ class SessionManagerTest {
|
|||||||
new WorktreeRequest("cb-581a", null));
|
new WorktreeRequest("cb-581a", null));
|
||||||
worktrees.failHasUncommittedWith(new WorktreeException("git status exited 128"));
|
worktrees.failHasUncommittedWith(new WorktreeException("git status exited 128"));
|
||||||
|
|
||||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
try (CapturedLog log = CapturedLog.at(SessionManager.class, Level.WARN)) {
|
||||||
ch.qos.logback.classic.Logger sessionLog =
|
|
||||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.setContext(ctx);
|
|
||||||
appender.start();
|
|
||||||
sessionLog.addAppender(appender);
|
|
||||||
sessionLog.setLevel(Level.WARN);
|
|
||||||
try {
|
|
||||||
assertDoesNotThrow(() -> sessions.release(s.paneId()),
|
assertDoesNotThrow(() -> sessions.release(s.paneId()),
|
||||||
"a throwing dirty check must not abort the release");
|
"a throwing dirty check must not abort the release");
|
||||||
|
|
||||||
assertTrue(worktrees.removeCalls().isEmpty(),
|
assertTrue(worktrees.removeCalls().isEmpty(),
|
||||||
"the worktree is preserved when its dirty state cannot be determined");
|
"the worktree is preserved when its dirty state cannot be determined");
|
||||||
String warn = appender.list.stream()
|
String warn = log.events().stream()
|
||||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||||
.map(ILoggingEvent::getFormattedMessage)
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
.filter(m -> m.contains(s.worktree()))
|
.filter(m -> m.contains(s.worktree()))
|
||||||
@@ -1224,8 +1329,6 @@ class SessionManagerTest {
|
|||||||
.orElse("no warn logged naming the worktree");
|
.orElse("no warn logged naming the worktree");
|
||||||
assertTrue(warn.contains(s.paneId()), "the WARN names the pane: " + warn);
|
assertTrue(warn.contains(s.paneId()), "the WARN names the pane: " + warn);
|
||||||
assertTrue(warn.contains(s.terminalId()), "the WARN names the terminal: " + warn);
|
assertTrue(warn.contains(s.terminalId()), "the WARN names the terminal: " + warn);
|
||||||
} finally {
|
|
||||||
sessionLog.detachAppender(appender);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1392,20 +1495,12 @@ class SessionManagerTest {
|
|||||||
// this itself, so it no longer propagates out of release() at all.
|
// this itself, so it no longer propagates out of release() at all.
|
||||||
worktrees.failRemoveFor(b.worktree());
|
worktrees.failRemoveFor(b.worktree());
|
||||||
|
|
||||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
|
||||||
ch.qos.logback.classic.Logger sessionLog =
|
|
||||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.setContext(ctx);
|
|
||||||
appender.start();
|
|
||||||
sessionLog.addAppender(appender);
|
|
||||||
sessionLog.setLevel(Level.WARN);
|
|
||||||
int reaped;
|
int reaped;
|
||||||
try {
|
try (CapturedLog log = CapturedLog.at(SessionManager.class, Level.WARN)) {
|
||||||
clock[0] = 100;
|
clock[0] = 100;
|
||||||
reaped = sessions.reapIdle(10);
|
reaped = sessions.reapIdle(10);
|
||||||
|
|
||||||
String warn = appender.list.stream()
|
String warn = log.events().stream()
|
||||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||||
.map(ILoggingEvent::getFormattedMessage)
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
.filter(m -> m.contains(b.paneId()))
|
.filter(m -> m.contains(b.paneId()))
|
||||||
@@ -1413,8 +1508,6 @@ class SessionManagerTest {
|
|||||||
.orElse("no worktree-removal-failure WARN logged");
|
.orElse("no worktree-removal-failure WARN logged");
|
||||||
assertTrue(warn.contains(b.terminalId()), "the WARN names the failed session's terminal: " + warn);
|
assertTrue(warn.contains(b.terminalId()), "the WARN names the failed session's terminal: " + warn);
|
||||||
assertTrue(warn.contains(b.worktree()), "the WARN names the failed session's worktree: " + warn);
|
assertTrue(warn.contains(b.worktree()), "the WARN names the failed session's worktree: " + warn);
|
||||||
} finally {
|
|
||||||
sessionLog.detachAppender(appender);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
assertEquals(3, reaped,
|
assertEquals(3, reaped,
|
||||||
@@ -1460,28 +1553,18 @@ class SessionManagerTest {
|
|||||||
// trigger reapIdle's own guard is for, now that #283 closed the worktree-removal trigger.
|
// trigger reapIdle's own guard is for, now that #283 closed the worktree-removal trigger.
|
||||||
herdr.paneCloseFailsForPane("w9:pRoot_2", "internal_error");
|
herdr.paneCloseFailsForPane("w9:pRoot_2", "internal_error");
|
||||||
|
|
||||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
|
||||||
ch.qos.logback.classic.Logger sessionLog =
|
|
||||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.setContext(ctx);
|
|
||||||
appender.start();
|
|
||||||
sessionLog.addAppender(appender);
|
|
||||||
sessionLog.setLevel(Level.WARN);
|
|
||||||
int reaped;
|
int reaped;
|
||||||
try {
|
try (CapturedLog log = CapturedLog.at(SessionManager.class, Level.WARN)) {
|
||||||
clock[0] = 100;
|
clock[0] = 100;
|
||||||
reaped = sessions.reapIdle(10);
|
reaped = sessions.reapIdle(10);
|
||||||
|
|
||||||
String warn = appender.list.stream()
|
String warn = log.events().stream()
|
||||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||||
.map(ILoggingEvent::getFormattedMessage)
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
.filter(m -> m.contains("reap failed") && m.contains(b.paneId()))
|
.filter(m -> m.contains("reap failed") && m.contains(b.paneId()))
|
||||||
.findFirst()
|
.findFirst()
|
||||||
.orElse("no reap-failed WARN logged for the failing session");
|
.orElse("no reap-failed WARN logged for the failing session");
|
||||||
assertTrue(warn.contains(b.terminalId()), "the WARN names the failed session's terminal: " + warn);
|
assertTrue(warn.contains(b.terminalId()), "the WARN names the failed session's terminal: " + warn);
|
||||||
} finally {
|
|
||||||
sessionLog.detachAppender(appender);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
assertEquals(2, reaped,
|
assertEquals(2, reaped,
|
||||||
|
|||||||
@@ -0,0 +1,91 @@
|
|||||||
|
package dev.ltms.fleet.session;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
|
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Set;
|
||||||
|
import java.util.concurrent.CountDownLatch;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.concurrent.atomic.AtomicBoolean;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertNotSame;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
class SessionReaperResilienceTest {
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void anErrorInOneIterationDoesNotStopTheNextIteration() throws Exception {
|
||||||
|
CountDownLatch errorThrown = new CountDownLatch(1);
|
||||||
|
CountDownLatch nextIteration = new CountDownLatch(1);
|
||||||
|
AtomicBoolean first = new AtomicBoolean(true);
|
||||||
|
LongSupplier clock = () -> {
|
||||||
|
if (first.compareAndSet(true, false)) {
|
||||||
|
errorThrown.countDown();
|
||||||
|
throw new AssertionError("test Error from the first reap iteration");
|
||||||
|
}
|
||||||
|
nextIteration.countDown();
|
||||||
|
return System.nanoTime();
|
||||||
|
};
|
||||||
|
SessionReaper reaper = new SessionReaper(sessionManager(clock), 60, 1);
|
||||||
|
reaper.start();
|
||||||
|
try {
|
||||||
|
assertTrue(errorThrown.await(2, TimeUnit.SECONDS),
|
||||||
|
"the first reap iteration must throw its test Error");
|
||||||
|
assertTrue(nextIteration.await(2, TimeUnit.SECONDS),
|
||||||
|
"the reaper must continue to the next iteration after an Error");
|
||||||
|
} finally {
|
||||||
|
reaper.stop();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void anAbnormalExitClearsRunningSoStartCreatesANewLoop() throws Exception {
|
||||||
|
SessionReaper reaper = new SessionReaper(sessionManager(System::nanoTime), 60, -1);
|
||||||
|
reaper.start();
|
||||||
|
Thread first = threadOf(reaper);
|
||||||
|
first.join(2000);
|
||||||
|
assertFalse(runningOf(reaper), "an abnormal loop exit must clear running");
|
||||||
|
|
||||||
|
reaper.start();
|
||||||
|
Thread restarted = threadOf(reaper);
|
||||||
|
try {
|
||||||
|
assertNotSame(first, restarted, "start() must create a new loop after an abnormal exit");
|
||||||
|
restarted.join(2000);
|
||||||
|
} finally {
|
||||||
|
reaper.stop();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static SessionManager sessionManager(LongSupplier clock) {
|
||||||
|
FakeHerdr herdr = new FakeHerdr();
|
||||||
|
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||||
|
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||||
|
List.of("ccs", "ltms-local"), "tab", "fleetd-workers",
|
||||||
|
"worker: {profile} #{n}", null, null, null);
|
||||||
|
ClaudeCodeLauncher launcher = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||||
|
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||||
|
return new SessionManager(launcher, new FakeWorktrees(), clock);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Thread threadOf(SessionReaper reaper) throws ReflectiveOperationException {
|
||||||
|
Field field = SessionReaper.class.getDeclaredField("thread");
|
||||||
|
field.setAccessible(true);
|
||||||
|
return (Thread) field.get(reaper);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static boolean runningOf(SessionReaper reaper) throws ReflectiveOperationException {
|
||||||
|
Field field = SessionReaper.class.getDeclaredField("running");
|
||||||
|
field.setAccessible(true);
|
||||||
|
return field.getBoolean(reaper);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,170 @@
|
|||||||
|
package dev.ltms.fleet.session;
|
||||||
|
|
||||||
|
import dev.ltms.fleet.config.FleetConfig;
|
||||||
|
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||||
|
import dev.ltms.fleet.herdr.AgentControl;
|
||||||
|
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||||
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
|
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
|
||||||
|
import java.lang.reflect.Field;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Map;
|
||||||
|
import java.util.Set;
|
||||||
|
import java.util.concurrent.CountDownLatch;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
import java.util.concurrent.atomic.AtomicLong;
|
||||||
|
import java.util.function.LongSupplier;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #544: {@link SessionReaper#health()} — the progress watchdog, not a liveness check. Every
|
||||||
|
* test drives {@link LoopWatchdog}'s clock explicitly (never real elapsed time), so the threshold
|
||||||
|
* crossing is deterministic rather than a race against the poll interval.
|
||||||
|
*/
|
||||||
|
class SessionReaperWatchdogTest {
|
||||||
|
|
||||||
|
private static final long INTERVAL_MILLIS = 50;
|
||||||
|
// Mirrors SessionReaper.staleAfterNanos(50): 50ms * the 12x multiplier = 600ms.
|
||||||
|
private static final long STALE_AFTER_NANOS = TimeUnit.MILLISECONDS.toNanos(INTERVAL_MILLIS) * 12;
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aLoopParkedInAReapCallThatNeverReturnsReportsStalled() throws Exception {
|
||||||
|
CountDownLatch entered = new CountDownLatch(1);
|
||||||
|
// reapIdle's very first statement reads sessions' own injected clock — blocking there
|
||||||
|
// parks the reaper thread mid-round, exactly as a wedged call would (fleetd #544).
|
||||||
|
SessionManager sessions = sessionManager(new BlocksOnFirstCall(entered));
|
||||||
|
AtomicLong watchdogClock = new AtomicLong(0);
|
||||||
|
SessionReaper reaper = new SessionReaper(sessions, 60, INTERVAL_MILLIS, watchdogClock::get);
|
||||||
|
reaper.start();
|
||||||
|
try {
|
||||||
|
assertTrue(entered.await(2, TimeUnit.SECONDS),
|
||||||
|
"the reaper must have entered the blocking reapIdle call");
|
||||||
|
|
||||||
|
assertEquals(LoopWatchdog.State.RUNNING, reaper.health(),
|
||||||
|
"freshly parked, before the threshold elapses, must still read RUNNING");
|
||||||
|
|
||||||
|
watchdogClock.set(STALE_AFTER_NANOS + 1);
|
||||||
|
assertEquals(LoopWatchdog.State.STALLED, reaper.health(),
|
||||||
|
"a loop parked mid-round past the staleness threshold must report STALLED");
|
||||||
|
} finally {
|
||||||
|
reaper.stop();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aLoopThatDiedReportsStalled() throws Exception {
|
||||||
|
AtomicLong watchdogClock = new AtomicLong(0);
|
||||||
|
// intervalMillis=-1: Thread.sleep(-1) throws IllegalArgumentException, uncaught by loop(),
|
||||||
|
// which kills the carrier thread — the same "died" shape SessionReaperResilienceTest uses.
|
||||||
|
SessionReaper reaper = new SessionReaper(sessionManager(System::nanoTime), 60, -1, watchdogClock::get);
|
||||||
|
reaper.start();
|
||||||
|
Thread thread = threadOf(reaper);
|
||||||
|
thread.join(2000);
|
||||||
|
assertFalse(thread.isAlive(), "the loop must have died from Thread.sleep(-1)");
|
||||||
|
|
||||||
|
// staleAfterNanos(-1) = TimeUnit.MILLISECONDS.toNanos(max(-1,1)) * 12 = 12ms.
|
||||||
|
watchdogClock.set(TimeUnit.MILLISECONDS.toNanos(1) * 12 + 1);
|
||||||
|
assertEquals(LoopWatchdog.State.STALLED, reaper.health(),
|
||||||
|
"a dead loop's last-completed round goes stale and must report STALLED");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aLoopStoppedOnPurposeReportsStoppedNotStalled() throws Exception {
|
||||||
|
AtomicLong watchdogClock = new AtomicLong(0);
|
||||||
|
SessionReaper reaper = new SessionReaper(sessionManager(System::nanoTime), 60, INTERVAL_MILLIS, watchdogClock::get);
|
||||||
|
reaper.start();
|
||||||
|
reaper.stop();
|
||||||
|
|
||||||
|
// Advance the clock far past any staleness threshold — an intentional stop must never
|
||||||
|
// be reported as STALLED no matter how old the last round looks (fleetd #512 shape).
|
||||||
|
watchdogClock.set(STALE_AFTER_NANOS * 100);
|
||||||
|
assertEquals(LoopWatchdog.State.STOPPED, reaper.health(),
|
||||||
|
"stop() must report STOPPED, never STALLED, however stale the last round looks");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aRestartedLoopReportsRunningAgainNotStoppedForever() throws Exception {
|
||||||
|
// fleetd #544 review (issue comment #16944): stoppedByCaller is sticky, and reset() —
|
||||||
|
// called only from start() — is the sole thing that clears it. start() is documented
|
||||||
|
// idempotent and loop()'s own error line says "it can be restarted", so stop() followed
|
||||||
|
// by start() is an anticipated path. Without the reset() call in start(), health() would
|
||||||
|
// report STOPPED forever after a restart even though the loop is genuinely running again.
|
||||||
|
AtomicLong watchdogClock = new AtomicLong(0);
|
||||||
|
SessionReaper reaper = new SessionReaper(sessionManager(System::nanoTime), 60, INTERVAL_MILLIS, watchdogClock::get);
|
||||||
|
reaper.start();
|
||||||
|
reaper.stop();
|
||||||
|
reaper.start();
|
||||||
|
|
||||||
|
assertEquals(LoopWatchdog.State.RUNNING, reaper.health(),
|
||||||
|
"an intentional stop must not outlive the restart that follows it");
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void aHealthyLoopReportsRunning() throws Exception {
|
||||||
|
// Real elapsed time on purpose, unlike the other tests here: a clock frozen at 0 would read
|
||||||
|
// RUNNING even if recordRoundComplete() were never wired into loop() at all, since the
|
||||||
|
// constructor's own initial timestamp already satisfies "not stale yet". Running for several
|
||||||
|
// multiples of this instance's own threshold (5ms interval * 12 = 60ms) and still reading
|
||||||
|
// RUNNING instead proves the loop's repeated round completions are the thing keeping the
|
||||||
|
// staleness deadline pushed forward.
|
||||||
|
long fastIntervalMillis = 5;
|
||||||
|
SessionReaper reaper = new SessionReaper(sessionManager(System::nanoTime), 60, fastIntervalMillis, System::nanoTime);
|
||||||
|
reaper.start();
|
||||||
|
try {
|
||||||
|
Thread.sleep(400);
|
||||||
|
assertEquals(LoopWatchdog.State.RUNNING, reaper.health(),
|
||||||
|
"a healthy loop's repeated round completions must keep pushing the staleness deadline forward");
|
||||||
|
} finally {
|
||||||
|
reaper.stop();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static SessionManager sessionManager(LongSupplier clock) {
|
||||||
|
FakeHerdr herdr = new FakeHerdr();
|
||||||
|
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||||
|
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||||
|
List.of("ccs", "ltms-local"), "tab", "fleetd-workers",
|
||||||
|
"worker: {profile} #{n}", null, null, null);
|
||||||
|
ClaudeCodeLauncher launcher = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||||
|
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||||
|
return new SessionManager(launcher, new FakeWorktrees(), clock);
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Thread threadOf(SessionReaper reaper) throws ReflectiveOperationException {
|
||||||
|
Field field = SessionReaper.class.getDeclaredField("thread");
|
||||||
|
field.setAccessible(true);
|
||||||
|
return (Thread) field.get(reaper);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Blocks forever on its first call, then falls back to real time if ever unblocked. */
|
||||||
|
private static final class BlocksOnFirstCall implements LongSupplier {
|
||||||
|
private final CountDownLatch entered;
|
||||||
|
private volatile boolean first = true;
|
||||||
|
|
||||||
|
private BlocksOnFirstCall(CountDownLatch entered) {
|
||||||
|
this.entered = entered;
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public synchronized long getAsLong() {
|
||||||
|
if (first) {
|
||||||
|
first = false;
|
||||||
|
entered.countDown();
|
||||||
|
try {
|
||||||
|
// Blocks forever, mirroring a genuinely wedged call (fleetd #544) — a test
|
||||||
|
// double standing in for a stuck dependency, not a real socket.
|
||||||
|
new CountDownLatch(1).await();
|
||||||
|
} catch (InterruptedException e) {
|
||||||
|
Thread.currentThread().interrupt();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return System.nanoTime();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -7,13 +7,17 @@ import dev.ltms.fleet.herdr.AgentControl;
|
|||||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||||
import ch.qos.logback.classic.Level;
|
import ch.qos.logback.classic.Level;
|
||||||
import ch.qos.logback.classic.LoggerContext;
|
|
||||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
import ch.qos.logback.core.read.ListAppender;
|
|
||||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||||
import dev.ltms.fleet.msg.TestTurnTokens;
|
import dev.ltms.fleet.msg.TestTurnTokens;
|
||||||
import dev.ltms.fleet.peer.MemberRole;
|
import dev.ltms.fleet.peer.MemberRole;
|
||||||
|
import dev.ltms.fleet.testing.CapturedLog;
|
||||||
|
import org.junit.jupiter.api.AfterAll;
|
||||||
|
import org.junit.jupiter.api.BeforeAll;
|
||||||
|
import org.junit.jupiter.api.MethodOrderer;
|
||||||
|
import org.junit.jupiter.api.Order;
|
||||||
import org.junit.jupiter.api.Test;
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.junit.jupiter.api.TestMethodOrder;
|
||||||
import org.slf4j.LoggerFactory;
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
import java.util.List;
|
import java.util.List;
|
||||||
@@ -27,9 +31,41 @@ import static org.junit.jupiter.api.Assertions.*;
|
|||||||
* CB-301-ext acceptance tests for worktree provisioning and config-parity overlay.
|
* CB-301-ext acceptance tests for worktree provisioning and config-parity overlay.
|
||||||
* No live git — every Worktrees call is handled by {@link FakeWorktrees} and every herdr
|
* No live git — every Worktrees call is handled by {@link FakeWorktrees} and every herdr
|
||||||
* call by {@link FakeHerdr}, matching the project's fake-based test style.
|
* call by {@link FakeHerdr}, matching the project's fake-based test style.
|
||||||
|
*
|
||||||
|
* <p>fleetd #529: only {@link #releasePreservesDirtyWorktreeAndLogsWarn} (explicitly
|
||||||
|
* {@link Order#value() @Order(1)}) and the proving test right after it
|
||||||
|
* ({@link #sharedSessionManagerLoggerLevelIsRestoredAfterDirtyWorktreeReleasePinsWarn},
|
||||||
|
* {@code @Order(2)}) care about method order — every other test here has no {@code @Order} and so
|
||||||
|
* runs after both (JUnit 5's {@link MethodOrderer.OrderAnnotation} gives an unannotated method the
|
||||||
|
* lowest priority), in whatever relative order it already ran in.
|
||||||
*/
|
*/
|
||||||
|
@TestMethodOrder(MethodOrderer.OrderAnnotation.class)
|
||||||
class WorktreeSessionManagerTest {
|
class WorktreeSessionManagerTest {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #529: the level {@link SessionManager}'s logger had when this class started, captured
|
||||||
|
* before any test here touches it, then forced to a distinctive, known value (TRACE) so {@link
|
||||||
|
* #sharedSessionManagerLoggerLevelIsRestoredAfterDirtyWorktreeReleasePinsWarn} can tell "the
|
||||||
|
* level came back to what it was" apart from "the level happens to already be WARN because
|
||||||
|
* some other test class in this JVM fork (surefire reuses forks by default) left it there".
|
||||||
|
*/
|
||||||
|
private static ch.qos.logback.classic.Level sessionManagerLevelBeforeThisClass;
|
||||||
|
|
||||||
|
@BeforeAll
|
||||||
|
static void pinSessionManagerLoggerToAKnownBaseline() {
|
||||||
|
ch.qos.logback.classic.Logger sessionLog =
|
||||||
|
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||||
|
sessionManagerLevelBeforeThisClass = sessionLog.getLevel();
|
||||||
|
sessionLog.setLevel(ch.qos.logback.classic.Level.TRACE);
|
||||||
|
}
|
||||||
|
|
||||||
|
@AfterAll
|
||||||
|
static void restoreSessionManagerLoggerLevel() {
|
||||||
|
ch.qos.logback.classic.Logger sessionLog =
|
||||||
|
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||||
|
sessionLog.setLevel(sessionManagerLevelBeforeThisClass);
|
||||||
|
}
|
||||||
|
|
||||||
private static MemberRegistry members() {
|
private static MemberRegistry members() {
|
||||||
return new MemberRegistry(new FleetConfig.Fleet(Map.of(),
|
return new MemberRegistry(new FleetConfig.Fleet(Map.of(),
|
||||||
Map.of("architect", new FleetConfig.Slot("ltms-local")),
|
Map.of("architect", new FleetConfig.Slot("ltms-local")),
|
||||||
@@ -254,6 +290,7 @@ class WorktreeSessionManagerTest {
|
|||||||
* path, the session, and the cause an operator needs to find the work.
|
* path, the session, and the cause an operator needs to find the work.
|
||||||
*/
|
*/
|
||||||
@Test
|
@Test
|
||||||
|
@Order(1)
|
||||||
void releasePreservesDirtyWorktreeAndLogsWarn() {
|
void releasePreservesDirtyWorktreeAndLogsWarn() {
|
||||||
FakeHerdr herdr = new FakeHerdr();
|
FakeHerdr herdr = new FakeHerdr();
|
||||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt")
|
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt")
|
||||||
@@ -262,21 +299,13 @@ class WorktreeSessionManagerTest {
|
|||||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||||
new WorktreeRequest("cb-576", null));
|
new WorktreeRequest("cb-576", null));
|
||||||
|
|
||||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
try (CapturedLog log = CapturedLog.at(SessionManager.class, Level.WARN)) {
|
||||||
ch.qos.logback.classic.Logger sessionLog =
|
|
||||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
|
||||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
|
||||||
appender.setContext(ctx);
|
|
||||||
appender.start();
|
|
||||||
sessionLog.addAppender(appender);
|
|
||||||
sessionLog.setLevel(Level.WARN);
|
|
||||||
try {
|
|
||||||
sessions.release(s.paneId());
|
sessions.release(s.paneId());
|
||||||
|
|
||||||
assertTrue(herdr.called("pane.close"), "release still tears the worker pane down");
|
assertTrue(herdr.called("pane.close"), "release still tears the worker pane down");
|
||||||
assertTrue(worktrees.removeCalls().isEmpty(),
|
assertTrue(worktrees.removeCalls().isEmpty(),
|
||||||
"a dirty worktree is never removed — it holds the only copy of the work");
|
"a dirty worktree is never removed — it holds the only copy of the work");
|
||||||
String warn = appender.list.stream()
|
String warn = log.events().stream()
|
||||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||||
.map(ILoggingEvent::getFormattedMessage)
|
.map(ILoggingEvent::getFormattedMessage)
|
||||||
.filter(m -> m.contains("dirty worktree"))
|
.filter(m -> m.contains("dirty worktree"))
|
||||||
@@ -285,11 +314,38 @@ class WorktreeSessionManagerTest {
|
|||||||
assertTrue(warn.contains(s.worktree()), "the WARN names the worktree path: " + warn);
|
assertTrue(warn.contains(s.worktree()), "the WARN names the worktree path: " + warn);
|
||||||
assertTrue(warn.contains(s.terminalId()), "the WARN names the session: " + warn);
|
assertTrue(warn.contains(s.terminalId()), "the WARN names the session: " + warn);
|
||||||
assertTrue(warn.contains("COMPLETED"), "the WARN names the release cause: " + warn);
|
assertTrue(warn.contains("COMPLETED"), "the WARN names the release cause: " + warn);
|
||||||
} finally {
|
|
||||||
sessionLog.detachAppender(appender);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #529 proving test: pins that the leak this ticket fixes stays fixed. {@link
|
||||||
|
* #releasePreservesDirtyWorktreeAndLogsWarn} above (which runs immediately before this, via
|
||||||
|
* {@code @Order}) pins the shared {@link SessionManager} logger to WARN through a {@link
|
||||||
|
* CapturedLog}; if {@link CapturedLog#close} only detached the appender — the original bug —
|
||||||
|
* the level would still read WARN here instead of the {@code TRACE} baseline this class's
|
||||||
|
* {@code @BeforeAll} set. Runs at {@code @Order(2)}, guaranteed after {@code @Order(1)} and
|
||||||
|
* before every other (unannotated) test in this class.
|
||||||
|
*
|
||||||
|
* <p>This proves only the WITHIN-CLASS case: JUnit 5's {@code @TestMethodOrder} orders methods
|
||||||
|
* inside one class, not test classes relative to each other, and surefire's default class order
|
||||||
|
* is not something a single test can force. The cross-class leak fleetd #525 measured — this
|
||||||
|
* class's {@code releasePreservesDirtyWorktreeAndLogsWarn} pinning WARN and bleeding into a
|
||||||
|
* later-running {@code SessionManagerTest} in the same fork — is fixed by the same {@link
|
||||||
|
* CapturedLog} mechanism proven here, but that cross-class ordering itself is NOT asserted by
|
||||||
|
* any test and remains unproven by construction.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
@Order(2)
|
||||||
|
void sharedSessionManagerLoggerLevelIsRestoredAfterDirtyWorktreeReleasePinsWarn() {
|
||||||
|
ch.qos.logback.classic.Logger sessionLog =
|
||||||
|
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||||
|
assertEquals(ch.qos.logback.classic.Level.TRACE, sessionLog.getLevel(),
|
||||||
|
"releasePreservesDirtyWorktreeAndLogsWarn pins the shared SessionManager logger to "
|
||||||
|
+ "WARN; its cleanup must restore the level it captured (TRACE, set by this "
|
||||||
|
+ "class's @BeforeAll) rather than leaving WARN pinned for every test that "
|
||||||
|
+ "runs after it");
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* CB-576 review (fleetd #116). A worktree that is already gone (operator cleanup,
|
* CB-576 review (fleetd #116). A worktree that is already gone (operator cleanup,
|
||||||
* {@code git worktree prune}, an earlier half-completed release) must not break teardown.
|
* {@code git worktree prune}, an earlier half-completed release) must not break teardown.
|
||||||
|
|||||||
@@ -0,0 +1,112 @@
|
|||||||
|
package dev.ltms.fleet.testing;
|
||||||
|
|
||||||
|
import ch.qos.logback.classic.Level;
|
||||||
|
import ch.qos.logback.classic.Logger;
|
||||||
|
import ch.qos.logback.classic.LoggerContext;
|
||||||
|
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||||
|
import ch.qos.logback.core.read.ListAppender;
|
||||||
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
import java.util.List;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #529 (promoted from {@code SessionManagerTest}, merged in #527 for fleetd #525): captures
|
||||||
|
* a logger's output and, on {@link #close}, restores <em>both</em> the appender and the level to
|
||||||
|
* what they were before.
|
||||||
|
*
|
||||||
|
* <p>{@code ch.qos.logback.classic.Logger} instances are cached per class and shared across the
|
||||||
|
* whole JVM — and surefire reuses forks by default — so a bare {@code addAppender}/{@code
|
||||||
|
* setLevel} pair whose {@code finally} only detaches the appender leaves the level pinned for
|
||||||
|
* every test that runs after it, in the same class or in a completely unrelated one sharing the
|
||||||
|
* fork. try-with-resources makes "restored the appender but not the level" impossible to write,
|
||||||
|
* because there is only one thing to close.
|
||||||
|
*
|
||||||
|
* <p>New code must use this rather than hand-rolling the {@code ListAppender} + {@code setLevel} +
|
||||||
|
* {@code finally detachAppender} pattern: use {@link #at} or {@link #of}. It is <em>not</em> yet
|
||||||
|
* the only instance of the pattern in this test tree, and the earlier wording here said it was —
|
||||||
|
* which would leave a reader who greps unable to tell a leftover from a violation.
|
||||||
|
*
|
||||||
|
* <p>Measured on main at af95897 (2026-09-12): nine test files still hand-roll it, with 42
|
||||||
|
* {@code setLevel} calls on a raw logback {@code Logger} between them — {@code
|
||||||
|
* FleetdStartupReportTest}, {@code GitHostShapeReportTest}, {@code MemberCredentialsGapReportTest},
|
||||||
|
* {@code MemberTrustModelReportTest}, {@code FleetHealthMonitorTest}, {@code
|
||||||
|
* ClaudeCodeLauncherTest}, {@code HerdrPeerLauncherAllowListWiringTest}, {@code
|
||||||
|
* HerdrPeerLauncherCharterTest} and {@code OpenCodeLauncherTest}. Every one of them pairs its pin
|
||||||
|
* with a restore, so none is the fleetd #525 leak and none was in fleetd #529's scope, which was
|
||||||
|
* the 19 <em>unrestored</em> pins only. They are unmigrated, not broken.
|
||||||
|
*
|
||||||
|
* <p>Re-measure with the two commands below, from the repo root. A file that appears in the first
|
||||||
|
* list and not the second still hand-rolls the pattern. When the first list comes back empty, this
|
||||||
|
* paragraph is spent and the sentence above can go back to saying "the one way" — delete the
|
||||||
|
* paragraph then rather than updating the count.
|
||||||
|
*
|
||||||
|
* <pre>{@code
|
||||||
|
* grep -rlE '\.setLevel\(' fleetd/src/test/java --include='*.java' | grep -v CapturedLog.java
|
||||||
|
* grep -rl 'CapturedLog' fleetd/src/test/java --include='*.java'
|
||||||
|
* }</pre>
|
||||||
|
*/
|
||||||
|
public final class CapturedLog implements AutoCloseable {
|
||||||
|
private final Logger logger;
|
||||||
|
private final Level originalLevel;
|
||||||
|
private final ListAppender<ILoggingEvent> appender;
|
||||||
|
|
||||||
|
private CapturedLog(Logger logger, Level pinnedLevel) {
|
||||||
|
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||||
|
this.logger = logger;
|
||||||
|
this.originalLevel = logger.getLevel();
|
||||||
|
this.appender = new ListAppender<>();
|
||||||
|
appender.setContext(ctx);
|
||||||
|
appender.start();
|
||||||
|
logger.addAppender(appender);
|
||||||
|
if (pinnedLevel != null) {
|
||||||
|
logger.setLevel(pinnedLevel);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Capture {@code loggerClass}'s output, pinning its level to {@code pinnedLevel} for the
|
||||||
|
* duration of the try-with-resources block. */
|
||||||
|
public static CapturedLog at(Class<?> loggerClass, Level pinnedLevel) {
|
||||||
|
return new CapturedLog((Logger) LoggerFactory.getLogger(loggerClass), pinnedLevel);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Capture {@code loggerClass}'s output without changing its level. */
|
||||||
|
public static CapturedLog of(Class<?> loggerClass) {
|
||||||
|
return new CapturedLog((Logger) LoggerFactory.getLogger(loggerClass), null);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Capture the named logger's output, pinning its level to {@code pinnedLevel}. For a logger
|
||||||
|
* obtained in production code via {@code LoggerFactory.getLogger("some-name")} rather than a
|
||||||
|
* class — e.g. {@code AuditLog}'s {@code "audit"} logger — where {@link #at(Class, Level)}
|
||||||
|
* would capture the wrong {@code Logger} instance (class name and logger name are different
|
||||||
|
* strings and resolve to different cached loggers).
|
||||||
|
*/
|
||||||
|
public static CapturedLog at(String loggerName, Level pinnedLevel) {
|
||||||
|
return new CapturedLog((Logger) LoggerFactory.getLogger(loggerName), pinnedLevel);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Capture the named logger's output without changing its level. See {@link #at(String, Level)}. */
|
||||||
|
public static CapturedLog of(String loggerName) {
|
||||||
|
return new CapturedLog((Logger) LoggerFactory.getLogger(loggerName), null);
|
||||||
|
}
|
||||||
|
|
||||||
|
public List<ILoggingEvent> events() {
|
||||||
|
return appender.list;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Re-pin the level while this capture is still open — for example to lower it further for one
|
||||||
|
* assertion inside a test whose fixture already pinned a coarser baseline in {@code @BeforeEach}.
|
||||||
|
* This does not change what {@link #close} restores: that is always the level captured when
|
||||||
|
* this instance was created, never a value set through this method.
|
||||||
|
*/
|
||||||
|
public void setLevel(Level level) {
|
||||||
|
logger.setLevel(level);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
public void close() {
|
||||||
|
logger.detachAppender(appender);
|
||||||
|
logger.setLevel(originalLevel);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,96 @@
|
|||||||
|
package dev.ltms.fleet.testing;
|
||||||
|
|
||||||
|
import ch.qos.logback.classic.Level;
|
||||||
|
import ch.qos.logback.classic.Logger;
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import org.slf4j.LoggerFactory;
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* fleetd #537: pins {@link CapturedLog#close}'s own contract — the appender detach half, the level
|
||||||
|
* restore half, and that {@link CapturedLog#setLevel} does not change what {@code close} restores.
|
||||||
|
* Before this, only the level-restore half was pinned (by {@code
|
||||||
|
* WorktreeSessionManagerTest.sharedSessionManagerLoggerLevelIsRestoredAfterDirtyWorktreeReleasePinsWarn}).
|
||||||
|
* Measured: deleting {@code logger.detachAppender(appender);} from {@code close()} still left
|
||||||
|
* {@code mvn clean install} green — 1701 tests, 0 failures — before this file existed.
|
||||||
|
*
|
||||||
|
* <p>Every test below uses a logger name no production class uses, and unique per test, so this
|
||||||
|
* file cannot become the next entry in fleetd #525's leak family: {@link CapturedLog}'s own
|
||||||
|
* javadoc re-measure commands (top of that file) would otherwise need to start naming this class.
|
||||||
|
*/
|
||||||
|
class CapturedLogTest {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The appender must be detached on close: an event logged through the raw logger after close
|
||||||
|
* must not land in {@link CapturedLog#events()}. Asserting on the observable list (rather than
|
||||||
|
* {@code logger.iteratorForAppenders()}) is what the ticket asked for, and it is also what a
|
||||||
|
* real leak would actually break — a later test's own {@code ListAppender} silently gaining
|
||||||
|
* events emitted by code under test that has nothing to do with it.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void closeDetachesTheAppenderSoALaterLogIsNotCaptured() {
|
||||||
|
String loggerName = "capturedlog-test-only.appender-detach";
|
||||||
|
Logger rawLogger = (Logger) LoggerFactory.getLogger(loggerName);
|
||||||
|
|
||||||
|
CapturedLog log = CapturedLog.of(loggerName);
|
||||||
|
rawLogger.info("while open");
|
||||||
|
int eventsWhileOpen = log.events().size();
|
||||||
|
assertEquals(1, eventsWhileOpen, "the event logged while open must be captured");
|
||||||
|
|
||||||
|
log.close();
|
||||||
|
rawLogger.info("after close");
|
||||||
|
|
||||||
|
assertEquals(eventsWhileOpen, log.events().size(),
|
||||||
|
"close() must detach the appender: an event logged after close must not be "
|
||||||
|
+ "captured, but the captured list grew from " + eventsWhileOpen + " to "
|
||||||
|
+ log.events().size());
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The helper's own headline contract, pinned in one place independent of any production
|
||||||
|
* class's behaviour: {@code close()} restores the level the logger had before {@link
|
||||||
|
* CapturedLog#at} pinned it.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void closeRestoresTheLevelCapturedAtOpen() {
|
||||||
|
String loggerName = "capturedlog-test-only.level-restore";
|
||||||
|
Logger rawLogger = (Logger) LoggerFactory.getLogger(loggerName);
|
||||||
|
rawLogger.setLevel(Level.DEBUG);
|
||||||
|
|
||||||
|
CapturedLog log = CapturedLog.at(loggerName, Level.ERROR);
|
||||||
|
assertEquals(Level.ERROR, rawLogger.getLevel(), "the pinned level took effect while open");
|
||||||
|
|
||||||
|
log.close();
|
||||||
|
|
||||||
|
assertEquals(Level.DEBUG, rawLogger.getLevel(),
|
||||||
|
"close() must restore the level captured when at() was called (DEBUG), not leave "
|
||||||
|
+ "the pinned level (ERROR) in place");
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* {@link CapturedLog#setLevel}'s javadoc claims that re-pinning the level mid-capture does not
|
||||||
|
* change what {@code close()} restores — that restore always uses the level captured when the
|
||||||
|
* instance was created, never a value set through {@code setLevel}. Nothing checked this
|
||||||
|
* before: open with a pinned WARN, call {@code setLevel(TRACE)}, close, and the result must be
|
||||||
|
* the level from BEFORE {@code at} — neither WARN nor TRACE.
|
||||||
|
*/
|
||||||
|
@Test
|
||||||
|
void setLevelDuringCaptureDoesNotChangeWhatCloseRestores() {
|
||||||
|
String loggerName = "capturedlog-test-only.setlevel-no-effect";
|
||||||
|
Logger rawLogger = (Logger) LoggerFactory.getLogger(loggerName);
|
||||||
|
rawLogger.setLevel(Level.DEBUG);
|
||||||
|
|
||||||
|
CapturedLog log = CapturedLog.at(loggerName, Level.WARN);
|
||||||
|
log.setLevel(Level.TRACE);
|
||||||
|
assertEquals(Level.TRACE, rawLogger.getLevel(), "setLevel took effect immediately");
|
||||||
|
|
||||||
|
log.close();
|
||||||
|
|
||||||
|
assertEquals(Level.DEBUG, rawLogger.getLevel(),
|
||||||
|
"close() must restore the level captured at open (DEBUG) regardless of any "
|
||||||
|
+ "later setLevel() call: it must be neither WARN (the level pinned by "
|
||||||
|
+ "at()) nor TRACE (the level set via setLevel() mid-capture), but got "
|
||||||
|
+ rawLogger.getLevel());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -64,7 +64,106 @@
|
|||||||
# The two outputs side by side are the finding: any name whose hash matches between them is a
|
# The two outputs side by side are the finding: any name whose hash matches between them is a
|
||||||
# credential the member holds in full.
|
# credential the member holds in full.
|
||||||
#
|
#
|
||||||
|
# One parse pass: line 1 = present (true/false/null), line 2 = policy mode (possibly blank),
|
||||||
|
# lines 3-5 = knownCount/allowedCount/blockedCount, remaining lines = the known[] names. A single
|
||||||
|
# pass avoids re-parsing (and re-risking a truthiness bug) five separate times.
|
||||||
|
#
|
||||||
|
# This used to feed the parser straight into `mapfile -t _FIELDS < <(producer)`. That form cannot
|
||||||
|
# see the producer fail: `<` `<(...)` is a process substitution, not a pipeline, so `set -o
|
||||||
|
# pipefail` does not reach inside it, and mapfile's own exit status reports whether the BUILTIN
|
||||||
|
# ran, not whether the command substituted into it succeeded — a failing jq or python3 there still
|
||||||
|
# leaves mapfile at rc=0 with an empty array, read as a parse that genuinely found nothing (fleetd
|
||||||
|
# #500). Capturing the parser's output with command substitution first, and checking ITS exit
|
||||||
|
# status, reports the producer's real failure while the fact still exists — before it is handed to
|
||||||
|
# mapfile at all.
|
||||||
|
#
|
||||||
|
# mapfile then reads from that captured string with `<<<` (a herestring), not `< <(...)`: `<<<`
|
||||||
|
# materialises the whole string in memory first, where `< <(...)` would stream it. That only
|
||||||
|
# matters for a large producer; this one is a short credential-name policy response, so the
|
||||||
|
# tradeoff is irrelevant here — noted because it would not be for every producer.
|
||||||
|
parse_policy_fields() {
|
||||||
|
if command -v jq >/dev/null 2>&1; then
|
||||||
|
_FIELDS_RAW="$(printf '%s' "$POLICY_JSON" | jq -r '
|
||||||
|
(.present | tostring),
|
||||||
|
(.policy // ""),
|
||||||
|
(.knownCount // 0 | tostring),
|
||||||
|
(.allowedCount // 0 | tostring),
|
||||||
|
(.blockedCount // 0 | tostring),
|
||||||
|
(.known[]? // empty)')"
|
||||||
|
_PARSE_STATUS=$?
|
||||||
|
_PARSER_NAME="jq"
|
||||||
|
else
|
||||||
|
_FIELDS_RAW="$(printf '%s' "$POLICY_JSON" | python3 - <<'PY'
|
||||||
|
import json, sys
|
||||||
|
data = json.load(sys.stdin)
|
||||||
|
print(str(data.get("present")))
|
||||||
|
print(data.get("policy") or "")
|
||||||
|
print(data.get("knownCount") if data.get("knownCount") is not None else 0)
|
||||||
|
print(data.get("allowedCount") if data.get("allowedCount") is not None else 0)
|
||||||
|
print(data.get("blockedCount") if data.get("blockedCount") is not None else 0)
|
||||||
|
for n in (data.get("known") or []):
|
||||||
|
print(n)
|
||||||
|
PY
|
||||||
|
)"
|
||||||
|
_PARSE_STATUS=$?
|
||||||
|
_PARSER_NAME="python3"
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [ "$_PARSE_STATUS" -ne 0 ]; then
|
||||||
|
echo "refusing to run: could not parse the policy fetched from $POLICY_URL — $_PARSER_NAME exited" \
|
||||||
|
"non-zero (status $_PARSE_STATUS). That is a parser failure, not a claim about the policy" \
|
||||||
|
"itself; the policy response has not been read." >&2
|
||||||
|
return 4
|
||||||
|
fi
|
||||||
|
|
||||||
|
# A herestring adds a newline, so mapfile would turn an empty parser result into one empty field.
|
||||||
|
# Keep that case separate so the refusal reports what the parser actually returned: zero fields.
|
||||||
|
if [ -z "$_FIELDS_RAW" ]; then
|
||||||
|
_FIELDS=()
|
||||||
|
else
|
||||||
|
mapfile -t _FIELDS <<< "$_FIELDS_RAW"
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Arity check — the CORRECTNESS fix (fleetd #500). A parser that exits 0 can still return fewer
|
||||||
|
# than the 5 fixed fields (present, policy mode, 3 counts) that every fixed-field read in main() expects,
|
||||||
|
# whatever the reason: a producer that printed nothing, malformed JSON that jq/python3 still
|
||||||
|
# accepted, or a schema change upstream. main()'s slice (`_FIELDS[@]:5`) does not fire
|
||||||
|
# `set -u` on an unset OR a short array, and every fixed-field read there used a `:-` default, so
|
||||||
|
# without this check a short `_FIELDS` reaches the "0 known names" guard further down with the
|
||||||
|
# same look as a policy that genuinely has 0 names. Check the count here, at the one point the
|
||||||
|
# fact is still present, before the slice consumes it.
|
||||||
|
if (( ${#_FIELDS[@]} < 5 )); then
|
||||||
|
echo "refusing to run: the policy parser ($_PARSER_NAME) returned ${#_FIELDS[@]} field(s); at" \
|
||||||
|
"least 5 are required (present, policy mode, knownCount, allowedCount, blockedCount). The" \
|
||||||
|
"parse ran but its shape is wrong — this is not a claim about how many names the policy" \
|
||||||
|
"knows." >&2
|
||||||
|
return 5
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
main() {
|
||||||
set -uo pipefail
|
set -uo pipefail
|
||||||
|
# `pipefail` is not what catches the parser failure handled in parse_policy_fields() above (fleetd #500): in
|
||||||
|
# `printf '%s' "$POLICY_JSON" | jq -r '...'`, jq is the LAST element of the pipe, so the pipeline's
|
||||||
|
# own exit status is already jq's status, with or without pipefail. It is kept as insurance for if
|
||||||
|
# a post-processing stage is ever appended after the parser (e.g. `| tail -n +2`) — at that point
|
||||||
|
# the parser would sit upstream and pipefail becomes the only thing that still reports its status.
|
||||||
|
|
||||||
|
# --- refuse on an interpreter that cannot run this script (fleetd #500) -------------------------
|
||||||
|
#
|
||||||
|
# mapfile, used below to parse the policy response, was added in bash 4.0. macOS ships bash 3.2.57
|
||||||
|
# at /bin/bash, which predates it. This script's own `set -uo pipefail` does not catch a missing
|
||||||
|
# mapfile: the builtin just fails with "command not found" on stderr, and every line below that
|
||||||
|
# reads the array it would have filled uses a `:-` default or a slice, neither of which `set -u`
|
||||||
|
# catches on an unset array. Left unguarded, that chain ends in the "0 known names" refusal further
|
||||||
|
# down — a claim about the POLICY, for a failure that is actually about the INTERPRETER. So the
|
||||||
|
# interpreter is checked once, explicitly, before it is asked to do anything mapfile depends on.
|
||||||
|
if (( ${BASH_VERSINFO[0]} < 4 )); then
|
||||||
|
echo "refusing to run: this script uses mapfile, which needs bash 4 or newer. This shell is bash" \
|
||||||
|
"${BASH_VERSION:-<unknown, no \$BASH_VERSION>}. Re-run it under a newer bash, for example:" \
|
||||||
|
"\"\$(command -v bash)\" \"$0\"" "$@" >&2
|
||||||
|
exit 3
|
||||||
|
fi
|
||||||
|
|
||||||
FLEETD_HOST="${FLEETD_HOST:-http://127.0.0.1:8765}"
|
FLEETD_HOST="${FLEETD_HOST:-http://127.0.0.1:8765}"
|
||||||
POLICY_URL="${FLEETD_HOST%/}/member-credentials"
|
POLICY_URL="${FLEETD_HOST%/}/member-credentials"
|
||||||
@@ -121,31 +220,10 @@ EOF
|
|||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# One parse pass: line 1 = present (true/false/null), line 2 = policy mode (possibly blank),
|
# Parse the policy in one pass and refuse on any of the three failure causes. The decision, the
|
||||||
# lines 3-5 = knownCount/allowedCount/blockedCount, remaining lines = the known[] names. A single
|
# three refusals and the reasoning behind each live in parse_policy_fields() above — kept there
|
||||||
# pass avoids re-parsing (and re-risking a truthiness bug) five separate times.
|
# with the code rather than here, so the explanation cannot drift away from what it explains.
|
||||||
if command -v jq >/dev/null 2>&1; then
|
parse_policy_fields || exit $?
|
||||||
mapfile -t _FIELDS < <(printf '%s' "$POLICY_JSON" | jq -r '
|
|
||||||
(.present | tostring),
|
|
||||||
(.policy // ""),
|
|
||||||
(.knownCount // 0 | tostring),
|
|
||||||
(.allowedCount // 0 | tostring),
|
|
||||||
(.blockedCount // 0 | tostring),
|
|
||||||
(.known[]? // empty)')
|
|
||||||
else
|
|
||||||
mapfile -t _FIELDS < <(printf '%s' "$POLICY_JSON" | python3 - <<'PY'
|
|
||||||
import json, sys
|
|
||||||
data = json.load(sys.stdin)
|
|
||||||
print(str(data.get("present")))
|
|
||||||
print(data.get("policy") or "")
|
|
||||||
print(data.get("knownCount") if data.get("knownCount") is not None else 0)
|
|
||||||
print(data.get("allowedCount") if data.get("allowedCount") is not None else 0)
|
|
||||||
print(data.get("blockedCount") if data.get("blockedCount") is not None else 0)
|
|
||||||
for n in (data.get("known") or []):
|
|
||||||
print(n)
|
|
||||||
PY
|
|
||||||
)
|
|
||||||
fi
|
|
||||||
|
|
||||||
PRESENT="${_FIELDS[0]:-null}"
|
PRESENT="${_FIELDS[0]:-null}"
|
||||||
POLICY_MODE="${_FIELDS[1]:-}"
|
POLICY_MODE="${_FIELDS[1]:-}"
|
||||||
@@ -164,19 +242,21 @@ case "$KNOWN_COUNT_REPORTED" in
|
|||||||
;;
|
;;
|
||||||
esac
|
esac
|
||||||
|
|
||||||
# --- guard the denominator explicitly — never proceed on a zero/short count ---------------------
|
# --- guard the denominator explicitly — never proceed on a zero count ---------------------------
|
||||||
#
|
#
|
||||||
# This is the exact trap named in the ticket: an empty (or truncated) NAMES array passes every
|
# This is the exact trap named in the ticket: an empty NAMES array passes every subsequent "is it
|
||||||
# subsequent "is it set" check vacuously and prints a table that LOOKS complete. So this is checked
|
# set" check vacuously and prints a table that LOOKS complete. By this point the interpreter gate,
|
||||||
# before anything else runs, with a message that says why, not just that it failed.
|
# the parser-exit-status check, and the arity check above have already ruled out "the interpreter
|
||||||
|
# couldn't run mapfile", "the parser failed", and "the parser returned the wrong shape" — so a zero
|
||||||
|
# count reaching here really does mean the policy itself reports 0 known names, not a swallowed
|
||||||
|
# failure upstream. That is still checked before anything else runs, with a message that says so.
|
||||||
if [ "${#NAMES[@]}" -eq 0 ] || [ "$KNOWN_COUNT_REPORTED" -eq 0 ]; then
|
if [ "${#NAMES[@]}" -eq 0 ] || [ "$KNOWN_COUNT_REPORTED" -eq 0 ]; then
|
||||||
cat >&2 <<EOF
|
cat >&2 <<EOF
|
||||||
refusing to run: the policy fetched from $POLICY_URL contains 0 known names (present=${PRESENT:-unknown}).
|
refusing to run: the policy fetched from $POLICY_URL contains 0 known names (present=${PRESENT:-unknown}).
|
||||||
|
|
||||||
Either memberCredentials: is absent/empty on the running daemon (nothing is protected — see fleetd's
|
memberCredentials: is absent or empty on the running daemon — nothing is protected (see fleetd's own
|
||||||
own startup warning), or the response could not be parsed. Either way, checking zero names would
|
startup warning). Checking zero names would print a clean-looking table for a policy that protects
|
||||||
print a clean-looking table for a policy that protects nothing, or for a probe that read nothing.
|
nothing. This is refused rather than reported as a pass.
|
||||||
This is refused rather than reported as a pass.
|
|
||||||
EOF
|
EOF
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
@@ -251,3 +331,8 @@ How to read this:
|
|||||||
hardcoded list did. If the daemon's policy changes, the next run of this script reflects it
|
hardcoded list did. If the daemon's policy changes, the next run of this script reflects it
|
||||||
with no edit to this file.
|
with no edit to this file.
|
||||||
EOF
|
EOF
|
||||||
|
}
|
||||||
|
|
||||||
|
if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then
|
||||||
|
main "$@"
|
||||||
|
fi
|
||||||
|
|||||||
+1071
-108
File diff suppressed because it is too large
Load Diff
Executable
+109
@@ -0,0 +1,109 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Self-contained checks for the policy parsing guards in probe-member-credentials.sh.
|
||||||
|
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||||
|
PROBE="$ROOT/scripts/probe-member-credentials.sh"
|
||||||
|
TMP="$(mktemp -d "$ROOT/.probe-member-credentials-test.XXXXXX")"
|
||||||
|
trap 'rm -rf "$TMP"' EXIT
|
||||||
|
|
||||||
|
# The SOURCED guard exposes this pure parser without contacting POLICY_URL.
|
||||||
|
source "$PROBE"
|
||||||
|
|
||||||
|
fail() {
|
||||||
|
printf 'FAIL: %s\n' "$*" >&2
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
assert_equals() {
|
||||||
|
local expected="$1" actual="$2" description="$3"
|
||||||
|
[ "$expected" = "$actual" ] || fail "$description: expected $expected, got $actual"
|
||||||
|
}
|
||||||
|
|
||||||
|
assert_contains() {
|
||||||
|
local needle="$1" text="$2" description="$3"
|
||||||
|
printf '%s' "$text" | grep -qF "$needle" || fail "$description: missing $needle"
|
||||||
|
}
|
||||||
|
|
||||||
|
make_jq() {
|
||||||
|
local body="$1"
|
||||||
|
mkdir -p "$TMP/bin"
|
||||||
|
printf '%s\n' '#!/usr/bin/env bash' "$body" > "$TMP/bin/jq"
|
||||||
|
chmod +x "$TMP/bin/jq"
|
||||||
|
}
|
||||||
|
|
||||||
|
run_parser() {
|
||||||
|
local output rc=0
|
||||||
|
POLICY_JSON="$(< "$TMP/policy.json")"
|
||||||
|
POLICY_URL="fixture://member-credentials"
|
||||||
|
output="$(PATH="$TMP/bin:$PATH" parse_policy_fields 2>&1)" || rc=$?
|
||||||
|
PARSER_OUTPUT="$output"
|
||||||
|
PARSER_RC="$rc"
|
||||||
|
}
|
||||||
|
|
||||||
|
test_bash_older_than_four_refuses() {
|
||||||
|
local output rc=0 version
|
||||||
|
version="$(/bin/bash -c 'printf %s "$BASH_VERSION"')"
|
||||||
|
output="$(/bin/bash "$PROBE" 2>&1)" || rc=$?
|
||||||
|
assert_equals 3 "$rc" "bash 3 refusal status"
|
||||||
|
assert_contains 'This shell is bash' "$output" "bash 3 refusal"
|
||||||
|
assert_contains "$version" "$output" "bash 3 refusal version"
|
||||||
|
}
|
||||||
|
|
||||||
|
test_parser_non_zero_refuses() {
|
||||||
|
make_jq 'exit 17'
|
||||||
|
run_parser
|
||||||
|
assert_equals 4 "$PARSER_RC" "parser failure status"
|
||||||
|
assert_contains 'jq exited non-zero (status 17)' "$PARSER_OUTPUT" "parser failure message"
|
||||||
|
}
|
||||||
|
|
||||||
|
test_short_parser_output_refuses() {
|
||||||
|
make_jq "printf '%s\\n' true enforce 3 2"
|
||||||
|
run_parser
|
||||||
|
assert_equals 5 "$PARSER_RC" "short parser output status"
|
||||||
|
assert_contains 'policy parser (jq) returned 4 field(s)' "$PARSER_OUTPUT" "short parser output count"
|
||||||
|
}
|
||||||
|
|
||||||
|
test_empty_parser_output_reports_zero_fields() {
|
||||||
|
make_jq ':'
|
||||||
|
run_parser
|
||||||
|
assert_equals 5 "$PARSER_RC" "empty parser output status"
|
||||||
|
assert_contains 'policy parser (jq) returned 0 field(s)' "$PARSER_OUTPUT" "empty parser output count"
|
||||||
|
}
|
||||||
|
|
||||||
|
test_well_formed_policy_prints_name_table() {
|
||||||
|
local output rc=0
|
||||||
|
make_jq "cat '$TMP/policy.fields'"
|
||||||
|
# Shell functions cannot be passed in an environment assignment. Run the executable through bash.
|
||||||
|
output="$(BRIDGED_MEMBER=1 FIXTURE="$TMP/policy.json" PROBE="$PROBE" PATH="$TMP/bin:$PATH" bash -c '
|
||||||
|
curl() { cat "$FIXTURE"; }
|
||||||
|
export -f curl
|
||||||
|
exec "$PROBE"
|
||||||
|
' 2>&1)" || rc=$?
|
||||||
|
assert_equals 0 "$rc" "well-formed policy status"
|
||||||
|
assert_contains 'ALPHA_TOKEN' "$output" "name table"
|
||||||
|
assert_contains 'BETA_TOKEN' "$output" "name table"
|
||||||
|
assert_contains 'GAMMA_TOKEN' "$output" "name table"
|
||||||
|
}
|
||||||
|
|
||||||
|
cat > "$TMP/policy.json" <<'JSON'
|
||||||
|
{"present":true,"policy":"enforce","knownCount":3,"allowedCount":2,"blockedCount":1,"known":["ALPHA_TOKEN","BETA_TOKEN","GAMMA_TOKEN"]}
|
||||||
|
JSON
|
||||||
|
cat > "$TMP/policy.fields" <<'FIELDS'
|
||||||
|
true
|
||||||
|
enforce
|
||||||
|
3
|
||||||
|
2
|
||||||
|
1
|
||||||
|
ALPHA_TOKEN
|
||||||
|
BETA_TOKEN
|
||||||
|
GAMMA_TOKEN
|
||||||
|
FIELDS
|
||||||
|
|
||||||
|
test_bash_older_than_four_refuses
|
||||||
|
test_parser_non_zero_refuses
|
||||||
|
test_short_parser_output_refuses
|
||||||
|
test_empty_parser_output_reports_zero_fields
|
||||||
|
test_well_formed_policy_prints_name_table
|
||||||
|
printf 'PASS: probe member credentials guards\n'
|
||||||
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user