Compare commits
281 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 2bce7e37b6 | |||
| 5cabd09705 | |||
| cc47672b7c | |||
| 37edd9134b | |||
| 80f167b1f7 | |||
| bd6547fca3 | |||
| 7e97f5bff5 | |||
| 7d41ccccee | |||
| bf616e192a | |||
| f8bd5d0c51 | |||
| ac40de1d30 | |||
| 7930a31b94 | |||
| 850fb12807 | |||
| b14b66ab03 | |||
| 15ff6bcde5 | |||
| 7822772905 | |||
| 0efe1567c0 | |||
| 837fed7690 | |||
| aa4ee64a34 | |||
| bb750cdba3 | |||
| d56c77b368 | |||
| 83e2ff06cf | |||
| f5deaafd06 | |||
| fe46311266 | |||
| 08968bb1b7 | |||
| 27bbd11f06 | |||
| 48d7841fbf | |||
| cc1df11f69 | |||
| fdfd4ac491 | |||
| d5dd5639ae | |||
| 7a120b3256 | |||
| 863d477966 | |||
| cec48832be | |||
| a36b7ccd7c | |||
| 32bf324a1e | |||
| 16de9df000 | |||
| 65c9deb4d1 | |||
| 28ae27b8e1 | |||
| 81a0cf4710 | |||
| 8a837a2830 | |||
| 0a2b3a4a56 | |||
| 613ece92dc | |||
| 8d4206c2b5 | |||
| 03286a589b | |||
| 88b9503c3b | |||
| c553d795d8 | |||
| 3ba6d6784c | |||
| 78ca24dc3f | |||
| 4fa6553db5 | |||
| 2124e043ce | |||
| e4f3620acb | |||
| 032a59a34d | |||
| e689090024 | |||
| 1cc34888fd | |||
| 0331ecd5d3 | |||
| 831a918c30 | |||
| 3db5277ae8 | |||
| 6939e0cbbc | |||
| f0095bf8b2 | |||
| 5206679efd | |||
| ac044e7573 | |||
| 6ebad2a91f | |||
| 3d10ed385c | |||
| 6d0c94dbdb | |||
| e01563a550 | |||
| 30e3225a3c | |||
| ef186a1516 | |||
| 5d5b3bdc76 | |||
| 2f8c98dac9 | |||
| 2db7189067 | |||
| 94476ac109 | |||
| 5fe02b7c98 | |||
| a1052f4fd1 | |||
| 8c9904a7c4 | |||
| 23af5dfdff | |||
| 3a10f6ad17 | |||
| a3842c873d | |||
| c29c3f063d | |||
| 758d62a396 | |||
| 6725642274 | |||
| a1f4dc365a | |||
| 165b62ee20 | |||
| 72d3481de3 | |||
| 29ccb747ad | |||
| 4749e27453 | |||
| e501d39988 | |||
| 6b6cf25862 | |||
| 180de840eb | |||
| 8fd2d7e5e7 | |||
| 129dd4a838 | |||
| e09cac6f1f | |||
| c00a86b32c | |||
| d3ae0350a2 | |||
| 2c2196a1f1 | |||
| 35ade14630 | |||
| 541df87272 | |||
| 337b6ccd6e | |||
| 42f46dfe9a | |||
| 9088d2b2c5 | |||
| 500bfa2c33 | |||
| 9ca9c43dfa | |||
| b525b0f08f | |||
| 1966c69994 | |||
| 976eff8ad1 | |||
| 0af902ec43 | |||
| 9118ce2537 | |||
| 2f48e08f1f | |||
| fec284e7cb | |||
| 8e2e4c5e73 | |||
| 0edc6615fc | |||
| b745e159de | |||
| aac29d604c | |||
| c884802b13 | |||
| 5f5573a24e | |||
| 74b0087ebb | |||
| 927e0151d4 | |||
| 5275922d1d | |||
| e186c7945a | |||
| 33a6e77f0e | |||
| 4e47489d53 | |||
| c76b2f149e | |||
| 826fffe05b | |||
| 75f57cdba7 | |||
| f556af5d4e | |||
| d5f33f0c6e | |||
| 16e17b32ad | |||
| 988494e18b | |||
| c4549a5e20 | |||
| 46fa4f38d5 | |||
| 20e0e68ad7 | |||
| 17468a234a | |||
| 8a53d5bfc6 | |||
| 695da7418e | |||
| abd26c796b | |||
| 24559d81ac | |||
| 6c1c2c3994 | |||
| 01fab15713 | |||
| 0cd00e71c3 | |||
| bf0ff2adbf | |||
| ed4bbc1c56 | |||
| c456402cc5 | |||
| 4f0bf667b1 | |||
| 61af9aa574 | |||
| 3b59b34e76 | |||
| 65b38997f7 | |||
| 7510f7649c | |||
| 619792a81c | |||
| cea1183f75 | |||
| e2af4c5ae4 | |||
| dfd5f82894 | |||
| e81944cef6 | |||
| 7bdd39ab9a | |||
| f4b38f040e | |||
| 799668e129 | |||
| 890190263e | |||
| f8522edacd | |||
| b958747855 | |||
| 078bde2c02 | |||
| 0b10ea987b | |||
| 1553d38182 | |||
| d04b075996 | |||
| 644927636d | |||
| 95e45007aa | |||
| 7a583c4045 | |||
| 65bce058bb | |||
| 48b437083b | |||
| fa0612859b | |||
| 4ffbcd0b7d | |||
| a0cd053fd9 | |||
| f9a5e066b5 | |||
| e9bc192160 | |||
| e0a57988ad | |||
| b414a74c26 | |||
| 2c3796d598 | |||
| 7e0ff9ab06 | |||
| 6d1565d8b6 | |||
| e18e002d2f | |||
| c18572ea9d | |||
| e9d02bc5e8 | |||
| a2108a8a14 | |||
| 57f8fa257a | |||
| 61944fc045 | |||
| 4b48d2d921 | |||
| 3aa145cbef | |||
| 4875127daa | |||
| d14a624421 | |||
| 246f50b778 | |||
| 15b53c6cfa | |||
| 3a7ef0adbd | |||
| 293a305748 | |||
| e5038c6d13 | |||
| 13ea79f6fd | |||
| f55d3c203b | |||
| 37c4d47d3a | |||
| 04e21c9243 | |||
| 83cac07f6e | |||
| f472e0f782 | |||
| 91c9f981c5 | |||
| dd906526c0 | |||
| ec3001796a | |||
| 86cf4c285a | |||
| cd18887b69 | |||
| 4637c68295 | |||
| 0114bd1fa7 | |||
| 6dee84ca71 | |||
| f004a0c654 | |||
| 6123576c68 | |||
| 21cfc09f8e | |||
| 509530e235 | |||
| d146a01422 | |||
| 91332723db | |||
| 244fbd98a5 | |||
| 5afe8e14d9 | |||
| 73f6b12dd8 | |||
| 793f2e7157 | |||
| cf48983046 | |||
| d038516750 | |||
| f8182e4514 | |||
| bc13b8e92c | |||
| fbe79258bf | |||
| ef1e014b41 | |||
| e3c8393d1b | |||
| 379e03f9d0 | |||
| e13921aa8a | |||
| 8a549d8610 | |||
| 84081b2bd8 | |||
| defe3365c4 | |||
| 4e6201ecd1 | |||
| ccf50f950e | |||
| a7f0211e2f | |||
| 049ce4828c | |||
| c1173346ef | |||
| 7ace184fe6 | |||
| 54b314ace5 | |||
| 224b344445 | |||
| 0b28b4cb0f | |||
| 83129e165c | |||
| 871b595954 | |||
| b67b1585c2 | |||
| 979b2b5632 | |||
| cf4ad186ab | |||
| 5100f215cf | |||
| f129e9b7cd | |||
| 4aed45de19 | |||
| 1e6daa5c73 | |||
| 4c015d76b7 | |||
| 37a11cd168 | |||
| 22ad24db6c | |||
| e94c1b8841 | |||
| cc0ec65714 | |||
| d67d30c58a | |||
| d75ee1cca5 | |||
| 6804676a96 | |||
| 3e5d742ac7 | |||
| 6da2a71050 | |||
| e32ac39faf | |||
| daa243d37a | |||
| 2773ab600d | |||
| 19cdf8dc9f | |||
| 9daf1ec5ba | |||
| c9f0ca9359 | |||
| 84c8a2d2f0 | |||
| ded226abfe | |||
| 9b8d55bc18 | |||
| 11f8709286 | |||
| b034f105c0 | |||
| 6e37722383 | |||
| ffce30afa2 | |||
| e724a59f2d | |||
| 4bf855d225 | |||
| d4c9704007 | |||
| 7c252b5f5f | |||
| f756933879 | |||
| a1aecbf4fc | |||
| 131e7b1ccd | |||
| d0ac6c435f | |||
| 2bc5f3a057 | |||
| ba6b4a5da9 | |||
| da5a987df0 | |||
| 7dd6c46156 | |||
| 3a5cdc5108 |
@@ -0,0 +1,18 @@
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the bridged MCP gateway.",
|
||||
"owner": {
|
||||
"name": "LTMS"
|
||||
},
|
||||
"plugins": [
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"source": "./plugin",
|
||||
"description": "Make a project bridge-ready: mount the bridged MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.1.0",
|
||||
"author": {
|
||||
"name": "LTMS"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
---
|
||||
name: architect
|
||||
description: Refine work into clear, independent units before implementation.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from bridged.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You are an architect in this fleet. You refine work before anyone builds it: scope,
|
||||
acceptance criteria, risks, and a unit split. You read the repo and write analysis.
|
||||
You never commit production code and never open a pull request.
|
||||
|
||||
A design task is worked by two architects. Design alone first, then exchange and
|
||||
say plainly where you disagree. Do not concede just to agree.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `bridge_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Report only work you actually did and the real output of checks you ran. Do not
|
||||
claim a result from a tool you could not use. The primary's IDE tools are not yours.
|
||||
A mounted forge tool may use a blocked credential and fail by design.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,29 @@
|
||||
---
|
||||
name: dev
|
||||
description: Implement one assigned unit, test it, and open a pull request.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from bridged.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You implement the one unit you were given and nothing else. Work in your assigned
|
||||
git worktree and branch. Never check out, rebase onto, or push to `main`. Confirm
|
||||
the worktree root and branch before you edit. Use only paths under that root.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `bridge_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Implement the change and run the full required build in your worktree. Read the
|
||||
complete output and report its real result. Do not hide failures with a pipe. State
|
||||
only checks you actually ran. The primary's IDE tools are not yours. A mounted forge
|
||||
tool may use a blocked credential and fail by design.
|
||||
|
||||
Stage only files you changed. Never use `git add -A` or `git add .`. Never commit
|
||||
`.mcp.json` or `wiki/`. Commit with a clear message, push your branch, and open your
|
||||
own pull request against `main`. Never merge.
|
||||
|
||||
Your handoff must name the pull request or why it was not created, the branch, the
|
||||
files changed, the build result, and any caveat for review.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,34 @@
|
||||
---
|
||||
name: reviewer
|
||||
description: Review one assigned scope and report the most important real issue.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from bridged.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You review the diff you were given. Report bugs, risks, and missing tests. You do
|
||||
not change code.
|
||||
|
||||
Read the whole assigned scope before judging it. Review only that scope. If you see
|
||||
something outside it, note it in one line and do not investigate it further. Do not
|
||||
run the build. The owner makes changes and runs checks.
|
||||
|
||||
Use `bridge_ask{question}` only when a decision belongs to the lead, such as an
|
||||
unclear requirement or two defensible fixes. Do not ask about something you can
|
||||
decide by reading more code.
|
||||
|
||||
Report the single most important real issue in this form:
|
||||
|
||||
```
|
||||
1. <path>:<line>
|
||||
2. issue: <one sentence: what is wrong and why it matters>
|
||||
3. fix: <one line: the concrete change>
|
||||
4. severity: high | medium | low
|
||||
```
|
||||
|
||||
If there is no real issue, report `NO ISSUE` and one line saying why. A clean review
|
||||
is valid. Do not invent an issue. Use high for a wrong result, data loss, security,
|
||||
or a hang or crash on a real path. Use medium for an edge-path bug or a correctness
|
||||
risk under load or concurrency. Use low for clarity, a latent foot-gun, or a smell
|
||||
with no current failure.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -1,24 +1,23 @@
|
||||
---
|
||||
name: implementer
|
||||
description: Implementer-role playbook for a bridged worker — you are in an isolated git worktree on a dedicated branch; implement the assigned task, commit, push, open your own PR to main, and hand off the PR URL via bridge_reply. You never merge. Load this when you have been delegated an implementation task over bridged.
|
||||
description: Implementer-role procedure for a bridged worker — verify your worktree, implement the scope, commit, push, open your own PR, and hand off the PR URL. Load this when the lead delegates you an implementation task over bridged.
|
||||
---
|
||||
|
||||
# Implementer worker
|
||||
# Implementer worker — procedure
|
||||
|
||||
You are an **implementer** in the claude-bridge fleet. The lead delegated you one scoped task,
|
||||
and you are running in an **isolated git worktree on your own branch** — a full peer of the
|
||||
primary (same `CLAUDE.md`, skills, memory, MCP), differing only in the model behind you and the
|
||||
branch you sit on. Your job for this turn: **implement the task, then hand off a PR the lead can
|
||||
review and merge.** You do the work; the lead (or human) is the merge gate — you never merge.
|
||||
The turn contract (one `bridge_reply`, `bridge_ask` for the lead's decisions, honest reporting,
|
||||
never merge, never commit `.mcp.json` or `wiki/`) is in **`CLAUDE.md` → Bridge communication →
|
||||
Worker** and already applies. This skill is only the *implement-and-hand-off procedure*.
|
||||
|
||||
Delivery mechanics (how the task reached you, how your reply resolves the lead's blocked send)
|
||||
are in [`docs/MCP-Contract.md`](../../../docs/MCP-Contract.md); the worktree/PR model is in
|
||||
[`docs/Worker-Git-Workflow.md`](../../../docs/Worker-Git-Workflow.md). You only need the steps
|
||||
below.
|
||||
You run in an **isolated git worktree on your own branch** — a full peer of the primary (same repo,
|
||||
`CLAUDE.md`, skills), differing in the model behind you and the branch you sit on. Your MCP surface
|
||||
is **only what your launcher mounted** (the bridge): the primary's IDE and forge servers are not
|
||||
yours, and the worktree's `.mcp.json` is deliberately emptied so you cannot inherit them.
|
||||
The worktree model is documented in [`docs/Worker-Git-Workflow.md`](../../../docs/Worker-Git-Workflow.md).
|
||||
|
||||
## 1. Confirm where you are — a worktree on a dedicated branch
|
||||
## 1. Confirm where you are — then never leave
|
||||
|
||||
Before touching anything, verify your ground truth:
|
||||
Before touching anything:
|
||||
|
||||
```bash
|
||||
git rev-parse --show-toplevel # your worktree root — NOT the primary's main tree
|
||||
@@ -26,46 +25,64 @@ git branch --show-current # your dedicated branch: worker/<ticket>-<nonce>
|
||||
git status # should be clean at the start
|
||||
```
|
||||
|
||||
Do **all** work here, on this branch. **Never** switch to `main`, never `git checkout main`,
|
||||
never rebase onto or push to `main` directly. The branch is your isolation — respect it.
|
||||
Do **all** work here, on this branch. Never `git checkout main`, never rebase onto or push to
|
||||
`main`. The branch is your isolation — respect it.
|
||||
|
||||
## 2. Implement the task
|
||||
|
||||
- Implement exactly the scope the lead named. Keep changes focused; if you notice something out
|
||||
of scope, note it in your reply rather than expanding the diff.
|
||||
- Match the surrounding code's style, naming, and idioms. Follow project `CLAUDE.md`.
|
||||
- **You cannot run the IDE MCP tools** (intellij-index / jetbrains are the primary's, not yours).
|
||||
So **never claim a file is "IDE-clean" or "diagnostics-clean"** — you cannot verify that. State
|
||||
only what you actually ran (e.g. `mvn`, a test) and its real output. A fabricated clean claim is
|
||||
worse than an honest "I could not verify inspections here."
|
||||
- Run whatever build/test you can and **report the true result** — including failures.
|
||||
|
||||
## 3. Commit — focused, and never the excluded files
|
||||
**Every path you read, edit, or build is relative to that root.** Work from `$PWD`; if a tool, a
|
||||
brief, or your own memory hands you an absolute path, check it starts with your worktree root
|
||||
before you touch it, and stop if it doesn't. An absolute path pointing anywhere else is the
|
||||
primary's checkout — editing there while building here means **every build you run is of code that
|
||||
does not contain your changes**, and it passes while your work goes nowhere. This has happened:
|
||||
a worker made all 59 of its edits in the primary's tree and never noticed.
|
||||
|
||||
```bash
|
||||
git add <the files you changed>
|
||||
test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-toplevel)"
|
||||
```
|
||||
|
||||
## 2. Implement
|
||||
|
||||
- Implement exactly the scope the lead named. Keep the diff focused; note anything out of scope
|
||||
in your reply instead of widening it.
|
||||
- Match the surrounding code's style, naming, and idioms.
|
||||
|
||||
**Acceptance criterion — a green build, quoted.** Your work is not done until this passes *inside
|
||||
your worktree*:
|
||||
|
||||
```bash
|
||||
cd "$(git rev-parse --show-toplevel)/bridged" && mvn clean install
|
||||
echo "exit=$?"
|
||||
```
|
||||
|
||||
Read its **full** output — never pipe it through `tail`/`head`/`grep`, which hide a failure behind
|
||||
a zero exit. Then quote the real `Tests run: … Failures: … Errors: …` line and the
|
||||
`BUILD SUCCESS`/`FAILURE` verbatim in your reply. If it does not go green, say so with the actual
|
||||
error; a failing build honestly reported is a usable result, a claimed-green one is not. You have
|
||||
no IDE MCP tools, so `mvn` is your only verification — never claim a check you had no way to run.
|
||||
|
||||
## 3. Commit
|
||||
|
||||
```bash
|
||||
git add <the files you changed> # explicitly — never `git add -A` / `git add .`
|
||||
git commit -m "<ticket>: <clear one-line summary>"
|
||||
```
|
||||
|
||||
**Excluded from every commit, always:** `.mcp.json` (the primary's local, session-modified copy —
|
||||
present only for parity) and `wiki/` (a separate submodule). Stage files explicitly; do **not**
|
||||
`git add -A` / `git add .` blindly, or you risk staging them. If `.mcp.json` shows as modified,
|
||||
leave it — it is flagged `--skip-worktree` and is not yours to commit.
|
||||
`.mcp.json` is neutralized and `--skip-worktree` in your worktree — never `git add` it, and never
|
||||
"restore" it from the primary's copy. Same for `wiki/` (a submodule with its own remote).
|
||||
|
||||
## 4. Push your branch
|
||||
## 4. Push
|
||||
|
||||
```bash
|
||||
git push -u origin HEAD
|
||||
```
|
||||
|
||||
Push is over SSH as the same user — no extra credential needed. Push the branch as-is; do not
|
||||
force-push over anything you did not create.
|
||||
Push is over SSH as the same user — no extra credential needed. Never force-push over anything
|
||||
you did not create.
|
||||
|
||||
## 5. Open your own PR to `main`
|
||||
|
||||
Open the PR via the gitea REST API. The daemon injected a **repo-scoped token** (`GITEA_TOKEN`)
|
||||
and the forge host (`GITEA_HOST`) into your env for exactly this — the token can create a PR but
|
||||
**cannot merge** (that stays the lead/human gate).
|
||||
Via the gitea REST API. The daemon injected a **repo-scoped token** (`GITEA_TOKEN`) and the forge
|
||||
host (`GITEA_HOST`) into your env for exactly this — the token can create a PR but **cannot
|
||||
merge**.
|
||||
|
||||
```bash
|
||||
API="${GITEA_HOST%/}/api/v1/repos/lms/claude-bridge/pulls"
|
||||
@@ -81,46 +98,40 @@ JSON
|
||||
)"
|
||||
```
|
||||
|
||||
The response JSON includes `"html_url"` — that is your PR URL. If the call fails (non-2xx), read
|
||||
the error body, fix the cause if it is yours (e.g. branch not pushed yet), and report the failure
|
||||
honestly in your reply rather than inventing a URL. If `GITEA_TOKEN` is unset, your profile was
|
||||
not granted PR-create — push the branch (step 4) and report the branch name so the lead opens the
|
||||
PR.
|
||||
The response JSON carries `"html_url"` — that is your PR URL. On a non-2xx, read the error body,
|
||||
fix it if the cause is yours (e.g. branch not pushed yet), and report the failure rather than
|
||||
inventing a URL. If `GITEA_TOKEN` is unset your profile was not granted PR-create: push the branch
|
||||
and report its name so the lead opens the PR.
|
||||
|
||||
## 6. Reply via `bridge_reply` — the PR is the handoff
|
||||
## 6. Hand off — what goes in `bridge_reply`
|
||||
|
||||
End your turn with **exactly one** `bridge_reply`. That reply is the entire handoff — the lead
|
||||
cannot see your terminal. Include:
|
||||
The reply is the entire handoff; the lead cannot see your terminal.
|
||||
|
||||
```
|
||||
PR: <html_url from step 5, or "not created: <reason>" + branch name>
|
||||
branch: <your branch>
|
||||
files: <the files you changed>
|
||||
tests: <what you ran and its REAL result — or "not run: <why>">
|
||||
root: <git rev-parse --show-toplevel — proves you worked in your own worktree>
|
||||
files: <worktree-relative paths you changed>
|
||||
build: <the verbatim "Tests run: …" and BUILD SUCCESS/FAILURE lines — or "not run: <why>">
|
||||
summary: <2-3 lines: what you implemented and any caveat the reviewer needs>
|
||||
```
|
||||
|
||||
Then stop. **Do not merge. Do not touch `.mcp.json` or `wiki/`.** One reply closes the turn.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
autonumber
|
||||
participant L as Lead
|
||||
participant B as bridged
|
||||
participant I as Implementer (you)
|
||||
participant G as git / gitea
|
||||
|
||||
L->>B: bridge_send(task) — blocks
|
||||
B-->>I: your assignment (in a worktree on your branch)
|
||||
I->>I: implement + build/test here
|
||||
L->>I: delegated task (you are in a worktree on your branch)
|
||||
I->>I: "implement here — every path under $PWD"
|
||||
I->>I: "mvn clean install in this worktree, unpiped, until green"
|
||||
I->>G: git commit (never .mcp.json / wiki)
|
||||
I->>G: git push -u origin HEAD
|
||||
I->>G: POST /pulls (GITEA_TOKEN) — open PR to main
|
||||
G-->>I: html_url
|
||||
I->>B: bridge_reply(PR url, branch, files, tests)
|
||||
B-->>L: { outcome:"reply", text }
|
||||
Note over L,G: lead reviews the PR, merges on green — you never merge
|
||||
I->>L: bridge_reply(PR url, branch, files, tests)
|
||||
Note over L,G: the lead reviews the PR and merges on green — you never merge
|
||||
```
|
||||
|
||||
*The implement turn: work in the worktree, commit → push → open the PR, hand off the URL. The
|
||||
lead is the merge gate.*
|
||||
*The implement turn: work in the worktree, commit → push → open the PR, hand off the URL.*
|
||||
|
||||
@@ -0,0 +1,180 @@
|
||||
---
|
||||
name: port-to-opencode
|
||||
description: Make an OpenCode session a first-class participant in a Claude Code workspace — instructions, MCP servers, and secrets — without duplicating config. Use when onboarding opencode to a project that already has CLAUDE.md and .mcp.json, or when an opencode peer needs the same tools and rules as the Claude session.
|
||||
---
|
||||
|
||||
# Porting a Claude Code workspace to OpenCode
|
||||
|
||||
**The headline: there is almost nothing to port.** OpenCode reads `CLAUDE.md` natively. The only
|
||||
artifact you create is one `opencode.json` mapping MCP servers. Do not translate instructions, do
|
||||
not generate a second rules file, and do not install a sync tool — every one of those makes the
|
||||
workspace worse.
|
||||
|
||||
Everything below was verified against `opencode 1.18.16` and the OpenCode docs.
|
||||
|
||||
## 1. Know what you get for free
|
||||
|
||||
OpenCode's instruction search order:
|
||||
|
||||
```
|
||||
1. walking up from cwd: AGENTS.md , then CLAUDE.md
|
||||
2. global: ~/.config/opencode/AGENTS.md
|
||||
3. Claude Code global: ~/.claude/CLAUDE.md (unless disabled)
|
||||
```
|
||||
|
||||
*"The first matching file wins in each category."*
|
||||
|
||||
Consequences that decide the whole procedure:
|
||||
|
||||
- **A project `CLAUDE.md` is already read.** No port needed.
|
||||
- **Your user-level `~/.claude/CLAUDE.md` is already read too.** Global preferences carry over.
|
||||
- **An `AGENTS.md` in the repo SHADOWS `CLAUDE.md`.** If one exists from a previous Codex port,
|
||||
**delete it** — otherwise opencode reads the stale translated copy instead of the real rules.
|
||||
This is the single most likely way to get this wrong.
|
||||
|
||||
## 2. Create `opencode.json` for MCP servers only
|
||||
|
||||
Project config lives at `opencode.json` in the repo root; the global one is
|
||||
`~/.config/opencode/opencode.json`. **Configs merge, they do not replace** — so machine-local
|
||||
servers belong in the global file and shared ones in the project file.
|
||||
|
||||
Map each entry from `.mcp.json`:
|
||||
|
||||
| `.mcp.json` | `opencode.json` |
|
||||
|---|---|
|
||||
| `"type": "http"` / `"sse"` | `"type": "remote"`, `"url"` |
|
||||
| `"type": "stdio"` | `"type": "local"`, `"command": ["bin", "arg"]` |
|
||||
| `"command"` + `"args"` | single `"command"` array |
|
||||
| `"env"` | `"environment"` |
|
||||
| `"headers"` | `"headers"` |
|
||||
|
||||
```json
|
||||
{
|
||||
"$schema": "https://opencode.ai/config.json",
|
||||
"instructions": ["CLAUDE.md"],
|
||||
"mcp": {
|
||||
"bridged": { "type": "remote", "url": "http://127.0.0.1:8765/mcp", "enabled": true },
|
||||
"context7": { "type": "remote", "url": "https://example.dev/mcp", "enabled": true,
|
||||
"headers": { "Authorization": "Bearer {env:CONTEXT7_TOKEN}" } },
|
||||
"gitea": { "type": "local", "command": ["gitea-mcp", "-t", "stdio"], "enabled": true,
|
||||
"environment": { "GITEA_ACCESS_TOKEN": "{env:GITEA_ACCESS_TOKEN}" } }
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Set `instructions` explicitly even though `CLAUDE.md` is found anyway — the fallback only applies
|
||||
while no `AGENTS.md` exists, and being explicit survives someone adding one later.
|
||||
|
||||
## 3. Reference secrets, never embed them
|
||||
|
||||
OpenCode substitutes at load time, in both `headers` and `environment`:
|
||||
|
||||
```
|
||||
{env:VARIABLE_NAME} value from the environment
|
||||
{file:~/.secrets/token} value read from a file
|
||||
```
|
||||
|
||||
**No credential ever belongs in `opencode.json`.** With `{env:…}` there is no reason to write one,
|
||||
which is what makes this file safe to commit — and it must be committable, because a peer running
|
||||
in a git worktree receives tracked files only.
|
||||
|
||||
**But `{env:…}` reads OpenCode's *process* environment — and OpenCode has no env store of its
|
||||
own.** A Claude Code `env` block in `~/.claude/settings.json` or `.claude/settings.local.json` does
|
||||
**not** reach it: those files are Claude Code's, and the variables exist only inside processes
|
||||
Claude Code spawned. Verify with a clean login shell, not the shell your agent hands you:
|
||||
|
||||
```bash
|
||||
env -u MY_TOKEN zsh -lc 'echo "${MY_TOKEN:-NOT IN PROFILE}"'
|
||||
```
|
||||
|
||||
So `{env:…}` only works if something puts the variable in the environment first. Pick the supply
|
||||
route by who launches opencode:
|
||||
|
||||
| Launcher | Route |
|
||||
|---|---|
|
||||
| a human, from a terminal | one central store, sourced by the login shell |
|
||||
| a spawner (bridge, CI, IDE) | `{env:…}`, with the spawner injecting the variable |
|
||||
|
||||
**For the human case, keep every credential in one file the login shell sources.** Here that file
|
||||
is `${SHARED_ENV}/tools/secrets.sh`, sourced from `${SHARED_ENV}/.ltms`, kept at mode 600 and never
|
||||
committed. `opencode.json` then names variables and holds no values:
|
||||
|
||||
```json
|
||||
"headers": { "Authorization": "Bearer {env:CONTEXT7_TOKEN}" }
|
||||
```
|
||||
|
||||
Both routes end at the same syntax, and that is the point. The file does not change when a human
|
||||
launches opencode instead of the bridge.
|
||||
|
||||
**`{file:…}` also works, and this project moved away from it.** Opencode resolves a relative
|
||||
`{file:}` path against the project root, so a gitignored `.secrets/` beside `opencode.json` needs no
|
||||
shell setup at all. It did not fail; the problem is that it makes a second copy of the token. The
|
||||
same secret then lives in two places, and the copy you forget is the one that leaks or goes stale.
|
||||
One store with many references is easier to rotate and to audit.
|
||||
|
||||
**One catch survives either choice, so state it out loud:** what a human's shell exports does not
|
||||
reach a spawned peer, and neither does a gitignored `.secrets/` — a git worktree receives tracked
|
||||
files only. Spawned peers must be fed through `{env:…}` by whatever launches them. Check the
|
||||
variable *names* match: a spawner often injects under a different name than your shell uses, and
|
||||
the config has no fallback. In this repo the bridge goes further: it neutralizes a worktree's
|
||||
`opencode.json`, so a member cannot inherit the primary's credentials by accident.
|
||||
|
||||
## 4. Do not port machine-local MCP servers
|
||||
|
||||
IDE indexes, language servers, editor bridges — anything bound to *your* checkout — stay out of the
|
||||
project file. Put them in `~/.config/opencode/opencode.json` if you want them personally.
|
||||
|
||||
A committed project config reaches every worktree. A worker that mounts servers whose paths point
|
||||
into the primary's checkout will edit the primary's files while building in its own — every build
|
||||
passes, every change lands in the wrong tree.
|
||||
|
||||
Port the servers the work needs. Leave the rest.
|
||||
|
||||
## 5. Verify against the running agent, not the file
|
||||
|
||||
A file on disk proves nothing about what the agent loaded.
|
||||
|
||||
```bash
|
||||
opencode run "In one line: state a rule from this project's instructions."
|
||||
```
|
||||
|
||||
The answer must reflect the actual `CLAUDE.md`. If it answers generically, the instructions did not
|
||||
reach the model and everything after this is built on sand.
|
||||
|
||||
Then confirm the tools are mounted:
|
||||
|
||||
```bash
|
||||
opencode mcp list
|
||||
```
|
||||
|
||||
**"connected" does not mean "working".** A stdio server with a missing credential still completes
|
||||
the MCP handshake and reports green; only a real tool call reveals it. Verified: `gitea` showed
|
||||
`✓ connected` with no token, then failed the first call with `token is required`. A remote server
|
||||
is more honest (`⚠ needs authentication`), but do not rely on that difference — **exercise one
|
||||
authenticated tool per server**:
|
||||
|
||||
```bash
|
||||
opencode run "Call <server>'s <tool>. Report the result or the exact error. One line."
|
||||
```
|
||||
|
||||
Check too that no server you deliberately withheld is present.
|
||||
|
||||
## 6. Report
|
||||
|
||||
State what you changed, which servers crossed and which you withheld and why, and quote the
|
||||
verification answer verbatim. If any server failed to connect, say so plainly — a partially mounted
|
||||
peer is worse than a missing one, because it looks configured.
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **A leftover `AGENTS.md` silently wins over `CLAUDE.md`.** Check for one before anything else.
|
||||
- **`opencode.json` is merged, not overridden** — a global entry and a project entry with the same
|
||||
server name both matter; keep names distinct unless you intend to layer them.
|
||||
- **`enabled: false`** turns a server off without deleting its config — prefer it over removal when
|
||||
you may want the server back.
|
||||
- **OAuth-based servers** store tokens in `~/.local/share/opencode/mcp-auth.json` after
|
||||
`opencode mcp auth <server>`; that is machine state, never config to commit.
|
||||
- **Skills and subagents do not port.** OpenCode uses its own agent markdown under
|
||||
`.opencode/agents/`; `.claude/skills/**` is not read. If a delegation brief tells a peer to load a
|
||||
skill by name, that instruction has no effect on an opencode peer — spell the procedure out in the
|
||||
brief, or author the equivalent agent file.
|
||||
@@ -1,69 +1,39 @@
|
||||
---
|
||||
name: reviewer
|
||||
description: Reviewer-role playbook for a bridged worker — read the assigned scope, find the real issues, ask the lead via bridge_ask when a decision is genuinely theirs, and report the finding via bridge_reply. Load this when you have been delegated a code review over bridged.
|
||||
description: Reviewer-role procedure for a bridged worker — how to work a review scope and the exact shape of the finding to report. Load this when the lead delegates you a code review over bridged.
|
||||
---
|
||||
|
||||
# Reviewer worker
|
||||
# Reviewer worker — procedure
|
||||
|
||||
You are a **reviewer** in the claude-bridge fleet. The lead delegated you one scoped review
|
||||
over `bridged`, and your whole job is **this single turn**: examine the scope it named, and
|
||||
report back. You are not the owner of the code and you do not merge anything — you surface
|
||||
what the owner needs to know, then hand the turn back.
|
||||
|
||||
Delivery mechanics (how the task reached you, how your reply resolves the lead's blocked
|
||||
send) are in [`docs/MCP-Contract.md`](../../../docs/MCP-Contract.md); you only need the three
|
||||
rules below.
|
||||
The turn contract (one `bridge_reply`, `bridge_ask` for the lead's decisions, honest reporting,
|
||||
never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and already applies. This
|
||||
skill is only the *review procedure*: how to work the scope, and the exact shape of what you
|
||||
send back.
|
||||
|
||||
## 1. Read the whole scope before you judge
|
||||
|
||||
The delegation names your scope — a file, a diff, a PR, a function. **Read all of it first.**
|
||||
A review that fires on a snippet misses the caller that makes it safe (or the one that makes
|
||||
it a bug). Reviewing only part of the scope and guessing the rest is the most common way a
|
||||
reviewer worker is wrong.
|
||||
A review that fires on a snippet misses the caller that makes it safe (or the one that makes it
|
||||
a bug). Reviewing part of the scope and guessing the rest is the most common way a reviewer is
|
||||
wrong.
|
||||
|
||||
## 2. Stay in your lane
|
||||
## 2. Stay in the scope
|
||||
|
||||
- Review **only** the assigned scope. If you notice something elsewhere, mention it in one
|
||||
line — do **not** go hunt it. Wandering is how two workers end up reporting the same thing
|
||||
and neither covers what it was given.
|
||||
- Do **not** edit files, run the build, or spawn other workers. You review; the owner acts.
|
||||
- You never set `ANTHROPIC_BASE_URL` and never touch herdr — you are a Claude Code process,
|
||||
not part of the transport.
|
||||
- Review **only** what you were assigned. Something elsewhere looks wrong? One line in your
|
||||
reply — do not go hunt it. Wandering is how two reviewers report the same thing and neither
|
||||
covers what it was given.
|
||||
- Do **not** edit files or run the build. You review; the owner acts.
|
||||
|
||||
## 3. When the decision is the lead's — ask, don't guess
|
||||
## 3. Reach for `bridge_ask` only for a genuine fork
|
||||
|
||||
Some things you cannot resolve from the code: an ambiguous requirement, a missing acceptance
|
||||
criterion, "is this behavior intended or a bug?", or a choice between two defensible fixes.
|
||||
Guessing there produces a confident-but-wrong finding. Instead **pause and ask the lead** with
|
||||
`bridge_ask` — a single crisp question. The call blocks; when the lead answers you **resume
|
||||
the same turn** with the answer and finish. Ask only when the answer changes your finding;
|
||||
don't narrate options you could decide yourself.
|
||||
Ambiguous requirement, a missing acceptance criterion, "intended or a bug?", or two defensible
|
||||
fixes with different consequences — those are the lead's call, and guessing produces a
|
||||
confident-but-wrong finding. Anything you could settle by reading more code is yours to settle.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant L as Lead
|
||||
participant B as bridged
|
||||
participant R as Reviewer (you)
|
||||
## 4. The finding — what goes in `bridge_reply`
|
||||
|
||||
L->>B: bridge_send(review scope) — blocks
|
||||
B-->>R: your assignment
|
||||
R->>R: read the full scope
|
||||
opt a decision only the lead can make
|
||||
R->>B: bridge_ask("intended, or a bug?") — you block
|
||||
B-->>L: { outcome:"question", turn_id }
|
||||
L->>B: bridge_send(answer, turn_id)
|
||||
B-->>R: { answer } — you resume the SAME turn
|
||||
end
|
||||
R->>B: bridge_reply(structured finding) — ends your turn
|
||||
B-->>L: { outcome:"reply", text }
|
||||
```
|
||||
|
||||
*The review turn, with the optional `bridge_ask` detour when the call is the lead's to make.*
|
||||
|
||||
## 4. Report with `bridge_reply` — one structured finding
|
||||
|
||||
End your turn with **exactly one** `bridge_reply`. Report the **single most important** real
|
||||
issue in the scope, in these four lines, under ~90 words:
|
||||
Report the **single most important** real issue in the scope, in these four lines, under
|
||||
~90 words:
|
||||
|
||||
```
|
||||
1. <path>:<line>
|
||||
@@ -72,13 +42,10 @@ issue in the scope, in these four lines, under ~90 words:
|
||||
4. severity: high | medium | low
|
||||
```
|
||||
|
||||
- Found nothing real after reading? Reply `NO ISSUE` and one line saying why — a clean review
|
||||
is a valid result, and a fabricated issue is worse than none.
|
||||
- **Nothing real after reading?** Reply `NO ISSUE` and one line saying why. A clean review is a
|
||||
valid result; a fabricated issue is worse than none.
|
||||
- **Severity:** `high` = wrong result, data loss, security, or a hang/crash on a real path ·
|
||||
`medium` = a real bug on an edge path, or a correctness risk under load/concurrency ·
|
||||
`low` = clarity, a latent foot-gun, or a smell with no current failure.
|
||||
- Be specific and verifiable: a line number and a one-line repro beat an adjective. If you
|
||||
can't point to where it goes wrong, you haven't found it yet.
|
||||
|
||||
One reply closes the turn. If you asked mid-turn, the answer you got is already folded into
|
||||
this finding — you do not ask again after replying.
|
||||
- Be specific and verifiable: a line number and a one-line repro beat an adjective. If you can't
|
||||
point at where it goes wrong, you haven't found it yet.
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
name: CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
branches: [main]
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
# The wiki submodule is docs only and is not needed to build — leave it unfetched so CI
|
||||
# does not depend on the wiki repo being reachable.
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
# The runner image ships an older default-jdk; bridged sets maven.compiler.release=25, so
|
||||
# provision the JDK explicitly rather than apt-installing whatever "default" means today.
|
||||
- name: Set up JDK 25
|
||||
uses: actions/setup-java@v4
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '25'
|
||||
cache: maven
|
||||
|
||||
# setup-java provisions the JDK only — it does NOT install Maven, and the runner image has
|
||||
# no mvn on PATH (a bare `mvn` exits 127). Install it separately. apt pulls a default JRE as
|
||||
# a dependency; JAVA_HOME from setup-java still wins, which the version check below proves.
|
||||
- name: Install Maven
|
||||
run: |
|
||||
apt-get update && apt-get install -y --no-install-recommends maven
|
||||
mvn -version
|
||||
|
||||
- name: Build and test
|
||||
working-directory: bridged
|
||||
# This IS the mock-socket surface CB-503 asks for: the pom's `default-excludes` profile
|
||||
# already sets excludedGroups=contract, so the @Tag("contract") tests — which need a live
|
||||
# herdr socket and a RabbitMQ container — are excluded without any flag here. Everything
|
||||
# that runs does so against the fake UDS herdr and fake ccs/claude stubs.
|
||||
run: mvn -B clean install
|
||||
|
||||
# Deliberately NOT actions/upload-artifact: this Gitea instance presents as GHES, and
|
||||
# @actions/artifact v2+ (i.e. upload-artifact@v4) refuses to run there —
|
||||
# "GHESNotSupportedError ... not currently supported on GHES", which red-Xes an otherwise
|
||||
# green build. Since the artifact could not be retrieved anyway, dump the failing tests into
|
||||
# the log instead, where they are actually readable.
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: bridged
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
grep -qE "Failures: [1-9]|Errors: [1-9]" "$f" && { echo "===== $f ====="; cat "$f"; }
|
||||
done
|
||||
exit 0
|
||||
|
||||
# CB-521 — actually run the AMQP contract test in CI, against a REAL broker. The broker is a
|
||||
# RabbitMQ SERVICE CONTAINER, not Testcontainers-with-Docker: the runner image has no Docker, so
|
||||
# AmqpReplyInboxContractTest reads AMQP_URI (set below to the service's network alias) and binds
|
||||
# straight to it — no Docker, no skipped tests. This separation (build job hermetic and
|
||||
# Docker-free; contract job broker-provided) is deliberate — see the default-excludes/contract
|
||||
# profiles in bridged/pom.xml. `setup-java` provides the JDK only; Maven is installed separately,
|
||||
# exactly as in the build job above.
|
||||
contract:
|
||||
runs-on: ubuntu-latest
|
||||
services:
|
||||
rabbitmq:
|
||||
image: rabbitmq:3.13 # same AMQP 0-9-1 engine the local Testcontainers fixture uses
|
||||
env:
|
||||
RABBITMQ_DEFAULT_USER: guest
|
||||
RABBITMQ_DEFAULT_PASS: guest
|
||||
ports:
|
||||
- 5672:5672
|
||||
env:
|
||||
# Service containers are reachable from the job by their network alias on their internal port.
|
||||
AMQP_URI: amqp://guest:guest@rabbitmq:5672
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Set up JDK 25
|
||||
uses: actions/setup-java@v4
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '25'
|
||||
cache: maven
|
||||
|
||||
- name: Install Maven
|
||||
run: |
|
||||
apt-get update && apt-get install -y --no-install-recommends maven
|
||||
mvn -version
|
||||
|
||||
# The `contract` profile clears the default-excludes group, so the @Tag("contract") AMQP test
|
||||
# runs against the RabbitMQ service container (AMQP_URI). Pinned to the one contract test to
|
||||
# avoid re-running the unit suite already covered by the `build` job.
|
||||
- name: Contract tests
|
||||
working-directory: bridged
|
||||
run: mvn -B -Pcontract test -Dtest=AmqpReplyInboxContractTest
|
||||
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: bridged
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
grep -qE "Failures: [1-9]|Errors: [1-9]" "$f" && { echo "===== $f ====="; cat "$f"; }
|
||||
done
|
||||
exit 0
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
# No secret belongs in this repo any more: every credential lives in one shell-level store
|
||||
# (${SHARED_ENV}/tools/secrets.sh), and opencode.json reads it as {env:...}. This line stays as a
|
||||
# backstop, so a workspace-scoped copy that someone re-creates by habit still cannot be committed.
|
||||
.secrets/
|
||||
|
||||
# Settings backups inherit the env block — and secrets with it.
|
||||
.claude/settings.local.json.bak*
|
||||
|
||||
# The default profile parityOverlay copies these primary→worktree, so they appear in EVERY worker
|
||||
# worktree. Two reasons they must be ignored. They hold environment values, which is reason enough.
|
||||
# And since CB-576 a release preserves any worktree that `git status --porcelain` calls dirty —
|
||||
# untracked files included, deliberately. An untracked overlay file would therefore make every
|
||||
# COMPLETED release preserve its worktree, and worktrees would pile up with no error to notice.
|
||||
.env
|
||||
.envrc
|
||||
|
||||
# Daemon runtime artefacts. bridged appends its log wherever it is launched from, so both the
|
||||
# repo root and bridged/ collect one; neither belongs in git.
|
||||
bridged.out
|
||||
bridged/bridged.out
|
||||
logs/
|
||||
@@ -0,0 +1,24 @@
|
||||
---
|
||||
description: Refine work into clear, independent units before implementation.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from bridged.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You are an architect in this fleet. You refine work before anyone builds it: scope,
|
||||
acceptance criteria, risks, and a unit split. You read the repo and write analysis.
|
||||
You never commit production code and never open a pull request.
|
||||
|
||||
A design task is worked by two architects. Design alone first, then exchange and
|
||||
say plainly where you disagree. Do not concede just to agree.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `bridge_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Report only work you actually did and the real output of checks you ran. Do not
|
||||
claim a result from a tool you could not use. The primary's IDE tools are not yours.
|
||||
A mounted forge tool may use a blocked credential and fail by design.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,29 @@
|
||||
---
|
||||
description: Implement one assigned unit, test it, and open a pull request.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from bridged.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You implement the one unit you were given and nothing else. Work in your assigned
|
||||
git worktree and branch. Never check out, rebase onto, or push to `main`. Confirm
|
||||
the worktree root and branch before you edit. Use only paths under that root.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `bridge_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Implement the change and run the full required build in your worktree. Read the
|
||||
complete output and report its real result. Do not hide failures with a pipe. State
|
||||
only checks you actually ran. The primary's IDE tools are not yours. A mounted forge
|
||||
tool may use a blocked credential and fail by design.
|
||||
|
||||
Stage only files you changed. Never use `git add -A` or `git add .`. Never commit
|
||||
`.mcp.json` or `wiki/`. Commit with a clear message, push your branch, and open your
|
||||
own pull request against `main`. Never merge.
|
||||
|
||||
Your handoff must name the pull request or why it was not created, the branch, the
|
||||
files changed, the build result, and any caveat for review.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,34 @@
|
||||
---
|
||||
description: Review one assigned scope and report the most important real issue.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from bridged.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You review the diff you were given. Report bugs, risks, and missing tests. You do
|
||||
not change code.
|
||||
|
||||
Read the whole assigned scope before judging it. Review only that scope. If you see
|
||||
something outside it, note it in one line and do not investigate it further. Do not
|
||||
run the build. The owner makes changes and runs checks.
|
||||
|
||||
Use `bridge_ask{question}` only when a decision belongs to the lead, such as an
|
||||
unclear requirement or two defensible fixes. Do not ask about something you can
|
||||
decide by reading more code.
|
||||
|
||||
Report the single most important real issue in this form:
|
||||
|
||||
```
|
||||
1. <path>:<line>
|
||||
2. issue: <one sentence: what is wrong and why it matters>
|
||||
3. fix: <one line: the concrete change>
|
||||
4. severity: high | medium | low
|
||||
```
|
||||
|
||||
If there is no real issue, report `NO ISSUE` and one line saying why. A clean review
|
||||
is valid. Do not invent an issue. Use high for a wrong result, data loss, security,
|
||||
or a hang or crash on a real path. Use medium for an edge-path bug or a correctness
|
||||
risk under load or concurrency. Use low for clarity, a latent foot-gun, or a smell
|
||||
with no current failure.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -1,7 +1,313 @@
|
||||
# claude-bridge — project instructions
|
||||
|
||||
## Bridge communication (enforced — read this first)
|
||||
|
||||
> **Canonical block.** Everything down to §Layering is the portable bridge charter, copied verbatim
|
||||
> into every project that mounts the bridge MCP. Keep it byte-identical with the template in the
|
||||
> wiki ([Use Cases](https://git.ltms.dev/lms/claude-bridge/wiki/7-Use-Cases) → *The portable
|
||||
> CLAUDE.md block*); improvements go to the template first, then out to each project. Anything
|
||||
> specific to *this* repo lives under §Project addendum below, never inline above it.
|
||||
|
||||
If no `bridge_*` MCP tools are mounted in this session, this section does not apply — skip it.
|
||||
|
||||
`bridged` is the **sole communication gateway** between agents here. The orchestrating session (the
|
||||
**primary**) and every delegated peer (a **member**) mount the *same* MCP server and talk only
|
||||
through its `bridge_*` tools. No session addresses a peer, a broker, or the network directly.
|
||||
|
||||
### Which role am I? — settle this before acting
|
||||
|
||||
**Every role reads this file.** A member runs in a git worktree of this same repo, so it inherits
|
||||
this `CLAUDE.md` verbatim, and every rule below is role-conditional.
|
||||
|
||||
**Call `bridge_whoami`.** It returns `primary`, `worker`, or `architect`, resolved by the daemon from
|
||||
your connection — unforgeable, and the same resolution its authorization gate uses. A worker also
|
||||
carries its `sessionId`, `profile`, `worktree` and `branch`; an architect carries the slot name it
|
||||
was bound to. Don't infer what you can ask.
|
||||
|
||||
Only if that call is unavailable, fall back to these — each is one-way, so keep reading until one
|
||||
fires: the reply charter in your system prompt (*"You are an off-subscription worker in the
|
||||
claude-bridge fleet"*) ⇒ **spawned member**; bridge tools prefixed `mcp__bridge__*` ⇒ **spawned
|
||||
member** (the launcher fixes that mount name; a primary's mount is named by whoever wrote its
|
||||
`.mcp.json`, so it varies); `ANTHROPIC_BASE_URL` set ⇒ **spawned member** (Claude-model members run
|
||||
on a clean env, so its *absence* proves nothing). None of these separate a worker from an architect —
|
||||
only `bridge_whoami` does. **Still unsure ⇒ act as a worker**, the most restricted member role. The
|
||||
two mistakes are not symmetric: a primary acting as a worker is refused by the authorization gate —
|
||||
loud and self-correcting — while a member acting as the primary ends its turn with no `bridge_reply`,
|
||||
and the sender silently receives nothing. Fail toward the recoverable error.
|
||||
|
||||
### Invariants — both roles, no exceptions
|
||||
|
||||
1. **Never set, export, or forward `ANTHROPIC_BASE_URL`** (or `ANTHROPIC_AUTH_TOKEN`). The primary
|
||||
stays on subscription; only the bridge puts a member off it, at spawn. Mounting the bridge must
|
||||
never move a session across that boundary.
|
||||
2. **The bridge is the only channel.** Text you print in your terminal reaches nobody — the other
|
||||
side cannot see your screen. An answer that isn't in a `bridge_*` call is silently discarded.
|
||||
3. **Identity comes from the connection, never an argument.** Workers never pass a target; you
|
||||
cannot act as another session. Spawn/stop/drain are lead-only; **send is lead or architect**;
|
||||
reply/ask are only-as-itself — any peer may answer for its own pane, and for no other. A call
|
||||
outside your role is refused, not queued.
|
||||
4. **Delivery is status-gated: one message per turn.** Don't busy-poll a peer's terminal and don't
|
||||
re-send because a call looks slow — the bridge delivers when the peer is `idle`, `blocked` or
|
||||
`done`. A spawned member must **also** have mounted the bridge MCP: until it has, it is not
|
||||
deliverable, and a send waits on that gate for ~60s and then fails without ever reaching its pane.
|
||||
5. **Never drive the terminal multiplexer directly** (no `herdr` CLI, no socket). The bridge owns
|
||||
policy; the multiplexer owns PTYs. Going around the bridge bypasses every rule above.
|
||||
|
||||
### Primary (lead) — run this on every task, in order
|
||||
|
||||
**Delegate by default — that is the job.** With the bridge mounted you are an orchestrator on a
|
||||
metered subscription, and workers are cheap, parallel, and disposable. The default answer to "who
|
||||
does this?" is **a worker**, not you. Reach for `bridge_send` before you reach for `Edit`. The steps
|
||||
below are the procedure — run them in order, every task, not only the big ones.
|
||||
|
||||
0. **Know your role** — `bridge_whoami`, once per session, before anything else.
|
||||
1. **Split.** Write the unit list. Every unit carries: scope · the files or PR in question ·
|
||||
acceptance criteria · exactly what to report back. A unit with no acceptance criteria is not
|
||||
ready to delegate — refine it or keep it.
|
||||
2. **Gate each unit** on one question: **"can I write a brief good enough for a worker to
|
||||
succeed?"** — *not* "could I do this faster myself?" (usually you could; doing it yourself costs
|
||||
your context and your subscription, while a wasted worker turn costs a worker turn). Yes ⇒
|
||||
delegate. The keep-list is closed: the conversation with the user, decomposition and planning,
|
||||
the final judgment call, verification, merges, and anything that depends on context only you
|
||||
hold. Nothing else is yours by default.
|
||||
3. **Spawn every delegated unit first** — `bridge_spawn{profile, worktree:true, ticket}`, one per
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model and cost, not in
|
||||
tier, so the default is rarely what you want.
|
||||
4. **Then send them all** — `bridge_send{sessionId, content, wait:false}`. Line 1 of every brief is
|
||||
`Load the <name> skill.` naming the worker's playbook; those skills are opt-in and that line is
|
||||
what makes them reliable. Where the project ships no such skill, spell the procedure out in the
|
||||
brief instead. The brief is self-contained — the worker sees your message and the repo, nothing
|
||||
of your context, your plan, or your screen.
|
||||
5. **Collect** — `bridge_poll{ticket}` → `bridge_ack{target, msgId}`. Answer a worker's `bridge_ask`
|
||||
with `bridge_send{turnId, content}` — **not** `sessionId`. A worker gone quiet is diagnosed with
|
||||
`bridge_status`, never by reading its terminal; it also reports an open question and the `turnId`
|
||||
that answers it. **A worker's ask waits ~55 seconds, and no nudge makes that longer** — so never
|
||||
brief a worker to "ask me". Decide before you delegate, or give it an explicit default.
|
||||
6. **Verify yourself.** Re-run the build and the checks. A worker cannot run your IDE tooling, any
|
||||
forge tools it appears to have hold a blocked credential and fail, and a piped command
|
||||
(`… | tail`) hides failures behind a zero exit — never promote a worker's "clean" to a fact.
|
||||
7. **Review — fan out.** Spawn reviewers against the diff, one per dimension or per file, with
|
||||
`wait:false`. Never the implementer of the scope it reviews, and brief them from the diff — not
|
||||
from the implementer's rationale, which carries its own blind spot. Dispatch each PR's reviewers
|
||||
as it lands; don't wait for the last implementer. Under ~50 changed lines, skip the fan-out and
|
||||
read it yourself.
|
||||
8. **Adjudicate, merge, tear down — yours alone.** Read the diff yourself: fully if it is small,
|
||||
targeted at the reported findings and the risky paths if it is large. Reviewer findings direct
|
||||
your attention; they never substitute for it. Then merge, then `bridge_stop{paneId}`.
|
||||
|
||||
**Steps 3 and 4 are separate on purpose** — spawning and sending in one loop is how parallel work
|
||||
silently becomes serial, and it is the most common way this layer is wasted. For the same reason,
|
||||
prefer `wait:false` + `bridge_poll` for anything non-trivial: a blocking `bridge_send` is capped by
|
||||
*your own* MCP client call timeout (~60s), well below the task's real runtime.
|
||||
|
||||
**Delegating does not delegate responsibility.** Workers open PRs; you are the gate. Never delegate
|
||||
the merge — and merging on a reviewer's word is delegating it by proxy.
|
||||
|
||||
| Intent | Tool |
|
||||
|---|---|
|
||||
| Confirm your own role | `bridge_whoami` |
|
||||
| See backends available | `bridge_profiles` |
|
||||
| Start a member | `bridge_spawn{role?, profile?, cwd?, worktree?, ticket?, sessionName?, resumeSessionId?}` → `sessionId` + `paneId` |
|
||||
| See the fleet | `bridge_list` → `leads` (your peers) + `members` (each carries `agentSessionId` when its backend knows one) · one peer's state: `bridge_status{sessionId}` |
|
||||
| Delegate (blocking) | `bridge_send{sessionId, content}` |
|
||||
| Delegate (long task) | `bridge_send{sessionId, content, wait:false}` → ticket → `bridge_poll{ticket}` |
|
||||
| Answer a member's `bridge_ask` | `bridge_send{turnId, content}` — **not** `sessionId` |
|
||||
| Message a **peer lead** | `bridge_send{sessionId: <their terminal>, content}` — `bridge_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `bridge_reply{content}` — the one case a lead replies |
|
||||
| Collect a held reply | `bridge_poll{target}` · then `bridge_ack{target, msgId}` |
|
||||
| Tear down a member | `bridge_stop{paneId}` |
|
||||
|
||||
### Lead ↔ lead — coordinate, never delegate
|
||||
|
||||
`bridge_list` returns `leads` alongside `members`; your own row carries `self: true`. Every other row
|
||||
is a peer — an orchestrator with its own context, its own members, and its own judgment. An empty
|
||||
`members` array means no members are spawned; it says nothing about peers.
|
||||
|
||||
**A lead never assigns a task to another lead.** Work goes to members — only ever downward, never
|
||||
sideways. Sending a peer a brief with acceptance criteria is a category error: a brief is a member's
|
||||
artefact, and a peer is not yours to task. If a unit needs doing and it falls in your area, spawn a
|
||||
member and delegate it yourself; if it falls in the peer's area, say so and let the peer assign it.
|
||||
The traffic between leads is coordination and nothing else:
|
||||
|
||||
1. **Divide the map, not the work.** Agree who owns which area, then each of you assigns inside your
|
||||
own. Split by **context ownership** — whoever already holds the context owns that area — and say
|
||||
who takes what, in one message, before either of you starts. Two leads silently working the same
|
||||
unit is the failure mode here, and neither notices until the merge.
|
||||
2. **Share findings, hazards, and corrections.** What you have already discovered, what broke, what
|
||||
the next person will trip on. This is the traffic that actually pays for the channel: it costs one
|
||||
message and saves a peer a rediscovery.
|
||||
3. **Verify a peer exactly as you verify yourself.** Peer status buys nothing: check the claim
|
||||
against the code, and re-run the build. A peer's correction gets the same treatment — right or
|
||||
wrong on the evidence, not on who said it. Neither of you merges the other's work unreviewed.
|
||||
|
||||
Being messaged by a peer does not make you its worker: answer with `bridge_reply`, and push back on
|
||||
the substance if it is wrong. A peer that simply complies has thrown away the reason there are two of
|
||||
you.
|
||||
|
||||
### Member (worker or architect) — the turn contract
|
||||
|
||||
1. **Load the playbook skill the lead named** before doing anything else.
|
||||
2. **Do the assigned scope only.** Note anything you spot outside it in one line; don't go hunt it.
|
||||
3. **`bridge_ask{question}`** when a decision is genuinely the lead's (ambiguous requirement, two
|
||||
defensible fixes, "bug or intended?"). It blocks and you resume the *same* turn with the answer.
|
||||
Don't ask what you could decide yourself.
|
||||
4. **End the turn with exactly one `bridge_reply{content}`**, carrying your complete answer. This is
|
||||
the whole handoff. No `bridge_reply` ⇒ the sender gets nothing and the exchange stalls.
|
||||
Do **not** lean on the completion fallback to carry your answer for you: when you end a turn
|
||||
without replying, the bridge scrapes your pane, and it can return only the last 4000 characters.
|
||||
A clipped scrape is marked as partial, but the missing text is gone — your report reaches the
|
||||
lead with its end cut off.
|
||||
5. **Report honestly.** State only what you actually ran and its real output, including failures,
|
||||
and never claim the result of a check you had no way to run. **Measure your own tools; do not
|
||||
assume them.** What you mount depends on your backend: an opencode member gets the bridge and
|
||||
nothing else, while a Claude Code member also inherits the operator's user-scope MCP servers,
|
||||
which the bridge never chose for you. Two rules follow. The primary's IDE tooling is still not
|
||||
yours, whatever you see. And **a mounted tool is not a working tool** — the forge server you may
|
||||
find there holds a deliberately blocked credential and fails every call, by design.
|
||||
6. **Never merge.** Stage files explicitly — never `git add -A` — and leave alone anything the
|
||||
project marks as not-yours-to-commit.
|
||||
|
||||
### Where each rule lives (don't duplicate — extend the right layer)
|
||||
|
||||
| Layer | Scope | Reaches |
|
||||
|---|---|---|
|
||||
| the launcher's reply charter | the one rule that must survive with no repo: *end every turn with `bridge_reply`* | every spawned member, at launch, every peer kind — never a lead |
|
||||
| **this section** | protocol + orchestration policy | primary **and** every member that reads the repo — tracked in git, so worktrees inherit it |
|
||||
| role agent definition files | role contract and per-job procedure | a member whose launcher binds its role to the matching file in its worktree |
|
||||
| role playbook skills | per-job procedure (commit/PR recipe, finding format) | a member told to load one |
|
||||
| the bridge's own docs | design detail, flows, error model | on demand |
|
||||
|
||||
A rule belongs in **exactly one** layer — the outermost one that must obey it. A member without a
|
||||
repo checkout still gets the launcher's reply charter, which is why that one rule stays there.
|
||||
Peers that don't read `CLAUDE.md` (non-Claude adapters) get the charter only, so any rule *they*
|
||||
must obey belongs in the charter, not here.
|
||||
|
||||
## Project addendum — claude-bridge (not part of the canonical block)
|
||||
|
||||
- **This repo is the bridge.** The daemon is `bridged`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
and the code behind the rules above is `mcp/BridgeMcp` (tools), `auth/Authz` (the role table),
|
||||
`mcp/ConnectionIdentity` (connection→role), and `worker/*Launcher` (`REPLY_CHARTER`).
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR) and
|
||||
`reviewer` (scoped review → one structured finding). Name one in every delegation.
|
||||
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace).
|
||||
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
||||
(a submodule with its own remote).
|
||||
- **Flows and the error model** — rendezvous, `bridge_ask`, detached delivery, turn-done fallback —
|
||||
are diagrammed in `docs/MCP-Contract.md` **§6 only**. The rest of that page is a pre-build design
|
||||
doc whose tool names, parameter names and REST paths never caught up with the code, so do not use
|
||||
it as the tool reference (CB-609). Section 6 is kept out of this file because this file loads into
|
||||
every session's context.
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
**A merge is not a deployment.** The running `bridged` holds the jar it was started with, so a
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. Saying "shipped"
|
||||
about code the live daemon has never loaded is a false report. The lead **may and should** redeploy
|
||||
rather than hand the job back to the operator.
|
||||
|
||||
Workers must never do this. A worker has no business restarting the daemon it is talking through,
|
||||
and stopping it kills the worker's own channel mid-turn.
|
||||
|
||||
**Use the script — do not hand-roll the steps.**
|
||||
|
||||
```bash
|
||||
scripts/redeploy-bridged.sh --check # report state, change nothing
|
||||
scripts/redeploy-bridged.sh # build, confirm drain, restart, verify
|
||||
scripts/redeploy-bridged.sh --yes # skip the drain prompt (fleet already checked)
|
||||
```
|
||||
|
||||
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
||||
old process to exit rather than assuming; it polls `/healthz`; and it anchors its log checks to a
|
||||
line marker taken before the restart, so old errors cannot be misread as new ones. Run `--check`
|
||||
first — it is read-only and reports whether the forge token resolves, which nothing else tells you.
|
||||
|
||||
The script encodes the five things below, each of which has gone wrong here before. Read them anyway:
|
||||
if the script is unavailable or a step fails, this is what it was protecting you from.
|
||||
|
||||
1. **Login shell, or workers silently lose their forge token.** The daemon inherits
|
||||
`WORKER_GITEA_TOKEN` from the shell that starts it, and that comes from
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Start it from a non-login shell and the variable is empty, the
|
||||
daemon starts fine, and the failure appears much later as workers that cannot open a PR. Nothing
|
||||
logs this at startup — the script's `--check` is the only thing that reports it, and it checks
|
||||
whether the name resolves without ever printing the value.
|
||||
2. **Drain live members first.** `bridge_list`, then `bridge_stop` each member, and collect anything
|
||||
you still want with `bridge_poll` before you kill anything. A restart drops in-flight tickets and
|
||||
rendezvous, and a member's report is not recoverable once its ticket is gone.
|
||||
3. **A restart is the only way deferred config keys take effect.** That is usually the reason to do
|
||||
it. The startup log names which keys it accepted and which it deferred — read those lines rather
|
||||
than assuming.
|
||||
4. **Re-check identity afterwards.** Call `bridge_whoami` and confirm it still answers `primary`. The
|
||||
lead is found by its tab label (`fleet.leaders.*.tab`), and a lead whose tab no longer matches is
|
||||
demoted to worker, which refuses every orchestration call.
|
||||
5. **Prove the new jar is the one running.** Confirm a *fresh* `bridged listening` line at the end of
|
||||
`bridged/bridged.out`, dated after the restart. An old daemon that never died looks identical from
|
||||
the outside.
|
||||
|
||||
**Permission.** A `CLAUDE.md` rule grants intent, not tool permission — the command classifier
|
||||
refuses a bare `kill` on the daemon whatever this file says. The script is the seam that fixes that:
|
||||
it is one auditable command, so the operator allow-lists it once instead of approving a stop and a
|
||||
start every time. The rule lives in the operator's Claude Code settings:
|
||||
|
||||
```json
|
||||
{ "permissions": { "allow": ["Bash(scripts/redeploy-bridged.sh:*)"] } }
|
||||
```
|
||||
|
||||
Granted by the operator on 2026-08-15. If a call is still refused, do **not** route around it by
|
||||
running the stop and start as separate commands — that is exactly the approval the script replaced.
|
||||
Say what you were going to run and why, and let the operator decide.
|
||||
|
||||
### The prompt is part of the product — update it with the code (mandatory)
|
||||
|
||||
This repo *is* the bridge, so the canonical block above is not documentation about someone else's
|
||||
system: it is the instruction surface this codebase ships. **Every change here must end by asking
|
||||
whether the block still tells the truth.** A code change that silently invalidates it is an
|
||||
incomplete change — the agents reading it have no other source.
|
||||
|
||||
Before you call any work done, check the row that matches what you touched:
|
||||
|
||||
| You changed… | Re-read and update… |
|
||||
|---|---|
|
||||
| a `bridge_*` tool — added, removed, renamed, or its params/semantics | the primary's intent→tool table; any rule that names that tool |
|
||||
| `Authz` / the role table | invariant 3, and the primary-only vs worker-only claims |
|
||||
| `ConnectionIdentity` / how a caller is resolved | the `bridge_whoami` paragraph and the fallback ladder |
|
||||
| `REPLY_CHARTER`, or a launcher's mount/flags | the fallback ladder (`mcp__bridge__*`), and the layering table's top row |
|
||||
| the injector / status gating | invariant 4 |
|
||||
| worktree provisioning or the parity overlay | the "both roles read this file" premise — it rests on the worker's worktree being a checkout of this repo |
|
||||
| `.claude/skills/**` | the addendum's skill list, and the "name the playbook" rule |
|
||||
| a new peer kind (non-Claude adapter) | what that peer can read — anything it must obey belongs in its charter, not in the block |
|
||||
| **anything an operator can use, configure, or observe** — an MCP tool, a `bridged.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
||||
|
||||
That last row is not bookkeeping. Chapters 1–10 answer *how is this built* and *why this way*;
|
||||
none of them has a home for *what can it do and how do I turn it on*, so for twenty tickets a
|
||||
shipped capability landed nowhere and the Roadmap went on claiming the stage was finished. The
|
||||
*why* line is the one that matters — without it a decision gets re-litigated from scratch a month
|
||||
later. Internal contract changes go to `wiki/9-Implementation.md` instead; test and coverage work
|
||||
is a Roadmap line. A change that touches none of the three earns no entry, and that is a normal
|
||||
outcome rather than an omission.
|
||||
|
||||
Then **propagate**: the block in this file and the template in the wiki
|
||||
([Use Cases](https://git.ltms.dev/lms/claude-bridge/wiki/7-Use-Cases) → *The portable `CLAUDE.md`
|
||||
block*) must stay byte-identical, and other projects carrying the block need the same edit. Verify
|
||||
rather than trust:
|
||||
|
||||
```bash
|
||||
python3 - <<'PY'
|
||||
import pathlib
|
||||
c = pathlib.Path("CLAUDE.md").read_text()
|
||||
w = pathlib.Path("wiki/7-Use-Cases.md").read_text()
|
||||
S, E = "## Bridge communication (enforced", "## Project addendum — claude-bridge"
|
||||
block = c[c.index(S):c.index(E)].rstrip() + "\n"
|
||||
i = w.index("```markdown\n") + len("```markdown\n")
|
||||
print("in sync:", w[i:w.index("\n```\n", i) + 1] == block)
|
||||
PY
|
||||
```
|
||||
|
||||
## IDE MCP tools & validation workflow (enforced)
|
||||
|
||||
> **Primary only.** Workers have no IDE MCP mount — if you are a worker, skip this section and
|
||||
> report the build/test output you actually ran (see §Bridge communication → Worker).
|
||||
|
||||
Two IDE MCP servers are connected: **intellij-index** (semantic code intelligence) and
|
||||
**jetbrains** (file problems, reformat, debugger). IntelliJ has multiple projects open; our
|
||||
module is **`bridged`**. Always pass these to IDE MCP tools:
|
||||
|
||||
@@ -18,8 +18,9 @@ one unified Claude setup and the **sole communication gateway** (REST/SSE stays
|
||||
clients; any broker is `bridged`-internal, below the gateway).
|
||||
herdr owns the PTYs, multiplexing, persistence, and **agent-status events**; `bridged` owns
|
||||
policy (subscription boundary, session lifecycle, status-gated delivery) and the client
|
||||
contract. The worker `claude` launches with `ANTHROPIC_BASE_URL=https://ollama.ltms.dev` + a
|
||||
bearer token; the primary Opus stays env-clean and calls `bridged`'s MCP tools.
|
||||
contract. A Claude member launches with `ANTHROPIC_BASE_URL` pointed at the gateway,
|
||||
`https://llm.ltms.dev/anthropic`, plus a bearer token; the lead stays env-clean and calls
|
||||
`bridged`'s MCP tools. See the wiki's **[13 User Guide](wiki/13-User-Guide.md)** to run it.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
@@ -31,7 +32,7 @@ flowchart LR
|
||||
end
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
M["ollama.ltms.dev<br/>(worker model)"]
|
||||
M["llm.ltms.dev<br/>(the one gateway)"]
|
||||
|
||||
OPUS -->|"MCP bridge_send (blocks)"| SRV
|
||||
W -.->|"MCP bridge_reply"| SRV
|
||||
@@ -92,7 +93,8 @@ bridge (code reviews delegated this way have produced committed bug fixes). Sele
|
||||
primary approach 2026-07-11, superseding the AgentAPI plan (2026-07-08); AgentAPI retained as a
|
||||
fallback injector.
|
||||
|
||||
**Shipped** (Java 25 · Maven · 105 tests green — unit/acceptance + live-herdr contract tests):
|
||||
**Shipped** (Java 25 · Maven · 266 unit/acceptance tests green; the live-herdr and broker contract
|
||||
tests run separately via `mvn test -Pcontract`):
|
||||
|
||||
- **Core gateway** — herdr socket client (contract-tested vs live 0.7.0); guard-checked worker
|
||||
spawn with `ANTHROPIC_BASE_URL` injected only into the worker's env; status-gated injector;
|
||||
@@ -106,7 +108,17 @@ fallback injector.
|
||||
- **Fleet** — multiple worker profiles, each with an independent base_url guard check; workers
|
||||
inherit the primary's working directory (never `$HOME`); a readiness gate holds delivery until
|
||||
a worker's Claude has connected the bridge MCP (no paste lost into its boot window).
|
||||
- **Blocked-worker path** — `bridge_ask` reverse rendezvous: a worker pauses its delegated turn to
|
||||
ask the primary and resumes the *same* turn with the answer (CB-205).
|
||||
- **Session lifecycle** — session manager with spawn/reuse/recycle, `idle_ttl` reaper, `context_cap`,
|
||||
and graceful drain on shutdown (CB-301/CB-303); per-worker git worktrees on their own branch with
|
||||
a config-parity overlay, so parallel implementers never stomp each other (CB-301-ext).
|
||||
- **Reliable worker→primary delivery** — a durable `ReplyInbox` (in-memory by default, AMQP/LavinMQ
|
||||
for cross-restart durability) holds a reply that arrives with no open send, and an active
|
||||
status-gated push loop nudges the primary to drain it (CB-307).
|
||||
- **Pluggable peers** — a `PeerLauncher` SPI with two in-tree adapters, `claude-code` and `opencode`,
|
||||
routed by a `kind:` discriminator (CB-401/CB-402).
|
||||
|
||||
**Next** (see the [roadmap](wiki/8-Roadmap.md)) — structured envelope schema, `bridge_ask`
|
||||
(blocked-worker path), session lifecycle / recycle / `idle_ttl`, split-host, and hardening
|
||||
(auth/TLS, `/metrics`, CI, systemd).
|
||||
**Next** (see the [roadmap](wiki/8-Roadmap.md)) — Stage 5 hardening (auth/TLS, `/metrics`, CI,
|
||||
service supervision, per-session authz + audit), then cross-host: CB-308 multi-host federation and
|
||||
CB-500 multi-tier coordination.
|
||||
|
||||
@@ -5,6 +5,9 @@ dependency-reduced-pom.xml
|
||||
# Local runtime config (copy from bridged.example.yaml)
|
||||
bridged.yaml
|
||||
|
||||
# CB-505 audit trail + daemon stdout/stderr — runtime records, never source
|
||||
logs/
|
||||
|
||||
# Editor / OS
|
||||
*.iml
|
||||
.idea/
|
||||
|
||||
+573
-13
@@ -3,20 +3,125 @@
|
||||
# bridged is the sole gateway between primary/worker Claude sessions and herdr.
|
||||
# It is NOT a Claude process and must never carry ANTHROPIC_BASE_URL.
|
||||
|
||||
# REST + MCP listen address. Keep it on loopback — bridged is same-host in Stage-1.
|
||||
# REST + MCP listen address. Keep it on loopback unless you also switch auth.mode to `token`
|
||||
# below — bridged REFUSES TO START on a non-loopback bind under loopback-trust (see auth).
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
|
||||
# API authentication (CB-501). Governs how a caller that is NOT an on-host worker pane proves it
|
||||
# is the primary. Worker identity never depends on this: a loopback peer PID that maps to a herdr
|
||||
# pane is unforgeable and is always honoured, so turning auth on cannot lock the fleet out.
|
||||
#
|
||||
# mode: loopback-trust → DEFAULT, and the historical behaviour: any loopback caller that is not
|
||||
# a worker is the primary, no credential needed. Sound ONLY because the
|
||||
# OS refuses remote connections to a loopback socket.
|
||||
# mode: token → such a caller must send `Authorization: Bearer <token>`; without it it
|
||||
# is anonymous and authorized for nothing. REQUIRED for a non-loopback
|
||||
# bind — the daemon fails fast otherwise, because "unauthenticated ⇒
|
||||
# primary" on a reachable port would hand spawn/stop/send to anyone.
|
||||
# tokenEnv → host env var holding the token (never the literal value). Default
|
||||
# BRIDGED_API_TOKEN. Read only in token mode; empty ⇒ startup fails.
|
||||
#
|
||||
# TLS is deliberately NOT terminated in the daemon (CB-501 D3): run a reverse proxy in front and
|
||||
# let it own certificate lifecycle, e.g.
|
||||
# location / { proxy_pass http://127.0.0.1:8765; proxy_set_header Authorization $http_authorization; }
|
||||
# The broker link gets TLS from its own URI (amqps://…) — see `broker` below.
|
||||
# auth:
|
||||
# mode: token
|
||||
# tokenEnv: BRIDGED_API_TOKEN
|
||||
|
||||
# Optional pinned primary terminal (CB-307). Names the herdr pane the PRIMARY itself runs in:
|
||||
# a caller whose connection maps to this pane resolves as the primary (no credential needed —
|
||||
# the pane mapping is as unforgeable as a worker's), and reply nudges are pushed to it.
|
||||
# REQUIRED when the primary runs inside a herdr pane — without it the pane match reads the
|
||||
# primary as a worker and refuses spawn/send/stop. Get the id from bridge_whoami; re-pin if
|
||||
# the primary moves panes.
|
||||
# primary:
|
||||
# terminal: term_0123456789abcd
|
||||
# pushReminders: 5 # max nudges before giving up (default 5)
|
||||
# pushBackoffMs: 15000 # delay between nudges (default 15000)
|
||||
|
||||
# CB-530: MORE THAN ONE LEAD. `primary:` above is singular by construction — every other pane
|
||||
# resolves as a worker — which is right for one lead driving a fleet and wrong the moment two leads
|
||||
# (say a Claude lead and an opencode lead) work as peers: the second is silently demoted and refused
|
||||
# every orchestration call. List each lead's pane here and all of them resolve as leads.
|
||||
#
|
||||
# tab → the ONLY field identity depends on (CB-579); the exact label of the tab hosting the lead.
|
||||
# Label the tab yourself, or let bridged label one it launches — see `fleet.leaders:` below.
|
||||
# kind/model → descriptive; they document what runs in the pane and are echoed by bridge_whoami
|
||||
#
|
||||
# A lead's tab must already carry its label (or be launched by bridged, which labels it) — there is
|
||||
# no terminal id to paste in and nothing to re-pin when the session restarts: the tab survives, so
|
||||
# the same label resolves the same lead again on the next scan.
|
||||
# `bridge_whoami` reports `{"role":"primary","leader":"<name>"}`; role stays "primary" because a lead
|
||||
# IS a primary for authorization, so nothing that keys on the role breaks.
|
||||
#
|
||||
# KEEP `primary:` when adding leads: it still addresses the CB-307 push loop, which needs a single
|
||||
# destination for its nudges, and is a separate mechanism from lead identity — see `fleet.leaders:`.
|
||||
#
|
||||
# Leads are configured under `fleet.leaders:` — see THE FLEET further down.
|
||||
#
|
||||
# Two things stop the tab-name convention from becoming a way to claim leadership: the configured
|
||||
# member spaces are excluded from the scan, so nothing bridged places can land in a matching tab;
|
||||
# and startup REFUSES a `tabPrefix` that the fleet tabLabel template, or any per-profile `tabLabel`
|
||||
# override, also matches — so the two namespaces cannot overlap by accident. The label is a NAME,
|
||||
# never a capability: what a pane may do is decided by the role the daemon resolves for it.
|
||||
|
||||
# CB-551: IDLE-LEAD HEARTBEAT — nudge the single lead back to work when it has been continuously
|
||||
# idle (no open bridge_send driving it) past the quiet period. The fleet is one lead + architects +
|
||||
# workers, so a lead that stalls is a single point of failure; the ReplyPushLoop only nudges when a
|
||||
# reply lands, and this timer catches the gap where nothing lands and the lead just sits idle.
|
||||
#
|
||||
# Opt-in on purpose — it SPENDS the operator's subscription on its own initiative (each nudge starts
|
||||
# a lead turn nobody asked for), so upgrading the daemon must never switch it on for you. Absent
|
||||
# block = feature off, exactly as before.
|
||||
#
|
||||
# Three knobs, each with a default that errs on the side of not burning context:
|
||||
# idleAfterSeconds: 300 # how long the lead must stay idle before the FIRST nudge (default 300 —
|
||||
# # absorbs normal post-turn pauses; re-prompting every pause burns context)
|
||||
# backoffMs: 60000 # re-check cadence / spacing between nudges past the quiet period (default 60000)
|
||||
# quietNudgeCap: 3 # cap on consecutive nudges that find NOTHING pending, then it stops
|
||||
# # until real state appears (default 3 — never nag an empty fleet forever)
|
||||
# leadHeartbeat:
|
||||
# idleAfterSeconds: 300
|
||||
# backoffMs: 60000
|
||||
# quietNudgeCap: 3
|
||||
|
||||
# Fleet health detection is dormant unless enabled (CB-573). It reads one whole-fleet agent list
|
||||
# per tick.
|
||||
# intervalSeconds → how often a tick runs (default 30). ENFORCED floor of 15: the code computes
|
||||
# Math.max(15, intervalSeconds), so a lower value is silently raised, not
|
||||
# rejected.
|
||||
# workingSuspectAfterSeconds, paneProbeIntervalSeconds → accepted and parsed, but NOT YET READ by
|
||||
# anything — the dormant monitor only consumes intervalSeconds today (CB-573
|
||||
# shipped ahead of the evidence publishers these two knobs are for). Setting
|
||||
# them changes nothing right now, and no minimum is enforced on either, because
|
||||
# nothing reads them to enforce one. They exist so a later build can start
|
||||
# honouring them without another config-shape change.
|
||||
# notifications.mode → "webhook" flips what bridge_list REPORTS (healthCoverage: "full" instead
|
||||
# of "detection-only") — it does NOT make bridged send any webhook call; no
|
||||
# delivery mechanism is implemented yet. Any other value, or omitting the
|
||||
# block, reports "detection-only".
|
||||
# health:
|
||||
# enabled: true
|
||||
# intervalSeconds: 30
|
||||
# workingSuspectAfterSeconds: 600
|
||||
# paneProbeIntervalSeconds: 60
|
||||
# notifications:
|
||||
# mode: disabled
|
||||
|
||||
# herdr Unix socket. Omit to use the client default
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
|
||||
# How worker sessions are spawned. Define one or more named profiles (backends) under
|
||||
# `workers`; each key is the profile name (also the ccs profile). `defaultWorker` picks
|
||||
# which one a no-argument spawn uses (bridge_spawn with no profile / POST /workers).
|
||||
# How member sessions are spawned. Define one or more named profiles (backends) under
|
||||
# `profiles`; each key is the profile name (also the ccs profile). A profile says only WHICH
|
||||
# BACKEND — model, CLI adapter, credentials, cost. It says nothing about what a member spawned on
|
||||
# it is for; that is the member's role, and roles live under `fleet:` below. Which profile an
|
||||
# unqualified spawn lands on comes from that role's pool, not from a global default.
|
||||
#
|
||||
# Shared knobs (placement/workspace/tabLabel) can be repeated per profile; they usually match.
|
||||
# Shared knobs (placement/workspace) can be repeated per profile; they usually match.
|
||||
# placement: tab → each worker lands in its OWN tab in a dedicated worker space (default).
|
||||
# Use `pane` for the legacy behaviour (split the focused tab).
|
||||
# mcpUrl → bridged mounts the bridge MCP (--mcp-config, inline) + reply charter
|
||||
@@ -26,25 +131,340 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# cwd → pin this profile's working directory (CB-112). Omit to inherit the primary's
|
||||
# cwd on an MCP spawn, else the daemon's cwd — never $HOME. See
|
||||
# docs/Worker-Startup-and-Trust.md.
|
||||
# configDir → CLAUDE_CONFIG_DIR for the worker, so it inherits that profile's
|
||||
# skills/MCP/hooks. Omit to leave the worker on the host default.
|
||||
# parityOverlay → repo-relative paths copied primary→worktree so a worker in a provisioned
|
||||
# worktree sees the same local config (CB-301-ext). Omit for the default set:
|
||||
# [.claude/settings.local.json, .env, .envrc].
|
||||
#
|
||||
# Do NOT add .mcp.json (CB-525). A worker's tools are whatever its launcher
|
||||
# mounts — the bridge, and nothing else. Replicating the primary's MCP config
|
||||
# handed a worker the primary's IDE servers, which are bound to the primary's
|
||||
# checkout, so its navigation returned paths OUTSIDE its own worktree: one
|
||||
# worker made all 59 of its edits in the primary tree while compiling its
|
||||
# worktree, and every build it ran was of code that did not contain them.
|
||||
# bridged neutralizes a provisioned worktree's .mcp.json for this reason;
|
||||
# listing it here would copy the primary's back over that.
|
||||
# gitTokenEnv → host env var holding the git-forge API token. When set, its value is injected
|
||||
# as GITEA_TOKEN so the worker can open its OWN PR at checkpoint (CB-302).
|
||||
# Opt-in by design — omit and the worker gets no PR-create grant (push over
|
||||
# SSH is unaffected). The token value itself is never stored in this file.
|
||||
# gitHostEnv → host env var holding the forge host (default GITEA_HOST). Injected as
|
||||
# GITEA_HOST *only* alongside a resolved gitTokenEnv.
|
||||
# exhaustedPattern → regex matched against a completion-fallback scrape (CB-578 stage A) to
|
||||
# classify a turn that ended with no bridge_reply as the backend having
|
||||
# refused on a subscription usage limit, rather than a real answer. Opt-in —
|
||||
# omit and this profile's completion fallback behaves exactly as before.
|
||||
# Every backend words its refusal differently, so this is config, never a
|
||||
# vendor string baked into bridged itself.
|
||||
# DEFERRED: compiled once into a startup pattern map — editing it needs a
|
||||
# daemon restart, same as this profile's model/baseUrl/argv.
|
||||
# credentialId → CB-578 stage B: the credential this profile quarantines WITH when a
|
||||
# BACKEND_EXHAUSTED classification fires. Two profiles that set the SAME
|
||||
# credentialId share one quarantine — the case this exists for is two models
|
||||
# on one account (e.g. sol and terra both billing one OpenAI credential): an
|
||||
# exhaustion on either one must lock out both, or the fleet just walks onto
|
||||
# the same dead account under the sibling's name. Opt-in — omit and this
|
||||
# profile quarantines alone, under its own name, exactly as if the field did
|
||||
# not exist. Cooldown length is the top-level quarantineCooldownSeconds below.
|
||||
# HOT: read live at every spawn/exhaustion check — no restart needed.
|
||||
# env → extra environment for this profile's workers, as a literal key/value map
|
||||
# (CB-511). Use it to give workers a toolchain.
|
||||
#
|
||||
# A worker's environment does NOT come from your shell. bridged hands herdr an
|
||||
# explicit env map and herdr merges it into ITS OWN process env — so before
|
||||
# CB-511 a worker inherited whatever PATH the herdr server happened to be
|
||||
# started with, which on a long-lived herdr can predate your toolchain entirely
|
||||
# and leave workers unable to run `mvn` or `java` at all.
|
||||
# bridged now propagates ITS OWN PATH to every worker by default; set `env:`
|
||||
# only to override that or add more (JAVA_HOME, …). Since the default is the
|
||||
# daemon's PATH, make sure the daemon is started with a good one — see the PATH
|
||||
# lines in deploy/dev.ltms.bridged.plist and deploy/bridged.service.
|
||||
#
|
||||
# Adapter-owned variables always win over `env:`: ANTHROPIC_BASE_URL and the
|
||||
# rest of the ANTHROPIC_*/CLAUDE_* wiring are applied after it, so an `env:`
|
||||
# entry cannot repoint a worker past the SubscriptionGuard — which is checked
|
||||
# against `baseUrl` alone.
|
||||
# Put `defaultMode: "auto"` in each ccs profile so the worker runs autonomously.
|
||||
workers:
|
||||
profiles:
|
||||
gx10: # ccs profile name (NOT a hostname)
|
||||
kind: claude-code # which adapter spawns this profile (default; may omit)
|
||||
baseUrl: http://gx01.gw:8000 # the vLLM host this profile targets (gx00.gw / gx01.gw)
|
||||
model: coder
|
||||
placement: tab
|
||||
workspace: bridged-workers
|
||||
tabLabel: "worker: {profile} #{n}" # {profile}/{model}/{n} substituted; {n} keeps sibling tabs distinct
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
tokenEnv: BRIDGED_WORKER_TOKEN
|
||||
argv: ["ccs", "gx10"]
|
||||
ollama:
|
||||
baseUrl: http://ollama.ltms.dev # local/self-hosted; usually no token
|
||||
# weight: relative selection weight for automatic placement (weighted, round-robin, and
|
||||
# fixed's fallback walk). Absent defaults to 1.0. An explicit 0 or negative value means
|
||||
# "never auto-select this profile" (CB-554) — it stays reachable via an explicit
|
||||
# `bridge_spawn{profile:"gx10"}`, which bypasses placement entirely; only automatic
|
||||
# selection skips it.
|
||||
weight: 0.5
|
||||
# maxLoad: max live workers on this profile. Omit for unlimited. An explicit 0 (CB-585) caps
|
||||
# the profile at zero live members — it is excluded from automatic placement and an explicit
|
||||
# `bridge_spawn{profile:"gx10"}` against it is refused too; a cap holds even when the profile
|
||||
# is named directly. Negative is refused at config load — there is no sane meaning for it.
|
||||
maxLoad: 2
|
||||
# subscription: true
|
||||
# THE KNOB THAT DECIDES WHO PAYS (CB-539). Default false. When true, this profile's members
|
||||
# run on the OPERATOR'S OWN Claude subscription instead of a metered endpoint — every spawn
|
||||
# bills your plan and eats your usage limit. Off-subscription is the whole point of this
|
||||
# daemon, so treat `true` as a deliberate exception, not a convenience.
|
||||
#
|
||||
# What changes when it is set (ClaudeCodeLauncher):
|
||||
# - no ANTHROPIC_BASE_URL and no ANTHROPIC_AUTH_TOKEN are injected — the member inherits
|
||||
# the operator's own Claude Code auth, which is exactly why it bills the plan;
|
||||
# - SubscriptionGuard never vets it, because there is no baseUrl to vet;
|
||||
# - no token is required, so `tokenEnv` is irrelevant here.
|
||||
#
|
||||
# MUTUALLY EXCLUSIVE with `baseUrl` — setting both is refused at config load (CB-542). On the
|
||||
# subscription path no guard would vet the URL, so allowing both would be a way around the
|
||||
# guard rather than a configuration.
|
||||
#
|
||||
# GOTCHA 1 — it is invisible to the startup secret check. `Bridged.reportRequiredSecrets`
|
||||
# skips subscription profiles on purpose (they need no token), so a boot log that reports
|
||||
# every secret as fine says nothing about these profiles.
|
||||
#
|
||||
# GOTCHA 2 — `maxLoad` is the ONLY throttle you have here. There is no metering, no budget
|
||||
# and no refusal on cost; the cap on live members is the single thing standing between a
|
||||
# fan-out and your monthly limit. Set it deliberately and keep it small.
|
||||
# gitTokenEnv: GITEA_TOKEN # opt-in: let this profile's workers open their own PR (CB-302)
|
||||
# gitHostEnv: GITEA_HOST # defaults to GITEA_HOST; injected only with gitTokenEnv
|
||||
# exhaustedPattern: "usage limit has been reached" # opt-in: classify a usage-limit refusal (CB-578)
|
||||
# credentialId: shared-openai # opt-in: quarantine together with every other profile sharing this id (CB-578)
|
||||
# configDir: /Users/me/.ccs/instances/gx10 # CLAUDE_CONFIG_DIR — inherit that profile's skills/MCP
|
||||
# cwd: /Users/me/src/myrepo # pin the working dir; omit to inherit the primary's
|
||||
# parityOverlay: [".claude/settings.local.json", ".env", ".envrc"] # never add .mcp.json — see above
|
||||
gx11: # a second backend, so `placement: weighted` has a choice
|
||||
baseUrl: http://gx01.gw:8000 # self-hosted; ccs handles the model + token
|
||||
placement: tab
|
||||
workspace: bridged-workers
|
||||
tabLabel: "worker: {profile} #{n}"
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
argv: ["ccs", "ollama"]
|
||||
defaultWorker: gx10
|
||||
argv: ["ccs", "gx11"]
|
||||
weight: 0.5
|
||||
maxLoad: 2
|
||||
# Pin an auto-compact window BELOW the served model's context ceiling. The global
|
||||
# ~/.claude/settings.json value is shared by every ccs instance and the primary, so the
|
||||
# per-profile override belongs here. Equal to the ceiling means auto-compact never fires
|
||||
# before the server rejects the prompt, which kills a worker mid-turn (CB-523).
|
||||
env:
|
||||
CLAUDE_CODE_AUTO_COMPACT_WINDOW: "280000"
|
||||
# CB-402: a second coding-agent kind, proving the PeerLauncher SPI is provider-neutral.
|
||||
# opencode is provider-agnostic and uses NONE of Claude's private seams: no ANTHROPIC_BASE_URL /
|
||||
# SubscriptionGuard (so it needs no `guard` host entry), no --mcp-config / --append-system-prompt.
|
||||
# The bridge MCP + reply charter mount via a generated OPENCODE_CONFIG file, and the model is a
|
||||
# `provider/model` selector. Placement, tabs, cwd, and the readiness gate are shared with Claude.
|
||||
#
|
||||
# Dogfood-verified 2026-07-29 against opencode 1.18.5 (spawn → readiness gate → bridge_send →
|
||||
# structured bridge_reply → teardown). The `opencode/*-free` models run on opencode's own gateway
|
||||
# and need NO credentials — check `opencode models` for the current free list, since the names
|
||||
# change. That also makes the worker off-subscription by construction.
|
||||
# opencode-free:
|
||||
# kind: opencode
|
||||
# model: opencode/north-mini-code-free # `provider/model` selector, injected as `-m`
|
||||
# placement: tab
|
||||
# workspace: bridged-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
#
|
||||
# CB-508: point an opencode profile at your OWN OpenAI-compatible endpoint (local vLLM, llama.cpp,
|
||||
# LM Studio, TGI…) instead of opencode's gateway. Setting `baseUrl` on a `kind: opencode` profile
|
||||
# makes the bridge emit a custom `provider` block into the generated opencode.json — opencode has
|
||||
# no ANTHROPIC_BASE_URL seam, so this is how the endpoint is pinned.
|
||||
# baseUrl → a bare host:port gets `/v1` appended (where these servers mount the API); a URL that
|
||||
# already has a path is used verbatim, so a custom mount point still works.
|
||||
# model → MUST be "<provider>/<model>". The provider half names the generated block; the model
|
||||
# half must match an id the server reports at /v1/models. One field drives both the
|
||||
# declaration and the `-m` flag, so they cannot drift apart. A bare model name with a
|
||||
# baseUrl set is rejected at spawn rather than silently using the default gateway.
|
||||
# tokenEnv → optional; its value becomes the provider apiKey. Most local servers ignore the key,
|
||||
# so a placeholder is used when unset (the AI SDK still requires a non-empty one).
|
||||
# NOTE: no `guard` entry is needed even with a baseUrl set. The SubscriptionGuard exists to stop a
|
||||
# worker borrowing the primary's Anthropic subscription, and an opencode process has no Anthropic
|
||||
# credential path at all.
|
||||
# opencode-local:
|
||||
# kind: opencode
|
||||
# baseUrl: http://127.0.0.1:8000
|
||||
# model: local-vllm/deepseek-v4-flash
|
||||
# placement: tab
|
||||
# workspace: bridged-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
# How an unqualified spawn chooses a profile: fixed (default, reproduces pre-CB-518 behaviour),
|
||||
# round-robin, or weighted. Omitting this key is a strict no-op for existing configs.
|
||||
#
|
||||
# `weighted` IS NOT "cheapest first" — read this before you set weights (CB-589).
|
||||
# It is smooth weighted round-robin: it spreads spawns across EVERY profile that has a free slot,
|
||||
# in weight ratio. It has no idea which profile costs money. So with local:10 / paid:2 you do not
|
||||
# get "use local, overflow to paid" — you get roughly one spawn in six going to the paid profile
|
||||
# while the local box still has a free slot.
|
||||
#
|
||||
# There is a sharper second effect. The policy's running score map lives for the daemon's whole
|
||||
# life. While a profile is at maxLoad it is filtered out and its score FREEZES, so the paid
|
||||
# profiles keep accumulating against it. When the local slot frees up it returns with a stale
|
||||
# score and can LOSE the next pick — a paid spawn while the free box sits idle.
|
||||
#
|
||||
# Until a real cost-first policy exists, the workaround is to make the ratio decisive rather than
|
||||
# proportional: give the free profile a weight so large that it wins every pick it is eligible
|
||||
# for, and paid profiles only ever take genuine overflow. On this host that is local weight 100
|
||||
# against paid weights of ~1.
|
||||
#
|
||||
# The gotcha with that workaround: it expresses a PREFERENCE ORDER through a RATIO knob. Add a
|
||||
# future profile at weight 150 and it silently outranks the free box, with nothing to warn you.
|
||||
# Re-check the weights whenever you add a profile.
|
||||
placement: weighted
|
||||
|
||||
# How long a credential sits out after a BACKEND_EXHAUSTED classification (CB-578 stage B), in
|
||||
# seconds, before a spawn may land on it again. Applies to every profile's effective credential
|
||||
# (its own name, or its credentialId if set above) — there is no per-profile override. Default
|
||||
# 1800 (30 minutes) when omitted or non-positive.
|
||||
# DEFERRED: baked once into the BackendQuarantine built at startup — a running quarantine keeps
|
||||
# its original cooldown regardless; a new value only applies to a quarantine that starts after a
|
||||
# restart. Editing this needs a daemon restart to take effect.
|
||||
# quarantineCooldownSeconds: 1800
|
||||
|
||||
# Re-read this file without restarting the daemon (CB-559). Off unless you add this block, so an
|
||||
# upgraded bridged keeps the old behaviour: the file is read once at boot and never again.
|
||||
# enabled → turn the watch on. bridged checks the file's modified time on a timer and
|
||||
# reloads when it moves.
|
||||
# intervalSeconds → how often to check (default 10). One `stat` per tick, so this is cheap.
|
||||
#
|
||||
# Not every key can move under a running daemon, and the difference is about what already exists
|
||||
# when the reload happens — not about how important the key is:
|
||||
# HOT → takes effect on the next spawn: the whole `fleet:` block (every role pool,
|
||||
# `charters`, and `tabLabel`), `placement:`, and an existing profile's weight / maxLoad
|
||||
# / credentialId. Those are hot because the placement policy (and, for credentialId,
|
||||
# the CB-578 stage B quarantine check) reads them through a supplier — being config is
|
||||
# not by itself enough to make a key hot.
|
||||
# EXCEPT `fleet.leaders`: Bridged.main reads it once at startup to build the lead tab
|
||||
# scanner and launcher, and neither is rebuilt on reload. A changed/added/removed
|
||||
# `fleet.leaders` entry is silently accepted — the reload reports "config reloaded"
|
||||
# with nothing in the deferred list — but has NO effect until you restart. Treat it
|
||||
# as deferred in practice, even though today's reload output does not say so.
|
||||
# DEFERRED → accepted into the new config, but the wiring built at startup keeps the old value
|
||||
# until you restart: `lifecycle:`, `leadHeartbeat:`, `guard:`, `worktreeRoot:`,
|
||||
# `spawnReadyTimeoutMs` / `spawnReadyPollMs`, `quarantineCooldownSeconds` (CB-578
|
||||
# stage B — baked once into the quarantine tracker built at startup), ADDING or
|
||||
# REMOVING a profile (a new backend needs its own launcher, and launchers are built
|
||||
# once), AND an existing profile's launch settings — model, baseUrl, argv, env,
|
||||
# configDir, mcpUrl, tabLabel, exhaustedPattern. The launcher takes a copy of
|
||||
# `profiles:` at startup and resolves every spawn out of that copy, so those never
|
||||
# reach a launch until you restart. The reload logs them by name rather than
|
||||
# pretending they applied.
|
||||
# COLD → cannot change at all: `bind:`, `herdrSocket:`, `broker:` and `auth:`. The socket is
|
||||
# bound, the broker connection is open, and the auth mode decides who may reach the
|
||||
# port that is already listening.
|
||||
#
|
||||
# A changed COLD key refuses the WHOLE reload — not the hot half applied and the cold half warned
|
||||
# about. A half-applied reload would leave the daemon matching no file on disk, which is the worst
|
||||
# thing a reload can do to an operator debugging one. A file that fails to parse or fails a startup
|
||||
# validator is refused the same way, and the running config stays live.
|
||||
# configReload:
|
||||
# enabled: true
|
||||
# intervalSeconds: 10
|
||||
|
||||
# THE FLEET (CB-557) — who the daemon may run, and under which role. This one block replaced four
|
||||
# older keys: `leaders:`, `members:`, `leadScan:` and `defaultProfile:`.
|
||||
#
|
||||
# A member is anything a lead spawns, and every member has two INDEPENDENT attributes:
|
||||
# role — which contract: architect, dev or reviewer. It picks the launch charter, the role
|
||||
# file, the playbook skill and the authz row.
|
||||
# profile — which backend: one of the `profiles:` keys above (model, CLI adapter, cost).
|
||||
# They vary on their own. A reviewer may run on the same profile as the dev whose diff it reads,
|
||||
# which is why the two cannot be one field.
|
||||
#
|
||||
# The ROLE IS THE CONTAINING KEY, not a `role:` field. That is not only tidier: a misspelled role
|
||||
# used to parse into a member with no contract at all, while a misspelled pool name here simply
|
||||
# declares nothing.
|
||||
#
|
||||
# Each pool lists the profiles that role MAY run on — these are pools, not identities. That is also
|
||||
# what replaced `defaultProfile:`: an unqualified spawn names a role, and that role's pool supplies
|
||||
# the candidates, in definition order. A dev and a reviewer staying anonymous is exactly compatible
|
||||
# with being listed here; the entry key just names the entry.
|
||||
fleet:
|
||||
# Optional launch-charter text, keyed only by the singular role wire names: architect, dev,
|
||||
# reviewer. Changes are HOT and reach the next spawn without a daemon restart. Do not put secrets
|
||||
# here: a later launch step writes this text to a world-readable temp file, and ${ENV} interpolation
|
||||
# is deliberately not supported.
|
||||
charters:
|
||||
architect: |-
|
||||
You are an architect in this fleet. You refine work before anyone builds it:
|
||||
scope, acceptance criteria, risks, and a unit split. You read the repo and
|
||||
write analysis. You never commit production code and never open a PR.
|
||||
A design task is worked by two architects. Design alone first, then exchange
|
||||
and say plainly where you disagree. Do not concede just to agree.
|
||||
dev: |-
|
||||
You implement the one unit you were given, and nothing else. You test it,
|
||||
commit it, and open your own pull request. You never merge.
|
||||
reviewer: |-
|
||||
You review the diff you were given. You report bugs, risks and missing tests.
|
||||
You do not change code.
|
||||
|
||||
# Optional. Template for a member tab's label; {role}, {profile}, {model} and {n} are substituted.
|
||||
# {n} counts per role+profile, so `dev: sonnet #2` really is the second sonnet dev. Because {role}
|
||||
# comes from a closed enum, a generated label can never begin with a lead's tabPrefix.
|
||||
# tabLabel: "{role}: {profile} #{n}"
|
||||
|
||||
# Panes that orchestrate rather than are orchestrated. A lead may now be CREATED as well as
|
||||
# recognised: give it a `profile:` and the daemon launches the shortfall when fewer than
|
||||
# `instances` are live. Omit `profile:` and it is recognise-only, as before.
|
||||
#
|
||||
# `tab:` (CB-579) is REQUIRED and is the only field identity depends on — the exact label of the
|
||||
# tab hosting the lead, matched case-insensitively. Label the tab yourself and put that same
|
||||
# string here, and the pane is recognised on the next rescan. Reopen the tab later, or the session
|
||||
# inside it restarts — the terminal id changes; the tab, and its label, do not, so no config edit
|
||||
# follows a restart.
|
||||
#
|
||||
# A lead the daemon launches is labelled BY the daemon with this same `tab:` value, so it is found
|
||||
# by the same scan. A lead counts as live only when herdr also reports a running agent in that
|
||||
# tab — a label left behind by a session that died does not block the relaunch, and a tab that is
|
||||
# gone entirely drops out of the next scan rather than being remembered forever.
|
||||
#
|
||||
# An auto-launched lead is NOT a member: it gets no worker reply charter, is never registered with
|
||||
# the session lifecycle (the idle reaper would kill your orchestrator), and stays on the
|
||||
# subscription — ANTHROPIC_BASE_URL/AUTH_TOKEN are stripped from its env whatever the profile says.
|
||||
#
|
||||
# GET THE `tab:` VALUE RIGHT. A pane that does not match any configured `tab:` (a typo, a renamed
|
||||
# tab, a pane no entry names at all) is not recognised as a lead — it resolves as an ordinary
|
||||
# WORKER instead, silently, and every orchestration call it makes (spawn/stop/send/drain) is
|
||||
# refused. There is no error at startup for this: an unmatched pane is simply not a lead. If your
|
||||
# primary suddenly can't spawn or send, check this section first.
|
||||
# leaders:
|
||||
# opus-5.0:
|
||||
# profile: opus # omit to never create this lead, only recognise it
|
||||
# instances: 1 # desired live count; only the shortfall is launched. 0 = off
|
||||
# tab: "lead: opus-5.0" # REQUIRED — the exact tab label this lead lives in
|
||||
# tabPrefix: "lead:" # only used to guard against a worker tabLabel colliding with
|
||||
# # this convention at startup; plays no part in matching a lead
|
||||
# scanIntervalSeconds: 10 # rescan cadence, and the worst case before a new tab is seen
|
||||
# workspace: leads # where a launched lead's tab is created (default "leads").
|
||||
# # MUST NOT be a member workspace — those are excluded from the
|
||||
# # scan, so a lead placed in one is never found again.
|
||||
# cwd: /path/to/repo # the launched lead's working directory (default: bridged's own)
|
||||
# kind: claude # descriptive; reported by bridge_whoami
|
||||
# gpt-sol-5.6:
|
||||
# tab: "lead: gpt-sol-5.6"
|
||||
# kind: opencode
|
||||
# model: openai/gpt-5.6-terra
|
||||
|
||||
# architects:
|
||||
# architect-1:
|
||||
# profile: opus # a strong model, on the operator's subscription
|
||||
# architect-2:
|
||||
# profile: sol # a different vendor on purpose — two architects that share a
|
||||
# # model share its blind spots
|
||||
developers:
|
||||
gx10:
|
||||
profile: gx10
|
||||
# reviewers:
|
||||
# gx10:
|
||||
# profile: gx10 # the same backend may serve two roles; that is the point
|
||||
|
||||
# Subscription boundary. A worker's base_url host MUST be one of these; the primary
|
||||
# must carry none. Every profile above must have its host listed here.
|
||||
@@ -52,14 +472,154 @@ guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
- gx01.gw
|
||||
- ollama.ltms.dev
|
||||
|
||||
# Member credential policy (CB-596, gitea issue #82). A herdr pane runs a LOGIN shell, and that
|
||||
# shell re-sources the operator's own secret store — so a spawned member inherits every credential
|
||||
# the operator's shell holds, not just the ones bridged means to give it. Measured on this host:
|
||||
# 31 credential names, all set, with only ONE (GITEA_ACCESS_TOKEN) blocked before this — and that
|
||||
# block was a single name hardcoded in HerdrPeerLauncher.java, not driven by this file. This block
|
||||
# replaces that hardcoded shadow with a config-driven list of names.
|
||||
#
|
||||
# ROUND-2 CORRECTION, measured live: the pane-creation env overlay below (applied at tab.create /
|
||||
# pane.split, BEFORE the pane's login shell runs) does NOT survive that login shell for any name
|
||||
# secrets.sh actually exports — the shell re-exports it afterwards and overwrites the sentinel.
|
||||
# Proof: GITEA_ACCESS_TOKEN comes back blocked only because secrets.sh itself carries a guarded
|
||||
# export (`[ -n "${BRIDGED_MEMBER:-}" ] || export GITEA_ACCESS_TOKEN=...`) — that guard, not this
|
||||
# file, is what wins. No other name in `known` below has a matching guard in secrets.sh yet (1
|
||||
# guard measured against 33 export lines there). So today this block's overlay is REAL protection
|
||||
# only for a name secrets.sh does not export, or a peer kind whose pane never runs a login shell —
|
||||
# for everything secrets.sh exports and guards, the guard in secrets.sh (out of scope for this
|
||||
# ticket) is what actually blocks it, not this list. An exec-time fix (winning after the login
|
||||
# shell finishes, before the agent process starts) was attempted and found to have no seam in the
|
||||
# current herdr protocol — AgentControl.start takes a fixed `kind` (herdr resolves the executable)
|
||||
# plus trailing CLI args for that binary, not an arbitrary argv or an env map; only tab.create /
|
||||
# pane.split accept `env`, and that is this same pane-creation overlay. See gitea #82 for the open
|
||||
# design question this leaves.
|
||||
#
|
||||
# DENY-BY-DEFAULT, NOT A DENY-LIST. A deny-list (name the bad ones, let everything else through) is
|
||||
# silently wrong the moment the operator's store gains a new secret — nothing would ever report it.
|
||||
# Deny-by-default inverts that: `known` bounds the blast radius to names actually enumerated below,
|
||||
# and EVERY one of them is blocked UNLESS it is also in `allow`. Omitting this block entirely (the
|
||||
# shipped default) blocks NOTHING — unlike most optional blocks in this file, absence here is a real
|
||||
# gap, not a safe "feature off". A name that is neither `known` nor `allow`-ed is not silently let
|
||||
# through either: the daemon logs a WARN naming any credential-shaped env var it finds on neither
|
||||
# list (never its value), so a secret added to the store later does not go unnoticed forever.
|
||||
#
|
||||
# policy → only "deny-by-default" exists today (an operator-authored deny-list was deliberately
|
||||
# rejected — see above). An unrecognized value refuses to start, naming it.
|
||||
# allow → credential names a member legitimately needs. Left OUT of the pane's env overlay
|
||||
# entirely, so the value the pane's own (login) shell exports passes through untouched.
|
||||
# known → every credential name the operator's store is known to export. Every name here NOT
|
||||
# also in `allow` is overlaid with a non-secret sentinel value before the pane's login
|
||||
# shell runs — real protection only for names the login shell does not itself re-export
|
||||
# (see the ROUND-2 CORRECTION note above for the ones it does).
|
||||
#
|
||||
# HOT-RELOADABLE the same way `fleet:` is (CB-559): read fresh on every spawn, so editing this list
|
||||
# and reloading config (or restarting) changes what the NEXT spawn inherits; already-running members
|
||||
# are unaffected either way.
|
||||
# memberCredentials:
|
||||
# policy: deny-by-default
|
||||
# allow:
|
||||
# - AI_GATEWAY_TOKEN # named in a profile's tokenEnv (local/gx) — a member reaching the
|
||||
# # gateway is by design, not a leak
|
||||
# - WORKER_GITEA_TOKEN # the repo-scoped forge token a member needs to open its own PR (CB-302)
|
||||
# - CONTEXT7_TOKEN # already decided as allowed by CB-593
|
||||
# - GITEA_HOST # not a credential — a hostname, paired with the forge token above
|
||||
# known:
|
||||
# - AI_GATEWAY_TOKEN
|
||||
# - BESZEL_ADMIN_EMAIL
|
||||
# - BESZEL_ADMIN_PASSWORD
|
||||
# - BESZEL_HUB_URL
|
||||
# - BESZEL_KEY
|
||||
# - BESZEL_UNIVERSAL_TOKEN
|
||||
# - BRAIN_MCP_TOKEN
|
||||
# - CF_ACCOUNT_ID
|
||||
# - CF_API_TOKEN
|
||||
# - CF_USER_TOKEN
|
||||
# - CONFLUENCE_API_TOKEN
|
||||
# - CONFLUENCE_USERNAME
|
||||
# - CONTEXT7_TOKEN
|
||||
# - GITEA_ACCESS_TOKEN
|
||||
# - GITEA_HOST
|
||||
# - GITLAB_OAUTH_CLIENT_SECRET
|
||||
# - GITLAB_PERSONAL_ACCESS_TOKEN
|
||||
# - GRAFANA_ADMIN_PASSWORD
|
||||
# - GRAFANA_ADMIN_USER
|
||||
# - HASS_TOKEN
|
||||
# - HW_PASSWORD
|
||||
# - HW_USER
|
||||
# - LTMS_API_KEY
|
||||
# - MEMORY_MCP_TOKEN
|
||||
# - METRICS_PUSH_TOKEN
|
||||
# - OPENCODE_AUTOMODE_MODEL
|
||||
# - TELEGRAM_BOT_TOKEN
|
||||
# - TELEGRAM_CHAT_ID
|
||||
# - TS_API_KEY
|
||||
# - TS_AUTHKEY
|
||||
# - WORKER_GITEA_TOKEN
|
||||
|
||||
# Spawn-readiness gate (CB-306). The launcher blocks until the worker's herdr status is
|
||||
# injectable (IDLE/BLOCKED/DONE) or the timeout elapses. 0 disables the gate.
|
||||
# NOTE: keys are camelCase — config is bound by plain Jackson with no naming strategy and
|
||||
# unknown keys are ignored, so a snake_case key would be silently dropped (default kept).
|
||||
# spawnReadyTimeoutMs: 20000
|
||||
# spawnReadyPollMs: 300
|
||||
|
||||
# Worktree provisioning root (CB-301-ext). Where per-worker git worktrees are checked out so
|
||||
# each worker owns an isolated branch instead of sharing the primary's tree. Omit to default
|
||||
# to a sibling directory of the repo root.
|
||||
# worktreeRoot: /Users/me/src/.bridged-worktrees
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
# the feature disabled. By default the daemon never reaps, caps, or drains sessions.
|
||||
# idleTtlSeconds → reap READY/DONE sessions idle longer than this (never BUSY/SPAWNING)
|
||||
# contextCap → force-release a session after this many delegated turns
|
||||
# drainTimeoutSeconds → seconds to wait for BUSY sessions on shutdown before forced teardown
|
||||
# clearAfterTurn → whether a reusable worker discards its conversation context after every
|
||||
# completed delegated turn (default false). Works for claude-code workers
|
||||
# only — any other peer kind (e.g. opencode) logs "context reset is
|
||||
# unsupported for peer kind …" once and the reset is a no-op.
|
||||
# lifecycle:
|
||||
# idleTtlSeconds: 300
|
||||
# contextCap: 10
|
||||
# drainTimeoutSeconds: 5
|
||||
# clearAfterTurn: false
|
||||
|
||||
# Durable reply delivery (CB-307 Stage 2). OMIT this block entirely to keep the default
|
||||
# in-memory, soft-state reply inbox (late worker replies are held only until a daemon bounce).
|
||||
# Set a broker uri to swap in the AMQP-backed inbox: worker replies with no open send are held
|
||||
# on a durable per-target queue (agent.<target>.inbox) and survive a restart — the broker
|
||||
# redelivers anything the primary had not yet drained. Production default is LavinMQ; a stock
|
||||
# RabbitMQ speaks the same AMQP 0-9-1, so it is a URI-only swap.
|
||||
# uri → AMQP connection URI. No trailing slash ⇒ the default vhost "/"; an empty path ("/")
|
||||
# is vhost "" and will NOT connect. Encode a named vhost as .../%2Fmyvhost.
|
||||
# prefetch → CB-527: consumer basicQos, capping how many unacked messages the inbox holds
|
||||
# in-heap per owned target (the rest sits on the broker's durable queue instead of
|
||||
# growing the JVM heap). Default 32 when omitted.
|
||||
# broker:
|
||||
# uri: amqp://guest:guest@127.0.0.1:5672
|
||||
# prefetch: 32
|
||||
|
||||
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open bridge_send,
|
||||
# the ReplyPushLoop injects a *drain nudge* (never the payload) into the primary's own herdr
|
||||
# pane — status-gated (only when injectable, never mid-turn) and bounded. Ack = drain: the loop
|
||||
# stops as soon as the primary's inbox is empty.
|
||||
# terminal → pin the primary's herdr terminal id. Omit to learn it from the connection on
|
||||
# the first orchestration-side MCP call (the normal case). An off-host or
|
||||
# non-herdr primary leaves this unresolved → the loop is a no-op and delivery
|
||||
# degrades to pull; the reply is still never lost.
|
||||
#
|
||||
# REQUIRED (CB-522) if the primary itself runs inside a herdr pane. Caller
|
||||
# identity resolves a loopback PID to its herdr pane, and PaneLocator scans
|
||||
# EVERY pane — not just bridged-spawned ones — so such a primary is otherwise
|
||||
# classified as a WORKER and refused SPAWN/SEND/STOP. That failure is
|
||||
# self-locking: the learned terminal is populated by the very orchestration
|
||||
# calls being refused, so only this pinned value can break the cycle. Read the
|
||||
# id off bridge_whoami (it reports the current terminal even while
|
||||
# misclassified) and re-pin whenever the primary moves panes.
|
||||
# pushReminders → max nudges before giving up (default 5)
|
||||
# pushBackoffMs → delay between nudges in ms (default 15000)
|
||||
# primary:
|
||||
# terminal: term_65619bd6174568
|
||||
# pushReminders: 5
|
||||
# pushBackoffMs: 15000
|
||||
|
||||
@@ -0,0 +1,142 @@
|
||||
# CB-307 — Active push-to-primary + reminder loop (the reliability layer)
|
||||
|
||||
**Status:** design (2026-07-19). Builds directly on the shipped durable landing zone
|
||||
(`AmqpReplyInbox`, main `2bc5f3a`, dogfooded live). gitea #5.
|
||||
|
||||
## Why this exists
|
||||
|
||||
Stage 2 gave a worker→primary reply a **durable place to wait** when no `bridge_send` is
|
||||
open: it lands in `agent.<target>.inbox` on the broker and survives a daemon bounce. But
|
||||
delivery is still **pull** — the primary only sees the reply if it happens to call
|
||||
`bridge_poll(target)` / `GET /sessions/{id}/replies`. A reply can sit indefinitely while
|
||||
the primary works on something else.
|
||||
|
||||
This layer makes delivery **active**: the bridge *pushes* a nudge to the primary the moment
|
||||
a reply lands, and keeps reminding (bounded) until the primary drains it. At-least-once,
|
||||
dedup by `msgId`, and — critically — it never loses the reply even if every push fails,
|
||||
because the durable inbox is the backstop.
|
||||
|
||||
## The hard constraint it works around
|
||||
|
||||
The bridge is an MCP **server**; the primary is an MCP **client**. A server cannot call
|
||||
into a client. So "push to the primary" cannot be an MCP response — it needs a *sideband*
|
||||
channel. The chosen channel: **inject a synthetic user-turn into the primary's own herdr
|
||||
terminal pane** — the same mechanism the bridge already uses to deliver tasks to workers,
|
||||
pointed at the primary's pane instead.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
W["worker"] -->|"bridge_reply (no open send)"| MS["MessageService.reply"]
|
||||
MS -->|"inbox.publish"| INBOX[("agent.<target>.inbox<br/>(durable, LavinMQ)")]
|
||||
MS -->|"notify"| LOOP["ReplyPushLoop"]
|
||||
LOOP -->|"status-gated inject"| PANE["primary's herdr pane"]
|
||||
PANE -->|"primary drains"| DRAIN["bridge_poll(target)<br/>= peek + ack"]
|
||||
DRAIN -->|"inbox now empty"| LOOP
|
||||
LOOP -.->|"still non-empty →<br/>re-inject on backoff"| PANE
|
||||
classDef store fill:#2c5282,stroke:#1a365d,color:#ffffff;
|
||||
class INBOX store
|
||||
```
|
||||
|
||||
*Figure 1 — a reply lands in the durable inbox; the push loop nudges the primary's pane;
|
||||
the primary's drain acks it; a still-full inbox triggers a bounded re-nudge.*
|
||||
|
||||
## Three increments
|
||||
|
||||
### Increment 1 — learn & store the primary's terminal_id
|
||||
|
||||
**Finding (seam map):** `ConnectionIdentity.resolve(remoteAddr, remotePort)` already returns
|
||||
the caller's herdr `terminal_id` for *every* MCP call, via `PaneLocator.terminalForPid`
|
||||
(walks `pane.list`, matches the caller PID to a pane's process tree). It is non-null whenever
|
||||
the caller runs in a herdr pane on this host. Today it's discarded for the primary
|
||||
(`presence.markPresent` is a no-op on it).
|
||||
|
||||
**Plan:** a single-slot `PrimaryRegistry` (thread-safe) holding the primary's `terminal_id`.
|
||||
Populate it from the **orchestration-side** MCP tools — `bridge_send`, `bridge_spawn`,
|
||||
`bridge_poll`, `bridge_list`, `bridge_status`, `bridge_profiles` — capturing
|
||||
`callerTerminal(exchange)` when it is (a) non-null and (b) **not** a registered worker
|
||||
session in `SessionManager`. That caller is, by construction, the primary. Worker-side tools
|
||||
(`bridge_reply`, `bridge_ask`) never set it.
|
||||
|
||||
- **Config override / pin:** a `primary: { terminal: "<id>" }` block in `BridgedConfig`
|
||||
(nested record, same shape as `Broker`). Lets an operator pin it, or supply it when
|
||||
derivation can't (see degrade case).
|
||||
- **Degrade:** if the primary is off-host or in a non-herdr terminal, `terminalForPid`
|
||||
returns null and no override is set → **the registry stays empty → the push loop is a
|
||||
no-op and we fall back to pull** (today's behaviour). The reply is never lost; it's just
|
||||
not actively pushed. This is a safe, explicit degradation, not a failure.
|
||||
|
||||
### Increment 2 — the push loop
|
||||
|
||||
A `ReplyPushLoop` component, notified at the single no-waiter call site
|
||||
(`MessageService.reply` → the `inbox.publish` branch, `MessageService.java:192`).
|
||||
|
||||
- **Inject a nudge, not the payload.** The injected turn tells the primary *to drain*
|
||||
(e.g. "Worker `<target>` returned a reply — run `bridge_poll(target=<target>)` to collect
|
||||
it"), it does **not** carry the reply text. Rationale: replies can be large/multiline and
|
||||
terminal injection would mangle them; the drain response is the clean transport. Keeps the
|
||||
push idempotent — re-nudging is harmless.
|
||||
- **Ack = drain.** The primary draining (`drainReplies` = peek + ack) is the acknowledgement.
|
||||
The loop's **stop condition is `inbox.peek(target).isEmpty()`** — the reply is gone from the
|
||||
inbox because it was acked. No new `bridge_ack` tool needed for v1 (see Increment 3).
|
||||
- **Status-gated injection (mechanism (b), chosen).** A dedicated lightweight scheduled loop,
|
||||
**not** the worker `Injector`. It injects via `AgentControl.send(primaryTerminal, nudge)`
|
||||
(the same herdr `agent.send` = `pane send-text` + submit that delivers to workers) only when
|
||||
`AgentControl.status(primaryTerminal).injectable()` (IDLE/BLOCKED) — never mid-turn. This keeps
|
||||
the primary path fully isolated from `WorkerPresence`/`StatusPoller` (which are worker-scoped),
|
||||
and makes it unit-testable with a fake `AgentControl` + an injected clock (per the CB-306
|
||||
`LongSupplier` clock + `Runnable` sleeper seam). Rejected (a) reuse-the-Injector: it would force
|
||||
the primary terminal into the worker poller set and couple to worker-presence semantics — more
|
||||
integration surface, harder to test, no real gain for a bounded reminder.
|
||||
- **Bounded reminder / backoff.** While `peek(target)` stays non-empty, re-inject on a
|
||||
backoff schedule up to a cap (N reminders or a max duration; config
|
||||
`primary.push_reminders` / `primary.push_backoff_ms`). After the cap, **stop reminding** —
|
||||
the reply remains in the durable inbox and the next natural poll (or a later worker reply's
|
||||
nudge) still surfaces it. Bounded so the bridge never spams the primary.
|
||||
|
||||
### Increment 3 — optional per-`msgId` `bridge_ack` tool (deferred)
|
||||
|
||||
Drain-as-ack is coarse: it clears *all* pending replies for a target at once. If finer
|
||||
control is ever needed (ack one reply, leave others held), add a `bridge_ack(msgId)` tool
|
||||
mapping to `inbox.ack(target, msgId)` — the port already supports per-`msgId` ack. Not built
|
||||
in v1; the stop-on-empty loop is sufficient.
|
||||
|
||||
## The two subtleties (decided here)
|
||||
|
||||
1. **Which caller is "the primary"?** Connection-derived, not self-reported: the caller whose
|
||||
resolved terminal is non-null **and not a registered worker session**, seen on an
|
||||
orchestration-side tool. This never mislabels a worker (workers are in `SessionManager`)
|
||||
and needs no new env var or argument (identity stays connection-derived, per the existing
|
||||
`BridgeMcp` invariant).
|
||||
|
||||
2. **Readiness-gate mismatch → dedicated loop.** The existing `Injector` gates delivery on
|
||||
`ready.test(target)` = `WorkerPresence` (the *worker's* MCP connected). The primary is not
|
||||
in `WorkerPresence`, so reusing `Injector` would mean forcing the primary terminal into the
|
||||
worker `StatusPoller` set and swapping the `ready` predicate — extra integration surface with
|
||||
worker-scoped machinery. Decision: **mechanism (b)** — a small dedicated scheduled loop that
|
||||
calls `AgentControl.status(primaryTerminal).injectable()` then `AgentControl.send(...)`, with
|
||||
an injected clock. Isolated from worker presence, trivially unit-testable, sufficient for a
|
||||
bounded reminder. (Verified live: this primary resolves to `term_656c8cc03e1f0b1`, pane
|
||||
`w2:pY` — the primary genuinely runs in a herdr pane on this host, so the path is exercisable.)
|
||||
|
||||
## Boundary note
|
||||
|
||||
This is the first time the bridge **writes into the primary's pane** — a new direction of
|
||||
control. It stays within the communication-bus identity: the injection is a **nudge** (a
|
||||
synthetic "go drain your replies" turn), **status-gated** so it never interrupts a turn,
|
||||
**bounded** so it never spams, carries **no env** and **never crosses the subscription
|
||||
boundary**. The bridge is signalling the primary that it has mail — not driving its work.
|
||||
|
||||
## Test plan
|
||||
|
||||
- **Unit (hermetic):** `PrimaryRegistry` set/clear/override; the "caller is primary iff
|
||||
non-null terminal AND not a registered session" predicate; the loop's stop-on-empty and
|
||||
bounded-reminder logic with an injected clock + a fake injector (no real herdr).
|
||||
- **Live dogfood (primary-side):** with the daemon on the broker jar + a real worker,
|
||||
delegate a task, let the worker reply after the `bridge_send` window closes, and observe the
|
||||
bridge inject a drain nudge into *this* primary pane; confirm draining stops the reminders;
|
||||
confirm an unreachable primary (registry empty) degrades to pull with no loss.
|
||||
|
||||
## Out of scope
|
||||
|
||||
Multi-host (CB-308) — the push loop is local-only; a remote primary is reached by its own
|
||||
local gateway, not cross-host injection. Federation reuses this loop per-gateway.
|
||||
+91
-1
@@ -6,7 +6,7 @@
|
||||
|
||||
<groupId>dev.ltms</groupId>
|
||||
<artifactId>bridged</artifactId>
|
||||
<version>0.1.0-SNAPSHOT</version>
|
||||
<version>1.0.0</version>
|
||||
<packaging>jar</packaging>
|
||||
|
||||
<name>bridged</name>
|
||||
@@ -24,6 +24,10 @@
|
||||
<slf4j.version>2.0.16</slf4j.version>
|
||||
<logback.version>1.5.18</logback.version>
|
||||
<junit.version>5.11.4</junit.version>
|
||||
<amqp.version>5.22.0</amqp.version>
|
||||
<testcontainers.version>1.20.4</testcontainers.version>
|
||||
<commons-compress.version>1.27.1</commons-compress.version>
|
||||
<commons-lang3.version>3.18.0</commons-lang3.version>
|
||||
</properties>
|
||||
|
||||
<!--
|
||||
@@ -62,6 +66,22 @@
|
||||
<artifactId>jackson-annotations</artifactId>
|
||||
<version>3.0-rc5</version>
|
||||
</dependency>
|
||||
<!-- Testcontainers 1.20.4 pulls commons-compress 1.24.0 (test scope), which carries
|
||||
CVE-2024-25710 (8.1) + CVE-2024-26308 — both fixed in 1.26.0. Pin the patched line.
|
||||
Test-scope only (never shipped in the jar), but bumped per the CVE policy. -->
|
||||
<dependency>
|
||||
<groupId>org.apache.commons</groupId>
|
||||
<artifactId>commons-compress</artifactId>
|
||||
<version>${commons-compress.version}</version>
|
||||
</dependency>
|
||||
<!-- Testcontainers 1.20.4 also pulls commons-lang3 3.16.0 (test scope): CVE-2025-48924
|
||||
(uncontrolled recursion in ClassUtils), fixed in 3.18.0. Pin the patched line.
|
||||
Test-scope only (never shipped in the jar), bumped per the CVE policy. -->
|
||||
<dependency>
|
||||
<groupId>org.apache.commons</groupId>
|
||||
<artifactId>commons-lang3</artifactId>
|
||||
<version>${commons-lang3.version}</version>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
</dependencyManagement>
|
||||
|
||||
@@ -93,6 +113,16 @@
|
||||
<version>${mcp.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Broker client (CB-307 Stage 2): AMQP 0-9-1. Default deploy targets LavinMQ; this same
|
||||
client speaks to RabbitMQ unchanged (URI-only swap), so integration tests run against a
|
||||
stock RabbitMQ container. Only wired when a broker: block is present in config; absent →
|
||||
the in-memory ReplyInbox. -->
|
||||
<dependency>
|
||||
<groupId>com.rabbitmq</groupId>
|
||||
<artifactId>amqp-client</artifactId>
|
||||
<version>${amqp.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Logging -->
|
||||
<dependency>
|
||||
<groupId>org.slf4j</groupId>
|
||||
@@ -112,6 +142,22 @@
|
||||
<version>${junit.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
|
||||
<!-- Testcontainers RabbitMQ: spins a real broker for the @Tag("contract") AMQP integration
|
||||
test only. Excluded from the default build (contract group), so `mvn clean install`
|
||||
stays hermetic and green without Docker; run under -Pcontract with Docker present. -->
|
||||
<dependency>
|
||||
<groupId>org.testcontainers</groupId>
|
||||
<artifactId>rabbitmq</artifactId>
|
||||
<version>${testcontainers.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.testcontainers</groupId>
|
||||
<artifactId>junit-jupiter</artifactId>
|
||||
<version>${testcontainers.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
|
||||
<build>
|
||||
@@ -123,6 +169,30 @@
|
||||
<version>3.14.0</version>
|
||||
</plugin>
|
||||
|
||||
<!--
|
||||
Coverage (CB-509). Build-time tooling only — never a compile/runtime dependency, so
|
||||
it adds nothing to the shipped jar. Report lands at target/site/jacoco/index.html and
|
||||
target/site/jacoco/jacoco.csv. No `check` rule / threshold is wired: a coverage gate
|
||||
rewards writing tests that execute lines, which is the failure mode this project is
|
||||
trying to avoid, not encourage.
|
||||
-->
|
||||
<plugin>
|
||||
<groupId>org.jacoco</groupId>
|
||||
<artifactId>jacoco-maven-plugin</artifactId>
|
||||
<version>0.8.13</version>
|
||||
<executions>
|
||||
<execution>
|
||||
<id>prepare-agent</id>
|
||||
<goals><goal>prepare-agent</goal></goals>
|
||||
</execution>
|
||||
<execution>
|
||||
<id>report</id>
|
||||
<phase>test</phase>
|
||||
<goals><goal>report</goal></goals>
|
||||
</execution>
|
||||
</executions>
|
||||
</plugin>
|
||||
|
||||
<!-- Unit tests run by default; contract tests (live herdr) are tag-excluded. -->
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
@@ -166,6 +236,26 @@
|
||||
<profile>
|
||||
<id>contract</id>
|
||||
<properties><excludedGroups/></properties>
|
||||
<build>
|
||||
<plugins>
|
||||
<plugin>
|
||||
<artifactId>maven-surefire-plugin</artifactId>
|
||||
<configuration>
|
||||
<!-- Docker-engine compat (see "Running the contract tests" in
|
||||
docs/CB-307-Reliable-Delivery.md): Testcontainers 1.20.4's docker-java
|
||||
client defaults to Docker API 1.32 when no version is set, but modern
|
||||
engines (OrbStack on this dev host, min 1.40) reject that as too old —
|
||||
which surfaces as "Could not find a valid Docker environment". Pinning
|
||||
api.version=1.43 works on OrbStack and Docker 24+, and is overridable
|
||||
per-host via -Dapi.version. Only active under -Pcontract, so the
|
||||
default hermetic build never sets it. -->
|
||||
<systemPropertyVariables>
|
||||
<api.version>1.43</api.version>
|
||||
</systemPropertyVariables>
|
||||
</configuration>
|
||||
</plugin>
|
||||
</plugins>
|
||||
</build>
|
||||
</profile>
|
||||
</profiles>
|
||||
</project>
|
||||
|
||||
@@ -1,33 +1,71 @@
|
||||
package dev.ltms.bridged;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.config.ConfigRef;
|
||||
import dev.ltms.bridged.config.ConfigWatcher;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrClient;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.LeadTabScanner;
|
||||
import dev.ltms.bridged.lead.LeadLauncher;
|
||||
import dev.ltms.bridged.herdr.PaneLocator;
|
||||
import dev.ltms.bridged.herdr.UnixSocketHerdrClient;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.inject.CompletionResolver;
|
||||
import dev.ltms.bridged.inject.ExhaustedPatternLookup;
|
||||
import dev.ltms.bridged.inject.ExhaustionSink;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.inject.StatusPoller;
|
||||
import dev.ltms.bridged.inject.TurnListener;
|
||||
import dev.ltms.bridged.inject.WorkerPresence;
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import dev.ltms.bridged.auth.MemberRegistry;
|
||||
import dev.ltms.bridged.auth.CallerResolver;
|
||||
import dev.ltms.bridged.mcp.BridgeMcp;
|
||||
import dev.ltms.bridged.mcp.ConnectionIdentity;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.health.FleetHealthMonitor;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.mcp.LsofPeerPidLookup;
|
||||
import dev.ltms.bridged.mcp.LsofProcessCwdLookup;
|
||||
import dev.ltms.bridged.msg.AmqpReplyInbox;
|
||||
import dev.ltms.bridged.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.msg.ReplyInbox;
|
||||
import dev.ltms.bridged.msg.LeadHeartbeatLoop;
|
||||
import dev.ltms.bridged.msg.ReplyPushLoop;
|
||||
import dev.ltms.bridged.rest.BridgedApp;
|
||||
import dev.ltms.bridged.session.GitWorktrees;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.session.SessionReaper;
|
||||
import dev.ltms.bridged.worker.ClaudeCodeLauncher;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.bridged.member.CompositePeerLauncher;
|
||||
import dev.ltms.bridged.member.HerdrPeerLauncher;
|
||||
import dev.ltms.bridged.member.OpenCodeLauncher;
|
||||
import dev.ltms.bridged.placement.BackendQuarantine;
|
||||
import io.javalin.Javalin;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* {@code bridged} entry point. Wires the real herdr socket client to the REST app and
|
||||
@@ -38,17 +76,46 @@ public final class Bridged {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(Bridged.class);
|
||||
|
||||
/** How often the injector samples a busy worker's status while it has queued work. */
|
||||
private static final long INJECT_POLL_MILLIS = 250;
|
||||
/** CB-504: how long to wait at startup for herdr's socket before serving degraded. */
|
||||
private static final long HERDR_WAIT_SECONDS = 30;
|
||||
private static final long HERDR_WAIT_POLL_MILLIS = 500;
|
||||
|
||||
static void main(String[] args) {
|
||||
Path configPath = Path.of(args.length > 0 ? args[0] : "bridged.yaml");
|
||||
BridgedConfig cfg = BridgedConfig.load(configPath);
|
||||
// CB-594: report which secret env vars the config actually needs, by name, before anything
|
||||
// else can fail on a silently-empty one. A daemon started without a login shell (launchd)
|
||||
// boots fine either way — this is the only thing that says so out loud.
|
||||
reportRequiredSecrets(cfg);
|
||||
// CB-596: an absent (or empty) memberCredentials: block blocks NOTHING — no credential
|
||||
// name is hardcoded any more to fall back on. Say so loudly, the same way a missing
|
||||
// secret is reported above, so upgrading past this commit never silently drops CB-592's
|
||||
// protection.
|
||||
reportMemberCredentialsGap(cfg);
|
||||
// CB-559: `cfg` stays the startup snapshot — every validation and every piece of one-time
|
||||
// wiring below reads it, and must, because those decisions cannot be unmade. `config` is the
|
||||
// live reference the hot paths read per use. Which keys can actually move is ConfigRef's
|
||||
// contract; adding a reader here does not make a key reloadable by itself.
|
||||
ConfigRef config = new ConfigRef(configPath, cfg);
|
||||
|
||||
// The primary/host env that launched bridged must not be tainted.
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
guard.assertPrimaryClean(System.getenv());
|
||||
|
||||
// CB-501: refuse to start if the bind is wider than the auth mode can defend. Under
|
||||
// loopback-trust, "not a known worker" means "the primary" — sound only because the OS
|
||||
// refuses remote connections to a loopback socket. This throws rather than warns so the
|
||||
// dangerous configuration cannot be reached by ignoring a log line.
|
||||
cfg.validateAuthExposure();
|
||||
cfg.validateLeadTabPrefixes();
|
||||
// CB-542: a subscription:true profile whose env: reseats ANTHROPIC_BASE_URL/AUTH_TOKEN would
|
||||
// reach an unguarded endpoint (the launcher skips SubscriptionGuard for it). Refuse at load.
|
||||
cfg.validateSubscriptionProfiles();
|
||||
cfg.validateCharters();
|
||||
// CB-548: every architect slot must name a configured workers: profile — the strong-model
|
||||
// backend the future spawn lifecycle would read. A stale reference dies here, not later.
|
||||
cfg.validateMembers();
|
||||
|
||||
Path socket = cfg.herdrSocket() != null && !cfg.herdrSocket().isBlank()
|
||||
? Path.of(cfg.herdrSocket())
|
||||
: UnixSocketHerdrClient.defaultSocketPath();
|
||||
@@ -57,11 +124,64 @@ public final class Bridged {
|
||||
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
WorkspaceControl spaces = new WorkspaceControl(herdr);
|
||||
PeerLauncher workers = new ClaudeCodeLauncher(agents, spaces, guard,
|
||||
cfg.workerProfiles(), cfg.defaultProfile(), System::getenv);
|
||||
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died with
|
||||
// the previous process — reap those leaked orphans now, before we start serving.
|
||||
workers.reapOrphanWorkers();
|
||||
// CB-402: one adapter per configured peer kind, fronted by a composite router. A profile's
|
||||
// `kind:` selects its adapter — claude-code (the default) and opencode partition the profile
|
||||
// set — and the composite dispatches each SPI call to the adapter that owns the profile/pane.
|
||||
Map<String, BridgedConfig.Profile> claudeProfiles = new LinkedHashMap<>();
|
||||
Map<String, BridgedConfig.Profile> opencodeProfiles = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, w) -> {
|
||||
if (w.isOpenCode()) {
|
||||
opencodeProfiles.put(name, w);
|
||||
} else {
|
||||
claudeProfiles.put(name, w);
|
||||
}
|
||||
});
|
||||
List<HerdrPeerLauncher> adapters = new ArrayList<>();
|
||||
// The claude-code adapter is the always-present default; keep it even with no profiles (so a
|
||||
// bridge configured with no workers, or opencode-only, still has a well-defined base adapter)
|
||||
// unless opencode is the only kind configured.
|
||||
if (!claudeProfiles.isEmpty() || opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new ClaudeCodeLauncher(agents, spaces, guard,
|
||||
claudeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet(),
|
||||
() -> config.get().memberCredentials()));
|
||||
}
|
||||
if (!opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new OpenCodeLauncher(agents, spaces,
|
||||
opencodeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet(),
|
||||
() -> config.get().memberCredentials()));
|
||||
}
|
||||
AtomicReference<Function<String, Integer>> liveCountRef = new AtomicReference<>(_ -> 0);
|
||||
// CB-578 stage B: one quarantine tracker for the whole daemon, shared between the launcher
|
||||
// (checked at spawn) and the exhaustion sink wired in below (written on BACKEND_EXHAUSTED).
|
||||
// The cooldown is deferred (see BridgedConfig#quarantineCooldownSeconds): it is read once
|
||||
// here, at startup, and a config reload only changes it for a daemon restart.
|
||||
BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));
|
||||
PeerLauncher workers = new CompositePeerLauncher(
|
||||
adapters,
|
||||
cfg.effectiveDefaultProfile(),
|
||||
config,
|
||||
profileName -> liveCountRef.get().apply(profileName),
|
||||
quarantine);
|
||||
// CB-504: under supervision (launchd/systemd) bridged can start before herdr's socket
|
||||
// exists. The client itself is lazy — it connects per call — but the orphan reap below is
|
||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||
// would crash the daemon into a restart loop. Wait, then degrade rather than die: serving
|
||||
// with /healthz reporting "degraded" is strictly more useful than exiting.
|
||||
boolean herdrUp = awaitHerdr(herdr);
|
||||
if (herdrUp) {
|
||||
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
||||
// with the previous process — reap those leaked orphans now, before we start serving.
|
||||
workers.reapOrphanWorkers();
|
||||
} else {
|
||||
log.warn("herdr did not answer within {}s — starting anyway; /healthz will report "
|
||||
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
HERDR_WAIT_SECONDS);
|
||||
}
|
||||
|
||||
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
||||
// CB-301-ext: worktree provisioning seam, optionally rooted at a configured directory.
|
||||
@@ -71,7 +191,12 @@ public final class Bridged {
|
||||
&& cfg.lifecycle().contextCap() > 0) {
|
||||
contextCap = cfg.lifecycle().contextCap();
|
||||
}
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(cfg.worktreeRoot()), contextCap);
|
||||
boolean clearAfterTurn = cfg.lifecycle() != null && cfg.lifecycle().clearAfterTurn();
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(cfg.worktreeRoot()),
|
||||
System::nanoTime, contextCap, clearAfterTurn);
|
||||
liveCountRef.set(profileName -> (int) sessions.roster().stream()
|
||||
.filter(s -> profileName.equals(s.profile()))
|
||||
.count());
|
||||
|
||||
// CB-303 part 1: idle-ttl reaper — only when configured, defaults to disabled.
|
||||
final SessionReaper reaper;
|
||||
@@ -84,14 +209,110 @@ public final class Bridged {
|
||||
reaper = null;
|
||||
}
|
||||
|
||||
// CB-530: every pane the config names as a lead, merged from `leaders:` and the legacy
|
||||
// singular pin. PrimaryRegistry below still tracks ONE terminal — it addresses the push
|
||||
// loop's nudges, which need a single destination — so it keeps the legacy pin.
|
||||
Map<String, String> leadTerminals = cfg.leaderTerminals();
|
||||
if (leadTerminals.size() > 1) {
|
||||
log.info("leads: {} panes recognised {}", leadTerminals.size(), leadTerminals.values());
|
||||
}
|
||||
// CB-531: on top of the legacy primary.terminal pin, discover leads by the tab labels the
|
||||
// operator writes. CB-557 moved the settings onto the lead they describe, so scanning is on
|
||||
// whenever a `fleet.leaders:` entry exists — with no leads configured the supplier is a
|
||||
// constant and never touches herdr, exactly as a missing `leadScan:` block used to behave.
|
||||
// CB-579: each lead now names its own exact `tab:` label, so one scanner discovers every
|
||||
// configured lead regardless of how differently their tabs are labelled — the old
|
||||
// single-shared-tabPrefix limitation (and its warning) is gone.
|
||||
final Supplier<Map<String, String>> leads;
|
||||
var leaders = cfg.fleet().leaders();
|
||||
if (!leaders.isEmpty()) {
|
||||
Set<String> memberSpaces = cfg.profiles().values().stream()
|
||||
.map(BridgedConfig.Profile::workspace)
|
||||
.filter(Objects::nonNull)
|
||||
.collect(Collectors.toSet());
|
||||
Map<String, String> tabToName = new LinkedHashMap<>();
|
||||
leaders.forEach((name, leader) -> {
|
||||
if (leader != null && leader.tab() != null && !leader.tab().isBlank()) {
|
||||
tabToName.put(leader.tab(), name);
|
||||
}
|
||||
});
|
||||
// One shared rescan cadence: still taken from the first entry, as before — it is an
|
||||
// operational cadence, not identity, so there is no correctness reason to give every
|
||||
// lead its own scanner.
|
||||
int scanIntervalSeconds = leaders.values().iterator().next().scanIntervalSeconds();
|
||||
leads = new LeadTabScanner(herdr, tabToName, memberSpaces,
|
||||
TimeUnit.SECONDS.toNanos(scanIntervalSeconds), System::nanoTime);
|
||||
log.info("lead scan: tabs {} host a lead (rescan every {}s, member spaces {} excluded)",
|
||||
tabToName.keySet(), scanIntervalSeconds, memberSpaces);
|
||||
} else {
|
||||
leads = () -> leadTerminals;
|
||||
}
|
||||
|
||||
// CB-558: start any declared lead that is not already running. After the scanner is built,
|
||||
// because both read the same tab labels and the ordering makes that dependency visible; and
|
||||
// only when herdr answered, because the launcher's whole safety property is that it can
|
||||
// count live leads first — it must never guess and risk a second orchestrator.
|
||||
if (herdrUp && !leaders.isEmpty()) {
|
||||
int launched = new LeadLauncher(agents, spaces, cfg).ensureLeads();
|
||||
if (launched > 0) {
|
||||
log.info("lead auto-launch: {} lead(s) started", launched);
|
||||
}
|
||||
}
|
||||
|
||||
// CB-548: config-declared architect slots. Config supplies only the stable name → profile
|
||||
// map; the terminal → slot binding is owned by the registry and is empty at startup, so no
|
||||
// pane resolves to an architect until the later spawn lifecycle binds one. The registry is
|
||||
// what CallerResolver resolves against and what that lifecycle will read profiles from;
|
||||
// nothing here spawns a slot.
|
||||
MemberRegistry members = new MemberRegistry(cfg.fleet());
|
||||
sessions.setMemberLifecycle(members);
|
||||
if (!members.slots().isEmpty()) {
|
||||
log.info("member slots: {} configured {} — none bound yet (a slot is idle until the "
|
||||
+ "spawn lifecycle binds a live terminal to it)",
|
||||
members.slots().size(), members.slots().keySet());
|
||||
}
|
||||
|
||||
// Status-gated injector (CB-103): the single writer into workers, fed by a poller.
|
||||
// The blocking message endpoint (CB-104) is the producer; the poller is inert until then.
|
||||
// CB-106: a confirmed turn completion resolves a blocked send whose worker never replied.
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous);
|
||||
// CB-578 stage A: classify a completion-fallback scrape that matches a profile's configured
|
||||
// usage-limit refusal as BACKEND_EXHAUSTED rather than handing it back as a real answer.
|
||||
// Compiled once at startup, keyed by profile name; a profile with no exhaustedPattern is
|
||||
// simply absent here, so its workers keep today's completion-fallback behaviour unchanged.
|
||||
Map<String, Pattern> exhaustedPatternsByProfile = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (profile.hasExhaustedPattern()) {
|
||||
exhaustedPatternsByProfile.put(name, Pattern.compile(profile.exhaustedPattern()));
|
||||
}
|
||||
});
|
||||
ExhaustedPatternLookup exhaustedPatterns = target -> sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(session -> exhaustedPatternsByProfile.get(session.profile()))
|
||||
.orElse(null);
|
||||
log.info("backend-exhausted classification (CB-578 stage A): {}",
|
||||
CompletionResolver.coverage(cfg.profiles().keySet(), exhaustedPatternsByProfile.keySet()));
|
||||
// CB-578 stage B: on a classification that actually wins, quarantine the exhausted profile's
|
||||
// CREDENTIAL — not the profile name — so a profile sharing that credential (e.g. two models
|
||||
// on one OpenAI account) is refused too, not just the one that happened to report it. Reads
|
||||
// the profile config live off `config`, so a credentialId edit is hot: no restart needed.
|
||||
ExhaustionSink exhaustionSink = (target, reason) -> sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(MemberSession::profile)
|
||||
.map(profileName -> config.get().profiles().get(profileName))
|
||||
.ifPresent(profile -> {
|
||||
String credentialId = profile.effectiveCredentialId();
|
||||
quarantine.quarantine(credentialId);
|
||||
log.warn("credential '{}' quarantined for {}s (profile '{}' classified "
|
||||
+ "BACKEND_EXHAUSTED): {}", credentialId,
|
||||
cfg.quarantineCooldownSeconds(), profile.profile(), reason);
|
||||
});
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous, exhaustedPatterns, exhaustionSink);
|
||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||
WorkerPresence presence = sessions.asPresence();
|
||||
MemberPresence presence = sessions.asPresence();
|
||||
TurnListener turnListener = new TurnListener() {
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
@@ -100,9 +321,20 @@ public final class Bridged {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target) {
|
||||
completion.onDelivered(target);
|
||||
sessions.onDelivered(target);
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
return sessions.hasPostTurnAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
completion.resolveBeforePostAction(target);
|
||||
return sessions.onTurnCompleteWithPostAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target, dev.ltms.bridged.msg.TurnToken token) {
|
||||
completion.onDelivered(target, token);
|
||||
sessions.onDelivered(target, token);
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -110,19 +342,169 @@ public final class Bridged {
|
||||
completion.onTurnFailed(target);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
completion.onTurnFailed(target, reason);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
};
|
||||
Injector injector = new Injector(agents, turnListener, presence::isPresent, presence::forget);
|
||||
StatusPoller poller = new StatusPoller(agents, injector, INJECT_POLL_MILLIS);
|
||||
Injector injector = new Injector(agents, turnListener, deliverableTo(presence, leads),
|
||||
presence::forget);
|
||||
StatusPoller poller = new StatusPoller(agents, injector, Injector.POLL_INTERVAL_MILLIS);
|
||||
poller.start();
|
||||
|
||||
MessageService messages = new MessageService(agents, injector, rendezvous);
|
||||
// CB-307: reply inbox. A broker: block (with a uri) selects the AMQP-backed durable adapter;
|
||||
// absent, bridged stays soft-state on the in-memory inbox. The AMQP inbox owns a broker
|
||||
// connection, so keep the reference to close it in the ordered shutdown hook.
|
||||
final ReplyInbox replyInbox;
|
||||
if (cfg.broker() != null && cfg.broker().isConfigured()) {
|
||||
replyInbox = AmqpReplyInbox.open(cfg.broker().uri(), cfg.broker().prefetchOrDefault());
|
||||
log.info("reply inbox: AMQP broker (durable) at {} (prefetch={})",
|
||||
cfg.broker().uri(), cfg.broker().prefetchOrDefault());
|
||||
} else {
|
||||
replyInbox = new InMemoryReplyInbox();
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
}
|
||||
// CB-307: learn the primary's terminal from orchestration tool calls (or pin from config).
|
||||
// The pin also feeds CallerResolver below: a primary running inside a herdr pane would
|
||||
// otherwise resolve as a worker and be refused every orchestration tool.
|
||||
String pinnedPrimaryTerminal = cfg.primary() != null ? cfg.primary().terminal() : null;
|
||||
PrimaryRegistry primaryRegistry = new PrimaryRegistry(pinnedPrimaryTerminal);
|
||||
// CB-532: `primary.terminal` is superseded and no longer needed for either of its jobs —
|
||||
// identity comes from `leaders:`/`leadScan:`, and reply nudges now follow the delegating
|
||||
// lead. Say so once at startup rather than leaving a redundant pin to look load-bearing.
|
||||
if (pinnedPrimaryTerminal != null && !pinnedPrimaryTerminal.isBlank()) {
|
||||
log.warn("primary.terminal is DEPRECATED (CB-532) and can be deleted: identity now comes "
|
||||
+ "from leaders:/leadScan:, and reply nudges follow the lead that delegated. "
|
||||
+ "It still works, and is still the fallback nudge destination when a restart "
|
||||
+ "has lost the delegation map. Its pushReminders/pushBackoffMs stay valid.");
|
||||
}
|
||||
// CB-307: active push-to-primary loop — nudge the primary when replies land without an
|
||||
// open bridge_send. Uses its own lightweight scheduled executor, separate from the injector.
|
||||
int maxReminders = cfg.primary() != null ? cfg.primary().remindersOrDefault() : 5;
|
||||
long backoffMs = cfg.primary() != null ? cfg.primary().backoffMsOrDefault() : 15_000L;
|
||||
var pushScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-push-").unstarted(r));
|
||||
// CB-502: the registry is built before the service and the push loop so send/reply outcomes
|
||||
// are counted at their single funnel rather than at each of the two caller-facing surfaces.
|
||||
// CB-512: the push loop takes it too, so nudge outcomes (delivered|exhausted) are counted.
|
||||
Metrics metrics = BridgedMetrics.create(sessions, replyInbox);
|
||||
var pushLoop = new ReplyPushLoop(primaryRegistry, agents, replyInbox,
|
||||
pushScheduler, maxReminders, backoffMs, metrics);
|
||||
// CB-551: idle-lead heartbeat. Opt-in; absent `leadHeartbeat:` this is never constructed, so
|
||||
// an upgraded daemon cannot silently start spending subscription on nudging an idle lead.
|
||||
// It has its own single-thread scheduler and holds its own scheduler shutdown via close().
|
||||
final LeadHeartbeatLoop heartbeat;
|
||||
var heartbeatScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-heartbeat-").unstarted(r));
|
||||
if (cfg.leadHeartbeat() != null) {
|
||||
var hb = cfg.leadHeartbeat();
|
||||
heartbeat = new LeadHeartbeatLoop(primaryRegistry, agents, replyInbox, sessions::roster,
|
||||
pushLoop, heartbeatScheduler, System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(hb.idleAfterSeconds()), hb.backoffMs(), hb.quietNudgeCap(),
|
||||
metrics);
|
||||
heartbeat.start();
|
||||
} else {
|
||||
heartbeat = null;
|
||||
heartbeatScheduler.shutdownNow();
|
||||
}
|
||||
MessageService messages = new MessageService(agents, injector, rendezvous, replyInbox,
|
||||
pushLoop, metrics);
|
||||
|
||||
// Health is a slow whole-fleet observer. Keep it separate from the 250ms delivery poller.
|
||||
final FleetHealthMonitor healthMonitor;
|
||||
var healthScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-health-").unstarted(r));
|
||||
if (cfg.health() != null && cfg.health().isEnabled()) {
|
||||
// CB-580: a member found GONE/NEVER_READY must fail whatever ticket is waiting on it,
|
||||
// through the same idempotent target-wide operation CB-516 already uses on release.
|
||||
healthMonitor = new FleetHealthMonitor(agents, sessions::roster, messages, healthScheduler,
|
||||
System::nanoTime, cfg.health().intervalOrDefault(), messages::abandon);
|
||||
String coverage = FleetHealthMonitor.coverage(true,
|
||||
cfg.health().notifications() != null && cfg.health().notifications().configured());
|
||||
if ("detection-only".equals(coverage)) {
|
||||
log.warn("fleet health: {} (no notification sink configured)", coverage);
|
||||
} else {
|
||||
log.info("fleet health: {}", coverage);
|
||||
}
|
||||
healthMonitor.start();
|
||||
} else {
|
||||
healthMonitor = null;
|
||||
healthScheduler.shutdownNow();
|
||||
}
|
||||
|
||||
// CB-520: the reply inbox only consumes for agents this gateway owns. own on acquire,
|
||||
// release on teardown. Do this before CB-516 so the inbox is owned before any reply can land.
|
||||
sessions.onAcquire(replyInbox::own);
|
||||
// CB-516: releasing a worker must fail whatever send was waiting on it. Without this a
|
||||
// torn-down delegation kept reporting PENDING until the 30-minute async timeout, and never
|
||||
// reached /metrics — the delegation was unresolvable and nothing said so.
|
||||
sessions.onRelease(detail -> {
|
||||
// CB-578 stage C, acceptance criterion 10: a failed ticket's detail should tell a lead
|
||||
// where to re-dispatch onto the same tree, not just that the worker vanished.
|
||||
String reason = "the worker session was released before it replied";
|
||||
if (detail.worktreePath() != null) {
|
||||
reason += "; worktree=" + detail.worktreePath() + " branch=" + detail.branch()
|
||||
+ " snapshot=" + (detail.snapshotRef() != null ? detail.snapshotRef() : "none");
|
||||
}
|
||||
// CB-584 (issue #65 criterion 5): also name the agent session, so a lead can resume the
|
||||
// member's conversation instead of only re-dispatching a fresh one onto the same files.
|
||||
if (detail.agentSessionId() != null) {
|
||||
reason += " agentSessionId=" + detail.agentSessionId();
|
||||
}
|
||||
messages.abandon(detail.terminalId(), reason);
|
||||
replyInbox.release(detail.terminalId());
|
||||
primaryRegistry.forgetDelegation(detail.terminalId()); // CB-532: don't leak the lead binding
|
||||
});
|
||||
|
||||
// MCP server face (CB-105): bridge_send/bridge_reply/bridge_status, mounted at /mcp.
|
||||
// Caller identity is resolved from the connection (peer PID → herdr pane), not arguments.
|
||||
ConnectionIdentity identity = new ConnectionIdentity(
|
||||
new PaneLocator(herdr), new LsofPeerPidLookup(), new LsofProcessCwdLookup());
|
||||
// Cast to ClaudeCodeLauncher: BridgeMcp is not yet migrated to PeerLauncher (Stage A scope).
|
||||
BridgeMcp mcp = new BridgeMcp(messages, rendezvous, (ClaudeCodeLauncher) workers, sessions, identity, presence);
|
||||
|
||||
// CB-501: one resolver behind both entry paths. Worker identity still comes from the
|
||||
// connection and is never token-gated, so enabling token mode cannot lock the fleet out.
|
||||
final CallerResolver callers;
|
||||
if (cfg.auth().tokenMode()) {
|
||||
String token = System.getenv(cfg.auth().tokenEnv());
|
||||
if (token == null || token.isBlank()) {
|
||||
throw new IllegalStateException("auth.mode=token but env var " + cfg.auth().tokenEnv()
|
||||
+ " is unset or empty — export it before starting bridged");
|
||||
}
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, true, token, leads, members);
|
||||
log.info("auth: token mode (bearer required for non-worker callers, env {})",
|
||||
cfg.auth().tokenEnv());
|
||||
} else {
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, false, null, leads, members);
|
||||
log.info("auth: loopback-trust (any loopback non-worker caller is the primary)");
|
||||
}
|
||||
|
||||
BridgeMcp mcp = new BridgeMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, metrics, new BridgeMcp.CapacitySource(profile -> liveCountRef.get().apply(profile),
|
||||
profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.maxLoad();
|
||||
}, () -> config.get().profiles().keySet(), System::nanoTime),
|
||||
new BridgeMcp.HealthCoverageSource(() -> {
|
||||
var health = config.get().health();
|
||||
return FleetHealthMonitor.coverage(health != null && health.isEnabled(),
|
||||
health != null && health.notifications() != null && health.notifications().configured());
|
||||
}),
|
||||
new BridgeMcp.QuarantineSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, quarantine));
|
||||
|
||||
// CB-559: opt-in config reload. With no `configReload:` block nothing is constructed, so an
|
||||
// upgraded daemon behaves exactly as before — the file is read once at boot and never again.
|
||||
final ConfigWatcher configWatcher;
|
||||
if (cfg.configReload() != null && cfg.configReload().isEnabled()) {
|
||||
configWatcher = new ConfigWatcher(config, cfg.configReload().intervalSeconds());
|
||||
configWatcher.start();
|
||||
} else {
|
||||
configWatcher = null;
|
||||
}
|
||||
|
||||
// CB-303 part 3: single ordered shutdown hook. Drain sessions first while herdr is still
|
||||
// open (so releases reach the daemon), then stop poller/message/mcp/reaper, and close herdr
|
||||
@@ -131,17 +513,171 @@ public final class Bridged {
|
||||
sessions.close(cfg.lifecycle() != null ? cfg.lifecycle().drainTimeoutSeconds() : null);
|
||||
poller.stop();
|
||||
messages.close();
|
||||
pushLoop.close();
|
||||
if (heartbeat != null) heartbeat.close(); // CB-551: stop the idle-lead heartbeat scheduler
|
||||
if (healthMonitor != null) healthMonitor.stop();
|
||||
if (configWatcher != null) configWatcher.stop(); // CB-559: stop polling the config file
|
||||
mcp.close();
|
||||
if (reaper != null) reaper.stop();
|
||||
// Release the broker connection last among message resources (no-op for the in-memory inbox).
|
||||
if (replyInbox instanceof AutoCloseable closeable) {
|
||||
try {
|
||||
closeable.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("reply inbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
herdr.close();
|
||||
}));
|
||||
|
||||
Javalin app = new BridgedApp(herdr, (ClaudeCodeLauncher) workers, sessions, messages, rendezvous, presence, mcp.servlet()).build();
|
||||
Javalin app = new BridgedApp(herdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics).build();
|
||||
app.start(cfg.bind().host(), cfg.bind().port());
|
||||
log.info("bridged listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@link Injector}'s readiness gate (CB-534): a target is deliverable if it is a spawned
|
||||
* member whose agent has connected the bridge MCP, <em>or</em> a lead.
|
||||
*
|
||||
* <p>The gate exists for one reason — to hold a delivery out of a <em>spawned</em> member's boot
|
||||
* window, where herdr already reports {@code idle} but the TUI would drop an injected paste. That
|
||||
* hazard is a property of spawning. A lead is never spawned: the operator started it and named it
|
||||
* (or labelled its tab) only once it was up, so there is no boot window to guard.
|
||||
*
|
||||
* <p>A lead is also never enrolled in {@link MemberPresence} — {@code BridgeMcp} marks presence
|
||||
* for every spawned member (worker and architect), deliberately, since that map doubles as the
|
||||
* member roster's availability signal and a lead counted there would show up as an available
|
||||
* member. So without the second disjunct a lead is permanently un-deliverable: every
|
||||
* lead→lead send sat on the gate for {@code READINESS_GRACE_POLLS} (~60s) and then failed
|
||||
* having never been typed into the pane.
|
||||
*
|
||||
* <p>The lead set is read through the supplier on each call rather than snapshotted, so a lead
|
||||
* discovered by {@code leadScan} after startup becomes deliverable without a restart.
|
||||
*/
|
||||
static Predicate<String> deliverableTo(MemberPresence presence, Supplier<Map<String, String>> leads) {
|
||||
return target -> presence.isPresent(target) || leads.get().containsKey(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-594: which env vars the loaded config actually needs, and why — every non-{@code
|
||||
* subscription} profile's {@code tokenEnv} (a subscription profile never reads one, see
|
||||
* {@link BridgedConfig.Profile#isSubscription()}), plus every profile's {@code gitTokenEnv}
|
||||
* where set (opt-in). Derived from the config, not hard-coded, so a new profile is covered for
|
||||
* free. A var required by more than one profile is one entry naming every profile that needs
|
||||
* it. Deliberately excludes {@code auth.tokenEnv}: that one is already enforced loudly, by a
|
||||
* startup throw in {@code main()} — about 370 lines <em>below</em> this method's call site
|
||||
* ({@link #reportRequiredSecrets(BridgedConfig)}), not a few lines above it. That throw only
|
||||
* fires when {@code auth.mode: token} is configured; under the default loopback-trust mode it
|
||||
* never runs, and {@code auth.tokenEnv} is simply not required.
|
||||
*
|
||||
* <p>Package-private and pure (no I/O, no logging) so the derivation is unit-testable without
|
||||
* capturing log output; {@link #reportRequiredSecrets(BridgedConfig)} is the logging caller.
|
||||
*/
|
||||
static Map<String, List<String>> requiredSecretEnvVars(BridgedConfig cfg) {
|
||||
Map<String, List<String>> requiredBy = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (!profile.isSubscription()) {
|
||||
requiredBy.computeIfAbsent(profile.tokenEnv(), _ -> new ArrayList<>())
|
||||
.add("profile '" + name + "' tokenEnv");
|
||||
}
|
||||
if (profile.hasGitToken()) {
|
||||
requiredBy.computeIfAbsent(profile.gitTokenEnv(), _ -> new ArrayList<>())
|
||||
.add("profile '" + name + "' gitTokenEnv");
|
||||
}
|
||||
});
|
||||
return requiredBy;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-594: log, by name only, which required env vars (see {@link #requiredSecretEnvVars}) are
|
||||
* set in the daemon's own process environment — the environment every profile's {@code
|
||||
* tokenEnv}/{@code gitTokenEnv} is read from at spawn time (see
|
||||
* {@code HerdrPeerLauncher.resolveEnv}). Never logs a value, a prefix, or a length.
|
||||
*
|
||||
* <p>A missing entry only warns — it must never refuse to start. A daemon that boots and says
|
||||
* what is wrong is strictly more useful than one that will not boot at all.
|
||||
*/
|
||||
private static void reportRequiredSecrets(BridgedConfig cfg) {
|
||||
Map<String, List<String>> requiredBy = requiredSecretEnvVars(cfg);
|
||||
if (requiredBy.isEmpty()) {
|
||||
log.info("startup secrets: no profile references a token env var — nothing to check");
|
||||
return;
|
||||
}
|
||||
Map<String, String> env = System.getenv();
|
||||
requiredBy.forEach((varName, sources) -> {
|
||||
String value = env.get(varName);
|
||||
if (value != null && !value.isBlank()) {
|
||||
log.info("startup secret {}: set ({})", varName, String.join(", ", sources));
|
||||
} else {
|
||||
log.warn("startup secret {}: MISSING ({}) — the daemon will start anyway, and this "
|
||||
+ "failure stays invisible until a worker actually needs it. Fix "
|
||||
+ "${SHARED_ENV}/tools/secrets.sh and restart bridged from a LOGIN "
|
||||
+ "shell (see scripts/redeploy-bridged.sh).",
|
||||
varName, String.join(", ", sources));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-596: {@code known:} empty (block absent entirely, or present but empty) means {@link
|
||||
* BridgedConfig.MemberCredentials#blockedSet()} is empty too — every member pane inherits the
|
||||
* operator's whole secret store, unblocked, exactly the defect this ticket fixes. Unlike a
|
||||
* missing token ({@link #reportRequiredSecrets}), there is no name to point at: the point is
|
||||
* that the block itself is missing. Warn once at startup and say what to add; never refuse to
|
||||
* start over it — see {@link #reportRequiredSecrets} for why a daemon that boots and says
|
||||
* what is wrong beats one that will not boot at all.
|
||||
*
|
||||
* <p>Package-private so the test can capture the log directly, the same way {@link
|
||||
* #requiredSecretEnvVars} is exposed for {@link #reportRequiredSecrets}'s own test.
|
||||
*/
|
||||
static void reportMemberCredentialsGap(BridgedConfig cfg) {
|
||||
BridgedConfig.MemberCredentials creds = cfg.memberCredentials();
|
||||
if (creds != null && !creds.known().isEmpty()) {
|
||||
log.info("memberCredentials: {} known name(s), {} allowed — blocking {} on every spawn",
|
||||
creds.known().size(), creds.allow().size(), creds.blockedSet().size());
|
||||
return;
|
||||
}
|
||||
log.warn("memberCredentials: absent or empty — the daemon will start anyway, and every "
|
||||
+ "member pane inherits the operator's WHOLE secret store, unblocked (CB-592's "
|
||||
+ "protection is lost). Add a memberCredentials: block (policy/allow/known) to "
|
||||
+ "bridged.yaml — see bridged.example.yaml — and restart.");
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll herdr's {@code ping} until it answers or {@link #HERDR_WAIT_SECONDS} elapses (CB-504).
|
||||
*
|
||||
* @return true if herdr answered, false if it never did
|
||||
*/
|
||||
private static boolean awaitHerdr(HerdrClient herdr) {
|
||||
long deadline = System.nanoTime() + HERDR_WAIT_SECONDS * 1_000_000_000L;
|
||||
boolean waited = false;
|
||||
while (true) {
|
||||
try {
|
||||
herdr.call("ping");
|
||||
if (waited) {
|
||||
log.info("herdr is up");
|
||||
}
|
||||
return true;
|
||||
} catch (HerdrException e) {
|
||||
if (System.nanoTime() >= deadline) {
|
||||
return false;
|
||||
}
|
||||
if (!waited) {
|
||||
log.info("waiting up to {}s for the herdr socket…", HERDR_WAIT_SECONDS);
|
||||
waited = true;
|
||||
}
|
||||
try {
|
||||
Thread.sleep(HERDR_WAIT_POLL_MILLIS);
|
||||
} catch (InterruptedException ie) {
|
||||
Thread.currentThread().interrupt();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private Bridged() {
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.time.Instant;
|
||||
import java.time.ZoneOffset;
|
||||
import java.time.format.DateTimeFormatter;
|
||||
|
||||
/**
|
||||
* Append-only record of privileged actions (CB-505).
|
||||
*
|
||||
* <p>Writes JSON lines to a dedicated {@code audit} logger — its own appender, separate from the
|
||||
* chatty app log — so the trail stays greppable and can later be shipped without dragging debug
|
||||
* noise along.
|
||||
*
|
||||
* <p><strong>Message content is never recorded.</strong> This bridge carries the user's source
|
||||
* code, diffs, and prompts; an audit trail that quietly accumulated them would be a transcript
|
||||
* archive wearing a security control's clothing. Records carry <em>who / what / against what /
|
||||
* outcome</em> and correlation ids only.
|
||||
*/
|
||||
public final class AuditLog {
|
||||
|
||||
private static final Logger AUDIT = LoggerFactory.getLogger("audit");
|
||||
private static final DateTimeFormatter TS =
|
||||
DateTimeFormatter.ofPattern("yyyy-MM-dd'T'HH:mm:ss.SSSXXX").withZone(ZoneOffset.UTC);
|
||||
|
||||
private AuditLog() {
|
||||
}
|
||||
|
||||
/** Record an allowed action. */
|
||||
public static void allowed(Principal caller, Authz.Action action, String target) {
|
||||
write(caller, action, target, "allowed", null);
|
||||
}
|
||||
|
||||
/** Record a refused action and why. */
|
||||
public static void denied(Principal caller, Authz.Action action, String target, String reason) {
|
||||
write(caller, action, target, "denied", reason);
|
||||
}
|
||||
|
||||
/** Record an action that was authorized but then failed downstream (guard, timeout, herdr). */
|
||||
public static void failed(Principal caller, Authz.Action action, String target, String reason) {
|
||||
write(caller, action, target, "failed", reason);
|
||||
}
|
||||
|
||||
private static void write(Principal caller, Authz.Action action, String target,
|
||||
String outcome, String reason) {
|
||||
Principal c = caller != null ? caller : Principal.anonymous();
|
||||
StringBuilder sb = new StringBuilder(200);
|
||||
// The timestamp is built here rather than by the appender pattern: a pattern that wrapped
|
||||
// literal braces around the message collides with logback's own variable substitution.
|
||||
sb.append("{\"ts\":\"").append(TS.format(Instant.now())).append('"')
|
||||
.append(",\"role\":\"").append(c.role()).append('"')
|
||||
.append(",\"actor\":\"").append(esc(c.describe())).append('"')
|
||||
.append(",\"pid\":").append(c.pid())
|
||||
.append(",\"action\":\"").append(action).append('"')
|
||||
.append(",\"target\":").append(target == null ? "null" : '"' + esc(target) + '"')
|
||||
.append(",\"outcome\":\"").append(outcome).append('"');
|
||||
if (reason != null) {
|
||||
sb.append(",\"reason\":\"").append(esc(reason)).append('"');
|
||||
}
|
||||
sb.append('}');
|
||||
// The appender supplies the timestamp, so it cannot disagree with the app log's clock.
|
||||
AUDIT.info(sb.toString());
|
||||
}
|
||||
|
||||
/** Minimal JSON string escaping — these values are ids and short reasons, never free text. */
|
||||
private static String esc(String s) {
|
||||
return s.replace("\\", "\\\\").replace("\"", "\\\"")
|
||||
.replace("\n", "\\n").replace("\r", "\\r").replace("\t", "\\t");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
/**
|
||||
* The authorization table (CB-505), stated once and enforced on both entry paths.
|
||||
*
|
||||
* <p>Most of these rules are already true de facto — {@code BridgeMcp} derives a worker's identity
|
||||
* from the connection rather than reading it from an argument, so a worker has never been able to
|
||||
* reply <em>as</em> another worker over MCP. What was missing is that the REST surface trusted the
|
||||
* session id in the URL path, and neither surface checked role at all. This class makes the
|
||||
* invariant explicit and testable rather than emergent.
|
||||
*/
|
||||
public final class Authz {
|
||||
|
||||
private Authz() {
|
||||
}
|
||||
|
||||
/** A privileged operation, named for the audit trail. */
|
||||
public enum Action {
|
||||
/** Spawn a worker peer. */
|
||||
SPAWN,
|
||||
/** Tear a worker peer down. */
|
||||
STOP,
|
||||
/** Deliver a turn to a session (or answer a worker's question). */
|
||||
SEND,
|
||||
/** A worker's terminal reply for its own turn. */
|
||||
REPLY,
|
||||
/** A worker's mid-turn question to the primary. */
|
||||
ASK,
|
||||
/** Collect held replies from a session's inbox. */
|
||||
DRAIN,
|
||||
/** Read-only observation: status, roster, profiles, task polling. */
|
||||
READ,
|
||||
/** Scrape the metrics endpoint. */
|
||||
METRICS
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code caller} may perform {@code action} against {@code targetSession}.
|
||||
*
|
||||
* @param targetSession the session id in the request path; only consulted for the worker-scoped
|
||||
* actions ({@code REPLY}, {@code ASK}), ignored otherwise, may be
|
||||
* {@code null}
|
||||
*/
|
||||
public static boolean permits(Principal caller, Action action, String targetSession) {
|
||||
if (caller == null || caller.isAnonymous()) {
|
||||
return false; // authenticated as nothing ⇒ authorized for nothing
|
||||
}
|
||||
return switch (action) {
|
||||
// Fleet lifecycle is the primary's alone — spawn, stop, drain. An architect
|
||||
// deliberately does NOT get these (CB-548), so it cannot tear down or stand up workers
|
||||
// even though it coordinates them; and a worker driving any of these would be a worker
|
||||
// escalating into the orchestrator role.
|
||||
case SPAWN, STOP, DRAIN -> caller.isPrimary();
|
||||
|
||||
// Delivering a turn is open to the primary and the architect: an architect delegates
|
||||
// to workers (that is the role's point) but still has no lifecycle rights. A worker is
|
||||
// excluded — sending would be it escalating.
|
||||
case SEND -> caller.isPrimary() || caller.isArchitect();
|
||||
|
||||
// The load-bearing rule: a caller acts only as the pane it occupies. CB-532 widened who
|
||||
// that can be — a lead answering another lead is replying for its OWN terminal, which
|
||||
// this already permits — while the rule itself is unchanged, and is what stops anyone
|
||||
// forging a reply for a rendezvous someone else is waiting on. An architect's own pane
|
||||
// passes through the same check, so it can answer a funnel that delegated to it. An
|
||||
// unnamed primary (token/loopback, no pane) owns nothing and is still excluded.
|
||||
case REPLY, ASK -> caller.ownsSession(targetSession);
|
||||
|
||||
// Observation is open to every authenticated role: a worker legitimately polls its own
|
||||
// status, and the roster carries no secrets.
|
||||
case READ, METRICS -> caller.isPrimary() || caller.isWorker() || caller.isArchitect();
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Why a request was refused, for the error body. Distinguishes "you are nobody" from "you are
|
||||
* somebody, but not the right somebody" — the first is a credential problem (401), the second
|
||||
* an authorization one (403).
|
||||
*/
|
||||
public static boolean isUnauthenticated(Principal caller) {
|
||||
return caller == null || caller.isAnonymous();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,272 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import dev.ltms.bridged.mcp.ConnectionIdentity;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.security.MessageDigest;
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* Resolves every caller to a {@link Principal}, for both entry paths into the core (CB-501).
|
||||
*
|
||||
* <p>There are two of them and they are not layered the way the docs suggest: {@code BridgeMcp}
|
||||
* calls the service layer directly and is mounted as a raw servlet (so it never passes through a
|
||||
* Javalin filter), while the REST routes historically resolved no identity at all. Both now
|
||||
* delegate here, so the authorization rules are stated once instead of drifting apart.
|
||||
*
|
||||
* <p><strong>Resolution order</strong> — connection identity first, token second, nothing third:
|
||||
* <ol>
|
||||
* <li>A loopback peer PID that maps to a pane named by {@code leaders:}, by the legacy
|
||||
* {@code primary.terminal} pin, or by an operator-labelled lead tab (CB-307, CB-530, CB-531)
|
||||
* ⇒ {@link Role#PRIMARY}, carrying that lead's
|
||||
* name. The pane mapping is as unforgeable as a worker's, and the config explicitly names
|
||||
* that pane as a lead's own — without this rule a lead running <em>inside</em> a herdr pane
|
||||
* is misread as a worker and locked out of orchestration. More than one pane may be named,
|
||||
* so two leads can work as peers rather than one being demoted.</li>
|
||||
* <li>A loopback peer PID that maps to a pane bound to a CB-548 architect slot ⇒
|
||||
* {@link Role#ARCHITECT}, carrying the slot name. Just unforgeable as a worker's, and
|
||||
* resolved from the <em>live</em> terminal→slot binding (never a request argument), before
|
||||
* the generic worker fallback.</li>
|
||||
* <li>A loopback peer PID that maps to any other herdr pane ⇒ {@link Role#WORKER}. This is
|
||||
* unforgeable (the OS reports the PID, herdr owns the PID→pane map) and is honoured
|
||||
* regardless of auth mode, so enabling auth never breaks the fleet.</li>
|
||||
* <li>Otherwise, under {@code token} mode, a valid bearer token ⇒ {@link Role#PRIMARY}.</li>
|
||||
* <li>Otherwise, under {@code loopback-trust}, a loopback caller ⇒ {@link Role#PRIMARY}
|
||||
* (the historical behaviour, now an explicit configured choice).</li>
|
||||
* <li>Otherwise {@link Role#ANONYMOUS}.</li>
|
||||
* </ol>
|
||||
*/
|
||||
public final class CallerResolver {
|
||||
|
||||
private final ConnectionIdentity identity;
|
||||
private final boolean tokenMode;
|
||||
private final byte[] expectedToken; // null unless tokenMode
|
||||
/**
|
||||
* terminal_id → lead name; empty when nothing is pinned. CB-530.
|
||||
*
|
||||
* <p>A supplier rather than a map because the registry is no longer fixed at startup: CB-531
|
||||
* discovers leads by scanning herdr for operator-labelled tabs, so a lead that opens its tab
|
||||
* after the daemon booted must still be recognised. Consulted per resolve; the scanner behind
|
||||
* it is TTL-cached, so this is a map lookup in the common case.
|
||||
*/
|
||||
private final Supplier<Map<String, String>> leadTerminals;
|
||||
/**
|
||||
* terminal_id → architect slot name; empty when nothing is configured. CB-548.
|
||||
*
|
||||
* <p>Like {@link #leadTerminals}, a supplier rather than a fixed map, so a binding injected
|
||||
* after startup — when the later spawn lifecycle establishes a live architect session, or an
|
||||
* operator pins one — takes effect without a restart. Consulted per resolve; today's wiring
|
||||
* in {@code Bridged} reads a constant from config, which is the degenerate live case.
|
||||
*/
|
||||
private final Supplier<Map<String, String>> architectTerminals;
|
||||
private final Function<String, MemberRole> memberSlotRoles;
|
||||
private final Function<String, String> memberSlotNames;
|
||||
|
||||
/** Loopback-trust resolver: no token required, historical behaviour. Test-only. */
|
||||
CallerResolver(ConnectionIdentity identity) {
|
||||
this(identity, false, null, Map.of());
|
||||
}
|
||||
|
||||
/** As {@link #CallerResolver(ConnectionIdentity, boolean, String, Map)} with no leads pinned. Test-only. */
|
||||
CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token) {
|
||||
this(identity, tokenMode, token, Map.of());
|
||||
}
|
||||
|
||||
/**
|
||||
* Single-pin form for the legacy {@code primary.terminal}-only configuration — one lead, named
|
||||
* {@code primary}.
|
||||
*
|
||||
* <p>A static factory rather than a fourth constructor overload on purpose: {@code String} and
|
||||
* {@code Map} overloads are ambiguous for a literal {@code null} argument, which is a compile
|
||||
* error at the call site and exactly the shape "unpinned" is written in.
|
||||
*
|
||||
* @param pinnedPrimaryTerminal the primary's own herdr {@code terminal_id}
|
||||
* ({@code null}/blank = unpinned)
|
||||
*/
|
||||
static CallerResolver pinnedTo(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token, String pinnedPrimaryTerminal) {
|
||||
return new CallerResolver(identity, tokenMode, token,
|
||||
pinnedPrimaryTerminal == null || pinnedPrimaryTerminal.isBlank()
|
||||
? Map.of() : Map.of(pinnedPrimaryTerminal, "primary"));
|
||||
}
|
||||
|
||||
/**
|
||||
* @param identity connection-based worker identification
|
||||
* @param tokenMode when true, a non-worker caller must present a valid bearer token
|
||||
* @param token the expected bearer token; required (non-blank) when {@code tokenMode}
|
||||
* @param leadTerminals herdr {@code terminal_id} → lead name for every configured lead
|
||||
* (CB-530). A caller resolving to one of these panes is that lead — a
|
||||
* {@link Role#PRIMARY} — rather than a worker. Empty = nothing pinned,
|
||||
* so every pane resolves as a worker.
|
||||
*/
|
||||
CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Map<String, String> leadTerminals) {
|
||||
this(identity, tokenMode, token, fixed(leadTerminals), null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Live-registry form: {@code leadTerminals} is consulted on every resolve, so leads discovered
|
||||
* after startup (CB-531's tab scan) take effect without a restart.
|
||||
*
|
||||
* <p>A static factory rather than a fourth constructor overload, for the same reason as
|
||||
* {@link #pinnedTo}: {@code Map} and {@code Supplier} overloads are ambiguous for a literal
|
||||
* {@code null}.
|
||||
*/
|
||||
static CallerResolver withLeads(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token,
|
||||
Supplier<Map<String, String>> leadTerminals) {
|
||||
return new CallerResolver(identity, tokenMode, token, leadTerminals, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Live registry form that can confirm a bound slot is an architect slot.
|
||||
*
|
||||
* <p>This is the only public construction path. It keeps terminal bindings and slot roles in
|
||||
* the same {@link MemberRegistry}, so a configured architect can resolve as an architect.
|
||||
*/
|
||||
public static CallerResolver withLeadsAndMembers(ConnectionIdentity identity,
|
||||
boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
MemberRegistry members) {
|
||||
return new CallerResolver(identity, tokenMode, token, leadTerminals,
|
||||
members == null ? null : members::snapshot,
|
||||
members == null ? null : members::roleForSlot,
|
||||
members == null ? null : members::nameForSlot);
|
||||
}
|
||||
|
||||
private static Supplier<Map<String, String>> fixed(Map<String, String> leadTerminals) {
|
||||
Map<String, String> snapshot = leadTerminals == null ? Map.of() : Map.copyOf(leadTerminals);
|
||||
return () -> snapshot;
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals) {
|
||||
this(identity, tokenMode, token, leadTerminals, architectTerminals, null);
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals,
|
||||
Function<String, MemberRole> memberSlotRoles) {
|
||||
this(identity, tokenMode, token, leadTerminals, architectTerminals, memberSlotRoles, null);
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals,
|
||||
Function<String, MemberRole> memberSlotRoles,
|
||||
Function<String, String> memberSlotNames) {
|
||||
if (tokenMode && (token == null || token.isBlank())) {
|
||||
throw new IllegalArgumentException(
|
||||
"auth.mode=token requires a non-empty token; check that the env var named by "
|
||||
+ "auth.tokenEnv is exported to the daemon's environment");
|
||||
}
|
||||
this.identity = identity;
|
||||
this.tokenMode = tokenMode;
|
||||
this.expectedToken = tokenMode ? token.getBytes(StandardCharsets.UTF_8) : null;
|
||||
this.leadTerminals = leadTerminals == null ? Map::of : leadTerminals;
|
||||
this.architectTerminals = architectTerminals == null ? Map::of : architectTerminals;
|
||||
this.memberSlotRoles = memberSlotRoles == null ? _ -> null : memberSlotRoles;
|
||||
this.memberSlotNames = memberSlotNames == null ? Function.identity() : memberSlotNames;
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-recognised leads, {@code terminal_id → name} (CB-535).
|
||||
*
|
||||
* <p>Deliberately read from the same supplier {@link #resolve} consults, rather than from a
|
||||
* second copy handed to the roster: a lead that is <em>listed</em> but would not <em>resolve</em>
|
||||
* (or the reverse) is an address a peer cannot actually reach, and the two answers drifting apart
|
||||
* is precisely the confusion this exists to end. Live, so a lead discovered by the tab scan after
|
||||
* startup appears without a restart.
|
||||
*/
|
||||
public Map<String, String> leads() {
|
||||
return leadTerminals.get();
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-recognised architect slots, {@code terminal_id → slot name} (CB-548).
|
||||
*
|
||||
* <p>Read from the same supplier {@link #resolve} consults, so a slot that is <em>listed</em>
|
||||
* here but would not <em>resolve</em> (or the reverse) cannot drift apart. Live for the same
|
||||
* reason as {@link #leads()}.
|
||||
*/
|
||||
public Map<String, String> members() {
|
||||
return architectTerminals.get();
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the caller of a request.
|
||||
*
|
||||
* @param remoteAddr the connection's remote address
|
||||
* @param remotePort the connection's remote port (used for the peer-PID lookup)
|
||||
* @param authorizationHeader the raw {@code Authorization} header, or {@code null}
|
||||
*/
|
||||
public Principal resolve(String remoteAddr, int remotePort, String authorizationHeader) {
|
||||
ConnectionIdentity.Caller c = identity.resolve(remoteAddr, remotePort);
|
||||
if (c.terminal() != null) {
|
||||
String lead = leadTerminals.get().get(c.terminal());
|
||||
if (lead != null) {
|
||||
// The config names this pane as a lead's own. The pane mapping is exactly as
|
||||
// unforgeable as a worker's, so it outranks the token path — no credential needed.
|
||||
// Checked before the architect registry so a pane named in BOTH is still the lead
|
||||
// (CB-548 preserves every existing leader behaviour).
|
||||
return Principal.leader(lead, c.terminal(), c.pid());
|
||||
}
|
||||
String slot = architectTerminals.get().get(c.terminal());
|
||||
if (slot != null && memberSlotRoles.apply(slot) == MemberRole.ARCHITECT) {
|
||||
// The config/live binding names this pane as an architect slot's own. Same
|
||||
// unforgeable pane mapping; the live binding, never a request argument, decides.
|
||||
// Check the slot role too: this defence in depth prevents a bad lifecycle bind from
|
||||
// escalating a dev or reviewer into an architect. Checked before the worker fallback.
|
||||
return Principal.architect(memberSlotNames.apply(slot), c.terminal(), c.pid());
|
||||
}
|
||||
return Principal.worker(c.terminal(), c.pid()); // unforgeable; never token-gated
|
||||
}
|
||||
|
||||
if (tokenMode) {
|
||||
return presentedTokenMatches(authorizationHeader)
|
||||
? Principal.primary(c.pid())
|
||||
: Principal.anonymous();
|
||||
}
|
||||
|
||||
// loopback-trust: same-host callers that are not workers are the primary. A non-loopback
|
||||
// caller is anonymous even here — and startup refuses that combination anyway
|
||||
// (BridgedConfig.validateAuthExposure), so this is defence in depth, not the control.
|
||||
return isLoopback(remoteAddr) ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
}
|
||||
|
||||
private boolean presentedTokenMatches(String authorizationHeader) {
|
||||
String presented = bearerValue(authorizationHeader);
|
||||
if (presented == null) {
|
||||
return false;
|
||||
}
|
||||
// Constant-time: MessageDigest.isEqual does not short-circuit on the first differing byte,
|
||||
// so a token cannot be recovered a byte at a time by timing the response.
|
||||
return MessageDigest.isEqual(presented.getBytes(StandardCharsets.UTF_8), expectedToken);
|
||||
}
|
||||
|
||||
/** Extract the credential from {@code Authorization: Bearer <token>}, or {@code null}. */
|
||||
private static String bearerValue(String header) {
|
||||
if (header == null) {
|
||||
return null;
|
||||
}
|
||||
String h = header.trim();
|
||||
if (h.length() < 7 || !h.regionMatches(true, 0, "Bearer ", 0, 7)) {
|
||||
return null;
|
||||
}
|
||||
String token = h.substring(7).trim();
|
||||
return token.isEmpty() ? null : token;
|
||||
}
|
||||
|
||||
private static boolean isLoopback(String remoteAddr) {
|
||||
if (remoteAddr == null) {
|
||||
return false;
|
||||
}
|
||||
return remoteAddr.equals("127.0.0.1") || remoteAddr.equals("::1")
|
||||
|| remoteAddr.equals("0:0:0:0:0:0:0:1") || remoteAddr.startsWith("127.");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
|
||||
/** Optional session lifecycle hook for live member-slot bindings. */
|
||||
public interface MemberLifecycle {
|
||||
|
||||
MemberLifecycle NONE = new MemberLifecycle() {
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
}
|
||||
};
|
||||
|
||||
void acquired(MemberRole role, String profile, String terminal);
|
||||
|
||||
void released(String terminal);
|
||||
}
|
||||
@@ -0,0 +1,232 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* The architect-slot registry (CB-548): every gateway-local architect name and the strong-model
|
||||
* profile it points at, plus the <em>live</em> bindings from a live architect's herdr terminal to
|
||||
* its slot.
|
||||
*
|
||||
* <p>Two halves, split by who owns each:
|
||||
* <ul>
|
||||
* <li><b>slots</b> — configured once, keyed by the gateway-local unique name; each carries the
|
||||
* {@code profile} reference the spawn lifecycle reads when it stands the slot up. A read-only
|
||||
* snapshot taken at construction.</li>
|
||||
* <li><b>terminal bindings</b> — owned by this registry and initially <em>empty</em>. Config
|
||||
* declares no architect terminal, so at startup every slot is idle and nothing resolves to an
|
||||
* architect; a session only becomes one when the spawn lifecycle {@linkplain #bind(String,
|
||||
* String) binds} its terminal to a slot. {@link CallerResolver} reads this through
|
||||
* {@link #snapshot()} to turn a pane into an {@link Role#ARCHITECT}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Spawning/lifecycle is deliberately a separate unit: this class only owns the bindings and
|
||||
* exposes the map the resolver resolves against plus the profile lookup lifecycle will call.
|
||||
* Nothing here creates or manages an architect session.
|
||||
*/
|
||||
public final class MemberRegistry implements MemberLifecycle {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(MemberRegistry.class);
|
||||
|
||||
/**
|
||||
* One flattened {@code fleet:} entry.
|
||||
*
|
||||
* <p>Flattened because a slot name is unique only <em>within</em> its pool — {@code sonnet} may
|
||||
* legitimately be both a developer and a reviewer — while a terminal binds to exactly one thing.
|
||||
* The qualified {@link #key()} is what that binding uses.
|
||||
*
|
||||
* @param name the slot's key inside its pool
|
||||
* @param role the pool it came from
|
||||
* @param profile the backend it runs on
|
||||
*/
|
||||
public record Entry(String name, MemberRole role, String profile) {
|
||||
/** {@code "architect:opus"} — unique across pools, unlike {@link #name()}. */
|
||||
public String key() {
|
||||
return role.wireName() + ":" + name;
|
||||
}
|
||||
}
|
||||
|
||||
private final Map<String, Entry> slots;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code this}. */
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one registry. Leaders are not members. */
|
||||
public MemberRegistry(BridgedConfig.Fleet fleet) {
|
||||
Map<String, Entry> flat = new LinkedHashMap<>();
|
||||
if (fleet != null) {
|
||||
for (MemberRole role : MemberRole.values()) {
|
||||
fleet.pool(role).forEach((name, slot) -> {
|
||||
if (slot != null) {
|
||||
Entry e = new Entry(name, role, slot.profile());
|
||||
flat.put(e.key(), e);
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
this.slots = Collections.unmodifiableMap(flat);
|
||||
}
|
||||
|
||||
/** The configured slots, keyed by qualified {@link Entry#key()}. Unmodifiable snapshot. */
|
||||
public Map<String, Entry> slots() {
|
||||
return slots;
|
||||
}
|
||||
|
||||
/** The slots belonging to {@code role}, in definition order. */
|
||||
public Map<String, Entry> slotsFor(MemberRole role) {
|
||||
Map<String, Entry> out = new LinkedHashMap<>();
|
||||
slots.forEach((key, e) -> {
|
||||
if (e.role() == role) {
|
||||
out.put(key, e);
|
||||
}
|
||||
});
|
||||
return Collections.unmodifiableMap(out);
|
||||
}
|
||||
|
||||
/**
|
||||
* An immutable copy of the live {@code terminal_id → slot name} bindings.
|
||||
*
|
||||
* <p>Passed to {@link CallerResolver} as the source of architect identity, and what
|
||||
* {@code bridge_whoami}/the roster will read to say which slot a pane hosts. Empty until the
|
||||
* spawn lifecycle binds a slot.
|
||||
*/
|
||||
public Map<String, String> snapshot() {
|
||||
synchronized (terminalToSlot) {
|
||||
return Map.copyOf(terminalToSlot);
|
||||
}
|
||||
}
|
||||
|
||||
/** The slot a live terminal is bound to, or {@code null} if it is not an architect slot. */
|
||||
public String slotForTerminal(String terminal) {
|
||||
if (terminal == null) {
|
||||
return null;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
return terminalToSlot.get(terminal);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The strong-model profile a slot runs under — what the spawn lifecycle reads.
|
||||
*
|
||||
* @return the slot's configured {@code profile}, or {@code null} if the slot is unknown or
|
||||
* declares none
|
||||
*/
|
||||
public String profileForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return (e == null || e.profile() == null) ? null : e.profile();
|
||||
}
|
||||
|
||||
/** The role a qualified slot key belongs to, or {@code null} when the key is unknown. */
|
||||
public MemberRole roleForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return e == null ? null : e.role();
|
||||
}
|
||||
|
||||
/** The unqualified configured name for a slot, or {@code null} if it is unknown. */
|
||||
public String nameForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return e == null ? null : e.name();
|
||||
}
|
||||
|
||||
/** True when {@code slotName} is a configured architect slot. */
|
||||
public boolean isSlot(String slotName) {
|
||||
return slots.containsKey(slotName);
|
||||
}
|
||||
|
||||
/**
|
||||
* Bind {@code terminal} to {@code slot} (CB-548).
|
||||
*
|
||||
* <p>The spawn lifecycle calls this when it stands a slot up. The bind is atomic and preserves
|
||||
* the two cardinality invariants: a terminal may occupy at most one slot, and a slot may host at
|
||||
* most one terminal. Binding the same terminal to the same slot again is a harmless no-op.
|
||||
*
|
||||
* @param slot a configured slot name, or the bind is refused
|
||||
* @param terminal the pane that will act as this architect
|
||||
* @return {@code true} if the binding is now {@code terminal → slot}; {@code false} if it was
|
||||
* refused — an unknown slot, a terminal already bound to a different slot, or a slot
|
||||
* already hosting a different terminal
|
||||
*/
|
||||
public boolean bind(String slot, String terminal) {
|
||||
if (slot == null || terminal == null || terminal.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
if (!isSlot(slot)) {
|
||||
return false; // unknown slot — nothing to bind to
|
||||
}
|
||||
String existingSlot = terminalToSlot.get(terminal);
|
||||
if (existingSlot != null) {
|
||||
return slot.equals(existingSlot); // already this slot (idempotent) or a different one
|
||||
}
|
||||
if (terminalToSlot.containsValue(slot)) {
|
||||
return false; // slot already hosts a terminal — no second one
|
||||
}
|
||||
terminalToSlot.put(terminal, slot);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Compare-safe unbind of {@code expectedTerminal} from {@code slot} (CB-548).
|
||||
*
|
||||
* <p>The spawn lifecycle calls this when it tears a slot down. Only the exact binding
|
||||
* {@code expectedTerminal → slot} is removed; if that terminal was since rebound to a different
|
||||
* slot (or the slot to a different terminal), the call is a no-op returning {@code false} — a
|
||||
* stale unbind must never remove a replacement.
|
||||
*
|
||||
* @param slot the slot the caller believes the terminal is bound to
|
||||
* @param expectedTerminal the terminal it expects to be bound there
|
||||
* @return {@code true} if {@code expectedTerminal → slot} was removed; {@code false} if nothing
|
||||
* was (no such binding, or the binding had already moved)
|
||||
*/
|
||||
public boolean unbind(String slot, String expectedTerminal) {
|
||||
if (slot == null || expectedTerminal == null) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
String current = terminalToSlot.get(expectedTerminal);
|
||||
if (current == null || !slot.equals(current)) {
|
||||
return false; // absent, or a replacement/moved binding — leave it in place
|
||||
}
|
||||
terminalToSlot.remove(expectedTerminal);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Bind only architect sessions to a free slot with the resolved profile.
|
||||
*
|
||||
* <p>The role check is lifecycle policy. {@link CallerResolver} repeats it when resolving a
|
||||
* binding, so a later lifecycle regression cannot turn a worker into an architect.
|
||||
*/
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
if (role != MemberRole.ARCHITECT || terminal == null || terminal.isBlank()) {
|
||||
return;
|
||||
}
|
||||
// slotsFor preserves definition order, so duplicate-profile slots use the first free one.
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if (Objects.equals(profile, entry.profile()) && bind(entry.key(), terminal)) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
log.info("member slot: no free architect slot for profile={}; session remains a worker", profile);
|
||||
}
|
||||
|
||||
/** Unbind a released terminal using the compare-safe registry operation. */
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
String slot = slotForTerminal(terminal);
|
||||
if (slot != null) {
|
||||
unbind(slot, terminal);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,127 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
/**
|
||||
* A resolved caller: its {@link Role}, and — for a worker — the herdr {@code terminal_id} that
|
||||
* identifies which worker it is (CB-501).
|
||||
*
|
||||
* @param role what this caller is authorized to act as
|
||||
* @param terminal the herdr {@code terminal_id} of the pane this caller occupies — a worker's, or
|
||||
* (since CB-532) a named lead's; {@code null} for an unnamed primary resolved off a
|
||||
* token or loopback trust, and for {@code ANONYMOUS}
|
||||
* @param pid the connecting process id, or {@code -1} when not resolvable (audit context)
|
||||
* @param name for a lead resolved from the CB-530 {@code leaders:} registry, which lead it is;
|
||||
* for an architect resolved from the CB-548 {@code architects:} registry, which
|
||||
* slot it occupies; {@code null} for every other caller, including an unnamed primary
|
||||
*/
|
||||
public record Principal(Role role, String terminal, long pid, String name) {
|
||||
|
||||
/**
|
||||
* Three-arg form for the callers that have no name to carry (workers, anonymous, and the
|
||||
* token/loopback primary paths). Kept so adding CB-530's {@code name} did not churn every
|
||||
* construction site — and so a reconstructed principal without a stashed name still works.
|
||||
*/
|
||||
public Principal(Role role, String terminal, long pid) {
|
||||
this(role, terminal, pid, null);
|
||||
}
|
||||
|
||||
/** A caller authenticated as nothing — the default when no check establishes anything else. */
|
||||
public static Principal anonymous() {
|
||||
return new Principal(Role.ANONYMOUS, null, -1);
|
||||
}
|
||||
|
||||
/** The orchestrating session, unnamed (token or loopback-trust path). */
|
||||
public static Principal primary(long pid) {
|
||||
return new Principal(Role.PRIMARY, null, pid);
|
||||
}
|
||||
|
||||
/**
|
||||
* A named lead from the {@code leaders:} registry (CB-530).
|
||||
*
|
||||
* <p>Carries {@link Role#PRIMARY}: a lead <em>is</em> a primary as far as authorization goes,
|
||||
* so every existing {@code isPrimary()} gate keeps working unchanged and the role table needed
|
||||
* no new entry. The name is reporting only — it lets {@code bridge_whoami} say <em>which</em>
|
||||
* lead is asking once more than one is configured.
|
||||
*
|
||||
* <p><strong>CB-532: a lead now carries the terminal it was matched by.</strong> Under CB-530 it
|
||||
* deliberately did not, because {@code terminal} meant "which worker pane" everywhere and a
|
||||
* non-null one would have enrolled the lead in the worker presence map. That reading was what
|
||||
* made a lead unaddressable: {@link #ownsSession} could never be true for it, so
|
||||
* {@code bridge_reply} was refused and one lead could send to another but never be answered.
|
||||
* The terminal now means "which pane is this caller", the presence map keys on
|
||||
* {@link #isSpawnedMember()} instead, and a lead is a peer that can both send and receive.
|
||||
*/
|
||||
public static Principal leader(String name, String terminal, long pid) {
|
||||
return new Principal(Role.PRIMARY, terminal, pid, name);
|
||||
}
|
||||
|
||||
/** A worker peer, identified by its herdr pane. */
|
||||
public static Principal worker(String terminal, long pid) {
|
||||
return new Principal(Role.WORKER, terminal, pid);
|
||||
}
|
||||
|
||||
/**
|
||||
* An architect (CB-548), identified by the slot it occupies and the pane bound to it.
|
||||
*
|
||||
* <p>Carries {@link Role#ARCHITECT}. {@code slotName} is reporting only — it lets
|
||||
* {@code bridge_whoami} say <em>which</em> architect slot is asking, and it is the key the
|
||||
* (future) spawn lifecycle reads a profile back from. Identity is the {@code terminal}: like a
|
||||
* worker's it comes from the connection and the live terminal→slot binding, so
|
||||
* {@code ownsSession} works exactly as it does for a worker — an architect acts as its own
|
||||
* pane and no other.
|
||||
*/
|
||||
public static Principal architect(String slotName, String terminal, long pid) {
|
||||
return new Principal(Role.ARCHITECT, terminal, pid, slotName);
|
||||
}
|
||||
|
||||
public boolean isPrimary() {
|
||||
return role == Role.PRIMARY;
|
||||
}
|
||||
|
||||
public boolean isArchitect() {
|
||||
return role == Role.ARCHITECT;
|
||||
}
|
||||
|
||||
public boolean isWorker() {
|
||||
return role == Role.WORKER;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether this caller is a spawned member with its own pane.
|
||||
*
|
||||
* <p>Both workers and architects are spawned members. A lead is excluded because recording it
|
||||
* as present would count it as an available member in the roster.
|
||||
*/
|
||||
public boolean isSpawnedMember() {
|
||||
return role == Role.WORKER || role == Role.ARCHITECT;
|
||||
}
|
||||
|
||||
public boolean isAnonymous() {
|
||||
return role == Role.ANONYMOUS;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether this caller may act <em>as</em> {@code sessionId} — the "own session only" rule that
|
||||
* keeps one peer from replying or asking on another's behalf.
|
||||
*
|
||||
* <p>The rule is about <em>identity</em>, not rank: a caller may act as the pane it demonstrably
|
||||
* occupies, and as no other. CB-532 dropped the extra {@code isWorker()} conjunct that used to
|
||||
* be here. It was not what enforced the rule — {@code terminal.equals(sessionId)} is, and that
|
||||
* terminal comes from the connection, so it cannot be forged either way. All the conjunct did
|
||||
* was make a lead permanently unable to answer anyone, since a lead's terminal was null and a
|
||||
* lead is not a worker. A caller with no terminal at all (an off-host or token-authenticated
|
||||
* primary) still owns nothing, which is the case the null check covers.
|
||||
*/
|
||||
public boolean ownsSession(String sessionId) {
|
||||
return terminal != null && terminal.equals(sessionId);
|
||||
}
|
||||
|
||||
/** Short, non-sensitive description for audit lines and error details. */
|
||||
public String describe() {
|
||||
return switch (role) {
|
||||
case WORKER -> "worker:" + terminal;
|
||||
case ARCHITECT -> "architect:" + name;
|
||||
case PRIMARY -> name == null ? "primary" : "leader:" + name;
|
||||
case ANONYMOUS -> "anonymous";
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
/**
|
||||
* What a caller is allowed to be on the bus (CB-501).
|
||||
*
|
||||
* <p>The ordering matters conceptually: {@link #PRIMARY} is the <em>most</em> privileged role
|
||||
* (it spawns, stops, sends to any session, and drains any inbox), not the least. Before CB-501
|
||||
* the daemon reached {@code PRIMARY} by <em>failing</em> every other check — any caller that did
|
||||
* not resolve to a known worker pane was treated as the primary. That is inverted here:
|
||||
* {@link #ANONYMOUS} is the fallback, and {@code PRIMARY} must be established.
|
||||
*/
|
||||
public enum Role {
|
||||
|
||||
/**
|
||||
* The orchestrating session. Established either by being a loopback caller that is not a
|
||||
* worker pane (under {@code loopback-trust}) or by presenting a valid bearer token (under
|
||||
* {@code token} mode).
|
||||
*/
|
||||
PRIMARY,
|
||||
|
||||
/**
|
||||
* A worker peer, identified by its herdr pane. Unforgeable: derived from the connection's
|
||||
* loopback peer PID via herdr's PID→pane map, never from a request argument.
|
||||
*/
|
||||
WORKER,
|
||||
|
||||
/**
|
||||
* A config-declared architect slot (CB-548): a gateway-local named session on a strong-model
|
||||
* profile that coordinates and delegates turns but does not own the fleet. Unforgeable like a
|
||||
* worker's — derived from the connection's pane and the live terminal→slot binding, never from
|
||||
* a request argument. May {@code SEND} a turn, {@code REPLY}/{@code ASK} only as its own pane,
|
||||
* and {@code READ}/{@code METRICS}; may <em>not</em> {@code SPAWN}/{@code STOP}/{@code DRAIN}
|
||||
* (those stay the primary's, to keep lifecycle in one pair of hands).
|
||||
*/
|
||||
ARCHITECT,
|
||||
|
||||
/** Authenticated as nothing. Authorized for nothing but {@code /healthz}. */
|
||||
ANONYMOUS
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,291 @@
|
||||
package dev.ltms.bridged.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The daemon's live configuration, re-readable without a restart (CB-559).
|
||||
*
|
||||
* <p>Consumers hold this, not a {@link BridgedConfig}, and read through {@link #get()} at the point
|
||||
* of use. A component that captures {@code ref.get()} into a field at construction has opted out of
|
||||
* reload — which is sometimes right (see <em>deferred</em> below), but it must then be a deliberate
|
||||
* choice rather than an accident of where the field was initialised.
|
||||
*
|
||||
* <h2>Not every key can change under a running daemon</h2>
|
||||
* Keys fall into three classes, and the difference is about what already exists when the reload
|
||||
* happens — not about how important the key is.
|
||||
*
|
||||
* <ul>
|
||||
* <li><strong>Hot</strong> — re-read per use, so a reload takes effect on the next spawn:
|
||||
* {@code fleet:} (every role pool, {@code charters}, and {@code tabLabel}),
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Those
|
||||
* three are read through a supplier on {@code CompositePeerLauncher}, which is what makes
|
||||
* them hot — not the fact that they are config. <strong>This does NOT include
|
||||
* {@code fleet.leaders}</strong>: {@code Bridged.main} reads {@code cfg.fleet().leaders()}
|
||||
* once at startup to build the {@code LeadTabScanner} and the {@code LeadLauncher}, and
|
||||
* neither is reconstructed on reload — so a lead added, removed, or re-{@code tab}'d under
|
||||
* {@code fleet.leaders} needs a restart, the same as any deferred key below.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code quarantineCooldownSeconds}
|
||||
* (CB-578 stage B — baked once into the {@code BackendQuarantine} built at startup),
|
||||
* {@code guard:}, {@code worktreeRoot:}, adding or removing a profile (a new backend needs its own launcher,
|
||||
* which is constructed once), <em>and an existing profile's launch settings</em> —
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl},
|
||||
* {@code exhaustedPattern} (CB-578 stage A — compiled once into {@code Bridged.main}'s
|
||||
* pattern map at startup), and the rest. {@code credentialId} (CB-578 stage B) is NOT on
|
||||
* this list — it is read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so it is hot instead.
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} at construction and resolves
|
||||
* each spawn out of that copy, so those never reach a launch until the daemon restarts. A
|
||||
* reload logs these rather than pretending they applied.</li>
|
||||
* <li><strong>Cold</strong> — cannot change at all under a running daemon: {@code bind:},
|
||||
* {@code herdrSocket:}, {@code broker:} and {@code auth:}. The socket is bound, the broker
|
||||
* connection is open, and the auth mode decides who may reach the port that is already
|
||||
* listening.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>A cold change refuses the whole reload.</strong> Not the hot half applied and the cold
|
||||
* half warned about: that would leave the running daemon in a state matching no file on disk, which
|
||||
* is the worst thing a reload can do to an operator debugging one. Refusing keeps the invariant that
|
||||
* the live config is always some version of the file, and the message names the keys that must
|
||||
* change through a restart.
|
||||
*
|
||||
* <p>A reload that fails to parse or fails validation is also refused, and the previous config keeps
|
||||
* running. A config file being edited is normally read once mid-save; degrading a working daemon
|
||||
* because it caught a half-written file would be a bad trade.
|
||||
*/
|
||||
public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigRef.class);
|
||||
|
||||
/** Keys that cannot change under a running daemon — see the class doc. */
|
||||
private static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "broker", "auth");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<BridgedConfig> current;
|
||||
|
||||
public ConfigRef(Path path, BridgedConfig initial) {
|
||||
this.path = path;
|
||||
this.current = new AtomicReference<>(Objects.requireNonNull(initial, "initial config"));
|
||||
}
|
||||
|
||||
/** A fixed reference that never reloads — for tests and for wiring built from a config in code. */
|
||||
public static ConfigRef fixed(BridgedConfig cfg) {
|
||||
return new ConfigRef(null, cfg);
|
||||
}
|
||||
|
||||
/** The live configuration. Read this per use; do not cache it in a field. */
|
||||
@Override
|
||||
public BridgedConfig get() {
|
||||
return current.get();
|
||||
}
|
||||
|
||||
/** The file this ref reloads from, or {@code null} for a {@link #fixed} ref. */
|
||||
public Path path() {
|
||||
return path;
|
||||
}
|
||||
|
||||
/**
|
||||
* What a reload attempt did.
|
||||
*
|
||||
* @param applied true when the new config is now live
|
||||
* @param coldKeys cold keys whose value changed, which is why an unapplied reload was refused
|
||||
* @param deferred keys that changed and were accepted, but whose effect waits for a restart
|
||||
* @param error the parse or validation failure that refused the reload, else {@code null}
|
||||
*/
|
||||
public record Outcome(boolean applied, List<String> coldKeys, List<String> deferred,
|
||||
String error) {
|
||||
|
||||
public Outcome {
|
||||
coldKeys = List.copyOf(coldKeys);
|
||||
deferred = List.copyOf(deferred);
|
||||
}
|
||||
|
||||
static Outcome refusedCold(List<String> keys) {
|
||||
return new Outcome(false, keys, List.of(), null);
|
||||
}
|
||||
|
||||
static Outcome failed(String error) {
|
||||
return new Outcome(false, List.of(), List.of(), error);
|
||||
}
|
||||
|
||||
/** A one-line summary for the operator — the reason, not just the verdict. */
|
||||
public String summary() {
|
||||
if (error != null) {
|
||||
return "config reload refused — " + error;
|
||||
}
|
||||
if (!applied) {
|
||||
return "config reload refused — these keys cannot change under a running daemon: "
|
||||
+ String.join(", ", coldKeys) + ". Restart bridged to apply them.";
|
||||
}
|
||||
if (!deferred.isEmpty()) {
|
||||
return "config reloaded; these changes need a restart to take effect: "
|
||||
+ String.join(", ", deferred);
|
||||
}
|
||||
return "config reloaded";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-read the file, validate it, and swap it in when nothing cold changed.
|
||||
*
|
||||
* <p>Never throws: a reload is a best-effort operation on a daemon that is already serving, and
|
||||
* a bad edit must not take it down. Every failure path leaves the previous config live and is
|
||||
* reported through the returned {@link Outcome}.
|
||||
*/
|
||||
public Outcome reload() {
|
||||
if (path == null) {
|
||||
return Outcome.failed("this config was built in code and has no file to reload from");
|
||||
}
|
||||
BridgedConfig old = current.get();
|
||||
BridgedConfig fresh;
|
||||
try {
|
||||
fresh = BridgedConfig.load(path);
|
||||
// The same gate startup runs. A config that would have refused to boot must not be able
|
||||
// to slip in through a reload — that is how a daemon ends up in a state it could never
|
||||
// have started in, which is the hardest kind to debug.
|
||||
fresh.validateAuthExposure();
|
||||
fresh.validateLeadTabPrefixes();
|
||||
fresh.validateSubscriptionProfiles();
|
||||
fresh.validateCharters();
|
||||
fresh.validateMembers();
|
||||
} catch (RuntimeException e) {
|
||||
String msg = e.getMessage() == null ? e.toString() : e.getMessage();
|
||||
log.warn("config reload from {} refused, keeping the running config: {}", path, msg);
|
||||
return Outcome.failed(msg);
|
||||
}
|
||||
|
||||
List<String> cold = changedColdKeys(old, fresh);
|
||||
if (!cold.isEmpty()) {
|
||||
Outcome out = Outcome.refusedCold(cold);
|
||||
log.warn(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
List<String> deferred = changedDeferredKeys(old, fresh);
|
||||
current.set(fresh);
|
||||
Outcome out = new Outcome(true, List.of(), deferred, null);
|
||||
log.info(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Cold keys whose value differs between the running config and the candidate. */
|
||||
private static List<String> changedColdKeys(BridgedConfig old, BridgedConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.bind(), fresh.bind())) {
|
||||
changed.add("bind");
|
||||
}
|
||||
if (!Objects.equals(old.herdrSocket(), fresh.herdrSocket())) {
|
||||
changed.add("herdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.broker(), fresh.broker())) {
|
||||
changed.add("broker");
|
||||
}
|
||||
if (!Objects.equals(old.auth(), fresh.auth())) {
|
||||
changed.add("auth");
|
||||
}
|
||||
// Kept in step with COLD_KEYS so the doc and the code cannot drift apart silently.
|
||||
assert COLD_KEYS.containsAll(changed) : "a cold key was reported that COLD_KEYS omits";
|
||||
return changed;
|
||||
}
|
||||
|
||||
/** Changed keys that were accepted but whose effect waits for a restart. */
|
||||
private static List<String> changedDeferredKeys(BridgedConfig old, BridgedConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.lifecycle(), fresh.lifecycle())) {
|
||||
changed.add("lifecycle");
|
||||
}
|
||||
if (!Objects.equals(old.leadHeartbeat(), fresh.leadHeartbeat())) {
|
||||
changed.add("leadHeartbeat");
|
||||
}
|
||||
if (!Objects.equals(old.guard(), fresh.guard())) {
|
||||
changed.add("guard");
|
||||
}
|
||||
if (!Objects.equals(old.worktreeRoot(), fresh.worktreeRoot())) {
|
||||
changed.add("worktreeRoot");
|
||||
}
|
||||
if (!Objects.equals(old.spawnReadyTimeoutMs(), fresh.spawnReadyTimeoutMs())
|
||||
|| !Objects.equals(old.spawnReadyPollMs(), fresh.spawnReadyPollMs())) {
|
||||
changed.add("spawnReady*");
|
||||
}
|
||||
// CB-578 stage B: baked once into the BackendQuarantine built at startup — a running
|
||||
// quarantine keeps its original cooldown regardless, and a new cooldown only applies to a
|
||||
// quarantine that starts after a restart.
|
||||
if (!Objects.equals(old.quarantineCooldownSeconds(), fresh.quarantineCooldownSeconds())) {
|
||||
changed.add("quarantineCooldownSeconds");
|
||||
}
|
||||
Map<String, BridgedConfig.Profile> before =
|
||||
old.profiles() == null ? Map.of() : old.profiles();
|
||||
Map<String, BridgedConfig.Profile> after =
|
||||
fresh.profiles() == null ? Map.of() : fresh.profiles();
|
||||
// Adding or removing a profile is deferred: a new backend needs its own launcher, and
|
||||
// launchers are built once at startup.
|
||||
if (!before.keySet().equals(after.keySet())) {
|
||||
Set<String> diff = new LinkedHashSet<>(before.keySet());
|
||||
diff.addAll(after.keySet());
|
||||
diff.removeIf(p -> before.containsKey(p) && after.containsKey(p));
|
||||
changed.add("profiles (added/removed: " + String.join(", ", diff) + ")");
|
||||
}
|
||||
// An EXISTING profile's launch settings are deferred too, and this is easy to get wrong:
|
||||
// `HerdrPeerLauncher` takes `Map.copyOf(profiles)` at construction and `spawn` resolves the
|
||||
// profile out of that snapshot, so a reloaded model/baseUrl/argv/env never reaches a launch.
|
||||
// Only weight and maxLoad are genuinely hot, because placement reads them through the
|
||||
// supplier on the composite rather than from the adapter's copy. Without this check a
|
||||
// changed model would report "config reloaded" and silently do nothing — the worst outcome
|
||||
// a reload can produce, because the operator has no reason to doubt it.
|
||||
List<String> relaunch = new ArrayList<>();
|
||||
before.forEach((name, was) -> {
|
||||
BridgedConfig.Profile now = after.get(name);
|
||||
if (now != null && !sameLaunchSettings(was, now)) {
|
||||
relaunch.add(name);
|
||||
}
|
||||
});
|
||||
if (!relaunch.isEmpty()) {
|
||||
changed.add("profiles." + String.join("/", relaunch) + " launch settings "
|
||||
+ "(model, baseUrl, argv, env, …) — the launcher holds a startup snapshot");
|
||||
}
|
||||
return changed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically. Compares every component
|
||||
* the launcher reads at spawn; {@code weight}, {@code maxLoad} and {@code credentialId} are
|
||||
* excluded because those are read live (by the placement policy and, for credentialId, by
|
||||
* {@code CompositePeerLauncher}/the CB-578 stage B exhaustion sink) and really do take effect on
|
||||
* the next spawn.
|
||||
*/
|
||||
private static boolean sameLaunchSettings(BridgedConfig.Profile a, BridgedConfig.Profile b) {
|
||||
return Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
&& Objects.equals(a.model(), b.model())
|
||||
&& Objects.equals(a.configDir(), b.configDir())
|
||||
&& Objects.equals(a.tokenEnv(), b.tokenEnv())
|
||||
&& Objects.equals(a.argv(), b.argv())
|
||||
&& Objects.equals(a.placement(), b.placement())
|
||||
&& Objects.equals(a.workspace(), b.workspace())
|
||||
&& Objects.equals(a.tabLabel(), b.tabLabel())
|
||||
&& Objects.equals(a.mcpUrl(), b.mcpUrl())
|
||||
&& Objects.equals(a.cwd(), b.cwd())
|
||||
&& Objects.equals(a.parityOverlay(), b.parityOverlay())
|
||||
&& Objects.equals(a.gitTokenEnv(), b.gitTokenEnv())
|
||||
&& Objects.equals(a.gitHostEnv(), b.gitHostEnv())
|
||||
&& Objects.equals(a.kind(), b.kind())
|
||||
&& Objects.equals(a.env(), b.env())
|
||||
&& Objects.equals(a.subscription(), b.subscription())
|
||||
// CB-578 stage B: exhaustedPattern is compiled once into Bridged.main's pattern map
|
||||
// at startup (see ExhaustedPatternLookup wiring) — a reload never re-reads it, so a
|
||||
// changed pattern must be reported as deferred, exactly like model/baseUrl/argv.
|
||||
&& Objects.equals(a.exhaustedPattern(), b.exhaustedPattern());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
package dev.ltms.bridged.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Polls {@code bridged.yaml}'s modified time and asks {@link ConfigRef} to reload when it moves
|
||||
* (CB-559). Opt-in through {@code configReload.enabled}.
|
||||
*
|
||||
* <p><strong>Why polling and not a filesystem watch.</strong> {@code WatchService} on macOS has no
|
||||
* native backend — it falls back to polling internally anyway, at an interval this code does not
|
||||
* control — and editors save config files in ways that produce a different event mix per editor
|
||||
* (write-in-place, write-and-rename, write-temp-and-swap). A modified-time check treats all of them
|
||||
* the same and is a single {@code stat} per tick, which at a ten-second cadence costs nothing worth
|
||||
* measuring.
|
||||
*
|
||||
* <p><strong>A missing or unreadable file is not a reason to act.</strong> Many editors briefly
|
||||
* unlink the file during a save. Reloading on "it vanished" would mean reloading from a file that no
|
||||
* longer exists; reporting an error every tick would bury the log. So an unreadable file is skipped
|
||||
* silently and the next tick tries again — the running config stays live, which is the correct
|
||||
* outcome either way.
|
||||
*/
|
||||
public final class ConfigWatcher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigWatcher.class);
|
||||
|
||||
private final ConfigRef ref;
|
||||
private final long intervalSeconds;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
|
||||
private volatile long lastSeenMillis;
|
||||
|
||||
public ConfigWatcher(ConfigRef ref, long intervalSeconds) {
|
||||
this.ref = ref;
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.lastSeenMillis = modifiedMillis(ref.path());
|
||||
this.scheduler = Executors.newSingleThreadScheduledExecutor(r -> {
|
||||
Thread t = new Thread(r, "config-watcher");
|
||||
// A daemon thread: an operator's config watch must never be the reason the JVM refuses
|
||||
// to exit after everything else has shut down.
|
||||
t.setDaemon(true);
|
||||
return t;
|
||||
});
|
||||
}
|
||||
|
||||
/** Begin watching. A ref with no file (a fixed one) is a no-op rather than an error. */
|
||||
public void start() {
|
||||
if (ref.path() == null) {
|
||||
log.debug("config watch not started — this config has no file behind it");
|
||||
return;
|
||||
}
|
||||
scheduler.scheduleWithFixedDelay(this::tick, intervalSeconds, intervalSeconds,
|
||||
TimeUnit.SECONDS);
|
||||
log.info("config watch: {} re-read when it changes (every {}s)", ref.path(), intervalSeconds);
|
||||
}
|
||||
|
||||
/** One poll. Never throws — an exception here would silently cancel the schedule. */
|
||||
void tick() {
|
||||
try {
|
||||
long now = modifiedMillis(ref.path());
|
||||
if (now == 0 || now == lastSeenMillis) {
|
||||
return;
|
||||
}
|
||||
// Stamp BEFORE reloading. A file whose reload is refused (a bad edit, or a cold key)
|
||||
// must not be retried every tick — that would log the same refusal forever. The next
|
||||
// save moves the timestamp again and earns a fresh attempt.
|
||||
lastSeenMillis = now;
|
||||
ref.reload();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("config watch tick failed, still watching: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
private static long modifiedMillis(Path path) {
|
||||
if (path == null) {
|
||||
return 0;
|
||||
}
|
||||
try {
|
||||
return Files.getLastModifiedTime(path).toMillis();
|
||||
} catch (IOException e) {
|
||||
return 0; // mid-save, or gone: say nothing and try again next tick
|
||||
}
|
||||
}
|
||||
|
||||
/** Stop polling. Called from the daemon's ordered shutdown hook, alongside the other loops. */
|
||||
public void stop() {
|
||||
scheduler.shutdownNow();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,40 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
|
||||
/**
|
||||
* Pure classifier. Collection and repair are deliberately outside this package.
|
||||
* {@link HealthState#ERROR_ON_SCREEN} is not decided yet because it needs a bounded pane detection
|
||||
* read and an adapter-specific fatal signature; status facts alone must not guess it.
|
||||
*/
|
||||
public final class FleetHealth {
|
||||
private FleetHealth() { }
|
||||
|
||||
public static HealthDecision decide(HealthSnapshot s, HealthPrior prior, long nowNanos) {
|
||||
if (s.controlLinkDown()) return result(HealthState.CONTROL_LINK_DOWN, false);
|
||||
if (s.targetNotFound()) return result(HealthState.GONE, false);
|
||||
if (s.sessionState() == MemberSession.State.SPAWNING && !s.present() && s.readinessGraceElapsed()) {
|
||||
return result(HealthState.NEVER_READY, false);
|
||||
}
|
||||
if (s.orphanedDelegation()) return result(HealthState.DELEGATION_ORPHANED, false);
|
||||
boolean disagreement = s.sessionState() == MemberSession.State.BUSY && s.acceptedDelivery()
|
||||
&& (s.liveStatus() == AgentStatus.IDLE || s.liveStatus() == AgentStatus.DONE);
|
||||
if (disagreement && prior.busyButDone()) return result(HealthState.TURN_BOUNDARY_LOST, true);
|
||||
if (s.stalled()) return result(HealthState.STALL_SUSPECTED, disagreement);
|
||||
if (s.replyStranded()) return result(HealthState.REPLY_STRANDED, disagreement);
|
||||
if (s.queuedDelivery() || s.inboxMessage()) return result(HealthState.WORK_PENDING, disagreement);
|
||||
if (s.sessionState() == MemberSession.State.SPAWNING) return result(HealthState.STARTING, disagreement);
|
||||
if (s.acceptedDelivery() && s.liveStatus() == AgentStatus.BLOCKED) {
|
||||
return result(HealthState.BLOCKED_AMBIGUOUS, disagreement);
|
||||
}
|
||||
// An accepted delivery remains bridge work even when herdr is late, unknown, or has already
|
||||
// reported DONE once. It cannot be IDLE until the delegation has resolved.
|
||||
if (s.acceptedDelivery()) return result(HealthState.WORKING, disagreement);
|
||||
return result(HealthState.IDLE, disagreement);
|
||||
}
|
||||
|
||||
private static HealthDecision result(HealthState state, boolean disagreement) {
|
||||
return new HealthDecision(state, new HealthPrior(disagreement));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,151 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/** Slow whole-fleet evidence collection. It is deliberately separate from the delivery poller. */
|
||||
public final class FleetHealthMonitor {
|
||||
private static final Logger log = LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
|
||||
/** Bounded attempts to run {@link #failTarget} for one transition. Never retried tick-to-tick (CB-580). */
|
||||
static final int MAX_FAIL_TARGET_ATTEMPTS = 3;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final MessageService messages;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final long intervalSeconds;
|
||||
private final BiConsumer<String, String> failTarget;
|
||||
private final Map<String, HealthPrior> priors = new HashMap<>();
|
||||
private final Map<String, HealthState> states = new HashMap<>();
|
||||
|
||||
// These facts need the evidence publishers introduced by later M4 units. They are not negatives.
|
||||
private static final boolean NOT_YET_OBSERVED = false;
|
||||
|
||||
/**
|
||||
* @param failTarget CB-568's idempotent target-wide failure operation (e.g. {@code messages::abandon}),
|
||||
* invoked once when a member transitions into a terminal health state. Required —
|
||||
* there is deliberately no defaulting overload; a caller that does not want the
|
||||
* fail-tickets-on-terminal-health behavior must pass an explicit inert value (see
|
||||
* {@code TestTurnTokens.inert} / {@code BridgeMcp.CapacitySource.none()} for the pattern).
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, long intervalSeconds,
|
||||
BiConsumer<String, String> failTarget) {
|
||||
this.agents = agents;
|
||||
this.roster = roster;
|
||||
this.messages = messages;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.failTarget = Objects.requireNonNull(failTarget, "failTarget");
|
||||
}
|
||||
|
||||
/** Pure per-member decision seam. */
|
||||
static HealthDecision decide(HealthSnapshot snapshot, HealthPrior prior, long nowNanos) {
|
||||
return FleetHealth.decide(snapshot, prior, nowNanos);
|
||||
}
|
||||
|
||||
public void start() { scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS); }
|
||||
public void stop() { scheduler.shutdownNow(); }
|
||||
|
||||
// Package-private so tests can run one tick without waiting.
|
||||
void tick() {
|
||||
try {
|
||||
List<Agent> agentsNow = agents.list(); // Exactly one list call for this complete observation.
|
||||
List<MemberSession> rosterNow = roster.get(); // One in-memory roster snapshot for this tick.
|
||||
Map<String, Agent> live = new HashMap<>();
|
||||
for (Agent agent : agentsNow) live.put(agent.terminalId(), agent);
|
||||
HashSet<String> current = new HashSet<>();
|
||||
for (MemberSession session : rosterNow) {
|
||||
current.add(session.terminalId());
|
||||
Agent agent = live.get(session.terminalId());
|
||||
AgentStatus status = agent == null ? AgentStatus.UNKNOWN : agent.status();
|
||||
boolean accepted = messages.hasAcceptedDelivery(session.terminalId());
|
||||
HealthSnapshot snapshot = new HealthSnapshot(session.state(), status, accepted, NOT_YET_OBSERVED,
|
||||
messages.hasInboxMessage(session.terminalId()), agent != null, NOT_YET_OBSERVED,
|
||||
NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED);
|
||||
HealthDecision decision = decide(snapshot, priors.getOrDefault(session.terminalId(), HealthPrior.NONE),
|
||||
clock.getAsLong());
|
||||
priors.put(session.terminalId(), decision.prior());
|
||||
reportTransition(session.terminalId(), decision.state());
|
||||
}
|
||||
priors.keySet().retainAll(current);
|
||||
states.keySet().retainAll(current);
|
||||
} catch (Throwable error) {
|
||||
// A list failure is health evidence, and must never kill the monitor's only scheduler task.
|
||||
log.warn("fleet health collection failed; will retry next tick", error);
|
||||
} finally {
|
||||
if (!scheduler.isShutdown()) {
|
||||
scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void reportTransition(String target, HealthState next) {
|
||||
HealthState previous = states.put(target, next);
|
||||
if (previous == next) return;
|
||||
if (fault(next)) {
|
||||
log.warn("fleet health member={} state={} previous={}", target, next, previous);
|
||||
} else if (previous != null && fault(previous)) {
|
||||
log.info("fleet health member={} recovered state={} previous={}", target, next, previous);
|
||||
}
|
||||
// CB-580: a member entering GONE/NEVER_READY must not leave its waiting tickets pending
|
||||
// forever. Fire exactly once per transition — never on a tick where the state is unchanged,
|
||||
// which is what made the rejected commit call abandon() once per tick for as long as a
|
||||
// member stayed terminal.
|
||||
if (terminal(next)) {
|
||||
failTerminalTarget(target, next);
|
||||
}
|
||||
}
|
||||
|
||||
private void failTerminalTarget(String target, HealthState state) {
|
||||
String reason = "fleet health: member reached terminal state " + state.name();
|
||||
RuntimeException last = null;
|
||||
for (int attempt = 1; attempt <= MAX_FAIL_TARGET_ATTEMPTS; attempt++) {
|
||||
try {
|
||||
failTarget.accept(target, reason);
|
||||
return;
|
||||
} catch (RuntimeException error) {
|
||||
last = error;
|
||||
log.warn("fleet health: failTarget attempt {}/{} failed for member={} state={}",
|
||||
attempt, MAX_FAIL_TARGET_ATTEMPTS, target, state, error);
|
||||
}
|
||||
}
|
||||
log.warn("fleet health: giving up on failTarget for member={} state={} after {} attempts",
|
||||
target, state, MAX_FAIL_TARGET_ATTEMPTS, last);
|
||||
}
|
||||
|
||||
private static boolean terminal(HealthState state) {
|
||||
return state == HealthState.GONE || state == HealthState.NEVER_READY;
|
||||
}
|
||||
|
||||
private static boolean fault(HealthState state) {
|
||||
return switch (state) {
|
||||
case NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN -> true;
|
||||
default -> false;
|
||||
};
|
||||
}
|
||||
|
||||
public static String coverage(boolean enabled, boolean notificationConfigured) {
|
||||
return !enabled ? "off" : notificationConfigured ? "full" : "detection-only";
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,4 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
/** Classification plus the private fact that the next pure decision needs. */
|
||||
public record HealthDecision(HealthState state, HealthPrior prior) { }
|
||||
@@ -0,0 +1,6 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
/** Private cross-tick observation. It is deliberately not a reported health value. */
|
||||
public record HealthPrior(boolean busyButDone) {
|
||||
public static final HealthPrior NONE = new HealthPrior(false);
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
|
||||
/** Read-only facts from one fleet collection tick. */
|
||||
public record HealthSnapshot(MemberSession.State sessionState, AgentStatus liveStatus,
|
||||
boolean acceptedDelivery, boolean queuedDelivery, boolean inboxMessage,
|
||||
boolean present, boolean targetNotFound, boolean controlLinkDown,
|
||||
boolean readinessGraceElapsed, boolean orphanedDelegation,
|
||||
boolean replyStranded, boolean stalled) { }
|
||||
@@ -0,0 +1,8 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
/** Health classifications reported for a member. */
|
||||
public enum HealthState {
|
||||
STARTING, IDLE, WORKING, WORK_PENDING, BLOCKED_AMBIGUOUS,
|
||||
NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* Counts turns that ended via the completion fallback instead of {@code bridge_reply}.
|
||||
* MUTE is an observation by target and profile, not a classifier state and never suppresses faults.
|
||||
*/
|
||||
public final class MuteCounter {
|
||||
private final Map<String, Integer> byTarget = new ConcurrentHashMap<>();
|
||||
private final Map<String, Integer> byProfile = new ConcurrentHashMap<>();
|
||||
|
||||
/** Record only fallback completion; a structured reply does not make a member mute. */
|
||||
public void observe(String target, String profile, Rendezvous.Kind kind) {
|
||||
if (kind != Rendezvous.Kind.COMPLETION) return;
|
||||
byTarget.merge(target, 1, Integer::sum);
|
||||
byProfile.merge(profile, 1, Integer::sum);
|
||||
}
|
||||
|
||||
public int forTarget(String target) { return byTarget.getOrDefault(target, 0); }
|
||||
public int forProfile(String profile) { return byProfile.getOrDefault(profile, 0); }
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/** Fixed pane-probe limits. Pane content is never retained here. */
|
||||
public final class PaneBudget {
|
||||
public static final long COOLDOWN_NANOS = 60_000_000_000L;
|
||||
public static final int MAX_PER_TICK = 2;
|
||||
private final Map<String, Long> lastProbe = new HashMap<>();
|
||||
private int cursor;
|
||||
|
||||
public List<String> choose(List<String> candidates, long nowNanos, long configuredCooldownNanos) {
|
||||
long cooldown = Math.max(COOLDOWN_NANOS, configuredCooldownNanos);
|
||||
List<String> out = new ArrayList<>();
|
||||
for (int n = 0; n < candidates.size() && out.size() < MAX_PER_TICK; n++) {
|
||||
String target = candidates.get((cursor + n) % candidates.size());
|
||||
Long last = lastProbe.get(target);
|
||||
if (last == null || nowNanos - last >= cooldown) { out.add(target); lastProbe.put(target, nowNanos); }
|
||||
}
|
||||
if (!candidates.isEmpty()) cursor = (cursor + 1) % candidates.size();
|
||||
return List.copyOf(out);
|
||||
}
|
||||
}
|
||||
@@ -6,93 +6,120 @@ import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* Domain layer over herdr's native {@code agent.*} namespace — the worker south side.
|
||||
* Chosen in the CB-102 spike over the pane + {@code send_text} fallback because
|
||||
* {@code agent.start} takes a first-class {@code env} map (clean, guard-checked
|
||||
* subscription injection) and herdr tracks each worker's Claude session UUID itself.
|
||||
* Chosen in the CB-102 spike over the pane + {@code send_text} fallback because herdr
|
||||
* tracks each worker's Claude session UUID itself.
|
||||
*
|
||||
* <p>Ported to herdr protocol 19 (herdr 0.8.0, CB-521): {@code agent.start} now starts a
|
||||
* <em>supported</em> agent ({@code kind}) into an <em>existing</em> pane, so the worker's
|
||||
* {@code env}/{@code cwd} move to pane creation ({@code tab.create}/{@code pane.split} — see
|
||||
* {@link WorkspaceControl}), and {@code agent.send} is replaced by {@code agent.prompt}
|
||||
* (which submits in one call) plus {@code agent.send_keys} for the raw Enter nudge.
|
||||
*
|
||||
* <p>Every method is one herdr call through the injected {@link HerdrClient}, so this
|
||||
* layer is unit-testable with a fake and contract-tested against a live daemon.
|
||||
*/
|
||||
public final class AgentControl {
|
||||
|
||||
/**
|
||||
* The keystroke that submits a prompt in the Claude Code TUI: a carriage return (Enter).
|
||||
* It must be delivered as its <em>own</em> {@code agent.send} call — herdr delivers a message's
|
||||
* text as a bracketed paste, and a {@code "\r"} appended to that same text is swallowed as
|
||||
* literal newline content, not a submit. Sent as a separate keystroke event it lands outside
|
||||
* the paste and submits. (A bare {@code "\n"} inserts a newline either way.) Verified live
|
||||
* against Claude Code v2.1.210: an injected task stayed unsubmitted with {@code "text\r"} in
|
||||
* one call, and submitted the instant a standalone {@code "\r"} was sent.
|
||||
*/
|
||||
static final String SUBMIT_KEY = "\r";
|
||||
|
||||
private final HerdrClient herdr;
|
||||
|
||||
/**
|
||||
* Protocol 19 dropped {@code terminal_id} as an {@code agent.*} target — herdr now resolves
|
||||
* targets by pane id or agent name only, while the bridge keys every session on the terminal.
|
||||
* This caches the terminal→pane mapping (stable for a worker's lifetime) so callers keep
|
||||
* addressing agents by terminal; entries are invalidated on {@code agent_not_found}.
|
||||
*/
|
||||
private final Map<String, String> paneByTerminal = new ConcurrentHashMap<>();
|
||||
|
||||
public AgentControl(HerdrClient herdr) {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
/** One agent-targeted call, translating a terminal id to its pane id (retrying once fresh). */
|
||||
private JsonNode agentCall(String method, String target, Map<String, Object> extra) {
|
||||
String resolved = resolveTarget(target);
|
||||
try {
|
||||
return herdr.call(method, withTarget(resolved, extra));
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_not_found".equals(e.code()) || resolved.equals(target)) throw e;
|
||||
paneByTerminal.remove(target); // the cached pane went away — re-resolve once
|
||||
String fresh = resolveTarget(target);
|
||||
if (fresh.equals(resolved)) throw e;
|
||||
return herdr.call(method, withTarget(fresh, extra));
|
||||
}
|
||||
}
|
||||
|
||||
private static Map<String, Object> withTarget(String target, Map<String, Object> extra) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("target", target);
|
||||
m.putAll(extra);
|
||||
return m;
|
||||
}
|
||||
|
||||
/** The pane id behind a terminal-id target, or the target verbatim for pane ids / names. */
|
||||
private String resolveTarget(String target) {
|
||||
if (target == null || !target.startsWith("term_")) {
|
||||
return target;
|
||||
}
|
||||
String cached = paneByTerminal.get(target);
|
||||
if (cached != null) {
|
||||
return cached;
|
||||
}
|
||||
for (JsonNode a : herdr.call("agent.list").path("agents")) {
|
||||
if (target.equals(a.path("terminal_id").asText(null))) {
|
||||
String pane = a.path("pane_id").asText(null);
|
||||
if (pane != null) {
|
||||
paneByTerminal.put(target, pane);
|
||||
return pane;
|
||||
}
|
||||
}
|
||||
}
|
||||
return target; // unknown terminal — let herdr report it against the original target
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn an agent. {@code env} is applied to the process environment verbatim — this
|
||||
* is where a worker's {@code ANTHROPIC_BASE_URL} lives, and the ONLY place it should.
|
||||
* Start an agent into {@code paneId}, which must be sitting at its interactive shell prompt —
|
||||
* the seed pane of a freshly-created worker tab, or a fresh split. The pane's shell already
|
||||
* carries the worker's env ({@code ANTHROPIC_BASE_URL}, token, …) and cwd from pane creation;
|
||||
* herdr resolves the executable from {@code kind} and waits (its default timeout) until the
|
||||
* agent is detected and ready for input.
|
||||
*
|
||||
* @param name label/kind for herdr status detection (e.g. {@code "claude"})
|
||||
* @param argv launch command, e.g. {@code ["claude"]}
|
||||
* @param env process environment additions ({@code ANTHROPIC_BASE_URL}, token, …)
|
||||
* @param name unique label for this agent ({@code <kind>-<profile>-<nonce>-<seq>})
|
||||
* @param kind supported agent kind and canonical executable, e.g. {@code "claude"},
|
||||
* {@code "opencode"}
|
||||
* @param args extra arguments after the executable, e.g. {@code --mcp-config …}
|
||||
* @param paneId the pane to start the agent in
|
||||
*/
|
||||
public Agent start(String name, List<String> argv, Map<String, String> env) {
|
||||
return start(name, argv, env, null);
|
||||
}
|
||||
|
||||
/** Spawn an agent into {@code tabId} at herdr's default cwd. */
|
||||
public Agent start(String name, List<String> argv, Map<String, String> env, String tabId) {
|
||||
return start(name, argv, env, tabId, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn an agent. With a non-null {@code tabId} the worker lands in that tab (the placement
|
||||
* policy's dedicated worker tab); with {@code null} herdr splits the currently-focused tab
|
||||
* (legacy pane placement). A non-blank {@code cwd} sets the worker process's working directory —
|
||||
* {@code agent.start} honours {@code cwd} directly (an agent pane does <em>not</em> inherit the
|
||||
* tab's or workspace's cwd, so this is the only way to root a worker in the primary's directory;
|
||||
* CB-112).
|
||||
*/
|
||||
public Agent start(String name, List<String> argv, Map<String, String> env, String tabId, String cwd) {
|
||||
public Agent start(String name, String kind, List<String> args, String paneId) {
|
||||
Map<String, Object> params = new LinkedHashMap<>();
|
||||
params.put("name", name);
|
||||
params.put("argv", argv);
|
||||
params.put("env", env);
|
||||
if (tabId != null) {
|
||||
params.put("tab_id", tabId);
|
||||
}
|
||||
if (cwd != null && !cwd.isBlank()) {
|
||||
params.put("cwd", cwd);
|
||||
}
|
||||
params.put("kind", kind);
|
||||
params.put("pane_id", paneId);
|
||||
params.put("args", args);
|
||||
JsonNode result = herdr.call("agent.start", params);
|
||||
return Agent.from(result.get("agent"));
|
||||
}
|
||||
|
||||
/**
|
||||
* Deliver {@code text} to an agent as its next prompt <em>and submit it</em> — two keystroke
|
||||
* events: the message (a bracketed paste, so any embedded newlines are preserved verbatim),
|
||||
* then a standalone {@link #SUBMIT_KEY} (Enter) that actually submits it. Without the second
|
||||
* event the text just sits in the worker's input box, never processed (see {@link #SUBMIT_KEY}).
|
||||
* Deliver {@code text} to an agent as its next prompt <em>and submit it</em> — herdr's
|
||||
* {@code agent.prompt} pastes the text (embedded newlines preserved verbatim) and submits it
|
||||
* in the same call, replacing the pre-protocol-19 two-event {@code agent.send} dance.
|
||||
*/
|
||||
public void send(String target, String text) {
|
||||
herdr.call("agent.send", Map.of("target", target, "text", text));
|
||||
herdr.call("agent.send", Map.of("target", target, "text", SUBMIT_KEY));
|
||||
agentCall("agent.prompt", target, Map.of("text", text));
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-send the submit keystroke (Enter) to {@code target}. The Enter that accompanies a delivery
|
||||
* can race the paste — especially right as the worker's TUI becomes interactive — leaving the
|
||||
* text unsubmitted; the injector nudges it with this until the worker actually picks up (CB-113).
|
||||
* Re-send the submit keystroke (Enter) to {@code target}. The submit that accompanies a
|
||||
* delivery can race the paste — especially right as the worker's TUI becomes interactive —
|
||||
* leaving the text unsubmitted; the injector nudges it with this until the worker actually
|
||||
* picks up (CB-113).
|
||||
*/
|
||||
public void submit(String target) {
|
||||
herdr.call("agent.send", Map.of("target", target, "text", SUBMIT_KEY));
|
||||
agentCall("agent.send_keys", target, Map.of("keys", List.of("enter")));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -101,13 +128,13 @@ public final class AgentControl {
|
||||
* @param source one of {@code visible|recent|recent_unwrapped|detection}
|
||||
*/
|
||||
public String read(String target, String source) {
|
||||
JsonNode result = herdr.call("agent.read", Map.of("target", target, "source", source));
|
||||
JsonNode result = agentCall("agent.read", target, Map.of("source", source));
|
||||
return result.path("read").path("text").asText("");
|
||||
}
|
||||
|
||||
/** Current agent record (status, session UUID, pane). */
|
||||
public Agent get(String target) {
|
||||
return Agent.from(herdr.call("agent.get", Map.of("target", target)).get("agent"));
|
||||
return Agent.from(agentCall("agent.get", target, Map.of()).get("agent"));
|
||||
}
|
||||
|
||||
/** Just the lifecycle status — what the status-gated injector checks before send. */
|
||||
|
||||
@@ -0,0 +1,188 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* Discovers which panes host a lead by scanning herdr for tabs the operator labelled by convention
|
||||
* (CB-531), and hands {@link dev.ltms.bridged.auth.CallerResolver} the resulting
|
||||
* {@code terminal_id → lead name} map.
|
||||
*
|
||||
* <p><strong>Why scan at all.</strong> A lead is never spawned — a human opens a tab and starts an
|
||||
* agent in it — so the daemon cannot learn a lead's {@code terminal_id} at creation time the way it
|
||||
* does a worker's. CB-530 solved that by having the operator paste each id into {@code leaders:},
|
||||
* which works but costs a config edit and a daemon restart per lead, and the id is only obtainable
|
||||
* by first starting the session and asking it. Scanning closes that loop: label the tab, and the
|
||||
* pane is recognised on the next resolve.
|
||||
*
|
||||
* <p><strong>CB-579 — matched by name, not prefix.</strong> This used to strip one shared
|
||||
* {@code tabPrefix} off a label to derive the lead's name, and merged a config-supplied
|
||||
* {@code terminal_id} pin over every scan result so the pin could never expire. Both are gone: each
|
||||
* lead now configures its own exact {@code tab} label ({@code fleet.leaders.<name>.tab}), so this
|
||||
* class is handed a {@code tab → name} map up front and matches labels against it exactly
|
||||
* (case-insensitively). There is no merge step — a scan result is the whole answer. That is the
|
||||
* fix for the bug this replaces: a {@code terminal_id} pin surviving in config after the pane it
|
||||
* named was gone, so the daemon kept treating a dead session as a live lead forever.
|
||||
*
|
||||
* <p><strong>Direction of trust.</strong> The label names the lead; it never <em>grants</em>
|
||||
* anything a pane could take for itself. Three properties keep that honest:
|
||||
* <ol>
|
||||
* <li>Worker spaces are excluded wholesale ({@code excludedWorkspaceLabels}), so a worker cannot
|
||||
* become a lead by being placed — as a split, say — inside a matching tab.</li>
|
||||
* <li>A worker cannot rename a tab: {@code tab.rename} is reachable only through
|
||||
* {@link WorkspaceControl}, which no {@code bridge_*} tool exposes. The label is writable by
|
||||
* the human at the terminal and by nobody the bridge is defending against.</li>
|
||||
* <li>The label is a <em>name</em>, not a capability. What a pane may do is decided by
|
||||
* {@code Authz} against the role {@code CallerResolver} returns; a tab that calls itself a
|
||||
* lead still cannot act as one unless the daemon's own registry agrees.</li>
|
||||
* </ol>
|
||||
*
|
||||
* <p><strong>CB-558 — bridged now writes lead labels too.</strong> This class used to be able to say
|
||||
* that bridged never renames a lead tab, so the label was always the human's own writing and there
|
||||
* was no round-trip from the daemon's rename back into its next decision.
|
||||
* {@code dev.ltms.bridged.lead.LeadLauncher} ends that: an auto-launched lead is labelled by the
|
||||
* daemon and found again by this scan. The trust direction above is unaffected — bridged writing a
|
||||
* name for a lead it just started is not a pane promoting itself — but <em>staleness</em> becomes
|
||||
* real: a label left behind by a session that has since died would read as a live lead forever.
|
||||
* This scanner does not solve that (its job is naming, and a stale name costs nothing here); the
|
||||
* launcher does, by requiring a running agent in the tab before it counts the lead as live. If you
|
||||
* ever make a decision that <em>removes</em> something based on this map, add the same check.
|
||||
* The remaining hazard is an <em>operator</em> one — a worker {@code tabLabel} template that
|
||||
* happens to start with the same prefix would promote the whole fleet — and that is refused at
|
||||
* startup by {@code BridgedConfig.validateLeadTabPrefixes} rather than documented here.
|
||||
*
|
||||
* <p><strong>Caching.</strong> {@link #get()} is on the request path (every resolve), so the scan
|
||||
* is TTL-cached and a stale-but-valid map is preferred to a herdr round-trip. A failed scan keeps
|
||||
* the previous answer instead of emptying it — a herdr hiccup must not silently demote a live lead
|
||||
* mid-session.
|
||||
*/
|
||||
public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadTabScanner.class);
|
||||
|
||||
private final HerdrClient herdr;
|
||||
private final Map<String, String> tabToName;
|
||||
private final Set<String> excludedWorkspaceLabels;
|
||||
private final long ttlNanos;
|
||||
private final LongSupplier clock;
|
||||
|
||||
private Map<String, String> cached = Map.of();
|
||||
private long scannedAtNanos;
|
||||
private boolean everScanned;
|
||||
|
||||
/**
|
||||
* @param herdr the herdr client to query ({@code workspace.list},
|
||||
* {@code tab.list}, {@code pane.list} — all read-only)
|
||||
* @param tabToName every configured lead's exact tab label → its name
|
||||
* ({@code fleet.leaders.<name>.tab}), matched case-insensitively
|
||||
* @param excludedWorkspaceLabels workspaces never scanned — the configured worker spaces
|
||||
* @param ttlNanos how long a scan result is reused before the next one
|
||||
* @param clock nanosecond time source ({@code System::nanoTime} in production)
|
||||
*/
|
||||
public LeadTabScanner(HerdrClient herdr, Map<String, String> tabToName,
|
||||
Set<String> excludedWorkspaceLabels, long ttlNanos, LongSupplier clock) {
|
||||
this.herdr = herdr;
|
||||
this.tabToName = normalize(tabToName);
|
||||
this.excludedWorkspaceLabels = excludedWorkspaceLabels == null
|
||||
? Set.of() : Set.copyOf(excludedWorkspaceLabels);
|
||||
this.ttlNanos = ttlNanos;
|
||||
this.clock = clock;
|
||||
}
|
||||
|
||||
/** Keys stripped and lower-cased once, so every lookup is a plain map hit. */
|
||||
private static Map<String, String> normalize(Map<String, String> tabToName) {
|
||||
if (tabToName == null || tabToName.isEmpty()) {
|
||||
return Map.of();
|
||||
}
|
||||
Map<String, String> out = new LinkedHashMap<>();
|
||||
tabToName.forEach((tab, name) -> {
|
||||
if (tab != null && !tab.isBlank() && name != null && !name.isBlank()) {
|
||||
out.put(tab.strip().toLowerCase(Locale.ROOT), name);
|
||||
}
|
||||
});
|
||||
return Collections.unmodifiableMap(out);
|
||||
}
|
||||
|
||||
/**
|
||||
* The current {@code terminal_id → lead name} map, rescanning when the cache has expired.
|
||||
*
|
||||
* <p>Synchronized so a burst of concurrent calls produces one scan rather than one each; a scan
|
||||
* is a handful of RPCs over a Unix socket and is rate-limited to one per TTL.
|
||||
*/
|
||||
@Override
|
||||
public synchronized Map<String, String> get() {
|
||||
long now = clock.getAsLong();
|
||||
if (everScanned && now - scannedAtNanos < ttlNanos) {
|
||||
return cached;
|
||||
}
|
||||
// Stamp before scanning, not after: a herdr that is down must cost one attempt per TTL, not
|
||||
// one per request.
|
||||
scannedAtNanos = now;
|
||||
everScanned = true;
|
||||
try {
|
||||
Map<String, String> fresh = scan();
|
||||
if (!fresh.equals(cached)) {
|
||||
log.info("lead panes: {}", fresh);
|
||||
}
|
||||
cached = fresh;
|
||||
} catch (HerdrException e) {
|
||||
log.warn("lead-tab scan failed, keeping the {} lead(s) already known: {}",
|
||||
cached.size(), e.getMessage());
|
||||
}
|
||||
return cached;
|
||||
}
|
||||
|
||||
/** One full pass: labelled tabs → their panes → those panes' terminals. */
|
||||
private Map<String, String> scan() {
|
||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
||||
for (JsonNode w : herdr.call("workspace.list").path("workspaces")) {
|
||||
Workspace ws = Workspace.from(w);
|
||||
if (ws.workspaceId() == null || excludedWorkspaceLabels.contains(ws.label())) {
|
||||
continue;
|
||||
}
|
||||
for (JsonNode t : herdr.call("tab.list", Map.of("workspace_id", ws.workspaceId())).path("tabs")) {
|
||||
Tab tab = Tab.from(t);
|
||||
String name = leadNameOf(tab.label());
|
||||
if (name != null && tab.tabId() != null) {
|
||||
nameByTab.put(tab.tabId(), name);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Map<String, String> byTerminal = new LinkedHashMap<>();
|
||||
if (!nameByTab.isEmpty()) {
|
||||
// One pane.list for every tab: panes carry tab_id, so the join is local.
|
||||
for (JsonNode p : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String name = nameByTab.get(p.path("tab_id").asText(null));
|
||||
String terminal = p.path("terminal_id").asText(null);
|
||||
if (name != null && terminal != null && !terminal.isBlank()) {
|
||||
byTerminal.put(terminal, name);
|
||||
}
|
||||
}
|
||||
}
|
||||
return Collections.unmodifiableMap(byTerminal);
|
||||
}
|
||||
|
||||
/**
|
||||
* The lead name a tab label declares, or {@code null} if it names none of the configured leads.
|
||||
*
|
||||
* <p>Exact match (case-insensitive, ends stripped) against {@link #tabToName} — no prefix
|
||||
* stripping, so an operator's {@code "lead: something-else"} tab is never mistaken for a
|
||||
* configured lead just because it shares a prefix.
|
||||
*/
|
||||
private String leadNameOf(String label) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
return tabToName.get(label.strip().toLowerCase(Locale.ROOT));
|
||||
}
|
||||
}
|
||||
@@ -23,9 +23,9 @@ public record Tab(String tabId, String workspaceId, String label, int paneCount)
|
||||
}
|
||||
|
||||
/**
|
||||
* A freshly-created tab together with the placeholder shell pane herdr seeds it with.
|
||||
* The caller starts the worker into {@link #tab()} then closes {@link #rootPaneId()} so
|
||||
* only the worker pane remains.
|
||||
* A freshly-created tab together with the shell pane herdr seeds it with. Under protocol 19
|
||||
* the caller starts the worker <em>into</em> {@link #rootPaneId()} — the seed pane's shell
|
||||
* carries the worker's cwd and env from {@code tab.create}, and becomes the worker pane.
|
||||
*/
|
||||
public record Created(Tab tab, String rootPaneId) {
|
||||
/** Project a {@code tab_created} result ({@code {tab, root_pane}}). */
|
||||
|
||||
@@ -5,6 +5,7 @@ import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
@@ -40,6 +41,16 @@ public final class WorkspaceControl {
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Every tab in {@code workspaceId}, in herdr's order. */
|
||||
public List<Tab> listTabs(String workspaceId) {
|
||||
JsonNode result = herdr.call("tab.list", Map.of("workspace_id", workspaceId));
|
||||
List<Tab> out = new ArrayList<>();
|
||||
for (JsonNode t : result.path("tabs")) {
|
||||
out.add(Tab.from(t));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/** The first workspace with this exact label, if any. */
|
||||
public Optional<Workspace> findByLabel(String label) {
|
||||
return listWorkspaces().stream()
|
||||
@@ -65,13 +76,37 @@ public final class WorkspaceControl {
|
||||
}
|
||||
|
||||
/**
|
||||
* A brand-new tab in {@code workspaceId} plus the placeholder shell pane herdr seeds it with.
|
||||
* Start the worker into the tab, then {@code pane.close} the root pane so the tab holds only the
|
||||
* worker. (The worker's own cwd is set on {@code agent.start}, not here — an {@code agent.start}
|
||||
* pane does not inherit the tab's cwd; see {@code AgentControl.start}.)
|
||||
* A brand-new tab in {@code workspaceId} plus the shell pane herdr seeds it with. Under
|
||||
* protocol 19 that seed pane is where the worker <em>starts</em>: its shell carries
|
||||
* {@code cwd} and {@code env} (the worker's {@code ANTHROPIC_BASE_URL} — this is the
|
||||
* subscription-injection seam now), and {@code agent.start} launches the agent into it.
|
||||
*/
|
||||
public Tab.Created createTab(String workspaceId) {
|
||||
return Tab.Created.from(herdr.call("tab.create", Map.of("workspace_id", workspaceId)));
|
||||
public Tab.Created createTab(String workspaceId, String cwd, Map<String, String> env) {
|
||||
Map<String, Object> params = new LinkedHashMap<>();
|
||||
params.put("workspace_id", workspaceId);
|
||||
if (cwd != null && !cwd.isBlank()) {
|
||||
params.put("cwd", cwd);
|
||||
}
|
||||
if (env != null && !env.isEmpty()) {
|
||||
params.put("env", env);
|
||||
}
|
||||
return Tab.Created.from(herdr.call("tab.create", params));
|
||||
}
|
||||
|
||||
/**
|
||||
* Split the currently-focused tab and return the new pane's id — the legacy pane placement's
|
||||
* seed pane, carrying {@code cwd} and {@code env} exactly as {@link #createTab}'s does.
|
||||
*/
|
||||
public String splitPane(String cwd, Map<String, String> env) {
|
||||
Map<String, Object> params = new LinkedHashMap<>();
|
||||
params.put("direction", "right");
|
||||
if (cwd != null && !cwd.isBlank()) {
|
||||
params.put("cwd", cwd);
|
||||
}
|
||||
if (env != null && !env.isEmpty()) {
|
||||
params.put("env", env);
|
||||
}
|
||||
return herdr.call("pane.split", params).path("pane").path("pane_id").asText(null);
|
||||
}
|
||||
|
||||
/** Give a worker's tab a human label in the tab bar. */
|
||||
|
||||
@@ -2,11 +2,17 @@ package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* The CB-106 completion fallback: bridges the {@link Injector}'s turn-completion signal to the
|
||||
@@ -55,8 +61,13 @@ public final class CompletionResolver implements TurnListener {
|
||||
/** Cap the scraped tail so a long transcript can't return an unbounded blob. */
|
||||
static final int MAX_SCRAPE_CHARS = 4000;
|
||||
|
||||
private static final String CLIPPED_PANE_TAIL_MARKER =
|
||||
"[Pane tail clipped: member did not call bridge_reply.]";
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ExhaustedPatternLookup exhaustedPatterns;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
|
||||
/**
|
||||
* Per-target record of the turn currently in flight: the exact {@link Rendezvous} waiter its
|
||||
@@ -74,23 +85,36 @@ public final class CompletionResolver implements TurnListener {
|
||||
|
||||
private final ConcurrentHashMap<String, InFlight> inFlight = new ConcurrentHashMap<>();
|
||||
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous) {
|
||||
/**
|
||||
* @param exhaustedPatterns CB-578 stage A: per-target lookup for a profile's configured
|
||||
* usage-limit refusal pattern. Required — there is deliberately no
|
||||
* defaulting overload; a caller that does not want the classification
|
||||
* must pass an explicit inert value ({@link ExhaustedPatternLookup#none()}).
|
||||
* @param exhaustionSink CB-578 stage B: notified when a {@code BACKEND_EXHAUSTED}
|
||||
* classification actually resolves a waiter. Required for the same
|
||||
* reason as {@code exhaustedPatterns} — pass {@link ExhaustionSink#none()}
|
||||
* to opt out.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this.agents = agents;
|
||||
this.rendezvous = rendezvous;
|
||||
this.exhaustedPatterns = Objects.requireNonNull(exhaustedPatterns, "exhaustedPatterns");
|
||||
this.exhaustionSink = Objects.requireNonNull(exhaustionSink, "exhaustionSink");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target) {
|
||||
public void onDelivered(String target, TurnToken token) {
|
||||
// Capture the exact waiter this turn belongs to (CB-116) and snapshot the pane's pre-turn
|
||||
// content — what it shows *before* the just-delivered turn produces output — as the staleness
|
||||
// reference (CB-115). Done synchronously (like the delivering send itself) so both are in
|
||||
// place before this turn's completion can fire.
|
||||
captureBaseline(target);
|
||||
captureBaseline(target, token);
|
||||
}
|
||||
|
||||
/** Capture the in-flight turn: its waiter and pre-turn baseline (the testable core of {@link #onDelivered}). */
|
||||
void captureBaseline(String target) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(target);
|
||||
void captureBaseline(String target, TurnToken token) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = token.waiter();
|
||||
if (waiter == null) {
|
||||
inFlight.remove(target); // no send is waiting on this delivery — nothing to resolve later
|
||||
return;
|
||||
@@ -122,10 +146,25 @@ public final class CompletionResolver implements TurnListener {
|
||||
Thread.ofVirtual().name("completion-" + target).start(() -> resolve(target, turn));
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the completed turn before adapter housekeeping can erase its rendered output. This is
|
||||
* intentionally synchronous and used only when a post-turn context reset is enabled; the normal
|
||||
* path remains off-loaded so polling is not blocked by a scrape.
|
||||
*/
|
||||
public void resolveBeforePostAction(String target) {
|
||||
resolve(target, inFlight.get(target));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn));
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn, null));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn, reason));
|
||||
}
|
||||
|
||||
/** Synchronous resolve (the unit-testable core of {@link #onTurnComplete}). */
|
||||
@@ -138,9 +177,15 @@ public final class CompletionResolver implements TurnListener {
|
||||
return;
|
||||
}
|
||||
String tail;
|
||||
String assistantBlock = null;
|
||||
int originalLength = 0;
|
||||
boolean clipped = false;
|
||||
boolean scrapeFailed = false;
|
||||
try {
|
||||
tail = clip(lastAssistantBlock(agents.read(target, SCRAPE_SOURCE)));
|
||||
assistantBlock = lastAssistantBlock(agents.read(target, SCRAPE_SOURCE));
|
||||
originalLength = assistantBlock.strip().length();
|
||||
clipped = originalLength > MAX_SCRAPE_CHARS;
|
||||
tail = clip(assistantBlock);
|
||||
} catch (RuntimeException e) {
|
||||
// The worker finished but we couldn't read its screen — still resolve the send so the
|
||||
// caller unblocks; an empty tail beats hanging until the caller's timeout.
|
||||
@@ -160,8 +205,33 @@ public final class CompletionResolver implements TurnListener {
|
||||
target);
|
||||
return; // keep the in-flight record: a later genuine completion still needs it
|
||||
}
|
||||
if (rendezvous.resolveCompletion(waiter, tail)) {
|
||||
// CB-578 stage A: a turn that ended with no bridge_reply AND whose scrape matches the
|
||||
// backend's configured usage-limit pattern is a refusal, not an answer. Classify it as
|
||||
// BACKEND_EXHAUSTED rather than handing the caller a scrape that reads like a real reply.
|
||||
if (!scrapeFailed) {
|
||||
Pattern exhausted = exhaustedPatterns.patternFor(target);
|
||||
String matchedLine = exhausted == null ? null : firstMatchingLine(assistantBlock, exhausted);
|
||||
if (matchedLine != null) {
|
||||
String reason = "backend exhausted (usage limit): " + matchedLine;
|
||||
if (rendezvous.resolveExhausted(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("completion for {} classified BACKEND_EXHAUSTED (no bridge_reply; scrape "
|
||||
+ "matched the profile's exhausted pattern): {}", target, reason);
|
||||
// CB-578 stage B: only on the resolution that actually won the race — a late
|
||||
// duplicate must never quarantine a credential twice for one refusal.
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
String completion = clipped ? tail + "\n" + CLIPPED_PANE_TAIL_MARKER : tail;
|
||||
if (rendezvous.resolveCompletion(waiter, completion)) {
|
||||
inFlight.remove(target, turn);
|
||||
if (clipped) {
|
||||
log.warn("completion scrape for {} clipped from {} chars to the {} char cap; "
|
||||
+ "member did not call bridge_reply, so the pane tail is partial",
|
||||
target, originalLength, MAX_SCRAPE_CHARS);
|
||||
}
|
||||
log.debug("resolved send to {} via turn-completion fallback ({} chars scraped)",
|
||||
target, tail.length());
|
||||
}
|
||||
@@ -169,6 +239,11 @@ public final class CompletionResolver implements TurnListener {
|
||||
|
||||
/** Synchronous fail (the unit-testable core of {@link #onTurnFailed}). */
|
||||
void fail(String target, InFlight turn) {
|
||||
fail(target, turn, null);
|
||||
}
|
||||
|
||||
/** Synchronous fail with an optional reason supplied by a dropped worker queue. */
|
||||
void fail(String target, InFlight turn, String explicitReason) {
|
||||
// A never-delivered readiness failure has no in-flight record but still has a blocked send;
|
||||
// fall back to the currently-registered waiter (unambiguous — that send never completed, so
|
||||
// no next turn exists to confuse it with).
|
||||
@@ -178,23 +253,64 @@ public final class CompletionResolver implements TurnListener {
|
||||
inFlight.remove(target, turn); // nobody blocked on this worker — nothing to fail
|
||||
return;
|
||||
}
|
||||
String reason;
|
||||
try {
|
||||
reason = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
} catch (RuntimeException e) {
|
||||
reason = "";
|
||||
}
|
||||
if (reason.isBlank()) {
|
||||
// No screen to scrape — either the worker is stuck (CB-109) or gone (CB-110).
|
||||
reason = "worker did not reply; its turn ended in an unrecoverable state "
|
||||
+ "(worker unreachable or stuck)";
|
||||
String reason = explicitReason;
|
||||
if (reason == null || reason.isBlank()) {
|
||||
try {
|
||||
reason = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
} catch (RuntimeException e) {
|
||||
reason = "";
|
||||
}
|
||||
if (reason.isBlank()) {
|
||||
// No screen to scrape — either the worker is stuck (CB-109) or gone (CB-110).
|
||||
reason = "worker did not reply; its turn ended in an unrecoverable state "
|
||||
+ "(worker unreachable or stuck)";
|
||||
}
|
||||
}
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.debug("failed send to {} via turn-stall fallback", target);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The first line of {@code text} matching {@code pattern}, stripped — the CB-578 stage A
|
||||
* evidence carried in a {@code BACKEND_EXHAUSTED} reason so the operator sees the real refusal
|
||||
* text, never a generic label. {@code null} if no line matches.
|
||||
*/
|
||||
static String firstMatchingLine(String text, Pattern pattern) {
|
||||
if (text == null || text.isEmpty()) return null;
|
||||
for (String line : text.split("\n", -1)) {
|
||||
if (pattern.matcher(line).find()) {
|
||||
return line.strip();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for the CB-578 stage A exhausted-pattern classification, logged at startup
|
||||
* the way {@link dev.ltms.bridged.health.FleetHealthMonitor#coverage} is — so an operator can
|
||||
* see whether the classification is on, and for which profiles, without reading every
|
||||
* profile's config by hand.
|
||||
*
|
||||
* @param allProfiles every configured profile name
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry an exhausted pattern
|
||||
*/
|
||||
public static String coverage(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
if (configuredProfiles.isEmpty()) {
|
||||
return "off (no profile has an exhaustedPattern configured; profiles: " + sorted(allProfiles) + ")";
|
||||
}
|
||||
Set<String> unconfigured = new TreeSet<>(allProfiles);
|
||||
unconfigured.removeAll(configuredProfiles);
|
||||
return unconfigured.isEmpty()
|
||||
? "full (all profiles configured: " + sorted(allProfiles) + ")"
|
||||
: "partial (configured: " + sorted(configuredProfiles) + "; not configured: " + sorted(unconfigured) + ")";
|
||||
}
|
||||
|
||||
private static List<String> sorted(Set<String> names) {
|
||||
return names.stream().sorted().toList();
|
||||
}
|
||||
|
||||
private static String clip(String s) {
|
||||
if (s == null) return "";
|
||||
String trimmed = s.strip();
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Per-target lookup for a profile's configured usage-limit refusal pattern (CB-578 stage A): how
|
||||
* {@link CompletionResolver} tells a backend that refused on a subscription usage limit — the
|
||||
* worker's pane stays healthy, but the account is exhausted — apart from a genuine completion.
|
||||
*
|
||||
* <p>The pattern is always profile config, never a vendor string in Java source: every backend
|
||||
* words its refusal differently, so a hardcoded sentence would only ever match one of them.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface ExhaustedPatternLookup {
|
||||
|
||||
/** The compiled pattern configured for {@code target}'s profile, or {@code null} if none. */
|
||||
Pattern patternFor(String target);
|
||||
|
||||
/**
|
||||
* Inert lookup — no profile has a pattern configured, so the classification never fires and
|
||||
* the completion fallback behaves exactly as before CB-578 stage A. The explicit stand-in a
|
||||
* caller (or a test not exercising this feature) passes instead of a defaulting overload.
|
||||
*/
|
||||
static ExhaustedPatternLookup none() {
|
||||
return target -> null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,31 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
/**
|
||||
* Notified when {@link CompletionResolver} actually delivers a {@code BACKEND_EXHAUSTED}
|
||||
* classification to a waiting send (CB-578 stage B) — never on a race that lost (see
|
||||
* {@link CompletionResolver#resolve}, which only calls this after
|
||||
* {@code Rendezvous.resolveExhausted} returns {@code true}).
|
||||
*
|
||||
* <p>{@link CompletionResolver} knows only {@code target} (a herdr terminal id); it has no notion of
|
||||
* profiles or credentials, so mapping {@code target} to whatever should be quarantined is entirely
|
||||
* the sink's job — see {@code Bridged.main}'s wiring, which resolves target → session → profile →
|
||||
* {@code effectiveCredentialId()} and calls {@code BackendQuarantine.quarantine} on it.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface ExhaustionSink {
|
||||
|
||||
/**
|
||||
* @param target the herdr terminal id whose turn was classified {@code BACKEND_EXHAUSTED}
|
||||
* @param reason the matched-line reason carried by the classification
|
||||
*/
|
||||
void onExhausted(String target, String reason);
|
||||
|
||||
/**
|
||||
* Inert sink — nothing happens on exhaustion. The explicit stand-in a caller (or a test not
|
||||
* exercising this feature) passes instead of a defaulting overload, exactly like
|
||||
* {@link ExhaustedPatternLookup#none()}.
|
||||
*/
|
||||
static ExhaustionSink none() {
|
||||
return (target, reason) -> { };
|
||||
}
|
||||
}
|
||||
@@ -2,6 +2,7 @@ package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -78,6 +79,15 @@ public final class Injector {
|
||||
*/
|
||||
private static final int READINESS_GRACE_POLLS = 240;
|
||||
|
||||
/**
|
||||
* The single source for the injector poll cadence — how often the {@link StatusPoller} drives
|
||||
* {@link #onStatus} at. {@code Bridged} passes this to every {@link StatusPoller} it constructs,
|
||||
* and this class reads it to state the readiness grace in seconds on the CB-562 expiry log
|
||||
* instead of hardcoding "60s". One constant, so a cadence change cannot silently desync a log
|
||||
* that claims a grace duration.
|
||||
*/
|
||||
public static final long POLL_INTERVAL_MILLIS = 250;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final TurnListener turnListener;
|
||||
private final Predicate<String> ready; // CB-113: a target is deliverable only when available
|
||||
@@ -120,7 +130,7 @@ public final class Injector {
|
||||
}
|
||||
|
||||
/** A pending message and the future that completes when it has been delivered. */
|
||||
private record Pending(String text, CompletableFuture<Void> delivered) {
|
||||
private record Pending(String text, TurnToken token, CompletableFuture<Void> delivered) {
|
||||
}
|
||||
|
||||
/** Per-worker delivery state, guarded by its own monitor (single writer per worker). */
|
||||
@@ -132,6 +142,10 @@ public final class Injector {
|
||||
boolean turnObserved; // saw a real `working` sample since that delivery (turn ran)
|
||||
int unknownSinceTurn; // consecutive `unknown` samples while a delegation is outstanding (CB-109)
|
||||
int notReadySincePoll; // consecutive injectable samples a queued message waited on the readiness gate (CB-114)
|
||||
boolean postTurnPending; // completion observed; adapter housekeeping has not started yet
|
||||
boolean awaitingPostTurnPickup;
|
||||
boolean postTurnObserved;
|
||||
int injectableSincePostTurnPickup;
|
||||
|
||||
synchronized void add(Pending p) {
|
||||
queue.add(p);
|
||||
@@ -146,9 +160,9 @@ public final class Injector {
|
||||
* <p>Uses an atomic map update so a concurrent {@link #drop} cannot slip between "find the
|
||||
* target" and "queue the message" and orphan it in a target it just removed.
|
||||
*/
|
||||
public CompletableFuture<Void> enqueue(String target, String text) {
|
||||
public CompletableFuture<Void> enqueue(String target, String text, TurnToken token) {
|
||||
CompletableFuture<Void> delivered = new CompletableFuture<>();
|
||||
Pending p = new Pending(text, delivered);
|
||||
Pending p = new Pending(text, token, delivered);
|
||||
targets.compute(target, (_, existing) -> {
|
||||
Target t = (existing != null) ? existing : new Target();
|
||||
t.add(p); // synchronized on the Target monitor — atomic with a concurrent drop
|
||||
@@ -173,9 +187,15 @@ public final class Injector {
|
||||
boolean turnCompleted = false;
|
||||
boolean turnFailed = false;
|
||||
boolean resubmit = false;
|
||||
boolean startPostTurn = false;
|
||||
List<Pending> notReady = null; // queued messages failed because the worker never became ready
|
||||
synchronized (t) {
|
||||
if (status == AgentStatus.WORKING) {
|
||||
if (t.awaitingPostTurnPickup) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
t.injectableSincePostTurnPickup = 0;
|
||||
t.postTurnObserved = true;
|
||||
}
|
||||
// Definitive pickup: the worker is busy on our last message, and (if a delivery is
|
||||
// outstanding) a real turn is now confirmed to be running.
|
||||
t.awaitingPickup = false;
|
||||
@@ -185,6 +205,16 @@ public final class Injector {
|
||||
if (t.awaitingCompletion) t.turnObserved = true;
|
||||
} else if (status.injectable()) { // IDLE or BLOCKED
|
||||
t.unknownSinceTurn = 0;
|
||||
if (t.awaitingPostTurnPickup) {
|
||||
if (++t.injectableSincePostTurnPickup >= PICKUP_GRACE_POLLS) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
t.injectableSincePostTurnPickup = 0;
|
||||
} else {
|
||||
resubmit = true;
|
||||
}
|
||||
} else if (t.postTurnObserved) {
|
||||
t.postTurnObserved = false;
|
||||
}
|
||||
if (t.awaitingPickup) {
|
||||
if (++t.injectableSincePickup >= PICKUP_GRACE_POLLS) {
|
||||
// Pickup edge was never sampled (turn faster than the poll, or status lag).
|
||||
@@ -209,11 +239,16 @@ public final class Injector {
|
||||
t.awaitingCompletion = false;
|
||||
t.turnObserved = false;
|
||||
turnCompleted = true;
|
||||
if (turnListener.hasPostTurnAction(target)) {
|
||||
t.postTurnPending = true;
|
||||
startPostTurn = true;
|
||||
}
|
||||
}
|
||||
// Deliver the next queued message only once the prior turn is fully settled, so a
|
||||
// completion is never confused with the pickup of the following message — and only
|
||||
// once the worker is available (CB-113), so we never paste into its boot window.
|
||||
if (!t.awaitingCompletion) {
|
||||
if (!t.awaitingCompletion && !t.postTurnPending
|
||||
&& !t.awaitingPostTurnPickup && !t.postTurnObserved) {
|
||||
Pending p = t.queue.peek();
|
||||
if (p != null && ready.test(target)) {
|
||||
t.notReadySincePoll = 0;
|
||||
@@ -239,6 +274,11 @@ public final class Injector {
|
||||
// fail every queued message and release the target (CB-114) instead of
|
||||
// polling it indefinitely with the caller's future never completing.
|
||||
notReady = new ArrayList<>(t.queue);
|
||||
log.warn("readiness grace for {} expired after {} polls ({}s): target never "
|
||||
+ "became deliverable, so failing {} queued message(s) that never "
|
||||
+ "reached its pane",
|
||||
target, READINESS_GRACE_POLLS,
|
||||
READINESS_GRACE_POLLS * POLL_INTERVAL_MILLIS / 1000, notReady.size());
|
||||
t.queue.clear();
|
||||
t.notReadySincePoll = 0;
|
||||
}
|
||||
@@ -263,7 +303,8 @@ public final class Injector {
|
||||
|
||||
// Reclaim the entry once the worker is fully quiescent (nothing queued, no pickup or
|
||||
// completion awaited), so the map cannot grow without bound across short-lived workers.
|
||||
if (t.queue.isEmpty() && !t.awaitingPickup && !t.awaitingCompletion) {
|
||||
if (t.queue.isEmpty() && !t.awaitingPickup && !t.awaitingCompletion
|
||||
&& !t.postTurnPending && !t.awaitingPostTurnPickup && !t.postTurnObserved) {
|
||||
targets.remove(target, t);
|
||||
}
|
||||
}
|
||||
@@ -290,7 +331,21 @@ public final class Injector {
|
||||
turnListener.onTurnFailed(target);
|
||||
}
|
||||
if (turnCompleted) {
|
||||
turnListener.onTurnComplete(target);
|
||||
if (startPostTurn) {
|
||||
boolean started = turnListener.onTurnCompleteWithPostAction(target);
|
||||
synchronized (t) {
|
||||
t.postTurnPending = false;
|
||||
if (started) {
|
||||
t.awaitingPostTurnPickup = true;
|
||||
t.injectableSincePostTurnPickup = 0;
|
||||
}
|
||||
if (t.queue.isEmpty() && !t.awaitingPostTurnPickup) {
|
||||
targets.remove(target, t);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
turnListener.onTurnComplete(target);
|
||||
}
|
||||
}
|
||||
if (turnFailed) {
|
||||
turnListener.onTurnFailed(target);
|
||||
@@ -302,7 +357,7 @@ public final class Injector {
|
||||
} else {
|
||||
// Baseline the pane's pre-turn content so a misattributed completion (no new output)
|
||||
// can't resolve this send with the previous turn's stale answer (CB-115).
|
||||
turnListener.onDelivered(target);
|
||||
turnListener.onDelivered(target, sent.token());
|
||||
sent.delivered().complete(null);
|
||||
}
|
||||
}
|
||||
@@ -317,7 +372,8 @@ public final class Injector {
|
||||
.filter(e -> {
|
||||
synchronized (e.getValue()) {
|
||||
Target t = e.getValue();
|
||||
return !t.queue.isEmpty() || t.awaitingPickup || t.awaitingCompletion;
|
||||
return !t.queue.isEmpty() || t.awaitingPickup || t.awaitingCompletion
|
||||
|| t.postTurnPending || t.awaitingPostTurnPickup || t.postTurnObserved;
|
||||
}
|
||||
})
|
||||
.map(java.util.Map.Entry::getKey)
|
||||
@@ -343,13 +399,20 @@ public final class Injector {
|
||||
hadDeliveredTurn = t.awaitingCompletion;
|
||||
t.awaitingCompletion = false;
|
||||
t.awaitingPickup = false;
|
||||
t.postTurnPending = false;
|
||||
t.awaitingPostTurnPickup = false;
|
||||
t.postTurnObserved = false;
|
||||
}
|
||||
log.warn("{} is gone, dropping its queue: {} message(s) failed{}; cause: {}", target,
|
||||
pending.size(),
|
||||
hadDeliveredTurn ? " (including one turn already in flight whose completion was never confirmed)" : "",
|
||||
cause.getMessage());
|
||||
forget.accept(target); // the worker is gone — clear its readiness/presence too (CB-114)
|
||||
for (Pending p : pending) {
|
||||
p.delivered().completeExceptionally(cause);
|
||||
}
|
||||
if (hadDeliveredTurn) {
|
||||
turnListener.onTurnFailed(target);
|
||||
}
|
||||
// A queued send has no in-flight record, while a delivered turn does. CompletionResolver
|
||||
// handles both forms and resolves its waiter at most once.
|
||||
turnListener.onTurnFailed(target, cause.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
+1
-1
@@ -15,7 +15,7 @@ import java.util.Set;
|
||||
* mounts the bridge MCP is never marked present — its sends stay queued until they time out, which is
|
||||
* correct (it could not have replied anyway).
|
||||
*/
|
||||
public class WorkerPresence {
|
||||
public class MemberPresence {
|
||||
|
||||
private final Set<String> present = ConcurrentHashMap.newKeySet();
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
|
||||
/**
|
||||
* Notified when a worker's delegated turn is observed to complete — a confirmed
|
||||
* {@code WORKING → IDLE} transition after a delivery. This is the CB-106 completion signal the
|
||||
@@ -13,6 +15,25 @@ public interface TurnListener {
|
||||
/** A worker's delegated turn finished (worker returned to idle after visibly working). */
|
||||
void onTurnComplete(String target);
|
||||
|
||||
/**
|
||||
* Whether completion must pause ordinary delivery while adapter-specific housekeeping starts.
|
||||
* This is queried before the injector considers the next queued message, closing the same-tick
|
||||
* ordering gap. Implementations must not mutate state here.
|
||||
*/
|
||||
default boolean hasPostTurnAction(String target) {
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Complete the delegated turn and start its non-turn housekeeping operation.
|
||||
*
|
||||
* @return true when the injector must observe that operation settle before the next delivery
|
||||
*/
|
||||
default boolean onTurnCompleteWithPostAction(String target) {
|
||||
onTurnComplete(target);
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* A worker that visibly ran a delegated turn then wedged in a non-idle, non-working state
|
||||
* (CB-109) — e.g. an error screen herdr classifies as {@code unknown} — so no
|
||||
@@ -22,6 +43,14 @@ public interface TurnListener {
|
||||
default void onTurnFailed(String target) {
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #onTurnFailed(String)}, carrying the reason a worker became unreachable. The default
|
||||
* keeps existing listeners working while allowing the completion resolver to report a useful cause.
|
||||
*/
|
||||
default void onTurnFailed(String target, String reason) {
|
||||
onTurnFailed(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* A message was just delivered into {@code target}'s pane (CB-115). Fired so the completion
|
||||
* resolver can snapshot the pane's pre-turn content: a later {@link #onTurnComplete} whose
|
||||
@@ -30,7 +59,7 @@ public interface TurnListener {
|
||||
* resolve the send with the previous turn's stale answer. A default no-op keeps the interface
|
||||
* functional for callers that don't scrape.
|
||||
*/
|
||||
default void onDelivered(String target) {
|
||||
default void onDelivered(String target, TurnToken token) {
|
||||
}
|
||||
|
||||
/** No-op default for callers that only need delivery, not completion signalling. */
|
||||
|
||||
@@ -0,0 +1,292 @@
|
||||
package dev.ltms.bridged.lead;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.Tab;
|
||||
import dev.ltms.bridged.herdr.Workspace;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* Starts the leads {@code fleet.leaders:} declares, when none is already running (CB-558).
|
||||
*
|
||||
* <p><strong>A lead is not a member, and this class exists to keep it that way.</strong> Every other
|
||||
* spawn path in the daemon goes through {@code HerdrPeerLauncher}, which does three things a lead
|
||||
* must never receive:
|
||||
* <ol>
|
||||
* <li>it appends the <em>reply charter</em> — "you are an off-subscription worker … end every turn
|
||||
* with {@code bridge_reply}". A lead is the orchestrator; telling it that it is a worker is
|
||||
* exactly backwards.</li>
|
||||
* <li>it registers the session with {@code SessionManager}, which subjects it to the idle reaper,
|
||||
* the context cap and the shutdown drain. An idle lead is the normal state of a lead, so the
|
||||
* reaper would kill the orchestrator for doing its job.</li>
|
||||
* <li>it can move a peer off the subscription via {@code ANTHROPIC_BASE_URL}. A lead stays on the
|
||||
* operator's subscription, always.</li>
|
||||
* </ol>
|
||||
* So this launcher talks to {@link AgentControl}/{@link WorkspaceControl} directly. The duplication
|
||||
* with the member launchers (argv and env assembly) is deliberate and is the cheaper half of the
|
||||
* trade: entangling the worker path with a not-a-worker case is how the three rules above get
|
||||
* broken later, quietly.
|
||||
*
|
||||
* <p><strong>Liveness, and the label round-trip.</strong> {@code LeadTabScanner} used to be able to
|
||||
* promise that bridged never writes a lead label. That is no longer true — an auto-launched lead is
|
||||
* labelled by this class, and the scanner reads that label back. The risk this opens is not
|
||||
* privilege escalation (the tab label never granted anything a pane could take for itself; see that
|
||||
* class's javadoc), but <em>staleness</em>: a label left behind by a crashed session would otherwise
|
||||
* read as a live lead forever, and the lead would never be relaunched. So a lead counts as live only
|
||||
* when herdr also reports a <em>running agent</em> in that tab — see {@link #liveLeads}. A labelled
|
||||
* tab with no agent in it is not a lead.
|
||||
*/
|
||||
public final class LeadLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadLauncher.class);
|
||||
|
||||
private final AgentControl agents;
|
||||
private final WorkspaceControl spaces;
|
||||
private final BridgedConfig cfg;
|
||||
|
||||
/**
|
||||
* @param agents herdr agent control (start, list)
|
||||
* @param spaces workspace / tab control (ensure, create, label, list)
|
||||
* @param cfg the loaded config — {@code fleet.leaders}, {@code profiles} and each lead's tab
|
||||
*/
|
||||
public LeadLauncher(AgentControl agents, WorkspaceControl spaces, BridgedConfig cfg) {
|
||||
this.agents = agents;
|
||||
this.spaces = spaces;
|
||||
this.cfg = cfg;
|
||||
}
|
||||
|
||||
/**
|
||||
* Bring every declared lead up to its {@code instances} count, and return how many were started.
|
||||
*
|
||||
* <p>Never throws: a daemon that cannot start a lead must still serve. herdr being unreachable,
|
||||
* a profile that does not exist, a failed {@code agent.start} — each is logged and skipped, and
|
||||
* the remaining leads are still attempted.
|
||||
*/
|
||||
public int ensureLeads() {
|
||||
Map<String, BridgedConfig.Leader> leaders = cfg.fleet().leaders();
|
||||
if (leaders.isEmpty()) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
Map<String, Integer> live;
|
||||
try {
|
||||
live = liveLeads(leaders);
|
||||
} catch (HerdrException e) {
|
||||
// Counting is the whole safety mechanism against double-spawning. If we cannot count, we
|
||||
// must not guess — spawning a second orchestrator is worse than starting none.
|
||||
log.warn("lead auto-launch skipped — cannot tell which leads are live: {}", e.getMessage());
|
||||
return 0;
|
||||
}
|
||||
|
||||
int started = 0;
|
||||
for (Map.Entry<String, BridgedConfig.Leader> e : leaders.entrySet()) {
|
||||
String name = e.getKey();
|
||||
BridgedConfig.Leader lead = e.getValue();
|
||||
int running = live.getOrDefault(name, 0);
|
||||
int wanted = lead.instances();
|
||||
|
||||
if (running >= wanted) {
|
||||
log.info("lead '{}': {} live, {} wanted — nothing to start", name, running, wanted);
|
||||
continue;
|
||||
}
|
||||
if (!lead.isCreatable()) {
|
||||
// A lead with a `tab:` but no `profile:` is recognise-only by design: the operator
|
||||
// opens it by hand. Say so once rather than looking like a silent failure.
|
||||
log.info("lead '{}' is not live, and names no profile — it can be recognised but not "
|
||||
+ "launched. Add `profile:` under fleet.leaders.{} to have bridged start it.",
|
||||
name, name);
|
||||
continue;
|
||||
}
|
||||
|
||||
BridgedConfig.Profile profile = cfg.profiles().get(lead.profile());
|
||||
if (profile == null) {
|
||||
log.warn("lead '{}' names profile '{}', which is not configured — not launching",
|
||||
name, lead.profile());
|
||||
continue;
|
||||
}
|
||||
|
||||
for (int i = running; i < wanted; i++) {
|
||||
if (launch(name, lead, profile)) {
|
||||
started++;
|
||||
}
|
||||
}
|
||||
}
|
||||
return started;
|
||||
}
|
||||
|
||||
/**
|
||||
* How many live leads exist per configured name: a running agent in a tab labelled with that
|
||||
* lead's exact {@code tab} (CB-579). Member workspaces are excluded, exactly as the scanner
|
||||
* excludes them: a member must not be counted as a lead because it happens to sit in a matching
|
||||
* tab.
|
||||
*
|
||||
* <p>There used to be a second path here — a running agent on the terminal a
|
||||
* {@code fleet.leaders.<name>.terminal} pin named, for a lead opened and pinned by hand. That
|
||||
* pin is retired: {@code tab} is now the only field identity depends on, and {@link Agent}
|
||||
* already carries {@link Agent#tabId()} directly, so a hand-opened lead is found the same way an
|
||||
* auto-launched one is — by labelling its tab to match.
|
||||
*/
|
||||
private Map<String, Integer> liveLeads(Map<String, BridgedConfig.Leader> leaders) {
|
||||
Set<String> memberSpaces = cfg.profiles().values().stream()
|
||||
.map(BridgedConfig.Profile::workspace)
|
||||
.filter(w -> w != null && !w.isBlank())
|
||||
.collect(Collectors.toSet());
|
||||
|
||||
// tabId → the lead name its label declares.
|
||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
||||
for (Workspace ws : spaces.listWorkspaces()) {
|
||||
if (ws.workspaceId() == null || memberSpaces.contains(ws.label())) {
|
||||
continue;
|
||||
}
|
||||
for (Tab tab : spaces.listTabs(ws.workspaceId())) {
|
||||
String declared = leadNameOf(tab.label(), leaders);
|
||||
if (declared != null && tab.tabId() != null) {
|
||||
nameByTab.put(tab.tabId(), declared);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Map<String, Integer> counts = new LinkedHashMap<>();
|
||||
for (Agent a : agents.list()) {
|
||||
String name = nameByTab.get(a.tabId());
|
||||
if (name != null) {
|
||||
counts.merge(name, 1, Integer::sum);
|
||||
}
|
||||
}
|
||||
return counts;
|
||||
}
|
||||
|
||||
/**
|
||||
* The configured lead a tab label names, or {@code null} for a label that names none.
|
||||
*
|
||||
* <p>Matched exactly (case-insensitively) against each lead's configured {@code tab}, so an
|
||||
* operator's {@code "lead: something-else"} tab is not mistaken for a configured lead.
|
||||
*/
|
||||
private String leadNameOf(String label, Map<String, BridgedConfig.Leader> leaders) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
String l = label.strip();
|
||||
for (Map.Entry<String, BridgedConfig.Leader> e : leaders.entrySet()) {
|
||||
String tab = e.getValue().tabLabel();
|
||||
if (tab != null && l.equalsIgnoreCase(tab.strip())) {
|
||||
return e.getKey();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Start one lead. Returns false (having logged) rather than throwing on any failure. */
|
||||
private boolean launch(String name, BridgedConfig.Leader lead, BridgedConfig.Profile profile) {
|
||||
String label = lead.tabLabel();
|
||||
String cwd = (lead.cwd() == null || lead.cwd().isBlank())
|
||||
? System.getProperty("user.dir") : lead.cwd();
|
||||
|
||||
Tab.Created tab = null;
|
||||
try {
|
||||
Workspace ws = spaces.ensureWorkspace(lead.workspace());
|
||||
tab = spaces.createTab(ws.workspaceId(), cwd, leadEnv(profile));
|
||||
if (tab.rootPaneId() == null) {
|
||||
throw new IllegalStateException("tab " + tab.tab().tabId()
|
||||
+ " came back with no seed pane — nowhere to start the lead");
|
||||
}
|
||||
// Same shape as the member launchers: herdr resolves the executable from `kind`, so
|
||||
// argv[0] (the configured launcher, e.g. `ccs`) is dropped and only the rest is passed.
|
||||
List<String> argv = leadArgv(profile);
|
||||
Agent started = agents.start("lead-" + name, herdrKind(profile),
|
||||
argv.isEmpty() ? argv : argv.subList(1, argv.size()), tab.rootPaneId());
|
||||
|
||||
// Label AFTER the start succeeds. A label written before would survive a failed start
|
||||
// and then read back as a live lead on the next boot, which is the exact staleness the
|
||||
// agent-liveness check exists to prevent — no need to create the case ourselves.
|
||||
spaces.renameTab(tab.tab().tabId(), label);
|
||||
|
||||
log.info("lead '{}' launched: profile={} tab={} pane={} terminal={} label='{}' cwd={}",
|
||||
name, profile.profile(), tab.tab().tabId(), started.paneId(),
|
||||
started.terminalId(), label, cwd);
|
||||
return true;
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("lead '{}' failed to launch on profile '{}': {}",
|
||||
name, profile.profile(), e.getMessage());
|
||||
if (tab != null && tab.tab() != null && tab.tab().tabId() != null) {
|
||||
try {
|
||||
spaces.closeTab(tab.tab().tabId());
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("could not close the orphaned lead tab {}: {}",
|
||||
tab.tab().tabId(), cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The herdr agent kind for this profile — the same value the matching member adapter passes, so
|
||||
* herdr resolves the same executable for a lead as it does for a member on that backend.
|
||||
*/
|
||||
private static String herdrKind(BridgedConfig.Profile profile) {
|
||||
return profile.isOpenCode() ? "opencode" : "claude";
|
||||
}
|
||||
|
||||
/**
|
||||
* The lead's argv: the profile's own command, the model pin, and the bridge MCP mount.
|
||||
*
|
||||
* <p>No {@code --append-system-prompt}. That flag carries the worker reply charter, and a lead
|
||||
* is not a worker — it reads its orchestration rules from the project's {@code CLAUDE.md} like
|
||||
* any other primary. This is the single most important difference from the member launchers;
|
||||
* do not "unify" it back.
|
||||
*/
|
||||
private List<String> leadArgv(BridgedConfig.Profile profile) {
|
||||
List<String> argv = new ArrayList<>(profile.argv());
|
||||
if (profile.hasMcp()) {
|
||||
argv.add("--mcp-config");
|
||||
argv.add("{\"mcpServers\":{\"bridge\":{\"type\":\"http\",\"url\":\""
|
||||
+ profile.mcpUrl() + "\"}}}");
|
||||
}
|
||||
// Appended last, for the same reason the member launcher does it (CB-533): the argv is
|
||||
// usually a wrapper such as `ccs <profile>`, which exports its own model family over
|
||||
// whatever it inherited, and --model outranks the environment.
|
||||
if (profile.model() != null && !profile.model().isBlank()) {
|
||||
argv.add("--model");
|
||||
argv.add(profile.model());
|
||||
}
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* The lead's environment: the profile's {@code env:} block, and nothing that could move it off
|
||||
* the subscription.
|
||||
*
|
||||
* <p>{@code ANTHROPIC_BASE_URL} and {@code ANTHROPIC_AUTH_TOKEN} are stripped unconditionally —
|
||||
* not defaulted, not guarded, stripped. A lead runs on the operator's subscription by
|
||||
* definition, so there is no configuration under which pointing it elsewhere is correct, and a
|
||||
* profile that carries them (a member profile reused as a lead's backend) must not leak them in.
|
||||
*/
|
||||
private Map<String, String> leadEnv(BridgedConfig.Profile profile) {
|
||||
Map<String, String> out = new LinkedHashMap<>();
|
||||
if (profile.env() != null) {
|
||||
out.putAll(profile.env());
|
||||
}
|
||||
out.remove("ANTHROPIC_BASE_URL");
|
||||
out.remove("ANTHROPIC_AUTH_TOKEN");
|
||||
if (profile.configDir() != null && !profile.configDir().isBlank()) {
|
||||
out.put("CLAUDE_CONFIG_DIR", profile.configDir());
|
||||
}
|
||||
// Deliberately no git token: a lead reviews and merges through the operator's own
|
||||
// credentials, and never needs the scoped write:repository token a member is granted.
|
||||
out.values().removeIf(Objects::isNull);
|
||||
return out;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
package dev.ltms.bridged.logging;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.turbo.TurboFilter;
|
||||
import ch.qos.logback.core.spi.FilterReply;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import org.slf4j.Marker;
|
||||
|
||||
/** Suppresses only the SDK warning for the normal MCP cancellation notification. */
|
||||
public final class McpCancelledNotificationFilter extends TurboFilter {
|
||||
|
||||
static final String LOGGER = "io.modelcontextprotocol.spec.McpStreamableServerSession";
|
||||
static final String UNHANDLED_NOTIFICATION = "No handler registered for notification method: {}";
|
||||
|
||||
@Override
|
||||
public FilterReply decide(Marker marker, Logger logger, Level level, String format, Object[] params,
|
||||
Throwable throwable) {
|
||||
if (level == Level.WARN
|
||||
&& LOGGER.equals(logger.getName())
|
||||
&& UNHANDLED_NOTIFICATION.equals(format)
|
||||
&& params != null
|
||||
&& params.length == 1
|
||||
&& params[0] instanceof McpSchema.JSONRPCNotification notification
|
||||
&& "notifications/cancelled".equals(notification.method())) {
|
||||
return FilterReply.DENY;
|
||||
}
|
||||
return FilterReply.NEUTRAL;
|
||||
}
|
||||
}
|
||||
@@ -1,15 +1,26 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
import dev.ltms.bridged.auth.AuditLog;
|
||||
import dev.ltms.bridged.auth.Authz;
|
||||
import dev.ltms.bridged.auth.CallerResolver;
|
||||
import dev.ltms.bridged.auth.Principal;
|
||||
import dev.ltms.bridged.auth.Role;
|
||||
import dev.ltms.bridged.guard.GuardException;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.inject.WorkerPresence;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.placement.BackendQuarantine;
|
||||
import dev.ltms.bridged.placement.PlacementException;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.session.WorkerSession;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import dev.ltms.bridged.session.WorktreeRequest;
|
||||
import dev.ltms.bridged.worker.ClaudeCodeLauncher;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import io.modelcontextprotocol.common.McpTransportContext;
|
||||
import io.modelcontextprotocol.json.McpJsonMapper;
|
||||
import io.modelcontextprotocol.json.jackson3.JacksonMcpJsonMapperSupplier;
|
||||
@@ -24,7 +35,11 @@ import jakarta.servlet.http.HttpServlet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
@@ -34,8 +49,8 @@ import java.util.stream.Collectors;
|
||||
* / {@code bridge_status}; the worker calls {@code bridge_reply}.
|
||||
*
|
||||
* <p>Beyond delegation the primary also manages the fleet here (CB-108): {@code bridge_spawn} /
|
||||
* {@code bridge_list} / {@code bridge_stop} adapt {@link ClaudeCodeLauncher} so a worker's whole lifecycle
|
||||
* is driven through MCP, with the subscription boundary still enforced inside {@code ClaudeCodeLauncher}.
|
||||
* {@code bridge_list} / {@code bridge_stop} drive the {@link PeerLauncher} SPI so a worker's whole
|
||||
* lifecycle is managed through MCP, with each adapter's subscription boundary enforced inside it.
|
||||
*
|
||||
* <p>The tool <em>logic</em> lives in package-private static methods returning a
|
||||
* {@link McpSchema.CallToolResult}, so it is unit-testable without standing up the HTTP transport;
|
||||
@@ -55,12 +70,55 @@ public final class BridgeMcp {
|
||||
static final String CALLER_TERMINAL = "callerTerminal";
|
||||
/** Transport-context key under which the extractor stashes the caller's PID (for cwd inherit). */
|
||||
static final String CALLER_PID = "callerPid";
|
||||
/** Transport-context key under which the extractor stashes the resolved {@link Role} (CB-501). */
|
||||
static final String CALLER_ROLE = "callerRole";
|
||||
/** Transport-context key for the lead's configured name, when the caller is one (CB-530). */
|
||||
static final String CALLER_NAME = "callerName";
|
||||
|
||||
private final HttpServletStreamableServerTransportProvider transport;
|
||||
private final McpSyncServer server;
|
||||
private final CallerResolver authz; // CB-501: null → authorization not enforced (legacy)
|
||||
private final Metrics metrics; // CB-502: null → auth failures not counted
|
||||
private final CapacitySource capacity;
|
||||
private final HealthCoverageSource healthCoverage;
|
||||
private final QuarantineSource quarantine;
|
||||
|
||||
public BridgeMcp(MessageService messages, Rendezvous rendezvous, ClaudeCodeLauncher workers,
|
||||
SessionManager sessions, ConnectionIdentity identity, WorkerPresence presence) {
|
||||
/** Capacity facts used by {@code bridge_list}; production must supply the placement live count. */
|
||||
public record CapacitySource(Function<String, Integer> liveCount, Function<String, Integer> maxLoad,
|
||||
Supplier<Set<String>> configuredProfiles, LongSupplier clock) {
|
||||
/** Inert test-only source. It omits capacity rather than inventing zero live counts. */
|
||||
public static CapacitySource none() { return new CapacitySource(_ -> 0, _ -> null, Set::of, System::nanoTime); }
|
||||
boolean available() { return !configuredProfiles.get().isEmpty(); }
|
||||
}
|
||||
|
||||
/** Coverage is supplied by the health wiring, not inferred from a missing dependency. */
|
||||
public record HealthCoverageSource(Supplier<String> value) { }
|
||||
|
||||
/**
|
||||
* CB-578 stage B quarantine facts used by {@code bridge_profiles}: a profile → credential id
|
||||
* lookup, plus the shared {@link BackendQuarantine} to read remaining cooldowns off.
|
||||
*/
|
||||
public record QuarantineSource(Function<String, String> credentialIdFor, BackendQuarantine quarantine) {
|
||||
/** Inert source — no profile is ever reported quarantined. Explicit stand-in, not a default. */
|
||||
public static QuarantineSource none() { return new QuarantineSource(_ -> null, BackendQuarantine.none()); }
|
||||
}
|
||||
|
||||
/**
|
||||
* @param callers resolves each call's {@link Principal}; {@code null} disables authorization.
|
||||
* This surface needs its own enforcement: {@code /mcp} is a raw servlet on
|
||||
* Jetty's context handler and never passes through Javalin's {@code before}
|
||||
* filter, so the REST guard does not cover it.
|
||||
* @param metrics registry for auth-failure counting; may be {@code null}
|
||||
* @param quarantine CB-578 stage B facts for {@code bridge_profiles}; required — pass
|
||||
* {@link QuarantineSource#none()} for a caller that does not want the feature
|
||||
*/
|
||||
public BridgeMcp(MessageService messages, PeerLauncher workers, SessionManager sessions,
|
||||
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||
CallerResolver callers, Metrics metrics, CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine) {
|
||||
this.capacity = capacity;
|
||||
this.quarantine = Objects.requireNonNull(quarantine, "quarantine");
|
||||
this.healthCoverage = healthCoverage;
|
||||
McpJsonMapper json = new JacksonMcpJsonMapperSupplier().get();
|
||||
this.transport = HttpServletStreamableServerTransportProvider.builder()
|
||||
.jsonMapper(json)
|
||||
@@ -70,53 +128,234 @@ public final class BridgeMcp {
|
||||
// inherit the primary's cwd (CB-112). Any contact from a worker marks it available
|
||||
// (CB-113) — its MCP initialize is the reliable "the agent is up" signal.
|
||||
.contextExtractor(req -> {
|
||||
ConnectionIdentity.Caller c = identity.resolve(req.getRemoteAddr(), req.getRemotePort());
|
||||
presence.markPresent(c.terminal()); // no-op for the primary (null terminal)
|
||||
// One resolution per call, shared with the REST surface via CallerResolver so
|
||||
// the two paths cannot drift on who a caller is.
|
||||
Principal p = callers != null
|
||||
? callers.resolve(req.getRemoteAddr(), req.getRemotePort(),
|
||||
req.getHeader("Authorization"))
|
||||
: legacyPrincipal(identity, req.getRemoteAddr(), req.getRemotePort());
|
||||
// CB-532: guard on the ROLE, not on the terminal being null. This excludes a
|
||||
// lead, which carries its pane too, while including every spawned member role.
|
||||
// Enrolling a lead would count it as an available member in the roster.
|
||||
markSpawnedMemberPresent(p, presence);
|
||||
return McpTransportContext.create(Map.of(
|
||||
CALLER_TERMINAL, orEmpty(c.terminal()),
|
||||
CALLER_PID, Long.toString(c.pid())));
|
||||
CALLER_TERMINAL, orEmpty(p.terminal()),
|
||||
CALLER_PID, Long.toString(p.pid()),
|
||||
CALLER_ROLE, p.role().name(),
|
||||
CALLER_NAME, orEmpty(p.name())));
|
||||
})
|
||||
.build();
|
||||
this.server = McpServer.sync(transport)
|
||||
.serverInfo("bridge", "0.1.0")
|
||||
.capabilities(McpSchema.ServerCapabilities.builder().tools(true).build())
|
||||
.toolCall(sendTool(), (_, req) -> {
|
||||
.toolCall(sendTool(), (exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.SEND,
|
||||
str(req.arguments(), "sessionId"));
|
||||
if (denied != null) return denied;
|
||||
String caller = callerTerminal(exchange);
|
||||
// CB-548: only a PRIMARY caller may claim the legacy singleton "primary" fallback.
|
||||
// An architect delegates as its own pane but must never become the fallback that
|
||||
// no-delegation inbox nudges target as if it were the primary (the per-target
|
||||
// delegation map does not cure the singleton).
|
||||
recordPrimarySingleton(primaryRegistry, caller, principal(exchange));
|
||||
Map<String, Object> a = req.arguments();
|
||||
String target = str(a, "sessionId");
|
||||
String content = str(a, "content");
|
||||
String turnId = str(a, "turnId");
|
||||
if (turnId != null && !turnId.isBlank()) {
|
||||
// Answering a worker's bridge_ask (CB-205): resolve its blocked question and
|
||||
// block for the worker's reply as it resumes the same turn.
|
||||
return answer(messages, turnId, str(a, "content"), timeoutMs(a));
|
||||
// block for the worker's reply as it resumes the same turn. This is the same
|
||||
// delegation, so ownership is left untouched (CB-548) — never re-recorded.
|
||||
return answer(messages, turnId, content, timeoutMs(a));
|
||||
}
|
||||
// CB-548: delegator ownership (which lead's reply nudge this worker routes to,
|
||||
// CB-532) is recorded only once the send is ACCEPTED — MessageService has won the
|
||||
// session lock and queued delivery — via the accepted-delivery callback, never at
|
||||
// request time. A concurrent sender that times out BUSY therefore cannot steal a
|
||||
// live turn's reply routing without ever owning the turn.
|
||||
Runnable onAccepted = () -> primaryRegistry.recordDelegation(target, caller);
|
||||
// wait defaults to true (block for the reply); wait:false is fire-and-poll.
|
||||
return Boolean.FALSE.equals(a.get("wait"))
|
||||
? sendAsync(messages, str(a, "sessionId"), str(a, "content"))
|
||||
: send(messages, str(a, "sessionId"), str(a, "content"), timeoutMs(a));
|
||||
? sendAsync(messages, target, content, onAccepted, workers.profiles())
|
||||
: send(messages, target, content, timeoutMs(a), onAccepted, workers.profiles());
|
||||
})
|
||||
// bridge_reply's identity is the CONNECTION, never an argument — so the authz check
|
||||
// is "is this caller a worker at all", and it can only ever reply as itself.
|
||||
.toolCall(replyTool(), (exchange, req) -> {
|
||||
String self = callerTerminal(exchange);
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.REPLY, self);
|
||||
if (denied != null) return denied;
|
||||
return reply(messages, self, str(req.arguments(), "content"));
|
||||
})
|
||||
// bridge_reply's identity is the CONNECTION, never an argument.
|
||||
.toolCall(replyTool(), (exchange, req) ->
|
||||
reply(rendezvous, callerTerminal(exchange), str(req.arguments(), "content")))
|
||||
// bridge_ask (CB-205): a worker's mid-turn question — identity from the CONNECTION.
|
||||
.toolCall(askTool(), (exchange, req) ->
|
||||
ask(messages, callerTerminal(exchange), str(req.arguments(), "question"), timeoutMs(req.arguments())))
|
||||
.toolCall(statusTool(), (_, req) ->
|
||||
status(messages, str(req.arguments(), "sessionId")))
|
||||
.toolCall(pollTool(), (_, req) ->
|
||||
poll(messages, str(req.arguments(), "ticket")))
|
||||
.toolCall(askTool(), (exchange, req) -> {
|
||||
String self = callerTerminal(exchange);
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.ASK, self);
|
||||
if (denied != null) return denied;
|
||||
return ask(messages, self, str(req.arguments(), "question"), timeoutMs(req.arguments()));
|
||||
})
|
||||
.toolCall(statusTool(), (exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
return status(messages, str(req.arguments(), "sessionId"));
|
||||
})
|
||||
.toolCall(pollTool(), (exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
Map<String, Object> a = req.arguments();
|
||||
return poll(messages, str(a, "ticket"), str(a, "target"));
|
||||
})
|
||||
// CB-307 Increment 3: per-msgId ack (not needed in v1 but supported by the inbox).
|
||||
// Acking removes a reply from the inbox, so it is a drain, not a read.
|
||||
.toolCall(ackTool(), (exchange, req) -> {
|
||||
Map<String, Object> a = req.arguments();
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.DRAIN, str(a, "target"));
|
||||
if (denied != null) return denied;
|
||||
return ack(messages, str(a, "target"), str(a, "msgId"));
|
||||
})
|
||||
// Fleet management (CB-108): spawn/list/stop over ClaudeCodeLauncher.
|
||||
.toolCall(spawnTool(), (exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.SPAWN, null);
|
||||
if (denied != null) return denied;
|
||||
String caller = callerTerminal(exchange);
|
||||
// SPAWN is already auth-gated to PRIMARY (architects can never call it), but
|
||||
// enforce the same invariant here: only a PRIMARY may claim the legacy singleton.
|
||||
recordPrimarySingleton(primaryRegistry, caller, principal(exchange));
|
||||
Map<String, Object> a = req.arguments();
|
||||
// CB-112: worker inherits the primary's cwd unless the call pins one.
|
||||
// CB-301: carry the caller's identity as the session owner (null for the primary).
|
||||
// CB-301-ext: optional isolated worktree for parallel implementers.
|
||||
String callerCwd = identity.cwdForPid(callerPid(exchange));
|
||||
return spawn(sessions, str(a, "profile"), str(a, "cwd"), callerCwd,
|
||||
callerTerminal(exchange), worktreeRequest(a));
|
||||
return spawn(sessions, str(a, "profile"), str(a, "role"), str(a, "cwd"), callerCwd,
|
||||
callerTerminal(exchange), worktreeRequest(a),
|
||||
str(a, "sessionName"), str(a, "resumeSessionId"));
|
||||
})
|
||||
.toolCall(listTool(), (exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine,
|
||||
callers == null ? Map.of() : callers.leads(),
|
||||
callerTerminal(exchange));
|
||||
})
|
||||
.toolCall(stopTool(), (exchange, req) -> {
|
||||
String paneId = str(req.arguments(), "paneId");
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.STOP, paneId);
|
||||
if (denied != null) return denied;
|
||||
return stop(sessions, paneId);
|
||||
})
|
||||
.toolCall(profilesTool(), (exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
return profiles(workers, quarantine);
|
||||
})
|
||||
.toolCall(whoamiTool(), (exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
return whoami(principal(exchange), sessions);
|
||||
})
|
||||
.toolCall(listTool(), (_, _) -> listWorkers(workers, sessions))
|
||||
.toolCall(stopTool(), (_, req) -> stop(sessions, str(req.arguments(), "paneId")))
|
||||
.toolCall(profilesTool(), (_, _) -> profiles(workers))
|
||||
.build();
|
||||
this.authz = callers;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pre-CB-501 identity: worker if the connection maps to a pane, otherwise the primary. Used
|
||||
* only by the legacy constructor, where authorization is not enforced anyway.
|
||||
*/
|
||||
private static Principal legacyPrincipal(ConnectionIdentity identity, String addr, int port) {
|
||||
ConnectionIdentity.Caller c = identity.resolve(addr, port);
|
||||
return c.terminal() != null
|
||||
? Principal.worker(c.terminal(), c.pid())
|
||||
: Principal.primary(c.pid());
|
||||
}
|
||||
|
||||
/** The caller reconstructed from the transport context. */
|
||||
private static Principal principal(McpSyncServerExchange exchange) {
|
||||
return principalFrom(exchange.transportContext().get(CALLER_ROLE),
|
||||
callerTerminal(exchange), callerPid(exchange), callerName(exchange));
|
||||
}
|
||||
|
||||
/**
|
||||
* Rebuild a {@link Principal} from the three values the context extractor stashed.
|
||||
*
|
||||
* <p>Split out from {@link #principal(McpSyncServerExchange)} so the identity rules are
|
||||
* reachable without an {@code McpSyncServerExchange} — that is an SDK type this project has no
|
||||
* mocking library to fabricate, which is why this logic had no test at all until CB-513.
|
||||
*
|
||||
* @param role the stashed {@link Role} name, or {@code null} on the legacy path
|
||||
* @param terminal the worker terminal, or {@code null} for a non-worker
|
||||
* @param pid the calling pid, or {@code -1}
|
||||
*/
|
||||
static Principal principalFrom(Object role, String terminal, long pid) {
|
||||
return principalFrom(role, terminal, pid, null);
|
||||
}
|
||||
|
||||
/** As {@link #principalFrom(Object, String, long)}, carrying a lead's name (CB-530). */
|
||||
static Principal principalFrom(Object role, String terminal, long pid, String name) {
|
||||
if (role == null) {
|
||||
// No role stashed (legacy path): fall back to the historical interpretation.
|
||||
return terminal != null ? Principal.worker(terminal, pid) : Principal.primary(pid);
|
||||
}
|
||||
return new Principal(Role.valueOf(role.toString()), terminal, pid, name);
|
||||
}
|
||||
|
||||
/**
|
||||
* Gate a tool call on the CB-505 table. Returns {@code null} when the call may proceed, or the
|
||||
* error result to return when it may not.
|
||||
*/
|
||||
private McpSchema.CallToolResult deny(McpSyncServerExchange exchange, Authz.Action action,
|
||||
String target) {
|
||||
return denyFor(principal(exchange), action, target);
|
||||
}
|
||||
|
||||
/**
|
||||
* The policy half of {@link #deny}: everything except pulling the caller out of the MCP
|
||||
* exchange. Kept separate so the authorization decision — the actual control — is unit-testable
|
||||
* without fabricating an SDK {@code McpSyncServerExchange}.
|
||||
*
|
||||
* <p>This surface exists because the enforcement was previously unreachable from a test: no
|
||||
* test constructs a {@code BridgeMcp}, so the whole MCP-side gate ran zero times in the suite
|
||||
* while the REST-side equivalent had ten tests. A security control nothing exercises is a
|
||||
* claim, not a control.
|
||||
*
|
||||
* @return {@code null} when the call may proceed, or the error result to return when it may not
|
||||
*/
|
||||
McpSchema.CallToolResult denyFor(Principal caller, Authz.Action action, String target) {
|
||||
// The enforcement switch lives HERE rather than in the exchange-facing wrapper: any future
|
||||
// tool that calls this directly must not be able to skip the gate by accident.
|
||||
if (authz == null) {
|
||||
return null; // legacy constructor: authorization not enforced
|
||||
}
|
||||
if (Authz.permits(caller, action, target)) {
|
||||
if (action != Authz.Action.READ) {
|
||||
AuditLog.allowed(caller, action, target); // reads would drown the trail
|
||||
}
|
||||
return null;
|
||||
}
|
||||
String reason = Authz.isUnauthenticated(caller) ? "unauthenticated" : "forbidden";
|
||||
AuditLog.denied(caller, action, target, reason);
|
||||
if (metrics != null) {
|
||||
metrics.inc(BridgedMetrics.AUTH_FAILURES, "reason", reason);
|
||||
}
|
||||
return error(reason + ": " + caller.describe() + " may not " + action);
|
||||
}
|
||||
|
||||
/**
|
||||
* Update the legacy singleton "primary" fallback used for no-delegation inbox nudges (CB-548).
|
||||
*
|
||||
* <p>Only {@link Role#PRIMARY} callers — the unnamed primary and named leads alike — may claim
|
||||
* it. An architect delegates as its own pane but must never become the fallback: the per-target
|
||||
* delegation map ({@code PrimaryRegistry#recordDelegation}) does not cure the singleton, so an
|
||||
* architect left here would draw nudges that belong to a primary. The decision uses the resolved
|
||||
* role, never name/kind sniffing. A null {@code caller} (legacy/no-auth path) records nothing.
|
||||
*
|
||||
* <p>Split out of the tool handlers so the guard is unit-testable without fabricating an SDK
|
||||
* {@code McpSyncServerExchange} (same pattern as {@link #denyFor}/{@link #principalFrom}).
|
||||
*/
|
||||
static void recordPrimarySingleton(PrimaryRegistry registry, String callerTerminal, Principal caller) {
|
||||
if (caller != null && caller.isPrimary()) {
|
||||
registry.record(callerTerminal);
|
||||
}
|
||||
}
|
||||
|
||||
/** The worker identity resolved from this call's connection, or {@code null} if the primary. */
|
||||
@@ -126,6 +365,13 @@ public final class BridgeMcp {
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
}
|
||||
|
||||
/** The lead name resolved from this call's connection, or {@code null} (CB-530). */
|
||||
private static String callerName(McpSyncServerExchange exchange) {
|
||||
Object v = exchange.transportContext().get(CALLER_NAME);
|
||||
String s = v == null ? null : v.toString();
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
}
|
||||
|
||||
/** The caller's PID resolved from this call's connection, or {@code -1} if unknown. */
|
||||
private static long callerPid(McpSyncServerExchange exchange) {
|
||||
Object v = exchange.transportContext().get(CALLER_PID);
|
||||
@@ -145,6 +391,13 @@ public final class BridgeMcp {
|
||||
return transport;
|
||||
}
|
||||
|
||||
/** Mark a connected spawned member available for the injector readiness gate. */
|
||||
static void markSpawnedMemberPresent(Principal caller, MemberPresence presence) {
|
||||
if (caller.isSpawnedMember()) {
|
||||
presence.markPresent(caller.terminal());
|
||||
}
|
||||
}
|
||||
|
||||
/** Graceful shutdown of the MCP server. */
|
||||
public void close() {
|
||||
server.closeGracefully();
|
||||
@@ -152,14 +405,25 @@ public final class BridgeMcp {
|
||||
|
||||
// --- tool logic (thin adapters over the services; unit-testable) ---------------------------
|
||||
|
||||
/** {@code bridge_send}: delegate {@code content} to a worker session and block for its reply. */
|
||||
static McpSchema.CallToolResult send(MessageService messages, String sessionId, String content, Long timeoutMs) {
|
||||
/**
|
||||
* {@code bridge_send}: delegate {@code content} to a worker session and block for its reply.
|
||||
* The configured profiles are required so a profile name can never bypass target validation.
|
||||
*
|
||||
* (CB-548): {@code onAccepted} records delegator ownership the instant the send is accepted, so
|
||||
* a BUSY interloper never claims a turn it did not win. {@code null} disables recording.
|
||||
*/
|
||||
static McpSchema.CallToolResult send(MessageService messages, String sessionId, String content,
|
||||
Long timeoutMs, Runnable onAccepted, Set<String> profiles) {
|
||||
if (isBlank(sessionId) || isBlank(content)) {
|
||||
return error("sessionId and content are required");
|
||||
}
|
||||
McpSchema.CallToolResult targetError = profileTargetError(sessionId, profiles);
|
||||
if (targetError != null) {
|
||||
return targetError;
|
||||
}
|
||||
long timeout = clamp(timeoutMs == null ? DEFAULT_TIMEOUT_MS : timeoutMs);
|
||||
try {
|
||||
return formatReply(messages.send(sessionId, content, timeout), timeout);
|
||||
return formatReply(messages.send(sessionId, content, timeout, onAccepted), timeout);
|
||||
} catch (HerdrException e) {
|
||||
return error("herdr error contacting session " + sessionId + ": " + e.getMessage());
|
||||
}
|
||||
@@ -212,6 +476,11 @@ public final class BridgeMcp {
|
||||
"[worker finished without a structured bridge_reply — transcript tail follows]\n" + r.text());
|
||||
// The worker ran the turn then wedged (CB-109) — surface the error context.
|
||||
case WORKER_FAILED -> text("[worker failed — turn ended in an unrecoverable state]\n" + r.text());
|
||||
// The backend refused on a subscription usage limit (CB-578 stage A) — the worker's
|
||||
// pane stayed healthy, but its account is exhausted. Distinct from WORKER_FAILED so the
|
||||
// primary gets the real cause, not a generic wedge.
|
||||
case BACKEND_EXHAUSTED -> text("[backend exhausted — the worker's account refused on a "
|
||||
+ "usage limit]\n" + r.text());
|
||||
// The worker paused mid-turn to ask (CB-205) — tell the primary how to answer in-turn.
|
||||
case QUESTION -> text("[question] the worker paused to ask before it can finish:\n" + r.text()
|
||||
+ "\n\nAnswer it by calling bridge_send again with turnId=\"" + r.turnId()
|
||||
@@ -226,19 +495,44 @@ public final class BridgeMcp {
|
||||
/**
|
||||
* {@code bridge_send} with {@code wait:false}: delegate {@code content} and return a ticket
|
||||
* immediately (fire-and-poll), so a long task isn't cut off by the caller's MCP call timeout.
|
||||
* The configured profiles are required so a profile name can never bypass target validation.
|
||||
*
|
||||
* This wires the accepted-delivery hook
|
||||
* (CB-548) so an async flooding send records delegator ownership exactly once it is accepted.
|
||||
*/
|
||||
static McpSchema.CallToolResult sendAsync(MessageService messages, String sessionId, String content) {
|
||||
static McpSchema.CallToolResult sendAsync(MessageService messages, String sessionId, String content,
|
||||
Runnable onAccepted, Set<String> profiles) {
|
||||
if (isBlank(sessionId) || isBlank(content)) {
|
||||
return error("sessionId and content are required");
|
||||
}
|
||||
String ticket = messages.sendAsync(sessionId, content);
|
||||
McpSchema.CallToolResult targetError = profileTargetError(sessionId, profiles);
|
||||
if (targetError != null) {
|
||||
return targetError;
|
||||
}
|
||||
String ticket = messages.sendAsync(sessionId, content, onAccepted);
|
||||
return text("accepted — task delegated. Poll bridge_poll with ticket=" + ticket);
|
||||
}
|
||||
|
||||
/** {@code bridge_poll}: check an async delegation by ticket (pending / done+reply / failed). */
|
||||
static McpSchema.CallToolResult poll(MessageService messages, String ticket) {
|
||||
/** A configured profile is never a send target; other unknown values may be herdr-owned panes. */
|
||||
private static McpSchema.CallToolResult profileTargetError(String sessionId, Set<String> profiles) {
|
||||
if (profiles.contains(sessionId)) {
|
||||
return error("unknown send target \"" + sessionId + "\": it is a configured profile name, not a "
|
||||
+ "session id. Call bridge_list to find a member or lead sessionId.");
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** {@code bridge_poll}: check an async delegation by ticket, or drain a worker's inbox by target. */
|
||||
static McpSchema.CallToolResult poll(MessageService messages, String ticket, String target) {
|
||||
if (!isBlank(target)) {
|
||||
var replies = messages.drainReplies(target);
|
||||
if (replies.isEmpty()) {
|
||||
return text("[]");
|
||||
}
|
||||
return text(json(replies));
|
||||
}
|
||||
if (isBlank(ticket)) {
|
||||
return error("ticket is required");
|
||||
return error("ticket (or target) is required");
|
||||
}
|
||||
MessageService.TaskView v = messages.poll(ticket);
|
||||
if (v == null) {
|
||||
@@ -249,16 +543,20 @@ public final class BridgeMcp {
|
||||
? "[done — worker finished without a structured bridge_reply; transcript tail follows]\n" + v.reply()
|
||||
: v.reply());
|
||||
case PENDING -> text("[pending — " + v.detail() + "]");
|
||||
case ASKING -> text("[question — worker is waiting for your answer]\n" + v.reply()
|
||||
+ "\n\nAnswer it by calling bridge_send again with turnId=\"" + v.turnId()
|
||||
+ "\" and content set to your answer; the worker resumes the same turn.");
|
||||
case FAILED -> text("[failed — " + v.detail() + "]");
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code bridge_reply}: the worker returns its structured answer, resolving the awaiting send.
|
||||
* {@code bridge_reply}: the worker returns its structured answer, resolving the awaiting send
|
||||
* or — when no send is open — queueing the reply in the inbox for later drain (CB-307).
|
||||
* {@code callerTerminal} is resolved from the connection (never an argument); a {@code null}
|
||||
* means the caller is not a known worker (e.g. the primary called it by mistake).
|
||||
*/
|
||||
static McpSchema.CallToolResult reply(Rendezvous rendezvous, String callerTerminal, String content) {
|
||||
static McpSchema.CallToolResult reply(MessageService messages, String callerTerminal, String content) {
|
||||
if (callerTerminal == null) {
|
||||
return error("bridge_reply is for workers only — could not identify the calling worker "
|
||||
+ "from the connection");
|
||||
@@ -266,48 +564,155 @@ public final class BridgeMcp {
|
||||
if (content == null) {
|
||||
return error("content is required");
|
||||
}
|
||||
return rendezvous.resolve(callerTerminal, content)
|
||||
? text("delivered")
|
||||
: error("no send is awaiting a reply for this worker");
|
||||
messages.reply(callerTerminal, content);
|
||||
return text("delivered");
|
||||
}
|
||||
|
||||
/** {@code bridge_status}: the live lifecycle status of a worker session. */
|
||||
/** {@code bridge_ack}: acknowledge (remove) a specific reply from the inbox. */
|
||||
static McpSchema.CallToolResult ack(MessageService messages, String target, String msgId) {
|
||||
if (isBlank(target) || isBlank(msgId)) {
|
||||
return error("target and msgId are required");
|
||||
}
|
||||
messages.ackReply(target, msgId);
|
||||
return text("acknowledged " + msgId);
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code bridge_status}: the live lifecycle status of a worker session, plus — when the worker
|
||||
* is paused mid-turn in an async {@code bridge_ask} (CB-582) — the open question and how to
|
||||
* answer it, so a lead on its normal poll cadence does not need the ticket to notice.
|
||||
*/
|
||||
static McpSchema.CallToolResult status(MessageService messages, String sessionId) {
|
||||
if (isBlank(sessionId)) {
|
||||
return error("sessionId is required");
|
||||
}
|
||||
try {
|
||||
return text(messages.status(sessionId).name().toLowerCase());
|
||||
String base = messages.status(sessionId).name().toLowerCase();
|
||||
MessageService.PendingAsk ask = messages.pendingAsk(sessionId);
|
||||
if (ask == null) {
|
||||
return text(base);
|
||||
}
|
||||
return text(base + "\n\n[question — worker is waiting for your answer]\n" + ask.question()
|
||||
+ "\n\nAnswer it by calling bridge_send again with turnId=\"" + ask.turnId()
|
||||
+ "\" and content set to your answer; the worker resumes the same turn."
|
||||
+ " (ticket " + ask.ticket() + ")");
|
||||
} catch (HerdrException e) {
|
||||
return error("herdr error for session " + sessionId + ": " + e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code bridge_whoami}: the caller's own identity, as the daemon already resolved it.
|
||||
*
|
||||
* <p>Every other tool <em>consumes</em> this identity — the authorization gate, the reply
|
||||
* rendezvous, the cwd inherit — but none reported it, so an agent had to infer its own role
|
||||
* from side channels the daemon does not control: a charter string in its system prompt, the
|
||||
* name its MCP mount happens to carry, or {@code ANTHROPIC_BASE_URL} (which Claude-model
|
||||
* workers do not set). The failure mode of guessing is asymmetric and silent: a primary that
|
||||
* mistakes itself for a worker is refused by {@link Authz} and learns immediately, while a
|
||||
* worker that mistakes itself for the primary ends its turn without {@code bridge_reply} and
|
||||
* the sender simply receives nothing. This tool removes the guess.
|
||||
*
|
||||
* <p>For a worker the session registry adds what it knows about that session. A worker the
|
||||
* registry has no record of — one that outlived a daemon restart — still gets its role and
|
||||
* {@code sessionId}, which is the load-bearing part.
|
||||
*/
|
||||
static McpSchema.CallToolResult whoami(Principal caller, SessionManager sessions) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("role", caller.role().name().toLowerCase());
|
||||
if (caller.isArchitect()) {
|
||||
// CB-548: the role reads "architect"; the name is the gateway-local slot the pane is
|
||||
// bound to, and the pane itself so a peer knows where to reach it.
|
||||
if (caller.name() != null) {
|
||||
m.put("architect", caller.name());
|
||||
}
|
||||
if (caller.terminal() != null) {
|
||||
m.put("sessionId", caller.terminal());
|
||||
}
|
||||
return text(json(m));
|
||||
}
|
||||
if (!caller.isWorker()) {
|
||||
// CB-530: which lead, once more than one pane is configured as one. `role` deliberately
|
||||
// still reads "primary" — the fallback ladder in CLAUDE.md keys on it, and a lead IS a
|
||||
// primary for authorization; the name is additive so no existing reader breaks.
|
||||
if (caller.name() != null) {
|
||||
m.put("leader", caller.name());
|
||||
}
|
||||
// CB-532: a lead's own pane, so it can tell a peer where to reach it — and so an
|
||||
// operator can read off which tab hosts which lead without going to herdr.
|
||||
if (caller.terminal() != null) {
|
||||
m.put("sessionId", caller.terminal());
|
||||
}
|
||||
return text(json(m));
|
||||
}
|
||||
m.put("sessionId", caller.terminal());
|
||||
sessions.roster().stream()
|
||||
.filter(s -> caller.terminal().equals(s.terminalId()))
|
||||
.findFirst()
|
||||
.ifPresent(s -> {
|
||||
m.put("paneId", s.paneId());
|
||||
m.put("profile", s.profile());
|
||||
m.put("state", s.state().name().toLowerCase());
|
||||
if (s.worktree() != null) {
|
||||
m.put("worktree", s.worktree());
|
||||
}
|
||||
if (s.branch() != null) {
|
||||
m.put("branch", s.branch());
|
||||
}
|
||||
if (s.ownerTerminal() != null) {
|
||||
m.put("owner", s.ownerTerminal());
|
||||
}
|
||||
});
|
||||
return text(json(m));
|
||||
}
|
||||
|
||||
// --- fleet management logic (CB-108 / CB-301) --------------------------------------------
|
||||
|
||||
/** {@code bridge_spawn} without cwd/caller context (default resolution). */
|
||||
static McpSchema.CallToolResult spawn(SessionManager sessions, String profile) {
|
||||
return spawn(sessions, profile, null, null, null, null);
|
||||
return spawn(sessions, profile, null, null, null, null, null, null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code bridge_spawn}: launch a guard-checked worker for {@code profile} (blank → the default
|
||||
* profile) and return its session id + pane id. The worker's cwd is {@code requestedCwd} if given,
|
||||
* else the profile's config, else {@code callerCwd} (the primary's directory), else the daemon's.
|
||||
* {@code bridge_spawn}: launch a guard-checked member for {@code profile} (blank → the default
|
||||
* profile) under {@code role} (blank → {@code dev}), and return its session id + pane id. The
|
||||
* member's cwd is {@code requestedCwd} if given, else the profile's config, else
|
||||
* {@code callerCwd} (the primary's directory), else the daemon's.
|
||||
* CB-301: the session is registered with {@code ownerTerminal} as its owner.
|
||||
* CB-301-ext: {@code worktreeRequest} non-null provisions an isolated git worktree.
|
||||
* CB-584: {@code sessionName}/{@code resumeSessionId} request agent session identity — a resumed
|
||||
* conversation requires an explicit {@code profile} whose adapter declares
|
||||
* {@link dev.ltms.bridged.peer.Capability#SESSION_RESUME}, or the spawn is refused rather than
|
||||
* silently starting a cold session.
|
||||
*
|
||||
* <p>{@code role} and {@code profile} are independent: the role picks the contract, the profile
|
||||
* picks the backend. A reviewer on the same profile as the dev it reviews is a normal spawn.
|
||||
*/
|
||||
static McpSchema.CallToolResult spawn(SessionManager sessions, String profile,
|
||||
static McpSchema.CallToolResult spawn(SessionManager sessions, String profile, String role,
|
||||
String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest worktreeRequest) {
|
||||
String ownerTerminal, WorktreeRequest worktreeRequest,
|
||||
String sessionName, String resumeSessionId) {
|
||||
MemberRole memberRole;
|
||||
try {
|
||||
WorkerSession worker = sessions.acquire(isBlank(profile) ? null : profile,
|
||||
requestedCwd, callerCwd, ownerTerminal, worktreeRequest);
|
||||
return text(json(workerView(worker)));
|
||||
memberRole = isBlank(role) ? MemberRole.DEV : MemberRole.parse(role);
|
||||
} catch (IllegalArgumentException e) {
|
||||
return error(e.getMessage());
|
||||
}
|
||||
try {
|
||||
MemberSession member = sessions.acquire(isBlank(profile) ? null : profile, memberRole,
|
||||
requestedCwd, callerCwd, ownerTerminal, worktreeRequest,
|
||||
isBlank(sessionName) ? null : sessionName, isBlank(resumeSessionId) ? null : resumeSessionId);
|
||||
return text(json(memberView(member)));
|
||||
} catch (GuardException e) {
|
||||
return error("subscription boundary: " + e.getMessage());
|
||||
} catch (PlacementException e) {
|
||||
// CB-599: no candidate had capacity (maxLoad, quarantine, or all-exhausted) — distinct
|
||||
// from "profile does not exist" below.
|
||||
return error("no capacity: " + e.getMessage());
|
||||
} catch (IllegalArgumentException e) {
|
||||
return error(e.getMessage()); // unknown / no-default profile
|
||||
return error(e.getMessage()); // unknown / no-default profile, or a refused resumeSessionId
|
||||
} catch (PeerUnreachableException e) {
|
||||
return error("spawn timed out — worker pane never reached injectable state: " + e.getMessage());
|
||||
} catch (HerdrException e) {
|
||||
return error("herdr error spawning worker: " + e.getMessage());
|
||||
}
|
||||
@@ -341,28 +746,165 @@ public final class BridgeMcp {
|
||||
return null;
|
||||
}
|
||||
|
||||
/** {@code bridge_profiles}: the configured worker profiles and the default. */
|
||||
static McpSchema.CallToolResult profiles(ClaudeCodeLauncher workers) {
|
||||
return text(json(Map.of(
|
||||
"profiles", workers.profiles(),
|
||||
"default", workers.defaultProfile() == null ? "" : workers.defaultProfile())));
|
||||
/**
|
||||
* {@code bridge_profiles}: the configured worker profiles, the default, and — CB-578 stage B —
|
||||
* which of them are currently quarantined (backend exhausted) and for how much longer. The
|
||||
* {@code quarantined} key is present only when at least one profile is, so a fleet where nothing
|
||||
* has ever been quarantined gets exactly the pre-stage-B shape.
|
||||
*/
|
||||
static McpSchema.CallToolResult profiles(PeerLauncher workers, QuarantineSource quarantine) {
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
result.put("profiles", workers.profiles());
|
||||
result.put("default", workers.defaultProfile() == null ? "" : workers.defaultProfile());
|
||||
Map<String, Object> quarantined = new LinkedHashMap<>();
|
||||
for (String profile : workers.profiles()) {
|
||||
String credentialId = quarantine.credentialIdFor().apply(profile);
|
||||
if (credentialId == null) {
|
||||
continue;
|
||||
}
|
||||
quarantine.quarantine().remainingSeconds(credentialId).ifPresent(remaining -> {
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("credentialId", credentialId);
|
||||
row.put("quarantinedForSeconds", remaining);
|
||||
quarantined.put(profile, row);
|
||||
});
|
||||
}
|
||||
if (!quarantined.isEmpty()) {
|
||||
result.put("quarantined", quarantined);
|
||||
}
|
||||
return text(json(result));
|
||||
}
|
||||
|
||||
/** {@code bridge_list}: bridge-owned roster merged with live herdr status by paneId. */
|
||||
static McpSchema.CallToolResult listWorkers(ClaudeCodeLauncher workers, SessionManager sessions) {
|
||||
/**
|
||||
* {@code bridge_list}: the whole fleet — {@code leads} and {@code workers} — each merged with
|
||||
* live herdr status. CB-519 decoupled the registry key (a host-unique id) from the herdr pane
|
||||
* coordinate, so the join is on the terminal id, which both the session and the live agent carry.
|
||||
*
|
||||
* <p>CB-535 added the {@code leads} half. Until then this listed the worker roster alone, and a
|
||||
* lead asking "who else is here?" got an empty array — which reads as <em>no peers</em> but
|
||||
* actually means <em>no workers spawned</em>. There was no way at all for a lead to learn a
|
||||
* peer's address; it had to be carried across by a human. Both halves are reported even when a
|
||||
* half is empty, so an empty {@code workers} can no longer be mistaken for an empty fleet.
|
||||
*
|
||||
* <p>Leads are drawn from the resolver rather than from a second registry, so an address listed
|
||||
* here is one that would actually resolve as a lead — see {@link CallerResolver#leads()}. The
|
||||
* caller's own row is flagged {@code "self": true}: a peer needs to tell its own pane apart from
|
||||
* a peer's, and the alternative is every lead calling {@code bridge_whoami} to subtract itself.
|
||||
*
|
||||
* <p>CB-583: the {@code capacity} rows reuse {@code quarantine} (the same {@link QuarantineSource}
|
||||
* {@code bridge_profiles} reads) so the two surfaces cannot disagree about which profile is
|
||||
* quarantined — see {@link #capacityView}.
|
||||
*
|
||||
* @param leads terminal_id → lead name, live from the resolver
|
||||
* @param selfTerm the calling pane's terminal id, or blank for a caller with no pane
|
||||
*/
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions,
|
||||
Map<String, String> leads, String selfTerm) {
|
||||
return listFleet(workers, sessions, null, CapacitySource.none(), new HealthCoverageSource(() -> "off"),
|
||||
QuarantineSource.none(), leads, selfTerm);
|
||||
}
|
||||
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, Map<String, String> leads, String selfTerm) {
|
||||
try {
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.filter(a -> a.paneId() != null)
|
||||
.collect(Collectors.toMap(Agent::paneId, Function.identity(), (_, b) -> b));
|
||||
List<Map<String, Object>> out = sessions.roster().stream()
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.paneId())))
|
||||
.map(Agent.class::cast)
|
||||
.filter(a -> a.terminalId() != null)
|
||||
.collect(Collectors.toMap(Agent::terminalId, Function.identity(), (_, b) -> b));
|
||||
List<Map<String, Object>> leadRows = leads.entrySet().stream()
|
||||
.sorted(Map.Entry.comparingByValue())
|
||||
.map(e -> leadView(e.getKey(), e.getValue(), live.get(e.getKey()), selfTerm))
|
||||
.toList();
|
||||
return text(json(Map.of("workers", out)));
|
||||
List<MemberSession> roster = sessions.roster();
|
||||
List<Map<String, Object>> out = roster.stream()
|
||||
.map(s -> memberCapacityView(s, live.get(s.terminalId()), messages, capacity.clock().getAsLong()))
|
||||
.toList();
|
||||
Set<String> profiles = new java.util.TreeSet<>(capacity.configuredProfiles().get());
|
||||
roster.stream().map(MemberSession::profile).forEach(profiles::add);
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
result.put("leads", leadRows); result.put("members", out);
|
||||
result.put("healthCoverage", healthCoverage.value().get());
|
||||
if (capacity.available()) result.put("capacity", profiles.stream()
|
||||
.map(profile -> capacityView(profile, capacity.liveCount(), capacity.maxLoad(), roster, messages,
|
||||
capacity.clock().getAsLong(), quarantine)).toList());
|
||||
return text(json(result));
|
||||
} catch (HerdrException e) {
|
||||
return error("herdr error listing workers: " + e.getMessage());
|
||||
return error("herdr error listing the fleet: " + e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Capacity is advisory only. {@code reclaimable} says there is no bridge work, not that bridged
|
||||
* may stop the member: the bridge has capacity facts but no work list, and choosing work needs
|
||||
* authority it does not have. {@code idleForSeconds} is derived from monotonic nanoTime and has
|
||||
* no wall-clock meaning across a daemon restart.
|
||||
*/
|
||||
private static Map<String, Object> memberCapacityView(MemberSession session, Agent live,
|
||||
MessageService messages, long nowNanos) {
|
||||
Map<String, Object> row = SessionManager.rosterView(session, live);
|
||||
boolean open = messages != null && messages.hasAcceptedDelivery(session.terminalId());
|
||||
boolean inbox = messages != null && messages.hasInboxMessage(session.terminalId());
|
||||
boolean reclaimable = (session.state() == MemberSession.State.READY || session.state() == MemberSession.State.DONE)
|
||||
&& !open && !inbox;
|
||||
row.put("reclaimable", reclaimable);
|
||||
row.put("idleForSeconds", reclaimable ? Math.max(0, (nowNanos - session.lastActivityAtNanos()) / 1_000_000_000L) : null);
|
||||
return row;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-583: {@code free} alone cannot tell a lead "busy, will free up" from "refusing, and
|
||||
* nothing changes for N seconds" — those need different decisions. So a quarantined profile
|
||||
* forces {@code free} to 0, whatever its {@code maxLoad}/{@code live} say, and the row carries
|
||||
* the same {@code credentialId}/{@code quarantinedForSeconds} facts {@code bridge_profiles}
|
||||
* reports, reusing {@link QuarantineSource} rather than a second lookup. Both new keys are
|
||||
* added only when the profile is actually quarantined, so an ordinary fleet's rows are
|
||||
* byte-identical to before this change.
|
||||
*/
|
||||
private static Map<String, Object> capacityView(String profile, Function<String, Integer> liveCount,
|
||||
Function<String, Integer> maxLoad, List<MemberSession> roster,
|
||||
MessageService messages, long nowNanos, QuarantineSource quarantine) {
|
||||
Integer cap = maxLoad.apply(profile);
|
||||
int live = liveCount.apply(profile);
|
||||
int reclaimable = (int) roster.stream().filter(s -> profile.equals(s.profile()))
|
||||
.filter(s -> (s.state() == MemberSession.State.READY || s.state() == MemberSession.State.DONE))
|
||||
.filter(s -> messages == null || (!messages.hasAcceptedDelivery(s.terminalId()) && !messages.hasInboxMessage(s.terminalId())))
|
||||
.count();
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("profile", profile); row.put("maxLoad", cap); row.put("live", live);
|
||||
row.put("free", cap == null ? null : Math.max(0, cap - live)); row.put("reclaimable", reclaimable);
|
||||
String credentialId = quarantine.credentialIdFor().apply(profile);
|
||||
if (credentialId != null) {
|
||||
quarantine.quarantine().remainingSeconds(credentialId).ifPresent(remaining -> {
|
||||
row.put("free", 0);
|
||||
row.put("credentialId", credentialId);
|
||||
row.put("quarantinedForSeconds", remaining);
|
||||
});
|
||||
}
|
||||
return row;
|
||||
}
|
||||
|
||||
/**
|
||||
* One lead's row: its address, its name, and whether it can be reached right now.
|
||||
*
|
||||
* <p>{@code status} is herdr's live view, and {@code unknown} when herdr is not tracking that
|
||||
* pane as an agent — the honest answer, and the one that matters: a lead whose pane herdr cannot
|
||||
* see is a lead a {@code bridge_send} cannot be typed into. It is reported rather than hidden,
|
||||
* because a peer that has gone unreachable is exactly what the sender needs to know.
|
||||
*/
|
||||
private static Map<String, Object> leadView(String terminal, String name, Agent live,
|
||||
String selfTerm) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("sessionId", terminal);
|
||||
m.put("name", name);
|
||||
m.put("status", live == null || live.status() == null
|
||||
? "unknown" : live.status().name().toLowerCase());
|
||||
if (terminal.equals(selfTerm)) {
|
||||
m.put("self", true);
|
||||
}
|
||||
return m;
|
||||
}
|
||||
|
||||
/** {@code bridge_stop}: tear a worker down by its pane id. */
|
||||
static McpSchema.CallToolResult stop(SessionManager sessions, String paneId) {
|
||||
if (isBlank(paneId)) {
|
||||
@@ -377,10 +919,14 @@ public final class BridgeMcp {
|
||||
}
|
||||
|
||||
/** CB-301 projection from the authoritative session registry. */
|
||||
private static Map<String, Object> workerView(WorkerSession s) {
|
||||
private static Map<String, Object> memberView(MemberSession s) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("sessionId", s.terminalId());
|
||||
m.put("paneId", s.paneId());
|
||||
// Echo the role back so the caller can see what it actually got, not what it meant to ask
|
||||
// for — a spawn that silently fell back to dev is otherwise invisible.
|
||||
m.put("role", s.role() == null ? "dev" : s.role().wireName());
|
||||
m.put("profile", s.profile());
|
||||
m.put("status", s.state().name().toLowerCase());
|
||||
if (s.worktree() != null) {
|
||||
m.put("worktree", s.worktree());
|
||||
@@ -435,37 +981,80 @@ public final class BridgeMcp {
|
||||
private static McpSchema.Tool pollTool() {
|
||||
return tool("bridge_poll",
|
||||
"Check an async delegation (a bridge_send with wait:false) by its ticket: "
|
||||
+ "pending, done (with the worker's reply), or failed.",
|
||||
+ "pending, done (with the worker's reply), or failed. When target (a worker "
|
||||
+ "session id) is present instead of ticket, drain that worker's inbox of "
|
||||
+ "replies delivered when no send was open.",
|
||||
objectSchema(Map.of(
|
||||
"ticket", stringProp("The ticket returned by bridge_send wait:false")),
|
||||
List.of("ticket")));
|
||||
"ticket", stringProp("The ticket returned by bridge_send wait:false"),
|
||||
"target", stringProp("Worker session id to drain pending replies from (optional)")),
|
||||
List.of()));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool ackTool() {
|
||||
return tool("bridge_ack",
|
||||
"Acknowledge (remove) a specific reply from a worker's inbox. Use when the primary "
|
||||
+ "has processed a reply and wants to confirm it, leaving other pending replies "
|
||||
+ "in the inbox for later drain.",
|
||||
objectSchema(Map.of(
|
||||
"target", stringProp("Worker session id whose inbox to ack from"),
|
||||
"msgId", stringProp("The message id to acknowledge")),
|
||||
List.of("target", "msgId")));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool spawnTool() {
|
||||
return tool("bridge_spawn",
|
||||
"Spawn a new off-subscription worker session. Pass a profile (from bridge_profiles) to "
|
||||
+ "pick the backend, or omit it for the default. The worker opens your current "
|
||||
+ "directory by default; pass cwd to pin a different one. Pass worktree:true (with "
|
||||
+ "ticket) or worktree:<ticket-slug> to provision an isolated git worktree. "
|
||||
+ "Returns the worker's sessionId (use with bridge_send) and paneId (use with bridge_stop).",
|
||||
"Spawn a new off-subscription member session. A member has two independent attributes: "
|
||||
+ "role (what it is for) and profile (which backend it runs on). Pass role to pick "
|
||||
+ "the contract — 'dev' implements a unit and opens its own PR, 'reviewer' reviews a "
|
||||
+ "diff it did not write, 'architect' refines a ticket before anyone builds it; omit "
|
||||
+ "it for 'dev'. Pass profile (from bridge_profiles) to pick the backend, or omit it "
|
||||
+ "for the default. The two are independent: a reviewer may run on the same profile "
|
||||
+ "as the dev it reviews. The member opens your current directory by default; pass "
|
||||
+ "cwd to pin a different one. Pass worktree:true (with ticket) or "
|
||||
+ "worktree:<ticket-slug> to provision an isolated git worktree. Pass resumeSessionId "
|
||||
+ "to relaunch onto a prior conversation instead of starting cold — this requires an "
|
||||
+ "explicit profile whose backend supports it (bridge_list shows agentSessionId for "
|
||||
+ "resumable members), and is refused otherwise rather than silently starting fresh. "
|
||||
+ "sessionName gives the member a display name in its own UI when the backend supports "
|
||||
+ "one. Returns the member's sessionId (use with bridge_send) and paneId (use with "
|
||||
+ "bridge_stop).",
|
||||
objectSchema(Map.of(
|
||||
"profile", stringProp("Worker profile to spawn (omit for the default profile)"),
|
||||
"cwd", stringProp("Working directory for the worker (omit to inherit yours)"),
|
||||
"role", stringProp("What the member is for: architect, dev or reviewer (default dev)"),
|
||||
"profile", stringProp("Which backend to run it on (omit for the default profile)"),
|
||||
"cwd", stringProp("Working directory for the member (omit to inherit yours)"),
|
||||
"worktree", Map.of("type", "string", "description", "'true' or a ticket slug — requests an isolated git worktree"),
|
||||
"ticket", stringProp("Ticket slug when worktree:true")),
|
||||
"ticket", stringProp("Ticket slug when worktree:true"),
|
||||
"sessionName", stringProp("Logical display name for the member's own session, when its backend supports one"),
|
||||
"resumeSessionId", stringProp("A prior member's agentSessionId (from bridge_list) to resume — requires an explicit profile that supports it")),
|
||||
List.of()));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool profilesTool() {
|
||||
return tool("bridge_profiles",
|
||||
"List the configured worker profiles (backends) and which one bridge_spawn uses by default.",
|
||||
"List the configured worker profiles (backends) and which one bridge_spawn uses by "
|
||||
+ "default. A 'quarantined' map is present when a backend-exhausted refusal put "
|
||||
+ "a profile's credential on cooldown — bridge_spawn onto it is refused until "
|
||||
+ "quarantinedForSeconds elapses; a profile sharing that credential is listed too.",
|
||||
objectSchema(Map.of(), List.of()));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool listTool() {
|
||||
return tool("bridge_list",
|
||||
"List the worker sessions the bridge tracks — each with its sessionId, paneId, profile, "
|
||||
+ "state, optional worktree/branch/owner, and live herdr status.",
|
||||
"List the whole fleet the bridge tracks, in two parts. 'leads' are your PEERS — other "
|
||||
+ "orchestrators, each with its sessionId (the address to bridge_send to), "
|
||||
+ "name, live status, and 'self': true on your own row; this is how you "
|
||||
+ "discover a peer lead without being told its address. 'members' are the "
|
||||
+ "sessions delegated to — each with sessionId, paneId, role (architect/dev/"
|
||||
+ "reviewer), profile (the backend it runs on), state, optional "
|
||||
+ "worktree/branch/owner/agentSessionId (the id to pass as bridge_spawn's "
|
||||
+ "resumeSessionId to relaunch onto that same conversation, when the backend "
|
||||
+ "supports it), and live herdr status. An empty 'members' "
|
||||
+ "means no members are spawned; it says nothing about peers. When capacity "
|
||||
+ "facts are configured, a 'capacity' row per profile also reports free: 0 for "
|
||||
+ "a quarantined profile's credential (see bridge_profiles), whatever its "
|
||||
+ "maxLoad/live — with credentialId and quarantinedForSeconds naming the "
|
||||
+ "quarantine, so 'free: 0, busy' can be told apart from 'free: 0, refusing "
|
||||
+ "for N seconds'.",
|
||||
objectSchema(Map.of(), List.of()));
|
||||
}
|
||||
|
||||
@@ -478,10 +1067,12 @@ public final class BridgeMcp {
|
||||
}
|
||||
|
||||
private static McpSchema.Tool replyTool() {
|
||||
// No session/target arg — the worker's identity is resolved from the connection.
|
||||
// No session/target arg — the caller's identity is resolved from the connection.
|
||||
return tool("bridge_reply",
|
||||
"Return your structured answer for the task you were delegated, "
|
||||
+ "resolving the caller's blocked bridge_send.",
|
||||
"Return your structured answer for a message you were sent, resolving the sender's "
|
||||
+ "blocked bridge_send. A worker MUST end every delegated turn with exactly "
|
||||
+ "one of these. A lead uses it only to answer another lead that messaged "
|
||||
+ "it — never to answer a worker, whose turn it is not.",
|
||||
objectSchema(Map.of(
|
||||
"content", stringProp("Your reply/answer")),
|
||||
List.of("content")));
|
||||
@@ -495,6 +1086,21 @@ public final class BridgeMcp {
|
||||
List.of("sessionId")));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool whoamiTool() {
|
||||
return tool("bridge_whoami",
|
||||
"Report who YOU are on the bridge — your role is resolved from your connection "
|
||||
+ "(unforgeable), never from anything you claim. Returns role 'primary' (you "
|
||||
+ "orchestrate: spawn/send/stop; reply ONLY to answer a peer lead that "
|
||||
+ "messaged you, never to answer a worker), 'architect' (you delegate turns "
|
||||
+ "and reply/ask as your own pane, but cannot spawn/stop/drain), or 'worker' "
|
||||
+ "(you were delegated to: you must end every turn with exactly one "
|
||||
+ "bridge_reply, and cannot spawn or send), plus 'leader'/'architect' naming "
|
||||
+ "which one you are, your own sessionId, and profile/worktree/branch when "
|
||||
+ "you are a worker. Call this first when following role-conditional "
|
||||
+ "instructions rather than guessing.",
|
||||
objectSchema(Map.of(), List.of()));
|
||||
}
|
||||
|
||||
// --- small helpers -------------------------------------------------------------------------
|
||||
|
||||
// The SDK 2.0.0 deprecates its own Tool builders without a stable replacement — isolate it here.
|
||||
|
||||
@@ -0,0 +1,117 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Optional;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
/**
|
||||
* Single-slot, thread-safe registry for the primary's herdr {@code terminal_id}.
|
||||
*
|
||||
* <p>Populated from the caller terminal of orchestration-side MCP tools
|
||||
* ({@code bridge_send}, {@code bridge_spawn}) — tools that only the primary calls.
|
||||
* A pinned terminal (from config) seeds the registry at construction and makes
|
||||
* subsequent {@link #record(String)} calls no-ops.
|
||||
*
|
||||
* <p>The push loop ({@code ReplyPushLoop}) uses {@link #isKnown()} to decide
|
||||
* whether active nudging is possible; an empty registry means the primary is
|
||||
* off-host or non-herdr and delivery falls back to pull.
|
||||
*/
|
||||
public final class PrimaryRegistry {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(PrimaryRegistry.class);
|
||||
|
||||
private final AtomicReference<String> terminal = new AtomicReference<>();
|
||||
private final boolean pinned;
|
||||
|
||||
/**
|
||||
* CB-532: worker terminal → the lead that delegated to it. The single slot above answers "who is
|
||||
* THE primary", a question with no correct answer once two leads orchestrate the same fleet:
|
||||
* whichever called {@code bridge_send} first captured every nudge, including nudges for the
|
||||
* other lead's delegations. This map answers the question that actually matters — "who is
|
||||
* waiting on THIS worker" — and is what lets {@code primary.terminal} be retired.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, String> leadByTarget = new ConcurrentHashMap<>();
|
||||
|
||||
/**
|
||||
* @param pinnedTerminal an optional pinned terminal from config ({@code null}/blank = unpinned)
|
||||
*/
|
||||
public PrimaryRegistry(String pinnedTerminal) {
|
||||
if (pinnedTerminal != null && !pinnedTerminal.isBlank()) {
|
||||
this.terminal.set(pinnedTerminal);
|
||||
this.pinned = true;
|
||||
log.info("primary terminal pinned: {}", pinnedTerminal);
|
||||
} else {
|
||||
this.pinned = false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Record a terminal_id. No-op when:
|
||||
* <ul>
|
||||
* <li>the registry is pinned (config override),
|
||||
* <li>{@code terminalId} is {@code null} or blank (non-herdr caller).
|
||||
* </ul>
|
||||
*/
|
||||
public void record(String terminalId) {
|
||||
if (pinned) return;
|
||||
if (terminalId == null || terminalId.isBlank()) return;
|
||||
String prev = terminal.getAndSet(terminalId);
|
||||
if (prev == null) {
|
||||
log.debug("primary terminal learned: {}", terminalId);
|
||||
} else if (!prev.equals(terminalId)) {
|
||||
log.debug("primary terminal changed: {} -> {}", prev, terminalId);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Record that {@code leadTerminal} owns the accepted delegation of worker {@code target} (CB-532).
|
||||
*
|
||||
* <p>Called from the {@code MessageService} accepted-delivery hook — only after a send has won
|
||||
* the session's send lock and queued delivery — where both halves are known (CB-548). It is
|
||||
* deliberately <em>not</em> called at {@code bridge_send} request time: a concurrent sender that
|
||||
* times out {@code BUSY} must not steal a live delegation's reply routing without ever owning
|
||||
* the turn. Last writer wins — if a second lead's later send is accepted, replies follow the
|
||||
* lead that most recently delegated to it, which is the one waiting.
|
||||
*/
|
||||
public void recordDelegation(String target, String leadTerminal) {
|
||||
if (target == null || target.isBlank() || leadTerminal == null || leadTerminal.isBlank()) {
|
||||
return;
|
||||
}
|
||||
leadByTarget.put(target, leadTerminal);
|
||||
}
|
||||
|
||||
/** Forget a worker's delegating lead — call on release, so a torn-down session leaks nothing. */
|
||||
public void forgetDelegation(String target) {
|
||||
if (target != null) {
|
||||
leadByTarget.remove(target);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Where a nudge about {@code target}'s reply should go: the lead that delegated to it, falling
|
||||
* back to the single known primary.
|
||||
*
|
||||
* <p>The fallback matters after a daemon restart, which loses the map while the durable inbox
|
||||
* keeps the reply. With one lead the fallback is unambiguous and correct. With several and no
|
||||
* recorded delegation there is no right answer, so this returns empty rather than guessing —
|
||||
* delivery degrades to pull, which is exactly what the durable inbox is for, instead of
|
||||
* interrupting the wrong lead with someone else's result.
|
||||
*/
|
||||
public Optional<String> nudgeTargetFor(String target) {
|
||||
String lead = target == null ? null : leadByTarget.get(target);
|
||||
return lead != null ? Optional.of(lead) : Optional.ofNullable(terminal.get());
|
||||
}
|
||||
|
||||
/** The known primary terminal, or empty if not yet learned (and not pinned). */
|
||||
public Optional<String> primaryTerminal() {
|
||||
return Optional.ofNullable(terminal.get());
|
||||
}
|
||||
|
||||
/** {@code true} once a terminal has been recorded (or was pinned at construction). */
|
||||
public boolean isKnown() {
|
||||
return terminal.get() != null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,427 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The {@link HerdrPeerLauncher} adapter for <strong>Claude Code</strong> — the safe path from a
|
||||
* delegation request to a running off-subscription Claude.
|
||||
*
|
||||
* <p>Everything transport-related (tab/pane placement, the CB-306 spawn-readiness gate, unique
|
||||
* naming, CB-117 orphan reap, teardown, listing, cwd resolution) lives in the base. This class
|
||||
* supplies only the two Claude-specific seams:
|
||||
* <ul>
|
||||
* <li>the {@code claude} name prefix (so reap matches {@code claude-*} panes, never another
|
||||
* adapter's), and</li>
|
||||
* <li>{@link #buildLaunch}, which encodes the subscription boundary: build the worker env with
|
||||
* {@code ANTHROPIC_BASE_URL}, assert that host is on the allowlist <em>before</em> touching
|
||||
* herdr, and mount the bridge MCP + reply charter as inline launch flags. A worker's base_url
|
||||
* lives in the env map handed to herdr and nowhere else; {@code bridged}'s own environment is
|
||||
* never mutated, and nothing is written to the worker's profile.</li>
|
||||
* </ul>
|
||||
*/
|
||||
public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
/** Label prefix for this adapter's herdr agent names (drives naming + orphan reap). */
|
||||
private static final String NAME_PREFIX = "claude";
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ClaudeCodeLauncher.class);
|
||||
|
||||
private final SubscriptionGuard guard;
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so
|
||||
* existing deployments and tests keep the legacy non-blocking spawn semantics.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env, 0,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(300));
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with spawn-ready gate enabled. The gate polls {@code agents.status()}
|
||||
* until the pane reports an injectable state or {@code spawnReadyTimeoutMs} elapses.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, spawnReadyPollMs, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor carrying the fleet-wide tab-label template (CB-557). The template comes
|
||||
* from {@code fleet.tabLabel}, which a profile cannot know because it names the member's
|
||||
* <em>role</em>; a profile may still override it with its own {@code tabLabel}.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
fleet);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor, plus the CB-596 {@code memberCredentials} policy supplier.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<BridgedConfig.Fleet> fleet,
|
||||
Supplier<BridgedConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
fleet, memberCredentials);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor. Every injectable collaborator is explicit so unit tests supply
|
||||
* fakes for the clock ({@code nowMillis}) and poll-loop wait ({@code sleeper}). The
|
||||
* {@code sleeper} is never called when the gate is disabled ({@code spawnReadyTimeoutMs == 0}).
|
||||
*
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param guard subscription-boundary guard (checked before spawning)
|
||||
* @param profiles configured worker profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (e.g. {@code () -> Thread.sleep(pollMs)}); it
|
||||
* already encodes the poll interval, so the 8th positional argument
|
||||
* (poll ms) is accepted for API symmetry but otherwise unused here
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the fleet-wide tab-label template (CB-557).
|
||||
*
|
||||
* @param fleet live fleet config, read once for each spawn
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the CB-596 {@code memberCredentials} policy supplier.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet,
|
||||
Supplier<BridgedConfig.MemberCredentials> memberCredentials) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus an injectable host-env-names source for the CB-596
|
||||
* criterion-4 gap detector. Test seam only — every production call site leaves this at the
|
||||
* default (the real {@code System.getenv()} key set) via the constructor above.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet,
|
||||
Supplier<BridgedConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<Set<String>> hostEnvNames) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials, hostEnvNames);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>The spawn sequence encodes the subscription boundary: assert the profile's base_url is on
|
||||
* the allowlist <em>before</em> any herdr call, then build the worker env with
|
||||
* {@code ANTHROPIC_*}, the parity-neutral git-forge grant, and the bridge MCP + reply charter
|
||||
* mounted as inline launch flags. When the request carries session identity (CB-547a) it is
|
||||
* applied here — see {@link #applySessionIdentity}.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg, LaunchSpec spec) {
|
||||
// CB-539: a profile may deliberately opt into the subscription (subscription: true) when no
|
||||
// off-subscription endpoint exists for it — e.g. `sonnet` on `ccs`. That profile gets no
|
||||
// ANTHROPIC_BASE_URL/AUTH_TOKEN (there is nothing to point them at) and the guard's base_url
|
||||
// requirement is skipped FOR IT ONLY. Every other profile keeps the hard boundary below.
|
||||
boolean onSubscription = cfg.isSubscription();
|
||||
String baseUrl = cfg.baseUrl();
|
||||
|
||||
if (onSubscription) {
|
||||
// NO SILENT CONTRADICTION: subscription:true + a baseUrl state opposite intents; refuse
|
||||
// loudly rather than pick a winner.
|
||||
if (baseUrl != null && !baseUrl.isBlank()) {
|
||||
throw new IllegalStateException("profile '" + cfg.profile()
|
||||
+ "' sets both subscription: true and a baseUrl ('" + baseUrl + "') — the two "
|
||||
+ "are contradictory: a subscription profile must not point at an endpoint. "
|
||||
+ "Drop baseUrl, or drop subscription: true.");
|
||||
}
|
||||
// Visible without anyone going looking for it: this worker bills the subscription.
|
||||
log.warn("spawning profile '{}' on the Claude subscription (subscription: true) — this "
|
||||
+ "worker WILL bill the operator's subscription", cfg.profile());
|
||||
} else {
|
||||
guard.assertWorker(baseUrl); // hard stop before we spawn anything
|
||||
}
|
||||
|
||||
Map<String, String> workerEnv = baseEnv(cfg);
|
||||
if (onSubscription) {
|
||||
// CB-542 belt-and-braces: on the subscription path no guard vets these two keys, and the
|
||||
// profile's env: is layered in by baseEnv — so strip any that rode in there. Config load
|
||||
// already rejects this (loudly, naming the profile); this makes the boundary hold even
|
||||
// for a profile built in code that never passed through that validation.
|
||||
workerEnv.remove("ANTHROPIC_BASE_URL");
|
||||
workerEnv.remove("ANTHROPIC_AUTH_TOKEN");
|
||||
} else {
|
||||
workerEnv.put("ANTHROPIC_BASE_URL", baseUrl);
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", env.apply(cfg.tokenEnv()));
|
||||
}
|
||||
putIfPresent(workerEnv, "ANTHROPIC_MODEL", cfg.model());
|
||||
putIfPresent(workerEnv, "CLAUDE_CONFIG_DIR", cfg.configDir());
|
||||
applyGitToken(workerEnv, cfg);
|
||||
|
||||
// CB-547a: Claude Code can MINT its own session id, so bridged chooses it — a fresh spawn
|
||||
// gets a UUID we pass as --session-id and return from agentSessionId(), so the resume
|
||||
// handle is known BEFORE the agent has written anything; a resume spawn adopts its prior
|
||||
// id via -r and passes no --session-id (the two conflict). Both are injected before the
|
||||
// model flag so --model keeps outranking the operator's own argv.
|
||||
// mutableArgv: argvWithBridge may hand back the profile's own (immutable) List.of when it
|
||||
// has neither MCP nor a charter — session flags must be added into a list we own.
|
||||
List<String> argv = mutableArgv(argvWithBridge(cfg, spec));
|
||||
String agentSessionId = applySessionIdentity(argv, spec.sessionName(), spec.resumeSessionId());
|
||||
return new Launch(workerEnv, argvWithModel(argv, cfg), agentSessionId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Add the Claude-specific session-identity flags to {@code argv} and return the peer's OWN
|
||||
* session id — the resume handle. A resume request passes the prior id via {@code -r} and
|
||||
* returns that id; a fresh named session mints a new UUID, passes it via {@code --session-id},
|
||||
* and returns the mint. The bridge's logical name rides along as {@code -n} when present. When
|
||||
* <em>no</em> identity is requested (sessionName and resumeSessionId both blank) this adds
|
||||
* nothing and returns {@code null}, keeping the legacy no-identity launch byte-identical.
|
||||
*/
|
||||
private static String applySessionIdentity(List<String> argv, String sessionName, String resumeSessionId) {
|
||||
boolean resuming = resumeSessionId != null && !resumeSessionId.isBlank();
|
||||
boolean named = sessionName != null && !sessionName.isBlank();
|
||||
if (!resuming && !named) {
|
||||
return null; // no identity requested — keep the legacy launch byte-identical
|
||||
}
|
||||
if (named) {
|
||||
argv.add("-n");
|
||||
argv.add(sessionName);
|
||||
}
|
||||
if (resuming) {
|
||||
argv.add("-r");
|
||||
argv.add(resumeSessionId);
|
||||
return resumeSessionId;
|
||||
}
|
||||
String minted = UUID.randomUUID().toString();
|
||||
argv.add("--session-id");
|
||||
argv.add(minted);
|
||||
return minted;
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv, plus an inline {@code --mcp-config} when {@code worker.mcpUrl} is set, the
|
||||
* CB-617 charter flags, and {@code --agent <role>} when the role has an agent-definition file
|
||||
* under the worker's cwd. Neither touches the profile's config; all are pure command-line flags.
|
||||
* This inline-flag mount is Claude Code specific — other adapters mount MCP and instructions
|
||||
* their own way.
|
||||
*
|
||||
* <p>CB-617: the role charter is operator-authored and often multi-line, so it can never be a
|
||||
* single inline argv element — herdr refuses to shell-encode a multi-line argument
|
||||
* ({@code invalid_agent_argument}). It is written to a temp file instead and mounted with
|
||||
* {@code --append-system-prompt-file}, which this host confirms Claude Code accepts for a
|
||||
* multi-line file.
|
||||
*
|
||||
* <p>CB-618: Claude Code refuses to start when BOTH {@code --append-system-prompt} and
|
||||
* {@code --append-system-prompt-file} are on the command line ("Cannot use both ... Please use
|
||||
* only one"), so the two charters can never travel on separate flags. When both are present they
|
||||
* are concatenated into the one file, role charter first and reply charter last — last is where
|
||||
* the reply rule must sit, because it is the rule that must survive. When only the reply charter
|
||||
* is present it keeps its proven inline {@code --append-system-prompt} delivery, which is also
|
||||
* the only form that reaches a member with no repo checkout.
|
||||
*/
|
||||
private List<String> argvWithBridge(BridgedConfig.Profile cfg, LaunchSpec spec) {
|
||||
String roleCharter = nonBlank(spec.roleCharter());
|
||||
String replyCharter = nonBlank(spec.replyCharter());
|
||||
Path agentFile = agentDefinitionFile(spec.cwd(), spec.role(), ".claude", "agents");
|
||||
if (!cfg.hasMcp() && roleCharter == null && replyCharter == null && agentFile == null) {
|
||||
return cfg.argv();
|
||||
}
|
||||
List<String> argv = mutableArgv(cfg.argv());
|
||||
if (cfg.hasMcp()) {
|
||||
String mcpJson = "{\"mcpServers\":{\"bridge\":{\"type\":\"http\",\"url\":\""
|
||||
+ cfg.mcpUrl() + "\"}}}";
|
||||
argv.add("--mcp-config");
|
||||
argv.add(mcpJson);
|
||||
}
|
||||
if (roleCharter != null) {
|
||||
String combined = replyCharter == null ? roleCharter : roleCharter + "\n\n" + replyCharter;
|
||||
argv.add("--append-system-prompt-file");
|
||||
argv.add(writeCharterFile(combined).toString());
|
||||
} else if (replyCharter != null) {
|
||||
argv.add("--append-system-prompt");
|
||||
argv.add(replyCharter);
|
||||
}
|
||||
if (agentFile != null) {
|
||||
argv.add("--agent");
|
||||
argv.add(spec.role().wireName());
|
||||
}
|
||||
return argv;
|
||||
}
|
||||
|
||||
/** {@code s}, or {@code null} when {@code s} is null/blank — the charter-presence test used above. */
|
||||
private static String nonBlank(String s) {
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
}
|
||||
|
||||
/**
|
||||
* Write the role charter to a fresh temp file so it can be mounted with
|
||||
* {@code --append-system-prompt-file} instead of riding inline in argv (CB-617). Best-effort
|
||||
* cleaned via {@code deleteOnExit} — the same disposable-worker-config cleanup
|
||||
* {@link OpenCodeLauncher#writeConfig} already uses for its charter file, since the process that
|
||||
* reads this file (the spawned peer) outlives this JVM call and there is no spawn-scoped teardown
|
||||
* hook to delete it synchronously.
|
||||
*/
|
||||
private static Path writeCharterFile(String charterText) {
|
||||
try {
|
||||
Path file = Files.createTempFile("bridged-role-charter-", ".md");
|
||||
Files.writeString(file, charterText);
|
||||
file.toFile().deleteOnExit();
|
||||
return file;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot write role charter temp file", e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Pin the model on the command line as well as in {@code ANTHROPIC_MODEL} (CB-533).
|
||||
*
|
||||
* <p>The env var alone is not a reliable pin for this adapter, because the argv is usually a
|
||||
* launcher rather than {@code claude} itself — {@code ["ccs", "<profile>"]} — and {@code ccs}
|
||||
* exports its profile's own model family ({@code ANTHROPIC_MODEL}, {@code DEFAULT_OPUS/SONNET/
|
||||
* HAIKU}, {@code CLAUDE_CODE_SUBAGENT_MODEL}) over whatever it inherited. A worker profile that
|
||||
* set {@code model:} therefore got silently overruled by its own launcher. Claude Code's
|
||||
* {@code --model} flag outranks the environment, and {@code ccs <profile> [claude-args...]}
|
||||
* passes trailing arguments through, so the flag survives the wrapper.
|
||||
*
|
||||
* <p>Appended last so it also outranks anything in the operator's own {@code argv}. Profiles
|
||||
* that deliberately leave {@code model:} unset (letting {@code ccs} own model selection, as
|
||||
* {@code gx10} does) are untouched — this adds nothing when there is nothing to add. This is
|
||||
* the {@code kind: claude} counterpart of the opencode adapter's {@code -m provider/model}.
|
||||
*/
|
||||
private static List<String> argvWithModel(List<String> argv, BridgedConfig.Profile cfg) {
|
||||
if (cfg.model() == null || cfg.model().isBlank()) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withModel = mutableArgv(argv);
|
||||
withModel.add("--model");
|
||||
withModel.add(cfg.model());
|
||||
return withModel;
|
||||
}
|
||||
|
||||
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
|
||||
|
||||
/** Spawn a worker for the default profile in the resolved default cwd. */
|
||||
public Agent spawn() {
|
||||
return spawnInternal(null, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile (null → default) in the resolved default cwd. */
|
||||
public Agent spawn(String profileName) {
|
||||
return spawnInternal(profileName, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile with an explicit requested/caller cwd (CB-112). */
|
||||
public Agent spawn(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd);
|
||||
}
|
||||
|
||||
// --- capabilities --------------------------------------------------------------------------
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
Set<Capability> caps = EnumSet.of(Capability.MID_TURN_ASK, Capability.WORKTREE,
|
||||
Capability.CONTEXT_RESET, Capability.ORPHAN_REAP,
|
||||
Capability.SESSION_NAME, Capability.SESSION_RESUME);
|
||||
if (hasGitTokenProfile()) {
|
||||
caps.add(Capability.SELF_PR);
|
||||
}
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
String target = agentTarget(id);
|
||||
if (target == null) {
|
||||
return false;
|
||||
}
|
||||
// This deliberately bypasses Injector: /clear is housekeeping, not a delegated turn.
|
||||
agents().send(target, "/clear");
|
||||
return true;
|
||||
}
|
||||
|
||||
/** Whether any configured profile opts into a git-forge token (required for {@link Capability#SELF_PR}). */
|
||||
private boolean hasGitTokenProfile() {
|
||||
return profileConfigs().stream().anyMatch(BridgedConfig.Profile::hasGitToken);
|
||||
}
|
||||
|
||||
// --- CB-117 reap predicate (Claude prefix), kept for direct unit testing -------------------
|
||||
|
||||
/**
|
||||
* Whether {@code name} is a Claude Code bridge worker started by a <em>different</em> process
|
||||
* than {@code currentNonce}. A thin {@code claude}-prefix binding of
|
||||
* {@link HerdrPeerLauncher#isForeignWorker(String, String, String)}.
|
||||
*/
|
||||
static boolean isForeignWorker(String name, String currentNonce) {
|
||||
return HerdrPeerLauncher.isForeignWorker(NAME_PREFIX, name, currentNonce);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,510 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import dev.ltms.bridged.placement.BackendQuarantine;
|
||||
import dev.ltms.bridged.placement.PlacementCandidate;
|
||||
import dev.ltms.bridged.placement.PlacementContext;
|
||||
import dev.ltms.bridged.placement.PlacementException;
|
||||
import dev.ltms.bridged.placement.PlacementPolicies;
|
||||
import dev.ltms.bridged.placement.PlacementPolicy;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.EnumSet;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* The {@link PeerLauncher} the core actually holds when more than one adapter is configured — a thin
|
||||
* router in front of one {@link HerdrPeerLauncher} per peer {@code kind} (Claude Code, opencode, …).
|
||||
* It owns no transport of its own; it dispatches each SPI call to the delegate that owns the profile
|
||||
* involved, and fans the fleet-wide queries (list/reap/caps/profiles) across all delegates.
|
||||
*
|
||||
* <p>Routing rules:
|
||||
* <ul>
|
||||
* <li><strong>By profile</strong> — {@link #spawn}, {@link #effectiveCwd}, {@link #parityOverlay}
|
||||
* resolve the profile (a null/blank name → the global {@link #defaultProfile}) and delegate to
|
||||
* the single adapter that declares it. Profiles partition cleanly across adapters: the
|
||||
* constructor rejects a name claimed by two.</li>
|
||||
* <li><strong>By pane id</strong> — {@link #stop} routes to the adapter that spawned that pane
|
||||
* (recorded at spawn time). A pane the composite never spawned (only real for a caller that
|
||||
* hand-rolls an id) falls back to the first delegate; teardown is pane-id addressed and
|
||||
* tab cleanup is single-occupant guarded, so it is safe either way.</li>
|
||||
* <li><strong>Fleet-wide</strong> — {@link #reapOrphanWorkers} and {@link #capabilities} fan out
|
||||
* and combine. {@link #list} is deduplicated by pane id because every herdr-backed delegate
|
||||
* shares one herdr connection and so reports the same global agent set.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>CB-518: an unqualified spawn is routed through a {@link PlacementPolicy}. The default
|
||||
* {@code fixed} policy reproduces the historical default-profile behaviour; {@code weighted} uses
|
||||
* smooth weighted round-robin with {@code maxLoad} gating. If a chosen profile fails with
|
||||
* {@link PeerUnreachableException}, the composite advances to the next available candidate and
|
||||
* retries, bounded by the number of candidates.
|
||||
*/
|
||||
public final class CompositePeerLauncher implements PeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(CompositePeerLauncher.class);
|
||||
|
||||
private final List<HerdrPeerLauncher> delegates;
|
||||
private final Map<String, HerdrPeerLauncher> byProfile;
|
||||
private final String defaultProfile;
|
||||
|
||||
/** paneId → the delegate that spawned it, so {@link #stop} tears down through the right adapter. */
|
||||
private final Map<String, HerdrPeerLauncher> spawnedBy = new ConcurrentHashMap<>();
|
||||
|
||||
private final Function<String, Integer> liveCount;
|
||||
|
||||
/**
|
||||
* CB-559: the placement inputs are read <em>per spawn</em>, not captured at construction, so a
|
||||
* config reload changes where the next member lands without a restart. These are the hot keys —
|
||||
* role pools, an existing profile's weight/maxLoad, and the placement policy. What cannot change
|
||||
* this way is the set of adapters ({@link #byProfile}), because a new backend needs a launcher
|
||||
* and launchers are built once; {@code ConfigRef} classifies that as deferred and says so.
|
||||
*/
|
||||
private final Supplier<Map<String, BridgedConfig.Profile>> profileConfigs;
|
||||
private final Supplier<PlacementPolicy> placementPolicy;
|
||||
|
||||
/** CB-578 stage B: credential cooldown, checked before an explicit spawn and filtered into placement. */
|
||||
private final BackendQuarantine quarantine;
|
||||
|
||||
/**
|
||||
* CB-557: the role pools an unqualified spawn draws its candidates from. A supplier that yields
|
||||
* {@code null}, and an empty pool for a role, both fall back to every configured profile — the
|
||||
* pre-CB-557 behaviour.
|
||||
*/
|
||||
private final Supplier<BridgedConfig.Fleet> fleet;
|
||||
|
||||
/**
|
||||
* Backward-compatible constructor: fixed placement, no live-counting. Use this for tests and
|
||||
* simple wiring; it preserves the pre-CB-518 behaviour exactly.
|
||||
*
|
||||
* @param delegates one adapter per configured peer kind; must be non-empty and declare
|
||||
* disjoint profile-name sets
|
||||
* @param defaultProfile the profile a no-argument spawn resolves to (may be null)
|
||||
* @throws IllegalArgumentException if {@code delegates} is empty or two adapters claim one profile
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates, String defaultProfile) {
|
||||
this(delegates, defaultProfile, Map.of(), PlacementPolicies.fixed(), _ -> 0);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with a placement policy and live-worker counter. Quarantine (CB-578
|
||||
* stage B) is off for this constructor — {@link BackendQuarantine#none()} — since it predates
|
||||
* the feature and existing callers of this exact overload never exercised it; use the 7-arg
|
||||
* overload below to wire a real {@link BackendQuarantine}.
|
||||
*
|
||||
* @param delegates one adapter per configured peer kind; must be non-empty and declare
|
||||
* disjoint profile-name sets
|
||||
* @param defaultProfile the profile a no-argument spawn resolves to under {@code fixed} policy
|
||||
* @param profileConfigs all configured worker profiles (used for candidate weights/caps)
|
||||
* @param placementPolicy which policy governs unqualified spawns
|
||||
* @param liveCount live worker count per profile (must never return {@code null})
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, BridgedConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, null, BackendQuarantine.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with role pools (CB-557). An unqualified spawn draws its candidates from
|
||||
* {@code fleet.<role>} instead of from every configured profile, so a reviewer is placed on a
|
||||
* reviewer backend and never on, say, the architect-only one. Quarantine is off for this
|
||||
* constructor too, for the same reason as the 5-arg overload above.
|
||||
*
|
||||
* @param fleet the configured role pools; {@code null} ⇒ every profile is a candidate for every
|
||||
* role, which is the pre-CB-557 behaviour
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, BridgedConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
BridgedConfig.Fleet fleet) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet, BackendQuarantine.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with role pools and quarantine (CB-578 stage B). The full-featured
|
||||
* non-reloading form; {@link #CompositePeerLauncher(List, String, Supplier, Function, BackendQuarantine)}
|
||||
* is what {@code Bridged.main} actually wires up.
|
||||
*
|
||||
* @param quarantine required — pass {@link BackendQuarantine#none()} for a caller that does not
|
||||
* want the feature, never a defaulting overload (CB-578 stage B's own rule).
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, BridgedConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
BridgedConfig.Fleet fleet,
|
||||
BackendQuarantine quarantine) {
|
||||
// LinkedHashMap, not Map.copyOf: candidates() promises definition order and the weighted
|
||||
// policy breaks exact-weight ties on it, so a salted iteration order would make placement
|
||||
// differ from one JVM run to the next.
|
||||
this(delegates, defaultProfile,
|
||||
constant(Collections.unmodifiableMap(new LinkedHashMap<>(profileConfigs))),
|
||||
constant(placementPolicy), liveCount, constant(fleet), quarantine);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor that re-reads its placement inputs per spawn (CB-559), so a config
|
||||
* reload retargets the next member without a restart.
|
||||
*
|
||||
* @param config the live configuration — read at every spawn, never captured
|
||||
* @param quarantine required — CB-578 stage B; pass {@link BackendQuarantine#none()} to opt out
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Supplier<BridgedConfig> config,
|
||||
Function<String, Integer> liveCount,
|
||||
BackendQuarantine quarantine) {
|
||||
this(delegates, defaultProfile,
|
||||
() -> config.get().profiles(),
|
||||
() -> PlacementPolicies.fromName(config.get().placement()),
|
||||
liveCount,
|
||||
() -> config.get().fleet(),
|
||||
quarantine);
|
||||
}
|
||||
|
||||
/** The all-suppliers form every other constructor funnels into. */
|
||||
private CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Supplier<Map<String, BridgedConfig.Profile>> profileConfigs,
|
||||
Supplier<PlacementPolicy> placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
Supplier<BridgedConfig.Fleet> fleet,
|
||||
BackendQuarantine quarantine) {
|
||||
this.fleet = fleet;
|
||||
this.quarantine = Objects.requireNonNull(quarantine, "quarantine");
|
||||
if (delegates.isEmpty()) {
|
||||
throw new IllegalArgumentException("at least one peer adapter must be configured");
|
||||
}
|
||||
this.delegates = List.copyOf(delegates);
|
||||
this.defaultProfile = defaultProfile;
|
||||
this.profileConfigs = profileConfigs;
|
||||
this.placementPolicy = placementPolicy;
|
||||
this.liveCount = liveCount;
|
||||
Map<String, HerdrPeerLauncher> index = new LinkedHashMap<>();
|
||||
for (HerdrPeerLauncher d : this.delegates) {
|
||||
for (String profile : d.profiles()) {
|
||||
HerdrPeerLauncher prev = index.putIfAbsent(profile, d);
|
||||
if (prev != null) {
|
||||
throw new IllegalArgumentException(
|
||||
"worker profile '" + profile + "' is claimed by two peer adapters");
|
||||
}
|
||||
}
|
||||
}
|
||||
// Order-preserving for the same reason, and because profiles() is user-visible (bridge_profiles).
|
||||
this.byProfile = Collections.unmodifiableMap(index);
|
||||
}
|
||||
|
||||
/** A supplier of a value fixed at construction — how the non-reloading constructors funnel in. */
|
||||
private static <T> Supplier<T> constant(T value) {
|
||||
return () -> value;
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-configured profiles, never null.
|
||||
*
|
||||
* <p>Read fresh on every call so a reload is visible; a caller that needs two consistent reads
|
||||
* takes one local, as {@link #poolFor} does.
|
||||
*/
|
||||
private Map<String, BridgedConfig.Profile> profiles0() {
|
||||
Map<String, BridgedConfig.Profile> m = profileConfigs.get();
|
||||
return m == null ? Map.of() : m;
|
||||
}
|
||||
|
||||
/** The adapter owning {@code profileName} (null/blank → the default). Throws on an unknown profile. */
|
||||
private HerdrPeerLauncher route(String profileName) {
|
||||
String resolved = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (resolved == null) {
|
||||
// No profile and no default configured — hand to the first delegate so it raises the
|
||||
// same "no default" error it would on its own; keeps the SPI contract single-sourced.
|
||||
return delegates.getFirst();
|
||||
}
|
||||
HerdrPeerLauncher d = byProfile.get(resolved);
|
||||
if (d == null) {
|
||||
throw new IllegalArgumentException("unknown worker profile: " + resolved);
|
||||
}
|
||||
return d;
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
String requestedProfile = req.profileName();
|
||||
if (requestedProfile != null && !requestedProfile.isBlank()) {
|
||||
// An explicit profile bypasses the placement policy, but not the capacity cap: maxLoad
|
||||
// is documented as an unconditional limit on this profile (BridgedConfig.Profile), and
|
||||
// the charter makes explicit-profile spawns the normal path — so skipping the check
|
||||
// here would leave the cap dead config in real operation.
|
||||
HerdrPeerLauncher d = route(requestedProfile);
|
||||
enforceNotQuarantined(requestedProfile);
|
||||
enforceMaxLoad(requestedProfile);
|
||||
PeerHandle handle = d.spawn(req);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
}
|
||||
|
||||
// CB-557: an unqualified spawn is placed inside the pool of the role it asked for, not across
|
||||
// the whole profile list. An EXPLICIT profile (above) is left alone on purpose — it is the
|
||||
// operator overriding, and refusing it would break `bridge_spawn{profile:"opus"}`, which
|
||||
// carries no role and so would be judged against the dev pool it was never meant for.
|
||||
List<PlacementCandidate> candidates = candidates(req.role());
|
||||
String roleDefault = defaultProfileFor(req.role());
|
||||
Set<String> unreachable = new HashSet<>();
|
||||
// CB-578 stage B: computed once up front — a quarantine's expiry cannot pass within one spawn
|
||||
// call, so re-deriving it per retry would only cost work, never change the answer.
|
||||
Set<String> quarantined = quarantinedProfiles(candidates);
|
||||
PlacementContext ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable, quarantined);
|
||||
|
||||
int maxAttempts = candidates.isEmpty() ? 1 : candidates.size();
|
||||
for (int attempt = 0; attempt < maxAttempts; attempt++) {
|
||||
// Deliberately uncaught: when no candidate is left (all at cap, or all unreachable) the
|
||||
// policy already throws a clear message. Catching it to rethrow a generic
|
||||
// PeerUnreachableException would replace a precise diagnosis with a vague one.
|
||||
PlacementCandidate chosen = placementPolicy.get().select(ctx);
|
||||
|
||||
HerdrPeerLauncher d = byProfile.get(chosen.profile());
|
||||
if (d == null) {
|
||||
// A configured profile with no adapter is a wiring bug; fail fast.
|
||||
unreachable.add(chosen.profile());
|
||||
continue;
|
||||
}
|
||||
|
||||
// CB-547a: route the chosen profile but keep the caller's session identity — dropping it
|
||||
// here would silently sever the resume handle on every policy-routed spawn. CB-557: the
|
||||
// role rides along for the same reason, or a routed spawn would be labelled as a dev.
|
||||
SpawnRequest routedReq = new SpawnRequest(chosen.profile(), req.requestedCwd(), req.callerCwd(),
|
||||
req.sessionName(), req.resumeSessionId(), req.role());
|
||||
try {
|
||||
PeerHandle handle = d.spawn(routedReq);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
} catch (PeerUnreachableException e) {
|
||||
log.warn("spawn on profile {} unreachable, will retry next candidate if any: {}",
|
||||
chosen.profile(), e.getMessage());
|
||||
unreachable.add(chosen.profile());
|
||||
// Update the context for the next selection so the policy excludes this profile.
|
||||
ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable, quarantined);
|
||||
}
|
||||
}
|
||||
|
||||
throw new PeerUnreachableException(
|
||||
"no reachable worker profile available after trying " + unreachable.size()
|
||||
+ " candidate(s): " + String.join(", ", unreachable));
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse an explicit-profile spawn when the profile is at its {@code maxLoad} cap.
|
||||
*
|
||||
* <p>maxLoad is a documented, unconditional capacity limit (see {@code BridgedConfig.Profile#maxLoad}),
|
||||
* and the charter makes explicit-profile spawns the normal path — so enforcing it only in placement
|
||||
* ({@code PlacementPolicyUtil}, package-private, hence not linked) would leave the cap dead config
|
||||
* on every call that names a profile. Same rule as placement: {@code live >= cap} is at capacity.
|
||||
*
|
||||
* <p>Deliberately no fallback to another profile: the caller named {@code profile} for a cost/model
|
||||
* reason, and silently re-routing a paid-tier (subscription) request elsewhere is worse than
|
||||
* refusing it. A caller that wants placement should omit the profile and let the policy pick.
|
||||
*
|
||||
* <p>Known TOCTOU limitation — documented, not fixed. {@link #liveCount} is read outside any lock and
|
||||
* {@code SessionManager} registers a session only after {@code launcher.spawn} returns, so two
|
||||
* genuinely concurrent spawns can both pass this check. The race already exists on the placement
|
||||
* path. Closing it needs slot reservation in the registry; serializing spawn here would block on
|
||||
* the readiness gate and is a far worse trade.
|
||||
*
|
||||
* @param profile the profile the caller explicitly named
|
||||
* @throws PlacementException when the profile is at capacity
|
||||
*/
|
||||
/**
|
||||
* Refuse an explicit-profile spawn whose credential is quarantined (CB-578 stage B): a prior
|
||||
* {@code BACKEND_EXHAUSTED} classification on this profile, or on another profile sharing its
|
||||
* {@code credentialId}, is still on cooldown.
|
||||
*
|
||||
* <p>Checked before {@link #enforceMaxLoad}, and for the same reason that check exists: an
|
||||
* explicit profile bypasses the placement policy's filtering entirely, so without this the cap
|
||||
* (there, quarantine here) would be dead config on the very path the charter calls normal.
|
||||
* Deliberately no fallback to another profile, matching {@link #enforceMaxLoad}'s own reasoning —
|
||||
* the caller named this profile for a cost/model reason.
|
||||
*
|
||||
* @throws PlacementException naming the profile, its credential, and the remaining cooldown
|
||||
*/
|
||||
private void enforceNotQuarantined(String profile) {
|
||||
String credentialId = credentialIdFor(profile);
|
||||
quarantine.remainingSeconds(credentialId).ifPresent(remaining -> {
|
||||
throw new PlacementException("worker profile '" + profile + "' is quarantined "
|
||||
+ "(credential '" + credentialId + "' exhausted; ~" + remaining
|
||||
+ "s remaining) — refusing spawn");
|
||||
});
|
||||
}
|
||||
|
||||
/** {@code profile}'s credential group (CB-578 stage B), or the profile's own name if unconfigured. */
|
||||
private String credentialIdFor(String profile) {
|
||||
BridgedConfig.Profile cfg = profiles0().get(profile);
|
||||
return cfg == null ? profile : cfg.effectiveCredentialId();
|
||||
}
|
||||
|
||||
/** The subset of {@code candidates} whose credential is currently quarantined (CB-578 stage B). */
|
||||
private Set<String> quarantinedProfiles(List<PlacementCandidate> candidates) {
|
||||
return candidates.stream()
|
||||
.map(PlacementCandidate::profile)
|
||||
.filter(p -> quarantine.isQuarantined(credentialIdFor(p)))
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
private void enforceMaxLoad(String profile) {
|
||||
// Absent config, or a config whose maxLoad normalized to null (ABSENT ⇒ unlimited at load),
|
||||
// means no cap — never cap what wasn't configured. Note "non-positive ⇒ unlimited" was true
|
||||
// until CB-585: an explicit `maxLoad: 0` now survives as 0 and is a real cap of zero, so the
|
||||
// check below refuses every spawn on that profile, and a negative value is refused at config
|
||||
// load rather than normalized away.
|
||||
BridgedConfig.Profile cfg = profiles0().get(profile);
|
||||
Integer cap = (cfg == null) ? null : cfg.maxLoad();
|
||||
if (cap == null) {
|
||||
return;
|
||||
}
|
||||
int live = liveCount.apply(profile);
|
||||
if (live >= cap) {
|
||||
throw new PlacementException("worker profile '" + profile + "' is at maxLoad: " + live
|
||||
+ " live >= " + cap + " cap; refusing spawn — no fallback to another profile");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The profile names {@code role} may be placed on, in definition order.
|
||||
*
|
||||
* <p>An empty or absent pool means "unconstrained", not "nothing allowed": a config that declares
|
||||
* no pool for a role must keep spawning, so it falls back to every configured profile. Names in a
|
||||
* pool that no adapter declares are dropped here rather than thrown — config load already rejects
|
||||
* a pool entry with no profile, so a survivor is a profile this particular composite does not own.
|
||||
*/
|
||||
private List<String> poolFor(MemberRole role) {
|
||||
Map<String, BridgedConfig.Profile> configured = profiles0();
|
||||
BridgedConfig.Fleet f = fleet.get();
|
||||
List<String> pool = (f == null) ? List.of() : f.profilesFor(role);
|
||||
List<String> known = pool.stream().filter(configured::containsKey).toList();
|
||||
return known.isEmpty() ? List.copyOf(configured.keySet()) : known;
|
||||
}
|
||||
|
||||
/** The profile an unqualified spawn for {@code role} falls back to under {@code fixed} placement. */
|
||||
private String defaultProfileFor(MemberRole role) {
|
||||
List<String> pool = poolFor(role);
|
||||
return pool.isEmpty() ? defaultProfile : pool.getFirst();
|
||||
}
|
||||
|
||||
/** Build the candidate list from {@code role}'s pool, in definition order. */
|
||||
private List<PlacementCandidate> candidates(MemberRole role) {
|
||||
List<PlacementCandidate> out = new ArrayList<>();
|
||||
for (String name : poolFor(role)) {
|
||||
BridgedConfig.Profile w = profiles0().get(name);
|
||||
if (w != null) {
|
||||
out.add(new PlacementCandidate(name, null, w.weight(), w.maxLoad()));
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
return route(req.profileName()).effectiveCwd(req);
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
return route(profileName).parityOverlay(profileName);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) {
|
||||
HerdrPeerLauncher d = spawnedBy.remove(id);
|
||||
if (d == null) {
|
||||
log.debug("stop({}) — no recorded owner, routing to the first adapter (pane-addressed)", id);
|
||||
d = delegates.getFirst();
|
||||
}
|
||||
d.stop(id);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
HerdrPeerLauncher delegate = spawnedBy.get(id);
|
||||
if (delegate == null) {
|
||||
log.debug("clearContext({}) ignored — no recorded owning adapter", id);
|
||||
return false;
|
||||
}
|
||||
return delegate.clearContext(id);
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return byProfile.keySet();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return defaultProfile;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Routes to the specific delegate {@code profileName} resolves to, not the fleet-wide
|
||||
* union {@link #capabilities()} returns — the whole reason this method exists (CB-584): in a
|
||||
* mixed fleet, one adapter's capability must never be read as every profile's.
|
||||
*/
|
||||
@Override
|
||||
public Set<Capability> capabilitiesFor(String profileName) {
|
||||
return route(profileName).capabilities();
|
||||
}
|
||||
|
||||
/** Every herdr agent, deduplicated by pane id (all delegates share one herdr and list globally). */
|
||||
@Override
|
||||
public List<Agent> list() {
|
||||
Map<String, Agent> byPane = new LinkedHashMap<>();
|
||||
for (HerdrPeerLauncher d : delegates) {
|
||||
for (Agent a : d.list()) {
|
||||
if (a.paneId() != null) {
|
||||
byPane.putIfAbsent(a.paneId(), a);
|
||||
}
|
||||
}
|
||||
}
|
||||
return List.copyOf(byPane.values());
|
||||
}
|
||||
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
int reaped = 0;
|
||||
for (HerdrPeerLauncher d : delegates) {
|
||||
reaped += d.reapOrphanWorkers();
|
||||
}
|
||||
return reaped;
|
||||
}
|
||||
|
||||
/** The union of every adapter's capabilities — a capability any adapter offers, the fleet offers. */
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
EnumSet<Capability> caps = EnumSet.noneOf(Capability.class);
|
||||
for (HerdrPeerLauncher d : delegates) {
|
||||
caps.addAll(d.capabilities());
|
||||
}
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,512 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.fasterxml.jackson.databind.node.ObjectNode;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.CharterReceipt;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The {@link HerdrPeerLauncher} adapter for <strong>opencode</strong> — an open-source,
|
||||
* provider-agnostic terminal coding agent. Its whole reason for existing is to prove the
|
||||
* {@code PeerLauncher} SPI is genuinely provider-neutral: opencode shares none of Claude Code's
|
||||
* private launch seams, yet reuses every line of shared transport in the base (tab/pane placement,
|
||||
* the CB-306 readiness gate, unique naming + CB-117 reap, teardown, listing, cwd).
|
||||
*
|
||||
* <p>The divergences from {@link ClaudeCodeLauncher}, all confined to {@link #buildLaunch}:
|
||||
* <ul>
|
||||
* <li><strong>No subscription boundary.</strong> opencode carries no {@code ANTHROPIC_BASE_URL}
|
||||
* and there is no {@link dev.ltms.bridged.guard.SubscriptionGuard} — the guard is a
|
||||
* Claude-private concern, not part of the SPI. opencode reads the operator's own provider
|
||||
* credentials from its global {@code auth.json}; the bridge injects none.</li>
|
||||
* <li><strong>File-based MCP mount + instructions.</strong> opencode has no inline
|
||||
* {@code --mcp-config}/{@code --append-system-prompt}. Instead the bridge writes an ephemeral
|
||||
* {@code opencode.json} that declares the bridge as a {@code remote} MCP server and lists a
|
||||
* member-charter file under {@code instructions}, then points the worker at it with
|
||||
* {@code OPENCODE_CONFIG}. This is the one place the launcher touches disk — Claude never did.</li>
|
||||
* <li><strong>Model as a flag.</strong> the {@code provider/model} selector is passed as
|
||||
* {@code -m}, not an env var.</li>
|
||||
* <li><strong>{@code opencode} name prefix</strong> so reap matches {@code opencode-*} panes and
|
||||
* never another adapter's.</li>
|
||||
* </ul>
|
||||
*/
|
||||
public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
/** Label prefix for this adapter's herdr agent names (drives naming + orphan reap). */
|
||||
private static final String NAME_PREFIX = "opencode";
|
||||
|
||||
/** Writer for the generated {@code opencode.json}. */
|
||||
private static final ObjectMapper JSON = new ObjectMapper();
|
||||
|
||||
/** Root under which per-spawn opencode config dirs are created (injectable for tests). */
|
||||
private final Path configRoot;
|
||||
|
||||
/**
|
||||
* Session discovery against opencode's on-disk storage ({@link OpenCodeSessionDiscovery}) —
|
||||
* the one seam that knows opencode's private session-file layout. Its root is injectable for
|
||||
* tests so they never touch the operator's real {@code ~/.local/share/opencode}.
|
||||
*/
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so it
|
||||
* matches the legacy non-blocking spawn semantics. Config dirs are created under the JVM temp dir.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, 0,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(300),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot());
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with the spawn-ready gate enabled. Polls {@code agents.status()} until
|
||||
* the pane reports an injectable state or {@code spawnReadyTimeoutMs} elapses.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs) {
|
||||
this(agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, spawnReadyPollMs, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor carrying the fleet-wide tab-label template (CB-557). The template comes
|
||||
* from {@code fleet.tabLabel}, which a profile cannot know because it names the member's
|
||||
* <em>role</em>; a profile may still override it with its own {@code tabLabel}.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor, plus the CB-596 {@code memberCredentials} policy supplier.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<BridgedConfig.Fleet> fleet,
|
||||
Supplier<BridgedConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet, memberCredentials);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor. Every injectable collaborator is explicit so unit tests supply a
|
||||
* fake clock ({@code nowMillis}), poll-loop wait ({@code sleeper}), and a temp {@code configRoot}
|
||||
* they can inspect the generated {@code opencode.json}/charter under.
|
||||
*
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param profiles configured worker profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (encodes the poll interval; never called when the
|
||||
* gate is disabled)
|
||||
* @param configRoot existing directory under which per-spawn config dirs are created
|
||||
* @param discoveryRoot opencode's on-disk storage root to scan for session records
|
||||
* (injectable for tests; opencode's layout is matched at
|
||||
* {@link OpenCodeSessionDiscovery})
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
nowMillis, sleeper, configRoot, discoveryRoot, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the fleet-wide tab-label template (CB-557).
|
||||
*
|
||||
* @param fleet live fleet config, read once for each spawn
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the CB-596 {@code memberCredentials} policy supplier.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot,
|
||||
Supplier<BridgedConfig.Fleet> fleet,
|
||||
Supplier<BridgedConfig.MemberCredentials> memberCredentials) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
}
|
||||
|
||||
private static Path defaultConfigRoot() {
|
||||
return Path.of(System.getProperty("java.io.tmpdir"));
|
||||
}
|
||||
|
||||
/** The default opencode storage root: {@code ~/.local/share/opencode} (the XDG data dir). */
|
||||
private static Path defaultDiscoveryRoot() {
|
||||
return Path.of(System.getProperty("user.home"), ".local", "share", "opencode");
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Builds the opencode launch: no {@code ANTHROPIC_*} and no guard (opencode reads its own
|
||||
* provider credentials); when the profile mounts the bridge MCP or has a member charter,
|
||||
* generate an ephemeral {@code opencode.json} (remote MCP server + member-charter instructions)
|
||||
* and point the worker at it via {@code OPENCODE_CONFIG}; carry the parity-neutral git-forge
|
||||
* grant; and select the model with {@code -m}.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg, LaunchSpec spec) {
|
||||
Map<String, String> workerEnv = baseEnv(cfg);
|
||||
// A config file is needed for the bridge MCP mount, a member charter, or a pinned endpoint (CB-508).
|
||||
if (cfg.hasMcp() || spec.charter() != null || hasCustomProvider(cfg)) {
|
||||
workerEnv.put("OPENCODE_CONFIG", writeConfig(cfg, spec.charter()).toString());
|
||||
}
|
||||
applyGitToken(workerEnv, cfg);
|
||||
List<String> argv = argvWithResume(argvWithModel(argvWithAuto(cfg), cfg), spec.resumeSessionId());
|
||||
return new Launch(workerEnv, argvWithAgent(argv, spec));
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv plus, when the role has an agent-definition file under the worker's cwd,
|
||||
* opencode's {@code --agent <role>} flag (CB-617). A role with no such file gets nothing added —
|
||||
* the member must still spawn.
|
||||
*/
|
||||
private List<String> argvWithAgent(List<String> argv, LaunchSpec spec) {
|
||||
Path agentFile = agentDefinitionFile(spec.cwd(), spec.role(), ".opencode", "agent");
|
||||
if (agentFile == null) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withAgent = mutableArgv(argv);
|
||||
withAgent.add("--agent");
|
||||
withAgent.add(spec.role().wireName());
|
||||
return withAgent;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when this profile pins its own OpenAI-compatible endpoint (CB-508) rather than using
|
||||
* whatever provider opencode resolves by default.
|
||||
*
|
||||
* <p>Note this reuses {@code baseUrl}, the same field the Claude adapter injects as
|
||||
* {@code ANTHROPIC_BASE_URL} — but it does <em>not</em> go through {@code SubscriptionGuard}.
|
||||
* That asymmetry is deliberate and safe: the guard exists to stop a worker borrowing the
|
||||
* primary's Anthropic subscription, and an opencode process has no Anthropic credential path
|
||||
* at all. Pointing it at a local vLLM cannot leak the subscription.
|
||||
*/
|
||||
private static boolean hasCustomProvider(BridgedConfig.Profile cfg) {
|
||||
return cfg.baseUrl() != null && !cfg.baseUrl().isBlank();
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv plus the unconditional {@code --auto} flag, which auto-approves the
|
||||
* permissions opencode does not explicitly deny. It is unconditional, not a preference: a
|
||||
* spawned peer has no human at its pane — the bridge spawned it — so one that stops at an
|
||||
* approval prompt is a wedged agent, indistinguishable from a legitimate mid-turn wait and
|
||||
* unable to end its turn with {@code bridge_reply}. opencode's own help calls this
|
||||
* "dangerous!", but the blast radius here is already bounded by design: a worker runs in its
|
||||
* own git worktree on its own branch, is off-subscription, and cannot merge — the lead is the
|
||||
* gate.
|
||||
*/
|
||||
private List<String> argvWithAuto(BridgedConfig.Profile cfg) {
|
||||
List<String> argv = mutableArgv(cfg.argv());
|
||||
argv.add("--auto");
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv plus, on a resumed spawn, opencode's {@code -s <id>} flag to continue a prior
|
||||
* conversation by its session id. {@code -s, --session <id>} resumes an existing session; on a
|
||||
* fresh spawn (no resume target) no flag is added, letting opencode start a brand-new session.
|
||||
* The id comes from the base launch spec.
|
||||
*/
|
||||
private List<String> argvWithResume(List<String> argv, String id) {
|
||||
if (id == null || id.isBlank()) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withResume = mutableArgv(argv);
|
||||
withResume.add("-s");
|
||||
withResume.add(id);
|
||||
return withResume;
|
||||
}
|
||||
|
||||
/** The launch argv plus, when a model is configured, the opencode {@code -m provider/model} flag. */
|
||||
private List<String> argvWithModel(List<String> argv, BridgedConfig.Profile cfg) {
|
||||
if (cfg.model() != null && !cfg.model().isBlank()) {
|
||||
argv.add("-m");
|
||||
argv.add(cfg.model());
|
||||
}
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* Write an ephemeral {@code opencode.json} (and the member-charter file it references) into a
|
||||
* fresh per-spawn directory under {@link #configRoot}, and return the config file's path for
|
||||
* {@code OPENCODE_CONFIG}. The dir is unique per spawn so concurrent workers never race on it;
|
||||
* it is best-effort cleaned on JVM exit (worker config is disposable — regenerated every spawn).
|
||||
*/
|
||||
private Path writeConfig(BridgedConfig.Profile cfg, String charterText) {
|
||||
try {
|
||||
Path dir = Files.createTempDirectory(configRoot, "bridged-opencode-");
|
||||
dir.toFile().deleteOnExit();
|
||||
|
||||
ObjectNode root = JSON.createObjectNode();
|
||||
root.put("$schema", "https://opencode.ai/config.json");
|
||||
// CB-523, opencode side: a worker that runs out of context dies mid-turn, and its reply
|
||||
// — the entire point of the turn — is lost with it. Auto-compaction is therefore not an
|
||||
// operator preference for a bridged worker, it is a condition of the turn contract.
|
||||
//
|
||||
// Stated deliberately even though it is redundant today: OPENCODE_CONFIG is MERGED over
|
||||
// ~/.config/opencode/config.json rather than replacing it, so a worker already inherits
|
||||
// an `auto: true` set at home. We do not want that inheritance to be what the guarantee
|
||||
// rests on — the home file is outside this repo, differs per machine, and is not ours.
|
||||
//
|
||||
// Know the cost before removing it: this key WINS over the home config (verified — an
|
||||
// OPENCODE_CONFIG value overrides the home value, it does not defer to it), so an
|
||||
// operator who sets `compaction.auto: false` at home cannot turn it off for bridged
|
||||
// workers. That is the intended trade for peers we spawn and whose turns we must land;
|
||||
// if per-profile control is ever wanted, add a profile knob rather than dropping this.
|
||||
root.putObject("compaction").put("auto", true);
|
||||
|
||||
if (charterText != null) {
|
||||
Path charter = dir.resolve("member-charter.md");
|
||||
Files.writeString(charter, charterText);
|
||||
charter.toFile().deleteOnExit();
|
||||
|
||||
root.putArray("instructions").add(charter.toAbsolutePath().toString());
|
||||
}
|
||||
|
||||
if (cfg.hasMcp()) {
|
||||
ObjectNode bridge = root.putObject("mcp").putObject("bridge");
|
||||
bridge.put("type", "remote");
|
||||
bridge.put("url", cfg.mcpUrl());
|
||||
bridge.put("enabled", true);
|
||||
}
|
||||
if (hasCustomProvider(cfg)) {
|
||||
addCustomProvider(root, cfg);
|
||||
}
|
||||
|
||||
Path cfgFile = dir.resolve("opencode.json");
|
||||
// Built with Jackson rather than string concatenation: the provider block is nested and
|
||||
// carries operator-supplied values (URL, model id, api key), so escaping must be real.
|
||||
Files.writeString(cfgFile, JSON.writerWithDefaultPrettyPrinter().writeValueAsString(root));
|
||||
cfgFile.toFile().deleteOnExit();
|
||||
return cfgFile;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException(
|
||||
"cannot write opencode config for profile " + cfg.profile(), e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Declare a custom OpenAI-compatible provider so the worker talks to a pinned endpoint (a local
|
||||
* vLLM, say) instead of opencode's default gateway (CB-508).
|
||||
*
|
||||
* <p>The provider id comes from the {@code provider/model} selector in {@code model:}, so one
|
||||
* field drives both the declaration and the {@code -m} flag and they cannot drift apart.
|
||||
*/
|
||||
private void addCustomProvider(ObjectNode root, BridgedConfig.Profile cfg) {
|
||||
String[] parts = splitModelSelector(cfg);
|
||||
String providerId = parts[0];
|
||||
String modelId = parts[1];
|
||||
|
||||
ObjectNode provider = root.putObject("provider").putObject(providerId);
|
||||
provider.put("npm", "@ai-sdk/openai-compatible");
|
||||
provider.put("name", providerId + " (bridged)");
|
||||
|
||||
ObjectNode options = provider.putObject("options");
|
||||
options.put("baseURL", openAiBaseUrl(cfg.baseUrl()));
|
||||
// vLLM and friends usually ignore the key, but the AI SDK still requires a non-empty one.
|
||||
String token = resolveEnv(cfg.tokenEnv());
|
||||
options.put("apiKey", (token == null || token.isBlank()) ? "bridged-local-noauth" : token);
|
||||
|
||||
provider.putObject("models").putObject(modelId).put("name", modelId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Split {@code model:} into its {@code provider} and {@code model} halves. A pinned endpoint
|
||||
* needs both, so a bare model name is rejected loudly rather than silently falling back to the
|
||||
* default gateway — a worker quietly talking to the wrong endpoint is the failure this avoids.
|
||||
*/
|
||||
private static String[] splitModelSelector(BridgedConfig.Profile cfg) {
|
||||
String model = cfg.model();
|
||||
int slash = model == null ? -1 : model.indexOf('/');
|
||||
if (model == null || model.isBlank() || slash <= 0 || slash == model.length() - 1) {
|
||||
throw new IllegalArgumentException(
|
||||
"profile " + cfg.profile() + " sets baseUrl (a pinned opencode endpoint) so"
|
||||
+ " model: must be \"<provider>/<model>\", e.g."
|
||||
+ " \"local-vllm/deepseek-v4-flash\"; got "
|
||||
+ (model == null ? "null" : '"' + model + '"'));
|
||||
}
|
||||
return new String[]{model.substring(0, slash), model.substring(slash + 1)};
|
||||
}
|
||||
|
||||
/**
|
||||
* The OpenAI-compatible base URL for {@code baseUrl}. A bare {@code host:port} gets {@code /v1}
|
||||
* appended (where these servers put the API); a URL that already carries a path is taken as-is,
|
||||
* so an endpoint mounted somewhere unusual is still reachable.
|
||||
*/
|
||||
private static String openAiBaseUrl(String baseUrl) {
|
||||
String trimmed = baseUrl.trim();
|
||||
while (trimmed.endsWith("/")) {
|
||||
trimmed = trimmed.substring(0, trimmed.length() - 1);
|
||||
}
|
||||
int schemeEnd = trimmed.indexOf("://");
|
||||
String afterScheme = schemeEnd < 0 ? trimmed : trimmed.substring(schemeEnd + 3);
|
||||
return afterScheme.contains("/") ? trimmed : trimmed + "/v1";
|
||||
}
|
||||
|
||||
/** Add lazy on-disk session discovery to the base handle. */
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
PeerHandle inner = super.spawn(req);
|
||||
return new SessionAwareHandle(inner, discovery, effectiveCwd(req));
|
||||
}
|
||||
|
||||
/**
|
||||
* A {@link PeerHandle} that delegates everything to the base's worker handle but resolves
|
||||
* {@link #agentSessionId()} lazily through opencode session discovery. Delegate-only, so the
|
||||
* base's id/terminalId/profile semantics (CB-519's host-unique routing key, herdr coordinates)
|
||||
* are untouched — only the opencode-specific identity answer is added. {@code sessionName()}
|
||||
* stays null: opencode has no display-name seam, so the logical name lives only in the bridge's
|
||||
* roster (see the SESSION_NAME capability).
|
||||
*/
|
||||
private static final class SessionAwareHandle implements PeerHandle {
|
||||
private final PeerHandle delegate;
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
private final String cwd;
|
||||
|
||||
SessionAwareHandle(PeerHandle delegate, OpenCodeSessionDiscovery discovery, String cwd) {
|
||||
this.delegate = delegate;
|
||||
this.discovery = discovery;
|
||||
this.cwd = cwd;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String id() {
|
||||
return delegate.id();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String terminalId() {
|
||||
return delegate.terminalId();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String profile() {
|
||||
return delegate.profile();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String sessionName() {
|
||||
return delegate.sessionName();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String agentSessionId() {
|
||||
// Lazy + retried, never a spawn-time blocker: opencode writes the session record only
|
||||
// when the session is first persisted, so null here is the correct interim answer and
|
||||
// the caller re-calls later (each call re-scans, picking up a record that has since
|
||||
// appeared).
|
||||
return discovery.sessionIdForDirectory(cwd);
|
||||
}
|
||||
|
||||
@Override
|
||||
public CharterReceipt charterReceipt() {
|
||||
return delegate.charterReceipt();
|
||||
}
|
||||
}
|
||||
|
||||
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
|
||||
|
||||
/** Spawn a worker for the default profile in the resolved default cwd. */
|
||||
public Agent spawn() {
|
||||
return spawnInternal(null, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile (null → default) in the resolved default cwd. */
|
||||
public Agent spawn(String profileName) {
|
||||
return spawnInternal(profileName, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile with an explicit requested/caller cwd (CB-112). */
|
||||
public Agent spawn(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd);
|
||||
}
|
||||
|
||||
// --- capabilities --------------------------------------------------------------------------
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
Set<Capability> caps = EnumSet.of(Capability.MID_TURN_ASK, Capability.WORKTREE,
|
||||
Capability.ORPHAN_REAP, Capability.SESSION_RESUME);
|
||||
if (hasGitTokenProfile()) {
|
||||
caps.add(Capability.SELF_PR);
|
||||
}
|
||||
// Deliberately NOT SESSION_NAME: opencode has no display-name flag, so the bridge's logical
|
||||
// name can't surface in the peer's own UI — declaring the capability would hide that
|
||||
// asymmetry rather than make it honest. For opencode the name lives only in the bridge's
|
||||
// roster (see PeerHandle.sessionName() returning null).
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
|
||||
/** Whether any configured profile opts into a git-forge token (required for {@link Capability#SELF_PR}). */
|
||||
private boolean hasGitTokenProfile() {
|
||||
return profileConfigs().stream().anyMatch(BridgedConfig.Profile::hasGitToken);
|
||||
}
|
||||
|
||||
// --- CB-117 reap predicate (opencode prefix), kept for direct unit testing -----------------
|
||||
|
||||
/**
|
||||
* Whether {@code name} is an opencode bridge worker started by a <em>different</em> process than
|
||||
* {@code currentNonce}. A thin {@code opencode}-prefix binding of
|
||||
* {@link HerdrPeerLauncher#isForeignWorker(String, String, String)}.
|
||||
*/
|
||||
static boolean isForeignWorker(String name, String currentNonce) {
|
||||
return HerdrPeerLauncher.isForeignWorker(NAME_PREFIX, name, currentNonce);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
/**
|
||||
* Resolves the opencode session id for a bridged worker from opencode's on-disk storage — the
|
||||
* only place this adapter touches opencode's private layout, and deliberately the <em>only</em>
|
||||
* class that does.
|
||||
*
|
||||
* <p><strong>Why this is isolated behind one seam.</strong> The layout is version-coupled and not a
|
||||
* stable contract: opencode writes one JSON file per session under
|
||||
* {@code <storageRoot>/session/<projectID>/<ses_*.json>}, and each record carries a
|
||||
* {@code "version"} field (e.g. {@code "1.1.31"}), so the exact directory shape, file naming, and
|
||||
* field names can move between opencode releases. opencode also ships a headless HTTP server that
|
||||
* may supersede file scanning entirely. Everything this adapter knows about that private storage —
|
||||
* its shape, naming, and field names — lives here, so a layout change, or a switch to the HTTP
|
||||
* server, changes exactly one class and nothing in {@link OpenCodeLauncher}.
|
||||
*
|
||||
* <p>The determinism that makes this useful is structural, not a guess: every bridged worker runs
|
||||
* in its own unique git worktree, so the record's {@code directory} (its project root) equals the
|
||||
* worker's cwd identifies <em>its</em> session unambiguously. We match on {@code directory} rather
|
||||
* than diffing {@code opencode session list} before/after — that races under concurrent spawns, and
|
||||
* the CLI listing does not even show the directory.
|
||||
*
|
||||
* <p>All reads are best-effort and never throw: a missing or unreadable storage root, a record that
|
||||
* fails to parse, or a directory with no record yet all yield {@code null}, and the caller (the
|
||||
* session handle) treats that as "identity not resolved yet" and retries later.
|
||||
*/
|
||||
final class OpenCodeSessionDiscovery {
|
||||
|
||||
private final Path storageRoot; // e.g. ~/.local/share/opencode (injectable for tests)
|
||||
private final ObjectMapper json;
|
||||
|
||||
OpenCodeSessionDiscovery(Path storageRoot) {
|
||||
this.storageRoot = storageRoot;
|
||||
this.json = new ObjectMapper();
|
||||
}
|
||||
|
||||
/**
|
||||
* The opencode session id whose record references {@code directory} (the worker's cwd), or
|
||||
* {@code null} when no record matches yet. When several records share the directory — e.g.
|
||||
* repeated spawns into the same worktree — the <em>most recently modified</em> one wins: it is
|
||||
* the session the pane most likely corresponds to.
|
||||
*
|
||||
* <p>Never throws: a missing {@code storageRoot}, an unreadable/malformed record, or a
|
||||
* directory that has not been persisted yet all resolve to {@code null} rather than failing a
|
||||
* spawn. A bridged worker's session record is written lazily (when the session is first
|
||||
* persisted), so {@code null} here is the normal answer right after the pane is ready, and the
|
||||
* caller retries later.
|
||||
*
|
||||
* @param directory the worker's cwd, as resolved for this spawn
|
||||
* @return the matching session id, or {@code null} if none is known yet
|
||||
*/
|
||||
String sessionIdForDirectory(String directory) {
|
||||
if (directory == null || directory.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
Path sessionRoot = storageRoot.resolve("session");
|
||||
if (!Files.isDirectory(sessionRoot)) {
|
||||
return null;
|
||||
}
|
||||
String best = null;
|
||||
long bestMtime = Long.MIN_VALUE;
|
||||
try (Stream<Path> projectDirs = Files.list(sessionRoot)) {
|
||||
for (Path projectDir : projectDirs.filter(Files::isDirectory).toList()) {
|
||||
try (Stream<Path> records = Files.list(projectDir)) {
|
||||
for (Path record : records.toList()) {
|
||||
String id = matchId(record, directory);
|
||||
if (id == null) {
|
||||
continue;
|
||||
}
|
||||
long mtime = lastModifiedEpochMillis(record);
|
||||
if (mtime > bestMtime) {
|
||||
bestMtime = mtime;
|
||||
best = id;
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// one project dir unreadable — skip it; another may still match
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// storage root vanished or became unreadable — "no session known yet"
|
||||
return null;
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
/**
|
||||
* The record's session id when it references {@code directory}, else {@code null}. A record
|
||||
* that is not JSON, lacks {@code id}/{@code directory}, or points at a different directory is
|
||||
* simply not our session; a malformed one is skipped, never fatal.
|
||||
*/
|
||||
private String matchId(Path record, String directory) {
|
||||
try {
|
||||
JsonNode node = json.readTree(record.toFile());
|
||||
JsonNode id = node == null ? null : node.get("id");
|
||||
JsonNode dir = node == null ? null : node.get("directory");
|
||||
if (id == null || dir == null || !directory.equals(dir.asText())) {
|
||||
return null;
|
||||
}
|
||||
return id.asText();
|
||||
} catch (IOException e) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** The record's last-modified epoch ms, or {@code Long.MIN_VALUE} if unreadable (never wins). */
|
||||
private static long lastModifiedEpochMillis(Path record) {
|
||||
try {
|
||||
return Files.getLastModifiedTime(record).toMillis();
|
||||
} catch (IOException e) {
|
||||
return Long.MIN_VALUE;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,101 @@
|
||||
package dev.ltms.bridged.metrics;
|
||||
|
||||
import dev.ltms.bridged.msg.ReplyInbox;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* The daemon's metric definitions (CB-502) — one place where every series is named, described, and
|
||||
* (for gauges) bound to live state.
|
||||
*
|
||||
* <p>The set is deliberately small: each series maps to a failure mode this project has actually
|
||||
* hit, not to whatever was easy to count. The two worth watching in practice are
|
||||
* {@code bridged_sends_total{outcome="completion_fallback"}} — a rising share means turn detection
|
||||
* is degrading, the CB-115/116/118 failure family — and
|
||||
* {@code bridged_push_nudges_total{outcome="exhausted"}}, which means the primary stopped draining
|
||||
* its inbox and CB-307's active push gave up.
|
||||
*/
|
||||
public final class BridgedMetrics {
|
||||
|
||||
/** Counter: delegated sends by terminal outcome. */
|
||||
public static final String SENDS = "bridged_sends_total";
|
||||
/** Counter: worker replies by the path that carried them (rendezvous vs stranded-to-inbox). */
|
||||
public static final String REPLIES = "bridged_replies_total";
|
||||
/** Counter: push-loop nudges to the primary, by outcome. */
|
||||
public static final String PUSH_NUDGES = "bridged_push_nudges_total";
|
||||
/** Counter: idle-lead heartbeat nudges to the lead, by outcome (CB-551). */
|
||||
public static final String HEARTBEAT_NUDGES = "bridged_lead_heartbeat_nudges_total";
|
||||
/** Counter: spawn attempts by peer kind and outcome. */
|
||||
public static final String SPAWNS = "bridged_spawns_total";
|
||||
/** Counter: herdr socket calls by method and outcome. */
|
||||
public static final String HERDR_CALLS = "bridged_herdr_calls_total";
|
||||
/** Counter: rejected requests by reason (CB-501). */
|
||||
public static final String AUTH_FAILURES = "bridged_auth_failures_total";
|
||||
/** Gauge: session census by lifecycle state. */
|
||||
public static final String SESSIONS = "bridged_sessions";
|
||||
/** Gauge: undrained replies held per target. */
|
||||
public static final String INBOX_DEPTH = "bridged_inbox_depth";
|
||||
|
||||
private BridgedMetrics() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the registry with its help text and live gauges bound.
|
||||
*
|
||||
* @param sessions the authoritative session registry (census gauge)
|
||||
* @param inbox the reply inbox; only used for a depth gauge when it can be inspected
|
||||
*/
|
||||
public static Metrics create(SessionManager sessions, ReplyInbox inbox) {
|
||||
Metrics m = new Metrics();
|
||||
|
||||
m.describe(SENDS, "counter",
|
||||
"Delegated sends by terminal outcome (replied|completion_fallback|timeout|failed).");
|
||||
m.describe(REPLIES, "counter",
|
||||
"Worker replies by delivery path (rendezvous=resolved an open send, inbox=stranded and held).");
|
||||
m.describe(PUSH_NUDGES, "counter",
|
||||
"CB-307 push-loop nudges to the primary (delivered|exhausted).");
|
||||
m.describe(HEARTBEAT_NUDGES, "counter",
|
||||
"CB-551 idle-lead heartbeat nudges (delivered|failed|exhausted). Quiet-cap exhaustion "
|
||||
+ "means the lead idled with nothing pending and was told to stand down.");
|
||||
m.describe(SPAWNS, "counter",
|
||||
"Worker spawn attempts by peer kind and outcome (ready|timeout|guard_rejected).");
|
||||
m.describe(HERDR_CALLS, "counter",
|
||||
"herdr socket calls by method and outcome — the dependency everything else rests on.");
|
||||
m.describe(AUTH_FAILURES, "counter",
|
||||
"Requests refused by CB-501/505 (unauthenticated|forbidden).");
|
||||
m.describe(SESSIONS, "gauge",
|
||||
"Registered worker sessions by lifecycle state.");
|
||||
m.describe(INBOX_DEPTH, "gauge",
|
||||
"Replies held for a target that the primary has not drained. Steady state is 0; "
|
||||
+ "a target stuck above 0 means CB-307 delivery is not completing.");
|
||||
|
||||
// One gauge per state so a scrape shows the whole census even when a state is empty —
|
||||
// an absent series and a zero series read very differently on a dashboard.
|
||||
for (MemberSession.State state : MemberSession.State.values()) {
|
||||
String label = state.name().toLowerCase();
|
||||
m.gauge(SESSIONS, () -> countIn(sessions, state), "state", label);
|
||||
}
|
||||
|
||||
// Depth is per live session, so the label set is only known at scrape time. peek() is the
|
||||
// port's non-destructive read — scraping metrics must never ack a reply out of the inbox.
|
||||
m.collector(INBOX_DEPTH, "target", () -> {
|
||||
Map<String, Number> depths = new LinkedHashMap<>();
|
||||
for (MemberSession s : sessions.roster()) {
|
||||
String target = s.terminalId();
|
||||
if (target == null) {
|
||||
continue;
|
||||
}
|
||||
depths.put(target, inbox.peek(target).size());
|
||||
}
|
||||
return depths;
|
||||
});
|
||||
return m;
|
||||
}
|
||||
|
||||
private static long countIn(SessionManager sessions, MemberSession.State state) {
|
||||
return sessions.roster().stream().filter(s -> s.state() == state).count();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,181 @@
|
||||
package dev.ltms.bridged.metrics;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.NavigableMap;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ConcurrentSkipListMap;
|
||||
import java.util.concurrent.atomic.LongAdder;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The daemon's metric registry and Prometheus text renderer (CB-502).
|
||||
*
|
||||
* <p>Deliberately dependency-free. The roadmap's tech-stack table specified Micrometer, but this
|
||||
* pom already carries an unusual reconciliation burden (a hand-pinned {@code jackson-annotations}
|
||||
* to make the MCP SDK's Jackson 3 coexist with our Jackson 2, a Jetty BOM import to stop version
|
||||
* skew, and four documented accepted-CVE advisories), and the dependency CVE gate this project
|
||||
* mandates could not be run when this landed. The metric set is small and fully known, and
|
||||
* Prometheus text exposition is a stable, well-specified format — so the registry is ~100 lines
|
||||
* here instead of a new transitive tree. {@code GET /metrics} is the swap seam if Micrometer's
|
||||
* ecosystem is ever wanted.
|
||||
*
|
||||
* <p>Thread-safe: counters are {@link LongAdder} (built for contended increment), gauges are
|
||||
* supplier-backed so they read live state at scrape time rather than needing to be pushed.
|
||||
*/
|
||||
public final class Metrics {
|
||||
|
||||
/** Counter series, keyed by the fully-rendered {@code name{labels}} sample id. */
|
||||
private final NavigableMap<String, LongAdder> counters = new ConcurrentSkipListMap<>();
|
||||
/** Gauge series, evaluated at scrape time. */
|
||||
private final NavigableMap<String, Supplier<Number>> gauges = new ConcurrentSkipListMap<>();
|
||||
/** Gauge families whose label set is only known at scrape time, keyed by metric name. */
|
||||
private final NavigableMap<String, Collector> collectors = new ConcurrentSkipListMap<>();
|
||||
/** HELP/TYPE metadata, keyed by bare metric name. */
|
||||
private final Map<String, String[]> meta = new ConcurrentHashMap<>();
|
||||
|
||||
/** A gauge family whose series are discovered per scrape (one label, many values). */
|
||||
private record Collector(String labelName, Supplier<Map<String, Number>> samples) {
|
||||
}
|
||||
|
||||
/** Declare a metric's help text and type once, so the exposition carries HELP/TYPE lines. */
|
||||
public Metrics describe(String name, String type, String help) {
|
||||
meta.put(name, new String[]{type, help});
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Increment a counter by one. */
|
||||
public void inc(String name, String... labelPairs) {
|
||||
add(name, 1, labelPairs);
|
||||
}
|
||||
|
||||
/** Increment a counter by {@code delta}. */
|
||||
public void add(String name, long delta, String... labelPairs) {
|
||||
counters.computeIfAbsent(sample(name, labelPairs), _ -> new LongAdder()).add(delta);
|
||||
}
|
||||
|
||||
/**
|
||||
* Register a live gauge. The supplier is called at scrape time, so it reflects current state
|
||||
* (session census, inbox depth) without anything having to remember to update it.
|
||||
*/
|
||||
public void gauge(String name, Supplier<Number> value, String... labelPairs) {
|
||||
gauges.put(sample(name, labelPairs), value);
|
||||
}
|
||||
|
||||
/**
|
||||
* Register a gauge family whose label values are not known up front — inbox depth per target,
|
||||
* for instance, where the set of targets changes as workers come and go. The supplier returns
|
||||
* {@code labelValue → value} and is evaluated once per scrape.
|
||||
*/
|
||||
public void collector(String name, String labelName, Supplier<Map<String, Number>> samples) {
|
||||
collectors.put(name, new Collector(labelName, samples));
|
||||
}
|
||||
|
||||
/** Current value of a counter series — for assertions in tests. */
|
||||
public long count(String name, String... labelPairs) {
|
||||
LongAdder a = counters.get(sample(name, labelPairs));
|
||||
return a == null ? 0 : a.sum();
|
||||
}
|
||||
|
||||
/**
|
||||
* Render the Prometheus text exposition format (version 0.0.4): optional {@code # HELP} and
|
||||
* {@code # TYPE} lines per metric family, then one line per sample.
|
||||
*/
|
||||
public String render() {
|
||||
StringBuilder out = new StringBuilder(1024);
|
||||
String lastFamily = null;
|
||||
for (Map.Entry<String, LongAdder> e : counters.entrySet()) {
|
||||
lastFamily = emitHeader(out, e.getKey(), lastFamily);
|
||||
out.append(e.getKey()).append(' ').append(e.getValue().sum()).append('\n');
|
||||
}
|
||||
for (Map.Entry<String, Supplier<Number>> e : gauges.entrySet()) {
|
||||
lastFamily = emitHeader(out, e.getKey(), lastFamily);
|
||||
Number v;
|
||||
try {
|
||||
v = e.getValue().get();
|
||||
} catch (RuntimeException ex) {
|
||||
continue; // a broken gauge must never break the whole scrape
|
||||
}
|
||||
if (v == null) {
|
||||
continue;
|
||||
}
|
||||
out.append(e.getKey()).append(' ').append(format(v)).append('\n');
|
||||
}
|
||||
for (Map.Entry<String, Collector> e : collectors.entrySet()) {
|
||||
Map<String, Number> samples;
|
||||
try {
|
||||
samples = e.getValue().samples().get();
|
||||
} catch (RuntimeException ex) {
|
||||
continue; // a broken collector must never break the whole scrape
|
||||
}
|
||||
if (samples == null || samples.isEmpty()) {
|
||||
continue;
|
||||
}
|
||||
lastFamily = emitHeader(out, e.getKey(), lastFamily);
|
||||
// Sort so repeated scrapes are byte-stable and diffable.
|
||||
new java.util.TreeMap<>(samples).forEach((label, v) -> {
|
||||
if (v != null) {
|
||||
out.append(sample(e.getKey(), e.getValue().labelName(), label))
|
||||
.append(' ').append(format(v)).append('\n');
|
||||
}
|
||||
});
|
||||
}
|
||||
return out.toString();
|
||||
}
|
||||
|
||||
/** Emit HELP/TYPE when the sample starts a new metric family; returns the current family. */
|
||||
private String emitHeader(StringBuilder out, String sampleId, String lastFamily) {
|
||||
String family = familyOf(sampleId);
|
||||
if (family.equals(lastFamily)) {
|
||||
return lastFamily;
|
||||
}
|
||||
String[] m = meta.get(family);
|
||||
if (m != null) {
|
||||
out.append("# HELP ").append(family).append(' ').append(m[1]).append('\n');
|
||||
out.append("# TYPE ").append(family).append(' ').append(m[0]).append('\n');
|
||||
}
|
||||
return family;
|
||||
}
|
||||
|
||||
private static String familyOf(String sampleId) {
|
||||
int brace = sampleId.indexOf('{');
|
||||
return brace < 0 ? sampleId : sampleId.substring(0, brace);
|
||||
}
|
||||
|
||||
/** Whole numbers render without a decimal point; everything else as-is. */
|
||||
private static String format(Number v) {
|
||||
double d = v.doubleValue();
|
||||
return (d == Math.rint(d) && !Double.isInfinite(d))
|
||||
? Long.toString((long) d)
|
||||
: Double.toString(d);
|
||||
}
|
||||
|
||||
/** Build the {@code name{k="v",k2="v2"}} sample id; labels are sorted for stable output. */
|
||||
private static String sample(String name, String... labelPairs) {
|
||||
if (labelPairs == null || labelPairs.length == 0) {
|
||||
return name;
|
||||
}
|
||||
if (labelPairs.length % 2 != 0) {
|
||||
throw new IllegalArgumentException("labels must be key/value pairs, got " + labelPairs.length);
|
||||
}
|
||||
NavigableMap<String, String> sorted = new java.util.TreeMap<>();
|
||||
for (int i = 0; i < labelPairs.length; i += 2) {
|
||||
sorted.put(labelPairs[i], labelPairs[i + 1] == null ? "" : labelPairs[i + 1]);
|
||||
}
|
||||
StringBuilder sb = new StringBuilder(name.length() + 16 * sorted.size());
|
||||
sb.append(name).append('{');
|
||||
boolean first = true;
|
||||
for (Map.Entry<String, String> e : sorted.entrySet()) {
|
||||
if (!first) {
|
||||
sb.append(',');
|
||||
}
|
||||
first = false;
|
||||
sb.append(e.getKey()).append("=\"").append(escapeLabel(e.getValue())).append('"');
|
||||
}
|
||||
return sb.append('}').toString();
|
||||
}
|
||||
|
||||
/** Label values are escaped per the exposition format: backslash, quote, newline. */
|
||||
private static String escapeLabel(String v) {
|
||||
return v.replace("\\", "\\\\").replace("\"", "\\\"").replace("\n", "\\n");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,451 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import com.rabbitmq.client.AMQP;
|
||||
import com.rabbitmq.client.Channel;
|
||||
import com.rabbitmq.client.Connection;
|
||||
import com.rabbitmq.client.ConnectionFactory;
|
||||
import com.rabbitmq.client.DeliverCallback;
|
||||
import com.rabbitmq.client.Recoverable;
|
||||
import com.rabbitmq.client.RecoveryListener;
|
||||
import com.rabbitmq.client.Return;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.NavigableMap;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ConcurrentSkipListMap;
|
||||
import java.util.concurrent.ExecutionException;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.TimeoutException;
|
||||
|
||||
/**
|
||||
* AMQP-backed {@link ReplyInbox} (CB-307 Stage 2): genuine cross-restart durability behind the same
|
||||
* port {@link InMemoryReplyInbox} implements as soft state.
|
||||
*
|
||||
* <p><strong>Mapping — consume-and-hold with deferred manual ack.</strong> Each target has a durable
|
||||
* queue {@code agent.<target>.inbox}. The gateway that owns the target starts a manual-ack consumer
|
||||
* ({@link #own}) that pulls persistent messages off that queue into an in-memory <em>held</em> map
|
||||
* (keyed by {@code msgId}) but does <em>not</em> ack them. {@link #peek} returns that snapshot;
|
||||
* {@link #ack} acks the broker delivery-tag and drops the entry. Because messages stay unacked until
|
||||
* the owning gateway actually drains them, a crash (or a {@code java -jar} bounce) before caller-ack
|
||||
* leaves them on the broker — it redelivers on reconnect. That is the durability the in-memory
|
||||
* adapter cannot give, with the port contract preserved.
|
||||
*
|
||||
* <p><strong>Prefetch bounds the held backlog (CB-527).</strong> The consumer channel calls
|
||||
* {@code basicQos} with a configurable prefetch count ({@link #DEFAULT_PREFETCH} unless the caller
|
||||
* passes another value to {@link #open(String, int)}) before starting any consumer. Without a bound,
|
||||
* the broker pushes its entire queue into {@link #held} the instant a target is {@link #own owned},
|
||||
* so an undrained primary grows the JVM heap without limit and any queue-level control
|
||||
* ({@code x-max-length}, per-message TTL) never fires because the queue never actually holds a
|
||||
* backlog. Prefetch keeps the backlog where it is visible — on the broker — until the owner drains it.
|
||||
*
|
||||
* <p><strong>Publishes require a confirmed, routable delivery (CB-528).</strong> {@link #publish}
|
||||
* runs on a channel separate from the consume/ack channel ({@link #channel}), so a slow or blocked
|
||||
* publish confirm can never hold {@link #channelLock} and stall an ack — the ack path never waits on
|
||||
* a publish confirm. That publish channel is in publisher-confirm mode and every publish sets the
|
||||
* {@code mandatory} flag, so an unroutable publish (queue not declared, e.g. the owner never called
|
||||
* {@link #own}) is returned by the broker instead of silently dropped. The broker sends the
|
||||
* <em>return</em> for an unroutable message before the <em>confirm</em> that covers it — the ack/nack
|
||||
* callback checks the returned-set at confirm time rather than assuming an ack means routed — so
|
||||
* "confirmed" here means "durably queued", not merely "accepted by the broker". A returned or nacked
|
||||
* (or un-confirmed within the timeout) publish surfaces as an {@link IllegalStateException} on the
|
||||
* caller's thread; the caller — {@link MessageService#reply} — must not report success for a
|
||||
* black-holed reply.
|
||||
*
|
||||
* <p><strong>Ownership is explicit.</strong> {@link #own} declares the queue and starts the consumer;
|
||||
* {@link #release} cancels it. {@link #publish} sends to the queue but does <em>not</em> imply ownership
|
||||
* and does not attach a consumer. This split is required by CB-308 federation, where one gateway may
|
||||
* publish to an agent owned by another gateway; in that case the publisher must not compete for
|
||||
* deliveries.
|
||||
*
|
||||
* <p><strong>Dedup.</strong> The consumer keys the held map by {@code msgId}; a redelivered duplicate
|
||||
* (at-least-once, or a producer double-publish) is acked-and-dropped on arrival, so it never
|
||||
* double-queues.
|
||||
*
|
||||
* <p><strong>Visibility.</strong> Unlike the in-memory adapter, publish → broker → consumer is
|
||||
* asynchronous, so a {@link #peek} immediately after {@link #publish} may not yet see the message
|
||||
* (broker delivery latency). Callers that need the reply drained poll (as the primary already does);
|
||||
* the contract test waits for visibility. This is inherent to broker-backed delivery, not a defect.
|
||||
*
|
||||
* <p>The default deploy targets LavinMQ; a stock RabbitMQ speaks the same AMQP 0-9-1 (URI-only swap),
|
||||
* so the {@code @Tag("contract")} integration test runs against a RabbitMQ container.
|
||||
*/
|
||||
public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(AmqpReplyInbox.class);
|
||||
|
||||
private static final String QUEUE_PREFIX = "agent.";
|
||||
private static final String QUEUE_SUFFIX = ".inbox";
|
||||
|
||||
/** CB-527: the prefetch used when a caller does not pass an explicit value to {@link #open(String, int)}. */
|
||||
public static final int DEFAULT_PREFETCH = 32;
|
||||
|
||||
/** How long {@link #publish} waits for its publisher confirm before failing the call (CB-528). */
|
||||
private static final long CONFIRM_TIMEOUT_MS = 10_000L;
|
||||
|
||||
private final Connection connection;
|
||||
private final Channel channel;
|
||||
/** All channel operations (publish/declare/ack/cancel) serialize on this — a Channel is not thread-safe. */
|
||||
private final Object channelLock = new Object();
|
||||
/** target → (msgId → held delivery). Per-target map is guarded by synchronizing on itself. */
|
||||
private final ConcurrentHashMap<String, LinkedHashMap<String, Held>> held = new ConcurrentHashMap<>();
|
||||
/** Targets whose queue is declared and consumer is running, mapped to their broker consumer tag. */
|
||||
private final ConcurrentHashMap<String, String> consumerTags = new ConcurrentHashMap<>();
|
||||
|
||||
/**
|
||||
* CB-528: a dedicated channel for {@link #publish}, kept separate from {@link #channel} (consume
|
||||
* + ack) so a publish confirm round trip never blocks under {@link #channelLock} and stalls an ack.
|
||||
*/
|
||||
private final Channel publishChannel;
|
||||
private final Object publishChannelLock = new Object();
|
||||
/** In-flight publishes awaiting their confirm, keyed by the publish channel's sequence number. */
|
||||
private final ConcurrentSkipListMap<Long, Pending> pendingBySeq = new ConcurrentSkipListMap<>();
|
||||
/**
|
||||
* The same in-flight publishes, keyed by {@code msgId} — a broker {@code Return} carries no delivery
|
||||
* tag. Assumes {@code msgId} is unique per in-flight publish: a second {@link #publish} for a
|
||||
* {@code msgId} still awaiting its confirm would overwrite this entry and misdirect
|
||||
* {@link #onReturn}'s lookup. Not reachable today — {@code MessageService.reply} generates a fresh
|
||||
* {@code UUID} per call — so no guard is added for it.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, Pending> pendingByMsgId = new ConcurrentHashMap<>();
|
||||
|
||||
/** A message pulled off the broker but not yet acked: its delivery-tag plus the port payload. */
|
||||
private record Held(long deliveryTag, InboxMessage message) {}
|
||||
|
||||
/** A publish awaiting its confirm; {@link #returned} records whether the broker already returned it. */
|
||||
private static final class Pending {
|
||||
final String msgId;
|
||||
final CompletableFuture<Void> confirmed = new CompletableFuture<>();
|
||||
volatile boolean returned;
|
||||
|
||||
Pending(String msgId) {
|
||||
this.msgId = msgId;
|
||||
}
|
||||
}
|
||||
|
||||
/** Connect to {@code uri} (e.g. {@code amqp://guest:guest@127.0.0.1:5672/}) with {@link #DEFAULT_PREFETCH}. */
|
||||
public static AmqpReplyInbox open(String uri) {
|
||||
return open(uri, DEFAULT_PREFETCH);
|
||||
}
|
||||
|
||||
/** As {@link #open(String)}, with an explicit consumer prefetch (CB-527: caps the held backlog per target). */
|
||||
public static AmqpReplyInbox open(String uri, int prefetch) {
|
||||
try {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares queues and re-attaches consumers.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
return new AmqpReplyInbox(factory.newConnection("bridged-reply-inbox"), prefetch);
|
||||
} catch (Exception e) {
|
||||
throw new IllegalStateException("cannot connect to AMQP broker at " + uri, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** Wrap an already-open connection with {@link #DEFAULT_PREFETCH} (injection seam for the contract test). */
|
||||
AmqpReplyInbox(Connection connection) {
|
||||
this(connection, DEFAULT_PREFETCH);
|
||||
}
|
||||
|
||||
/** As above, with an explicit prefetch (injection seam for the contract test). */
|
||||
AmqpReplyInbox(Connection connection, int prefetch) {
|
||||
this.connection = connection;
|
||||
try {
|
||||
this.channel = connection.createChannel();
|
||||
// CB-527: bound the held backlog per owned target — must be set before any own()/basicConsume.
|
||||
this.channel.basicQos(prefetch);
|
||||
this.publishChannel = connection.createChannel();
|
||||
this.publishChannel.confirmSelect();
|
||||
this.publishChannel.addReturnListener(this::onReturn);
|
||||
this.publishChannel.addConfirmListener(this::onAck, this::onNack);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot open AMQP channel", e);
|
||||
}
|
||||
// On automatic recovery the broker redelivers unacked messages with FRESH delivery-tags; the
|
||||
// tags we were holding are now stale. Drop the held snapshot so the re-attached consumer
|
||||
// repopulates it with valid tags (dedup by msgId still prevents any double-queue). Any publish
|
||||
// confirm still in flight when the connection dropped is equally stale — its sequence number
|
||||
// meant nothing on the old channel and means nothing on the recovered one, so fail it now
|
||||
// rather than let it silently ride out CONFIRM_TIMEOUT_MS.
|
||||
if (connection instanceof Recoverable recoverable) {
|
||||
recoverable.addRecoveryListener(new RecoveryListener() {
|
||||
@Override
|
||||
public void handleRecovery(Recoverable recoverable) {
|
||||
held.clear();
|
||||
failPendingPublishesOnRecovery();
|
||||
log.info("AMQP connection recovered; cleared held replies for fresh redelivery");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void handleRecoveryStarted(Recoverable recoverable) {
|
||||
// no-op: we act once recovery completes
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void own(String target) {
|
||||
synchronized (channelLock) {
|
||||
if (consumerTags.containsKey(target)) {
|
||||
return; // already owning this target
|
||||
}
|
||||
String queue = queueName(target);
|
||||
try {
|
||||
channel.queueDeclare(queue, true, false, false, null); // durable, non-exclusive, keep on idle
|
||||
String tag = channel.basicConsume(queue, false, deliverCallback(target), _ -> { });
|
||||
consumerTags.put(target, tag);
|
||||
log.debug("AMQP inbox owns queue {} for target {}", queue, target);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot own queue " + queue, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
synchronized (channelLock) {
|
||||
String tag = consumerTags.remove(target);
|
||||
held.remove(target); // stale delivery tags must not survive release
|
||||
if (tag == null) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
channel.basicCancel(tag);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot cancel consumer for " + target, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Publish {@code content} and block until the broker's publisher confirm for it lands (CB-528).
|
||||
* Throws {@link IllegalStateException} if the message is returned as unroutable, nacked, or not
|
||||
* confirmed within {@link #CONFIRM_TIMEOUT_MS} — the caller must treat that as a failed publish,
|
||||
* not a lost-and-forgotten one.
|
||||
*/
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
AMQP.BasicProperties props = new AMQP.BasicProperties.Builder()
|
||||
.messageId(msgId)
|
||||
.deliveryMode(2) // persistent — survives a broker restart
|
||||
.contentType("text/plain")
|
||||
.build();
|
||||
Pending pending = new Pending(msgId);
|
||||
long seq;
|
||||
synchronized (publishChannelLock) {
|
||||
seq = publishChannel.getNextPublishSeqNo();
|
||||
pendingBySeq.put(seq, pending);
|
||||
pendingByMsgId.put(msgId, pending);
|
||||
try {
|
||||
publishChannel.basicPublish("", queueName(target), true, props,
|
||||
content.getBytes(StandardCharsets.UTF_8));
|
||||
} catch (IOException e) {
|
||||
pendingBySeq.remove(seq, pending);
|
||||
pendingByMsgId.remove(msgId, pending);
|
||||
throw new IllegalStateException("cannot publish reply to " + queueName(target), e);
|
||||
}
|
||||
}
|
||||
try {
|
||||
pending.confirmed.get(CONFIRM_TIMEOUT_MS, TimeUnit.MILLISECONDS);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (TimeoutException e) {
|
||||
throw new IllegalStateException("publish confirm for reply " + msgId + " to " + queueName(target)
|
||||
+ " timed out after " + CONFIRM_TIMEOUT_MS + "ms — broker may be unreachable or overloaded", e);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting publish confirm for " + msgId, e);
|
||||
} finally {
|
||||
pendingBySeq.remove(seq, pending);
|
||||
pendingByMsgId.remove(msgId, pending);
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
return List.of();
|
||||
}
|
||||
synchronized (perTarget) {
|
||||
return perTarget.values().stream().map(Held::message).toList();
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
return;
|
||||
}
|
||||
Held h;
|
||||
synchronized (perTarget) {
|
||||
h = perTarget.remove(msgId);
|
||||
}
|
||||
if (h == null) {
|
||||
return; // never held (or already acked) — no-op
|
||||
}
|
||||
try {
|
||||
synchronized (channelLock) {
|
||||
channel.basicAck(h.deliveryTag(), false);
|
||||
}
|
||||
} catch (IOException e) {
|
||||
// Ack didn't reach the broker: restore the entry so a later ack (or a redelivery after
|
||||
// reconnect) can retry. Keeps the at-least-once contract — a reply is never silently lost.
|
||||
synchronized (perTarget) {
|
||||
perTarget.putIfAbsent(msgId, h);
|
||||
}
|
||||
throw new IllegalStateException("cannot ack reply " + msgId + " on " + queueName(target), e);
|
||||
}
|
||||
}
|
||||
|
||||
private DeliverCallback deliverCallback(String target) {
|
||||
return (_, delivery) -> {
|
||||
String msgId = delivery.getProperties().getMessageId();
|
||||
long tag = delivery.getEnvelope().getDeliveryTag();
|
||||
if (msgId == null || msgId.isBlank()) {
|
||||
msgId = Long.toHexString(tag); // synthesize an id so dedup still has a key
|
||||
}
|
||||
String content = new String(delivery.getBody(), StandardCharsets.UTF_8);
|
||||
var perTarget = held.computeIfAbsent(target, _ -> new LinkedHashMap<>());
|
||||
boolean duplicate;
|
||||
synchronized (perTarget) {
|
||||
if (perTarget.containsKey(msgId)) {
|
||||
duplicate = true;
|
||||
} else {
|
||||
perTarget.put(msgId, new Held(tag, new InboxMessage(msgId, target, content)));
|
||||
duplicate = false;
|
||||
}
|
||||
}
|
||||
if (duplicate) {
|
||||
// Redelivered duplicate: ack the new tag and drop it so the broker stops resending.
|
||||
synchronized (channelLock) {
|
||||
channel.basicAck(tag, false);
|
||||
}
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
/** Broker return for an unroutable {@code mandatory} publish — arrives BEFORE its confirm (CB-528). */
|
||||
private void onReturn(Return r) {
|
||||
String msgId = r.getProperties() == null ? null : r.getProperties().getMessageId();
|
||||
Pending pending = msgId == null ? null : pendingByMsgId.get(msgId);
|
||||
if (pending != null) {
|
||||
pending.returned = true;
|
||||
} else {
|
||||
log.warn("AMQP return for reply {} (routingKey={}, {} {}) with no matching in-flight publish"
|
||||
+ " — already resolved by a prior confirm", msgId, r.getRoutingKey(), r.getReplyCode(),
|
||||
r.getReplyText());
|
||||
}
|
||||
}
|
||||
|
||||
private void onAck(long seq, boolean multiple) {
|
||||
resolveConfirm(seq, multiple, true);
|
||||
}
|
||||
|
||||
private void onNack(long seq, boolean multiple) {
|
||||
resolveConfirm(seq, multiple, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve every pending publish covered by this confirm (a single seq, or — {@code multiple} —
|
||||
* every seq up to and including it). Checks {@link Pending#returned} at confirm time: since the
|
||||
* broker's return for an unroutable message always precedes its confirm, an ack that arrives after
|
||||
* a return means "confirmed but never routed", not "durably queued".
|
||||
*/
|
||||
private void resolveConfirm(long seq, boolean multiple, boolean ack) {
|
||||
NavigableMap<Long, Pending> covered = multiple
|
||||
? pendingBySeq.headMap(seq, true)
|
||||
: pendingBySeq.subMap(seq, true, seq, true);
|
||||
for (var it = covered.entrySet().iterator(); it.hasNext(); ) {
|
||||
Pending pending = it.next().getValue();
|
||||
it.remove();
|
||||
pendingByMsgId.remove(pending.msgId, pending);
|
||||
if (ack && !pending.returned) {
|
||||
pending.confirmed.complete(null);
|
||||
} else if (ack) {
|
||||
pending.confirmed.completeExceptionally(new IllegalStateException(
|
||||
"reply " + pending.msgId + " was returned as unroutable (queue not declared/owned)"));
|
||||
} else {
|
||||
pending.confirmed.completeExceptionally(new IllegalStateException(
|
||||
"broker nacked publish of reply " + pending.msgId));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fail every publish still awaiting its confirm — their sequence numbers are stale after recovery.
|
||||
* Guarded by {@link #publishChannelLock}, the same lock {@link #publish} holds while it takes its
|
||||
* sequence number and registers its {@link Pending}: without it, a {@link #publish} that starts
|
||||
* after the connection has already recovered (so it publishes — and will be confirmed — on the
|
||||
* <em>new</em> channel) can register between this sweep's iteration and its clear, and this sweep
|
||||
* then fails a publish that actually succeeded. {@link #publish} only holds the lock for the
|
||||
* seq/map-put/{@code basicPublish} — it awaits the confirm outside it — so this sweep can only ever
|
||||
* wait for an in-flight {@code basicPublish} call to return, never for a broker round trip. No
|
||||
* deadlock.
|
||||
*
|
||||
* <p>Package-private (rather than {@code private}) only so the unit test can drive it directly
|
||||
* against a concurrent {@link #publish} without a live broker reconnect.
|
||||
*/
|
||||
void failPendingPublishesOnRecovery() {
|
||||
synchronized (publishChannelLock) {
|
||||
for (var it = pendingBySeq.entrySet().iterator(); it.hasNext(); ) {
|
||||
Pending pending = it.next().getValue();
|
||||
it.remove();
|
||||
pendingByMsgId.remove(pending.msgId, pending);
|
||||
pending.confirmed.completeExceptionally(new IllegalStateException(
|
||||
"AMQP connection recovered mid-publish; confirm status of reply " + pending.msgId
|
||||
+ " is unknown"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fail every publish still awaiting its confirm with a clear, immediate error instead of leaving it
|
||||
* to time out after {@link #CONFIRM_TIMEOUT_MS} once the channels are closed underneath it. Guarded
|
||||
* by {@link #publishChannelLock} for the same reason as {@link #failPendingPublishesOnRecovery}.
|
||||
*/
|
||||
private void failPendingPublishesOnClose() {
|
||||
synchronized (publishChannelLock) {
|
||||
for (var it = pendingBySeq.entrySet().iterator(); it.hasNext(); ) {
|
||||
Pending pending = it.next().getValue();
|
||||
it.remove();
|
||||
pendingByMsgId.remove(pending.msgId, pending);
|
||||
pending.confirmed.completeExceptionally(new IllegalStateException(
|
||||
"AMQP reply inbox closed while publish of reply " + pending.msgId
|
||||
+ " was still awaiting its confirm"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private static String queueName(String target) {
|
||||
return QUEUE_PREFIX + target + QUEUE_SUFFIX;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
failPendingPublishesOnClose();
|
||||
try {
|
||||
channel.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("AMQP channel close: {}", e.toString());
|
||||
}
|
||||
try {
|
||||
publishChannel.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("AMQP publish channel close: {}", e.toString());
|
||||
}
|
||||
try {
|
||||
connection.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("AMQP connection close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* Soft-state {@link ReplyInbox} backed by a {@link ConcurrentHashMap} keyed by target session.
|
||||
* Per-target FIFO ordering (insertion order via {@link LinkedHashMap}). Dedup by {@code msgId}
|
||||
* within a target. Thread-safe for concurrent publish vs. drain.
|
||||
*
|
||||
* <p><strong>Ownership is explicit.</strong> {@link #own} marks a target as locally owned so that
|
||||
* {@link #peek} and {@link #ack} operate on it; {@link #publish} works whether or not the target is
|
||||
* owned. {@link #release} clears the local snapshot. This mirrors the AMQP adapter's contract so the
|
||||
* non-broker path stays interchangeable.
|
||||
*
|
||||
* <p><strong>This is soft-state, NOT persistence.</strong> Lost on a {@code java -jar} bounce — that
|
||||
* is correct and consistent with "bridged stays soft-state." The Stage-2 AMQP adapter replaces this.
|
||||
*/
|
||||
public final class InMemoryReplyInbox implements ReplyInbox {
|
||||
|
||||
private final ConcurrentHashMap<String, LinkedHashMap<String, InboxMessage>> store = new ConcurrentHashMap<>();
|
||||
private final Set<String> owned = ConcurrentHashMap.newKeySet();
|
||||
|
||||
@Override
|
||||
public void own(String target) {
|
||||
owned.add(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
owned.remove(target);
|
||||
store.remove(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
var perTarget = store.computeIfAbsent(target, _ -> new LinkedHashMap<>());
|
||||
//noinspection SynchronizationOnLocalVariableOrMethodParameter
|
||||
synchronized (perTarget) {
|
||||
perTarget.putIfAbsent(msgId, new InboxMessage(msgId, target, content));
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
if (!owned.contains(target)) {
|
||||
return List.of();
|
||||
}
|
||||
var perTarget = store.get(target);
|
||||
if (perTarget == null) {
|
||||
return List.of();
|
||||
}
|
||||
//noinspection SynchronizationOnLocalVariableOrMethodParameter
|
||||
synchronized (perTarget) {
|
||||
return List.copyOf(perTarget.values());
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
if (!owned.contains(target)) {
|
||||
return;
|
||||
}
|
||||
var perTarget = store.get(target);
|
||||
if (perTarget != null) {
|
||||
//noinspection SynchronizationOnLocalVariableOrMethodParameter
|
||||
synchronized (perTarget) {
|
||||
perTarget.remove(msgId);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,351 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* CB-551: an opt-in heartbeat that nudges the single idle lead back to work once it has been
|
||||
* continuously idle past a quiet period with no open {@code bridge_send} driving it.
|
||||
*
|
||||
* <p>Why this exists: the fleet is ONE lead + architects + workers, so an idle, stalled lead is a
|
||||
* single point of failure for the fleet's progress. {@link ReplyPushLoop} nudges the lead only when
|
||||
* a worker reply lands; this loop is the timer that catches the gap where nothing lands and the
|
||||
* lead simply sits idle with nobody to prompt it onward.
|
||||
*
|
||||
* <p>Mechanism mirrors {@link ReplyPushLoop}: a pure, unit-testable decision function
|
||||
* ({@link #decide(long, Long, int, AgentStatus, boolean, boolean, FleetState)}) plus a thin
|
||||
* scheduler around it. The loop is <em>entirely opt-in</em> ({@code leadHeartbeat:} in config); with
|
||||
* no such block it is never constructed, so upgrading the daemon cannot silently acquire a behaviour
|
||||
* that spends the operator's model subscription on its own initiative (constraint 1).
|
||||
*
|
||||
* <p>Four invariants keep it from becoming a runaway subscription burner:
|
||||
* <ol>
|
||||
* <li><b>Status-gated</b> — a {@code WORKING} lead is making progress and is never touched; only an
|
||||
* injectable (idle/done/blocked) lead is even considered (constraint 2).</li>
|
||||
* <li><b>Debounced</b> — a nudge only fires once the lead has been <em>continuously</em> idle past
|
||||
* {@code idleAfterSeconds}, so a lead that just finished a turn is not re-prompted into every
|
||||
* natural pause (constraint 3).</li>
|
||||
* <li><b>Bounded when quiet</b> — consecutive nudges that find no pending fleet state are capped at
|
||||
* {@code quietNudgeCap}, then the loop stops nudging until real state returns (constraint 4).</li>
|
||||
* <li><b>Never races {@link ReplyPushLoop}</b> — while that loop is actively nudging any target this
|
||||
* loop stands down, so two competing injections never start two turns in the same pane
|
||||
* (constraint 6).</li>
|
||||
* </ol>
|
||||
*/
|
||||
public final class LeadHeartbeatLoop {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadHeartbeatLoop.class);
|
||||
|
||||
/** Sentinel for {@link #idleSinceNanos}: the lead is not currently in an idle stretch. */
|
||||
private static final long NOT_IDLE = Long.MIN_VALUE;
|
||||
|
||||
private final PrimaryRegistry primaryRegistry;
|
||||
private final AgentControl agents;
|
||||
private final ReplyInbox inbox;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final ReplyPushLoop pushLoop;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final long idleAfterNanos;
|
||||
private final long backoffMs;
|
||||
private final int quietNudgeCap;
|
||||
private final Metrics metrics; // CB-512 pattern: nullable — no registry in unit tests
|
||||
|
||||
/** When the current idle stretch began (nanos), or {@link #NOT_IDLE}. Single scheduler thread only. */
|
||||
private long idleSinceNanos = NOT_IDLE;
|
||||
/** Consecutive nudges that found no pending fleet state. Single scheduler thread only. */
|
||||
private int quietCount = 0;
|
||||
|
||||
/** Constructor with an injectable clock and no metric registry (unit tests, or wiring that opts out). */
|
||||
public LeadHeartbeatLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
Supplier<List<MemberSession>> roster, ReplyPushLoop pushLoop,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock,
|
||||
long idleAfterNanos, long backoffMs, int quietNudgeCap) {
|
||||
this(primaryRegistry, agents, inbox, roster, pushLoop, scheduler, clock,
|
||||
idleAfterNanos, backoffMs, quietNudgeCap, null);
|
||||
}
|
||||
|
||||
/** As above, with a metric registry (the CB-512 pattern) so nudge outcomes are counted. */
|
||||
public LeadHeartbeatLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
Supplier<List<MemberSession>> roster, ReplyPushLoop pushLoop,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock,
|
||||
long idleAfterNanos, long backoffMs, int quietNudgeCap, Metrics metrics) {
|
||||
this.primaryRegistry = primaryRegistry;
|
||||
this.agents = agents;
|
||||
this.inbox = inbox;
|
||||
this.roster = roster;
|
||||
this.pushLoop = pushLoop;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.idleAfterNanos = idleAfterNanos;
|
||||
this.backoffMs = backoffMs;
|
||||
this.quietNudgeCap = quietNudgeCap;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Count one nudge outcome when a registry is wired; a no-op in unit tests. */
|
||||
private void countNudge(String outcome) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(BridgedMetrics.HEARTBEAT_NUDGES, "outcome", outcome);
|
||||
}
|
||||
}
|
||||
|
||||
// --- decision logic (package-private for unit-testing) --------------------------------------
|
||||
|
||||
/** The action the loop should take on one tick. */
|
||||
enum Action {
|
||||
/** Send a heartbeat nudge to the lead now. */
|
||||
INJECT,
|
||||
/** Lead is idle but not yet past the quiet period — keep waiting, do nothing. */
|
||||
WAIT_IDLE,
|
||||
/** Lead is making progress (or its status cannot be read) — reset the idle window and re-arm. */
|
||||
LEAD_BUSY,
|
||||
/** Idle past the quiet period with nothing pending and the cap exhausted — stop until new state appears. */
|
||||
QUIET_DONE,
|
||||
/** {@link ReplyPushLoop} is actively nudging — stand aside rather than start a competing turn. */
|
||||
STAND_DOWN
|
||||
}
|
||||
|
||||
/** The outcome of one decision: the action plus the state to persist for the next tick. */
|
||||
record Decision(Action action, Long idleSinceNanos, int quietCount) {}
|
||||
|
||||
/**
|
||||
* Pure decision function: given the current loop state and fleet/lead facts, return what to do
|
||||
* next and the exact state to carry forward. Pure means no I/O and no mutation — the caller
|
||||
* ({@link #tick}) applies {@link Decision} to its fields. This is what makes the loop unit-testable
|
||||
* with a fake clock and no sleeping.
|
||||
*
|
||||
* @param nowNanos the current time (an injected clock in tests, {@link System#nanoTime()} live)
|
||||
* @param idleSinceNanos when the current idle stretch began, or {@code null} if the lead is not idle
|
||||
* @param quietCount consecutive nudges so far that found no pending fleet state
|
||||
* @param status the lead's herdr status
|
||||
* @param pushLoopActive whether {@link ReplyPushLoop} is currently nudging some target (constraint 6)
|
||||
* @param leadKnown whether a lead terminal is known to nudge at all
|
||||
* @param fleet a snapshot of the pending fleet state (constraint 5)
|
||||
* @return the action to take and the state to persist
|
||||
*/
|
||||
Decision decide(long nowNanos, Long idleSinceNanos, int quietCount, AgentStatus status,
|
||||
boolean pushLoopActive, boolean leadKnown, FleetState fleet) {
|
||||
// Constraint 6: while ReplyPushLoop is actively nudging the lead, injecting a second,
|
||||
// competing prompt into the same pane would start a second turn — racing loops multiply
|
||||
// turns and context burn. Stand aside, and treat the active push as real state (re-arm the
|
||||
// quiet counter), because the reply that drove it is exactly the kind of new state that
|
||||
// should reset the cap.
|
||||
if (pushLoopActive) {
|
||||
return new Decision(Action.STAND_DOWN, idleSinceNanos, 0);
|
||||
}
|
||||
// Constraint 2: a WORKING lead is making progress and must NOT be touched; an unreadable
|
||||
// status (read failure, or the agent is gone) is safest treated the same way — never inject
|
||||
// into a state we cannot read. Either way, reset the idle window and the quiet counter: the
|
||||
// lead was / may be active, so the next idle stretch must count its own quiet period fresh.
|
||||
if (status == null || !status.injectable()) {
|
||||
return new Decision(Action.LEAD_BUSY, null, 0);
|
||||
}
|
||||
if (idleSinceNanos == null) {
|
||||
// The lead just became injectable — record the start of an idle stretch and wait out the
|
||||
// debounce quiet period before ever nudging (constraint 3).
|
||||
return new Decision(Action.WAIT_IDLE, nowNanos, quietCount);
|
||||
}
|
||||
if (nowNanos - idleSinceNanos < idleAfterNanos) {
|
||||
// Still within the quiet period: the lead that just finished a turn sits momentarily idle
|
||||
// and must not be re-prompted into every natural pause.
|
||||
return new Decision(Action.WAIT_IDLE, idleSinceNanos, quietCount);
|
||||
}
|
||||
// Past the quiet period with an injectable lead: it is a genuine candidate for a nudge. Two
|
||||
// gating facts decide whether and how:
|
||||
if (!leadKnown) {
|
||||
// No lead terminal is known yet (e.g. the registry has not learned one) — there is nobody
|
||||
// to nudge. Keep waiting; the window stays open so discovery re-arms it without a fresh
|
||||
// quiet period.
|
||||
return new Decision(Action.WAIT_IDLE, idleSinceNanos, quietCount);
|
||||
}
|
||||
if (fleet.hasPending()) {
|
||||
// Real fleet state is waiting — a worker reply or a DONE session. This is new state, so
|
||||
// it resets the quiet counter (constraint 4) and the lead is nudged to go collect it.
|
||||
return new Decision(Action.INJECT, idleSinceNanos, 0);
|
||||
}
|
||||
if (quietCount < quietNudgeCap) {
|
||||
// Nothing is pending, but the cap is not exhausted: nudge anyway, telling the lead
|
||||
// exactly that nothing is waiting so it can choose to stand down rather than hunt
|
||||
// (constraint 5). Count it toward the consecutive-quiet cap.
|
||||
return new Decision(Action.INJECT, idleSinceNanos, quietCount + 1);
|
||||
}
|
||||
// Nothing pending and the cap is exhausted: stop nudging until real state appears again
|
||||
// (constraint 4). The loop still ticks on backoff so a genuinely new reply or session change
|
||||
// re-arms it — QUIET_DONE stops injection, not observation.
|
||||
return new Decision(Action.QUIET_DONE, idleSinceNanos, quietCount);
|
||||
}
|
||||
|
||||
// --- loop ----------------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Start the heartbeat. Schedules the first tick after one backoff so a freshly-booting daemon
|
||||
* does not evaluate the lead's idle state before the fleet has settled.
|
||||
*/
|
||||
public void start() {
|
||||
log.info("idle-lead heartbeat: on — nudge lead after {}s idle (recheck {}ms, quiet cap {})",
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleAfterNanos), backoffMs, quietNudgeCap);
|
||||
scheduler.schedule(this::tick, backoffMs, TimeUnit.MILLISECONDS);
|
||||
}
|
||||
|
||||
/** One loop tick, every {@link #backoffMs} — the thin scheduler around {@link #decide}. */
|
||||
private void tick() {
|
||||
boolean leadKnown = primaryRegistry.primaryTerminal().isPresent();
|
||||
FleetState fleet = snapshot(inbox, roster);
|
||||
AgentStatus status = AgentStatus.UNKNOWN;
|
||||
if (leadKnown) {
|
||||
try {
|
||||
status = agents.status(primaryRegistry.primaryTerminal().orElseThrow());
|
||||
} catch (RuntimeException e) {
|
||||
// A failed status read degrades to "unknown" — decide() treats that like a busy lead
|
||||
// and never injects into a state it cannot read. Retry on the next backoff.
|
||||
log.debug("idle-heartbeat: status check failed for lead, will retry: {}", e.toString());
|
||||
}
|
||||
}
|
||||
|
||||
Decision d = decide(clock.getAsLong(),
|
||||
idleSinceNanos == NOT_IDLE ? null : idleSinceNanos,
|
||||
quietCount, status, pushLoop.isActive(), leadKnown, fleet);
|
||||
applyDecision(d);
|
||||
switch (d.action()) {
|
||||
case INJECT -> injectNudge(fleet);
|
||||
case QUIET_DONE -> countNudge("exhausted");
|
||||
case WAIT_IDLE, LEAD_BUSY, STAND_DOWN -> { /* nothing to inject, nothing to count */ }
|
||||
}
|
||||
scheduleNext();
|
||||
}
|
||||
|
||||
/** Persist the state a decision returned, so the next tick starts from it. */
|
||||
private void applyDecision(Decision d) {
|
||||
idleSinceNanos = d.idleSinceNanos() == null ? NOT_IDLE : d.idleSinceNanos();
|
||||
quietCount = d.quietCount();
|
||||
}
|
||||
|
||||
/** Send the nudge to the known lead. */
|
||||
private void injectNudge(FleetState fleet) {
|
||||
var lead = primaryRegistry.primaryTerminal();
|
||||
if (lead.isEmpty()) {
|
||||
return; // the lead disappeared between the decision and the injection
|
||||
}
|
||||
String leadTerminal = lead.get();
|
||||
try {
|
||||
agents.send(leadTerminal, fleet.nudgeText());
|
||||
log.debug("idle-heartbeat: nudge sent to lead {} (quiet nudges so far in this stretch: {})",
|
||||
leadTerminal, quietCount);
|
||||
countNudge("delivered");
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("idle-heartbeat: failed to nudge lead {}: {}", leadTerminal, e.toString());
|
||||
countNudge("failed");
|
||||
}
|
||||
}
|
||||
|
||||
/** Schedule the next tick on the scheduler thread pool. */
|
||||
private void scheduleNext() {
|
||||
scheduler.schedule(this::tick, backoffMs, TimeUnit.MILLISECONDS);
|
||||
}
|
||||
|
||||
// --- lifecycle -----------------------------------------------------------------------------
|
||||
|
||||
/** Shut down the scheduler; outstanding ticks are cancelled. */
|
||||
public void stop() {
|
||||
scheduler.shutdownNow();
|
||||
}
|
||||
|
||||
/** @see #stop() */
|
||||
public void close() {
|
||||
stop();
|
||||
}
|
||||
|
||||
/**
|
||||
* A read-only snapshot of the fleet facts the heartbeat reports to an idle lead. Carries exact
|
||||
* counts and names the targets that hold replies, so the nudge is actionable rather than a
|
||||
* content-free "keep working" that would cost the lead a turn just to discover what there is to
|
||||
* do (constraint 5).
|
||||
*
|
||||
* @param pendingReplies total undrained worker replies across owned targets
|
||||
* @param doneSessions session count in the DONE state — finished turns awaiting teardown
|
||||
* @param liveWorkers sessions still registered (acquired minus released)
|
||||
* @param replyTargets the terminals whose inboxes currently hold at least one reply
|
||||
*/
|
||||
record FleetState(int pendingReplies, int doneSessions, int liveWorkers, List<String> replyTargets) {
|
||||
|
||||
/** True when the lead has something real to collect — a reply or a session awaiting teardown. */
|
||||
boolean hasPending() {
|
||||
return pendingReplies > 0 || doneSessions > 0;
|
||||
}
|
||||
|
||||
/** The nudge body, phrased for the two cases the heartbeat distinguishes. */
|
||||
String nudgeText() {
|
||||
StringBuilder sb = new StringBuilder(
|
||||
"Heartbeat: you are idle and no bridge_send is waiting on you.");
|
||||
if (hasPending()) {
|
||||
sb.append(" The fleet has state to collect: ").append(pendingDetail());
|
||||
} else {
|
||||
sb.append(" Fleet is quiet — nothing pending to collect (")
|
||||
.append(pendingReplies).append(" worker replies, ")
|
||||
.append(doneSessions).append(" DONE sessions); ")
|
||||
.append(liveWorkers).append(" workers live. You may stand down until new state re-arms me.");
|
||||
}
|
||||
return sb.toString();
|
||||
}
|
||||
|
||||
private String pendingDetail() {
|
||||
StringBuilder sb = new StringBuilder();
|
||||
sb.append(pendingReplies).append(" worker ")
|
||||
.append(pendingReplies == 1 ? "reply" : "replies")
|
||||
.append(" pending collection");
|
||||
if (!replyTargets.isEmpty()) {
|
||||
// Render each as the exact command so the lead can act without parsing: the nearest
|
||||
// analogue to ReplyPushLoop's bridge_poll(target=...) nudge.
|
||||
sb.append(" (")
|
||||
.append(String.join(", ",
|
||||
replyTargets.stream().map(t -> "bridge_poll(target=" + t + ")").toList()))
|
||||
.append(")");
|
||||
}
|
||||
sb.append(", ").append(doneSessions).append(" DONE session").append(doneSessions == 1 ? "" : "s")
|
||||
.append(" awaiting teardown")
|
||||
.append(", ").append(liveWorkers).append(" worker").append(liveWorkers == 1 ? "" : "s")
|
||||
.append(" live");
|
||||
return sb.toString();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a {@link FleetState} snapshot over the live roster and the reply inbox — the same
|
||||
* sources the metrics collector uses, so the heartbeat reports facts the operator can cross-check
|
||||
* against {@code /metrics}. Package-private and pure (reads only) so it is unit-testable.
|
||||
*/
|
||||
static FleetState snapshot(ReplyInbox inbox, Supplier<List<MemberSession>> roster) {
|
||||
List<MemberSession> sessions = roster.get();
|
||||
int replies = 0;
|
||||
int done = 0;
|
||||
List<String> replyTargets = new ArrayList<>();
|
||||
for (MemberSession s : sessions) {
|
||||
if (s.terminalId() == null) {
|
||||
continue;
|
||||
}
|
||||
int depth = inbox.peek(s.terminalId()).size();
|
||||
if (depth > 0) {
|
||||
replies += depth;
|
||||
replyTargets.add(s.terminalId());
|
||||
}
|
||||
if (s.state() == MemberSession.State.DONE) {
|
||||
done++;
|
||||
}
|
||||
}
|
||||
return new FleetState(replies, done, sessions.size(), List.copyOf(replyTargets));
|
||||
}
|
||||
}
|
||||
@@ -3,9 +3,13 @@ package dev.ltms.bridged.msg;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.CompletionException;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
@@ -16,6 +20,7 @@ import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.TimeoutException;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.locks.ReentrantLock;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
* The blocking delegation feature (CB-104): deliver {@code content} into a worker and block until
|
||||
@@ -48,8 +53,12 @@ public final class MessageService {
|
||||
*/
|
||||
private static final long ASYNC_TIMEOUT_MS = 30 * 60 * 1_000L;
|
||||
|
||||
/** How long a finished (terminal) ticket is retained for polling before it is pruned. */
|
||||
private static final long TICKET_TTL_NANOS = 10 * 60 * 1_000_000_000L;
|
||||
/**
|
||||
* How long a finished (terminal) ticket is retained for polling before it is pruned. Package-
|
||||
* private (not {@code private}) so a test can advance an injected clock past it deterministically
|
||||
* instead of duplicating the magic number or sleeping for real.
|
||||
*/
|
||||
static final long TICKET_TTL_NANOS = 10 * 60 * 1_000_000_000L;
|
||||
|
||||
/** Outcome of a blocking send. */
|
||||
public enum Outcome {
|
||||
@@ -65,6 +74,14 @@ public final class MessageService {
|
||||
* failure context (e.g. the error screen). Terminal, but not a successful completion.
|
||||
*/
|
||||
WORKER_FAILED,
|
||||
/**
|
||||
* The turn finished without a {@code bridge_reply} and the scrape matched the backend's
|
||||
* configured usage-limit refusal pattern (CB-578 stage A); {@code text} is the reason,
|
||||
* carrying the matched line. The worker's pane is healthy — only its account is refusing —
|
||||
* so this is never reported as a completed reply, and is kept distinct from
|
||||
* {@link #WORKER_FAILED} (a wedged worker) and a session simply going {@code GONE}.
|
||||
*/
|
||||
BACKEND_EXHAUSTED,
|
||||
/**
|
||||
* The worker paused mid-turn to ask the primary a question (CB-205); {@code text} is the
|
||||
* question and {@code turnId} correlates the answer. Not terminal — the primary answers with
|
||||
@@ -123,6 +140,8 @@ public final class MessageService {
|
||||
public enum Phase {
|
||||
/** Delegated and in flight — queued for the worker or being worked. */
|
||||
PENDING,
|
||||
/** The worker is paused in {@code bridge_ask}; {@link TaskView#reply} and {@link TaskView#turnId} identify it. */
|
||||
ASKING,
|
||||
/** The worker's turn finished; {@link TaskView#reply} holds the answer. */
|
||||
DONE,
|
||||
/** The delegation could not complete (timed out, worker gone, or busy). */
|
||||
@@ -132,31 +151,109 @@ public final class MessageService {
|
||||
/**
|
||||
* A poll snapshot of an async delegation.
|
||||
*
|
||||
* @param reply the answer when {@link #phase} is {@link Phase#DONE}, else {@code null}
|
||||
* @param reply the answer when {@link #phase} is {@link Phase#DONE}, or the question when
|
||||
* {@link #phase} is {@link Phase#ASKING}; otherwise {@code null}
|
||||
* @param replySource {@code "reply"} (structured {@code bridge_reply}) or {@code "transcript"}
|
||||
* (completion scrape) when {@link Phase#DONE}, else {@code null}
|
||||
* @param detail a human note (live worker status while pending, or the failure reason)
|
||||
* @param detail a human note (live worker status while pending, ask state, or failure reason)
|
||||
* @param turnId correlation id for an {@link Phase#ASKING} ticket, else {@code null}
|
||||
*/
|
||||
public record TaskView(String ticket, Phase phase, String reply, String replySource, String detail) {
|
||||
public record TaskView(String ticket, Phase phase, String reply, String replySource, String detail,
|
||||
String turnId) {
|
||||
}
|
||||
|
||||
/** An in-flight or finished async delegation, keyed by its ticket. */
|
||||
private record Task(String target, CompletableFuture<Reply> future, long createdNanos) {
|
||||
private static final class Task {
|
||||
private final String ticket;
|
||||
private final String target;
|
||||
private final CompletableFuture<Reply> future = new CompletableFuture<>();
|
||||
private final long createdNanos;
|
||||
private volatile Reply question;
|
||||
private volatile String turnId;
|
||||
|
||||
private Task(String ticket, String target, long createdNanos) {
|
||||
this.ticket = ticket;
|
||||
this.target = target;
|
||||
this.createdNanos = createdNanos;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A worker session's currently-open {@code bridge_ask} question, surfaced so {@code bridge_status}
|
||||
* can show it without the caller needing the ticket first (CB-582). Only covers async
|
||||
* (fire-and-poll) delegations, which track the question on their {@link Task}; a blocking
|
||||
* ({@code wait:true}) send already hands the question straight back to its own caller, so there is
|
||||
* nothing hidden left for {@code bridge_status} to surface in that case.
|
||||
*/
|
||||
public record PendingAsk(String ticket, String question, String turnId) {
|
||||
}
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Injector injector;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ReplyInbox inbox;
|
||||
private final ReplyPushLoop pushLoop;
|
||||
private final Metrics metrics; // CB-502: nullable — no registry in unit tests
|
||||
// CB-588: injectable so pruneTerminalTickets' 10-minute TICKET_TTL_NANOS can be exercised in a
|
||||
// test without a real wait — same seam SessionManager already uses for its idle reaper (nowNanos).
|
||||
private final LongSupplier nowNanos;
|
||||
private final ConcurrentHashMap<String, ReentrantLock> sessionLocks = new ConcurrentHashMap<>();
|
||||
private final ConcurrentHashMap<String, Task> tasks = new ConcurrentHashMap<>();
|
||||
/** Async task that owns each exact forward rendezvous waiter. */
|
||||
private final ConcurrentHashMap<CompletableFuture<Rendezvous.Resolution>, Task> asyncTasksByWaiter =
|
||||
new ConcurrentHashMap<>();
|
||||
/** Async tickets paused on a specific {@code bridge_ask} turn. */
|
||||
private final ConcurrentHashMap<String, Task> asyncTasksByTurn = new ConcurrentHashMap<>();
|
||||
private final AtomicLong ticketSeq = new AtomicLong();
|
||||
private final ExecutorService asyncExecutor = Executors.newThreadPerTaskExecutor(
|
||||
Thread.ofVirtual().name("bridge-async-", 0).factory());
|
||||
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous) {
|
||||
/**
|
||||
* Create with an explicit {@link ReplyInbox} and optional {@link ReplyPushLoop}.
|
||||
*
|
||||
* @param pushLoop nullable — when non-null, the push loop is notified on the no-waiter reply
|
||||
* branch ({@link #reply}) so it can nudge the primary to drain the inbox,
|
||||
* (CB-588) whenever an async ticket started by {@link #sendAsync} reaches a
|
||||
* terminal phase, whenever {@link #poll} hands a terminal ticket to its caller,
|
||||
* and (CB-582) whenever an async ticket's worker pauses mid-turn in
|
||||
* {@code bridge_ask} or that pause ends (answered or lapsed)
|
||||
*/
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous,
|
||||
ReplyInbox inbox, ReplyPushLoop pushLoop) {
|
||||
this(agents, injector, rendezvous, inbox, pushLoop, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, with a metric registry (CB-502). Instrumenting here rather than at the REST and MCP
|
||||
* edges means both surfaces are counted by one piece of code and cannot drift.
|
||||
*
|
||||
* @param metrics nullable — when null, nothing is recorded
|
||||
*/
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous,
|
||||
ReplyInbox inbox, ReplyPushLoop pushLoop, Metrics metrics) {
|
||||
this(agents, injector, rendezvous, inbox, pushLoop, metrics, System::nanoTime);
|
||||
}
|
||||
|
||||
/** Test constructor with an injectable clock (CB-588: exercise the ticket-prune TTL without a real wait). */
|
||||
MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous, ReplyInbox inbox,
|
||||
ReplyPushLoop pushLoop, Metrics metrics, LongSupplier nowNanos) {
|
||||
this.agents = agents;
|
||||
this.injector = injector;
|
||||
this.rendezvous = rendezvous;
|
||||
this.inbox = inbox;
|
||||
this.pushLoop = pushLoop;
|
||||
this.metrics = metrics;
|
||||
this.nowNanos = nowNanos;
|
||||
}
|
||||
|
||||
/** Create with an explicit {@link ReplyInbox} and no push loop. */
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous, ReplyInbox inbox) {
|
||||
this(agents, injector, rendezvous, inbox, null);
|
||||
}
|
||||
|
||||
/** Backward-compatible constructor that uses a default {@link InMemoryReplyInbox}. */
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous) {
|
||||
this(agents, injector, rendezvous, new InMemoryReplyInbox());
|
||||
}
|
||||
|
||||
/** Current lifecycle status of a worker (the {@code GET /sessions/{id}/status} surface). */
|
||||
@@ -164,11 +261,171 @@ public final class MessageService {
|
||||
return agents.status(target);
|
||||
}
|
||||
|
||||
/** Read-only delegation fact for fleet views. */
|
||||
public boolean hasAcceptedDelivery(String target) {
|
||||
return rendezvous.isWaiting(target);
|
||||
}
|
||||
|
||||
/** Read-only inbox fact for fleet views. */
|
||||
public boolean hasInboxMessage(String target) {
|
||||
return !inbox.peek(target).isEmpty();
|
||||
}
|
||||
|
||||
/**
|
||||
* Route a worker's explicit {@code bridge_reply}: resolve an open send, or queue it in the
|
||||
* inbox if no send is currently open. Unlike the bare {@link Rendezvous#resolve}, a no-waiter
|
||||
* result is <em>not</em> a failure — the reply is held for later drain.
|
||||
*
|
||||
* <p><strong>Do NOT use this for mid-turn questions.</strong> {@code bridge_ask} /
|
||||
* {@link Rendezvous#resolveQuestion} must keep today's {@code NO_WAITER} behaviour — questions
|
||||
* are interactive and must never be queued.
|
||||
*
|
||||
* @return always {@code true} — the reply either resolved a live send or was queued
|
||||
*/
|
||||
public boolean reply(String session, String content) {
|
||||
if (rendezvous.resolve(session, content)) {
|
||||
count(BridgedMetrics.REPLIES, "path", "rendezvous");
|
||||
return true; // a live send took it — unchanged fast path
|
||||
}
|
||||
inbox.publish(session, UUID.randomUUID().toString(), content);
|
||||
// A rising inbox share is the signal CB-307 exists to make visible: the worker finished but
|
||||
// nobody was waiting, so delivery now depends on the push loop and a drain.
|
||||
count(BridgedMetrics.REPLIES, "path", "inbox");
|
||||
if (pushLoop != null) {
|
||||
pushLoop.onReplyQueued(session);
|
||||
}
|
||||
return true; // held, not lost
|
||||
}
|
||||
|
||||
/** Record a counter sample when a registry is wired; a no-op in unit tests. */
|
||||
private void count(String name, String... labels) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(name, labels);
|
||||
}
|
||||
}
|
||||
|
||||
/** Count a send's terminal outcome and pass the reply through unchanged. */
|
||||
private Reply recorded(Reply r) {
|
||||
String label = sendOutcomeLabel(r.outcome());
|
||||
if (label != null) {
|
||||
count(BridgedMetrics.SENDS, "outcome", label);
|
||||
}
|
||||
return r;
|
||||
}
|
||||
|
||||
/** Map a terminal send outcome to its metric label, or {@code null} for non-terminal ones. */
|
||||
private static String sendOutcomeLabel(Outcome o) {
|
||||
return switch (o) {
|
||||
case REPLIED -> "replied";
|
||||
case COMPLETED_UNREPLIED -> "completion_fallback";
|
||||
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, BUSY -> "timeout";
|
||||
case WORKER_FAILED -> "failed";
|
||||
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
||||
case STALE_TURN, QUESTION -> null; // not a completed delegation
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Abandon any send still waiting on {@code target} because its session has gone away (CB-516).
|
||||
*
|
||||
* <p>Without this, tearing a worker down left its rendezvous waiter open: a blocking
|
||||
* {@code bridge_send} kept blocking, and an async one kept reporting {@code PENDING} until
|
||||
* {@link #ASYNC_TIMEOUT_MS} — thirty minutes — even though the worker provably no longer
|
||||
* existed and the delegation could never complete. Worse, {@code poll} already had the evidence
|
||||
* (it calls {@code liveStatus} to build its detail string and gets back {@code "unknown"}) and
|
||||
* reported {@code PENDING} anyway.
|
||||
*
|
||||
* <p>Resolving the waiter as a failure — rather than letting it time out — also means the
|
||||
* outcome is counted, so a torn-down delegation stops being invisible to {@code /metrics}.
|
||||
*
|
||||
* @return true if a live waiter was failed
|
||||
*/
|
||||
public boolean abandon(String target, String reason) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(target);
|
||||
boolean failed = waiter != null && !waiter.isDone() && rendezvous.resolveFailure(waiter, reason);
|
||||
boolean asyncFailed = false;
|
||||
for (Task task : tasks.values()) {
|
||||
if (target.equals(task.target) && task.question == null
|
||||
&& task.future.complete(new Reply(Outcome.WORKER_FAILED, reason))) {
|
||||
asyncFailed = true;
|
||||
}
|
||||
}
|
||||
if (failed) {
|
||||
log.warn("abandoning the blocked send to {}: {}", target, reason);
|
||||
}
|
||||
return failed || asyncFailed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Acknowledge a specific reply by {@code msgId} for {@code target}. Removes it from the inbox
|
||||
* so that a subsequent drain or peek no longer returns it.
|
||||
*/
|
||||
public void ackReply(String target, String msgId) {
|
||||
inbox.ack(target, msgId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Drain (peek + ack) all pending inbox replies for {@code target}.
|
||||
*
|
||||
* <p><strong>The ack happens here, before the caller has the messages</strong> — before the MCP
|
||||
* or REST response carrying them has been written, and long before the client has processed
|
||||
* them. That ordering is what the two adapters disagree about, so do not read this method as
|
||||
* "at-least-once" without qualifying which inbox is behind it (CB-529):
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@code InMemoryReplyInbox} — the ack only drops an entry from a local map. The messages
|
||||
* are already in the returned list, so nothing can be lost after this point.
|
||||
* <li>{@code AmqpReplyInbox} — the ack is a broker-side {@code basicAck}. Once it lands the
|
||||
* broker has forgotten the message. If the daemon dies while writing the response, the
|
||||
* reply is gone from the broker <em>and</em> the client never received it. Re-polling
|
||||
* cannot recover it, because there is nothing left to re-deliver.
|
||||
* </ul>
|
||||
*
|
||||
* <p>So the loss window is the response write, and it is a genuine loss rather than a
|
||||
* redelivery. This is accepted, not overlooked: the alternative — ack on the next poll — turns
|
||||
* every normal drain into a double delivery, which costs more than the window it closes. A
|
||||
* caller that needs certainty re-polls; that is idempotent for every case except this one.
|
||||
*
|
||||
* <p>Any change here must be checked against <em>both</em> adapters. The previous version of
|
||||
* this javadoc claimed "the ack is local", which was true when only the in-memory inbox existed
|
||||
* and silently became false when the AMQP adapter landed.
|
||||
*
|
||||
* @return the drained messages, newest last (FIFO); empty list if none
|
||||
*/
|
||||
public List<ReplyInbox.InboxMessage> drainReplies(String target) {
|
||||
var messages = inbox.peek(target);
|
||||
for (var msg : messages) {
|
||||
inbox.ack(target, msg.msgId());
|
||||
}
|
||||
return messages;
|
||||
}
|
||||
|
||||
/**
|
||||
* Deliver {@code content} to {@code target} (a herdr {@code terminal_id}) and block until the
|
||||
* worker replies via {@link Rendezvous} or {@code timeoutMillis} elapses.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis) {
|
||||
return send(target, content, timeoutMillis, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #send(String, String, long)}, but with an accepted-delivery hook.
|
||||
*
|
||||
* <p>{@code onAccepted} is invoked exactly once, once this send has won {@code target}'s send
|
||||
* lock and so become the <em>accepted target turn</em> — it runs <em>before</em> delivery is
|
||||
* queued, so a throwing hook fails the send cleanly (the waiter it already opened is closed and
|
||||
* nothing is left queued). It is <em>not</em> invoked when the send is {@link Outcome#BUSY}
|
||||
* (lock never taken). A caller uses this to record that <em>it</em> now owns the delegation's
|
||||
* reply routing (CB-548: {@code PrimaryRegistry} delegator ownership) — recording only on
|
||||
* acceptance means a concurrent sender that times out {@code BUSY} can never steal ownership it
|
||||
* never earned. {@code null} disables the hook.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis, Runnable onAccepted) {
|
||||
return send(target, content, timeoutMillis, onAccepted, null);
|
||||
}
|
||||
|
||||
/** Run a send, optionally stopping an async task that teardown already failed before acceptance. */
|
||||
private Reply send(String target, String content, long timeoutMillis, Runnable onAccepted, Task task) {
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(target, _ -> new ReentrantLock());
|
||||
|
||||
@@ -176,22 +433,48 @@ public final class MessageService {
|
||||
return new Reply(Outcome.BUSY, null); // another send held the session the whole window
|
||||
}
|
||||
try {
|
||||
CompletableFuture<Void> delivered = injector.enqueue(target, content);
|
||||
if (task != null && task.future.isDone()) {
|
||||
return task.future.getNow(null);
|
||||
}
|
||||
if (hasAsyncQuestion(target)) {
|
||||
return new Reply(Outcome.BUSY, null); // the worker's current turn is paused for its lead
|
||||
}
|
||||
// Open the waiter BEFORE queueing delivery (CB-548). A fast reply — the worker already
|
||||
// injectable the instant we enqueue — otherwise arrives before the waiter is registered
|
||||
// and orphans into the inbox while this send blocks to the timeout (the enqueue-before-
|
||||
// open race). Opening first also means a throwing onAccepted (fired before enqueue) or an
|
||||
// enqueue failure is safely closed by the finally below: nothing is left queued, and the
|
||||
// failed send leaves no stale waiter behind.
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(target);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
} catch (TimeoutException e) {
|
||||
boolean wasDelivered = delivered.isDone() && !delivered.isCompletedExceptionally();
|
||||
log.debug("send to {} timed out (delivered={})", target, wasDelivered);
|
||||
return new Reply(wasDelivered ? Outcome.TIMED_OUT_WORKING : Outcome.TIMED_OUT_QUEUED, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + target, e);
|
||||
if (task != null) {
|
||||
asyncTasksByWaiter.put(reply, task);
|
||||
}
|
||||
TurnToken token = new TurnToken(target, reply);
|
||||
// The send has won the lock; the accepted-delivery hook records delegator ownership
|
||||
// here (CB-548). It runs BEFORE enqueue so a throwing hook — onAccepted is now a
|
||||
// public callback — fails the send without queuing a message that would orphan.
|
||||
if (onAccepted != null) {
|
||||
onAccepted.run();
|
||||
}
|
||||
CompletableFuture<Void> delivered = injector.enqueue(target, content, token);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return recorded(new Reply(outcomeOf(r.kind()), r.text(), r.turnId()));
|
||||
} catch (TimeoutException e) {
|
||||
boolean wasDelivered = delivered.isDone() && !delivered.isCompletedExceptionally();
|
||||
log.debug("send to {} timed out (delivered={})", target, wasDelivered);
|
||||
return recorded(new Reply(
|
||||
wasDelivered ? Outcome.TIMED_OUT_WORKING : Outcome.TIMED_OUT_QUEUED, null));
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + target, e);
|
||||
}
|
||||
} finally {
|
||||
asyncTasksByWaiter.remove(reply);
|
||||
rendezvous.close(target, reply);
|
||||
}
|
||||
} finally {
|
||||
@@ -217,16 +500,31 @@ public final class MessageService {
|
||||
if (ticket.fresh()) {
|
||||
// Register the reverse waiter first, then surface the question — so the answer, which can
|
||||
// arrive the instant the primary reacts, always finds an open waiter to resolve.
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(workerSession);
|
||||
Task task = markAsyncQuestion(waiter, question, ticket.turnId());
|
||||
if (!rendezvous.resolveQuestion(workerSession, question, ticket.turnId())) {
|
||||
if (task != null) {
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
}
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
return new AskResult(AskOutcome.NO_WAITER, null); // no primary is blocked on this worker
|
||||
}
|
||||
// CB-582: the question just became visible via bridge_poll (Phase.ASKING) for an async
|
||||
// (wait:false) delegation — nudge the lead's own pane the same way a terminal ticket does
|
||||
// (CB-588), since the lead's normal poll cadence is minutes away and the reverse-rendezvous
|
||||
// window (~55s, see BridgeMcp/BridgedApp) is far shorter. A blocking (wait:true) send has
|
||||
// no Task and gets the question directly in its own reply, so task == null there — nothing
|
||||
// to nudge.
|
||||
if (task != null && pushLoop != null) {
|
||||
pushLoop.onQuestionOpened(task.ticket, workerSession, ticket.turnId(), question);
|
||||
}
|
||||
}
|
||||
try {
|
||||
String answer = ticket.answer().get(timeoutMillis, TimeUnit.MILLISECONDS);
|
||||
return new AskResult(AskOutcome.ANSWERED, answer);
|
||||
} catch (TimeoutException e) {
|
||||
log.debug("bridge_ask from {} went unanswered in {}ms", workerSession, timeoutMillis);
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
return new AskResult(AskOutcome.TIMED_OUT, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
@@ -238,6 +536,14 @@ public final class MessageService {
|
||||
// Only the fresh owner tears down the shared turn; a duplicate must leave it open.
|
||||
if (ticket.fresh()) {
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
// CB-582: tear the push loop's copy down at the same point, not only on the three
|
||||
// paths that call clearAsyncQuestion. The answer future can complete exceptionally
|
||||
// (ExecutionException) or the thread be interrupted, and both leave this method by
|
||||
// throwing — the question would stay pending forever, keep being named in nudges
|
||||
// until its own cap, and never be removed from the map. Already-closed is a no-op.
|
||||
if (pushLoop != null) {
|
||||
pushLoop.questionClosed(ticket.turnId());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -269,9 +575,12 @@ public final class MessageService {
|
||||
rendezvous.close(workerSession, reply);
|
||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||
}
|
||||
clearAsyncQuestion(turnId, false);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
Reply result = new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
finishAsyncTask(turnId, result);
|
||||
return result;
|
||||
} catch (TimeoutException e) {
|
||||
// The worker resumed but hasn't replied yet — no completion fallback arms an answered
|
||||
// turn (it never re-entered the injector), so a silent worker rides out the window.
|
||||
@@ -298,10 +607,50 @@ public final class MessageService {
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content) {
|
||||
return sendAsync(target, content, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #sendAsync(String, String)}, with the accepted-delivery hook of
|
||||
* {@link #send(String, String, long, Runnable)} — the running {@code send} invokes {@code onAccepted}
|
||||
* the moment it becomes the accepted target turn, so async flooding records delegator ownership
|
||||
* exactly as the blocking path does (CB-548).
|
||||
*
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content, Runnable onAccepted) {
|
||||
String ticket = "task-" + ticketSeq.incrementAndGet();
|
||||
CompletableFuture<Reply> future =
|
||||
CompletableFuture.supplyAsync(() -> send(target, content, ASYNC_TIMEOUT_MS), asyncExecutor);
|
||||
tasks.put(ticket, new Task(target, future, System.nanoTime()));
|
||||
Task task = new Task(ticket, target, nowNanos.getAsLong());
|
||||
tasks.put(ticket, task);
|
||||
if (pushLoop != null) {
|
||||
// CB-588: task.future only ever completes on a terminal phase (DONE or a failure) — a
|
||||
// worker paused in bridge_ask leaves it running, per finishAsyncTask's own contract — so
|
||||
// this fires exactly once, from whichever path completes it: finishAsyncTask(task, result)
|
||||
// below on any non-QUESTION outcome of send() — a worker's bridge_reply, the CB-106
|
||||
// completion fallback, a CB-109 wedge, TIMED_OUT, BUSY, or BACKEND_EXHAUSTED — the same
|
||||
// finishAsyncTask reached via answer()'s finishAsyncTask(turnId, result) once a QUESTION
|
||||
// is resolved, completeExceptionally(t) just below when send() itself throws, or a CB-516
|
||||
// abandon() on teardown. Without this, MessageService.reply's rendezvous fast path (the
|
||||
// one an async ticket always takes) never told the push loop anything happened — see the
|
||||
// class javadoc on sendAsync/CB-107.
|
||||
task.future.whenComplete((reply, ex) -> {
|
||||
boolean failed = ex != null || reply == null || !reply.completed();
|
||||
pushLoop.onTicketTerminal(ticket, target, failed);
|
||||
});
|
||||
}
|
||||
asyncExecutor.submit(() -> {
|
||||
try {
|
||||
Reply result = send(target, content, ASYNC_TIMEOUT_MS, onAccepted, task);
|
||||
if (result.outcome() == Outcome.QUESTION) {
|
||||
// Keep the accepted owner until answer() finishes it. markAsyncQuestion may run
|
||||
// just after resolveQuestion wakes this thread.
|
||||
} else {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
} catch (Throwable t) {
|
||||
task.future.completeExceptionally(t);
|
||||
}
|
||||
});
|
||||
pruneTerminalTickets();
|
||||
log.debug("async send {} -> {}", ticket, target);
|
||||
return ticket;
|
||||
@@ -317,27 +666,39 @@ public final class MessageService {
|
||||
if (task == null) {
|
||||
return null;
|
||||
}
|
||||
CompletableFuture<Reply> f = task.future();
|
||||
CompletableFuture<Reply> f = task.future;
|
||||
if (!f.isDone()) {
|
||||
return new TaskView(ticket, Phase.PENDING, null, null, "worker " + liveStatus(task.target()));
|
||||
Reply question = task.question;
|
||||
if (question != null) {
|
||||
return new TaskView(ticket, Phase.ASKING, question.text(), null,
|
||||
"worker is waiting for your answer", question.turnId());
|
||||
}
|
||||
return new TaskView(ticket, Phase.PENDING, null, null, "worker " + liveStatus(task.target), null);
|
||||
}
|
||||
// CB-588: the ticket is terminal and being handed to the caller right here — tell the push
|
||||
// loop it is collected so a later tick's nudge never names a ticket the lead already has.
|
||||
if (pushLoop != null) {
|
||||
pushLoop.ticketCollected(ticket);
|
||||
}
|
||||
Reply r;
|
||||
try {
|
||||
r = f.getNow(null);
|
||||
} catch (CompletionException | java.util.concurrent.CancellationException e) {
|
||||
Throwable cause = (e instanceof CompletionException ce && ce.getCause() != null) ? ce.getCause() : e;
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, cause.getMessage());
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, cause.getMessage(), null);
|
||||
}
|
||||
if (r.completed()) {
|
||||
String source = r.outcome() == Outcome.REPLIED ? "reply" : "transcript";
|
||||
return new TaskView(ticket, Phase.DONE, r.text(), source, null);
|
||||
return new TaskView(ticket, Phase.DONE, r.text(), source, null, null);
|
||||
}
|
||||
// A wedged worker (CB-109) carries the error context as its reason; the timeout/busy
|
||||
// outcomes carry none, so fall back to the outcome name.
|
||||
String detail = r.outcome() == Outcome.WORKER_FAILED && r.text() != null
|
||||
// A wedged worker (CB-109) or a backend-exhausted classification (CB-578 stage A) carries
|
||||
// the real cause as its reason; the timeout/busy outcomes carry none, so fall back to the
|
||||
// outcome name.
|
||||
boolean carriesReason = r.outcome() == Outcome.WORKER_FAILED || r.outcome() == Outcome.BACKEND_EXHAUSTED;
|
||||
String detail = carriesReason && r.text() != null
|
||||
? r.text()
|
||||
: "no reply — " + r.outcome().name().toLowerCase();
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, detail);
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, detail, null);
|
||||
}
|
||||
|
||||
/** Best-effort live worker status for a pending poll; never throws (a lookup error is just noise). */
|
||||
@@ -349,10 +710,94 @@ public final class MessageService {
|
||||
}
|
||||
}
|
||||
|
||||
/** Drop finished tickets older than the TTL so the registry cannot grow without bound. */
|
||||
/**
|
||||
* Drop finished tickets older than the TTL so {@link #tasks} cannot grow without bound.
|
||||
*
|
||||
* <p>{@code tasks} is the sole authority on whether a ticket still exists — {@link #poll} returns
|
||||
* {@code null} the instant a ticket is gone from here, before it ever reaches the terminal branch
|
||||
* that calls {@link ReplyPushLoop#ticketCollected}. Without telling the push loop about a prune
|
||||
* too, its own {@code pendingTickets} entry would outlive the ticket it names: an unpolled ticket
|
||||
* (or one the reminder cap already gave up on) is pruned here but never collected there, so it
|
||||
* lingers in {@code pendingTickets} forever and rides along on every later nudge to the same lead
|
||||
* — naming a ticket {@code bridge_poll} can no longer find (CB-588 follow-up).
|
||||
*/
|
||||
private void pruneTerminalTickets() {
|
||||
long cutoff = System.nanoTime() - TICKET_TTL_NANOS;
|
||||
tasks.values().removeIf(t -> t.future().isDone() && t.createdNanos() < cutoff);
|
||||
long cutoff = nowNanos.getAsLong() - TICKET_TTL_NANOS;
|
||||
tasks.entrySet().removeIf(e -> {
|
||||
Task t = e.getValue();
|
||||
boolean expired = t.future.isDone() && t.createdNanos < cutoff;
|
||||
if (expired && pushLoop != null) {
|
||||
pushLoop.ticketCollected(e.getKey());
|
||||
}
|
||||
return expired;
|
||||
});
|
||||
}
|
||||
|
||||
/** Record the active question for an async ticket; blocking sends have no entry and stay unchanged. */
|
||||
private Task markAsyncQuestion(CompletableFuture<Rendezvous.Resolution> waiter, String text, String turnId) {
|
||||
Task task = waiter == null ? null : asyncTasksByWaiter.get(waiter);
|
||||
if (task != null) {
|
||||
task.question = new Reply(Outcome.QUESTION, text, turnId);
|
||||
task.turnId = turnId;
|
||||
asyncTasksByTurn.put(turnId, task);
|
||||
}
|
||||
return task;
|
||||
}
|
||||
|
||||
/** Clear an answered or lapsed question, but only when it matches the ticket's current turn. */
|
||||
private void clearAsyncQuestion(String turnId, boolean forgetTurn) {
|
||||
// CB-582: tell the push loop first — like ticketCollected, a removal for a turnId it never
|
||||
// nudged about (or already dropped) is a harmless no-op, so this is safe to call unconditionally
|
||||
// rather than threading the guard below through it.
|
||||
if (pushLoop != null) {
|
||||
pushLoop.questionClosed(turnId);
|
||||
}
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null && turnId.equals(task.turnId)) {
|
||||
task.question = null;
|
||||
if (forgetTurn) {
|
||||
asyncTasksByTurn.remove(turnId, task);
|
||||
task.turnId = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete and detach an async ticket after its worker's actual terminal reply. */
|
||||
private void finishAsyncTask(Task task, Reply result) {
|
||||
task.future.complete(result);
|
||||
if (task.turnId != null) {
|
||||
asyncTasksByTurn.remove(task.turnId, task);
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete the async ticket correlated to a specific answered turn. */
|
||||
private void finishAsyncTask(String turnId, Reply result) {
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null) {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
}
|
||||
|
||||
/** A new send must not open a waiter while an async ticket owns this worker's paused turn. */
|
||||
private boolean hasAsyncQuestion(String target) {
|
||||
return asyncTasksByTurn.values().stream().anyMatch(task -> target.equals(task.target));
|
||||
}
|
||||
|
||||
/**
|
||||
* The question {@code workerSession} is currently paused on via {@code bridge_ask}, if any
|
||||
* (CB-582) — {@code bridge_status} uses this to show a pending question without the caller
|
||||
* needing the ticket. {@code null} when the session has no open async question (including a
|
||||
* session mid a <em>blocking</em> {@code bridge_ask}, which has no {@link Task} to look up — see
|
||||
* {@link PendingAsk}).
|
||||
*/
|
||||
public PendingAsk pendingAsk(String workerSession) {
|
||||
for (Task task : tasks.values()) {
|
||||
Reply q = task.question;
|
||||
if (q != null && workerSession.equals(task.target)) {
|
||||
return new PendingAsk(task.ticket, q.text(), q.turnId());
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Release the async executor. */
|
||||
@@ -366,6 +811,7 @@ public final class MessageService {
|
||||
case REPLY -> Outcome.REPLIED;
|
||||
case COMPLETION -> Outcome.COMPLETED_UNREPLIED;
|
||||
case FAILED -> Outcome.WORKER_FAILED;
|
||||
case BACKEND_EXHAUSTED -> Outcome.BACKEND_EXHAUSTED;
|
||||
case QUESTION -> Outcome.QUESTION;
|
||||
};
|
||||
}
|
||||
|
||||
@@ -33,6 +33,13 @@ public final class Rendezvous {
|
||||
COMPLETION,
|
||||
/** The worker ran the turn then wedged (CB-109); {@code text} is the failure context. */
|
||||
FAILED,
|
||||
/**
|
||||
* The turn finished without a {@code bridge_reply}, and the scrape matched the backend's
|
||||
* configured usage-limit refusal pattern (CB-578 stage A); {@code text} is the reason,
|
||||
* carrying the matched line. The pane is healthy — only the account is refusing — so this
|
||||
* is kept separate from a session simply going {@code GONE}.
|
||||
*/
|
||||
BACKEND_EXHAUSTED,
|
||||
/**
|
||||
* The worker paused mid-turn to ask the primary a question (CB-205 reverse rendezvous);
|
||||
* {@code text} is the question and {@code turnId} correlates the primary's answer back to
|
||||
@@ -74,15 +81,30 @@ public final class Rendezvous {
|
||||
/**
|
||||
* Register a waiter for {@code session} — the await side of the public {@code resolve*} methods.
|
||||
* The caller must hold that session's send lock.
|
||||
*
|
||||
* <p>Atomic fail-if-present (CB-548): if a waiter is already registered for {@code session}, an
|
||||
* {@link IllegalStateException} is thrown rather than replacing the first — so any future
|
||||
* invariant violation fails loudly instead of silently swapping the waiter another send is
|
||||
* blocked on. {@code MessageService} serializes sends per session (the send lock), so in correct
|
||||
* code a double open is impossible; this is a tripwire for the day that no longer holds.
|
||||
*/
|
||||
public CompletableFuture<Resolution> open(String session) {
|
||||
CompletableFuture<Resolution> waiter = new CompletableFuture<>();
|
||||
waiters.put(session, waiter);
|
||||
CompletableFuture<Resolution> existing = waiters.putIfAbsent(session, waiter);
|
||||
if (existing != null) {
|
||||
throw new IllegalStateException(
|
||||
"rendezvous double-open for session " + session + " — a waiter is already registered");
|
||||
}
|
||||
return waiter;
|
||||
}
|
||||
|
||||
/** Remove {@code waiter} for {@code session} (only if it is still the registered one). */
|
||||
void close(String session, CompletableFuture<Resolution> waiter) {
|
||||
/**
|
||||
* Remove {@code waiter} for {@code session}, only if it is still the registered one. The
|
||||
* symmetric complement of {@link #open}: a terminal send deregisters its waiter so the next
|
||||
* send on the session may {@link #open} a fresh one (CB-548 makes double-open an error, so a
|
||||
* successful {@code open} after a finished turn requires this close to have happened first).
|
||||
*/
|
||||
public void close(String session, CompletableFuture<Resolution> waiter) {
|
||||
waiters.remove(session, waiter);
|
||||
}
|
||||
|
||||
@@ -209,6 +231,19 @@ public final class Rendezvous {
|
||||
return waiter != null && waiter.complete(new Resolution(Kind.FAILED, reason));
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a specific captured {@code waiter} as {@link Kind#BACKEND_EXHAUSTED} (CB-578 stage A):
|
||||
* the turn finished with no {@code bridge_reply} and the scrape matched the backend's configured
|
||||
* usage-limit pattern; {@code reason} carries the matched line. Like
|
||||
* {@link #resolveCompletion(CompletableFuture, String)} it targets the exact captured send
|
||||
* (CB-116). A no-op if that waiter was already resolved — first resolution wins.
|
||||
*
|
||||
* @return {@code true} if this call resolved the waiter, {@code false} if it was null or already resolved
|
||||
*/
|
||||
public boolean resolveExhausted(CompletableFuture<Resolution> waiter, String reason) {
|
||||
return waiter != null && waiter.complete(new Resolution(Kind.BACKEND_EXHAUSTED, reason));
|
||||
}
|
||||
|
||||
private boolean complete(String session, Resolution resolution) {
|
||||
CompletableFuture<Resolution> waiter = waiters.get(session);
|
||||
return waiter != null && waiter.complete(resolution);
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Holds terminal worker→primary replies that arrive with no live send to resolve, keyed by worker
|
||||
* session (target), until the primary drains them. Soft-state in Stage 1 (in-memory, lost on restart);
|
||||
* the Stage 2 AMQP adapter implements the same contract with cross-restart durability.
|
||||
*
|
||||
* <p><strong>This interface is the port.</strong> {@link InMemoryReplyInbox} is the Stage-1 adapter;
|
||||
* an AMQP-backed adapter (Stage 2) must implement the same contract (idempotent publish, FIFO peek,
|
||||
* at-least-once ack).
|
||||
*
|
||||
* <p><strong>Ownership is explicit.</strong> A gateway {@link #own owns} the inbox for each agent it
|
||||
* spawned; only the owner consumes and drains it. {@link #publish} sends a reply to the target's
|
||||
* inbox but does <em>not</em> imply ownership or start a consumer. This separation is required by
|
||||
* CB-308 federation, where one gateway may publish to an agent owned by another gateway.
|
||||
*/
|
||||
public interface ReplyInbox {
|
||||
|
||||
/** A queued reply: an idempotency id, the worker session it came from, and the reply text. */
|
||||
record InboxMessage(String msgId, String target, String content) {}
|
||||
|
||||
/**
|
||||
* Start owning (consuming) the inbox for {@code target}. Idempotent: multiple calls for the same
|
||||
* target are no-ops. The owner is the only gateway that may {@link #peek} and {@link #ack} replies
|
||||
* for this target.
|
||||
*/
|
||||
void own(String target);
|
||||
|
||||
/**
|
||||
* Stop owning (consuming) the inbox for {@code target}. Idempotent. Any replies held locally but
|
||||
* not yet acked are dropped from the local snapshot; the underlying durable queue keeps
|
||||
* unacked messages for redelivery when the target is re-owned.
|
||||
*/
|
||||
void release(String target);
|
||||
|
||||
/**
|
||||
* Queue {@code content} from worker {@code target} under {@code msgId}. Idempotent: publishing an
|
||||
* already-present {@code msgId} for {@code target} is a no-op (dedup), so an at-least-once Stage-2
|
||||
* redelivery cannot double-queue. Publishing does <em>not</em> imply ownership and must not start a
|
||||
* consumer.
|
||||
*/
|
||||
void publish(String target, String msgId, String content);
|
||||
|
||||
/** Non-destructive snapshot of pending replies for {@code target} (FIFO), empty list if none. */
|
||||
List<InboxMessage> peek(String target);
|
||||
|
||||
/** Remove the reply {@code msgId} for {@code target} once the primary has taken it. No-op if absent. */
|
||||
void ack(String target, String msgId);
|
||||
}
|
||||
@@ -0,0 +1,636 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* A status-gated push loop that nudges a lead's own herdr pane when it has uncollected work
|
||||
* waiting: a worker reply queued with no live {@code bridge_send} to resolve it (CB-307), an
|
||||
* async delegation ticket ({@code bridge_send(wait:false)}) that reached a terminal phase
|
||||
* (CB-588), or an async ticket's worker pausing mid-turn in {@code bridge_ask} to await an answer
|
||||
* (CB-582).
|
||||
*
|
||||
* <p><strong>CB-590: one schedule per lead.</strong> All three kinds of work are triggered
|
||||
* through their own entry point — {@link #onReplyQueued(String)},
|
||||
* {@link #onTicketTerminal(String, String, boolean)}, and
|
||||
* {@link #onQuestionOpened(String, String, String, String)} — but each resolves the lead that
|
||||
* should be nudged and coalesces onto a single per-lead reminder schedule, tracked in
|
||||
* {@link #activeLeads}. Earlier this was two independent schedules (one keyed by worker target
|
||||
* for replies, one keyed by lead for tickets) that could both decide to inject into the same pane
|
||||
* in the same window — a race, not routine behaviour, but the expensive kind: it interrupts the
|
||||
* lead's live turn twice. Collapsing to one schedule per lead makes that structurally impossible:
|
||||
* at most one scheduled tick chain is ever live for a given lead (guarded by {@link #activeLeads}'
|
||||
* compare-and-set), so at most one {@code agents.send} to that lead's pane is ever in flight.
|
||||
*
|
||||
* <p>Each tick examines <em>everything</em> pending for that lead — reply targets whose inbox
|
||||
* still holds an unacked message ({@link #pendingReplies}), tickets not yet collected
|
||||
* ({@link #pendingTickets}), and open questions not yet answered or lapsed
|
||||
* ({@link #pendingQuestions}) — and sends at most one combined nudge per tick
|
||||
* ({@link #injectNudge(String, int, int, int)}). Work that arrives while the lead is busy is
|
||||
* never lost: it is re-read fresh on every tick until the lead is injectable or its own reminder
|
||||
* cap ({@link #maxReminders}) is reached — each source spends from its own budget, so one source
|
||||
* exhausting its cap does not stop nudges about the others (post-CB-590 regression fix; see
|
||||
* {@link #decide}) — whichever the durable inbox / pending set doesn't already answer via
|
||||
* {@code STOP}.
|
||||
*/
|
||||
public final class ReplyPushLoop {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ReplyPushLoop.class);
|
||||
static final String NUDGE_FORMAT = "Worker %s returned a reply — run bridge_poll(target=%s) to collect it";
|
||||
/** Coalesced form, several uncollected replies for the same lead. */
|
||||
static final String REPLIES_NUDGE_FORMAT =
|
||||
"%d workers returned replies — run bridge_poll(target=...) for each to collect them: %s";
|
||||
/** Singular form, one uncollected ticket. */
|
||||
static final String TICKET_NUDGE_FORMAT =
|
||||
"Ticket %s finished%s — run bridge_poll(ticket=%s) to collect it";
|
||||
/** Coalesced form, several uncollected tickets for the same lead. */
|
||||
static final String TICKETS_NUDGE_FORMAT =
|
||||
"%d tickets finished%s — run bridge_poll(ticket=...) for each to collect them: %s";
|
||||
/** Singular form, one worker paused mid-turn in bridge_ask (CB-582) — names the answer call directly. */
|
||||
static final String QUESTION_NUDGE_FORMAT =
|
||||
"Worker %s asked a question (ticket %s) — answer it with bridge_send(turnId=\"%s\", "
|
||||
+ "content=...) to resume its turn:\n%s";
|
||||
/** Coalesced form, several open questions for the same lead. */
|
||||
static final String QUESTIONS_NUDGE_FORMAT =
|
||||
"%d workers are paused on a question — run bridge_poll(ticket=...) for each, then answer "
|
||||
+ "with bridge_send(turnId=..., content=...): %s";
|
||||
|
||||
private final PrimaryRegistry primaryRegistry;
|
||||
private final AgentControl agents;
|
||||
private final ReplyInbox inbox;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final int maxReminders;
|
||||
private final long backoffMs;
|
||||
private final Metrics metrics; // CB-512: nullable — no registry in unit tests
|
||||
|
||||
/**
|
||||
* Worker targets with a reply queued, keyed by target. Each entry carries its own nudge
|
||||
* count (CB-598) rather than sharing one counter per lead per source: a target's count only
|
||||
* ever reflects nudges that actually named that target, so a target that joins while the
|
||||
* schedule is already deep into another target's reminders still reads as fresh.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, ReplyEntry> pendingReplies = new ConcurrentHashMap<>();
|
||||
/** Tickets that have gone terminal but not yet been polled, keyed by ticket. */
|
||||
private final ConcurrentHashMap<String, PendingTicket> pendingTickets = new ConcurrentHashMap<>();
|
||||
/**
|
||||
* Open {@code bridge_ask} questions not yet answered or lapsed, keyed by {@code turnId}
|
||||
* (CB-582). A question's own nudge count is tracked the same per-item way as
|
||||
* {@link #pendingTickets} (CB-598): a fresh question keeps its source eligible regardless of
|
||||
* how depleted an older, still-open question's count is.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, PendingQuestion> pendingQuestions = new ConcurrentHashMap<>();
|
||||
/** CB-590: leads with an active combined reminder schedule (replies and/or tickets and/or questions). */
|
||||
private final ConcurrentHashMap<String, Boolean> activeLeads = new ConcurrentHashMap<>();
|
||||
|
||||
public ReplyPushLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
ScheduledExecutorService scheduler,
|
||||
int maxReminders, long backoffMs) {
|
||||
this(primaryRegistry, agents, inbox, scheduler, maxReminders, backoffMs, null);
|
||||
}
|
||||
|
||||
/** As above, with a metric registry (CB-512) so push outcomes are counted. */
|
||||
public ReplyPushLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
ScheduledExecutorService scheduler,
|
||||
int maxReminders, long backoffMs, Metrics metrics) {
|
||||
this.primaryRegistry = primaryRegistry;
|
||||
this.agents = agents;
|
||||
this.inbox = inbox;
|
||||
this.scheduler = scheduler;
|
||||
this.maxReminders = maxReminders;
|
||||
this.backoffMs = backoffMs;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Count one nudge outcome when a registry is wired; a no-op in unit tests. */
|
||||
private void countNudge(String outcome) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(BridgedMetrics.PUSH_NUDGES, "outcome", outcome);
|
||||
}
|
||||
}
|
||||
|
||||
// --- pending-work lookups (package-private for unit-testing) -------------------------------
|
||||
|
||||
/** The action the loop should take for a lead at the given reminder count. */
|
||||
enum Action { INJECT, WAIT_BUSY, STOP }
|
||||
|
||||
/**
|
||||
* Reply targets still pending for {@code lead} — registered via {@link #onReplyQueued} and
|
||||
* whose inbox still holds an unacked message. A target whose inbox has since drained (acked,
|
||||
* or collected via a live {@code bridge_send} rendezvous instead) is dropped from
|
||||
* {@link #pendingReplies} here rather than lingering forever; there is no explicit "reply
|
||||
* collected" callback the way {@link #ticketCollected} exists for tickets, so the inbox itself
|
||||
* is the only signal.
|
||||
*/
|
||||
private Set<String> pendingReplyTargetsFor(String lead) {
|
||||
Set<String> result = new HashSet<>();
|
||||
for (var entry : pendingReplies.entrySet()) {
|
||||
String target = entry.getKey();
|
||||
ReplyEntry owning = entry.getValue();
|
||||
if (!lead.equals(owning.lead())) continue;
|
||||
if (inbox.peek(target).isEmpty()) {
|
||||
pendingReplies.remove(target, owning);
|
||||
continue;
|
||||
}
|
||||
result.add(target);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/** A pending reply target: which lead to nudge, and how many nudges have named it so far. */
|
||||
private record ReplyEntry(String lead, int nudgeCount) {
|
||||
}
|
||||
|
||||
/**
|
||||
* A ticket awaiting collection: which lead to nudge, whether it ended in failure, and how
|
||||
* many nudges have named it so far (CB-598 — tracked per ticket, not per lead per source).
|
||||
*/
|
||||
private record PendingTicket(String ticket, String lead, boolean failed, int nudgeCount) {
|
||||
}
|
||||
|
||||
/** Tickets still pending for {@code lead}, snapshotted fresh for one tick. */
|
||||
private List<PendingTicket> pendingTicketsFor(String lead) {
|
||||
return pendingTickets.values().stream().filter(t -> lead.equals(t.lead())).toList();
|
||||
}
|
||||
|
||||
/** Ticket ids still pending for {@code lead} — a plain snapshot for race comparison. */
|
||||
private Set<String> pendingTicketIdsFor(String lead) {
|
||||
return pendingTicketsFor(lead).stream().map(PendingTicket::ticket)
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* An open question awaiting the lead's answer: which ticket it belongs to, which worker asked,
|
||||
* which lead to nudge, the question text, and how many nudges have named it so far (CB-598 —
|
||||
* tracked per question, not per lead per source).
|
||||
*/
|
||||
private record PendingQuestion(String turnId, String ticket, String target, String lead,
|
||||
String question, int nudgeCount) {
|
||||
}
|
||||
|
||||
/** Questions still open for {@code lead}, snapshotted fresh for one tick. */
|
||||
private List<PendingQuestion> pendingQuestionsFor(String lead) {
|
||||
return pendingQuestions.values().stream().filter(q -> lead.equals(q.lead())).toList();
|
||||
}
|
||||
|
||||
/** Question turnIds still open for {@code lead} — a plain snapshot for race comparison. */
|
||||
private Set<String> pendingQuestionTurnIdsFor(String lead) {
|
||||
return pendingQuestionsFor(lead).stream().map(PendingQuestion::turnId)
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* The reply-source reminder count {@link #decide} should see for {@code lead} on this tick:
|
||||
* the <em>minimum</em> nudge count among the reply targets currently pending for it (CB-598).
|
||||
*
|
||||
* <p>Before this, the count passed to {@code decide} was a single counter carried forward
|
||||
* across scheduled ticks ({@code scheduleNext(lead, count + 1, ...)}), incremented whenever
|
||||
* the source had <em>any</em> pending work — not tied to which target that work was. A target
|
||||
* that joined while an older target's count was already near the cap inherited that count on
|
||||
* its very next tick, even though no nudge had ever named it. Taking the minimum over what is
|
||||
* actually pending now means a fresh target (count 0) keeps the source eligible regardless of
|
||||
* how many times an older, still-undrained target has already been nudged; that older target
|
||||
* keeps riding along in the combined nudge text without spending any more of its own budget
|
||||
* (see {@link #bumpNudgeCounts}). Returns 0 when nothing is pending — {@link #decide} never
|
||||
* consults the count in that case, since {@code hasReplyWork} is false.
|
||||
*/
|
||||
private int minReplyNudgeCountFor(String lead) {
|
||||
int min = Integer.MAX_VALUE;
|
||||
for (String target : pendingReplyTargetsFor(lead)) {
|
||||
ReplyEntry entry = pendingReplies.get(target);
|
||||
if (entry != null) {
|
||||
min = Math.min(min, entry.nudgeCount());
|
||||
}
|
||||
}
|
||||
return min == Integer.MAX_VALUE ? 0 : min;
|
||||
}
|
||||
|
||||
/** As {@link #minReplyNudgeCountFor}, for the ticket source. */
|
||||
private int minTicketNudgeCountFor(String lead) {
|
||||
int min = Integer.MAX_VALUE;
|
||||
for (PendingTicket ticket : pendingTicketsFor(lead)) {
|
||||
min = Math.min(min, ticket.nudgeCount());
|
||||
}
|
||||
return min == Integer.MAX_VALUE ? 0 : min;
|
||||
}
|
||||
|
||||
/** As {@link #minReplyNudgeCountFor}, for the question source (CB-582). */
|
||||
private int minQuestionNudgeCountFor(String lead) {
|
||||
int min = Integer.MAX_VALUE;
|
||||
for (PendingQuestion q : pendingQuestionsFor(lead)) {
|
||||
min = Math.min(min, q.nudgeCount());
|
||||
}
|
||||
return min == Integer.MAX_VALUE ? 0 : min;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure decision function: examine everything pending for {@code lead} — reply targets and
|
||||
* tickets alike — and return what the loop should do.
|
||||
*
|
||||
* <p><strong>CB-590-fix: one schedule, two budgets.</strong> The single per-lead schedule
|
||||
* (CB-590) still ticks once for both sources, but each source is capped independently —
|
||||
* {@code replyReminderCount} against a reply target still pending, {@code ticketReminderCount}
|
||||
* against a ticket still pending. A busy reply stream that exhausts its own cap must not stop
|
||||
* the loop from nudging about a ticket that still has budget left, and vice versa: either
|
||||
* source being eligible (has pending work AND is under its own cap) is enough for
|
||||
* {@link Action#INJECT}. Only when neither source has eligible work does the loop
|
||||
* {@link Action#STOP}.
|
||||
*
|
||||
* <p><strong>CB-598: the counts are per-item, not per-tick.</strong> {@link #tick} no longer
|
||||
* carries these counts forward across scheduled calls — it recomputes them fresh every tick via
|
||||
* {@link #minReplyNudgeCountFor} / {@link #minTicketNudgeCountFor}, so this function itself did
|
||||
* not need to change; only what its caller feeds it did.
|
||||
*
|
||||
* @param lead the lead terminal to nudge
|
||||
* @param replyReminderCount the lowest nudge count among reply targets pending for this lead
|
||||
* @param ticketReminderCount the lowest nudge count among tickets pending for this lead
|
||||
* @return the action the caller should take
|
||||
*/
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount) {
|
||||
return decide(lead, replyReminderCount, ticketReminderCount, minQuestionNudgeCountFor(lead));
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #decide(String, int, int)}, with the question source (CB-582) folded in on the
|
||||
* same footing as replies and tickets: its own eligibility (has open questions AND under its
|
||||
* own {@link #maxReminders} budget) is enough on its own to {@link Action#INJECT}, exactly like
|
||||
* the other two.
|
||||
*
|
||||
* @param questionReminderCount the lowest nudge count among questions open for this lead
|
||||
*/
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount) {
|
||||
boolean hasReplyWork = !pendingReplyTargetsFor(lead).isEmpty();
|
||||
boolean hasTicketWork = !pendingTicketIdsFor(lead).isEmpty();
|
||||
boolean hasQuestionWork = !pendingQuestionTurnIdsFor(lead).isEmpty();
|
||||
if (!hasReplyWork && !hasTicketWork && !hasQuestionWork) {
|
||||
log.debug("push: nothing pending for lead {}, stopping reminder", lead);
|
||||
return Action.STOP;
|
||||
}
|
||||
boolean replyEligible = hasReplyWork && replyReminderCount < maxReminders;
|
||||
boolean ticketEligible = hasTicketWork && ticketReminderCount < maxReminders;
|
||||
boolean questionEligible = hasQuestionWork && questionReminderCount < maxReminders;
|
||||
if (!replyEligible && !ticketEligible && !questionEligible) {
|
||||
log.debug("push: reminder cap ({}) reached for lead {} on every source with pending work, stopping",
|
||||
maxReminders, lead);
|
||||
countNudge("exhausted");
|
||||
return Action.STOP;
|
||||
}
|
||||
AgentStatus status;
|
||||
try {
|
||||
status = agents.status(lead);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("push: status check failed for lead {}, will retry", lead, e);
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
if (status.injectable()) {
|
||||
return Action.INJECT;
|
||||
}
|
||||
log.debug("push: lead {} is {} (not injectable), waiting", lead, status);
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
|
||||
// --- public entrypoints ----------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Called when a reply is queued for {@code target}. Resolves the lead delegating to
|
||||
* {@code target} (CB-532) and coalesces onto that lead's single reminder schedule — starting
|
||||
* one if none is active, joining an already-active one otherwise. A no-op if no lead is known
|
||||
* to be waiting on {@code target}: there is nobody to nudge yet, and the durable inbox is the
|
||||
* backstop until a lead is recorded.
|
||||
*/
|
||||
public void onReplyQueued(String target) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on {}, skipping reminder", target);
|
||||
return;
|
||||
}
|
||||
pendingReplies.compute(target, (t, existing) ->
|
||||
new ReplyEntry(lead.get(), existing == null ? 0 : existing.nudgeCount()));
|
||||
startOrCoalesce(lead.get());
|
||||
}
|
||||
|
||||
/**
|
||||
* Called when an async delegation ticket ({@code bridge_send(wait:false)}, CB-107) reaches a
|
||||
* terminal phase — DONE or a failure. Unlike {@link #onReplyQueued}, which nudges about the
|
||||
* durable-inbox no-waiter path, this covers the path {@code MessageService.reply} takes when a
|
||||
* fire-and-poll send's own rendezvous waiter resolves the reply directly: that path returns
|
||||
* before {@link #onReplyQueued} is ever called, so without this entry point a ticket finishing
|
||||
* that way never nudged anyone (CB-588 / gitea #72).
|
||||
*
|
||||
* <p>Resolves the delegating lead the same way {@link #onReplyQueued} does and coalesces onto
|
||||
* the same per-lead schedule (CB-590) — several tickets, or a ticket and a reply, finishing
|
||||
* for the same lead while its schedule is already active all ride the existing schedule's next
|
||||
* tick rather than firing a nudge each.
|
||||
*
|
||||
* @param ticket the ticket to nudge about
|
||||
* @param target the worker session the ticket was sent to — resolves which lead delegated it
|
||||
* @param failed whether the ticket ended in a failure phase rather than {@code DONE}
|
||||
*/
|
||||
public void onTicketTerminal(String ticket, String target, boolean failed) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on ticket {} (target {}), skipping nudge",
|
||||
ticket, target);
|
||||
return;
|
||||
}
|
||||
pendingTickets.compute(ticket, (id, existing) ->
|
||||
new PendingTicket(ticket, lead.get(), failed, existing == null ? 0 : existing.nudgeCount()));
|
||||
startOrCoalesce(lead.get());
|
||||
}
|
||||
|
||||
/**
|
||||
* Called when a ticket's terminal state has been collected via {@code bridge_poll}. Removes it
|
||||
* from the pending set so a scheduled tick — and any nudge it sends — never names a ticket the
|
||||
* lead already has. A ticket that was never pending (unknown ticket, or one nudged with no push
|
||||
* loop configured) is a no-op.
|
||||
*/
|
||||
public void ticketCollected(String ticket) {
|
||||
pendingTickets.remove(ticket);
|
||||
}
|
||||
|
||||
/**
|
||||
* Called when an async ticket's worker pauses mid-turn in {@code bridge_ask} (CB-582): the
|
||||
* question is now visible via {@code bridge_poll} (Phase.ASKING), but the reverse-rendezvous
|
||||
* window it opened with (~55s default, see {@code BridgeMcp}/{@code BridgedApp}) is far shorter
|
||||
* than a lead's normal minutes-long poll cadence — exactly the gap this closes. Resolves the
|
||||
* delegating lead the same way {@link #onTicketTerminal} does and coalesces onto the same
|
||||
* per-lead schedule (CB-590).
|
||||
*
|
||||
* @param ticket the async ticket the question belongs to (for {@code bridge_poll})
|
||||
* @param target the worker session that asked
|
||||
* @param turnId correlation id the lead answers with ({@code bridge_send turnId=...})
|
||||
* @param question the question text
|
||||
*/
|
||||
public void onQuestionOpened(String ticket, String target, String turnId, String question) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on {}'s question (turnId {}), skipping nudge",
|
||||
target, turnId);
|
||||
return;
|
||||
}
|
||||
pendingQuestions.put(turnId, new PendingQuestion(turnId, ticket, target, lead.get(), question, 0));
|
||||
startOrCoalesce(lead.get());
|
||||
}
|
||||
|
||||
/**
|
||||
* Called when a worker's {@code bridge_ask} resolves — answered or lapsed unanswered — so a
|
||||
* scheduled tick never nudges about a question the lead already handled. A {@code turnId} that
|
||||
* was never pending (never nudged, or already closed) is a no-op.
|
||||
*/
|
||||
public void questionClosed(String turnId) {
|
||||
pendingQuestions.remove(turnId);
|
||||
}
|
||||
|
||||
// --- the schedule ----------------------------------------------------------------------------
|
||||
|
||||
/** Start a reminder schedule for {@code lead}, or join the one already running. */
|
||||
private void startOrCoalesce(String lead) {
|
||||
if (activeLeads.putIfAbsent(lead, Boolean.TRUE) != null) {
|
||||
log.debug("push: reminder loop already active for lead {}, work coalesced in", lead);
|
||||
return;
|
||||
}
|
||||
log.debug("push: starting reminder loop for lead {}", lead);
|
||||
scheduleNext(lead);
|
||||
}
|
||||
|
||||
/**
|
||||
* Execute one loop tick — called on the scheduler thread (or directly by a test; package-private
|
||||
* for the same reason as {@link #stopOrRestart}).
|
||||
*
|
||||
* <p><strong>CB-598.</strong> The reminder counts fed into {@link #decide} are recomputed fresh
|
||||
* every tick from what is actually pending right now ({@link #minReplyNudgeCountFor} /
|
||||
* {@link #minTicketNudgeCountFor}), rather than carried forward as running counters across
|
||||
* scheduled calls. A counter carried forward has no memory of which item it was counting for:
|
||||
* a target or ticket that joined mid-backoff — after the previous tick fired but before this one
|
||||
* did — is already sitting in {@code repliesBefore} / {@code ticketsBefore} below by the time this
|
||||
* tick takes its snapshot, indistinguishable at that point from backlog the cap is meant to
|
||||
* silence. Recomputing from the per-item counts fixes that: a newly-joined item's own count is
|
||||
* still 0, so it keeps its source eligible regardless of how depleted an older, still-undrained
|
||||
* item's count is.
|
||||
*/
|
||||
void tick(String lead) {
|
||||
Set<String> repliesBefore = pendingReplyTargetsFor(lead);
|
||||
Set<String> ticketsBefore = pendingTicketIdsFor(lead);
|
||||
Set<String> questionsBefore = pendingQuestionTurnIdsFor(lead);
|
||||
int replyReminderCount = minReplyNudgeCountFor(lead);
|
||||
int ticketReminderCount = minTicketNudgeCountFor(lead);
|
||||
int questionReminderCount = minQuestionNudgeCountFor(lead);
|
||||
var action = decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount);
|
||||
switch (action) {
|
||||
case INJECT -> {
|
||||
injectNudge(lead, replyReminderCount, ticketReminderCount, questionReminderCount);
|
||||
scheduleNext(lead);
|
||||
}
|
||||
// Re-check after the configured backoff; the lead may become injectable soon.
|
||||
case WAIT_BUSY -> scheduleNext(lead);
|
||||
case STOP -> stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Release {@code lead}'s active-schedule slot, then restart it only if work landed that
|
||||
* {@code repliesBefore} / {@code ticketsBefore} — the snapshots taken just before this tick's
|
||||
* decision — did not already account for. {@link #onReplyQueued} / {@link #onTicketTerminal}
|
||||
* read {@link #activeLeads} to decide whether to coalesce onto an existing schedule or start
|
||||
* one, so work that lands between {@link #decide} returning {@link Action#STOP} and this
|
||||
* removal running sees the (soon-to-be-stale) slot as occupied, coalesces onto a schedule that
|
||||
* is about to die, and gets no nudge scheduled at all — a lost nudge, exactly what CB-588 (and
|
||||
* now CB-590) exist to remove (originally found in review, gitea PR #73, for the ticket-only
|
||||
* loop; carried forward here for the unified one).
|
||||
*
|
||||
* <p>Restarting on ANY non-empty pending set would be wrong: when STOP is reached because the
|
||||
* reminder cap was hit rather than the backlog draining, the same never-collected work is
|
||||
* expected to still be sitting there — that is the cap doing its job — and restarting would
|
||||
* nudge about it forever, defeating the bound. Diffing the current pending sets against the
|
||||
* "before" snapshots tells the two cases apart: an item present before this tick's decision is
|
||||
* stale backlog, not a race; only an item absent from the "before" snapshot can only have
|
||||
* arrived during the decision-to-release window, which is exactly the race this method closes.
|
||||
*
|
||||
* <p>Package-private so a test can drive the interleaving directly rather than trying to force a
|
||||
* genuine thread race.
|
||||
*
|
||||
* <p>Terminates rather than spinning: this method restarts the schedule at most once per call,
|
||||
* and a fresh {@link #onReplyQueued} / {@link #onTicketTerminal} racing the recheck below still
|
||||
* terminates in one of two ways — either it observes the slot already vacated (by the
|
||||
* {@code activeLeads.remove} above, which happens-before this recheck in program order) and
|
||||
* claims it itself, or it lands first and this recheck then observes its work in
|
||||
* {@link #pendingReplies} / {@link #pendingTickets} and reclaims the slot instead. Exactly one
|
||||
* side always wins; neither can miss the other, so this never loops on its own account.
|
||||
*/
|
||||
void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore) {
|
||||
stopOrRestart(lead, repliesBefore, ticketsBefore, pendingQuestionTurnIdsFor(lead));
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #stopOrRestart(String, Set, Set)}, with the question source's (CB-582) own "before"
|
||||
* snapshot folded into the same race check: a question that raced in during the
|
||||
* decision-to-release window reclaims the schedule slot exactly like a raced-in reply or ticket.
|
||||
*/
|
||||
void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
|
||||
Set<String> questionsBefore) {
|
||||
activeLeads.remove(lead);
|
||||
boolean racedIn = pendingReplyTargetsFor(lead).stream().anyMatch(t -> !repliesBefore.contains(t))
|
||||
|| pendingTicketIdsFor(lead).stream().anyMatch(t -> !ticketsBefore.contains(t))
|
||||
|| pendingQuestionTurnIdsFor(lead).stream().anyMatch(t -> !questionsBefore.contains(t));
|
||||
if (racedIn && activeLeads.putIfAbsent(lead, Boolean.TRUE) == null) {
|
||||
log.debug("push: new work for lead {} raced the reminder loop's stop — restarting", lead);
|
||||
scheduleNext(lead);
|
||||
return;
|
||||
}
|
||||
log.debug("push: reminder loop ended for lead {}", lead);
|
||||
}
|
||||
|
||||
/** Send one combined nudge covering everything currently pending for {@code lead}. */
|
||||
private void injectNudge(String lead, int replyReminderCount, int ticketReminderCount,
|
||||
int questionReminderCount) {
|
||||
// Re-read rather than threading it down from decide(): a reply can drain, a ticket be
|
||||
// collected, or a question be answered (or another arrive), between the decision and the
|
||||
// injection.
|
||||
Set<String> replyTargets = pendingReplyTargetsFor(lead);
|
||||
List<PendingTicket> tickets = pendingTicketsFor(lead);
|
||||
List<PendingQuestion> questions = pendingQuestionsFor(lead);
|
||||
if (replyTargets.isEmpty() && tickets.isEmpty() && questions.isEmpty()) {
|
||||
log.debug("push: pending work for lead {} drained before the nudge could be sent", lead);
|
||||
return;
|
||||
}
|
||||
String nudge = formatNudge(replyTargets, tickets, questions);
|
||||
try {
|
||||
agents.send(lead, nudge);
|
||||
log.debug("push: nudge sent to lead {} (reply {}/{}, ticket {}/{}, question {}/{}; "
|
||||
+ "{} reply target(s), {} ticket(s), {} question(s))",
|
||||
lead, replyReminderCount + 1, maxReminders, ticketReminderCount + 1, maxReminders,
|
||||
questionReminderCount + 1, maxReminders,
|
||||
replyTargets.size(), tickets.size(), questions.size());
|
||||
countNudge("delivered");
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("push: failed to nudge lead {} (reply {}/{}, ticket {}/{}, question {}/{}): {}",
|
||||
lead, replyReminderCount + 1, maxReminders, ticketReminderCount + 1, maxReminders,
|
||||
questionReminderCount + 1, maxReminders, e.toString());
|
||||
}
|
||||
// Bump every item actually named in this nudge, not just whatever the shared source-level
|
||||
// eligibility used to gate (CB-598) — each item's own count is what the next tick's
|
||||
// minReplyNudgeCountFor / minTicketNudgeCountFor / minQuestionNudgeCountFor will read. An
|
||||
// item already at or over the cap keeps riding along in the text (still pending, still
|
||||
// named) but its extra bumps here are inert: decide() already treats it as ineligible once
|
||||
// its count reaches maxReminders.
|
||||
bumpNudgeCounts(replyTargets, tickets, questions);
|
||||
}
|
||||
|
||||
/** Record that every one of these items was just named in a sent (or attempted) nudge. */
|
||||
private void bumpNudgeCounts(Set<String> replyTargets, List<PendingTicket> tickets,
|
||||
List<PendingQuestion> questions) {
|
||||
for (String target : replyTargets) {
|
||||
pendingReplies.computeIfPresent(target, (t, e) -> new ReplyEntry(e.lead(), e.nudgeCount() + 1));
|
||||
}
|
||||
for (PendingTicket ticket : tickets) {
|
||||
pendingTickets.computeIfPresent(ticket.ticket(),
|
||||
(id, e) -> new PendingTicket(e.ticket(), e.lead(), e.failed(), e.nudgeCount() + 1));
|
||||
}
|
||||
for (PendingQuestion question : questions) {
|
||||
pendingQuestions.computeIfPresent(question.turnId(), (id, e) ->
|
||||
new PendingQuestion(e.turnId(), e.ticket(), e.target(), e.lead(), e.question(),
|
||||
e.nudgeCount() + 1));
|
||||
}
|
||||
}
|
||||
|
||||
/** Schedule the next tick on the scheduler thread pool. */
|
||||
private void scheduleNext(String lead) {
|
||||
scheduler.schedule(() -> tick(lead),
|
||||
backoffMs, TimeUnit.MILLISECONDS);
|
||||
}
|
||||
|
||||
// --- nudge formatting ------------------------------------------------------------------------
|
||||
|
||||
/** Render everything pending for one lead as a single nudge line. */
|
||||
private static String formatNudge(Set<String> replyTargets, List<PendingTicket> tickets,
|
||||
List<PendingQuestion> questions) {
|
||||
List<String> parts = new ArrayList<>();
|
||||
if (!replyTargets.isEmpty()) {
|
||||
parts.add(formatRepliesNudge(replyTargets));
|
||||
}
|
||||
if (!tickets.isEmpty()) {
|
||||
parts.add(formatTicketsNudge(tickets));
|
||||
}
|
||||
if (!questions.isEmpty()) {
|
||||
parts.add(formatQuestionsNudge(questions));
|
||||
}
|
||||
return String.join(" | ", parts);
|
||||
}
|
||||
|
||||
/** Render one or several pending reply targets. */
|
||||
private static String formatRepliesNudge(Set<String> targets) {
|
||||
if (targets.size() == 1) {
|
||||
String target = targets.iterator().next();
|
||||
return NUDGE_FORMAT.formatted(target, target);
|
||||
}
|
||||
String ids = String.join(", ", targets);
|
||||
return REPLIES_NUDGE_FORMAT.formatted(targets.size(), ids);
|
||||
}
|
||||
|
||||
/** Render one or several pending tickets. */
|
||||
private static String formatTicketsNudge(List<PendingTicket> pending) {
|
||||
if (pending.size() == 1) {
|
||||
PendingTicket t = pending.get(0);
|
||||
return TICKET_NUDGE_FORMAT.formatted(t.ticket(), t.failed() ? " (FAILED)" : "", t.ticket());
|
||||
}
|
||||
long failedCount = pending.stream().filter(PendingTicket::failed).count();
|
||||
String ids = pending.stream()
|
||||
.map(t -> t.failed() ? t.ticket() + " (FAILED)" : t.ticket())
|
||||
.collect(Collectors.joining(", "));
|
||||
String failedNote = failedCount > 0 ? " (%d failed)".formatted(failedCount) : "";
|
||||
return TICKETS_NUDGE_FORMAT.formatted(pending.size(), failedNote, ids);
|
||||
}
|
||||
|
||||
/** Render one or several open questions (CB-582). */
|
||||
private static String formatQuestionsNudge(List<PendingQuestion> pending) {
|
||||
if (pending.size() == 1) {
|
||||
PendingQuestion q = pending.get(0);
|
||||
return QUESTION_NUDGE_FORMAT.formatted(q.target(), q.ticket(), q.turnId(), q.question());
|
||||
}
|
||||
String ids = pending.stream()
|
||||
.map(q -> q.ticket() + " (turnId=" + q.turnId() + ")")
|
||||
.collect(Collectors.joining(", "));
|
||||
return QUESTIONS_NUDGE_FORMAT.formatted(pending.size(), ids);
|
||||
}
|
||||
|
||||
// --- lifecycle -----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Whether any reminder loop is currently active for some lead (CB-551). The idle-lead heartbeat
|
||||
* uses this to stand aside: while the push loop is actively nudging a lead, a concurrent
|
||||
* heartbeat injection would start a second competing turn in the same pane — racing loops
|
||||
* multiply turns and context burn. "Active" means a schedule exists in {@link #activeLeads},
|
||||
* which now covers reply-queued (CB-307), ticket-terminal (CB-588), and question-open (CB-582)
|
||||
* work (CB-590) — bounded by what has been triggered, not by any persistent state.
|
||||
*/
|
||||
public boolean isActive() {
|
||||
return !activeLeads.isEmpty();
|
||||
}
|
||||
|
||||
/** Shut down the scheduler. Outstanding reminders are cancelled. */
|
||||
public void stop() {
|
||||
scheduler.shutdownNow();
|
||||
activeLeads.clear();
|
||||
pendingReplies.clear();
|
||||
pendingTickets.clear();
|
||||
pendingQuestions.clear();
|
||||
}
|
||||
|
||||
/** @see #stop() */
|
||||
public void close() {
|
||||
stop();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
|
||||
/**
|
||||
* Identity for one accepted send. The session turn is deliberately absent: CompletionResolver's
|
||||
* delivery callback runs before SessionManager.onDelivered, so binding it needs a later ordering design.
|
||||
*/
|
||||
public final class TurnToken {
|
||||
private final String target;
|
||||
private final CompletableFuture<Rendezvous.Resolution> waiter;
|
||||
|
||||
public TurnToken(String target, CompletableFuture<Rendezvous.Resolution> waiter) {
|
||||
this.target = target;
|
||||
this.waiter = waiter;
|
||||
}
|
||||
|
||||
public String target() { return target; }
|
||||
public CompletableFuture<Rendezvous.Resolution> waiter() { return waiter; }
|
||||
}
|
||||
@@ -26,10 +26,33 @@ public enum Capability {
|
||||
*/
|
||||
WORKTREE,
|
||||
|
||||
/**
|
||||
* The peer can discard its current conversation context without starting a delegated bridge
|
||||
* turn. The command or API used to do that is adapter-specific.
|
||||
*/
|
||||
CONTEXT_RESET,
|
||||
|
||||
/**
|
||||
* The spawner can reconcile orphaned peers on boot — workers that outlived a prior daemon
|
||||
* process and whose pane ids died with it (CB-117). Claude Code over herdr supports this
|
||||
* via name-based matching against the herdr agent list.
|
||||
*/
|
||||
ORPHAN_REAP
|
||||
ORPHAN_REAP,
|
||||
|
||||
/**
|
||||
* The peer surfaces the bridge's logical name ({@link SpawnRequest#sessionName()}) in its own
|
||||
* UI at spawn — e.g. Claude Code's {@code -n} display name, which shows in the prompt box, the
|
||||
* {@code /resume} picker, and the terminal title. This is what lets an operator tell the
|
||||
* bridge's worker apart from a user's own session in the same terminal after a restart.
|
||||
*/
|
||||
SESSION_NAME,
|
||||
|
||||
/**
|
||||
* The peer can be relaunched onto a prior conversation via that conversation's own session id
|
||||
* ({@link SpawnRequest#resumeSessionId()}) — e.g. Claude Code's {@code -r}, which adopts the id
|
||||
* as the agent's own identity rather than starting a new conversation. A launcher that declares
|
||||
* this mints/resolves that id at spawn and exposes it on the returned
|
||||
* {@link PeerHandle#agentSessionId()}, so a later resume addresses the same conversation.
|
||||
*/
|
||||
SESSION_RESUME
|
||||
}
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
package dev.ltms.bridged.peer;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.security.MessageDigest;
|
||||
import java.security.NoSuchAlgorithmException;
|
||||
import java.util.HexFormat;
|
||||
|
||||
/**
|
||||
* CB-571: a fingerprint of the exact charter bytes handed to a spawned member.
|
||||
*
|
||||
* <p>Lets an operator prove <em>which</em> charter a member actually got, without ever logging the
|
||||
* charter text. The digest covers the exact composed UTF-8 string {@code HerdrPeerLauncher} passes
|
||||
* to its adapter as {@code LaunchSpec.charter()}, so every adapter that receives the same string —
|
||||
* Claude inlining it, OpenCode writing it to a file — produces the same digest for the same config.
|
||||
* Two spawns of the same role from the same config agree; editing the charter changes the digest.
|
||||
*
|
||||
* <p>Deliberately places no charter prose. A charter is operator-authored text that may name
|
||||
* internal projects or unreleased plans, and logs get tailed, shipped, and pasted into tickets.
|
||||
* The {@code charterSource} key is what the operator wants to confirm, and it carries no content.
|
||||
*/
|
||||
public record CharterReceipt(
|
||||
MemberRole role,
|
||||
String profile,
|
||||
String charterSource,
|
||||
String charterSha256,
|
||||
int charterBytes) {
|
||||
|
||||
/** Source reported when the role has no configured charter, so the field is never omitted. */
|
||||
public static final String NO_SOURCE = "none";
|
||||
|
||||
/**
|
||||
* The config key that supplied the role's charter text, e.g. {@code fleet.charters.architect}.
|
||||
*/
|
||||
public static String sourceKey(MemberRole role) {
|
||||
return "fleet.charters." + (role == null ? "?" : role.wireName());
|
||||
}
|
||||
|
||||
/**
|
||||
* Fingerprint the composed charter for {@code role} on {@code profile}. {@code configured} is
|
||||
* the role's charter text as read from config ({@code null} when none is configured);
|
||||
* {@code composed} is the exact string the launcher will pass to the adapter — the reply
|
||||
* charter may be appended to {@code configured}, or stand alone when no role charter exists.
|
||||
*
|
||||
* <p>No composed charter at all is reported as an explicit absence — a {@code null} digest and
|
||||
* a zero byte count — never a digest of the empty string, which would hide the fact that no
|
||||
* text was supplied. {@code configured} being {@code null} while {@code composed} is the reply
|
||||
* charter alone is a normal case, and the source says so.
|
||||
*/
|
||||
public static CharterReceipt compose(MemberRole role, String profile,
|
||||
String configured, String composed) {
|
||||
String source = (configured == null || configured.isBlank())
|
||||
? NO_SOURCE : sourceKey(role);
|
||||
if (composed == null) {
|
||||
return new CharterReceipt(role, profile, source, null, 0);
|
||||
}
|
||||
byte[] bytes = composed.getBytes(StandardCharsets.UTF_8);
|
||||
return new CharterReceipt(role, profile, source, digestOf(composed), bytes.length);
|
||||
}
|
||||
|
||||
/** Whether the composed charter was absent (no text was given to the member). */
|
||||
public boolean absent() {
|
||||
return charterSha256 == null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The stable SHA-256 hex digest of {@code text}, or {@code null} for null/blank text. Used both
|
||||
* for the receipt's fingerprint and to redact a charter argument in a spawn log.
|
||||
*/
|
||||
public static String digestOf(String text) {
|
||||
if (text == null || text.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
return sha256Hex(text.getBytes(StandardCharsets.UTF_8));
|
||||
}
|
||||
|
||||
private static String sha256Hex(byte[] bytes) {
|
||||
try {
|
||||
MessageDigest md = MessageDigest.getInstance("SHA-256");
|
||||
return HexFormat.of().formatHex(md.digest(bytes));
|
||||
} catch (NoSuchAlgorithmException e) {
|
||||
throw new IllegalStateException("SHA-256 is unavailable", e);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
package dev.ltms.bridged.peer;
|
||||
|
||||
import java.util.Locale;
|
||||
|
||||
/**
|
||||
* What a member is <em>for</em> — the contract it runs under.
|
||||
*
|
||||
* <p>A member is anything a lead spawns. Every member carries two independent attributes:
|
||||
*
|
||||
* <ul>
|
||||
* <li><b>role</b> (this enum) — <em>which contract</em>: the launch charter it is given, the role
|
||||
* file it reads, the playbook skill it loads, and its authorization row.</li>
|
||||
* <li><b>profile</b> (a {@code profiles:} key) — <em>which backend</em>: model, CLI adapter,
|
||||
* credentials, cost.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>These are separate axes on purpose. A {@code REVIEWER} may run on the same profile as the
|
||||
* {@code DEV} whose diff it reviews — same backend, different contract. That case is what proves
|
||||
* role and profile cannot be collapsed into one field.
|
||||
*
|
||||
* <p>A lead is deliberately <em>not</em> a role here. A lead is not spawned: it is a pre-existing,
|
||||
* human-facing session that config recognises. Only spawned peers have a member role.
|
||||
*/
|
||||
public enum MemberRole {
|
||||
|
||||
/**
|
||||
* Refines a ticket before anyone implements it: scope, acceptance criteria, risks, unit split.
|
||||
*
|
||||
* <p>Reads the repo and writes analysis. Never commits code and never opens a pull request —
|
||||
* an architect that starts implementing has stopped doing the job that makes it useful.
|
||||
*
|
||||
* <p>Architects are the one member kind declared in config, because a lead addresses the same
|
||||
* slots across many tickets and needs a stable name for them.
|
||||
*/
|
||||
ARCHITECT,
|
||||
|
||||
/**
|
||||
* Implements one unit of work: provisions a worktree, writes the code, commits, pushes, and
|
||||
* opens its own pull request.
|
||||
*
|
||||
* <p>Never merges. The lead is the gate, and a member that merged its own work would remove the
|
||||
* only independent check in the flow.
|
||||
*/
|
||||
DEV,
|
||||
|
||||
/**
|
||||
* Reviews a diff it did not write and reports one structured finding.
|
||||
*
|
||||
* <p>Never commits, never merges, and is never the member that wrote the scope under review.
|
||||
* Briefed from the diff rather than from the implementer's rationale, because that rationale
|
||||
* carries the same blind spot that produced the bug.
|
||||
*/
|
||||
REVIEWER;
|
||||
|
||||
/** The lowercase spelling used in config and on the wire ({@code architect}, {@code dev}, …). */
|
||||
public String wireName() {
|
||||
return name().toLowerCase(Locale.ROOT);
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code fleet:} block that holds this role's pool — {@code architects},
|
||||
* {@code developers}, {@code reviewers}.
|
||||
*
|
||||
* <p>Plural, and not always the wire name: the pool of things a {@code dev} may run on reads
|
||||
* naturally as {@code developers:}. The wire name stays the singular {@code dev}, because that
|
||||
* is what a tab label and a roster row say.
|
||||
*/
|
||||
public String configKey() {
|
||||
return switch (this) {
|
||||
case ARCHITECT -> "architects";
|
||||
case DEV -> "developers";
|
||||
case REVIEWER -> "reviewers";
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* The role owning the {@code fleet:} pool named {@code key}, or {@code null} when the key is not
|
||||
* a role pool ({@code leaders}, {@code tabLabel}, …). Null rather than a throw: callers use this
|
||||
* to sort a fleet block's children, where a non-pool key is normal rather than an error.
|
||||
*/
|
||||
public static MemberRole fromConfigKey(String key) {
|
||||
if (key == null) {
|
||||
return null;
|
||||
}
|
||||
String k = key.trim().toLowerCase(Locale.ROOT);
|
||||
for (MemberRole r : values()) {
|
||||
if (r.configKey().equals(k)) {
|
||||
return r;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a config/wire spelling, case-insensitively.
|
||||
*
|
||||
* @param s the spelling to parse
|
||||
* @return the matching role
|
||||
* @throws IllegalArgumentException when {@code s} is null, blank, or not a known role — the
|
||||
* message lists the valid spellings, because a typo'd role in
|
||||
* config should fail at startup with the fix in the error
|
||||
*/
|
||||
public static MemberRole parse(String s) {
|
||||
if (s != null && !s.isBlank()) {
|
||||
String t = s.trim().toLowerCase(Locale.ROOT);
|
||||
for (MemberRole r : values()) {
|
||||
if (r.wireName().equals(t)) {
|
||||
return r;
|
||||
}
|
||||
}
|
||||
}
|
||||
StringBuilder valid = new StringBuilder();
|
||||
for (MemberRole r : values()) {
|
||||
if (!valid.isEmpty()) {
|
||||
valid.append(", ");
|
||||
}
|
||||
valid.append(r.wireName());
|
||||
}
|
||||
throw new IllegalArgumentException(
|
||||
"unknown member role '" + s + "'; valid roles are: " + valid);
|
||||
}
|
||||
}
|
||||
@@ -12,9 +12,13 @@ package dev.ltms.bridged.peer;
|
||||
public interface PeerHandle {
|
||||
|
||||
/**
|
||||
* The registry/routing key — an opaque, launcher-assigned identifier. For the herdr-backed
|
||||
* launcher this is the herdr pane id; for other launchers it is whatever their transport
|
||||
* uses. Guaranteed to be non-null and unique among live peers within a single daemon process.
|
||||
* The registry/routing key — an opaque, launcher-assigned identifier (CB-519). Multiple
|
||||
* daemon processes may run on one host, so the contract is <em>host-unique</em>, not merely
|
||||
* process-unique: the herdr-backed launcher mints a fresh UUID per spawn, and a non-herdr
|
||||
* launcher is likewise expected to return an identifier that cannot collide across processes
|
||||
* on the same host. This id is the routing key and is deliberately decoupled from any launcher
|
||||
* transport coordinate (e.g. a herdr pane id), which stays launcher-private. Guaranteed to be
|
||||
* non-null and unique among live peers on the host.
|
||||
*/
|
||||
String id();
|
||||
|
||||
@@ -26,4 +30,59 @@ public interface PeerHandle {
|
||||
default String terminalId() {
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The worker profile that spawned this peer, if the launcher resolved one. A launcher that
|
||||
* performs dynamic profile selection (e.g. CB-518 weighted placement) sets this so the
|
||||
* session registry records the actual profile rather than the requested/default one.
|
||||
*
|
||||
* @return the profile name, or {@code null} when the launcher leaves it unspecified
|
||||
*/
|
||||
default String profile() {
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The bridge's logical name for this session, as assigned at spawn
|
||||
* ({@link SpawnRequest#sessionName()}). Stable across restarts and meaningful to an operator,
|
||||
* unlike the transport identifiers above; a launch with no name leaves the peer's display
|
||||
* identity to the launcher to derive.
|
||||
*
|
||||
* @return the bridge-assigned logical session name, or {@code null} if none was assigned
|
||||
*/
|
||||
default String sessionName() {
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The peer's OWN session id — the handle that resumes this conversation later (the id a
|
||||
* later {@link Capability#SESSION_RESUME resume} spawn would pass back). Null when the adapter
|
||||
* cannot determine it — the contract for an adapter that declines
|
||||
* {@link Capability#SESSION_RESUME}; an adapter that declares that capability returns this
|
||||
* non-null for a spawn that requested session identity, because it knows the id before the
|
||||
* peer has written anything.
|
||||
*
|
||||
* <p>Deliberately not a {@code default} (CB-584, the same fix CB-571 made for
|
||||
* {@link #charterReceipt()} one method below): a decorator that forgets to override this
|
||||
* silently answers {@code null} for a question it has no basis to answer, and the gap surfaces
|
||||
* only as a resume that quietly starts a cold session, not a compile error. Every
|
||||
* implementation must answer explicitly.
|
||||
*
|
||||
* @return the peer's own session id, or {@code null} when not determinable
|
||||
*/
|
||||
String agentSessionId();
|
||||
|
||||
/**
|
||||
* The charter receipt (CB-571) for this peer's launch — the fingerprint of the exact charter
|
||||
* bytes it was started with. {@code null} when the launcher records none (a non-instrumented
|
||||
* adapter, or a launcher before this field); the session registry stores it so the spawn result
|
||||
* and the roster row can show an operator which charter a member actually got.
|
||||
*
|
||||
* <p>Deliberately not a {@code default}: a decorator that forgets to override this silently
|
||||
* answers {@code null} for a question it has no basis to answer, and the gap surfaces only as
|
||||
* a missing roster field, not a compile error. Every implementation must answer explicitly.
|
||||
*
|
||||
* @return the fingerprint, or {@code null} when the launcher carries none
|
||||
*/
|
||||
CharterReceipt charterReceipt();
|
||||
}
|
||||
|
||||
@@ -25,6 +25,19 @@ public interface PeerLauncher {
|
||||
*/
|
||||
Set<Capability> capabilities();
|
||||
|
||||
/**
|
||||
* The capabilities of the adapter that {@code profileName} resolves to (null/blank → the
|
||||
* default profile, the same resolution {@link #spawn} uses). Distinct from {@link
|
||||
* #capabilities()}, which unions every configured adapter: a caller that must know whether
|
||||
* <em>this</em> profile's backend supports a capability — e.g. {@link Capability#SESSION_RESUME}
|
||||
* before honoring {@link SpawnRequest#resumeSessionId()} — needs the per-profile answer, not
|
||||
* the fleet-wide union, or a mixed fleet could OK a resume that lands on a non-supporting
|
||||
* adapter (CB-584).
|
||||
*
|
||||
* @throws IllegalArgumentException if the profile is unknown and no default is configured
|
||||
*/
|
||||
Set<Capability> capabilitiesFor(String profileName);
|
||||
|
||||
/**
|
||||
* {@code profileName}/requestedCwd null/blank → default resolution. Returns after the peer
|
||||
* process is live (env + argv + placement complete). Never returns {@code null}.
|
||||
@@ -82,4 +95,13 @@ public interface PeerLauncher {
|
||||
* when safe to do so.
|
||||
*/
|
||||
void stop(String id);
|
||||
|
||||
/**
|
||||
* Discard the context of the peer identified by {@code id}. Implementations must bypass normal
|
||||
* bridge delivery/turn accounting. Unsupported peer kinds return {@code false} without sending
|
||||
* a guessed command.
|
||||
*
|
||||
* @return {@code true} when a reset was sent and its status transition must settle before reuse
|
||||
*/
|
||||
boolean clearContext(String id);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
package dev.ltms.bridged.peer;
|
||||
|
||||
/**
|
||||
* Thrown when a {@link PeerLauncher} starts a peer process but the peer
|
||||
* does not reach an injectable (ready-to-receive) state within the configured
|
||||
* timeout. The launcher MUST clean up any resources it created (pane, tab)
|
||||
* before throwing — no orphaned peer or pane is left behind.
|
||||
*
|
||||
* <p>This is a spawn-time failure, distinct from a post-spawn disconnect.
|
||||
* Callers treat this as a clean spawn error (the peer never materialized
|
||||
* into a usable session), not a mid-life session fault.
|
||||
*/
|
||||
public final class PeerUnreachableException extends RuntimeException {
|
||||
|
||||
public PeerUnreachableException(String message) {
|
||||
super(message);
|
||||
}
|
||||
}
|
||||
@@ -1,13 +1,32 @@
|
||||
package dev.ltms.bridged.peer;
|
||||
|
||||
/**
|
||||
* Parameters for a {@link PeerLauncher#spawn(SpawnRequest)} call — the peer-neutral
|
||||
* aggregation of what the core knows at delegation time: which profile to use, the caller's
|
||||
* requested working directory, and the caller's own cwd (to inherit when no other cwd is set).
|
||||
* What a caller asks for when spawning a peer.
|
||||
*
|
||||
* <p>A null or blank {@code profileName} means "use the launcher's default profile."
|
||||
* A null or blank {@code requestedCwd} means "inherit from config or caller."
|
||||
* A null {@code callerCwd} means "the request came from the daemon itself (not a primary)."
|
||||
* @param profileName the {@code profiles:} entry to spawn on; {@code null}/blank ⇒ the caller
|
||||
* did not choose, and the role's pool supplies the candidates
|
||||
* @param requestedCwd working directory asked for by the caller ({@code null} ⇒ unset)
|
||||
* @param callerCwd the caller's own working directory, used when nothing else pins one
|
||||
* @param sessionName agent session name (CB-547a); {@code null} ⇒ the launcher mints one
|
||||
* @param resumeSessionId prior agent session to resume; {@code null} ⇒ a fresh session
|
||||
* @param role the contract this member runs under (CB-557). Picks the tab label and,
|
||||
* with the profile, the counter its tab number comes from. {@code null} is
|
||||
* read as {@link MemberRole#DEV} — an unqualified spawn is a unit of work
|
||||
*/
|
||||
public record SpawnRequest(String profileName, String requestedCwd, String callerCwd) {
|
||||
public record SpawnRequest(String profileName, String requestedCwd, String callerCwd,
|
||||
String sessionName, String resumeSessionId, MemberRole role) {
|
||||
|
||||
public SpawnRequest {
|
||||
role = (role == null) ? MemberRole.DEV : role;
|
||||
}
|
||||
|
||||
public SpawnRequest(String profileName, String requestedCwd, String callerCwd) {
|
||||
this(profileName, requestedCwd, callerCwd, null, null, null);
|
||||
}
|
||||
|
||||
/** Pre-CB-557 shape: session identity without an explicit role (defaults to {@code dev}). */
|
||||
public SpawnRequest(String profileName, String requestedCwd, String callerCwd,
|
||||
String sessionName, String resumeSessionId) {
|
||||
this(profileName, requestedCwd, callerCwd, sessionName, resumeSessionId, null);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,115 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
* Where a credential (not a profile — see {@code BridgedConfig.Profile#effectiveCredentialId()})
|
||||
* sits out a cooldown after a {@code BACKEND_EXHAUSTED} classification (CB-578 stage B), so a fresh
|
||||
* spawn does not walk straight back onto the account that just refused on a usage limit.
|
||||
*
|
||||
* <p>Keyed by credential id, never by profile name: two profiles sharing one credential (e.g. two
|
||||
* models on the same OpenAI account) share one quarantine — {@link #quarantine} one credential id
|
||||
* and every profile whose {@code effectiveCredentialId()} equals it is quarantined too, without this
|
||||
* class knowing anything about profiles at all. That mapping is the caller's job (see
|
||||
* {@code CompositePeerLauncher} and {@code dev.ltms.bridged.inject.ExhaustionSink}).
|
||||
*
|
||||
* <p>The clock is injected ({@link LongSupplier}, conventionally {@code System::nanoTime} like
|
||||
* {@code FleetHealthMonitor}), never read inline, so a quarantine's expiry is testable without a
|
||||
* real sleep.
|
||||
*/
|
||||
public final class BackendQuarantine {
|
||||
|
||||
private final ConcurrentHashMap<String, Long> quarantinedUntilNanos = new ConcurrentHashMap<>();
|
||||
private final LongSupplier nowNanos;
|
||||
private final long cooldownNanos;
|
||||
/** True only for {@link #none()}. See {@link #quarantine} for why this exists. */
|
||||
private final boolean inert;
|
||||
|
||||
/**
|
||||
* @param nowNanos monotonic clock, injected for testability
|
||||
* @param cooldownNanos how long a fresh {@link #quarantine} call blocks the credential for;
|
||||
* must be positive
|
||||
*/
|
||||
public BackendQuarantine(LongSupplier nowNanos, long cooldownNanos) {
|
||||
this(nowNanos, cooldownNanos, false);
|
||||
}
|
||||
|
||||
private BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, boolean inert) {
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
if (cooldownNanos <= 0) {
|
||||
throw new IllegalArgumentException("cooldownNanos must be positive: " + cooldownNanos);
|
||||
}
|
||||
this.cooldownNanos = cooldownNanos;
|
||||
this.inert = inert;
|
||||
}
|
||||
|
||||
/**
|
||||
* Inert quarantine — {@link #quarantine} does nothing on this instance, so nothing is ever
|
||||
* quarantined. The explicit stand-in a caller (or a test not exercising this feature) passes
|
||||
* instead of a defaulting overload, exactly like {@code ExhaustedPatternLookup.none()}.
|
||||
*/
|
||||
public static BackendQuarantine none() {
|
||||
return new BackendQuarantine(() -> 0L, 1, true);
|
||||
}
|
||||
|
||||
/**
|
||||
* Quarantine {@code credentialId} for the configured cooldown, starting now. A repeat call while
|
||||
* already quarantined restarts the cooldown at full length — a fresh refusal is fresh evidence the
|
||||
* account is still exhausted, not a reason to let an earlier, shorter wait stand.
|
||||
*
|
||||
* <p>On {@link #none()} this is a no-op. It has to be: that instance holds a clock frozen at 0,
|
||||
* so recording a deadline would produce a quarantine that never expires — a credential locked out
|
||||
* for the life of the daemon. Two production {@code CompositePeerLauncher} constructors default to
|
||||
* {@code none()}, so the failure would be silent and permanent. An inert stand-in must omit the
|
||||
* fact, never invent one.
|
||||
*/
|
||||
public void quarantine(String credentialId) {
|
||||
Objects.requireNonNull(credentialId, "credentialId");
|
||||
if (inert) {
|
||||
return;
|
||||
}
|
||||
quarantinedUntilNanos.put(credentialId, nowNanos.getAsLong() + cooldownNanos);
|
||||
}
|
||||
|
||||
/** Whether {@code credentialId} is quarantined right now. */
|
||||
public boolean isQuarantined(String credentialId) {
|
||||
return remainingNanos(credentialId) > 0;
|
||||
}
|
||||
|
||||
/** Seconds left on {@code credentialId}'s quarantine, or empty when it is not quarantined. */
|
||||
public OptionalLong remainingSeconds(String credentialId) {
|
||||
long remaining = remainingNanos(credentialId);
|
||||
return remaining > 0 ? OptionalLong.of(toSecondsRoundedUp(remaining)) : OptionalLong.empty();
|
||||
}
|
||||
|
||||
/**
|
||||
* Every currently-quarantined credential id and its remaining seconds (CB-578 stage B fleet
|
||||
* reporting) — expired entries are never included. Not pruned from the backing map here: it stays
|
||||
* small (bounded by the number of distinct credentials ever exhausted) and a lazily-stale entry is
|
||||
* harmless, since every read already checks the deadline.
|
||||
*/
|
||||
public Map<String, Long> activeRemainingSeconds() {
|
||||
Map<String, Long> out = new LinkedHashMap<>();
|
||||
quarantinedUntilNanos.forEach((credentialId, deadline) -> {
|
||||
long remaining = deadline - nowNanos.getAsLong();
|
||||
if (remaining > 0) {
|
||||
out.put(credentialId, toSecondsRoundedUp(remaining));
|
||||
}
|
||||
});
|
||||
return out;
|
||||
}
|
||||
|
||||
private long remainingNanos(String credentialId) {
|
||||
Long deadline = quarantinedUntilNanos.get(credentialId);
|
||||
return deadline == null ? 0L : deadline - nowNanos.getAsLong();
|
||||
}
|
||||
|
||||
private static long toSecondsRoundedUp(long nanos) {
|
||||
return (nanos + 999_999_999L) / 1_000_000_000L;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
/**
|
||||
* Backward-compatible placement: an unqualified spawn always resolves to the configured default
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. This ignores caps and
|
||||
* reachability so that a pre-existing config behaves identically after upgrade.
|
||||
*
|
||||
* <p>Two exceptions walk past the default instead of returning it unconditionally:
|
||||
* <ul>
|
||||
* <li>Quarantine (CB-578 stage B): a quarantined default is a credential that just refused on
|
||||
* a usage limit, not a transient capacity or reachability concern.
|
||||
* <li>Weight 0 (CB-554): {@code fixed} is still automatic selection, so a profile the operator
|
||||
* marked "never auto-select me" ({@code weight <= 0}) must be skipped here exactly as
|
||||
* {@code weighted}/{@code round-robin} skip it — an explicit {@code bridge_spawn} naming
|
||||
* the profile is unaffected, only this automatic fallback walk.
|
||||
* </ul>
|
||||
* A fleet where nothing is ever quarantined or weight-0 never exercises either path, so today's
|
||||
* behaviour is unchanged.
|
||||
*/
|
||||
final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
|
||||
@Override
|
||||
public PlacementCandidate select(PlacementContext ctx) {
|
||||
String d = ctx.defaultProfile();
|
||||
if (d != null && !d.isBlank() && !ctx.quarantined().contains(d) && !weightExcluded(ctx, d)) {
|
||||
return new PlacementCandidate(d, null, 1.0f, null);
|
||||
}
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (!ctx.quarantined().contains(c.profile()) && !c.excluded()) {
|
||||
return new PlacementCandidate(c.profile(), null, c.weight(), c.maxLoad());
|
||||
}
|
||||
}
|
||||
if (d != null && !d.isBlank()) {
|
||||
boolean dQuarantined = ctx.quarantined().contains(d);
|
||||
boolean dWeightExcluded = weightExcluded(ctx, d);
|
||||
if (dQuarantined && dWeightExcluded) {
|
||||
throw new PlacementException("worker profile '" + d + "' is quarantined (backend "
|
||||
+ "exhausted) and has weight 0 (excluded from automatic selection), and no "
|
||||
+ "available candidate remains");
|
||||
}
|
||||
if (dWeightExcluded) {
|
||||
throw new PlacementException("worker profile '" + d + "' has weight 0 (excluded "
|
||||
+ "from automatic selection) and no available candidate remains");
|
||||
}
|
||||
if (dQuarantined) {
|
||||
throw new PlacementException("worker profile '" + d + "' is quarantined (backend "
|
||||
+ "exhausted) and no un-quarantined candidate is available");
|
||||
}
|
||||
}
|
||||
if (!ctx.candidates().isEmpty()) {
|
||||
throw new PlacementException(
|
||||
"all worker profiles are excluded from automatic selection (quarantined or weight-0)");
|
||||
}
|
||||
throw new PlacementException("no worker profiles configured");
|
||||
}
|
||||
|
||||
/** Whether {@code profile} carries {@code weight <= 0} (CB-554) among {@code ctx}'s candidates. */
|
||||
private static boolean weightExcluded(PlacementContext ctx, String profile) {
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (c.profile().equals(profile)) {
|
||||
return c.excluded();
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
/**
|
||||
* A profile (and, in CB-308, a host) that can be chosen by a {@link PlacementPolicy}.
|
||||
*
|
||||
* <p>Keeping this as a small descriptor rather than a bare profile name lets CB-308 widen
|
||||
* selection to {@code (host, profile)} pairs without changing the policy interface.
|
||||
*/
|
||||
public record PlacementCandidate(String profile, String host, float weight, Integer maxLoad) {
|
||||
|
||||
/** A candidate with no explicit host (the single-host default) and the given weight/cap. */
|
||||
public static PlacementCandidate profile(String profile, float weight, Integer maxLoad) {
|
||||
return new PlacementCandidate(profile, null, weight, maxLoad);
|
||||
}
|
||||
|
||||
/** A candidate with no explicit host, unit weight, and no cap. */
|
||||
public static PlacementCandidate profile(String profile) {
|
||||
return new PlacementCandidate(profile, null, 1.0f, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* True when this candidate carries an explicit {@code weight <= 0} (CB-554) and must be
|
||||
* skipped by every automatic policy — the same way a quarantined or unreachable candidate is
|
||||
* skipped. {@code BridgedConfig.Profile}'s compact constructor already normalises "absent" to
|
||||
* {@code 1.0} and "negative" to {@code 0.0}, so this is a plain threshold check here; it does
|
||||
* not need to distinguish "explicit 0" from "absent" itself.
|
||||
*
|
||||
* <p>Exclusion is about <em>automatic</em> selection only — an explicit
|
||||
* {@code bridge_spawn{profile:"..."}} bypasses placement entirely and is unaffected.
|
||||
*/
|
||||
public boolean excluded() {
|
||||
return weight <= 0.0f;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
|
||||
/**
|
||||
* Everything a {@link PlacementPolicy} needs to make one selection.
|
||||
*
|
||||
* @param defaultProfile profile a {@code fixed} policy should return (may be {@code null})
|
||||
* @param candidates every configured candidate; the policy filters out those at cap or unreachable
|
||||
* @param liveCount current live worker count per profile (from the session registry)
|
||||
* @param unreachable profiles already known to have failed in this spawn attempt
|
||||
* @param quarantined profiles whose credential is currently quarantined (CB-578 stage B) — a
|
||||
* {@code BACKEND_EXHAUSTED} classification put it, or a profile it shares a
|
||||
* credential with, on cooldown. Filtered the same way as {@code unreachable}.
|
||||
*/
|
||||
public record PlacementContext(String defaultProfile,
|
||||
List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount,
|
||||
Set<String> unreachable,
|
||||
Set<String> quarantined) {
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
/**
|
||||
* Thrown when a {@link PlacementPolicy} has no candidate available. Kept as a distinct type so
|
||||
* callers can distinguish "no capacity" from a spawn-time transport failure.
|
||||
*/
|
||||
public final class PlacementException extends IllegalStateException {
|
||||
|
||||
public PlacementException(String message) {
|
||||
super(message);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,41 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
/**
|
||||
* Factory for the built-in placement policies.
|
||||
*/
|
||||
public final class PlacementPolicies {
|
||||
|
||||
private PlacementPolicies() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a policy name from config. Absent/blank values and {@code "fixed"} return the
|
||||
* backward-compatible fixed policy; unknown names throw.
|
||||
*/
|
||||
public static PlacementPolicy fromName(String name) {
|
||||
String n = (name == null) ? "" : name.toLowerCase();
|
||||
if (n.isBlank() || "fixed".equals(n)) {
|
||||
return fixed();
|
||||
}
|
||||
if ("weighted".equals(n)) {
|
||||
return weighted();
|
||||
}
|
||||
if ("round-robin".equals(n)) {
|
||||
return roundRobin();
|
||||
}
|
||||
throw new IllegalArgumentException("unknown placement policy '" + name
|
||||
+ "' — must be one of: fixed, round-robin, weighted");
|
||||
}
|
||||
|
||||
public static PlacementPolicy fixed() {
|
||||
return new FixedPlacementPolicy();
|
||||
}
|
||||
|
||||
public static PlacementPolicy weighted() {
|
||||
return new WeightedRoundRobinPolicy();
|
||||
}
|
||||
|
||||
public static PlacementPolicy roundRobin() {
|
||||
return new RoundRobinPlacementPolicy();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
/**
|
||||
* How {@code bridged} chooses a worker profile when a spawn names none. Implementations are
|
||||
* deterministic and unit-testable; the caller (the composite launcher) handles failover retries.
|
||||
*/
|
||||
public interface PlacementPolicy {
|
||||
|
||||
/**
|
||||
* Pick one candidate from the configured set.
|
||||
*
|
||||
* @throws java.lang.IllegalStateException when no candidate is available, with a message naming
|
||||
* whether every profile is at capacity or unreachable
|
||||
*/
|
||||
PlacementCandidate select(PlacementContext ctx);
|
||||
}
|
||||
@@ -0,0 +1,85 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Shared filtering and empty-set reporting used by the built-in placement policies.
|
||||
*/
|
||||
final class PlacementPolicyUtil {
|
||||
|
||||
private PlacementPolicyUtil() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Candidates that are not weight-excluded (CB-554: explicit {@code weight <= 0}, checked
|
||||
* first because it is a static config choice rather than transient state), not
|
||||
* known-unreachable, not quarantined (CB-578 stage B), and have not reached their maxLoad.
|
||||
* A {@code null} maxLoad means unlimited.
|
||||
*/
|
||||
static List<PlacementCandidate> available(PlacementContext ctx) {
|
||||
List<PlacementCandidate> out = new ArrayList<>();
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (c.excluded() || ctx.unreachable().contains(c.profile())
|
||||
|| ctx.quarantined().contains(c.profile())) {
|
||||
continue;
|
||||
}
|
||||
Integer cap = c.maxLoad();
|
||||
if (cap != null) {
|
||||
int live = ctx.liveCount().apply(c.profile());
|
||||
if (live >= cap) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
out.add(c);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a clear exception describing why every candidate was dropped: all weight-0, all
|
||||
* quarantined, all at capacity, all unreachable, or a mix. Each candidate is counted into
|
||||
* exactly one bucket (weight-excluded takes priority) so a candidate excluded for more than
|
||||
* one reason is never double-counted.
|
||||
*/
|
||||
static PlacementException emptyException(PlacementContext ctx) {
|
||||
int weightExcluded = 0;
|
||||
int atCap = 0;
|
||||
int unreachable = 0;
|
||||
int quarantined = 0;
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
Integer cap = c.maxLoad();
|
||||
if (c.excluded()) {
|
||||
weightExcluded++;
|
||||
} else if (ctx.quarantined().contains(c.profile())) {
|
||||
quarantined++;
|
||||
} else if (ctx.unreachable().contains(c.profile())) {
|
||||
unreachable++;
|
||||
} else if (cap != null && ctx.liveCount().apply(c.profile()) >= cap) {
|
||||
atCap++;
|
||||
}
|
||||
}
|
||||
|
||||
int total = ctx.candidates().size();
|
||||
if (total == 0) {
|
||||
return new PlacementException("no worker profiles configured");
|
||||
}
|
||||
if (weightExcluded == total) {
|
||||
return new PlacementException(
|
||||
"all worker profiles have weight 0 (excluded from automatic selection)");
|
||||
}
|
||||
if (quarantined == total) {
|
||||
return new PlacementException("all worker profiles are quarantined (backend exhausted)");
|
||||
}
|
||||
if (atCap == total) {
|
||||
return new PlacementException("all worker profiles are at maxLoad");
|
||||
}
|
||||
if (unreachable == total) {
|
||||
return new PlacementException("all worker profiles are unreachable");
|
||||
}
|
||||
return new PlacementException("no worker profile available: " + atCap + " at maxLoad, "
|
||||
+ unreachable + " unreachable, " + quarantined + " quarantined, "
|
||||
+ weightExcluded + " weight-0, "
|
||||
+ (total - atCap - unreachable - quarantined - weightExcluded) + " remaining");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
|
||||
/**
|
||||
* Deterministic round-robin over the profiles that still have capacity and are not known to be
|
||||
* unreachable. The index advances only on successful selections so the distribution stays even
|
||||
* across spawns.
|
||||
*/
|
||||
final class RoundRobinPlacementPolicy implements PlacementPolicy {
|
||||
|
||||
private final AtomicInteger index = new AtomicInteger(0);
|
||||
|
||||
@Override
|
||||
public synchronized PlacementCandidate select(PlacementContext ctx) {
|
||||
List<PlacementCandidate> available = PlacementPolicyUtil.available(ctx);
|
||||
if (available.isEmpty()) {
|
||||
throw PlacementPolicyUtil.emptyException(ctx);
|
||||
}
|
||||
return available.get(index.getAndIncrement() % available.size());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* Smooth weighted round-robin (nginx-style): for each selection, add the candidate's weight to
|
||||
* its current score, pick the highest score, then subtract the total weight of all available
|
||||
* candidates from the winner. Weights are not required to sum to 1.0; only their ratios matter.
|
||||
*
|
||||
* <p>The state is per-policy instance and protected by {@code synchronized} so concurrent spawns
|
||||
* see a consistent, deterministic sequence rather than interleaving updates.
|
||||
*/
|
||||
final class WeightedRoundRobinPolicy implements PlacementPolicy {
|
||||
|
||||
private final Map<String, Double> current = new ConcurrentHashMap<>();
|
||||
|
||||
@Override
|
||||
public synchronized PlacementCandidate select(PlacementContext ctx) {
|
||||
List<PlacementCandidate> available = PlacementPolicyUtil.available(ctx);
|
||||
if (available.isEmpty()) {
|
||||
throw PlacementPolicyUtil.emptyException(ctx);
|
||||
}
|
||||
|
||||
double total = 0.0;
|
||||
for (PlacementCandidate c : available) {
|
||||
total += c.weight();
|
||||
}
|
||||
if (total <= 0.0) {
|
||||
throw new PlacementException("all available profiles have non-positive weight");
|
||||
}
|
||||
|
||||
PlacementCandidate best = null;
|
||||
double bestScore = Double.NEGATIVE_INFINITY;
|
||||
for (PlacementCandidate c : available) {
|
||||
double score = current.merge(c.profile(), (double) c.weight(), (old, add) -> old + add);
|
||||
if (score > bestScore) {
|
||||
bestScore = score;
|
||||
best = c;
|
||||
}
|
||||
}
|
||||
if (best == null) {
|
||||
throw new PlacementException("no placement candidate could be selected");
|
||||
}
|
||||
|
||||
current.put(best.profile(), current.get(best.profile()) - total);
|
||||
return best;
|
||||
}
|
||||
}
|
||||
@@ -2,17 +2,24 @@ package dev.ltms.bridged.rest;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.bridged.auth.AuditLog;
|
||||
import dev.ltms.bridged.auth.Authz;
|
||||
import dev.ltms.bridged.auth.CallerResolver;
|
||||
import dev.ltms.bridged.auth.Principal;
|
||||
import dev.ltms.bridged.guard.GuardException;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.HerdrClient;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.inject.WorkerPresence;
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.placement.PlacementException;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.session.WorkerSession;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import dev.ltms.bridged.session.WorktreeRequest;
|
||||
import dev.ltms.bridged.worker.ClaudeCodeLauncher;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import io.javalin.Javalin;
|
||||
import io.javalin.http.Context;
|
||||
import jakarta.servlet.http.HttpServlet;
|
||||
@@ -43,25 +50,47 @@ public final class BridgedApp {
|
||||
private static final long DEFAULT_ASK_TIMEOUT_MS = 55_000;
|
||||
private static final long MAX_ASK_TIMEOUT_MS = 115_000;
|
||||
|
||||
/** Context attribute under which the resolved caller is stashed by the auth filter. */
|
||||
private static final String CALLER = "bridged.caller";
|
||||
|
||||
private final HerdrClient herdr;
|
||||
private final ClaudeCodeLauncher workers;
|
||||
private final PeerLauncher workers;
|
||||
private final SessionManager sessions; // CB-301: authoritative session registry
|
||||
private final MessageService messages;
|
||||
private final Rendezvous rendezvous;
|
||||
private final WorkerPresence presence; // CB-113: which workers are MCP-connected (available)
|
||||
private final MemberPresence presence; // CB-113: which workers are MCP-connected (available)
|
||||
private final HttpServlet mcpServlet; // MCP Streamable-HTTP endpoint, mounted at /mcp (nullable)
|
||||
private final CallerResolver auth; // CB-501: null → authz not enforced (legacy behaviour)
|
||||
private final Metrics metrics; // CB-502: null → /metrics not exposed
|
||||
private final ObjectMapper mapper = new ObjectMapper();
|
||||
|
||||
public BridgedApp(HerdrClient herdr, ClaudeCodeLauncher workers, SessionManager sessions,
|
||||
MessageService messages, Rendezvous rendezvous, WorkerPresence presence,
|
||||
/**
|
||||
* Legacy constructor — no identity resolution and no authorization, exactly as the REST surface
|
||||
* behaved before CB-501. Retained so existing acceptance tests keep exercising handler
|
||||
* behaviour without each needing an auth fixture.
|
||||
*/
|
||||
public BridgedApp(HerdrClient herdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet) {
|
||||
this(herdr, workers, sessions, messages, presence, mcpServlet, null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param auth resolves each request's {@link Principal}; {@code null} disables authorization
|
||||
* entirely (legacy). {@code main} always supplies one.
|
||||
* @param metrics registry to instrument and expose at {@code GET /metrics}; {@code null} omits
|
||||
* the endpoint
|
||||
*/
|
||||
public BridgedApp(HerdrClient herdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics) {
|
||||
this.herdr = herdr;
|
||||
this.workers = workers;
|
||||
this.sessions = sessions;
|
||||
this.messages = messages;
|
||||
this.rendezvous = rendezvous;
|
||||
this.presence = presence;
|
||||
this.mcpServlet = mcpServlet;
|
||||
this.auth = auth;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Wire routes onto a fresh, unstarted Javalin instance. Caller starts it. */
|
||||
@@ -74,21 +103,81 @@ public final class BridgedApp {
|
||||
h.addServlet(new ServletHolder(mcpServlet), "/mcp"));
|
||||
}
|
||||
});
|
||||
// CB-501: resolve identity once per request, before any handler. /mcp does NOT pass through
|
||||
// here — it is a raw servlet on Jetty's context handler — so BridgeMcp enforces separately
|
||||
// against the same CallerResolver. Any check that lives in only one place is not a control.
|
||||
if (auth != null) {
|
||||
app.before(ctx -> ctx.attribute(CALLER,
|
||||
auth.resolve(ctx.req().getRemoteAddr(), ctx.req().getRemotePort(),
|
||||
ctx.header("Authorization"))));
|
||||
}
|
||||
app.get("/healthz", this::healthz);
|
||||
if (metrics != null) {
|
||||
app.get("/metrics", this::metrics);
|
||||
}
|
||||
app.get("/sessions", this::sessions);
|
||||
app.get("/agents", this::agents);
|
||||
app.get("/workers", this::listWorkers); // CB-304: registry roster + live herdr status
|
||||
app.get("/profiles", this::profiles); // configured worker profiles
|
||||
app.post("/workers", this::spawnWorker); // optional ?profile= or {"profile":…}
|
||||
app.delete("/workers/{paneId}", this::stopWorker);
|
||||
app.get("/members", this::listMembers); // CB-304: registry roster + live herdr status
|
||||
app.get("/profiles", this::profiles); // configured backend profiles
|
||||
app.post("/members", this::spawnMember); // optional ?role=&profile= or {"role":…,"profile":…}
|
||||
app.delete("/members/{paneId}", this::stopMember);
|
||||
app.post("/sessions/{id}/message", this::sendMessage); // bridge_send (primary; blocking, wait:false, or answer via turnId)
|
||||
app.post("/sessions/{id}/reply", this::replyMessage); // bridge_reply (worker)
|
||||
app.get("/sessions/{id}/replies", this::drainReplies); // drain reply inbox (CB-307)
|
||||
app.post("/sessions/{id}/ask", this::askMessage); // bridge_ask (worker → primary, CB-205)
|
||||
app.get("/sessions/{id}/status", this::sessionStatus); // bridge_status
|
||||
app.get("/tasks/{ticket}", this::taskStatus); // poll an async (wait:false) send
|
||||
return app;
|
||||
}
|
||||
|
||||
/**
|
||||
* Gate a handler on the CB-505 authorization table. Returns {@code true} when the request may
|
||||
* proceed; otherwise writes the error response and returns {@code false}.
|
||||
*
|
||||
* <p>401 vs 403 is a real distinction here: 401 means "you presented no usable identity" (a
|
||||
* credential problem the caller can fix), 403 means "you are authenticated, but this is not
|
||||
* yours" (a worker reaching for another worker's session, or for orchestration).
|
||||
*/
|
||||
private boolean allow(Context ctx, Authz.Action action, String target) {
|
||||
if (auth == null) {
|
||||
return true; // legacy: authorization not enforced
|
||||
}
|
||||
Principal caller = ctx.attribute(CALLER);
|
||||
if (Authz.permits(caller, action, target)) {
|
||||
if (action != Authz.Action.READ && action != Authz.Action.METRICS) {
|
||||
AuditLog.allowed(caller, action, target); // reads would drown the trail
|
||||
}
|
||||
return true;
|
||||
}
|
||||
if (Authz.isUnauthenticated(caller)) {
|
||||
AuditLog.denied(caller, action, target, "unauthenticated");
|
||||
countAuthFailure("unauthenticated");
|
||||
ctx.status(401).json(Map.of("error", "unauthenticated",
|
||||
"detail", "present Authorization: Bearer <token>"));
|
||||
} else {
|
||||
AuditLog.denied(caller, action, target, "forbidden");
|
||||
countAuthFailure("forbidden");
|
||||
ctx.status(403).json(Map.of("error", "forbidden",
|
||||
"detail", caller.describe() + " may not " + action + " on "
|
||||
+ (target == null ? "this resource" : target)));
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
private void countAuthFailure(String reason) {
|
||||
if (metrics != null) {
|
||||
metrics.inc("bridged_auth_failures_total", "reason", reason);
|
||||
}
|
||||
}
|
||||
|
||||
/** Prometheus scrape endpoint (CB-502). */
|
||||
private void metrics(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.METRICS, null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).contentType("text/plain; version=0.0.4; charset=utf-8").result(metrics.render());
|
||||
}
|
||||
|
||||
/** Liveness + herdr reachability. 200 when herdr answers ping, 503 otherwise. */
|
||||
private void healthz(Context ctx) {
|
||||
try {
|
||||
@@ -108,6 +197,9 @@ public final class BridgedApp {
|
||||
|
||||
/** Sessions view derived from herdr {@code workspace.list} (one workspace → one row). */
|
||||
private void sessions(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
return;
|
||||
}
|
||||
JsonNode result = herdr.call("workspace.list");
|
||||
List<Map<String, Object>> out = new ArrayList<>();
|
||||
for (JsonNode w : result.path("workspaces")) {
|
||||
@@ -123,22 +215,41 @@ public final class BridgedApp {
|
||||
|
||||
/** Discovery: every agent herdr tracks, keyed by its Claude session UUID. */
|
||||
private void agents(Context ctx) {
|
||||
ctx.status(200).json(Map.of("agents", workers.list().stream().map(BridgedApp::view).toList()));
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).json(Map.of("agents",
|
||||
workers.list().stream().map(Agent.class::cast).map(BridgedApp::view).toList()));
|
||||
}
|
||||
|
||||
/** CB-304: bridge-owned roster merged with live herdr status by paneId. */
|
||||
private void listWorkers(Context ctx) {
|
||||
private void listMembers(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
return;
|
||||
}
|
||||
// CB-519: the registry key is a host-unique id, not the pane coordinate — join on terminal.
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.filter(a -> a.paneId() != null)
|
||||
.collect(Collectors.toMap(Agent::paneId, Function.identity(), (_, b) -> b));
|
||||
.map(Agent.class::cast)
|
||||
.filter(a -> a.terminalId() != null)
|
||||
.collect(Collectors.toMap(Agent::terminalId, Function.identity(), (_, b) -> b));
|
||||
List<Map<String, Object>> out = sessions.roster().stream()
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.paneId())))
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.terminalId())))
|
||||
.toList();
|
||||
ctx.status(200).json(Map.of("workers", out));
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
body.put("workers", out);
|
||||
// CB-586: operator visibility for the refs/wip snapshot store without shelling into the
|
||||
// repo — how many snapshot refs exist and roughly what they cost. Present only once a
|
||||
// worktree session has established the repo, so a never-snapshotted fleet reports nothing.
|
||||
sessions.wipRefs().ifPresent(st -> body.put("wipRefs",
|
||||
Map.of("count", st.count(), "costBytes", st.costBytes())));
|
||||
ctx.status(200).json(body);
|
||||
}
|
||||
|
||||
/** The configured worker profiles and which one a no-argument spawn uses. */
|
||||
private void profiles(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).json(Map.of(
|
||||
"profiles", workers.profiles(),
|
||||
"default", workers.defaultProfile() == null ? "" : workers.defaultProfile()));
|
||||
@@ -149,7 +260,11 @@ public final class BridgedApp {
|
||||
* body) picks which configured profile; omitted → the default. 403 if the base_url would breach
|
||||
* the subscription boundary, 400 for an unknown profile.
|
||||
*/
|
||||
private void spawnWorker(Context ctx) {
|
||||
private void spawnMember(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.SPAWN, null)) {
|
||||
return;
|
||||
}
|
||||
String role = ctx.queryParam("role");
|
||||
String profile = ctx.queryParam("profile");
|
||||
String cwd = ctx.queryParam("cwd");
|
||||
String worktree = ctx.queryParam("worktree");
|
||||
@@ -160,6 +275,7 @@ public final class BridgedApp {
|
||||
String body = ctx.body();
|
||||
if (!body.isBlank()) {
|
||||
JsonNode b = mapper.readTree(body);
|
||||
if (role == null || role.isBlank()) role = b.path("role").asText(null);
|
||||
if (profile == null || profile.isBlank()) profile = b.path("profile").asText(null);
|
||||
if (cwd == null || cwd.isBlank()) cwd = b.path("cwd").asText(null);
|
||||
if (worktree == null || worktree.isBlank()) worktree = b.path("worktree").asText(null);
|
||||
@@ -170,14 +286,29 @@ public final class BridgedApp {
|
||||
}
|
||||
}
|
||||
WorktreeRequest wt = worktreeRequest(worktree, ticket);
|
||||
MemberRole memberRole;
|
||||
try {
|
||||
memberRole = (role == null || role.isBlank()) ? MemberRole.DEV : MemberRole.parse(role);
|
||||
} catch (IllegalArgumentException e) {
|
||||
ctx.status(400).json(Map.of("error", "unknown_role", "detail", e.getMessage()));
|
||||
return;
|
||||
}
|
||||
try {
|
||||
// No MCP caller over REST, so callerCwd and ownerTerminal are null.
|
||||
WorkerSession worker = sessions.acquire(blankToNull(profile), blankToNull(cwd), null, null, wt);
|
||||
ctx.status(201).json(view(worker));
|
||||
MemberSession member = sessions.acquire(blankToNull(profile), memberRole,
|
||||
blankToNull(cwd), null, null, wt);
|
||||
ctx.status(201).json(view(member));
|
||||
} catch (GuardException e) {
|
||||
ctx.status(403).json(Map.of("error", "subscription_boundary", "detail", e.getMessage()));
|
||||
} catch (PlacementException e) {
|
||||
// CB-599: no candidate had capacity (maxLoad, quarantine, or all-exhausted) — a benign,
|
||||
// likely-transient refusal, distinct from "profile does not exist" below. 503: the
|
||||
// request was valid and will likely succeed later.
|
||||
ctx.status(503).json(Map.of("error", "no_capacity", "detail", e.getMessage()));
|
||||
} catch (IllegalArgumentException e) {
|
||||
ctx.status(400).json(Map.of("error", "unknown_profile", "detail", e.getMessage()));
|
||||
} catch (PeerUnreachableException e) {
|
||||
ctx.status(502).json(Map.of("error", "spawn_timeout", "detail", e.getMessage()));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -199,8 +330,12 @@ public final class BridgedApp {
|
||||
}
|
||||
|
||||
/** Tear a worker down by pane id. */
|
||||
private void stopWorker(Context ctx) {
|
||||
sessions.release(ctx.pathParam("paneId"));
|
||||
private void stopMember(Context ctx) {
|
||||
String paneId = ctx.pathParam("paneId");
|
||||
if (!allow(ctx, Authz.Action.STOP, paneId)) {
|
||||
return;
|
||||
}
|
||||
sessions.release(paneId);
|
||||
ctx.status(204);
|
||||
}
|
||||
|
||||
@@ -212,6 +347,9 @@ public final class BridgedApp {
|
||||
*/
|
||||
private void sendMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.SEND, id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
String turnId;
|
||||
long timeout;
|
||||
@@ -278,9 +416,12 @@ public final class BridgedApp {
|
||||
case TIMED_OUT_QUEUED -> "queued";
|
||||
case BUSY -> "busy";
|
||||
case WORKER_FAILED -> "failed";
|
||||
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
||||
default -> "done"; // unreachable (terminal outcomes handled above)
|
||||
},
|
||||
"detail", reply.outcome() == MessageService.Outcome.WORKER_FAILED && reply.text() != null
|
||||
"detail", (reply.outcome() == MessageService.Outcome.WORKER_FAILED
|
||||
|| reply.outcome() == MessageService.Outcome.BACKEND_EXHAUSTED)
|
||||
&& reply.text() != null
|
||||
? reply.text()
|
||||
: "no reply within " + timeout + "ms; poll status or retry"));
|
||||
}
|
||||
@@ -293,6 +434,9 @@ public final class BridgedApp {
|
||||
*/
|
||||
private void askMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.ASK, id)) {
|
||||
return;
|
||||
}
|
||||
String question;
|
||||
long timeout;
|
||||
try {
|
||||
@@ -322,10 +466,16 @@ public final class BridgedApp {
|
||||
|
||||
/**
|
||||
* The worker's structured reply ({@code bridge_reply}) — resolves the blocking send awaiting
|
||||
* on this session. 200 if a send was waiting, 409 if none was (late or spurious reply).
|
||||
* on this session, or queues the reply in the inbox when no send is open (CB-307).
|
||||
*/
|
||||
private void replyMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
// The rule that matters: a worker may reply only as itself. Over MCP this was already true
|
||||
// structurally (identity comes from the connection, never an argument); over REST the path
|
||||
// id was simply trusted, so this is where the invariant actually gets enforced.
|
||||
if (!allow(ctx, Authz.Action.REPLY, id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
try {
|
||||
content = mapper.readTree(ctx.body()).path("content").asText("");
|
||||
@@ -333,13 +483,25 @@ public final class BridgedApp {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "body must be JSON"));
|
||||
return;
|
||||
}
|
||||
if (rendezvous.resolve(id, content)) {
|
||||
ctx.status(200).json(Map.of("sessionId", id, "delivered", true));
|
||||
} else {
|
||||
ctx.status(409).json(Map.of(
|
||||
"sessionId", id, "error", "no_pending_send",
|
||||
"detail", "no send is awaiting a reply for this session"));
|
||||
messages.reply(id, content);
|
||||
ctx.status(200).json(Map.of("sessionId", id, "delivered", true));
|
||||
}
|
||||
|
||||
/**
|
||||
* Drain the reply inbox for a worker session — peek + ack any replies that arrived when no send
|
||||
* was open. At-least-once: draining removes them from the inbox so a subsequent read returns
|
||||
* nothing; an in-flight failure between the drain and the caller's processing re-surfaces them.
|
||||
*/
|
||||
private void drainReplies(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.DRAIN, id)) {
|
||||
return;
|
||||
}
|
||||
var replies = messages.drainReplies(id);
|
||||
ctx.status(200).json(Map.of("sessionId", id, "replies",
|
||||
replies.stream().map(m -> Map.of(
|
||||
"msgId", m.msgId(),
|
||||
"content", m.content())).toList()));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -350,11 +512,24 @@ public final class BridgedApp {
|
||||
*/
|
||||
private void sessionStatus(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.READ, id)) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
ctx.status(200).json(Map.of(
|
||||
"sessionId", id,
|
||||
"status", messages.status(id).name().toLowerCase(),
|
||||
"ready", presence.isPresent(id)));
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
body.put("sessionId", id);
|
||||
body.put("status", messages.status(id).name().toLowerCase());
|
||||
body.put("ready", presence.isPresent(id));
|
||||
// CB-582: a worker paused mid-turn in an async bridge_ask is otherwise invisible to a
|
||||
// status poll — surface the open question and how to answer it, same as bridge_poll's
|
||||
// Phase.ASKING view.
|
||||
MessageService.PendingAsk ask = messages.pendingAsk(id);
|
||||
if (ask != null) {
|
||||
body.put("question", ask.question());
|
||||
body.put("turnId", ask.turnId());
|
||||
body.put("ticket", ask.ticket());
|
||||
}
|
||||
ctx.status(200).json(body);
|
||||
} catch (HerdrException e) {
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
@@ -362,6 +537,9 @@ public final class BridgedApp {
|
||||
|
||||
/** Poll an async (wait:false) delegation by ticket. 404 for an unknown/expired ticket. */
|
||||
private void taskStatus(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
return;
|
||||
}
|
||||
MessageService.TaskView v = messages.poll(ctx.pathParam("ticket"));
|
||||
if (v == null) {
|
||||
ctx.status(404).json(Map.of("error", "unknown_ticket", "detail", "no such task (or it has expired)"));
|
||||
@@ -377,6 +555,11 @@ public final class BridgedApp {
|
||||
if (v.detail() != null) {
|
||||
body.put("detail", v.detail());
|
||||
}
|
||||
// CB-582: Phase.ASKING carries the question in v.reply() (handled above) and its answer-
|
||||
// correlation id here — a REST caller polling this ticket otherwise has no way to answer it.
|
||||
if (v.turnId() != null) {
|
||||
body.put("turnId", v.turnId());
|
||||
}
|
||||
ctx.status(200).json(body);
|
||||
}
|
||||
|
||||
@@ -403,7 +586,7 @@ public final class BridgedApp {
|
||||
}
|
||||
|
||||
/** CB-301 projection of an authoritative bridge-owned session. */
|
||||
private static Map<String, Object> view(WorkerSession s) {
|
||||
private static Map<String, Object> view(MemberSession s) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("terminalId", s.terminalId());
|
||||
m.put("paneId", s.paneId());
|
||||
|
||||
@@ -12,7 +12,12 @@ import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.security.SecureRandom;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.stream.Collectors;
|
||||
@@ -27,6 +32,54 @@ public final class GitWorktrees implements Worktrees {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(GitWorktrees.class);
|
||||
|
||||
/** Project-level MCP config. Present in the repo, so every worktree would otherwise inherit the
|
||||
* primary's IDE server mounts (a CB-523 worker edited the primary checkout; see the isolation
|
||||
* javadoc). Neutralized unconditionally. */
|
||||
private static final String MCP_CONFIG = ".mcp.json";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code .mcp.json}: a valid, explicitly empty server map. */
|
||||
private static final String NEUTRAL_MCP_CONFIG = "{\n \"mcpServers\": {}\n}\n";
|
||||
|
||||
/** OpenCode's repo-level config. Tracked here, so it lands in every worktree, and it mounts the
|
||||
* primary's gitea and context7 servers with the primary's credentials. Neutralized so the worker
|
||||
* gets only the config its launcher writes via {@code OPENCODE_CONFIG}.
|
||||
*
|
||||
* <p>The reason has changed shape and is now stronger. It used to be a crash: the file carried
|
||||
* {@code {file:.secrets/...}} references to gitignored files that never reached a worktree, and
|
||||
* opencode refuses to start on a dangling reference (CB-543). Those credentials now live in one
|
||||
* shell-level store and the file reads them as {@code {env:...}}, so in a worktree the reference
|
||||
* resolves instead of failing. That is worse, not better: a member would silently inherit the
|
||||
* primary's admin-scoped {@code GITEA_ACCESS_TOKEN}. A loud crash became a quiet privilege leak,
|
||||
* so this entry protects a boundary now rather than papering over a startup error. */
|
||||
private static final String OPENCODE_CONFIG = "opencode.json";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code opencode.json}: a valid, empty JSON object. */
|
||||
private static final String NEUTRAL_OPENCODE_CONFIG = "{}\n";
|
||||
|
||||
/** Autoenv's repo-level config. Not tracked today, but re-landing it must stay safe: autoenv
|
||||
* authorizes by path, so a fresh worktree path is always unauthorized and its interactive prompt
|
||||
* would block every spawn — neutralize it so it can never be committed. */
|
||||
private static final String AUTOENV_CONFIG = ".autoenv";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code .autoenv}: a valid, empty env file. */
|
||||
private static final String NEUTRAL_AUTOENV_CONFIG = "";
|
||||
|
||||
/**
|
||||
* A tracked project config that is hostile in a provisioned worktree, and what to replace it
|
||||
* with. {@link #file} is the repo-relative path; {@link #stub} is a neutral but VALID payload for
|
||||
* that file's format — a malformed stub would only trade one crash for another;
|
||||
* {@link #createIfAbsent} keeps {@code .mcp.json}'s long-standing behaviour of writing its stub
|
||||
* even when the repo carries no such file, whereas the others are only touched when present.
|
||||
*/
|
||||
private record WorktreeHostileConfig(String file, String stub, boolean createIfAbsent) {}
|
||||
|
||||
/** The worktree-hostile configs neutralized in every provisioned worktree, in order. */
|
||||
private static final List<WorktreeHostileConfig> WORKTREE_HOSTILE_CONFIGS = List.of(
|
||||
new WorktreeHostileConfig(MCP_CONFIG, NEUTRAL_MCP_CONFIG, true),
|
||||
new WorktreeHostileConfig(OPENCODE_CONFIG, NEUTRAL_OPENCODE_CONFIG, false),
|
||||
new WorktreeHostileConfig(AUTOENV_CONFIG, NEUTRAL_AUTOENV_CONFIG, false)
|
||||
);
|
||||
|
||||
private final String configuredRoot;
|
||||
private final SecureRandom random = new SecureRandom();
|
||||
private final AtomicLong seq = new AtomicLong();
|
||||
@@ -55,9 +108,59 @@ public final class GitWorktrees implements Worktrees {
|
||||
String wt = path.toAbsolutePath().toString();
|
||||
log.info("adding worktree branch={} path={} base={}", branch, wt, base);
|
||||
exec("git", "-C", repoRoot, "worktree", "add", wt, "-b", branch, base);
|
||||
isolateToolSurface(wt);
|
||||
return wt;
|
||||
}
|
||||
|
||||
/**
|
||||
* Neutralize the worktree's worktree-hostile project configs so a worker inherits only the tools
|
||||
* and environment its launcher mounts (the bridge via {@code --mcp-config}, the opencode config
|
||||
* via {@code OPENCODE_CONFIG}) — never the primary's.
|
||||
*
|
||||
* <p>This is unconditional, and it is not the same job as the parity overlay. The repo's own
|
||||
* committed {@code .mcp.json} declares the primary's IDE servers, so a fresh checkout mounts them
|
||||
* whether or not the overlay copies anything; a worker that inherits them navigates and edits
|
||||
* through tools bound to the <em>primary's</em> IntelliJ project, which silently hands it absolute
|
||||
* paths outside its own worktree. That is not hypothetical: a CB-523 worker made all 59 of its
|
||||
* edits in the primary checkout while compiling its worktree, so every build it ran was of code
|
||||
* that did not contain its changes. {@code opencode.json} is the same trap one tool over — tracked,
|
||||
* so it lands in every worktree, and it mounts gitea and context7 with the primary's own
|
||||
* credentials, which a member must never hold. {@code .autoenv} extends the principle to a
|
||||
* config that is not tracked today: autoenv authorizes by path, so a fresh worktree path is always
|
||||
* unauthorized and its interactive prompt would block every spawn, so re-landing one must be safe.
|
||||
*
|
||||
* <p>Where a config exists it is replaced by a valid neutral stub (an explicitly empty
|
||||
* map/object, or an empty env file — never a deletion, which would still let a later
|
||||
* {@code git checkout} restore the hostile copy). The {@code --skip-worktree} bit keeps the
|
||||
* neutralized copy from ever showing up as a local modification the worker might commit. A config
|
||||
* the repo does not carry is skipped silently — no stub is invented for a file the repo does not
|
||||
* have, and one missing file must never fail provisioning.
|
||||
*/
|
||||
private void isolateToolSurface(String worktreePath) {
|
||||
Path root = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
for (WorktreeHostileConfig cfg : WORKTREE_HOSTILE_CONFIGS) {
|
||||
neutralize(root, worktreePath, cfg);
|
||||
}
|
||||
}
|
||||
|
||||
private void neutralize(Path root, String worktreePath, WorktreeHostileConfig cfg) {
|
||||
Path target = root.resolve(cfg.file());
|
||||
if (!Files.exists(target) && !cfg.createIfAbsent()) {
|
||||
log.debug("{} absent in the worktree — skipping (repo does not carry it)", cfg.file());
|
||||
return;
|
||||
}
|
||||
try {
|
||||
Files.writeString(target, cfg.stub());
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot neutralize " + cfg.file() + " in the worktree: "
|
||||
+ e.getMessage(), e);
|
||||
}
|
||||
if (isTracked(root, cfg.file())) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", cfg.file());
|
||||
}
|
||||
log.debug("neutralized {} — worker tool surface is launcher-mounted only", cfg.file());
|
||||
}
|
||||
|
||||
@Override
|
||||
public void remove(String repoRoot, String worktreePath) {
|
||||
Path p = Path.of(worktreePath);
|
||||
@@ -69,6 +172,22 @@ public final class GitWorktrees implements Worktrees {
|
||||
exec("git", "-C", repoRoot, "worktree", "remove", "--force", worktreePath);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasUncommitted(String worktreePath) {
|
||||
// A worktree that is already gone holds no work to lose, and it must not break teardown:
|
||||
// git -C <missing-dir> status exits non-zero and would throw where release() is mid-way
|
||||
// through stopping a pane. Mirror remove()'s already-gone tolerance by treating it as clean.
|
||||
Path p = Path.of(worktreePath);
|
||||
if (!Files.exists(p)) {
|
||||
log.debug("worktree {} already gone — nothing can be uncommitted", worktreePath);
|
||||
return false;
|
||||
}
|
||||
// No --untracked-files=no: the exact shape of the work lost in CB-576 was a new file
|
||||
// that was never added, so an untracked-only worktree is still dirty.
|
||||
String out = exec("git", "-C", worktreePath, "status", "--porcelain");
|
||||
return !out.isBlank();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void overlayParity(String repoRoot, String worktreePath, List<String> overlay) {
|
||||
if (overlay == null || overlay.isEmpty()) {
|
||||
@@ -103,6 +222,209 @@ public final class GitWorktrees implements Worktrees {
|
||||
return Path.of(out.trim()).toAbsolutePath().normalize().toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-578 stage C, fixed by CB-587. Stages into a <em>temporary</em> index (never the worktree's
|
||||
* real one, which the worker may still be writing to) — but that temp index is first <em>seeded</em>
|
||||
* from the worktree's real one, rather than starting empty:
|
||||
*
|
||||
* <pre>
|
||||
* cp $(git -C worktree rev-parse --git-path index) <temp>
|
||||
* GIT_INDEX_FILE=<temp> git -C worktree add -A
|
||||
* tree=$(GIT_INDEX_FILE=<temp> git -C worktree write-tree)
|
||||
* commit=$(git -C worktree commit-tree $tree -p HEAD -m message)
|
||||
* git -C worktree update-ref refs/wip/branch $commit
|
||||
* </pre>
|
||||
*
|
||||
* A fresh empty index carries none of the real index's {@code --skip-worktree} /
|
||||
* {@code --assume-unchanged} bits, so {@code add -A} into it stages a skip-worktree file's local
|
||||
* on-disk content even though {@code git status --porcelain} correctly hides that file (CB-587).
|
||||
* Seeding from the real index preserves those bits, so {@code add -A} then skips exactly what
|
||||
* {@code git status} skips. {@code add -A} (never {@code -f}) also still respects
|
||||
* {@code .gitignore} exactly as it would in the real index — a gitignored file staying ignored is
|
||||
* what keeps secrets and local config out of the snapshot's tree. The temporary index file is
|
||||
* removed afterwards regardless of outcome; the worker's real index is never opened for writing.
|
||||
*/
|
||||
@Override
|
||||
public Optional<String> snapshot(String worktreePath, String branch, String message) {
|
||||
if (!Files.exists(Path.of(worktreePath))) {
|
||||
log.debug("worktree {} already gone — nothing to snapshot", worktreePath);
|
||||
return Optional.empty();
|
||||
}
|
||||
Path tempIndex;
|
||||
try {
|
||||
tempIndex = Files.createTempFile("bridged-wip-index-", ".tmp");
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot create a temporary index for snapshot: " + e.getMessage(), e);
|
||||
}
|
||||
Map<String, String> indexEnv = Map.of("GIT_INDEX_FILE", tempIndex.toAbsolutePath().toString());
|
||||
try {
|
||||
Path realIndex = resolveRealIndex(worktreePath);
|
||||
try {
|
||||
Files.copy(realIndex, tempIndex, StandardCopyOption.REPLACE_EXISTING);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot copy the worktree's real index (" + realIndex
|
||||
+ ") into the temporary snapshot index: " + e.getMessage(), e);
|
||||
}
|
||||
exec(indexEnv, "git", "-C", worktreePath, "add", "-A");
|
||||
String tree = exec(indexEnv, "git", "-C", worktreePath, "write-tree").trim();
|
||||
String commit = exec("git", "-C", worktreePath, "commit-tree", tree, "-p", "HEAD", "-m", message).trim();
|
||||
exec("git", "-C", worktreePath, "update-ref", "refs/wip/" + branch, commit);
|
||||
log.info("snapshotted worktree {} to refs/wip/{} commit={}", worktreePath, branch, commit);
|
||||
return Optional.of(commit);
|
||||
} finally {
|
||||
try {
|
||||
Files.deleteIfExists(tempIndex);
|
||||
} catch (IOException e) {
|
||||
log.debug("could not delete temporary snapshot index {}: {}", tempIndex, e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the path of {@code worktreePath}'s real index. Never assume {@code <worktree>/.git/index}:
|
||||
* in a linked worktree {@code .git} is a <em>file</em> pointing at the main repo's
|
||||
* {@code worktrees/<name>/} directory, and that is where the real per-worktree index lives.
|
||||
* {@code git rev-parse --git-path index} resolves this correctly for both a linked worktree and
|
||||
* the main checkout. Throws {@link WorktreeException} — same as every other failure in this
|
||||
* class — if the command fails or the resolved path does not exist, rather than silently
|
||||
* snapshotting from an empty index.
|
||||
*/
|
||||
private Path resolveRealIndex(String worktreePath) {
|
||||
String out = exec("git", "-C", worktreePath, "rev-parse", "--git-path", "index").trim();
|
||||
Path index = Path.of(out);
|
||||
if (!index.isAbsolute()) {
|
||||
index = Path.of(worktreePath).resolve(index).normalize();
|
||||
}
|
||||
if (!Files.exists(index)) {
|
||||
throw new WorktreeException("worktree's real index not found at resolved path " + index
|
||||
+ " (git rev-parse --git-path index reported '" + out + "')");
|
||||
}
|
||||
return index;
|
||||
}
|
||||
|
||||
/**
|
||||
* One {@code refs/wip/<branch>} snapshot ref as read by {@link #listWipRefs}: its full ref name,
|
||||
* the snapshot commit's sha, and that commit's committer time in unix millis (the age of the
|
||||
* snapshot — a snapshot is written once and never rewritten, so the commit date is the ref's).
|
||||
*/
|
||||
private record WipRef(String refName, String sha, long committerMillis) {
|
||||
String branch() {
|
||||
return refName.substring("refs/wip/".length());
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public WipRefStats wipRefs(String repoRoot) {
|
||||
List<WipRef> refs = listWipRefs(repoRoot);
|
||||
long costBytes = 0;
|
||||
for (WipRef ref : refs) {
|
||||
costBytes += treeSize(repoRoot, ref.sha());
|
||||
}
|
||||
return new WipRefStats(refs.size(), costBytes);
|
||||
}
|
||||
|
||||
@Override
|
||||
public int pruneWipRefs(String repoRoot, long minAgeMillis) {
|
||||
// The rule is documented on Worktrees#pruneWipRefs: delete only a snapshot whose tree
|
||||
// content is already reachable from main AND that is older than minAgeMillis. Reachability
|
||||
// is the floor that keeps a worker's last copy; the age floor keeps a just-written snapshot
|
||||
// from being swept while a lead may still be looking at it.
|
||||
List<WipRef> refs = listWipRefs(repoRoot);
|
||||
if (refs.isEmpty()) {
|
||||
return 0;
|
||||
}
|
||||
long nowMillis = System.currentTimeMillis();
|
||||
// Resolve what main carries once per sweep, not once per ref.
|
||||
Set<String> mainObjects = reachableObjectsFromMain(repoRoot);
|
||||
int deleted = 0;
|
||||
for (WipRef ref : refs) {
|
||||
long ageMillis = nowMillis - ref.committerMillis();
|
||||
if (ageMillis <= minAgeMillis) {
|
||||
continue; // too recent — never swept, even if it looks recoverable (CB-586)
|
||||
}
|
||||
String tree = exec("git", "-C", repoRoot, "rev-parse", ref.sha() + "^{tree}").trim();
|
||||
if (!mainObjects.contains(tree)) {
|
||||
// Last copy of the snapshot's content — the worker's work exists nowhere else.
|
||||
// Never delete automatically (CB-586 criterion 2).
|
||||
continue;
|
||||
}
|
||||
exec("git", "-C", repoRoot, "update-ref", "-d", ref.refName());
|
||||
deleted++;
|
||||
log.info("pruned snapshot ref refs/wip/{} commit={} (age {}h): its tree is already "
|
||||
+ "reachable from main, so the work is preserved; recover from reflog via "
|
||||
+ "git update-ref refs/wip/{} {}",
|
||||
ref.branch(), ref.sha(), TimeUnit.MILLISECONDS.toHours(ageMillis),
|
||||
ref.branch(), ref.sha());
|
||||
}
|
||||
return deleted;
|
||||
}
|
||||
|
||||
/**
|
||||
* Every {@code refs/wip/*} ref (see {@link WipRef}). The committer date is read as a unix
|
||||
* count of seconds and converted to millis. {@code %00} (NUL) separates the fields because a
|
||||
* branch name may contain spaces.
|
||||
*/
|
||||
private List<WipRef> listWipRefs(String repoRoot) {
|
||||
String out = exec("git", "-C", repoRoot, "for-each-ref",
|
||||
"--format=%(refname)%00%(objectname)%00%(committerdate:unix)", "refs/wip/");
|
||||
List<WipRef> refs = new ArrayList<>();
|
||||
for (String line : out.split("\\R")) {
|
||||
if (line.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
String[] parts = line.split("\u0000", -1);
|
||||
if (parts.length == 3 && !parts[1].isBlank()) {
|
||||
refs.add(new WipRef(parts[0], parts[1], Long.parseLong(parts[2]) * 1000L));
|
||||
}
|
||||
}
|
||||
return refs;
|
||||
}
|
||||
|
||||
/**
|
||||
* The set of object shas reachable from {@code main}, or an empty set when {@code main} cannot
|
||||
* be resolved. An empty set is the safe direction: the retention sweep then concludes nothing
|
||||
* is recoverable, so it deletes nothing — a repo with no {@code main} must never cause a
|
||||
* worker's last copy of a snapshot to be dropped on a reachability misreading.
|
||||
*/
|
||||
private Set<String> reachableObjectsFromMain(String repoRoot) {
|
||||
if (exitCode("git", "-C", repoRoot, "rev-parse", "--verify", "main") != 0) {
|
||||
log.debug("refs/wip retention: no 'main' ref in {} — treating nothing as reachable", repoRoot);
|
||||
return Set.of();
|
||||
}
|
||||
String out = exec("git", "-C", repoRoot, "rev-list", "--objects", "main");
|
||||
Set<String> objects = new HashSet<>();
|
||||
for (String line : out.split("\\R")) {
|
||||
if (line.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
int sp = line.indexOf(' ');
|
||||
objects.add(sp < 0 ? line : line.substring(0, sp));
|
||||
}
|
||||
return objects;
|
||||
}
|
||||
|
||||
/** Approximate cost of a snapshot: the sum of every blob's size in its committed tree. */
|
||||
private long treeSize(String repoRoot, String sha) {
|
||||
String out = exec("git", "-C", repoRoot, "ls-tree", "-r", "-l", sha);
|
||||
long total = 0;
|
||||
for (String line : out.split("\\R")) {
|
||||
if (line.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
// ls-tree -l row: "<mode> <type> <object> <size>\t<path>"; the size is only numeric for
|
||||
// blobs (trees read "-"), so gate on the type token and take the 4th whitespace field.
|
||||
String[] parts = line.split("\\s+");
|
||||
if (parts.length >= 4 && "blob".equals(parts[1])) {
|
||||
try {
|
||||
total += Long.parseLong(parts[3]);
|
||||
} catch (NumberFormatException ignored) {
|
||||
// a '-' size (or any anomaly) contributes nothing to the rough figure
|
||||
}
|
||||
}
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
/** Resolve the directory that will hold per-session worktree checkouts. */
|
||||
private Path resolveRoot(String repoRoot) {
|
||||
if (configuredRoot != null && !configuredRoot.isBlank()) {
|
||||
@@ -125,11 +447,20 @@ public final class GitWorktrees implements Worktrees {
|
||||
* stdout and stderr (merged by redirectErrorStream).
|
||||
*/
|
||||
private String exec(String... command) {
|
||||
return exec(Map.of(), command);
|
||||
}
|
||||
|
||||
/** Same as {@link #exec(String...)}, with extra environment variables set on the child process. */
|
||||
private String exec(Map<String, String> extraEnv, String... command) {
|
||||
String out;
|
||||
int code;
|
||||
Process p;
|
||||
try {
|
||||
p = new ProcessBuilder(command).redirectErrorStream(true).start();
|
||||
ProcessBuilder pb = new ProcessBuilder(command).redirectErrorStream(true);
|
||||
if (extraEnv != null && !extraEnv.isEmpty()) {
|
||||
pb.environment().putAll(extraEnv);
|
||||
}
|
||||
p = pb.start();
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("failed to start " + command[0] + ": " + e.getMessage(), e);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import dev.ltms.bridged.peer.CharterReceipt;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
|
||||
/**
|
||||
* A bridge-owned worker session — the authoritative in-daemon record of a worker this
|
||||
* process spawned. Immutable; state transitions are performed by replacing the record in
|
||||
* {@link SessionManager}'s registry.
|
||||
*
|
||||
* @param paneId the host-unique opaque id (CB-519) — the registry key and the argument to
|
||||
* teardown. Despite the historical name this is the {@link
|
||||
* dev.ltms.bridged.peer.PeerHandle#id()}, a UUID, and is distinct from the
|
||||
* launcher-private herdr pane coordinate.
|
||||
* @param terminalId herdr terminal handle — the {@code target} for send/read/status
|
||||
* @param profile the profile name that spawned this session — WHICH BACKEND
|
||||
* @param role the contract this member runs under — WHAT IT IS FOR. Independent of
|
||||
* {@code profile}: a reviewer may run on the same profile as the dev
|
||||
* whose diff it reads. {@code null} only for a session recorded before
|
||||
* the role was known
|
||||
* @param cwd the resolved working directory the worker started in
|
||||
* @param ownerTerminal the caller that requested this worker ({@code null} = daemon/anon)
|
||||
* @param spawnedAtNanos {@link System#nanoTime()} when the session was registered
|
||||
* @param lastActivityAtNanos {@link System#nanoTime()} of the most recent lifecycle event
|
||||
* @param turnCount number of delegated turns that have been delivered to this session
|
||||
* @param state current lifecycle state in the one-shot FSM
|
||||
* @param charterReceipt the fingerprint (CB-571) of the charter bytes this member was started
|
||||
* with; {@code null} for a session whose launcher recorded none
|
||||
* @param agentSessionId the peer's OWN session id (CB-584) — the handle a later
|
||||
* {@code resumeSessionId} spawn would pass back to resume this exact
|
||||
* conversation; {@code null} when the launcher could not determine one
|
||||
* (an adapter that declines {@code Capability.SESSION_RESUME}, or one
|
||||
* that resolves it lazily and has not yet)
|
||||
*/
|
||||
public record MemberSession(
|
||||
String paneId,
|
||||
String terminalId,
|
||||
String profile,
|
||||
MemberRole role,
|
||||
String cwd,
|
||||
String ownerTerminal,
|
||||
long spawnedAtNanos,
|
||||
long lastActivityAtNanos,
|
||||
int turnCount,
|
||||
State state,
|
||||
String worktree,
|
||||
String branch,
|
||||
CharterReceipt charterReceipt,
|
||||
String agentSessionId) {
|
||||
|
||||
/** One-shot worker lifecycle states. */
|
||||
public enum State {
|
||||
SPAWNING,
|
||||
READY,
|
||||
BUSY,
|
||||
DONE,
|
||||
FAILED,
|
||||
RELEASED
|
||||
}
|
||||
|
||||
/**
|
||||
* Backward-compatible shape: a session with no charter receipt and no agent session id (a test
|
||||
* or a launcher before CB-571 / CB-584). A separate constructor rather than a new parameter on
|
||||
* the canonical one, so existing call sites that have nothing to record keep compiling
|
||||
* unchanged.
|
||||
*/
|
||||
public MemberSession(String paneId, String terminalId, String profile, MemberRole role,
|
||||
String cwd, String ownerTerminal, long spawnedAtNanos,
|
||||
long lastActivityAtNanos, int turnCount, State state,
|
||||
String worktree, String branch) {
|
||||
this(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, null, null);
|
||||
}
|
||||
|
||||
/** Return a copy of this session in {@code state}. */
|
||||
public MemberSession withState(State state) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId);
|
||||
}
|
||||
|
||||
/** Return a copy with {@code lastActivityAtNanos} updated to {@code nowNanos}. */
|
||||
public MemberSession withActivity(long nowNanos) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId);
|
||||
}
|
||||
|
||||
/** Return a copy with the turn count incremented and activity timestamped at {@code nowNanos}. */
|
||||
public MemberSession bumpTurn(long nowNanos) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount + 1, state, worktree, branch, charterReceipt, agentSessionId);
|
||||
}
|
||||
}
|
||||
@@ -1,8 +1,12 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import dev.ltms.bridged.auth.MemberLifecycle;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.inject.TurnListener;
|
||||
import dev.ltms.bridged.inject.WorkerPresence;
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
@@ -14,9 +18,11 @@ import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
@@ -26,11 +32,10 @@ import java.util.function.LongSupplier;
|
||||
* teardown on top.
|
||||
*
|
||||
* <p>The state machine is intentionally one-shot / no-reuse: every acquired worker is fresh,
|
||||
* and a finished or released worker is torn down, never pooled. {@link #recycle} is a convenience
|
||||
* for {@code release + acquire} with a new distinct pane id.
|
||||
* and a finished or released worker is torn down, never pooled or reused.
|
||||
*
|
||||
* <p>The manager implements {@link TurnListener} so the injector's turn boundaries drive
|
||||
* {@code READY → BUSY → DONE} (or {@code FAILED}). It exposes a {@link WorkerPresence} view via
|
||||
* {@code READY → BUSY → DONE} (or {@code FAILED}). It exposes a {@link MemberPresence} view via
|
||||
* {@link #asPresence()}: any MCP contact from a worker marks it present and simultaneously
|
||||
* transitions the session {@code SPAWNING → READY}.
|
||||
*/
|
||||
@@ -40,122 +45,437 @@ public final class SessionManager implements TurnListener {
|
||||
|
||||
private final PeerLauncher launcher;
|
||||
private final Worktrees worktrees;
|
||||
private final ConcurrentHashMap<String /*paneId*/, WorkerSession> registry = new ConcurrentHashMap<>();
|
||||
private final WorkerPresence presence;
|
||||
private final ConcurrentHashMap<String /*paneId*/, MemberSession> registry = new ConcurrentHashMap<>();
|
||||
private final MemberPresence presence;
|
||||
private final SecureRandom nonceRandom = new SecureRandom();
|
||||
private final AtomicLong nonceSeq = new AtomicLong();
|
||||
private final LongSupplier nowNanos;
|
||||
private final int contextCap;
|
||||
private final boolean clearAfterTurn;
|
||||
private volatile MemberLifecycle memberLifecycle = MemberLifecycle.NONE;
|
||||
/**
|
||||
* CB-586: the repo root the fleet actually works in, remembered the first time a worktree
|
||||
* session is spawned (worktrees are checkouts of it). {@code refs/wip/*} live there, and this
|
||||
* single cached value is what the snapshot retention sweep and the operator-visible census run
|
||||
* against. The daemon is bridged into one project at a time, so "the first worktree's repo" is
|
||||
* the repo; {@code null} until any worktree is spawned, meaning nothing to sweep or measure.
|
||||
*/
|
||||
private volatile String fleetRepoRoot;
|
||||
|
||||
/** CB-520: notified with a terminalId on every acquire; no-op until wired. */
|
||||
private final List<Consumer<String>> acquireListeners = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
/** CB-516: notified with a {@link ReleaseDetail} on every release; no-op until wired. */
|
||||
private final List<Consumer<ReleaseDetail>> releaseListeners = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
|
||||
/** Backward-compatible constructor: shared-tree sessions, production git seam. */
|
||||
public SessionManager(PeerLauncher launcher) {
|
||||
this(launcher, new GitWorktrees(), System::nanoTime, 0);
|
||||
this(launcher, new GitWorktrees(), System::nanoTime, 0, false);
|
||||
}
|
||||
|
||||
/** Backward-compatible constructor with an injectable worktree seam. */
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees) {
|
||||
this(launcher, worktrees, System::nanoTime, 0);
|
||||
this(launcher, worktrees, System::nanoTime, 0, false);
|
||||
}
|
||||
|
||||
/** Test constructor with an injectable clock. */
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos) {
|
||||
this(launcher, worktrees, nowNanos, 0);
|
||||
this(launcher, worktrees, nowNanos, 0, false);
|
||||
}
|
||||
|
||||
/** Production constructor with a configured context turn cap. */
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, int contextCap) {
|
||||
this(launcher, worktrees, System::nanoTime, contextCap);
|
||||
this(launcher, worktrees, System::nanoTime, contextCap, false);
|
||||
}
|
||||
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap) {
|
||||
int contextCap) {
|
||||
this(launcher, worktrees, nowNanos, contextCap, false);
|
||||
}
|
||||
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap, boolean clearAfterTurn) {
|
||||
this.launcher = launcher;
|
||||
this.worktrees = worktrees;
|
||||
this.presence = new PresenceBridge(this);
|
||||
this.nowNanos = nowNanos;
|
||||
this.contextCap = contextCap;
|
||||
this.clearAfterTurn = clearAfterTurn;
|
||||
}
|
||||
|
||||
/**
|
||||
* The single {@link WorkerPresence} view of this manager: it records availability and forwards
|
||||
* The single {@link MemberPresence} view of this manager: it records availability and forwards
|
||||
* the signal to the {@code SPAWNING → READY} transition. Pass this to the {@code Injector} and
|
||||
* {@code BridgeMcp} where they previously accepted a plain {@link WorkerPresence}. The same
|
||||
* {@code BridgeMcp} where they previously accepted a plain {@link MemberPresence}. The same
|
||||
* instance is returned every call — presence is shared state, so a fresh bridge per call would
|
||||
* fragment the {@code present} set and lose signals across callers.
|
||||
*/
|
||||
public WorkerPresence asPresence() {
|
||||
public MemberPresence asPresence() {
|
||||
return presence;
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a worker and register it as {@link WorkerSession.State#SPAWNING}. The caller's
|
||||
* Spawn a worker and register it as {@link MemberSession.State#SPAWNING}. The caller's
|
||||
* identity is recorded as {@code ownerTerminal} ({@code null} for daemon/anon callers).
|
||||
*/
|
||||
public WorkerSession acquire(String profile, String requestedCwd, String callerCwd,
|
||||
public MemberSession acquire(String profile, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal) {
|
||||
return acquire(profile, requestedCwd, callerCwd, ownerTerminal, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a worker, optionally inside a fresh git worktree. When {@code wt} is non-null the
|
||||
* worktree is provisioned, parity-overlaid, and its path becomes the worker's cwd. On any
|
||||
* failure before registration the worktree is removed so no dangling checkout is left.
|
||||
* Spawn a member, optionally inside a fresh git worktree, defaulting the role to
|
||||
* {@link MemberRole#DEV}.
|
||||
*
|
||||
* <p>{@code DEV} is the right default because it is exactly what the old "worker" meant: an
|
||||
* unqualified spawn is a unit of implementation work. An architect or a reviewer is always
|
||||
* asked for on purpose, so neither is ever what a caller silently gets.
|
||||
*/
|
||||
public WorkerSession acquire(String profile, String requestedCwd, String callerCwd,
|
||||
public MemberSession acquire(String profile, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt) {
|
||||
return acquire(profile, MemberRole.DEV, requestedCwd, callerCwd, ownerTerminal, wt);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a member, optionally inside a fresh git worktree, with no session identity requested.
|
||||
* Equivalent to {@link #acquire(String, MemberRole, String, String, String, WorktreeRequest,
|
||||
* String, String)} with both trailing args {@code null}.
|
||||
*
|
||||
* @param profile which backend to run on — a {@code profiles:} key
|
||||
* @param role which contract the member runs under; never {@code null}
|
||||
*/
|
||||
public MemberSession acquire(String profile, MemberRole role, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt) {
|
||||
return acquire(profile, role, requestedCwd, callerCwd, ownerTerminal, wt, null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a member, optionally inside a fresh git worktree, optionally onto a chosen or resumed
|
||||
* agent session (CB-547a / CB-584). When {@code wt} is non-null the worktree is provisioned,
|
||||
* parity-overlaid, and its path becomes the member's cwd. On any failure before registration the
|
||||
* worktree is removed so no dangling checkout is left.
|
||||
*
|
||||
* <p>A non-blank {@code resumeSessionId} requires an explicit {@code profile}: a resumed
|
||||
* conversation is tied to the specific backend that started it, so an unqualified spawn (whose
|
||||
* backend a placement policy picks at spawn time) has no safe candidate to check the capability
|
||||
* against. It also requires that profile's adapter to declare {@link Capability#SESSION_RESUME};
|
||||
* refusing rather than silently delivering a cold session on a non-supporting adapter is the
|
||||
* whole point of checking before spawn, not after (CB-584).
|
||||
*
|
||||
* @param profile which backend to run on — a {@code profiles:} key
|
||||
* @param role which contract the member runs under; never {@code null}
|
||||
* @param sessionName the bridge's logical name for the session, or {@code null}
|
||||
* @param resumeSessionId the prior agent session to resume, or {@code null} for a fresh one
|
||||
* @throws IllegalArgumentException if {@code resumeSessionId} is set with no explicit profile,
|
||||
* or the resolved profile's adapter lacks
|
||||
* {@link Capability#SESSION_RESUME}
|
||||
*/
|
||||
public MemberSession acquire(String profile, MemberRole role, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt,
|
||||
String sessionName, String resumeSessionId) {
|
||||
MemberRole memberRole = (role == null) ? MemberRole.DEV : role;
|
||||
requireResumeCapability(profile, resumeSessionId);
|
||||
if (wt == null) {
|
||||
SpawnRequest req = new SpawnRequest(profile, requestedCwd, callerCwd);
|
||||
PeerHandle handle = launcher.spawn(req);
|
||||
String resolvedProfile = (profile == null || profile.isBlank())
|
||||
? launcher.defaultProfile() : profile;
|
||||
String cwd = launcher.effectiveCwd(req);
|
||||
// CB-557: the role must ride on the SpawnRequest, not stay a local. The launcher needs it
|
||||
// to pick the profile out of that role's pool and to label the tab; a role kept only on
|
||||
// the MemberSession is recorded after the spawn it was supposed to steer.
|
||||
SpawnRequest req = new SpawnRequest(profile, requestedCwd, callerCwd, sessionName, resumeSessionId, memberRole);
|
||||
PeerHandle handle;
|
||||
try {
|
||||
handle = launcher.spawn(req);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("spawn failed for profile={} role={}: {}", profile, memberRole, e.getMessage());
|
||||
throw e;
|
||||
}
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, requestedCwd, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
WorkerSession session = new WorkerSession(
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
memberRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
now,
|
||||
0,
|
||||
WorkerSession.State.SPAWNING,
|
||||
MemberSession.State.SPAWNING,
|
||||
null,
|
||||
null);
|
||||
null,
|
||||
handle.charterReceipt(),
|
||||
handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
memberLifecycle.acquired(session.role(), session.profile(), session.terminalId());
|
||||
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
return acquireWithWorktree(profile, requestedCwd, callerCwd, ownerTerminal, wt);
|
||||
return acquireWithWorktree(profile, memberRole, requestedCwd, callerCwd, ownerTerminal, wt,
|
||||
sessionName, resumeSessionId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse a {@code resumeSessionId} that either names no explicit profile or names one whose
|
||||
* adapter does not declare {@link Capability#SESSION_RESUME}. A no-op when
|
||||
* {@code resumeSessionId} is blank — the ordinary, no-identity spawn path.
|
||||
*/
|
||||
private void requireResumeCapability(String profile, String resumeSessionId) {
|
||||
if (resumeSessionId == null || resumeSessionId.isBlank()) {
|
||||
return;
|
||||
}
|
||||
if (profile == null || profile.isBlank()) {
|
||||
throw new IllegalArgumentException("resumeSessionId requires an explicit profile — "
|
||||
+ "a resumed conversation is tied to the backend that started it, so it cannot "
|
||||
+ "be left to placement to pick");
|
||||
}
|
||||
Set<Capability> caps = launcher.capabilitiesFor(profile);
|
||||
if (!caps.contains(Capability.SESSION_RESUME)) {
|
||||
throw new IllegalArgumentException("worker profile '" + profile + "' does not declare "
|
||||
+ "Capability.SESSION_RESUME — refusing resumeSessionId rather than silently "
|
||||
+ "starting a cold session");
|
||||
}
|
||||
}
|
||||
|
||||
/** Tear a worker down by pane id and remove it from the registry. Idempotent. */
|
||||
public void release(String paneId) {
|
||||
WorkerSession removed = registry.remove(paneId);
|
||||
release(paneId, ReleaseCause.COMPLETED);
|
||||
}
|
||||
|
||||
/**
|
||||
* Core teardown: always stops the worker pane and deregisters the session; whether the worker's
|
||||
* git worktree is also removed depends on {@code cause}.
|
||||
*
|
||||
* <p>CB-544: these are two concerns that used to be fused. Stopping the pane is correct on every
|
||||
* teardown — the worker process must end. Removing the worktree is a destructive act that is only
|
||||
* correct for a deliberately-finished teardown (an explicit stop of a completed session, the
|
||||
* reaper releasing a genuinely idle one, or a context-capped session). A shutdown drain
|
||||
* must stop panes but preserve worktrees: a worker's uncommitted work exists in exactly one
|
||||
* place — its worktree — so deleting it while the daemon simply goes down is silent data loss,
|
||||
* with no copy and no error. Do NOT fuse these back together; the cost of an orphaned worktree
|
||||
* is a logged path an operator can reclaim, the cost of a deleted one is unrecoverable work.
|
||||
*/
|
||||
private void release(String paneId, ReleaseCause cause) {
|
||||
MemberSession removed = registry.remove(paneId);
|
||||
boolean preserveWorktree = cause == ReleaseCause.SHUTDOWN;
|
||||
String snapshotRef = null;
|
||||
if (removed != null) {
|
||||
log.debug("releasing session pane={} terminal={} state={}",
|
||||
removed.paneId(), removed.terminalId(), removed.state());
|
||||
try {
|
||||
memberLifecycle.released(removed.terminalId());
|
||||
log.debug("releasing session pane={} terminal={} state={} cause={}",
|
||||
removed.paneId(), removed.terminalId(), removed.state(), cause);
|
||||
boolean dirty = removed.worktree() != null && worktrees.hasUncommitted(removed.worktree());
|
||||
if (preserveWorktree && removed.worktree() != null) {
|
||||
logPreservedForShutdown(removed);
|
||||
} else if (dirty) {
|
||||
// CB-576: a release that would otherwise remove the worktree finds it holding
|
||||
// uncommitted work the bridge cannot see. A worker that ends a turn without
|
||||
// committing (normally because it stopped to ask a question or refused the turn)
|
||||
// has its only copy of that work in the worktree. Remove would --force-delete it,
|
||||
// so preserve the directory and tell an operator where to find it.
|
||||
preserveWorktree = true;
|
||||
log.warn("release {} preserves dirty worktree {} for pane={} terminal={}: "
|
||||
+ "the worktree holds uncommitted changes that --force remove would destroy",
|
||||
cause, removed.worktree(), removed.paneId(), removed.terminalId());
|
||||
}
|
||||
if (dirty) {
|
||||
// CB-578 stage C: preserving on disk is not saving — the directory is one
|
||||
// `worktree remove --force`, or an operator tidying up, away from gone. Commit
|
||||
// its full state to a ref before the preserve-or-remove decision above can be
|
||||
// undone by anything else, regardless of why this release fired.
|
||||
snapshotRef = trySnapshot(removed, cause);
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
// CB-581: hasUncommitted shells out to `git status` and can throw on a non-zero
|
||||
// exit. We can no longer tell whether the worktree holds uncommitted work, so fail
|
||||
// toward the safe answer and preserve it — deleting on a guess can destroy work
|
||||
// that has no other copy (CB-576), while keeping it on a false alarm only costs
|
||||
// disk. The exception must not propagate: the pane still has to stop below.
|
||||
preserveWorktree = true;
|
||||
log.warn("release {} could not tell whether worktree {} for pane={} terminal={} has "
|
||||
+ "uncommitted changes; preserving it rather than risk destroying unsaved work: {}",
|
||||
cause, removed.worktree(), removed.paneId(), removed.terminalId(), e.toString());
|
||||
} finally {
|
||||
// CB-516/CB-581: a send still waiting on this worker can never be answered now, no
|
||||
// matter what happened above. Tell the listener BEFORE the pane is torn down, so a
|
||||
// blocked caller fails fast with a real reason instead of sitting on a rendezvous
|
||||
// nothing will ever resolve. CB-578 stage C: carry the worktree/branch/snapshot ref
|
||||
// too, so a failed ticket's detail can point a lead at the same tree to re-dispatch.
|
||||
// CB-584 (issue #65 criterion 5): carry agentSessionId alongside them, so a lead can
|
||||
// also resume the member's conversation, not just re-dispatch onto its files.
|
||||
notifyReleased(new ReleaseDetail(removed.terminalId(), removed.worktree(),
|
||||
removed.branch(), snapshotRef, removed.agentSessionId()));
|
||||
}
|
||||
}
|
||||
// CB-581: the pane must always stop, even if the dirty check above threw. A session removed
|
||||
// from the registry with no pane stop is an orphaned pane — a live terminal burning a fleet
|
||||
// slot that no longer appears in the roster and can never be reclaimed.
|
||||
launcher.stop(paneId);
|
||||
if (removed != null && removed.worktree() != null) {
|
||||
if (removed != null && !preserveWorktree && removed.worktree() != null) {
|
||||
worktrees.remove(worktrees.repoRoot(removed.cwd()), removed.worktree());
|
||||
}
|
||||
}
|
||||
|
||||
private WorkerSession acquireWithWorktree(String profile, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt) {
|
||||
String resolvedProfile = (profile == null || profile.isBlank())
|
||||
/**
|
||||
* Best-effort snapshot of a dirty worktree into {@code refs/wip/<branch>} (CB-578 stage C). A
|
||||
* failure here must never escalate: the caller has already decided to preserve the worktree
|
||||
* regardless of whether this succeeds, so the only cost of a failed snapshot is a WARN and a
|
||||
* missing ref — never a lost pane stop or a lost release notification.
|
||||
*/
|
||||
private String trySnapshot(MemberSession session, ReleaseCause cause) {
|
||||
if (session.worktree() == null || session.branch() == null) {
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
Optional<String> ref = worktrees.snapshot(session.worktree(), session.branch(),
|
||||
snapshotMessage(session, cause));
|
||||
ref.ifPresent(sha -> log.info(
|
||||
"snapshotted dirty worktree {} to refs/wip/{} commit={} for pane={} terminal={}",
|
||||
session.worktree(), session.branch(), sha, session.paneId(), session.terminalId()));
|
||||
return ref.orElse(null);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("snapshot of dirty worktree {} failed for pane={} terminal={} branch={}: the "
|
||||
+ "worktree is still preserved on disk, just not committed to refs/wip/{}: {}",
|
||||
session.worktree(), session.paneId(), session.terminalId(), session.branch(),
|
||||
session.branch(), e.toString());
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Commit message for a CB-578 stage C snapshot — names the member so an operator can tell runs apart. */
|
||||
private String snapshotMessage(MemberSession session, ReleaseCause cause) {
|
||||
return "CB-578 stage C: snapshot of a released worker\n\n"
|
||||
+ "terminal: " + session.terminalId() + "\n"
|
||||
+ "profile: " + session.profile() + "\n"
|
||||
+ "branch: " + session.branch() + "\n"
|
||||
+ "cause: " + cause;
|
||||
}
|
||||
|
||||
/**
|
||||
* Facts about a released session that a listener needs beyond the bare terminal id — enough
|
||||
* for a caller to point a lead at where to re-dispatch onto the same tree after a failed
|
||||
* release (CB-578 stage C, acceptance criterion 10). {@code worktreePath} and {@code branch}
|
||||
* are {@code null} for a shared-tree session; {@code snapshotRef} is {@code null} unless this
|
||||
* release snapshotted a dirty worktree into {@code refs/wip/<branch>}.
|
||||
*/
|
||||
public record ReleaseDetail(String terminalId, String worktreePath, String branch, String snapshotRef,
|
||||
String agentSessionId) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Why a session is being released — governs whether its worktree is preserved or removed.
|
||||
* Worktree removal is reserved for the one case that is genuinely finished; everything else
|
||||
* must keep the worker's only copy of its work.
|
||||
*/
|
||||
public enum ReleaseCause {
|
||||
/** Deliberate teardown of a finished session. Stops the pane and removes the worktree. */
|
||||
COMPLETED,
|
||||
/** Daemon shutdown drain. Stops the pane but PRESERVES the worktree. */
|
||||
SHUTDOWN
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-544 shutdown drain log for a worktree we deliberately kept. A session still {@code BUSY}
|
||||
* when the drain timeout expired was abandoned mid-turn — that work may be uncommitted and is
|
||||
* the only copy — so the message is loud and points at the path an operator needs to reclaim.
|
||||
*/
|
||||
private void logPreservedForShutdown(MemberSession session) {
|
||||
if (session.state() == MemberSession.State.BUSY) {
|
||||
log.warn("shutdown drain abandoned BUSY session pane={} terminal={} mid-turn; "
|
||||
+ "worktree preserved at {}", session.paneId(), session.terminalId(),
|
||||
session.worktree());
|
||||
} else {
|
||||
log.info("shutdown drain preserved worktree at {} for pane={}",
|
||||
session.worktree(), session.paneId());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Register a callback invoked with a session's {@code terminalId} whenever it is acquired
|
||||
* (CB-520). This is the hook that lets the reply inbox {@code own} a target's queue.
|
||||
*/
|
||||
public void onAcquire(Consumer<String> listener) {
|
||||
if (listener != null) {
|
||||
acquireListeners.add(listener);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Register a callback invoked with a session's {@code terminalId} whenever it is released
|
||||
* (CB-516). Every teardown path funnels through {@link #release}, so one hook covers the REST
|
||||
* and MCP stop tools, the idle-TTL reaper, and shutdown drain alike.
|
||||
*
|
||||
* <p>Added rather than injected because {@code MessageService} — one intended listener — is
|
||||
* constructed after this manager (it needs the injector and rendezvous, which need the session
|
||||
* presence view this manager exposes). Wiring it at construction would require breaking that
|
||||
* cycle for one callback.
|
||||
*/
|
||||
public void onRelease(Consumer<ReleaseDetail> listener) {
|
||||
if (listener != null) {
|
||||
releaseListeners.add(listener);
|
||||
}
|
||||
}
|
||||
|
||||
/** Inject the optional member-slot lifecycle after construction without changing constructors. */
|
||||
public void setMemberLifecycle(MemberLifecycle memberLifecycle) {
|
||||
this.memberLifecycle = memberLifecycle == null ? MemberLifecycle.NONE : memberLifecycle;
|
||||
}
|
||||
|
||||
/** A listener failure must never prevent the acquisition it is reacting to. */
|
||||
private void notifyAcquired(String terminalId) {
|
||||
if (terminalId == null) {
|
||||
return;
|
||||
}
|
||||
for (Consumer<String> listener : acquireListeners) {
|
||||
try {
|
||||
listener.accept(terminalId);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("acquire listener failed for terminal {}: {}", terminalId, e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** A listener failure must never prevent the teardown it is reacting to. */
|
||||
private void notifyReleased(ReleaseDetail detail) {
|
||||
if (detail.terminalId() == null) {
|
||||
return;
|
||||
}
|
||||
for (Consumer<ReleaseDetail> listener : releaseListeners) {
|
||||
try {
|
||||
listener.accept(detail);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("release listener failed for terminal {}: {}", detail.terminalId(), e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private MemberSession acquireWithWorktree(String profile, MemberRole memberRole, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt,
|
||||
String sessionName, String resumeSessionId) {
|
||||
String preResolvedProfile = (profile == null || profile.isBlank())
|
||||
? launcher.defaultProfile() : profile;
|
||||
String repoRoot = worktrees.repoRoot(firstNonBlank(requestedCwd, callerCwd));
|
||||
// CB-507: resolve through the launcher's CB-112 chain (requested → profile cwd → caller →
|
||||
// daemon cwd → "."), never the raw args. A plain REST spawn supplies neither a requested
|
||||
// nor a caller cwd, so taking the first non-blank of those two yielded null and put
|
||||
// `git -C null` on the command line — an NPE out of ProcessBuilder, surfacing as HTTP 500.
|
||||
// The non-worktree path always used this chain; only this branch was missed.
|
||||
String repoRoot = worktrees.repoRoot(
|
||||
launcher.effectiveCwd(new SpawnRequest(preResolvedProfile, requestedCwd, callerCwd)));
|
||||
if (fleetRepoRoot == null) {
|
||||
// CB-586: remember the repo whose worktrees the fleet spawns — its refs/wip/* are the
|
||||
// snapshot store the retention sweep and the operator census operate on.
|
||||
fleetRepoRoot = repoRoot;
|
||||
}
|
||||
String branch = "worker/" + slug(wt.ticketSlug()) + "-" + nonce();
|
||||
String path = null;
|
||||
PeerHandle handle;
|
||||
try {
|
||||
path = worktrees.add(repoRoot, branch, wt.baseRef());
|
||||
worktrees.overlayParity(repoRoot, path, launcher.parityOverlay(resolvedProfile));
|
||||
handle = launcher.spawn(new SpawnRequest(profile, path, callerCwd));
|
||||
worktrees.overlayParity(repoRoot, path, launcher.parityOverlay(preResolvedProfile));
|
||||
handle = launcher.spawn(new SpawnRequest(profile, path, callerCwd, sessionName, resumeSessionId, memberRole));
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("spawn failed for profile={} role={} branch={} path={}: {}",
|
||||
preResolvedProfile, memberRole, branch, path, e.getMessage());
|
||||
if (path != null) {
|
||||
try {
|
||||
worktrees.remove(repoRoot, path);
|
||||
@@ -165,22 +485,29 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, path, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
WorkerSession session = new WorkerSession(
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
resolveCwd(path, profile, callerCwd),
|
||||
memberRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
now,
|
||||
0,
|
||||
WorkerSession.State.SPAWNING,
|
||||
MemberSession.State.SPAWNING,
|
||||
path,
|
||||
branch);
|
||||
branch,
|
||||
handle.charterReceipt(),
|
||||
handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
memberLifecycle.acquired(session.role(), session.profile(), session.terminalId());
|
||||
log.debug("acquired worktree session id={} terminal={} profile={} branch={} path={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.branch(), session.worktree());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
|
||||
@@ -192,33 +519,29 @@ public final class SessionManager implements TurnListener {
|
||||
return String.format("%06x", nonceRandom.nextInt(1 << 24)) + "-" + nonceSeq.incrementAndGet();
|
||||
}
|
||||
|
||||
private static String firstNonBlank(String... values) {
|
||||
for (String v : values) {
|
||||
if (v != null && !v.isBlank()) return v;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Release the old session and acquire a fresh one with the same profile and working directory.
|
||||
* The new session is guaranteed to have a pane id distinct from the old one (no-reuse invariant).
|
||||
* The profile to record for a session. A launcher that performed dynamic selection tells us
|
||||
* the actual profile via {@link PeerHandle#profile()}; otherwise fall back to what the caller
|
||||
* requested (or the launcher's default for a no-profile spawn).
|
||||
*/
|
||||
public WorkerSession recycle(String paneId) {
|
||||
WorkerSession old = registry.get(paneId);
|
||||
if (old == null) {
|
||||
throw new IllegalArgumentException("no session for paneId " + paneId);
|
||||
private String resolveProfile(PeerHandle handle, String requestedProfile) {
|
||||
String fromHandle = handle.profile();
|
||||
if (fromHandle != null && !fromHandle.isBlank()) {
|
||||
return fromHandle;
|
||||
}
|
||||
release(paneId);
|
||||
return acquire(old.profile(), old.cwd(), old.cwd(), old.ownerTerminal());
|
||||
if (requestedProfile != null && !requestedProfile.isBlank()) {
|
||||
return requestedProfile;
|
||||
}
|
||||
return launcher.defaultProfile();
|
||||
}
|
||||
|
||||
/** The session for {@code paneId}, if it is still registered and not released. */
|
||||
public Optional<WorkerSession> get(String paneId) {
|
||||
public Optional<MemberSession> get(String paneId) {
|
||||
return Optional.ofNullable(registry.get(paneId));
|
||||
}
|
||||
|
||||
/** Bridge-owned roster: all registered sessions (acquired minus released). */
|
||||
public List<WorkerSession> roster() {
|
||||
public List<MemberSession> roster() {
|
||||
return List.copyOf(registry.values());
|
||||
}
|
||||
|
||||
@@ -226,11 +549,15 @@ public final class SessionManager implements TurnListener {
|
||||
* CB-304 merged roster+live view. The registry is authoritative for worktree, branch,
|
||||
* profile, owner, and state; the optional live agent supplies the herdr-reported status.
|
||||
*/
|
||||
public static Map<String, Object> rosterView(WorkerSession session, Agent live) {
|
||||
public static Map<String, Object> rosterView(MemberSession session, Agent live) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("sessionId", session.terminalId());
|
||||
m.put("paneId", session.paneId());
|
||||
// Both axes, always: profile says which backend this member runs on, role says what it is
|
||||
// for. A lead reading the roster needs both — two rows may share a profile and still be
|
||||
// allowed to do entirely different things.
|
||||
m.put("profile", session.profile());
|
||||
m.put("role", session.role() == null ? "dev" : session.role().wireName());
|
||||
m.put("state", session.state().name().toLowerCase());
|
||||
if (session.worktree() != null) {
|
||||
m.put("worktree", session.worktree());
|
||||
@@ -241,13 +568,29 @@ public final class SessionManager implements TurnListener {
|
||||
if (session.ownerTerminal() != null) {
|
||||
m.put("owner", session.ownerTerminal());
|
||||
}
|
||||
// CB-584: which conversation this member holds — the id a later resumeSessionId spawn would
|
||||
// pass back. Absent for an adapter that declines Capability.SESSION_RESUME, or one that
|
||||
// resolves it lazily and has not yet.
|
||||
if (session.agentSessionId() != null) {
|
||||
m.put("agentSessionId", session.agentSessionId());
|
||||
}
|
||||
// CB-571: which charter this member was started with — never the charter text itself. The
|
||||
// digest lets a lead tell at a glance whether all members got the same charter; the source
|
||||
// records whether a role charter was configured ("fleet.charters.<role>") or only the reply
|
||||
// charter was composed ("none").
|
||||
if (session.charterReceipt() != null) {
|
||||
m.put("charterSource", session.charterReceipt().charterSource());
|
||||
if (session.charterReceipt().charterSha256() != null) {
|
||||
m.put("charterSha256", session.charterReceipt().charterSha256());
|
||||
}
|
||||
}
|
||||
m.put("liveStatus", live == null ? "unknown" : live.status().name().toLowerCase());
|
||||
return m;
|
||||
}
|
||||
|
||||
/** Lifecycle hook: worker became available on the bridge MCP. */
|
||||
void onReady(String terminalId) {
|
||||
transitionByTerminal(terminalId, WorkerSession.State.SPAWNING, WorkerSession.State.READY);
|
||||
transitionByTerminal(terminalId, MemberSession.State.SPAWNING, MemberSession.State.READY);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -256,14 +599,14 @@ public final class SessionManager implements TurnListener {
|
||||
* can be re-delivered for multi-turn reuse until it is released.
|
||||
*/
|
||||
@Override
|
||||
public void onDelivered(String target) {
|
||||
WorkerSession current = findByTerminal(target);
|
||||
public void onDelivered(String target, TurnToken token) {
|
||||
MemberSession current = findByTerminal(target);
|
||||
if (current == null) return;
|
||||
if (current.state() != WorkerSession.State.READY && current.state() != WorkerSession.State.DONE) {
|
||||
if (current.state() != MemberSession.State.READY && current.state() != MemberSession.State.DONE) {
|
||||
return;
|
||||
}
|
||||
long now = nowNanos.getAsLong();
|
||||
WorkerSession updated = current.withState(WorkerSession.State.BUSY).bumpTurn(now);
|
||||
MemberSession updated = current.withState(MemberSession.State.BUSY).bumpTurn(now);
|
||||
if (replace(current, updated)) {
|
||||
log.debug("session transitioned terminal={} pane={} {} -> BUSY turn={}",
|
||||
target, current.paneId(), current.state(), updated.turnCount());
|
||||
@@ -273,16 +616,42 @@ public final class SessionManager implements TurnListener {
|
||||
/** Lifecycle hook: the worker's delegated turn completed successfully. */
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
WorkerSession current = findByTerminal(target);
|
||||
if (current == null || current.state() != WorkerSession.State.BUSY) return;
|
||||
completeTurn(target, false);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
if (!clearAfterTurn) return false;
|
||||
MemberSession current = findByTerminal(target);
|
||||
return current != null && current.state() == MemberSession.State.BUSY
|
||||
&& (contextCap <= 0 || current.turnCount() < contextCap);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
return completeTurn(target, true);
|
||||
}
|
||||
|
||||
private boolean completeTurn(String target, boolean startContextReset) {
|
||||
MemberSession current = findByTerminal(target);
|
||||
if (current == null || current.state() != MemberSession.State.BUSY) return false;
|
||||
long now = nowNanos.getAsLong();
|
||||
WorkerSession updated = current.withState(WorkerSession.State.DONE).withActivity(now);
|
||||
MemberSession updated = current.withState(MemberSession.State.DONE).withActivity(now);
|
||||
if (replace(current, updated)) {
|
||||
log.debug("session transitioned terminal={} pane={} BUSY -> DONE turn={}",
|
||||
target, current.paneId(), updated.turnCount());
|
||||
}
|
||||
if (contextCap > 0 && updated.turnCount() >= contextCap) {
|
||||
release(current.paneId());
|
||||
return false;
|
||||
}
|
||||
if (!startContextReset || !clearAfterTurn) return false;
|
||||
try {
|
||||
return launcher.clearContext(current.paneId());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("context reset failed for terminal={} pane={}; continuing without reset: {}",
|
||||
target, current.paneId(), e.getMessage());
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -294,11 +663,13 @@ public final class SessionManager implements TurnListener {
|
||||
|
||||
/** Lifecycle hook: the worker vanished or was dropped mid-life. */
|
||||
void onFailed(String target) {
|
||||
WorkerSession current = findByTerminal(target);
|
||||
MemberSession current = findByTerminal(target);
|
||||
if (current == null) return;
|
||||
if (current.state() == WorkerSession.State.RELEASED) return;
|
||||
if (replace(current, current.withState(WorkerSession.State.FAILED))) {
|
||||
log.debug("session marked failed terminal={} pane={}", target, current.paneId());
|
||||
if (current.state() == MemberSession.State.RELEASED) return;
|
||||
MemberSession.State priorState = current.state();
|
||||
if (replace(current, current.withState(MemberSession.State.FAILED))) {
|
||||
log.warn("member terminal={} pane={} can no longer be delegated to: its turn never resolved "
|
||||
+ "(was {} when it failed)", target, current.paneId(), priorState);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -310,32 +681,49 @@ public final class SessionManager implements TurnListener {
|
||||
int reapIdle(long idleTtlNanos) {
|
||||
long now = nowNanos.getAsLong();
|
||||
int reaped = 0;
|
||||
for (WorkerSession s : roster()) {
|
||||
if (s.state() != WorkerSession.State.READY && s.state() != WorkerSession.State.DONE) {
|
||||
for (MemberSession s : roster()) {
|
||||
if (s.state() != MemberSession.State.READY && s.state() != MemberSession.State.DONE) {
|
||||
continue;
|
||||
}
|
||||
if (now - s.lastActivityAtNanos() > idleTtlNanos) {
|
||||
release(s.paneId());
|
||||
reaped++;
|
||||
long idleNanos = now - s.lastActivityAtNanos();
|
||||
if (idleNanos > idleTtlNanos) {
|
||||
log.debug("reaping idle session terminal={} pane={}: idle {}s exceeds the {}s ttl",
|
||||
s.terminalId(), s.paneId(), TimeUnit.NANOSECONDS.toSeconds(idleNanos),
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos));
|
||||
// CB-581: one session that fails to release must not abort the whole reaping pass —
|
||||
// match drainAll's per-session try/catch so the rest of the roster still gets reaped.
|
||||
try {
|
||||
release(s.paneId());
|
||||
reaped++;
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("reap failed for pane={} terminal={} worktree={}; continuing with "
|
||||
+ "remaining sessions", s.paneId(), s.terminalId(), s.worktree(), e);
|
||||
}
|
||||
}
|
||||
}
|
||||
return reaped;
|
||||
}
|
||||
|
||||
/**
|
||||
* Gracefully drain all registered sessions. For each session that is {@code BUSY}, poll up to
|
||||
* {@code timeoutNanos} for it to leave {@code BUSY}, then release it regardless. Non-busy
|
||||
* sessions are released immediately. A failure releasing one session is logged and does not
|
||||
* abort the rest.
|
||||
* Gracefully drain all registered sessions on daemon shutdown. For each session that is
|
||||
* {@code BUSY}, poll up to {@code timeoutNanos} for it to leave {@code BUSY}, then release it
|
||||
* regardless. Non-busy sessions are released immediately. A failure releasing one session is
|
||||
* logged and does not abort the rest.
|
||||
*
|
||||
* <p>CB-544: this is a {@link ReleaseCause#SHUTDOWN} release — the worker's pane is stopped
|
||||
* (the process must end) but its worktree is preserved and its path logged. Shutdown is never
|
||||
* a reason to delete a worker's only copy of its uncommitted work. A session still {@code BUSY}
|
||||
* when the timeout expired is abandoned mid-turn and logged loudly so an operator can find its
|
||||
* kept worktree.
|
||||
*/
|
||||
void drainAll(long timeoutNanos) {
|
||||
long deadline = System.nanoTime() + timeoutNanos;
|
||||
for (WorkerSession s : roster()) {
|
||||
for (MemberSession s : roster()) {
|
||||
try {
|
||||
if (s.state() == WorkerSession.State.BUSY) {
|
||||
if (s.state() == MemberSession.State.BUSY) {
|
||||
while (System.nanoTime() < deadline) {
|
||||
WorkerSession current = registry.get(s.paneId());
|
||||
if (current == null || current.state() != WorkerSession.State.BUSY) {
|
||||
MemberSession current = registry.get(s.paneId());
|
||||
if (current == null || current.state() != MemberSession.State.BUSY) {
|
||||
break;
|
||||
}
|
||||
try {
|
||||
@@ -347,7 +735,7 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
}
|
||||
}
|
||||
release(s.paneId());
|
||||
release(s.paneId(), ReleaseCause.SHUTDOWN);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("drain failed for pane={}; continuing with remaining sessions", s.paneId(), e);
|
||||
}
|
||||
@@ -368,16 +756,48 @@ public final class SessionManager implements TurnListener {
|
||||
return registry.size();
|
||||
}
|
||||
|
||||
private WorkerSession findByTerminal(String terminalId) {
|
||||
for (WorkerSession s : registry.values()) {
|
||||
/**
|
||||
* CB-586: the operator-visible census of {@code refs/wip/*} in the repo the fleet works in —
|
||||
* how many snapshot refs exist and roughly what they cost. Empty (no repo known) until at
|
||||
* least one worktree session has been spawned, exactly so a fleet that has never snapshotted
|
||||
* anything surfaces nothing new, as it did before CB-586.
|
||||
*/
|
||||
public Optional<Worktrees.WipRefStats> wipRefs() {
|
||||
String repo = fleetRepoRoot;
|
||||
return repo == null ? Optional.empty() : Optional.of(worktrees.wipRefs(repo));
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-586: run the snapshot retention sweep in the fleet's repo (a no-op until a worktree has
|
||||
* been spawned, which establishes the repo). Returns how many {@code refs/wip/*} it deleted.
|
||||
*/
|
||||
public int sweepWipRefs(long minAgeMillis) {
|
||||
String repo = fleetRepoRoot;
|
||||
return repo == null ? 0 : worktrees.pruneWipRefs(repo, minAgeMillis);
|
||||
}
|
||||
|
||||
/**
|
||||
* The registered session owning {@code terminalId}, or {@code null} if none does.
|
||||
*
|
||||
* <p>A null {@code terminalId} is a normal input, not a caller bug: every lifecycle hook here is
|
||||
* fed from the MCP transport, where the <em>primary</em> resolves to a {@link
|
||||
* dev.ltms.bridged.auth.Principal} with no terminal. {@code BridgeMcp} documents that contact as
|
||||
* a no-op, and {@link dev.ltms.bridged.inject.MemberPresence#markPresent} honours it — but
|
||||
* {@code PresenceBridge} then forwards the same null here. Matching on a null id can never
|
||||
* succeed anyway (a registered session always has a terminal), so answer "no match" rather than
|
||||
* throwing: an NPE on this path takes down an unrelated tool call for the primary.
|
||||
*/
|
||||
private MemberSession findByTerminal(String terminalId) {
|
||||
if (terminalId == null) return null;
|
||||
for (MemberSession s : registry.values()) {
|
||||
if (terminalId.equals(s.terminalId())) return s;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private void transitionByTerminal(String terminalId, WorkerSession.State from,
|
||||
WorkerSession.State to) {
|
||||
WorkerSession current = findByTerminal(terminalId);
|
||||
private void transitionByTerminal(String terminalId, MemberSession.State from,
|
||||
MemberSession.State to) {
|
||||
MemberSession current = findByTerminal(terminalId);
|
||||
if (current == null || current.state() != from) return;
|
||||
long now = nowNanos.getAsLong();
|
||||
if (replace(current, current.withState(to).withActivity(now))) {
|
||||
@@ -386,16 +806,12 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
}
|
||||
|
||||
private boolean replace(WorkerSession expected, WorkerSession updated) {
|
||||
private boolean replace(MemberSession expected, MemberSession updated) {
|
||||
return registry.replace(expected.paneId(), expected, updated);
|
||||
}
|
||||
|
||||
private String resolveCwd(String requestedCwd, String profileName, String callerCwd) {
|
||||
return launcher.effectiveCwd(new SpawnRequest(profileName, requestedCwd, callerCwd));
|
||||
}
|
||||
|
||||
/** WorkerPresence bridge that also drives the manager's READY transition. */
|
||||
private static final class PresenceBridge extends WorkerPresence {
|
||||
/** MemberPresence bridge that also drives the manager's READY transition. */
|
||||
private static final class PresenceBridge extends MemberPresence {
|
||||
private final SessionManager sessions;
|
||||
|
||||
PresenceBridge(SessionManager sessions) {
|
||||
@@ -404,6 +820,9 @@ public final class SessionManager implements TurnListener {
|
||||
|
||||
@Override
|
||||
public void markPresent(String terminal) {
|
||||
if (terminal == null || terminal.isBlank()) {
|
||||
return; // the primary's contact carries no worker terminal — not a readiness signal
|
||||
}
|
||||
super.markPresent(terminal);
|
||||
sessions.onReady(terminal);
|
||||
}
|
||||
|
||||
@@ -14,12 +14,29 @@ public final class SessionReaper {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(SessionReaper.class);
|
||||
private static final long DEFAULT_INTERVAL_MILLIS = 5000;
|
||||
/** CB-586: the refs/wip age floor — never sweep a snapshot younger than 24h (the CB-586 rule). */
|
||||
private static final long WIP_MIN_AGE_MILLIS = TimeUnit.HOURS.toMillis(24);
|
||||
/**
|
||||
* CB-586: how often the retention sweep runs. Given the 24h age floor, running it every few
|
||||
* hours means a ref is dropped within hours of becoming eligible, never within minutes.
|
||||
*/
|
||||
private static final long WIP_SWEEP_INTERVAL_NANOS = TimeUnit.HOURS.toNanos(6);
|
||||
|
||||
private final SessionManager sessions;
|
||||
private final long idleTtlNanos;
|
||||
private final long intervalMillis;
|
||||
private volatile boolean running;
|
||||
private Thread thread;
|
||||
/**
|
||||
* When the retention sweep last ran, and whether it ever has. The flag is not a convenience:
|
||||
* a "never yet" sentinel value cannot be compared by subtraction. {@code Long.MIN_VALUE} was
|
||||
* the obvious choice and it silently overflows — {@code System.nanoTime()} is positive on this
|
||||
* platform, so {@code now - Long.MIN_VALUE} wraps to a large negative number, the interval gate
|
||||
* reads it as "swept moments ago", and it returns before ever assigning the field. The sweep
|
||||
* then never runs at all, for the life of the process, with nothing in the log to say so.
|
||||
*/
|
||||
private volatile boolean sweptOnce;
|
||||
private volatile long lastWipSweepNanos;
|
||||
|
||||
/** Construct a reaper with the default 5-second polling interval. */
|
||||
public SessionReaper(SessionManager sessions, long idleTtlSeconds) {
|
||||
@@ -49,10 +66,39 @@ public final class SessionReaper {
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("session reaper iteration failed; continuing", e);
|
||||
}
|
||||
maybeSweepWipRefs();
|
||||
sleep();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-586: run the refs/wip retention sweep on a slow cadence (hours, not the per-iteration
|
||||
* millisecond loop). Best-effort — a failure must never take the idle-reap loop down with it.
|
||||
*/
|
||||
private void maybeSweepWipRefs() {
|
||||
long now = System.nanoTime();
|
||||
// The first pass always sweeps: a restart is a fine moment to sweep, the 24h age floor
|
||||
// makes it safe, and it means the feature is observable right after a redeploy instead of
|
||||
// six hours later. Only after that does the interval gate apply, and by then both operands
|
||||
// come from nanoTime, so the subtraction is well-defined.
|
||||
if (sweptOnce && now - lastWipSweepNanos < WIP_SWEEP_INTERVAL_NANOS) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
int deleted = sessions.sweepWipRefs(WIP_MIN_AGE_MILLIS);
|
||||
if (deleted > 0) {
|
||||
log.info("refs/wip retention sweep deleted {} snapshot ref(s) older than 24h whose "
|
||||
+ "content was already reachable from main", deleted);
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("refs/wip retention sweep failed; continuing", e);
|
||||
}
|
||||
// Set even when the sweep threw, so a broken repo is retried on the slow cadence rather
|
||||
// than hammering git on every 5-second iteration.
|
||||
lastWipSweepNanos = now;
|
||||
sweptOnce = true;
|
||||
}
|
||||
|
||||
private void sleep() {
|
||||
try {
|
||||
Thread.sleep(intervalMillis);
|
||||
|
||||
@@ -1,58 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
/**
|
||||
* A bridge-owned worker session — the authoritative in-daemon record of a worker this
|
||||
* process spawned. Immutable; state transitions are performed by replacing the record in
|
||||
* {@link SessionManager}'s registry.
|
||||
*
|
||||
* @param paneId herdr pane handle — the registry key and the argument to teardown
|
||||
* @param terminalId herdr terminal handle — the {@code target} for send/read/status
|
||||
* @param profile the worker profile name that spawned this session
|
||||
* @param cwd the resolved working directory the worker started in
|
||||
* @param ownerTerminal the caller that requested this worker ({@code null} = daemon/anon)
|
||||
* @param spawnedAtNanos {@link System#nanoTime()} when the session was registered
|
||||
* @param lastActivityAtNanos {@link System#nanoTime()} of the most recent lifecycle event
|
||||
* @param turnCount number of delegated turns that have been delivered to this session
|
||||
* @param state current lifecycle state in the one-shot FSM
|
||||
*/
|
||||
public record WorkerSession(
|
||||
String paneId,
|
||||
String terminalId,
|
||||
String profile,
|
||||
String cwd,
|
||||
String ownerTerminal,
|
||||
long spawnedAtNanos,
|
||||
long lastActivityAtNanos,
|
||||
int turnCount,
|
||||
State state,
|
||||
String worktree,
|
||||
String branch) {
|
||||
|
||||
/** One-shot worker lifecycle states. */
|
||||
public enum State {
|
||||
SPAWNING,
|
||||
READY,
|
||||
BUSY,
|
||||
DONE,
|
||||
FAILED,
|
||||
RELEASED
|
||||
}
|
||||
|
||||
/** Return a copy of this session in {@code state}. */
|
||||
public WorkerSession withState(State state) {
|
||||
return new WorkerSession(paneId, terminalId, profile, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch);
|
||||
}
|
||||
|
||||
/** Return a copy with {@code lastActivityAtNanos} updated to {@code nowNanos}. */
|
||||
public WorkerSession withActivity(long nowNanos) {
|
||||
return new WorkerSession(paneId, terminalId, profile, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount, state, worktree, branch);
|
||||
}
|
||||
|
||||
/** Return a copy with the turn count incremented and activity timestamped at {@code nowNanos}. */
|
||||
public WorkerSession bumpTurn(long nowNanos) {
|
||||
return new WorkerSession(paneId, terminalId, profile, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount + 1, state, worktree, branch);
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Optional;
|
||||
|
||||
/** Seam between {@link SessionManager} and git worktree operations. Tests use a recording fake. */
|
||||
public interface Worktrees {
|
||||
@@ -10,9 +11,91 @@ public interface Worktrees {
|
||||
/** git -C <repoRoot> worktree remove --force <path>. Idempotent (already-gone tolerated). */
|
||||
void remove(String repoRoot, String worktreePath);
|
||||
|
||||
/**
|
||||
* True when the worktree holds uncommitted changes the bridge cannot see: tracked
|
||||
* modifications, staged files, or untracked files. {@code git status --porcelain} is the
|
||||
* test; an empty result means clean. Callers use this to decide whether removing the
|
||||
* worktree would silently destroy a worker's only copy of its work.
|
||||
*
|
||||
* <p>An already-gone worktree is reported as clean (no throw), matching {@link #remove}'s
|
||||
* idempotent contract: a path that does not exist holds no work to lose, and must not break
|
||||
* a teardown that is mid-way through stopping the pane.
|
||||
*/
|
||||
boolean hasUncommitted(String worktreePath);
|
||||
|
||||
/** Copy each existing overlay path repoRoot→worktree; mark tracked ones --skip-worktree. */
|
||||
void overlayParity(String repoRoot, String worktreePath, List<String> overlay);
|
||||
|
||||
/** git -C <cwd> rev-parse --show-toplevel — the repo root that owns cwd. */
|
||||
String repoRoot(String cwd);
|
||||
|
||||
/**
|
||||
* Commit the worktree's full on-disk state — tracked and untracked, respecting
|
||||
* {@code .gitignore} — to {@code refs/wip/<branch>}, so a release that would otherwise leave
|
||||
* the work as loose, unprotected files has a durable git object to fall back on (CB-578 stage
|
||||
* C). Built on a <em>temporary</em> index: the worker's own index, working tree, and HEAD are
|
||||
* never touched, since the worker may still be mid-write. The commit is parented on the
|
||||
* worktree's current HEAD.
|
||||
*
|
||||
* <p>Never writes under {@code refs/heads/} — the ref must not appear in {@code git branch},
|
||||
* must not be pushed by default, and must not be swept by a later {@code git branch -d}.
|
||||
*
|
||||
* <p>Callers are expected to have already confirmed {@link #hasUncommitted} before reaching
|
||||
* for this; it always stages and commits whatever {@code git add -A} finds, so calling it on
|
||||
* a clean worktree still produces a (harmless, tree-identical-to-HEAD) commit rather than
|
||||
* detecting cleanliness itself.
|
||||
*
|
||||
* @param worktreePath absolute path of the worktree to snapshot
|
||||
* @param branch the worktree's own branch — keys {@code refs/wip/<branch>}
|
||||
* @param message the commit message; should name the member, its branch, and the release
|
||||
* cause so an operator can tell which run produced it
|
||||
* @return the created commit's sha, or {@link Optional#empty()} if {@code worktreePath} does
|
||||
* not exist (mirrors {@link #remove} and {@link #hasUncommitted}'s already-gone
|
||||
* tolerance — a worktree that is gone holds nothing to snapshot)
|
||||
*/
|
||||
Optional<String> snapshot(String worktreePath, String branch, String message);
|
||||
|
||||
/**
|
||||
* CB-586: how many {@code refs/wip/*} snapshot refs exist in {@code repoRoot} and roughly what
|
||||
* they cost. This is the operator-visible surface for the snapshot growth CB-578 stage C left
|
||||
* behind — counts of refs alone hide that each one pins a whole tree for {@code git gc}.
|
||||
*
|
||||
* @param repoRoot the repository to scan
|
||||
* @return count of snapshot refs, and {@code costBytes} = the approximate total working-tree
|
||||
* size of every snapshot's committed content (summed per ref, so shared objects are
|
||||
* counted once per ref that carries them)
|
||||
*/
|
||||
WipRefStats wipRefs(String repoRoot);
|
||||
|
||||
/**
|
||||
* CB-586: run the {@code refs/wip/*} retention sweep and return how many refs it deleted.
|
||||
*
|
||||
* <p>The retention rule is <em>reachability plus an age floor</em>. A snapshot ref is deleted
|
||||
* only when <strong>both</strong> hold:
|
||||
* <ol>
|
||||
* <li>its commit's <em>tree content</em> is already reachable from {@code main} — the work
|
||||
* the snapshot preserved has been recovered, so dropping the ref loses nothing; and</li>
|
||||
* <li>the ref is older than {@code minAgeMillis} — a very recent snapshot is never swept
|
||||
* while a lead may still be looking at it.</li>
|
||||
* </ol>
|
||||
*
|
||||
* <p>Reachability is the safety property. A snapshot exists precisely because the work was not
|
||||
* committed anywhere else, so a snapshot whose content is <em>not</em> reachable from
|
||||
* {@code main} is the <strong>last copy</strong> of a worker's work and must never be deleted
|
||||
* automatically — that is the failure CB-576 and CB-578 stage C were built to stop. Age alone
|
||||
* must never drive a deletion, because age-based sweeping is exactly how the last copy gets
|
||||
* destroyed. (Both numbers and the rule are CB-586's decision; this method only implements it.)
|
||||
*
|
||||
* <p>Every deletion logs the ref name and the commit sha, so an operator who finds they lost
|
||||
* the wrong thing can still recover it from git's reflog.
|
||||
*
|
||||
* @param repoRoot the repository whose {@code refs/wip/*} to sweep
|
||||
* @param minAgeMillis the age floor; a ref younger than this is never touched
|
||||
* @return the number of snapshot refs deleted
|
||||
*/
|
||||
int pruneWipRefs(String repoRoot, long minAgeMillis);
|
||||
|
||||
/** CB-586: the operator-visible census of {@code refs/wip/*} in one repository. */
|
||||
record WipRefStats(int count, long costBytes) {
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,488 +0,0 @@
|
||||
package dev.ltms.bridged.worker;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.Tab;
|
||||
import dev.ltms.bridged.herdr.Workspace;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.security.SecureRandom;
|
||||
import java.util.ArrayList;
|
||||
import java.util.EnumSet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Function;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Spawns and lists worker sessions — the safe path from a delegation request to a
|
||||
* running off-subscription Claude.
|
||||
*
|
||||
* <p>The spawn sequence encodes the subscription boundary: build the worker env with
|
||||
* {@code ANTHROPIC_BASE_URL}, assert that host is on the allowlist <em>before</em>
|
||||
* touching herdr, and only then {@code agent.start}. A worker's base_url lives in the
|
||||
* env map handed to herdr and nowhere else; {@code bridged}'s own environment is never
|
||||
* mutated.
|
||||
*
|
||||
* <p>Placement: in the default {@code tab} policy a worker lands in its own tab inside a
|
||||
* dedicated worker space (found-or-created once, then shared), so workers never split or
|
||||
* clutter the user's real work spaces. Teardown removes the worker's pane <em>and</em> its
|
||||
* now-empty tab, tolerating an already-gone worker so a repeated DELETE is harmless.
|
||||
*/
|
||||
public final class ClaudeCodeLauncher implements PeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ClaudeCodeLauncher.class);
|
||||
|
||||
/** herdr rejects a duplicate agent {@code name}; we retry a bumped name this many times. */
|
||||
private static final int NAME_RETRIES = 8;
|
||||
|
||||
/**
|
||||
* A bridge-spawned worker label {@code claude-<profile>-<nonce>-<seq>} (see
|
||||
* {@link #startUniquelyNamed}); group 1 captures the 6-hex per-process {@code nonce}. The
|
||||
* profile segment may itself contain {@code -}, so the nonce/seq are anchored at the tail.
|
||||
* Names not matching this shape are not workers we started and are never reaped (CB-117).
|
||||
*/
|
||||
private static final Pattern WORKER_NAME = Pattern.compile("claude-.*-([0-9a-f]{6})-\\d+");
|
||||
|
||||
private final AgentControl agents;
|
||||
private final WorkspaceControl spaces;
|
||||
private final SubscriptionGuard guard;
|
||||
private final Map<String, BridgedConfig.Worker> profiles; // profile name → spawn settings
|
||||
private final String defaultProfile; // profile a no-arg spawn uses (nullable)
|
||||
private final Function<String, String> env; // host env lookup (injectable for tests)
|
||||
private final AtomicLong nameSeq = new AtomicLong(); // per-worker counter (also the tab #)
|
||||
|
||||
/**
|
||||
* Standing instruction appended to the worker's system prompt so it returns its result via
|
||||
* {@code bridge_reply}. Injected as a launch flag, so nothing is written to the worker's
|
||||
* profile — it is guidance, and a worker that never replies is caught by the send's timeout.
|
||||
*/
|
||||
static final String REPLY_CHARTER =
|
||||
"You are an off-subscription worker in the claude-bridge fleet. Every message you "
|
||||
+ "receive arrives through the bridge, and the ONLY channel back to the sender is the "
|
||||
+ "bridge_reply MCP tool. Text you write in your terminal is NOT sent anywhere — the "
|
||||
+ "sender cannot see your screen, so an in-terminal answer is silently discarded. "
|
||||
+ "Therefore you MUST end EVERY turn by calling bridge_reply with `content` set to your "
|
||||
+ "complete response. This holds for every message without exception — tasks, questions, "
|
||||
+ "clarifications, acknowledgements, and ordinary back-and-forth conversation. Call "
|
||||
+ "bridge_reply exactly once, as the final action of your turn, with your full answer in "
|
||||
+ "`content`; never wait for confirmation first. If you end a turn without calling "
|
||||
+ "bridge_reply, the sender receives nothing and the exchange stalls.";
|
||||
|
||||
// Per-process token mixed into each worker name so a fresh process (nameSeq back at 0)
|
||||
// cannot collide with same-profile workers that outlived a restart. See startUniquelyNamed.
|
||||
private final String nameNonce = String.format("%06x", new SecureRandom().nextInt(1 << 24));
|
||||
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Worker> profiles, String defaultProfile,
|
||||
Function<String, String> env) {
|
||||
this.agents = agents;
|
||||
this.spaces = spaces;
|
||||
this.guard = guard;
|
||||
this.profiles = Map.copyOf(profiles);
|
||||
this.defaultProfile = defaultProfile;
|
||||
this.env = env;
|
||||
}
|
||||
|
||||
/** The configured worker profile names (what {@code spawn(profile)} accepts). */
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return profiles.keySet();
|
||||
}
|
||||
|
||||
/** The parity-overlay file list for {@code profileName} (default list when unset). */
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
String name = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (name == null || name.isBlank()) {
|
||||
return List.of();
|
||||
}
|
||||
BridgedConfig.Worker cfg = profiles.get(name);
|
||||
return cfg == null ? List.of() : cfg.parityOverlay();
|
||||
}
|
||||
|
||||
/** The profile a no-argument {@link #spawn()} uses, or {@code null} if none is configured. */
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return defaultProfile;
|
||||
}
|
||||
|
||||
/** Spawn a worker for the default profile in the resolved default cwd. */
|
||||
public Agent spawn() {
|
||||
return spawn(null, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile (null → default) in the resolved default cwd. */
|
||||
public Agent spawn(String profileName) {
|
||||
return spawn(profileName, null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a worker. {@code profileName} null/blank → the default profile. The worker's working
|
||||
* directory (CB-112) is resolved by {@link #resolveCwd}: an explicit {@code requestedCwd} (a
|
||||
* spawn argument), else the profile's configured {@code cwd}, else {@code callerCwd} (the
|
||||
* primary's cwd, when the spawn came from the primary over MCP), else the daemon's cwd — never
|
||||
* assumed to be {@code $HOME}. Guard runs before any herdr call.
|
||||
*/
|
||||
public Agent spawn(String profileName, String requestedCwd, String callerCwd) {
|
||||
String name = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (name == null || name.isBlank()) {
|
||||
throw new IllegalArgumentException("no default worker profile is configured — "
|
||||
+ "pass a profile; configured: " + profiles.keySet());
|
||||
}
|
||||
BridgedConfig.Worker cfg = profiles.get(name);
|
||||
if (cfg == null) {
|
||||
throw new IllegalArgumentException("unknown worker profile '" + name
|
||||
+ "' — configured: " + profiles.keySet());
|
||||
}
|
||||
String baseUrl = cfg.baseUrl();
|
||||
guard.assertWorker(baseUrl); // hard stop before we spawn anything
|
||||
|
||||
Map<String, String> workerEnv = new LinkedHashMap<>();
|
||||
workerEnv.put("ANTHROPIC_BASE_URL", baseUrl);
|
||||
putIfPresent(workerEnv, "ANTHROPIC_MODEL", cfg.model());
|
||||
putIfPresent(workerEnv, "CLAUDE_CONFIG_DIR", cfg.configDir());
|
||||
String token = env.apply(cfg.tokenEnv());
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", token);
|
||||
|
||||
// CB-302: the worker checkpoint (commit → push → open its own PR). Push is free over SSH;
|
||||
// the only incremental grant is PR-create, a repo-scoped forge token injected here — opt-in
|
||||
// per profile via gitTokenEnv, and never mutating bridged's own env. The paired forge host
|
||||
// rides along only when a token is actually granted, so non-implementer profiles get neither.
|
||||
if (cfg.hasGitToken()) {
|
||||
String gitToken = resolveEnv(cfg.gitTokenEnv());
|
||||
if (gitToken != null) {
|
||||
workerEnv.put("GITEA_TOKEN", gitToken);
|
||||
putIfPresent(workerEnv, "GITEA_HOST", resolveEnv(cfg.gitHostEnv()));
|
||||
}
|
||||
}
|
||||
|
||||
// Mount the bridge MCP + reply charter as launch FLAGS (non-invasive: nothing written to
|
||||
// the worker's profile/config dir). Identity is connection-based, so the mount is shared.
|
||||
List<String> argv = argvWithBridge(cfg);
|
||||
String cwd = resolveCwd(requestedCwd, cfg, callerCwd);
|
||||
|
||||
return cfg.tabPlacement()
|
||||
? spawnInTab(cfg, workerEnv, argv, cwd)
|
||||
: spawnAsPane(cfg, workerEnv, argv, cwd);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-112 cwd resolution: spawn arg → profile config → the primary's cwd → the daemon's cwd.
|
||||
* Never returns {@code null}/blank: {@code "."} (the daemon's own working directory) is the
|
||||
* guaranteed last resort so a pathological environment with an unset {@code user.dir} still
|
||||
* honours the "never assume {@code $HOME}" contract rather than letting herdr default the pane.
|
||||
*/
|
||||
private static String resolveCwd(String requestedCwd, BridgedConfig.Worker cfg, String callerCwd) {
|
||||
return firstNonBlank(requestedCwd, cfg.cwd(), callerCwd, System.getProperty("user.dir"), ".");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-301: the effective working directory a spawn for {@code profileName} would use, without
|
||||
* actually spawning. Used by {@link dev.ltms.bridged.session.SessionManager} to record the
|
||||
* resolved cwd in the session registry.
|
||||
*/
|
||||
public String effectiveCwd(String profileName, String requestedCwd, String callerCwd) {
|
||||
String name = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (name == null || name.isBlank()) {
|
||||
throw new IllegalArgumentException("no default worker profile is configured — "
|
||||
+ "pass a profile; configured: " + profiles.keySet());
|
||||
}
|
||||
BridgedConfig.Worker cfg = profiles.get(name);
|
||||
if (cfg == null) {
|
||||
throw new IllegalArgumentException("unknown worker profile '" + name
|
||||
+ "' — configured: " + profiles.keySet());
|
||||
}
|
||||
return resolveCwd(requestedCwd, cfg, callerCwd);
|
||||
}
|
||||
|
||||
private static String firstNonBlank(String... values) {
|
||||
for (String v : values) {
|
||||
if (v != null && !v.isBlank()) return v;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv, plus — when {@code worker.mcpUrl} is set — inline {@code --mcp-config} for
|
||||
* the bridge server and {@code --append-system-prompt} for the {@link #REPLY_CHARTER}. Neither
|
||||
* touches the profile's config; both are pure command-line flags.
|
||||
*/
|
||||
private List<String> argvWithBridge(BridgedConfig.Worker cfg) {
|
||||
if (!cfg.hasMcp()) {
|
||||
return cfg.argv();
|
||||
}
|
||||
String mcpJson = "{\"mcpServers\":{\"bridge\":{\"type\":\"http\",\"url\":\""
|
||||
+ cfg.mcpUrl() + "\"}}}";
|
||||
List<String> argv = new ArrayList<>(cfg.argv());
|
||||
argv.add("--mcp-config");
|
||||
argv.add(mcpJson);
|
||||
argv.add("--append-system-prompt");
|
||||
argv.add(REPLY_CHARTER);
|
||||
return argv;
|
||||
}
|
||||
|
||||
/** Dedicated worker space → own tab → start the worker (rooted at {@code cwd}) → drop the shell. */
|
||||
private Agent spawnInTab(BridgedConfig.Worker cfg, Map<String, String> workerEnv,
|
||||
List<String> argv, String cwd) {
|
||||
Workspace space = spaces.ensureWorkspace(cfg.workspace());
|
||||
Tab.Created tab = spaces.createTab(space.workspaceId());
|
||||
log.info("spawning worker profile={} base_url={} space={} tab={} cwd={}",
|
||||
cfg.profile(), cfg.baseUrl(), space.workspaceId(), tab.tab().tabId(), cwd);
|
||||
|
||||
Started started;
|
||||
try {
|
||||
started = startUniquelyNamed(cfg, workerEnv, argv, tab.tab().tabId(), cwd);
|
||||
} catch (RuntimeException e) {
|
||||
// The worker never started — don't leave the tab we just created orphaned.
|
||||
// Best-effort cleanup; never let it mask the real spawn failure.
|
||||
try {
|
||||
spaces.closeTab(tab.tab().tabId());
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to close orphaned tab {} after spawn error: {}",
|
||||
tab.tab().tabId(), cleanup.getMessage());
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
|
||||
// The worker is LIVE now. The remaining steps are cosmetic (drop herdr's seed shell
|
||||
// so the tab holds only the worker; label the tab). They must not fail the spawn or
|
||||
// orphan the running worker — on error we log and still return it so the caller gets
|
||||
// its paneId and can tear it down.
|
||||
if (tab.rootPaneId() != null) {
|
||||
tidy("close seed pane " + tab.rootPaneId(), () -> agents.close(tab.rootPaneId()));
|
||||
} else {
|
||||
log.warn("tab {} had no seed pane in the create response; worker tab may hold an extra pane",
|
||||
tab.tab().tabId());
|
||||
}
|
||||
tidy("label tab " + tab.tab().tabId(),
|
||||
() -> spaces.renameTab(tab.tab().tabId(), cfg.renderTabLabel(started.seq())));
|
||||
log.info("worker started pane={} tab={} terminal={}",
|
||||
started.agent().paneId(), started.agent().tabId(), started.agent().terminalId());
|
||||
return started.agent();
|
||||
}
|
||||
|
||||
/** Run a best-effort post-start cleanup step, logging (not throwing) on failure. */
|
||||
private void tidy(String what, Runnable step) {
|
||||
try {
|
||||
step.run();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("post-start step failed ({}) — worker is running regardless: {}", what, e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/** Legacy placement: herdr splits the currently-focused tab; the worker still starts in {@code cwd}. */
|
||||
private Agent spawnAsPane(BridgedConfig.Worker cfg, Map<String, String> workerEnv,
|
||||
List<String> argv, String cwd) {
|
||||
log.info("spawning worker (pane placement) profile={} base_url={} cwd={} argv={}",
|
||||
cfg.profile(), cfg.baseUrl(), cwd, argv);
|
||||
Agent worker = startUniquelyNamed(cfg, workerEnv, argv, null, cwd).agent();
|
||||
log.info("worker started pane={} terminal={}", worker.paneId(), worker.terminalId());
|
||||
return worker;
|
||||
}
|
||||
|
||||
/** A started worker together with the sequence its unique name/label used. */
|
||||
private record Started(Agent agent, long seq) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Start the worker under a unique herdr agent name. herdr requires each running
|
||||
* agent's {@code name} to be distinct (a 2nd {@code name:"claude"} fails
|
||||
* {@code agent_name_taken}) — the exact case that makes multiple workers useful. The name
|
||||
* is {@code claude-<profile>-<nonce>-<seq>}: {@code seq} distinguishes workers within this
|
||||
* process, and the per-process {@code nonce} keeps a fresh process (whose {@code seq}
|
||||
* restarts at 0) from colliding with same-profile workers that outlived a restart. The
|
||||
* retry is a belt-and-braces backstop for the astronomically unlikely nonce+seq clash;
|
||||
* the name is a label only — herdr detects kind and status from terminal output, not it.
|
||||
*/
|
||||
private Started startUniquelyNamed(BridgedConfig.Worker cfg, Map<String, String> workerEnv,
|
||||
List<String> argv, String tabId, String cwd) {
|
||||
HerdrException last = null;
|
||||
for (int attempt = 0; attempt < NAME_RETRIES; attempt++) {
|
||||
long seq = nameSeq.incrementAndGet();
|
||||
String name = "claude-" + cfg.profile() + "-" + nameNonce + "-" + seq;
|
||||
try {
|
||||
return new Started(agents.start(name, argv, workerEnv, tabId, cwd), seq);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_name_taken".equals(e.code())) throw e;
|
||||
log.debug("worker name '{}' taken, retrying", name);
|
||||
last = e;
|
||||
}
|
||||
}
|
||||
throw last;
|
||||
}
|
||||
|
||||
/** All herdr-tracked agents — discovery for "what workers exist". */
|
||||
@Override
|
||||
public List<Agent> list() {
|
||||
return agents.list();
|
||||
}
|
||||
|
||||
/**
|
||||
* Reap worker panes left behind by an earlier daemon process (CB-117). herdr keeps a worker's
|
||||
* pane alive across a daemon restart <em>by design</em>, and that pane's id is held only by its
|
||||
* spawner — so a worker whose owning process exited before issuing the matching teardown leaks
|
||||
* with nothing tracking it (there is no registry; {@link #list()} only asks herdr). On boot we
|
||||
* scan herdr for agents whose name matches our {@code claude-<profile>-<nonce>-<seq>} scheme with
|
||||
* a nonce <em>other</em> than this process's {@link #nameNonce}, and tear each one down (its pane
|
||||
* and, via {@link #stop}, its now-empty dedicated tab). A current-nonce worker is ours and live,
|
||||
* so it is left running; a user's own {@code claude} session carries no such name and is never
|
||||
* touched. Best-effort: a failed listing, or a failure to stop any one worker, is logged and
|
||||
* never aborts startup.
|
||||
*
|
||||
* @return the number of orphaned workers reaped
|
||||
*/
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
List<Agent> all;
|
||||
try {
|
||||
all = agents.list();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("orphan-worker reap skipped — agent.list failed: {}", e.getMessage());
|
||||
return 0;
|
||||
}
|
||||
int reaped = 0;
|
||||
for (Agent a : all) {
|
||||
if (!isForeignWorker(a.name(), nameNonce)) continue;
|
||||
try {
|
||||
stop(a.paneId());
|
||||
reaped++;
|
||||
log.info("reaped orphan worker {} (pane={} tab={}) left by a prior daemon",
|
||||
a.name(), a.paneId(), a.tabId());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("could not reap orphan worker {} (pane={}): {}",
|
||||
a.name(), a.paneId(), e.getMessage());
|
||||
}
|
||||
}
|
||||
if (reaped > 0) {
|
||||
log.info("orphan-worker reap complete — {} stale worker(s) removed at startup", reaped);
|
||||
}
|
||||
return reaped;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code name} is a bridge worker started by a <em>different</em> process than
|
||||
* {@code currentNonce} — the reap predicate (CB-117). True only for our naming scheme with a
|
||||
* foreign nonce: a non-worker name (no match, e.g. a user session) or our own live nonce is
|
||||
* excluded. Pure and package-private so the decision is unit-testable without herdr.
|
||||
*/
|
||||
static boolean isForeignWorker(String name, String currentNonce) {
|
||||
String nonce = workerNonce(name);
|
||||
return nonce != null && !nonce.equals(currentNonce);
|
||||
}
|
||||
|
||||
/** The 6-hex nonce embedded in a bridge worker name, or {@code null} if {@code name} isn't one. */
|
||||
static String workerNonce(String name) {
|
||||
if (name == null) return null;
|
||||
Matcher m = WORKER_NAME.matcher(name);
|
||||
return m.matches() ? m.group(1) : null;
|
||||
}
|
||||
|
||||
/** This process's worker-name nonce (a label component only; exposed for reaper tests). */
|
||||
String nameNonce() {
|
||||
return nameNonce;
|
||||
}
|
||||
|
||||
/**
|
||||
* Tear a worker down by pane id: close the pane, and close its tab <em>only</em> when the
|
||||
* worker is that tab's sole occupant. The single-pane check is what makes this safe
|
||||
* regardless of how the worker was placed (or a placement-config change across a restart):
|
||||
* a pane-placement worker sitting in one of the user's shared tabs has siblings, so its
|
||||
* tab is never closed — we only ever remove a tab we created to hold one worker.
|
||||
*
|
||||
* <p>Resolves the tab from the pane <em>before</em> closing it. An already-gone pane/tab
|
||||
* (repeated DELETE, crashed worker) is treated as success; any other failure propagates so
|
||||
* a genuinely failed teardown is not reported as done.
|
||||
*/
|
||||
@Override
|
||||
public void stop(String paneId) {
|
||||
// Teardown knows only the paneId, not which profile spawned it. Attempt tab cleanup when any
|
||||
// profile uses tab placement (so the bridge may have created a dedicated worker tab); the
|
||||
// single-occupant check below is what actually protects the user's shared tabs.
|
||||
WorkspaceControl.PaneLocation loc = usesTabPlacement() ? spaces.locatePane(paneId) : null;
|
||||
try {
|
||||
agents.close(paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!isAlreadyGone(e)) throw e;
|
||||
log.debug("pane.close({}) ignored — already gone: {}", paneId, e.getMessage());
|
||||
}
|
||||
if (loc != null && loc.tabPaneCount() == 1) {
|
||||
spaces.closeTab(loc.tabId());
|
||||
} else if (loc != null) {
|
||||
log.debug("not closing tab {} — it holds {} panes (not a dedicated worker tab)",
|
||||
loc.tabId(), loc.tabPaneCount());
|
||||
}
|
||||
}
|
||||
|
||||
/** Whether any configured profile places workers in their own tab (so tabs may need cleanup). */
|
||||
private boolean usesTabPlacement() {
|
||||
return profiles.values().stream().anyMatch(BridgedConfig.Worker::tabPlacement);
|
||||
}
|
||||
|
||||
/** True when a herdr error means the target is already gone (safe to treat as done). */
|
||||
private static boolean isAlreadyGone(HerdrException e) {
|
||||
return e.code() != null && e.code().endsWith("_not_found");
|
||||
}
|
||||
|
||||
// --- PeerLauncher SPI -------------------------------------------------------------------
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
Set<Capability> caps = EnumSet.of(Capability.MID_TURN_ASK, Capability.WORKTREE, Capability.ORPHAN_REAP);
|
||||
if (hasGitTokenProfile()) {
|
||||
caps.add(Capability.SELF_PR);
|
||||
}
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
|
||||
/** Whether any configured profile opts into a git-forge token (required for {@link Capability#SELF_PR}). */
|
||||
private boolean hasGitTokenProfile() {
|
||||
return profiles.values().stream().anyMatch(BridgedConfig.Worker::hasGitToken);
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Delegates to the three-arg {@link #spawn(String, String, String)} and wraps the
|
||||
* resulting herdr {@link Agent} in a {@link WorkerHandle} whose {@link PeerHandle#id()}
|
||||
* equals the agent's paneId.
|
||||
*/
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
Agent agent = spawn(req.profileName(), req.requestedCwd(), req.callerCwd());
|
||||
return new WorkerHandle(agent.paneId(), agent.terminalId());
|
||||
}
|
||||
|
||||
/** A concrete {@link PeerHandle} wrapping herdr agent coordinates. */
|
||||
private record WorkerHandle(String id, String terminalId) implements PeerHandle {
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
return effectiveCwd(req.profileName(), req.requestedCwd(), req.callerCwd());
|
||||
}
|
||||
|
||||
private static void putIfPresent(Map<String, String> m, String k, String v) {
|
||||
if (v != null && !v.isBlank()) {
|
||||
m.put(k, v);
|
||||
}
|
||||
}
|
||||
|
||||
/** Host env lookup that tolerates an unconfigured (null/blank) var name — returns null then. */
|
||||
private String resolveEnv(String name) {
|
||||
return (name == null || name.isBlank()) ? null : env.apply(name);
|
||||
}
|
||||
}
|
||||
@@ -1,14 +1,52 @@
|
||||
<configuration>
|
||||
<!--
|
||||
CB-575: MCP SDK 2.0.0 has no public notification registration API. It registers only
|
||||
notifications/initialized and notifications/roots/list_changed, so other client notifications
|
||||
still warn when unhandled. Clients may legitimately send notifications/cancelled; suppress only
|
||||
that SDK WARN because it would devalue the action-needed WARN level used by M4 fleet health.
|
||||
If a later SDK handles cancellation, this filter simply stops matching and can be removed.
|
||||
-->
|
||||
<turboFilter class="dev.ltms.bridged.logging.McpCancelledNotificationFilter"/>
|
||||
|
||||
<appender name="STDOUT" class="ch.qos.logback.core.ConsoleAppender">
|
||||
<encoder>
|
||||
<pattern>%d{HH:mm:ss.SSS} %-5level [%thread] %logger{28} - %msg%n</pattern>
|
||||
</encoder>
|
||||
</appender>
|
||||
|
||||
<!--
|
||||
CB-505 audit trail. Its own file, deliberately not the app log: privileged actions
|
||||
(spawn/stop/send/reply/drain) must stay greppable and shippable without dragging DEBUG noise
|
||||
along. AuditLog emits a complete JSON object including its own ISO-8601 "ts" field, so the
|
||||
pattern is a bare %msg — a pattern that spliced literal braces around the message would
|
||||
collide with logback's own variable substitution. Rolls daily, 30 days retained, 100MB cap.
|
||||
|
||||
NOTE: records carry who/what/target/outcome only. Message CONTENT is never written here —
|
||||
this bridge carries source code and prompts, and an audit log that accumulated them would be
|
||||
a transcript archive rather than a control.
|
||||
-->
|
||||
<appender name="AUDIT" class="ch.qos.logback.core.rolling.RollingFileAppender">
|
||||
<file>logs/audit.log</file>
|
||||
<rollingPolicy class="ch.qos.logback.core.rolling.SizeAndTimeBasedRollingPolicy">
|
||||
<fileNamePattern>logs/audit.%d{yyyy-MM-dd}.%i.log</fileNamePattern>
|
||||
<maxFileSize>10MB</maxFileSize>
|
||||
<maxHistory>30</maxHistory>
|
||||
<totalSizeCap>100MB</totalSizeCap>
|
||||
</rollingPolicy>
|
||||
<encoder>
|
||||
<pattern>%msg%n</pattern>
|
||||
</encoder>
|
||||
</appender>
|
||||
|
||||
<logger name="dev.ltms.bridged" level="DEBUG"/>
|
||||
<logger name="io.javalin" level="INFO"/>
|
||||
<logger name="org.eclipse.jetty" level="WARN"/>
|
||||
|
||||
<!-- additivity=false keeps the audit stream out of stdout; it is its own record. -->
|
||||
<logger name="audit" level="INFO" additivity="false">
|
||||
<appender-ref ref="AUDIT"/>
|
||||
</logger>
|
||||
|
||||
<root level="INFO">
|
||||
<appender-ref ref="STDOUT"/>
|
||||
</root>
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
package dev.ltms.bridged;
|
||||
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.Map;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-534: the injector's readiness gate must open for a lead as well as for a present worker.
|
||||
*
|
||||
* <p>The bug these cover was silent and slow: a lead was never marked present (only workers are), so
|
||||
* every lead→lead delivery sat on the gate for the full readiness grace and failed ~60s later without
|
||||
* a keystroke ever reaching the pane.
|
||||
*/
|
||||
class BridgedDeliverabilityTest {
|
||||
|
||||
private static Supplier<Map<String, String>> leads(Map<String, String> m) {
|
||||
return () -> m;
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a worker that has connected its MCP is deliverable")
|
||||
void presentWorkerIsDeliverable() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
presence.markPresent("term_worker");
|
||||
|
||||
assertTrue(Bridged.deliverableTo(presence, leads(Map.of())).test("term_worker"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a worker still in its boot window is held back")
|
||||
void absentWorkerIsNotDeliverable() {
|
||||
assertFalse(Bridged.deliverableTo(new MemberPresence(), leads(Map.of())).test("term_booting"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a lead is deliverable without ever being marked present")
|
||||
void leadIsDeliverableWithoutPresence() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
Predicate<String> deliverable =
|
||||
Bridged.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")));
|
||||
|
||||
assertFalse(presence.isPresent("term_lead"), "a lead is never enrolled in worker presence");
|
||||
assertTrue(deliverable.test("term_lead"), "…and must be deliverable anyway");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an unknown terminal is deliverable to neither")
|
||||
void strangerIsNotDeliverable() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
presence.markPresent("term_worker");
|
||||
|
||||
assertFalse(Bridged.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")))
|
||||
.test("term_stranger"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a lead discovered after startup becomes deliverable with no restart")
|
||||
void leadSetIsReadThroughOnEveryCall() {
|
||||
Map<String, String> discovered = new HashMap<>();
|
||||
Predicate<String> deliverable = Bridged.deliverableTo(new MemberPresence(), leads(discovered));
|
||||
|
||||
assertFalse(deliverable.test("term_late"));
|
||||
discovered.put("term_late", "gpt-sol-5.6"); // leadScan picks up a newly labelled tab
|
||||
assertTrue(deliverable.test("term_late"), "the supplier must be re-read, not snapshotted");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("forgetting a torn-down worker does not strip a lead of its deliverability")
|
||||
void forgetDoesNotDisarmALead() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
Predicate<String> deliverable =
|
||||
Bridged.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")));
|
||||
|
||||
presence.forget("term_lead"); // the injector's cleanup path runs against every target
|
||||
|
||||
assertTrue(deliverable.test("term_lead"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,108 @@
|
||||
package dev.ltms.bridged;
|
||||
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-596: an absent (or empty) {@code memberCredentials:} block blocks nothing — no name is
|
||||
* hardcoded any more to fall back on, so the daemon must say so out loud at startup rather than
|
||||
* silently dropping CB-592's protection. Mirrors {@link RequiredSecretEnvVarsTest}'s pattern for
|
||||
* the CB-594 startup-secrets report, capturing the real log via a {@link ListAppender}.
|
||||
*/
|
||||
class MemberCredentialsGapReportTest {
|
||||
|
||||
private static BridgedConfig load(Path dir, String yaml) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml);
|
||||
return BridgedConfig.load(f);
|
||||
}
|
||||
|
||||
private static ListAppender<ILoggingEvent> attach() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Bridged.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
||||
((Logger) LoggerFactory.getLogger(Bridged.class)).detachAppender(appender);
|
||||
}
|
||||
|
||||
@Test
|
||||
void anAbsentBlockWarnsThatEveryMemberInheritsTheWholeStore(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, "bind:\n host: 127.0.0.1\n port: 8765\n");
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Bridged.reportMemberCredentialsGap(cfg);
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
assertTrue(appender.list.stream().anyMatch(e ->
|
||||
e.getLevel() == ch.qos.logback.classic.Level.WARN
|
||||
&& e.getFormattedMessage().contains("memberCredentials")
|
||||
&& e.getFormattedMessage().contains("WHOLE secret store")),
|
||||
"an absent block must WARN that protection is lost, not stay silent");
|
||||
}
|
||||
|
||||
@Test
|
||||
void anEmptyKnownListWarnsTheSameAsAbsent(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, """
|
||||
memberCredentials:
|
||||
policy: deny-by-default
|
||||
""");
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Bridged.reportMemberCredentialsGap(cfg);
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
assertTrue(appender.list.stream().anyMatch(e ->
|
||||
e.getLevel() == ch.qos.logback.classic.Level.WARN
|
||||
&& e.getFormattedMessage().contains("memberCredentials")),
|
||||
"policy: with no known: names still blocks nothing and must warn the same way");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPopulatedKnownListLogsInfoNotWarn(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, """
|
||||
memberCredentials:
|
||||
policy: deny-by-default
|
||||
allow: [AI_GATEWAY_TOKEN]
|
||||
known: [AI_GATEWAY_TOKEN, GITEA_ACCESS_TOKEN]
|
||||
""");
|
||||
|
||||
// logback-test.xml pins dev.ltms.bridged to WARN (see its own comment); raise it here so
|
||||
// the INFO line this test asserts on actually reaches the appender, and restore after.
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Bridged.class);
|
||||
ch.qos.logback.classic.Level original = logger.getLevel();
|
||||
logger.setLevel(ch.qos.logback.classic.Level.INFO);
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Bridged.reportMemberCredentialsGap(cfg);
|
||||
} finally {
|
||||
detach(appender);
|
||||
logger.setLevel(original);
|
||||
}
|
||||
|
||||
assertFalse(appender.list.stream().anyMatch(e -> e.getLevel() == ch.qos.logback.classic.Level.WARN),
|
||||
"a configured, non-empty known: list must not warn — the block is doing its job");
|
||||
assertTrue(appender.list.stream().anyMatch(e -> e.getFormattedMessage().contains("blocking 1")),
|
||||
"the INFO line should say how many names are actually blocked (known minus allow)");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,114 @@
|
||||
package dev.ltms.bridged;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-594: {@link Bridged#requiredSecretEnvVars(BridgedConfig)} is what decides what the startup
|
||||
* secret report checks — it must derive that set from the config, not a hand-written list, or a
|
||||
* new profile's token silently stops being reported.
|
||||
*/
|
||||
class RequiredSecretEnvVarsTest {
|
||||
|
||||
private static BridgedConfig load(Path dir, String yaml) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml);
|
||||
return BridgedConfig.load(f);
|
||||
}
|
||||
|
||||
@Test
|
||||
void collectsATokenEnvPerNonSubscriptionProfile(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
""");
|
||||
|
||||
Map<String, List<String>> required = Bridged.requiredSecretEnvVars(cfg);
|
||||
|
||||
assertTrue(required.containsKey("AI_GATEWAY_TOKEN"));
|
||||
assertEquals(List.of("profile 'local' tokenEnv"), required.get("AI_GATEWAY_TOKEN"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aSubscriptionProfileNeedsNoTokenEnv(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, """
|
||||
profiles:
|
||||
opus:
|
||||
subscription: true
|
||||
model: claude-opus-5
|
||||
""");
|
||||
|
||||
assertTrue(Bridged.requiredSecretEnvVars(cfg).isEmpty(),
|
||||
"subscription: true never reads ANTHROPIC_AUTH_TOKEN — see Profile#isSubscription");
|
||||
}
|
||||
|
||||
@Test
|
||||
void gitTokenEnvIsOptInAndCollectedWhenSet(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
gitTokenEnv: WORKER_GITEA_TOKEN
|
||||
""");
|
||||
|
||||
Map<String, List<String>> required = Bridged.requiredSecretEnvVars(cfg);
|
||||
|
||||
assertTrue(required.containsKey("WORKER_GITEA_TOKEN"));
|
||||
assertEquals(List.of("profile 'local' gitTokenEnv"), required.get("WORKER_GITEA_TOKEN"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void noGitTokenEnvMeansNothingIsRequiredForIt(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
""");
|
||||
|
||||
assertFalse(Bridged.requiredSecretEnvVars(cfg).containsKey("WORKER_GITEA_TOKEN"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aVarSharedByTwoProfilesIsReportedOnceNamingBoth(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
gitTokenEnv: WORKER_GITEA_TOKEN
|
||||
gx:
|
||||
kind: opencode
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
gitTokenEnv: WORKER_GITEA_TOKEN
|
||||
""");
|
||||
|
||||
Map<String, List<String>> required = Bridged.requiredSecretEnvVars(cfg);
|
||||
|
||||
assertEquals(List.of("profile 'local' tokenEnv", "profile 'gx' tokenEnv"),
|
||||
required.get("AI_GATEWAY_TOKEN"));
|
||||
assertEquals(List.of("profile 'local' gitTokenEnv", "profile 'gx' gitTokenEnv"),
|
||||
required.get("WORKER_GITEA_TOKEN"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void noProfilesMeansNothingIsRequired(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, "bind:\n host: 127.0.0.1\n port: 8765\n");
|
||||
|
||||
assertTrue(Bridged.requiredSecretEnvVars(cfg).isEmpty());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,100 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.LoggerContext;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-505 — the audit record's shape.
|
||||
*
|
||||
* <p>These exist because the first cut of this feature emitted lines that were <em>not</em> valid
|
||||
* JSON: the timestamp was spliced on by a logback pattern whose literal braces collided with
|
||||
* logback's variable substitution. The appender failed to parse, and nothing in the build noticed.
|
||||
* An audit trail that silently stops being machine-readable is worse than none.
|
||||
*/
|
||||
class AuditLogTest {
|
||||
|
||||
private final ObjectMapper mapper = new ObjectMapper();
|
||||
private ListAppender<ILoggingEvent> appender;
|
||||
private ch.qos.logback.classic.Logger auditLogger;
|
||||
|
||||
@BeforeEach
|
||||
void attach() {
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
auditLogger = ctx.getLogger("audit");
|
||||
appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
auditLogger.addAppender(appender);
|
||||
auditLogger.setLevel(Level.INFO);
|
||||
}
|
||||
|
||||
@AfterEach
|
||||
void detach() {
|
||||
auditLogger.detachAppender(appender);
|
||||
}
|
||||
|
||||
private JsonNode onlyRecord() throws Exception {
|
||||
assertEquals(1, appender.list.size(), "exactly one audit line expected");
|
||||
String line = appender.list.getFirst().getFormattedMessage();
|
||||
return mapper.readTree(line); // throws if the line is not valid JSON
|
||||
}
|
||||
|
||||
@Test
|
||||
void anAllowedActionIsRecordedAsValidJson() throws Exception {
|
||||
AuditLog.allowed(Principal.primary(4242), Authz.Action.SPAWN, "term_a");
|
||||
|
||||
JsonNode r = onlyRecord();
|
||||
assertEquals("PRIMARY", r.path("role").asText());
|
||||
assertEquals("primary", r.path("actor").asText());
|
||||
assertEquals(4242, r.path("pid").asLong());
|
||||
assertEquals("SPAWN", r.path("action").asText());
|
||||
assertEquals("term_a", r.path("target").asText());
|
||||
assertEquals("allowed", r.path("outcome").asText());
|
||||
assertFalse(r.path("ts").asText().isBlank(), "every record carries its own timestamp");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aDenialRecordsTheReason() throws Exception {
|
||||
AuditLog.denied(Principal.worker("term_b", 7), Authz.Action.REPLY, "term_a", "forbidden");
|
||||
|
||||
JsonNode r = onlyRecord();
|
||||
assertEquals("WORKER", r.path("role").asText());
|
||||
assertEquals("worker:term_b", r.path("actor").asText());
|
||||
assertEquals("denied", r.path("outcome").asText());
|
||||
assertEquals("forbidden", r.path("reason").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNullCallerIsRecordedAsAnonymousRatherThanCrashing() throws Exception {
|
||||
AuditLog.failed(null, Authz.Action.SEND, null, "herdr unreachable");
|
||||
|
||||
JsonNode r = onlyRecord();
|
||||
assertEquals("ANONYMOUS", r.path("role").asText());
|
||||
assertTrue(r.path("target").isNull(), "an absent target is JSON null, not the string \"null\"");
|
||||
assertEquals("failed", r.path("outcome").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
void hostileValuesAreEscapedAndCannotForgeAnExtraRecord() throws Exception {
|
||||
// A target id containing a quote and a newline must not be able to terminate the JSON
|
||||
// object early and inject a second, attacker-shaped audit line.
|
||||
AuditLog.denied(Principal.worker("term_a", 1), Authz.Action.REPLY,
|
||||
"evil\",\"outcome\":\"allowed\"}\n{\"forged\":true", "forbidden");
|
||||
|
||||
JsonNode r = onlyRecord();
|
||||
assertEquals("denied", r.path("outcome").asText(),
|
||||
"the injected outcome must not override the real one");
|
||||
assertTrue(r.path("target").asText().contains("forged"),
|
||||
"the hostile text survives as inert data inside the target field");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,124 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static dev.ltms.bridged.auth.Authz.Action.*;
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/** CB-505 — the authorization table, pinned so it cannot drift silently. */
|
||||
class AuthzTest {
|
||||
|
||||
private static final Principal PRIMARY = Principal.primary(100);
|
||||
private static final Principal WORKER_A = Principal.worker("term_a", 200);
|
||||
private static final Principal WORKER_B = Principal.worker("term_b", 300);
|
||||
private static final Principal ANON = Principal.anonymous();
|
||||
private static final Principal ARCH_DESIGN = Principal.architect("lead-designer", "term_design", 400);
|
||||
private static final Principal ARCH_OTHER = Principal.architect("reviewer", "term_review", 500);
|
||||
|
||||
@Test
|
||||
void anonymousIsAuthorizedForNothing() {
|
||||
for (Authz.Action a : Authz.Action.values()) {
|
||||
assertFalse(Authz.permits(ANON, a, "term_a"),
|
||||
a + " must be refused to an unauthenticated caller");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNullCallerIsTreatedAsAnonymous() {
|
||||
assertFalse(Authz.permits(null, READ, null));
|
||||
assertTrue(Authz.isUnauthenticated(null));
|
||||
}
|
||||
|
||||
@Test
|
||||
void orchestrationBelongsToThePrimaryAlone() {
|
||||
for (Authz.Action a : new Authz.Action[]{SPAWN, STOP, SEND, DRAIN}) {
|
||||
assertTrue(Authz.permits(PRIMARY, a, "term_a"), "the primary orchestrates: " + a);
|
||||
assertFalse(Authz.permits(WORKER_A, a, "term_a"),
|
||||
"a worker performing " + a + " would be escalating into the orchestrator role");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerMayReplyAndAskOnlyAsItself() {
|
||||
assertTrue(Authz.permits(WORKER_A, REPLY, "term_a"));
|
||||
assertTrue(Authz.permits(WORKER_A, ASK, "term_a"));
|
||||
|
||||
assertFalse(Authz.permits(WORKER_A, REPLY, "term_b"),
|
||||
"worker A must not be able to reply on worker B's session");
|
||||
assertFalse(Authz.permits(WORKER_B, ASK, "term_a"),
|
||||
"worker B must not be able to ask as worker A");
|
||||
}
|
||||
|
||||
@Test
|
||||
void thePrimaryMayNotForgeAWorkersReply() {
|
||||
// Not a hypothetical nicety: a forged reply would resolve the rendezvous the primary is
|
||||
// itself blocked on, corrupting the correlation between a turn and its answer.
|
||||
assertFalse(Authz.permits(PRIMARY, REPLY, "term_a"));
|
||||
assertFalse(Authz.permits(PRIMARY, ASK, "term_a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerWithNoTargetCannotReply() {
|
||||
assertFalse(Authz.permits(WORKER_A, REPLY, null),
|
||||
"an absent session id must not satisfy the own-session rule");
|
||||
}
|
||||
|
||||
// ── CB-548: the architect matrix ───────────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void anArchitectMaySendButNotSpawnStopOrDrain() {
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, SEND, "term_worker"),
|
||||
"delegating a turn to a worker IS the architect's job");
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, SEND, null));
|
||||
|
||||
for (Authz.Action a : new Authz.Action[]{SPAWN, STOP, DRAIN}) {
|
||||
assertFalse(Authz.permits(ARCH_DESIGN, a, null),
|
||||
"an architect must not " + a + " — fleet lifecycle is the primary's alone, so "
|
||||
+ "a coordinator cannot also stand up or tear down the fleet");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectMayReplyAndAskOnlyAsItsOwnPane() {
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, REPLY, "term_design"), "its own pane is its own");
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, ASK, "term_design"));
|
||||
|
||||
assertFalse(Authz.permits(ARCH_DESIGN, REPLY, "term_review"),
|
||||
"architect 'lead-designer' must not reply on reviewer's pane");
|
||||
assertFalse(Authz.permits(ARCH_OTHER, ASK, "term_design"),
|
||||
"reviewer must not ask as lead-designer — no terminal is another's");
|
||||
assertFalse(Authz.permits(ARCH_DESIGN, REPLY, null),
|
||||
"an absent target must not pass the own-session rule");
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectMayReadAndScrapeMetrics() {
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, READ, null));
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, METRICS, null));
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectIsNotCountedAsPrimaryOrWorker() {
|
||||
assertFalse(Authz.permits(ARCH_DESIGN, SPAWN, null), "not a primary — no lifecycle");
|
||||
assertFalse(ARCH_DESIGN.isPrimary());
|
||||
assertFalse(ARCH_DESIGN.isWorker(), "an architect is its own role, not a widened worker");
|
||||
assertTrue(ARCH_DESIGN.isArchitect());
|
||||
}
|
||||
|
||||
@Test
|
||||
void observationIsOpenToBothAuthenticatedRoles() {
|
||||
assertTrue(Authz.permits(PRIMARY, READ, null));
|
||||
assertTrue(Authz.permits(WORKER_A, READ, null));
|
||||
assertTrue(Authz.permits(PRIMARY, METRICS, null));
|
||||
assertTrue(Authz.permits(WORKER_A, METRICS, null));
|
||||
}
|
||||
|
||||
@Test
|
||||
void unauthenticatedIsDistinguishedFromMerelyForbidden() {
|
||||
// Drives the 401-vs-403 split: a missing credential is fixable by the caller, a wrong role
|
||||
// is not.
|
||||
assertTrue(Authz.isUnauthenticated(ANON));
|
||||
assertFalse(Authz.isUnauthenticated(WORKER_A));
|
||||
assertFalse(Authz.isUnauthenticated(PRIMARY));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,416 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.PaneLocator;
|
||||
import dev.ltms.bridged.mcp.ConnectionIdentity;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-501. The behaviour under test is the inversion of the pre-CB-501 default: failing every
|
||||
* identity check must yield {@link Role#ANONYMOUS}, not {@code PRIMARY}.
|
||||
*/
|
||||
class CallerResolverTest {
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
|
||||
/** Identity resolving the canned worker pane, keyed off a faked peer-PID lookup. */
|
||||
private ConnectionIdentity identity(long pid) {
|
||||
return new ConnectionIdentity(new PaneLocator(herdr), _ -> pid);
|
||||
}
|
||||
|
||||
/** A PID that owns a worker pane in the fake. */
|
||||
private ConnectionIdentity workerIdentity() {
|
||||
return identity(FakeHerdr.WORKER_PID);
|
||||
}
|
||||
|
||||
/** A PID that owns no pane — i.e. the primary, or any other local process. */
|
||||
private ConnectionIdentity nonWorkerIdentity() {
|
||||
return identity(999_999);
|
||||
}
|
||||
|
||||
private static MemberRegistry boundMembers(String slot, MemberRole role) {
|
||||
Map<String, BridgedConfig.Slot> architects = role == MemberRole.ARCHITECT
|
||||
? Map.of("lead-designer", new BridgedConfig.Slot("sonnet")) : Map.of();
|
||||
Map<String, BridgedConfig.Slot> devs = role == MemberRole.DEV
|
||||
? Map.of("builder", new BridgedConfig.Slot("sonnet")) : Map.of();
|
||||
MemberRegistry members = new MemberRegistry(
|
||||
new BridgedConfig.Fleet(Map.of(), architects, devs, Map.of(), null));
|
||||
assertTrue(members.bind(slot, "term_a"));
|
||||
return members;
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLoopbackWorkerPaneResolvesToWorkerRegardlessOfAuthMode() {
|
||||
Principal underTrust = new CallerResolver(workerIdentity()).resolve("127.0.0.1", 42, null);
|
||||
Principal underToken = new CallerResolver(workerIdentity(), true, "s3cret")
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, underTrust.role());
|
||||
assertEquals("term_a", underTrust.terminal());
|
||||
assertEquals(Role.WORKER, underToken.role(),
|
||||
"worker identity is unforgeable and must never be token-gated — otherwise enabling "
|
||||
+ "auth would lock the whole fleet out of bridge_reply");
|
||||
assertEquals("term_a", underToken.terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPinnedPrimaryTerminalResolvesToPrimaryNotWorker() {
|
||||
// The primary's own session lives in a herdr pane (term_a here). Without the pin the pane
|
||||
// match wins and the primary is locked out of spawn/send/stop as a misread worker.
|
||||
Principal p = CallerResolver.pinnedTo(workerIdentity(), false, null, "term_a")
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPinnedPrimaryTerminalNeedsNoTokenEvenInTokenMode() {
|
||||
Principal p = CallerResolver.pinnedTo(workerIdentity(), true, "s3cret", "term_a")
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role(),
|
||||
"the pane mapping is as unforgeable as a worker's — the pin outranks the token path");
|
||||
}
|
||||
|
||||
@Test
|
||||
void otherPanesRemainWorkersWhenAPinIsSet() {
|
||||
Principal p = CallerResolver.pinnedTo(workerIdentity(), false, null, "term_someone_else")
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role());
|
||||
assertEquals("term_a", p.terminal());
|
||||
}
|
||||
|
||||
/** The pin is optional config, so an absent or whitespace one must change nothing at all. */
|
||||
@Test
|
||||
void aBlankPinLeavesWorkerResolutionUntouched() {
|
||||
assertEquals(Role.WORKER,
|
||||
CallerResolver.pinnedTo(workerIdentity(), false, null, " ").resolve("127.0.0.1", 42, null).role());
|
||||
assertEquals(Role.WORKER,
|
||||
CallerResolver.pinnedTo(workerIdentity(), false, null, null).resolve("127.0.0.1", 42, null).role());
|
||||
}
|
||||
|
||||
@Test
|
||||
void loopbackTrustTreatsANonWorkerLoopbackCallerAsThePrimary() {
|
||||
Principal p = new CallerResolver(nonWorkerIdentity()).resolve("127.0.0.1", 99, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role(), "the historical behaviour, now an explicit choice");
|
||||
}
|
||||
|
||||
@Test
|
||||
void tokenModeRefusesANonWorkerCallerThatPresentsNoToken() {
|
||||
Principal p = new CallerResolver(nonWorkerIdentity(), true, "s3cret")
|
||||
.resolve("127.0.0.1", 99, null);
|
||||
|
||||
assertEquals(Role.ANONYMOUS, p.role(),
|
||||
"no credential must mean NOTHING, not the most privileged role on the bus");
|
||||
}
|
||||
|
||||
@Test
|
||||
void tokenModeAcceptsAValidBearerTokenAsThePrimary() {
|
||||
Principal p = new CallerResolver(nonWorkerIdentity(), true, "s3cret")
|
||||
.resolve("127.0.0.1", 99, "Bearer s3cret");
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
}
|
||||
|
||||
@Test
|
||||
void tokenModeRejectsAWrongOrMalformedCredential() {
|
||||
CallerResolver r = new CallerResolver(nonWorkerIdentity(), true, "s3cret");
|
||||
|
||||
assertEquals(Role.ANONYMOUS, r.resolve("127.0.0.1", 99, "Bearer wrong").role());
|
||||
assertEquals(Role.ANONYMOUS, r.resolve("127.0.0.1", 99, "s3cret").role(), "scheme required");
|
||||
assertEquals(Role.ANONYMOUS, r.resolve("127.0.0.1", 99, "Bearer ").role(), "empty credential");
|
||||
assertEquals(Role.ANONYMOUS, r.resolve("127.0.0.1", 99, "Basic s3cret").role(), "wrong scheme");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theBearerSchemeIsCaseInsensitivePerRfc7235() {
|
||||
CallerResolver r = new CallerResolver(nonWorkerIdentity(), true, "s3cret");
|
||||
|
||||
assertEquals(Role.PRIMARY, r.resolve("127.0.0.1", 99, "bearer s3cret").role());
|
||||
assertEquals(Role.PRIMARY, r.resolve("127.0.0.1", 99, "BEARER s3cret").role());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonLoopbackCallerIsNeverThePrimaryUnderLoopbackTrust() {
|
||||
// Defence in depth: startup already refuses this pairing (validateAuthExposure), but if a
|
||||
// proxy ever forwards a remote peer onto the loopback listener, the resolver must not
|
||||
// hand it the primary role.
|
||||
Principal p = new CallerResolver(nonWorkerIdentity()).resolve("10.0.0.7", 99, null);
|
||||
|
||||
assertEquals(Role.ANONYMOUS, p.role());
|
||||
}
|
||||
|
||||
// ── CB-530: the leaders registry ────────────────────────────────────────────────────────────
|
||||
// The fake resolves exactly one pane (term_a) from a PID, so "two leads both resolve" is
|
||||
// asserted at the config layer (BridgedConfigTest#leaderTerminals…). What matters here is that
|
||||
// resolution is a REGISTRY LOOKUP rather than a single equality test against one pin.
|
||||
|
||||
@Test
|
||||
void aRegisteredLeadPaneResolvesToPrimaryCarryingItsName() {
|
||||
Principal p = new CallerResolver(workerIdentity(), false, null, Map.of("term_a", "opus-5.0"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
assertEquals("opus-5.0", p.name(), "whoami must be able to say WHICH lead is asking");
|
||||
assertEquals("term_a", p.terminal(),
|
||||
"CB-532: a lead carries the pane it was matched by. Without it ownsSession() can "
|
||||
+ "never be true for a lead, so it can send to a peer but never answer one");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aSecondLeadIsRecognisedRatherThanSilentlyDemoted() {
|
||||
// The regression this feature exists for: with a singular pin, whichever lead was not the
|
||||
// pin resolved as a worker and was refused every orchestration call.
|
||||
Principal p = new CallerResolver(workerIdentity(), false, null,
|
||||
Map.of("term_elsewhere", "gpt-sol-5.6", "term_a", "opus-5.0"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
assertEquals("opus-5.0", p.name());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPaneAbsentFromTheRegistryIsStillAWorker() {
|
||||
Principal p = new CallerResolver(workerIdentity(), false, null,
|
||||
Map.of("term_elsewhere", "gpt-sol-5.6"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role());
|
||||
assertEquals("term_a", p.terminal());
|
||||
assertNull(p.name());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aRegisteredLeadNeedsNoTokenEvenInTokenMode() {
|
||||
Principal p = new CallerResolver(workerIdentity(), true, "s3cret", Map.of("term_a", "opus"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role(),
|
||||
"the pane mapping is as unforgeable as a worker's — it outranks the token path");
|
||||
assertEquals("opus", p.name());
|
||||
}
|
||||
|
||||
/** The pre-CB-530 spelling must keep working, exactly, including for configs that never migrate. */
|
||||
@Test
|
||||
void theLegacySinglePinBehavesAsALeadNamedPrimary() {
|
||||
Principal p = CallerResolver.pinnedTo(workerIdentity(), false, null, "term_a")
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
assertEquals("primary", p.name());
|
||||
}
|
||||
|
||||
@Test
|
||||
void anEmptyRegistryLeavesEveryPaneAWorker() {
|
||||
Map<String, String> noLeads = null;
|
||||
assertEquals(Role.WORKER,
|
||||
new CallerResolver(workerIdentity(), false, null, Map.of())
|
||||
.resolve("127.0.0.1", 42, null).role());
|
||||
assertEquals(Role.WORKER,
|
||||
new CallerResolver(workerIdentity(), false, null, noLeads)
|
||||
.resolve("127.0.0.1", 42, null).role());
|
||||
// CB-531: and the same for the live-registry form, whose supplier may also be absent.
|
||||
assertEquals(Role.WORKER,
|
||||
CallerResolver.withLeads(workerIdentity(), false, null, null)
|
||||
.resolve("127.0.0.1", 42, null).role());
|
||||
}
|
||||
|
||||
/** The audit line must distinguish leads once several exist, or a log says nothing useful. */
|
||||
@Test
|
||||
void describeNamesTheLeadButStillReadsPrimaryWhenUnnamed() {
|
||||
assertEquals("leader:opus-5.0", Principal.leader("opus-5.0", "term_a", 1).describe());
|
||||
assertEquals("primary", Principal.primary(1).describe());
|
||||
assertEquals("worker:term_a", Principal.worker("term_a", 1).describe());
|
||||
}
|
||||
|
||||
// ── CB-532: a lead is an addressable peer, not only a sender ────────────────────────────────
|
||||
|
||||
/**
|
||||
* The regression this ticket exists for: two leads could both be recognised (CB-530/531) and
|
||||
* still not converse, because REPLY is gated on ownsSession() and a lead owned nothing.
|
||||
*/
|
||||
@Test
|
||||
void aLeadOwnsItsOwnPaneSoItMayAnswerAPeer() {
|
||||
Principal lead = new CallerResolver(workerIdentity(), false, null, Map.of("term_a", "opus-5.0"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertTrue(lead.ownsSession("term_a"));
|
||||
assertTrue(Authz.permits(lead, Authz.Action.REPLY, "term_a"),
|
||||
"a lead answering a peer replies for its OWN terminal — the rendezvous the sender "
|
||||
+ "opened is keyed on exactly that");
|
||||
assertTrue(Authz.permits(lead, Authz.Action.ASK, "term_a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLeadStillCannotActAsAnyoneElse() {
|
||||
Principal lead = new CallerResolver(workerIdentity(), false, null, Map.of("term_a", "opus-5.0"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertFalse(lead.ownsSession("term_someone_else"));
|
||||
assertFalse(Authz.permits(lead, Authz.Action.REPLY, "term_someone_else"),
|
||||
"widening WHO may reply must not widen WHAT they may reply as");
|
||||
}
|
||||
|
||||
/** A primary with no pane — token mode, or off-host — owns nothing and must stay a sender only. */
|
||||
@Test
|
||||
void anUnnamedPrimaryWithNoPaneOwnsNothing() {
|
||||
Principal p = new CallerResolver(nonWorkerIdentity(), true, "s3cret")
|
||||
.resolve("127.0.0.1", 99, "Bearer s3cret");
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
assertNull(p.terminal());
|
||||
assertFalse(p.ownsSession(null), "a null terminal must never match a null session id");
|
||||
assertFalse(Authz.permits(p, Authz.Action.REPLY, null));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLeadKeepsEveryOrchestrationRightItAlreadyHad() {
|
||||
Principal lead = new CallerResolver(workerIdentity(), false, null, Map.of("term_a", "opus-5.0"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertTrue(Authz.permits(lead, Authz.Action.SPAWN, null));
|
||||
assertTrue(Authz.permits(lead, Authz.Action.SEND, "term_worker"));
|
||||
assertTrue(Authz.permits(lead, Authz.Action.STOP, null));
|
||||
assertTrue(Authz.permits(lead, Authz.Action.DRAIN, null));
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-531: the registry is read per resolve, not snapshotted at construction — a lead that
|
||||
* labels its tab after the daemon booted is recognised without a restart.
|
||||
*/
|
||||
@Test
|
||||
void aLeadRegisteredAfterConstructionIsHonouredWithoutRebuildingTheResolver() {
|
||||
Map<String, String> live = new java.util.HashMap<>();
|
||||
CallerResolver r = CallerResolver.withLeads(workerIdentity(), false, null, () -> live);
|
||||
|
||||
assertEquals(Role.WORKER, r.resolve("127.0.0.1", 42, null).role());
|
||||
|
||||
live.put("term_a", "gpt-sol-5.6"); // the scanner sees a newly-labelled tab
|
||||
|
||||
Principal p = r.resolve("127.0.0.1", 42, null);
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
assertEquals("gpt-sol-5.6", p.name());
|
||||
}
|
||||
|
||||
/** The map form must stay a snapshot: a caller handing over a map is not offering live state. */
|
||||
@Test
|
||||
void theMapFormIsCopiedSoLaterMutationCannotGrantLeadership() {
|
||||
Map<String, String> mutable = new java.util.HashMap<>();
|
||||
CallerResolver r = new CallerResolver(workerIdentity(), false, null, mutable);
|
||||
|
||||
mutable.put("term_a", "sneaky");
|
||||
|
||||
assertEquals(Role.WORKER, r.resolve("127.0.0.1", 42, null).role());
|
||||
}
|
||||
|
||||
// ── CB-548: architect slots ─────────────────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void aBoundArchitectPaneResolvesToArchitectBeforeTheWorkerFallback() {
|
||||
// This is the production construction path used by Bridged.
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, boundMembers("architect:lead-designer", MemberRole.ARCHITECT))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.ARCHITECT, p.role(),
|
||||
"a terminal bound to an architect slot is an architect, NOT the generic worker it "
|
||||
+ "would otherwise resolve to");
|
||||
assertEquals("lead-designer", p.name(), "whoami must say WHICH slot is asking");
|
||||
assertEquals("term_a", p.terminal(), "the pane identity is carried so ownsSession works");
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectNeedsNoTokenEvenInTokenMode() {
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), true, "s3cret",
|
||||
Map::of, boundMembers("architect:lead-designer", MemberRole.ARCHITECT))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.ARCHITECT, p.role(),
|
||||
"the pane mapping is as unforgeable as a worker's — it outranks the token path");
|
||||
}
|
||||
|
||||
@Test
|
||||
void anUnboundPaneStillResolvesAsAWorker() {
|
||||
MemberRegistry members = new MemberRegistry(new BridgedConfig.Fleet(Map.of(),
|
||||
Map.of("lead-designer", new BridgedConfig.Slot("sonnet")), Map.of(), Map.of(), null));
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, members).resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role());
|
||||
assertNull(p.name());
|
||||
}
|
||||
|
||||
/** CB-548 precedence: lead > architect > worker, so a pane named in BOTH is still a lead. */
|
||||
@Test
|
||||
void aLeadWinsOverAnArchitectBindingForTheSamePane() {
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
() -> Map.of("term_a", "opus-5.0"),
|
||||
boundMembers("architect:lead-designer", MemberRole.ARCHITECT))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role(),
|
||||
"a pane the config calls a lead must keep resolving as a lead — no behaviour change "
|
||||
+ "when an architect binding is added to an existing fleet");
|
||||
assertEquals("opus-5.0", p.name());
|
||||
}
|
||||
|
||||
/** The registry is live, like leads: a binding injected after construction is honoured. */
|
||||
@Test
|
||||
void anArchitectBoundAfterConstructionIsHonouredWithoutRebuildingTheResolver() {
|
||||
MemberRegistry members = new MemberRegistry(new BridgedConfig.Fleet(Map.of(),
|
||||
Map.of("lead-designer", new BridgedConfig.Slot("sonnet")), Map.of(), Map.of(), null));
|
||||
CallerResolver r = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, members);
|
||||
|
||||
assertEquals(Role.WORKER, r.resolve("127.0.0.1", 42, null).role());
|
||||
|
||||
assertTrue(members.bind("architect:lead-designer", "term_a")); // the later lifecycle binds the slot
|
||||
|
||||
assertEquals(Role.ARCHITECT, r.resolve("127.0.0.1", 42, null).role());
|
||||
assertEquals("architect:lead-designer", r.members().get("term_a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void describeNamesTheArchitectSlot() {
|
||||
assertEquals("architect:lead-designer",
|
||||
Principal.architect("lead-designer", "term_a", 1).describe());
|
||||
}
|
||||
|
||||
/** An architect acts only as its own pane — the same ownsSession rule as a worker or lead. */
|
||||
@Test
|
||||
void anArchitectOwnsItsOwnPaneAndNoOther() {
|
||||
Principal arch = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, boundMembers("architect:lead-designer", MemberRole.ARCHITECT))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertTrue(arch.ownsSession("term_a"));
|
||||
assertTrue(Authz.permits(arch, Authz.Action.REPLY, "term_a"));
|
||||
assertFalse(arch.ownsSession("term_b"));
|
||||
assertFalse(Authz.permits(arch, Authz.Action.REPLY, "term_b"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBoundNonArchitectSlotStillResolvesAsAWorker() {
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, boundMembers("dev:builder", MemberRole.DEV))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role(), "a dev binding must never grant architect rights");
|
||||
}
|
||||
|
||||
@Test
|
||||
void tokenModeRequiresANonEmptyConfiguredToken() {
|
||||
ConnectionIdentity id = nonWorkerIdentity();
|
||||
|
||||
assertThrows(IllegalArgumentException.class, () -> new CallerResolver(id, true, null));
|
||||
assertThrows(IllegalArgumentException.class, () -> new CallerResolver(id, true, " "));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,269 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.Future;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-548 — the architect-slot registry: the config snapshot of slot → profile, and the live
|
||||
* terminal → slot bindings it owns. The role a binding produces is asserted in
|
||||
* {@link CallerResolverTest}; this pins the registry object itself — its invariants and their
|
||||
* thread-safety.
|
||||
*/
|
||||
class MemberRegistryTest {
|
||||
|
||||
/**
|
||||
* Slot keys are qualified by role (CB-557): a bare name is unique only within its pool, so
|
||||
* {@code sonnet} can be both a developer and a reviewer, while a terminal binds to exactly one.
|
||||
*/
|
||||
private static final String DESIGNER = "architect:lead-designer";
|
||||
private static final String REVIEWER = "architect:code-reviewer";
|
||||
|
||||
private static BridgedConfig.Fleet fleetWith(Map<String, BridgedConfig.Slot> architects) {
|
||||
return new BridgedConfig.Fleet(Map.of(), architects, Map.of(), Map.of(), null);
|
||||
}
|
||||
|
||||
private static Map<String, BridgedConfig.Slot> architects() {
|
||||
Map<String, BridgedConfig.Slot> pool = new LinkedHashMap<>();
|
||||
pool.put("lead-designer", new BridgedConfig.Slot("sonnet"));
|
||||
pool.put("code-reviewer", new BridgedConfig.Slot("gx10"));
|
||||
return pool;
|
||||
}
|
||||
|
||||
private final MemberRegistry registry = new MemberRegistry(fleetWith(architects()));
|
||||
|
||||
@Test
|
||||
void exposesTheConfiguredSlotsQualifiedByRole() {
|
||||
assertEquals(Set.of(DESIGNER, REVIEWER), registry.slots().keySet());
|
||||
assertTrue(registry.isSlot(REVIEWER));
|
||||
assertFalse(registry.isSlot("nope"));
|
||||
assertFalse(registry.isSlot("lead-designer"),
|
||||
"the bare name is not the key — it is unique only inside its pool");
|
||||
}
|
||||
|
||||
/** The case the pool shape exists for: one profile serving two roles is not a duplicate. */
|
||||
@Test
|
||||
void oneProfileMayServeTwoRolesUnderTheSameSlotName() {
|
||||
Map<String, BridgedConfig.Slot> devs = Map.of("sonnet", new BridgedConfig.Slot("sonnet"));
|
||||
Map<String, BridgedConfig.Slot> revs = Map.of("sonnet", new BridgedConfig.Slot("sonnet"));
|
||||
MemberRegistry r = new MemberRegistry(
|
||||
new BridgedConfig.Fleet(Map.of(), Map.of(), devs, revs, null));
|
||||
|
||||
assertEquals(Set.of("dev:sonnet", "reviewer:sonnet"), r.slots().keySet());
|
||||
assertEquals(MemberRole.DEV, r.roleForSlot("dev:sonnet"));
|
||||
assertEquals(MemberRole.REVIEWER, r.roleForSlot("reviewer:sonnet"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void theSpawnLifecycleReadsTheProfileBackFromASlot() {
|
||||
assertEquals("sonnet", registry.profileForSlot(DESIGNER));
|
||||
assertEquals("gx10", registry.profileForSlot(REVIEWER));
|
||||
assertNull(registry.profileForSlot("unknown"), "an unknown slot has no profile");
|
||||
}
|
||||
|
||||
@Test
|
||||
void startsEmptySoNoTerminalResolvesToAnArchitect() {
|
||||
assertTrue(registry.snapshot().isEmpty());
|
||||
assertNull(registry.slotForTerminal("term_design"),
|
||||
"config declares no architect terminal — nothing is recognised until a bind");
|
||||
assertNull(registry.slotForTerminal(null), "no terminal ⇒ no slot");
|
||||
}
|
||||
|
||||
// ── bind ──────────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void bindResolvesTheTerminalToTheSlot() {
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
assertEquals(DESIGNER, registry.slotForTerminal("term_design"));
|
||||
assertEquals(Map.of("term_design", DESIGNER), registry.snapshot());
|
||||
}
|
||||
|
||||
@Test
|
||||
void bindRefusesAnUnknownSlot() {
|
||||
assertFalse(registry.bind("nope", "term_x"),
|
||||
"a slot that is not configured must be refused — bind is not a way to invent one");
|
||||
assertNull(registry.slotForTerminal("term_x"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void bindRefusesATerminalInTwoSlots() {
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
assertFalse(registry.bind(REVIEWER, "term_design"),
|
||||
"a terminal may occupy at most one slot");
|
||||
assertEquals(DESIGNER, registry.slotForTerminal("term_design"),
|
||||
"the first binding survives the refused second");
|
||||
}
|
||||
|
||||
@Test
|
||||
void bindRefusesASlotWithTwoTerminals() {
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
assertFalse(registry.bind(DESIGNER, "term_other"),
|
||||
"a slot may host at most one terminal");
|
||||
assertEquals(DESIGNER, registry.slotForTerminal("term_design"),
|
||||
"the first binding survives the refused second");
|
||||
assertNull(registry.slotForTerminal("term_other"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void rebindingTheSamePairIsAnIdempotentNoOp() {
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"),
|
||||
"the same terminal → slot is harmless to repeat");
|
||||
assertEquals(1, registry.snapshot().size());
|
||||
}
|
||||
|
||||
// ── unbind ────────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void unbindRemovesTheExactBinding() {
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
assertTrue(registry.unbind(DESIGNER, "term_design"));
|
||||
assertNull(registry.slotForTerminal("term_design"));
|
||||
assertTrue(registry.snapshot().isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aStaleUnbindDoesNotRemoveAReplacement() {
|
||||
// Bind, tear down, and stand the slot back up with a NEW terminal.
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
registry.unbind(DESIGNER, "term_design");
|
||||
assertTrue(registry.bind(DESIGNER, "term_new"));
|
||||
|
||||
// A late unbind naming the OLD terminal must not remove the replacement binding.
|
||||
assertFalse(registry.unbind(DESIGNER, "term_design"));
|
||||
assertEquals(DESIGNER, registry.slotForTerminal("term_new"),
|
||||
"the replacement terminal stays bound");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aStaleUnbindForATerminalThatMovedSlotsDoesNothing() {
|
||||
// term_design starts in lead-designer, is torn down, and stands back up in a FREE slot.
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
registry.unbind(DESIGNER, "term_design");
|
||||
assertTrue(registry.bind(REVIEWER, "term_design"));
|
||||
|
||||
// Unbinding against the slot it no longer occupies is refused; the new binding is intact.
|
||||
assertFalse(registry.unbind(DESIGNER, "term_design"),
|
||||
"the old slot must not unbind a terminal that moved elsewhere");
|
||||
assertEquals(REVIEWER, registry.slotForTerminal("term_design"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void unbindOfNothingIsAFalseNoOp() {
|
||||
assertFalse(registry.unbind(DESIGNER, "term_design"),
|
||||
"nothing was bound, so nothing is removed");
|
||||
}
|
||||
|
||||
// ── snapshot ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void theSnapshotIsAnImmutableCopyNotAliveState() {
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
Map<String, String> snap = registry.snapshot();
|
||||
|
||||
assertThrows(UnsupportedOperationException.class, () -> snap.put("x", "y"),
|
||||
"a handed-out snapshot cannot be mutated in place");
|
||||
|
||||
// Later binds must not leak into an earlier snapshot.
|
||||
assertTrue(registry.bind(REVIEWER, "term_review"));
|
||||
assertFalse(snap.containsKey("term_review"),
|
||||
"a snapshot is a point-in-time copy, not a live view");
|
||||
}
|
||||
|
||||
// ── concurrency (CB-548 invariants hold under contention) ─────────────────────────────────
|
||||
|
||||
@Test
|
||||
void concurrentBindsNeverGiveASlotTwoTerminals() throws Exception {
|
||||
int n = 16;
|
||||
ExecutorService pool = Executors.newFixedThreadPool(n);
|
||||
try {
|
||||
CountDownLatch go = new CountDownLatch(1);
|
||||
List<Future<Boolean>> results = new ArrayList<>();
|
||||
for (int i = 0; i < n; i++) {
|
||||
final String term = "term_" + i; // every thread races for the SAME slot
|
||||
results.add(pool.submit(() -> {
|
||||
go.await();
|
||||
return registry.bind(DESIGNER, term);
|
||||
}));
|
||||
}
|
||||
go.countDown();
|
||||
|
||||
int won = 0;
|
||||
for (Future<Boolean> r : results) {
|
||||
if (r.get()) {
|
||||
won++;
|
||||
}
|
||||
}
|
||||
assertEquals(1, won, "exactly one terminal may win the sole slot, got " + won);
|
||||
assertEquals(1, registry.snapshot().size(),
|
||||
"the slot hosts at most one terminal after the race");
|
||||
} finally {
|
||||
pool.shutdownNow();
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void concurrentBindsNeverPutOneTerminalInTwoSlots() throws Exception {
|
||||
int n = 16;
|
||||
ExecutorService pool = Executors.newFixedThreadPool(n);
|
||||
try {
|
||||
CountDownLatch go = new CountDownLatch(1);
|
||||
List<Future<String>> results = new ArrayList<>();
|
||||
for (int i = 0; i < n; i++) {
|
||||
final String slot = (i % 2 == 0) ? DESIGNER : REVIEWER; // all race for ONE terminal
|
||||
results.add(pool.submit(() -> {
|
||||
go.await();
|
||||
return registry.bind(slot, "shared_term")
|
||||
? registry.slotForTerminal("shared_term") : null;
|
||||
}));
|
||||
}
|
||||
go.countDown();
|
||||
|
||||
// Rebinding the same terminal to the same slot is a harmless idempotent true, so count
|
||||
// winners is not the assertion — agreement is: every thread that reported success must
|
||||
// have seen the terminal in the SAME slot, never in two at once.
|
||||
String bound = null;
|
||||
boolean conflict = false;
|
||||
for (Future<String> r : results) {
|
||||
String s = r.get();
|
||||
if (s != null) {
|
||||
if (bound == null) {
|
||||
bound = s;
|
||||
} else if (!bound.equals(s)) {
|
||||
conflict = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
assertFalse(conflict, "a terminal was observed in two slots at once");
|
||||
assertNotNull(bound, "at least one thread bound the terminal");
|
||||
assertEquals(1, registry.snapshot().size(),
|
||||
"the terminal occupies exactly one slot in the final snapshot");
|
||||
assertEquals(bound, registry.slotForTerminal("shared_term"));
|
||||
} finally {
|
||||
pool.shutdownNow();
|
||||
}
|
||||
}
|
||||
|
||||
/** A handed-over slot map is snapshotted at construction, not offered as live state. */
|
||||
@Test
|
||||
void theSlotSnapshotIsFixedByConstruction() {
|
||||
Map<String, BridgedConfig.Slot> mutable = architects();
|
||||
MemberRegistry r = new MemberRegistry(fleetWith(mutable));
|
||||
|
||||
mutable.put("hijack", new BridgedConfig.Slot("gx10"));
|
||||
|
||||
assertFalse(r.isSlot("architect:hijack"), "a handed-over map is not offered as live state");
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user