Compare commits
233 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 541df87272 | |||
| 42f46dfe9a | |||
| 9088d2b2c5 | |||
| 500bfa2c33 | |||
| b525b0f08f | |||
| 976eff8ad1 | |||
| 0af902ec43 | |||
| 9118ce2537 | |||
| 2f48e08f1f | |||
| fec284e7cb | |||
| 8e2e4c5e73 | |||
| 0edc6615fc | |||
| b745e159de | |||
| aac29d604c | |||
| c884802b13 | |||
| 5f5573a24e | |||
| 74b0087ebb | |||
| 927e0151d4 | |||
| 5275922d1d | |||
| e186c7945a | |||
| 33a6e77f0e | |||
| 4e47489d53 | |||
| c76b2f149e | |||
| 826fffe05b | |||
| 75f57cdba7 | |||
| f556af5d4e | |||
| d5f33f0c6e | |||
| 16e17b32ad | |||
| 988494e18b | |||
| c4549a5e20 | |||
| 46fa4f38d5 | |||
| 20e0e68ad7 | |||
| 17468a234a | |||
| 8a53d5bfc6 | |||
| 695da7418e | |||
| abd26c796b | |||
| 24559d81ac | |||
| 6c1c2c3994 | |||
| 01fab15713 | |||
| 0cd00e71c3 | |||
| bf0ff2adbf | |||
| ed4bbc1c56 | |||
| c456402cc5 | |||
| 4f0bf667b1 | |||
| 61af9aa574 | |||
| 3b59b34e76 | |||
| 65b38997f7 | |||
| 7510f7649c | |||
| 619792a81c | |||
| cea1183f75 | |||
| e2af4c5ae4 | |||
| dfd5f82894 | |||
| e81944cef6 | |||
| 7bdd39ab9a | |||
| f4b38f040e | |||
| 799668e129 | |||
| 890190263e | |||
| f8522edacd | |||
| b958747855 | |||
| 078bde2c02 | |||
| 0b10ea987b | |||
| 1553d38182 | |||
| d04b075996 | |||
| 644927636d | |||
| 95e45007aa | |||
| 7a583c4045 | |||
| 65bce058bb | |||
| 48b437083b | |||
| fa0612859b | |||
| 4ffbcd0b7d | |||
| a0cd053fd9 | |||
| f9a5e066b5 | |||
| e9bc192160 | |||
| e0a57988ad | |||
| b414a74c26 | |||
| 2c3796d598 | |||
| 7e0ff9ab06 | |||
| 6d1565d8b6 | |||
| e18e002d2f | |||
| c18572ea9d | |||
| e9d02bc5e8 | |||
| a2108a8a14 | |||
| 57f8fa257a | |||
| 61944fc045 | |||
| 4b48d2d921 | |||
| 3aa145cbef | |||
| 4875127daa | |||
| d14a624421 | |||
| 246f50b778 | |||
| 15b53c6cfa | |||
| 3a7ef0adbd | |||
| 293a305748 | |||
| e5038c6d13 | |||
| 13ea79f6fd | |||
| f55d3c203b | |||
| 37c4d47d3a | |||
| 04e21c9243 | |||
| 83cac07f6e | |||
| f472e0f782 | |||
| 91c9f981c5 | |||
| dd906526c0 | |||
| ec3001796a | |||
| 86cf4c285a | |||
| cd18887b69 | |||
| 4637c68295 | |||
| 0114bd1fa7 | |||
| 6dee84ca71 | |||
| f004a0c654 | |||
| 6123576c68 | |||
| 21cfc09f8e | |||
| 509530e235 | |||
| d146a01422 | |||
| 91332723db | |||
| 244fbd98a5 | |||
| 5afe8e14d9 | |||
| 73f6b12dd8 | |||
| 793f2e7157 | |||
| cf48983046 | |||
| d038516750 | |||
| f8182e4514 | |||
| bc13b8e92c | |||
| fbe79258bf | |||
| ef1e014b41 | |||
| e3c8393d1b | |||
| 379e03f9d0 | |||
| e13921aa8a | |||
| 8a549d8610 | |||
| 84081b2bd8 | |||
| defe3365c4 | |||
| 4e6201ecd1 | |||
| ccf50f950e | |||
| a7f0211e2f | |||
| 049ce4828c | |||
| c1173346ef | |||
| 7ace184fe6 | |||
| 54b314ace5 | |||
| 224b344445 | |||
| 0b28b4cb0f | |||
| 83129e165c | |||
| 871b595954 | |||
| b67b1585c2 | |||
| 979b2b5632 | |||
| cf4ad186ab | |||
| 5100f215cf | |||
| f129e9b7cd | |||
| 4aed45de19 | |||
| 1e6daa5c73 | |||
| 4c015d76b7 | |||
| 37a11cd168 | |||
| 22ad24db6c | |||
| e94c1b8841 | |||
| cc0ec65714 | |||
| d67d30c58a | |||
| d75ee1cca5 | |||
| 6804676a96 | |||
| 3e5d742ac7 | |||
| 6da2a71050 | |||
| e32ac39faf | |||
| daa243d37a | |||
| 2773ab600d | |||
| 19cdf8dc9f | |||
| 9daf1ec5ba | |||
| c9f0ca9359 | |||
| 84c8a2d2f0 | |||
| ded226abfe | |||
| 9b8d55bc18 | |||
| 11f8709286 | |||
| b034f105c0 | |||
| 6e37722383 | |||
| ffce30afa2 | |||
| e724a59f2d | |||
| 4bf855d225 | |||
| d4c9704007 | |||
| 7c252b5f5f | |||
| f756933879 | |||
| a1aecbf4fc | |||
| 131e7b1ccd | |||
| d0ac6c435f | |||
| 2bc5f3a057 | |||
| ba6b4a5da9 | |||
| da5a987df0 | |||
| 7dd6c46156 | |||
| 3a5cdc5108 | |||
| 3aa69a9e32 | |||
| e056c7e1fa | |||
| d63273d082 | |||
| 0efb65ca0a | |||
| 9fe04bfb08 | |||
| 09d3948acf | |||
| 954351a80b | |||
| 8d51066ddd | |||
| 84102baab4 | |||
| 64e70efdf1 | |||
| 97ecc7136e | |||
| f9073e2320 | |||
| 54d907c314 | |||
| 82c7d6553a | |||
| 19b10e3216 | |||
| 0f79e6bed5 | |||
| aa0cf814ff | |||
| 4aa9d03da2 | |||
| 37abfdfd6e | |||
| d5c3ede215 | |||
| 426855e378 | |||
| 358c6970b5 | |||
| 55ebd5b949 | |||
| 2a61fe69f1 | |||
| ff6aacdc78 | |||
| 5f0ec034d9 | |||
| 31e34d177b | |||
| b2d85af78b | |||
| 968a5c68b6 | |||
| 3c05823f19 | |||
| 2fb46f670f | |||
| 37f7ad4185 | |||
| 926724a279 | |||
| 959f04bc96 | |||
| 7f1b6b3a0d | |||
| ab771ea24e | |||
| a629a7ee73 | |||
| 2052929768 | |||
| a97c287aee | |||
| 8ed2370fbe | |||
| 5b26caca0c | |||
| 6988bfe88f | |||
| 07722a6007 | |||
| fa570ab32a | |||
| 46f87b9cd3 | |||
| 8ffbfd1f06 | |||
| 597ac2e562 | |||
| bc04637694 | |||
| c7b58f2195 | |||
| 01cfda7965 |
@@ -0,0 +1,18 @@
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the bridged MCP gateway.",
|
||||
"owner": {
|
||||
"name": "LTMS"
|
||||
},
|
||||
"plugins": [
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"source": "./plugin",
|
||||
"description": "Make a project bridge-ready: mount the bridged MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.1.0",
|
||||
"author": {
|
||||
"name": "LTMS"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
---
|
||||
name: implementer
|
||||
description: Implementer-role procedure for a bridged worker — verify your worktree, implement the scope, commit, push, open your own PR, and hand off the PR URL. Load this when the lead delegates you an implementation task over bridged.
|
||||
---
|
||||
|
||||
# Implementer worker — procedure
|
||||
|
||||
The turn contract (one `bridge_reply`, `bridge_ask` for the lead's decisions, honest reporting,
|
||||
never merge, never commit `.mcp.json` or `wiki/`) is in **`CLAUDE.md` → Bridge communication →
|
||||
Worker** and already applies. This skill is only the *implement-and-hand-off procedure*.
|
||||
|
||||
You run in an **isolated git worktree on your own branch** — a full peer of the primary (same repo,
|
||||
`CLAUDE.md`, skills), differing in the model behind you and the branch you sit on. Your MCP surface
|
||||
is **only what your launcher mounted** (the bridge): the primary's IDE and forge servers are not
|
||||
yours, and the worktree's `.mcp.json` is deliberately emptied so you cannot inherit them.
|
||||
The worktree model is documented in [`docs/Worker-Git-Workflow.md`](../../../docs/Worker-Git-Workflow.md).
|
||||
|
||||
## 1. Confirm where you are — then never leave
|
||||
|
||||
Before touching anything:
|
||||
|
||||
```bash
|
||||
git rev-parse --show-toplevel # your worktree root — NOT the primary's main tree
|
||||
git branch --show-current # your dedicated branch: worker/<ticket>-<nonce>
|
||||
git status # should be clean at the start
|
||||
```
|
||||
|
||||
Do **all** work here, on this branch. Never `git checkout main`, never rebase onto or push to
|
||||
`main`. The branch is your isolation — respect it.
|
||||
|
||||
**Every path you read, edit, or build is relative to that root.** Work from `$PWD`; if a tool, a
|
||||
brief, or your own memory hands you an absolute path, check it starts with your worktree root
|
||||
before you touch it, and stop if it doesn't. An absolute path pointing anywhere else is the
|
||||
primary's checkout — editing there while building here means **every build you run is of code that
|
||||
does not contain your changes**, and it passes while your work goes nowhere. This has happened:
|
||||
a worker made all 59 of its edits in the primary's tree and never noticed.
|
||||
|
||||
```bash
|
||||
test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-toplevel)"
|
||||
```
|
||||
|
||||
## 2. Implement
|
||||
|
||||
- Implement exactly the scope the lead named. Keep the diff focused; note anything out of scope
|
||||
in your reply instead of widening it.
|
||||
- Match the surrounding code's style, naming, and idioms.
|
||||
|
||||
**Acceptance criterion — a green build, quoted.** Your work is not done until this passes *inside
|
||||
your worktree*:
|
||||
|
||||
```bash
|
||||
cd "$(git rev-parse --show-toplevel)/bridged" && mvn clean install
|
||||
echo "exit=$?"
|
||||
```
|
||||
|
||||
Read its **full** output — never pipe it through `tail`/`head`/`grep`, which hide a failure behind
|
||||
a zero exit. Then quote the real `Tests run: … Failures: … Errors: …` line and the
|
||||
`BUILD SUCCESS`/`FAILURE` verbatim in your reply. If it does not go green, say so with the actual
|
||||
error; a failing build honestly reported is a usable result, a claimed-green one is not. You have
|
||||
no IDE MCP tools, so `mvn` is your only verification — never claim a check you had no way to run.
|
||||
|
||||
## 3. Commit
|
||||
|
||||
```bash
|
||||
git add <the files you changed> # explicitly — never `git add -A` / `git add .`
|
||||
git commit -m "<ticket>: <clear one-line summary>"
|
||||
```
|
||||
|
||||
`.mcp.json` is neutralized and `--skip-worktree` in your worktree — never `git add` it, and never
|
||||
"restore" it from the primary's copy. Same for `wiki/` (a submodule with its own remote).
|
||||
|
||||
## 4. Push
|
||||
|
||||
```bash
|
||||
git push -u origin HEAD
|
||||
```
|
||||
|
||||
Push is over SSH as the same user — no extra credential needed. Never force-push over anything
|
||||
you did not create.
|
||||
|
||||
## 5. Open your own PR to `main`
|
||||
|
||||
Via the gitea REST API. The daemon injected a **repo-scoped token** (`GITEA_TOKEN`) and the forge
|
||||
host (`GITEA_HOST`) into your env for exactly this — the token can create a PR but **cannot
|
||||
merge**.
|
||||
|
||||
```bash
|
||||
API="${GITEA_HOST%/}/api/v1/repos/lms/claude-bridge/pulls"
|
||||
BRANCH="$(git branch --show-current)"
|
||||
curl -sS -X POST "$API" \
|
||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d "$(cat <<JSON
|
||||
{"head": "${BRANCH}", "base": "main",
|
||||
"title": "<ticket>: <concise change summary>",
|
||||
"body": "<what changed and why; reference the ticket; note tests run and their result>"}
|
||||
JSON
|
||||
)"
|
||||
```
|
||||
|
||||
The response JSON carries `"html_url"` — that is your PR URL. On a non-2xx, read the error body,
|
||||
fix it if the cause is yours (e.g. branch not pushed yet), and report the failure rather than
|
||||
inventing a URL. If `GITEA_TOKEN` is unset your profile was not granted PR-create: push the branch
|
||||
and report its name so the lead opens the PR.
|
||||
|
||||
## 6. Hand off — what goes in `bridge_reply`
|
||||
|
||||
The reply is the entire handoff; the lead cannot see your terminal.
|
||||
|
||||
```
|
||||
PR: <html_url from step 5, or "not created: <reason>" + branch name>
|
||||
branch: <your branch>
|
||||
root: <git rev-parse --show-toplevel — proves you worked in your own worktree>
|
||||
files: <worktree-relative paths you changed>
|
||||
build: <the verbatim "Tests run: …" and BUILD SUCCESS/FAILURE lines — or "not run: <why>">
|
||||
summary: <2-3 lines: what you implemented and any caveat the reviewer needs>
|
||||
```
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
autonumber
|
||||
participant L as Lead
|
||||
participant I as Implementer (you)
|
||||
participant G as git / gitea
|
||||
|
||||
L->>I: delegated task (you are in a worktree on your branch)
|
||||
I->>I: "implement here — every path under $PWD"
|
||||
I->>I: "mvn clean install in this worktree, unpiped, until green"
|
||||
I->>G: git commit (never .mcp.json / wiki)
|
||||
I->>G: git push -u origin HEAD
|
||||
I->>G: POST /pulls (GITEA_TOKEN) — open PR to main
|
||||
G-->>I: html_url
|
||||
I->>L: bridge_reply(PR url, branch, files, tests)
|
||||
Note over L,G: the lead reviews the PR and merges on green — you never merge
|
||||
```
|
||||
|
||||
*The implement turn: work in the worktree, commit → push → open the PR, hand off the URL.*
|
||||
@@ -0,0 +1,180 @@
|
||||
---
|
||||
name: port-to-opencode
|
||||
description: Make an OpenCode session a first-class participant in a Claude Code workspace — instructions, MCP servers, and secrets — without duplicating config. Use when onboarding opencode to a project that already has CLAUDE.md and .mcp.json, or when an opencode peer needs the same tools and rules as the Claude session.
|
||||
---
|
||||
|
||||
# Porting a Claude Code workspace to OpenCode
|
||||
|
||||
**The headline: there is almost nothing to port.** OpenCode reads `CLAUDE.md` natively. The only
|
||||
artifact you create is one `opencode.json` mapping MCP servers. Do not translate instructions, do
|
||||
not generate a second rules file, and do not install a sync tool — every one of those makes the
|
||||
workspace worse.
|
||||
|
||||
Everything below was verified against `opencode 1.18.16` and the OpenCode docs.
|
||||
|
||||
## 1. Know what you get for free
|
||||
|
||||
OpenCode's instruction search order:
|
||||
|
||||
```
|
||||
1. walking up from cwd: AGENTS.md , then CLAUDE.md
|
||||
2. global: ~/.config/opencode/AGENTS.md
|
||||
3. Claude Code global: ~/.claude/CLAUDE.md (unless disabled)
|
||||
```
|
||||
|
||||
*"The first matching file wins in each category."*
|
||||
|
||||
Consequences that decide the whole procedure:
|
||||
|
||||
- **A project `CLAUDE.md` is already read.** No port needed.
|
||||
- **Your user-level `~/.claude/CLAUDE.md` is already read too.** Global preferences carry over.
|
||||
- **An `AGENTS.md` in the repo SHADOWS `CLAUDE.md`.** If one exists from a previous Codex port,
|
||||
**delete it** — otherwise opencode reads the stale translated copy instead of the real rules.
|
||||
This is the single most likely way to get this wrong.
|
||||
|
||||
## 2. Create `opencode.json` for MCP servers only
|
||||
|
||||
Project config lives at `opencode.json` in the repo root; the global one is
|
||||
`~/.config/opencode/opencode.json`. **Configs merge, they do not replace** — so machine-local
|
||||
servers belong in the global file and shared ones in the project file.
|
||||
|
||||
Map each entry from `.mcp.json`:
|
||||
|
||||
| `.mcp.json` | `opencode.json` |
|
||||
|---|---|
|
||||
| `"type": "http"` / `"sse"` | `"type": "remote"`, `"url"` |
|
||||
| `"type": "stdio"` | `"type": "local"`, `"command": ["bin", "arg"]` |
|
||||
| `"command"` + `"args"` | single `"command"` array |
|
||||
| `"env"` | `"environment"` |
|
||||
| `"headers"` | `"headers"` |
|
||||
|
||||
```json
|
||||
{
|
||||
"$schema": "https://opencode.ai/config.json",
|
||||
"instructions": ["CLAUDE.md"],
|
||||
"mcp": {
|
||||
"bridged": { "type": "remote", "url": "http://127.0.0.1:8765/mcp", "enabled": true },
|
||||
"context7": { "type": "remote", "url": "https://example.dev/mcp", "enabled": true,
|
||||
"headers": { "Authorization": "Bearer {env:CONTEXT7_TOKEN}" } },
|
||||
"gitea": { "type": "local", "command": ["gitea-mcp", "-t", "stdio"], "enabled": true,
|
||||
"environment": { "GITEA_ACCESS_TOKEN": "{env:GITEA_ACCESS_TOKEN}" } }
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Set `instructions` explicitly even though `CLAUDE.md` is found anyway — the fallback only applies
|
||||
while no `AGENTS.md` exists, and being explicit survives someone adding one later.
|
||||
|
||||
## 3. Reference secrets, never embed them
|
||||
|
||||
OpenCode substitutes at load time, in both `headers` and `environment`:
|
||||
|
||||
```
|
||||
{env:VARIABLE_NAME} value from the environment
|
||||
{file:~/.secrets/token} value read from a file
|
||||
```
|
||||
|
||||
**No credential ever belongs in `opencode.json`.** With `{env:…}` there is no reason to write one,
|
||||
which is what makes this file safe to commit — and it must be committable, because a peer running
|
||||
in a git worktree receives tracked files only.
|
||||
|
||||
**But `{env:…}` reads OpenCode's *process* environment — and OpenCode has no env store of its
|
||||
own.** A Claude Code `env` block in `~/.claude/settings.json` or `.claude/settings.local.json` does
|
||||
**not** reach it: those files are Claude Code's, and the variables exist only inside processes
|
||||
Claude Code spawned. Verify with a clean login shell, not the shell your agent hands you:
|
||||
|
||||
```bash
|
||||
env -u MY_TOKEN zsh -lc 'echo "${MY_TOKEN:-NOT IN PROFILE}"'
|
||||
```
|
||||
|
||||
So `{env:…}` only works if something puts the variable in the environment first. Pick the supply
|
||||
route by who launches opencode:
|
||||
|
||||
| Launcher | Route |
|
||||
|---|---|
|
||||
| a human, from a terminal | one central store, sourced by the login shell |
|
||||
| a spawner (bridge, CI, IDE) | `{env:…}`, with the spawner injecting the variable |
|
||||
|
||||
**For the human case, keep every credential in one file the login shell sources.** Here that file
|
||||
is `${SHARED_ENV}/tools/secrets.sh`, sourced from `${SHARED_ENV}/.ltms`, kept at mode 600 and never
|
||||
committed. `opencode.json` then names variables and holds no values:
|
||||
|
||||
```json
|
||||
"headers": { "Authorization": "Bearer {env:CONTEXT7_TOKEN}" }
|
||||
```
|
||||
|
||||
Both routes end at the same syntax, and that is the point. The file does not change when a human
|
||||
launches opencode instead of the bridge.
|
||||
|
||||
**`{file:…}` also works, and this project moved away from it.** Opencode resolves a relative
|
||||
`{file:}` path against the project root, so a gitignored `.secrets/` beside `opencode.json` needs no
|
||||
shell setup at all. It did not fail; the problem is that it makes a second copy of the token. The
|
||||
same secret then lives in two places, and the copy you forget is the one that leaks or goes stale.
|
||||
One store with many references is easier to rotate and to audit.
|
||||
|
||||
**One catch survives either choice, so state it out loud:** what a human's shell exports does not
|
||||
reach a spawned peer, and neither does a gitignored `.secrets/` — a git worktree receives tracked
|
||||
files only. Spawned peers must be fed through `{env:…}` by whatever launches them. Check the
|
||||
variable *names* match: a spawner often injects under a different name than your shell uses, and
|
||||
the config has no fallback. In this repo the bridge goes further: it neutralizes a worktree's
|
||||
`opencode.json`, so a member cannot inherit the primary's credentials by accident.
|
||||
|
||||
## 4. Do not port machine-local MCP servers
|
||||
|
||||
IDE indexes, language servers, editor bridges — anything bound to *your* checkout — stay out of the
|
||||
project file. Put them in `~/.config/opencode/opencode.json` if you want them personally.
|
||||
|
||||
A committed project config reaches every worktree. A worker that mounts servers whose paths point
|
||||
into the primary's checkout will edit the primary's files while building in its own — every build
|
||||
passes, every change lands in the wrong tree.
|
||||
|
||||
Port the servers the work needs. Leave the rest.
|
||||
|
||||
## 5. Verify against the running agent, not the file
|
||||
|
||||
A file on disk proves nothing about what the agent loaded.
|
||||
|
||||
```bash
|
||||
opencode run "In one line: state a rule from this project's instructions."
|
||||
```
|
||||
|
||||
The answer must reflect the actual `CLAUDE.md`. If it answers generically, the instructions did not
|
||||
reach the model and everything after this is built on sand.
|
||||
|
||||
Then confirm the tools are mounted:
|
||||
|
||||
```bash
|
||||
opencode mcp list
|
||||
```
|
||||
|
||||
**"connected" does not mean "working".** A stdio server with a missing credential still completes
|
||||
the MCP handshake and reports green; only a real tool call reveals it. Verified: `gitea` showed
|
||||
`✓ connected` with no token, then failed the first call with `token is required`. A remote server
|
||||
is more honest (`⚠ needs authentication`), but do not rely on that difference — **exercise one
|
||||
authenticated tool per server**:
|
||||
|
||||
```bash
|
||||
opencode run "Call <server>'s <tool>. Report the result or the exact error. One line."
|
||||
```
|
||||
|
||||
Check too that no server you deliberately withheld is present.
|
||||
|
||||
## 6. Report
|
||||
|
||||
State what you changed, which servers crossed and which you withheld and why, and quote the
|
||||
verification answer verbatim. If any server failed to connect, say so plainly — a partially mounted
|
||||
peer is worse than a missing one, because it looks configured.
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **A leftover `AGENTS.md` silently wins over `CLAUDE.md`.** Check for one before anything else.
|
||||
- **`opencode.json` is merged, not overridden** — a global entry and a project entry with the same
|
||||
server name both matter; keep names distinct unless you intend to layer them.
|
||||
- **`enabled: false`** turns a server off without deleting its config — prefer it over removal when
|
||||
you may want the server back.
|
||||
- **OAuth-based servers** store tokens in `~/.local/share/opencode/mcp-auth.json` after
|
||||
`opencode mcp auth <server>`; that is machine state, never config to commit.
|
||||
- **Skills and subagents do not port.** OpenCode uses its own agent markdown under
|
||||
`.opencode/agents/`; `.claude/skills/**` is not read. If a delegation brief tells a peer to load a
|
||||
skill by name, that instruction has no effect on an opencode peer — spell the procedure out in the
|
||||
brief, or author the equivalent agent file.
|
||||
@@ -0,0 +1,51 @@
|
||||
---
|
||||
name: reviewer
|
||||
description: Reviewer-role procedure for a bridged worker — how to work a review scope and the exact shape of the finding to report. Load this when the lead delegates you a code review over bridged.
|
||||
---
|
||||
|
||||
# Reviewer worker — procedure
|
||||
|
||||
The turn contract (one `bridge_reply`, `bridge_ask` for the lead's decisions, honest reporting,
|
||||
never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and already applies. This
|
||||
skill is only the *review procedure*: how to work the scope, and the exact shape of what you
|
||||
send back.
|
||||
|
||||
## 1. Read the whole scope before you judge
|
||||
|
||||
The delegation names your scope — a file, a diff, a PR, a function. **Read all of it first.**
|
||||
A review that fires on a snippet misses the caller that makes it safe (or the one that makes it
|
||||
a bug). Reviewing part of the scope and guessing the rest is the most common way a reviewer is
|
||||
wrong.
|
||||
|
||||
## 2. Stay in the scope
|
||||
|
||||
- Review **only** what you were assigned. Something elsewhere looks wrong? One line in your
|
||||
reply — do not go hunt it. Wandering is how two reviewers report the same thing and neither
|
||||
covers what it was given.
|
||||
- Do **not** edit files or run the build. You review; the owner acts.
|
||||
|
||||
## 3. Reach for `bridge_ask` only for a genuine fork
|
||||
|
||||
Ambiguous requirement, a missing acceptance criterion, "intended or a bug?", or two defensible
|
||||
fixes with different consequences — those are the lead's call, and guessing produces a
|
||||
confident-but-wrong finding. Anything you could settle by reading more code is yours to settle.
|
||||
|
||||
## 4. The finding — what goes in `bridge_reply`
|
||||
|
||||
Report the **single most important** real issue in the scope, in these four lines, under
|
||||
~90 words:
|
||||
|
||||
```
|
||||
1. <path>:<line>
|
||||
2. issue: <one sentence — what is wrong and why it matters>
|
||||
3. fix: <one line — the concrete change>
|
||||
4. severity: high | medium | low
|
||||
```
|
||||
|
||||
- **Nothing real after reading?** Reply `NO ISSUE` and one line saying why. A clean review is a
|
||||
valid result; a fabricated issue is worse than none.
|
||||
- **Severity:** `high` = wrong result, data loss, security, or a hang/crash on a real path ·
|
||||
`medium` = a real bug on an edge path, or a correctness risk under load/concurrency ·
|
||||
`low` = clarity, a latent foot-gun, or a smell with no current failure.
|
||||
- Be specific and verifiable: a line number and a one-line repro beat an adjective. If you can't
|
||||
point at where it goes wrong, you haven't found it yet.
|
||||
@@ -0,0 +1,107 @@
|
||||
name: CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
branches: [main]
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
# The wiki submodule is docs only and is not needed to build — leave it unfetched so CI
|
||||
# does not depend on the wiki repo being reachable.
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
# The runner image ships an older default-jdk; bridged sets maven.compiler.release=25, so
|
||||
# provision the JDK explicitly rather than apt-installing whatever "default" means today.
|
||||
- name: Set up JDK 25
|
||||
uses: actions/setup-java@v4
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '25'
|
||||
cache: maven
|
||||
|
||||
# setup-java provisions the JDK only — it does NOT install Maven, and the runner image has
|
||||
# no mvn on PATH (a bare `mvn` exits 127). Install it separately. apt pulls a default JRE as
|
||||
# a dependency; JAVA_HOME from setup-java still wins, which the version check below proves.
|
||||
- name: Install Maven
|
||||
run: |
|
||||
apt-get update && apt-get install -y --no-install-recommends maven
|
||||
mvn -version
|
||||
|
||||
- name: Build and test
|
||||
working-directory: bridged
|
||||
# This IS the mock-socket surface CB-503 asks for: the pom's `default-excludes` profile
|
||||
# already sets excludedGroups=contract, so the @Tag("contract") tests — which need a live
|
||||
# herdr socket and a RabbitMQ container — are excluded without any flag here. Everything
|
||||
# that runs does so against the fake UDS herdr and fake ccs/claude stubs.
|
||||
run: mvn -B clean install
|
||||
|
||||
# Deliberately NOT actions/upload-artifact: this Gitea instance presents as GHES, and
|
||||
# @actions/artifact v2+ (i.e. upload-artifact@v4) refuses to run there —
|
||||
# "GHESNotSupportedError ... not currently supported on GHES", which red-Xes an otherwise
|
||||
# green build. Since the artifact could not be retrieved anyway, dump the failing tests into
|
||||
# the log instead, where they are actually readable.
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: bridged
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
grep -qE "Failures: [1-9]|Errors: [1-9]" "$f" && { echo "===== $f ====="; cat "$f"; }
|
||||
done
|
||||
exit 0
|
||||
|
||||
# CB-521 — actually run the AMQP contract test in CI, against a REAL broker. The broker is a
|
||||
# RabbitMQ SERVICE CONTAINER, not Testcontainers-with-Docker: the runner image has no Docker, so
|
||||
# AmqpReplyInboxContractTest reads AMQP_URI (set below to the service's network alias) and binds
|
||||
# straight to it — no Docker, no skipped tests. This separation (build job hermetic and
|
||||
# Docker-free; contract job broker-provided) is deliberate — see the default-excludes/contract
|
||||
# profiles in bridged/pom.xml. `setup-java` provides the JDK only; Maven is installed separately,
|
||||
# exactly as in the build job above.
|
||||
contract:
|
||||
runs-on: ubuntu-latest
|
||||
services:
|
||||
rabbitmq:
|
||||
image: rabbitmq:3.13 # same AMQP 0-9-1 engine the local Testcontainers fixture uses
|
||||
env:
|
||||
RABBITMQ_DEFAULT_USER: guest
|
||||
RABBITMQ_DEFAULT_PASS: guest
|
||||
ports:
|
||||
- 5672:5672
|
||||
env:
|
||||
# Service containers are reachable from the job by their network alias on their internal port.
|
||||
AMQP_URI: amqp://guest:guest@rabbitmq:5672
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Set up JDK 25
|
||||
uses: actions/setup-java@v4
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '25'
|
||||
cache: maven
|
||||
|
||||
- name: Install Maven
|
||||
run: |
|
||||
apt-get update && apt-get install -y --no-install-recommends maven
|
||||
mvn -version
|
||||
|
||||
# The `contract` profile clears the default-excludes group, so the @Tag("contract") AMQP test
|
||||
# runs against the RabbitMQ service container (AMQP_URI). Pinned to the one contract test to
|
||||
# avoid re-running the unit suite already covered by the `build` job.
|
||||
- name: Contract tests
|
||||
working-directory: bridged
|
||||
run: mvn -B -Pcontract test -Dtest=AmqpReplyInboxContractTest
|
||||
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: bridged
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
grep -qE "Failures: [1-9]|Errors: [1-9]" "$f" && { echo "===== $f ====="; cat "$f"; }
|
||||
done
|
||||
exit 0
|
||||
+13
@@ -0,0 +1,13 @@
|
||||
# No secret belongs in this repo any more: every credential lives in one shell-level store
|
||||
# (${SHARED_ENV}/tools/secrets.sh), and opencode.json reads it as {env:...}. This line stays as a
|
||||
# backstop, so a workspace-scoped copy that someone re-creates by habit still cannot be committed.
|
||||
.secrets/
|
||||
|
||||
# Settings backups inherit the env block — and secrets with it.
|
||||
.claude/settings.local.json.bak*
|
||||
|
||||
# Daemon runtime artefacts. bridged appends its log wherever it is launched from, so both the
|
||||
# repo root and bridged/ collect one; neither belongs in git.
|
||||
bridged.out
|
||||
bridged/bridged.out
|
||||
logs/
|
||||
@@ -1,7 +1,245 @@
|
||||
# claude-bridge — project instructions
|
||||
|
||||
## Bridge communication (enforced — read this first)
|
||||
|
||||
> **Canonical block.** Everything down to §Layering is the portable bridge charter, copied verbatim
|
||||
> into every project that mounts the bridge MCP. Keep it byte-identical with the template in the
|
||||
> wiki ([Use Cases](https://git.ltms.dev/lms/claude-bridge/wiki/7-Use-Cases) → *The portable
|
||||
> CLAUDE.md block*); improvements go to the template first, then out to each project. Anything
|
||||
> specific to *this* repo lives under §Project addendum below, never inline above it.
|
||||
|
||||
If no `bridge_*` MCP tools are mounted in this session, this section does not apply — skip it.
|
||||
|
||||
`bridged` is the **sole communication gateway** between agents here. The orchestrating session (the
|
||||
**primary**) and every delegated peer (a **member**) mount the *same* MCP server and talk only
|
||||
through its `bridge_*` tools. No session addresses a peer, a broker, or the network directly.
|
||||
|
||||
### Which role am I? — settle this before acting
|
||||
|
||||
**Every role reads this file.** A member runs in a git worktree of this same repo, so it inherits
|
||||
this `CLAUDE.md` verbatim, and every rule below is role-conditional.
|
||||
|
||||
**Call `bridge_whoami`.** It returns `primary`, `worker`, or `architect`, resolved by the daemon from
|
||||
your connection — unforgeable, and the same resolution its authorization gate uses. A worker also
|
||||
carries its `sessionId`, `profile`, `worktree` and `branch`; an architect carries the slot name it
|
||||
was bound to. Don't infer what you can ask.
|
||||
|
||||
Only if that call is unavailable, fall back to these — each is one-way, so keep reading until one
|
||||
fires: the reply charter in your system prompt (*"You are an off-subscription worker in the
|
||||
claude-bridge fleet"*) ⇒ **spawned member**; bridge tools prefixed `mcp__bridge__*` ⇒ **spawned
|
||||
member** (the launcher fixes that mount name; a primary's mount is named by whoever wrote its
|
||||
`.mcp.json`, so it varies); `ANTHROPIC_BASE_URL` set ⇒ **spawned member** (Claude-model members run
|
||||
on a clean env, so its *absence* proves nothing). None of these separate a worker from an architect —
|
||||
only `bridge_whoami` does. **Still unsure ⇒ act as a worker**, the most restricted member role. The
|
||||
two mistakes are not symmetric: a primary acting as a worker is refused by the authorization gate —
|
||||
loud and self-correcting — while a member acting as the primary ends its turn with no `bridge_reply`,
|
||||
and the sender silently receives nothing. Fail toward the recoverable error.
|
||||
|
||||
### Invariants — both roles, no exceptions
|
||||
|
||||
1. **Never set, export, or forward `ANTHROPIC_BASE_URL`** (or `ANTHROPIC_AUTH_TOKEN`). The primary
|
||||
stays on subscription; only the bridge puts a member off it, at spawn. Mounting the bridge must
|
||||
never move a session across that boundary.
|
||||
2. **The bridge is the only channel.** Text you print in your terminal reaches nobody — the other
|
||||
side cannot see your screen. An answer that isn't in a `bridge_*` call is silently discarded.
|
||||
3. **Identity comes from the connection, never an argument.** Workers never pass a target; you
|
||||
cannot act as another session. Spawn/stop/drain are lead-only; **send is lead or architect**;
|
||||
reply/ask are only-as-itself — any peer may answer for its own pane, and for no other. A call
|
||||
outside your role is refused, not queued.
|
||||
4. **Delivery is status-gated: one message per turn.** Don't busy-poll a peer's terminal and don't
|
||||
re-send because a call looks slow — the bridge delivers when the peer is `idle`, `blocked` or
|
||||
`done`. A spawned member must **also** have mounted the bridge MCP: until it has, it is not
|
||||
deliverable, and a send waits on that gate for ~60s and then fails without ever reaching its pane.
|
||||
5. **Never drive the terminal multiplexer directly** (no `herdr` CLI, no socket). The bridge owns
|
||||
policy; the multiplexer owns PTYs. Going around the bridge bypasses every rule above.
|
||||
|
||||
### Primary (lead) — run this on every task, in order
|
||||
|
||||
**Delegate by default — that is the job.** With the bridge mounted you are an orchestrator on a
|
||||
metered subscription, and workers are cheap, parallel, and disposable. The default answer to "who
|
||||
does this?" is **a worker**, not you. Reach for `bridge_send` before you reach for `Edit`. The steps
|
||||
below are the procedure — run them in order, every task, not only the big ones.
|
||||
|
||||
0. **Know your role** — `bridge_whoami`, once per session, before anything else.
|
||||
1. **Split.** Write the unit list. Every unit carries: scope · the files or PR in question ·
|
||||
acceptance criteria · exactly what to report back. A unit with no acceptance criteria is not
|
||||
ready to delegate — refine it or keep it.
|
||||
2. **Gate each unit** on one question: **"can I write a brief good enough for a worker to
|
||||
succeed?"** — *not* "could I do this faster myself?" (usually you could; doing it yourself costs
|
||||
your context and your subscription, while a wasted worker turn costs a worker turn). Yes ⇒
|
||||
delegate. The keep-list is closed: the conversation with the user, decomposition and planning,
|
||||
the final judgment call, verification, merges, and anything that depends on context only you
|
||||
hold. Nothing else is yours by default.
|
||||
3. **Spawn every delegated unit first** — `bridge_spawn{profile, worktree:true, ticket}`, one per
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model and cost, not in
|
||||
tier, so the default is rarely what you want.
|
||||
4. **Then send them all** — `bridge_send{sessionId, content, wait:false}`. Line 1 of every brief is
|
||||
`Load the <name> skill.` naming the worker's playbook; those skills are opt-in and that line is
|
||||
what makes them reliable. Where the project ships no such skill, spell the procedure out in the
|
||||
brief instead. The brief is self-contained — the worker sees your message and the repo, nothing
|
||||
of your context, your plan, or your screen.
|
||||
5. **Collect** — `bridge_poll{ticket}` → `bridge_ack{ticket, msgId}`. Answer a worker's `bridge_ask`
|
||||
with `bridge_send{turnId, content}` — **not** `sessionId`. A worker gone quiet is diagnosed with
|
||||
`bridge_status`, never by reading its terminal.
|
||||
6. **Verify yourself.** Re-run the build and the checks. A worker mounts only the bridge MCP and
|
||||
cannot run your other tooling, and a piped command (`… | tail`) hides failures behind a zero
|
||||
exit — never promote a worker's "clean" to a fact.
|
||||
7. **Review — fan out.** Spawn reviewers against the diff, one per dimension or per file, with
|
||||
`wait:false`. Never the implementer of the scope it reviews, and brief them from the diff — not
|
||||
from the implementer's rationale, which carries its own blind spot. Dispatch each PR's reviewers
|
||||
as it lands; don't wait for the last implementer. Under ~50 changed lines, skip the fan-out and
|
||||
read it yourself.
|
||||
8. **Adjudicate, merge, tear down — yours alone.** Read the diff yourself: fully if it is small,
|
||||
targeted at the reported findings and the risky paths if it is large. Reviewer findings direct
|
||||
your attention; they never substitute for it. Then merge, then `bridge_stop{paneId}`.
|
||||
|
||||
**Steps 3 and 4 are separate on purpose** — spawning and sending in one loop is how parallel work
|
||||
silently becomes serial, and it is the most common way this layer is wasted. For the same reason,
|
||||
prefer `wait:false` + `bridge_poll` for anything non-trivial: a blocking `bridge_send` is capped by
|
||||
*your own* MCP client call timeout (~60s), well below the task's real runtime.
|
||||
|
||||
**Delegating does not delegate responsibility.** Workers open PRs; you are the gate. Never delegate
|
||||
the merge — and merging on a reviewer's word is delegating it by proxy.
|
||||
|
||||
| Intent | Tool |
|
||||
|---|---|
|
||||
| Confirm your own role | `bridge_whoami` |
|
||||
| See backends available | `bridge_profiles` |
|
||||
| Start a member | `bridge_spawn{role?, profile?, cwd?, worktree?, ticket?}` → `sessionId` + `paneId` |
|
||||
| See the fleet | `bridge_list` → `leads` (your peers) + `members` · one peer's state: `bridge_status{sessionId}` |
|
||||
| Delegate (blocking) | `bridge_send{sessionId, content}` |
|
||||
| Delegate (long task) | `bridge_send{sessionId, content, wait:false}` → ticket → `bridge_poll{ticket}` |
|
||||
| Answer a member's `bridge_ask` | `bridge_send{turnId, content}` — **not** `sessionId` |
|
||||
| Message a **peer lead** | `bridge_send{sessionId: <their terminal>, content}` — `bridge_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `bridge_reply{content}` — the one case a lead replies |
|
||||
| Collect a held reply | `bridge_poll{target}` · then `bridge_ack{target, msgId}` |
|
||||
| Tear down a member | `bridge_stop{paneId}` |
|
||||
|
||||
### Lead ↔ lead — coordinate, never delegate
|
||||
|
||||
`bridge_list` returns `leads` alongside `members`; your own row carries `self: true`. Every other row
|
||||
is a peer — an orchestrator with its own context, its own members, and its own judgment. An empty
|
||||
`members` array means no members are spawned; it says nothing about peers.
|
||||
|
||||
**A lead never assigns a task to another lead.** Work goes to members — only ever downward, never
|
||||
sideways. Sending a peer a brief with acceptance criteria is a category error: a brief is a member's
|
||||
artefact, and a peer is not yours to task. If a unit needs doing and it falls in your area, spawn a
|
||||
member and delegate it yourself; if it falls in the peer's area, say so and let the peer assign it.
|
||||
The traffic between leads is coordination and nothing else:
|
||||
|
||||
1. **Divide the map, not the work.** Agree who owns which area, then each of you assigns inside your
|
||||
own. Split by **context ownership** — whoever already holds the context owns that area — and say
|
||||
who takes what, in one message, before either of you starts. Two leads silently working the same
|
||||
unit is the failure mode here, and neither notices until the merge.
|
||||
2. **Share findings, hazards, and corrections.** What you have already discovered, what broke, what
|
||||
the next person will trip on. This is the traffic that actually pays for the channel: it costs one
|
||||
message and saves a peer a rediscovery.
|
||||
3. **Verify a peer exactly as you verify yourself.** Peer status buys nothing: check the claim
|
||||
against the code, and re-run the build. A peer's correction gets the same treatment — right or
|
||||
wrong on the evidence, not on who said it. Neither of you merges the other's work unreviewed.
|
||||
|
||||
Being messaged by a peer does not make you its worker: answer with `bridge_reply`, and push back on
|
||||
the substance if it is wrong. A peer that simply complies has thrown away the reason there are two of
|
||||
you.
|
||||
|
||||
### Member (worker or architect) — the turn contract
|
||||
|
||||
1. **Load the playbook skill the lead named** before doing anything else.
|
||||
2. **Do the assigned scope only.** Note anything you spot outside it in one line; don't go hunt it.
|
||||
3. **`bridge_ask{question}`** when a decision is genuinely the lead's (ambiguous requirement, two
|
||||
defensible fixes, "bug or intended?"). It blocks and you resume the *same* turn with the answer.
|
||||
Don't ask what you could decide yourself.
|
||||
4. **End the turn with exactly one `bridge_reply{content}`**, carrying your complete answer. This is
|
||||
the whole handoff. No `bridge_reply` ⇒ the sender gets nothing and the exchange stalls.
|
||||
Do **not** lean on the completion fallback to carry your answer for you: when you end a turn
|
||||
without replying, the bridge scrapes your pane, and it can return only the last 4000 characters.
|
||||
A clipped scrape is marked as partial, but the missing text is gone — your report reaches the
|
||||
lead with its end cut off.
|
||||
5. **Report honestly.** State only what you actually ran and its real output, including failures.
|
||||
You mount **only** the bridge MCP — the primary's other servers (IDE, forge, docs) are not yours,
|
||||
so never claim the result of a check you had no way to run.
|
||||
6. **Never merge.** Stage files explicitly — never `git add -A` — and leave alone anything the
|
||||
project marks as not-yours-to-commit.
|
||||
|
||||
### Where each rule lives (don't duplicate — extend the right layer)
|
||||
|
||||
| Layer | Scope | Reaches |
|
||||
|---|---|---|
|
||||
| the launcher's reply charter | the one rule that must survive with no repo: *end every turn with `bridge_reply`* | every spawned member, at launch, every peer kind — never a lead |
|
||||
| **this section** | protocol + orchestration policy | primary **and** every member that reads the repo — tracked in git, so worktrees inherit it |
|
||||
| role playbook skills | per-job procedure (commit/PR recipe, finding format) | a member told to load one |
|
||||
| the bridge's own docs | design detail, flows, error model | on demand |
|
||||
|
||||
A rule belongs in **exactly one** layer — the outermost one that must obey it. Peers that don't read
|
||||
`CLAUDE.md` (non-Claude adapters) get the charter only, so any rule *they* must obey belongs in the
|
||||
charter, not here.
|
||||
|
||||
## Project addendum — claude-bridge (not part of the canonical block)
|
||||
|
||||
- **This repo is the bridge.** The daemon is `bridged`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
and the code behind the rules above is `mcp/BridgeMcp` (tools), `auth/Authz` (the role table),
|
||||
`mcp/ConnectionIdentity` (connection→role), and `worker/*Launcher` (`REPLY_CHARTER`).
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR) and
|
||||
`reviewer` (scoped review → one structured finding). Name one in every delegation.
|
||||
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace).
|
||||
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
||||
(a submodule with its own remote).
|
||||
- **Flows and the error model** — rendezvous, `bridge_ask`, detached delivery, turn-done fallback —
|
||||
are diagrammed in `docs/MCP-Contract.md` §6, kept out of this file because it loads into every
|
||||
session's context.
|
||||
|
||||
### The prompt is part of the product — update it with the code (mandatory)
|
||||
|
||||
This repo *is* the bridge, so the canonical block above is not documentation about someone else's
|
||||
system: it is the instruction surface this codebase ships. **Every change here must end by asking
|
||||
whether the block still tells the truth.** A code change that silently invalidates it is an
|
||||
incomplete change — the agents reading it have no other source.
|
||||
|
||||
Before you call any work done, check the row that matches what you touched:
|
||||
|
||||
| You changed… | Re-read and update… |
|
||||
|---|---|
|
||||
| a `bridge_*` tool — added, removed, renamed, or its params/semantics | the primary's intent→tool table; any rule that names that tool |
|
||||
| `Authz` / the role table | invariant 3, and the primary-only vs worker-only claims |
|
||||
| `ConnectionIdentity` / how a caller is resolved | the `bridge_whoami` paragraph and the fallback ladder |
|
||||
| `REPLY_CHARTER`, or a launcher's mount/flags | the fallback ladder (`mcp__bridge__*`), and the layering table's top row |
|
||||
| the injector / status gating | invariant 4 |
|
||||
| worktree provisioning or the parity overlay | the "both roles read this file" premise — it rests on the worker's worktree being a checkout of this repo |
|
||||
| `.claude/skills/**` | the addendum's skill list, and the "name the playbook" rule |
|
||||
| a new peer kind (non-Claude adapter) | what that peer can read — anything it must obey belongs in its charter, not in the block |
|
||||
| **anything an operator can use, configure, or observe** — an MCP tool, a `bridged.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
||||
|
||||
That last row is not bookkeeping. Chapters 1–10 answer *how is this built* and *why this way*;
|
||||
none of them has a home for *what can it do and how do I turn it on*, so for twenty tickets a
|
||||
shipped capability landed nowhere and the Roadmap went on claiming the stage was finished. The
|
||||
*why* line is the one that matters — without it a decision gets re-litigated from scratch a month
|
||||
later. Internal contract changes go to `wiki/9-Implementation.md` instead; test and coverage work
|
||||
is a Roadmap line. A change that touches none of the three earns no entry, and that is a normal
|
||||
outcome rather than an omission.
|
||||
|
||||
Then **propagate**: the block in this file and the template in the wiki
|
||||
([Use Cases](https://git.ltms.dev/lms/claude-bridge/wiki/7-Use-Cases) → *The portable `CLAUDE.md`
|
||||
block*) must stay byte-identical, and other projects carrying the block need the same edit. Verify
|
||||
rather than trust:
|
||||
|
||||
```bash
|
||||
python3 - <<'PY'
|
||||
import pathlib
|
||||
c = pathlib.Path("CLAUDE.md").read_text()
|
||||
w = pathlib.Path("wiki/7-Use-Cases.md").read_text()
|
||||
S, E = "## Bridge communication (enforced", "## Project addendum — claude-bridge"
|
||||
block = c[c.index(S):c.index(E)].rstrip() + "\n"
|
||||
i = w.index("```markdown\n") + len("```markdown\n")
|
||||
print("in sync:", w[i:w.index("\n```\n", i) + 1] == block)
|
||||
PY
|
||||
```
|
||||
|
||||
## IDE MCP tools & validation workflow (enforced)
|
||||
|
||||
> **Primary only.** Workers have no IDE MCP mount — if you are a worker, skip this section and
|
||||
> report the build/test output you actually ran (see §Bridge communication → Worker).
|
||||
|
||||
Two IDE MCP servers are connected: **intellij-index** (semantic code intelligence) and
|
||||
**jetbrains** (file problems, reformat, debugger). IntelliJ has multiple projects open; our
|
||||
module is **`bridged`**. Always pass these to IDE MCP tools:
|
||||
@@ -20,6 +258,15 @@ module is **`bridged`**. Always pass these to IDE MCP tools:
|
||||
test run). A per-file-clean file can still break the build or another module. This is the
|
||||
whole-project gate before declaring work done or committing.
|
||||
|
||||
**Whenever dependencies change (or a `pom.xml` edit), validate CVEs with
|
||||
`jetbrains get_file_problems{filePath: "bridged/pom.xml"}`** — its Mend.io check reflects the
|
||||
dependencies on disk. (Note: `ide_diagnostics` / intellij-index does NOT re-resolve dependencies
|
||||
after a pom edit without a full Maven reimport, so it reports stale CVE results — don't trust it
|
||||
for this.) Treat a CVE warning like any other: bump to a patched version and confirm
|
||||
`mvn clean install` still passes. If the latest available version is still flagged (EOL line,
|
||||
"insufficient information", or config-file-only advisories), document it as accepted in the pom
|
||||
rather than chasing a fix that doesn't exist.
|
||||
|
||||
### Use IDE MCP tools for navigation, refactoring, and diagnostics only
|
||||
|
||||
- **Navigate (prefer over Grep/Read for symbols):** `ide_find_definition`, `ide_find_class`,
|
||||
|
||||
@@ -50,8 +50,9 @@ flowchart LR
|
||||
plain daemon (no Anthropic quota), so it may poll/subscribe freely.
|
||||
- **One gateway (unified MCP setup):** `bridged` is the **sole communication path** for every
|
||||
Claude session. Primary and workers each mount it as an MCP server (one `claude mcp add`
|
||||
line, same on both) and talk over MCP tools — `bridge_send` / `bridge_reply` / `bridge_ask` /
|
||||
`bridge_status`. **No Claude session ever addresses a broker, a peer, or the network
|
||||
line, same on both) and talk over MCP tools — `bridge_send` / `bridge_reply` /
|
||||
`bridge_status` (with `bridge_ask` planned for the blocked-worker path). **No Claude session
|
||||
ever addresses a broker, a peer, or the network
|
||||
directly**; any queue is `bridged`-internal. MCP tool I/O never sets `ANTHROPIC_BASE_URL`, so
|
||||
mounting the bridge is subscription-safe by construction.
|
||||
- **How the primary consumes a reply:** a single **blocking MCP call** (`bridge_send`);
|
||||
@@ -85,6 +86,38 @@ Gitea wiki.
|
||||
|
||||
## Status
|
||||
|
||||
🟢 Design — herdr-centric **`bridged`** message server selected as the primary approach
|
||||
(2026-07-11), superseding the AgentAPI plan (2026-07-08). AgentAPI retained as fallback
|
||||
injector.
|
||||
🟢 **Implemented & dogfooded** — the herdr-centric **`bridged`** message server is built and in
|
||||
real use: an Opus primary delegates tasks to off-subscription workers that reply through the
|
||||
bridge (code reviews delegated this way have produced committed bug fixes). Selected as the
|
||||
primary approach 2026-07-11, superseding the AgentAPI plan (2026-07-08); AgentAPI retained as a
|
||||
fallback injector.
|
||||
|
||||
**Shipped** (Java 25 · Maven · 266 unit/acceptance tests green; the live-herdr and broker contract
|
||||
tests run separately via `mvn test -Pcontract`):
|
||||
|
||||
- **Core gateway** — herdr socket client (contract-tested vs live 0.7.0); guard-checked worker
|
||||
spawn with `ANTHROPIC_BASE_URL` injected only into the worker's env; status-gated injector;
|
||||
blocking `bridge_send` with reply rendezvous; MCP server as a thin adapter over the REST core.
|
||||
- **MCP tools** — `bridge_send` / `bridge_reply` / `bridge_status` (messaging) and `bridge_spawn`
|
||||
/ `bridge_list` / `bridge_stop` / `bridge_profiles` / `bridge_poll` (fleet). Caller identity is
|
||||
connection-based (loopback peer PID → herdr pane), so the same mount serves primary and workers.
|
||||
- **Delivery reliability** — completion fallback (a confirmed `working→idle` turn resolves a
|
||||
send); async fire-and-poll (beats the caller's MCP call timeout for long tasks); and failure
|
||||
detection for wedged (`unknown`), vanished, and never-ready workers so a send never hangs.
|
||||
- **Fleet** — multiple worker profiles, each with an independent base_url guard check; workers
|
||||
inherit the primary's working directory (never `$HOME`); a readiness gate holds delivery until
|
||||
a worker's Claude has connected the bridge MCP (no paste lost into its boot window).
|
||||
- **Blocked-worker path** — `bridge_ask` reverse rendezvous: a worker pauses its delegated turn to
|
||||
ask the primary and resumes the *same* turn with the answer (CB-205).
|
||||
- **Session lifecycle** — session manager with spawn/reuse/recycle, `idle_ttl` reaper, `context_cap`,
|
||||
and graceful drain on shutdown (CB-301/CB-303); per-worker git worktrees on their own branch with
|
||||
a config-parity overlay, so parallel implementers never stomp each other (CB-301-ext).
|
||||
- **Reliable worker→primary delivery** — a durable `ReplyInbox` (in-memory by default, AMQP/LavinMQ
|
||||
for cross-restart durability) holds a reply that arrives with no open send, and an active
|
||||
status-gated push loop nudges the primary to drain it (CB-307).
|
||||
- **Pluggable peers** — a `PeerLauncher` SPI with two in-tree adapters, `claude-code` and `opencode`,
|
||||
routed by a `kind:` discriminator (CB-401/CB-402).
|
||||
|
||||
**Next** (see the [roadmap](wiki/8-Roadmap.md)) — Stage 5 hardening (auth/TLS, `/metrics`, CI,
|
||||
service supervision, per-session authz + audit), then cross-host: CB-308 multi-host federation and
|
||||
CB-500 multi-tier coordination.
|
||||
|
||||
@@ -5,6 +5,9 @@ dependency-reduced-pom.xml
|
||||
# Local runtime config (copy from bridged.example.yaml)
|
||||
bridged.yaml
|
||||
|
||||
# CB-505 audit trail + daemon stdout/stderr — runtime records, never source
|
||||
logs/
|
||||
|
||||
# Editor / OS
|
||||
*.iml
|
||||
.idea/
|
||||
|
||||
+414
-16
@@ -3,31 +3,429 @@
|
||||
# bridged is the sole gateway between primary/worker Claude sessions and herdr.
|
||||
# It is NOT a Claude process and must never carry ANTHROPIC_BASE_URL.
|
||||
|
||||
# REST + MCP listen address. Keep it on loopback — bridged is same-host in Stage-1.
|
||||
# REST + MCP listen address. Keep it on loopback unless you also switch auth.mode to `token`
|
||||
# below — bridged REFUSES TO START on a non-loopback bind under loopback-trust (see auth).
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8080
|
||||
port: 8765
|
||||
|
||||
# API authentication (CB-501). Governs how a caller that is NOT an on-host worker pane proves it
|
||||
# is the primary. Worker identity never depends on this: a loopback peer PID that maps to a herdr
|
||||
# pane is unforgeable and is always honoured, so turning auth on cannot lock the fleet out.
|
||||
#
|
||||
# mode: loopback-trust → DEFAULT, and the historical behaviour: any loopback caller that is not
|
||||
# a worker is the primary, no credential needed. Sound ONLY because the
|
||||
# OS refuses remote connections to a loopback socket.
|
||||
# mode: token → such a caller must send `Authorization: Bearer <token>`; without it it
|
||||
# is anonymous and authorized for nothing. REQUIRED for a non-loopback
|
||||
# bind — the daemon fails fast otherwise, because "unauthenticated ⇒
|
||||
# primary" on a reachable port would hand spawn/stop/send to anyone.
|
||||
# tokenEnv → host env var holding the token (never the literal value). Default
|
||||
# BRIDGED_API_TOKEN. Read only in token mode; empty ⇒ startup fails.
|
||||
#
|
||||
# TLS is deliberately NOT terminated in the daemon (CB-501 D3): run a reverse proxy in front and
|
||||
# let it own certificate lifecycle, e.g.
|
||||
# location / { proxy_pass http://127.0.0.1:8765; proxy_set_header Authorization $http_authorization; }
|
||||
# The broker link gets TLS from its own URI (amqps://…) — see `broker` below.
|
||||
# auth:
|
||||
# mode: token
|
||||
# tokenEnv: BRIDGED_API_TOKEN
|
||||
|
||||
# Optional pinned primary terminal (CB-307). Names the herdr pane the PRIMARY itself runs in:
|
||||
# a caller whose connection maps to this pane resolves as the primary (no credential needed —
|
||||
# the pane mapping is as unforgeable as a worker's), and reply nudges are pushed to it.
|
||||
# REQUIRED when the primary runs inside a herdr pane — without it the pane match reads the
|
||||
# primary as a worker and refuses spawn/send/stop. Get the id from bridge_whoami; re-pin if
|
||||
# the primary moves panes.
|
||||
# primary:
|
||||
# terminal: term_0123456789abcd
|
||||
# pushReminders: 5 # max nudges before giving up (default 5)
|
||||
# pushBackoffMs: 15000 # delay between nudges (default 15000)
|
||||
|
||||
# CB-530: MORE THAN ONE LEAD. `primary:` above is singular by construction — every other pane
|
||||
# resolves as a worker — which is right for one lead driving a fleet and wrong the moment two leads
|
||||
# (say a Claude lead and an opencode lead) work as peers: the second is silently demoted and refused
|
||||
# every orchestration call. List each lead's pane here and all of them resolve as leads.
|
||||
#
|
||||
# tab → the ONLY field identity depends on (CB-579); the exact label of the tab hosting the lead.
|
||||
# Label the tab yourself, or let bridged label one it launches — see `fleet.leaders:` below.
|
||||
# kind/model → descriptive; they document what runs in the pane and are echoed by bridge_whoami
|
||||
#
|
||||
# A lead's tab must already carry its label (or be launched by bridged, which labels it) — there is
|
||||
# no terminal id to paste in and nothing to re-pin when the session restarts: the tab survives, so
|
||||
# the same label resolves the same lead again on the next scan.
|
||||
# `bridge_whoami` reports `{"role":"primary","leader":"<name>"}`; role stays "primary" because a lead
|
||||
# IS a primary for authorization, so nothing that keys on the role breaks.
|
||||
#
|
||||
# KEEP `primary:` when adding leads: it still addresses the CB-307 push loop, which needs a single
|
||||
# destination for its nudges, and is a separate mechanism from lead identity — see `fleet.leaders:`.
|
||||
#
|
||||
# Leads are configured under `fleet.leaders:` — see THE FLEET further down.
|
||||
#
|
||||
# Two things stop the tab-name convention from becoming a way to claim leadership: the configured
|
||||
# member spaces are excluded from the scan, so nothing bridged places can land in a matching tab;
|
||||
# and startup REFUSES a `tabPrefix` that the fleet tabLabel template, or any per-profile `tabLabel`
|
||||
# override, also matches — so the two namespaces cannot overlap by accident. The label is a NAME,
|
||||
# never a capability: what a pane may do is decided by the role the daemon resolves for it.
|
||||
|
||||
# CB-551: IDLE-LEAD HEARTBEAT — nudge the single lead back to work when it has been continuously
|
||||
# idle (no open bridge_send driving it) past the quiet period. The fleet is one lead + architects +
|
||||
# workers, so a lead that stalls is a single point of failure; the ReplyPushLoop only nudges when a
|
||||
# reply lands, and this timer catches the gap where nothing lands and the lead just sits idle.
|
||||
#
|
||||
# Opt-in on purpose — it SPENDS the operator's subscription on its own initiative (each nudge starts
|
||||
# a lead turn nobody asked for), so upgrading the daemon must never switch it on for you. Absent
|
||||
# block = feature off, exactly as before.
|
||||
#
|
||||
# Three knobs, each with a default that errs on the side of not burning context:
|
||||
# idleAfterSeconds: 300 # how long the lead must stay idle before the FIRST nudge (default 300 —
|
||||
# # absorbs normal post-turn pauses; re-prompting every pause burns context)
|
||||
# backoffMs: 60000 # re-check cadence / spacing between nudges past the quiet period (default 60000)
|
||||
# quietNudgeCap: 3 # cap on consecutive nudges that find NOTHING pending, then it stops
|
||||
# # until real state appears (default 3 — never nag an empty fleet forever)
|
||||
# leadHeartbeat:
|
||||
# idleAfterSeconds: 300
|
||||
# backoffMs: 60000
|
||||
# quietNudgeCap: 3
|
||||
|
||||
# Fleet health detection is dormant unless enabled. It reads one whole-fleet agent list per tick.
|
||||
# It can run without a webhook; bridge_list then reports healthCoverage: detection-only.
|
||||
# health:
|
||||
# enabled: true
|
||||
# intervalSeconds: 30 # minimum 15
|
||||
# workingSuspectAfterSeconds: 600 # minimum 300
|
||||
# paneProbeIntervalSeconds: 60 # minimum 60
|
||||
# notifications:
|
||||
# mode: disabled # disabled (default) or webhook
|
||||
|
||||
# herdr Unix socket. Omit to use the client default
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
|
||||
# How a worker session is spawned. Stage-1 uses the existing ccs `ltms-local`
|
||||
# profile, whose .claude.json routes to the gx00 vLLM below.
|
||||
worker:
|
||||
profile: ltms-local
|
||||
baseUrl: http://gx00.gw:8000 # the gx00 vLLM (models: coder / deepseek-v4-flash)
|
||||
model: coder
|
||||
# Placement: each worker lands in its OWN tab inside a dedicated worker space, so it
|
||||
# never splits or clutters your real work spaces. Use `pane` for the legacy behaviour
|
||||
# (split the currently-focused tab).
|
||||
placement: tab # tab | pane
|
||||
workspace: bridged-workers # the dedicated worker space (found-or-created, shared)
|
||||
tabLabel: "worker: {profile} #{n}" # {profile}/{model}/{n} substituted; {n} keeps sibling tabs distinct
|
||||
# How member sessions are spawned. Define one or more named profiles (backends) under
|
||||
# `profiles`; each key is the profile name (also the ccs profile). A profile says only WHICH
|
||||
# BACKEND — model, CLI adapter, credentials, cost. It says nothing about what a member spawned on
|
||||
# it is for; that is the member's role, and roles live under `fleet:` below. Which profile an
|
||||
# unqualified spawn lands on comes from that role's pool, not from a global default.
|
||||
#
|
||||
# Shared knobs (placement/workspace) can be repeated per profile; they usually match.
|
||||
# placement: tab → each worker lands in its OWN tab in a dedicated worker space (default).
|
||||
# Use `pane` for the legacy behaviour (split the focused tab).
|
||||
# mcpUrl → bridged mounts the bridge MCP (--mcp-config, inline) + reply charter
|
||||
# (--append-system-prompt) as launch flags; nothing is written to the profile.
|
||||
# tokenEnv → host env var holding the worker's auth token (value never stored in config);
|
||||
# omit for a backend that needs no token (e.g. a local ollama).
|
||||
# cwd → pin this profile's working directory (CB-112). Omit to inherit the primary's
|
||||
# cwd on an MCP spawn, else the daemon's cwd — never $HOME. See
|
||||
# docs/Worker-Startup-and-Trust.md.
|
||||
# configDir → CLAUDE_CONFIG_DIR for the worker, so it inherits that profile's
|
||||
# skills/MCP/hooks. Omit to leave the worker on the host default.
|
||||
# parityOverlay → repo-relative paths copied primary→worktree so a worker in a provisioned
|
||||
# worktree sees the same local config (CB-301-ext). Omit for the default set:
|
||||
# [.claude/settings.local.json, .env, .envrc].
|
||||
#
|
||||
# Do NOT add .mcp.json (CB-525). A worker's tools are whatever its launcher
|
||||
# mounts — the bridge, and nothing else. Replicating the primary's MCP config
|
||||
# handed a worker the primary's IDE servers, which are bound to the primary's
|
||||
# checkout, so its navigation returned paths OUTSIDE its own worktree: one
|
||||
# worker made all 59 of its edits in the primary tree while compiling its
|
||||
# worktree, and every build it ran was of code that did not contain them.
|
||||
# bridged neutralizes a provisioned worktree's .mcp.json for this reason;
|
||||
# listing it here would copy the primary's back over that.
|
||||
# gitTokenEnv → host env var holding the git-forge API token. When set, its value is injected
|
||||
# as GITEA_TOKEN so the worker can open its OWN PR at checkpoint (CB-302).
|
||||
# Opt-in by design — omit and the worker gets no PR-create grant (push over
|
||||
# SSH is unaffected). The token value itself is never stored in this file.
|
||||
# gitHostEnv → host env var holding the forge host (default GITEA_HOST). Injected as
|
||||
# GITEA_HOST *only* alongside a resolved gitTokenEnv.
|
||||
# exhaustedPattern → regex matched against a completion-fallback scrape (CB-578 stage A) to
|
||||
# classify a turn that ended with no bridge_reply as the backend having
|
||||
# refused on a subscription usage limit, rather than a real answer. Opt-in —
|
||||
# omit and this profile's completion fallback behaves exactly as before.
|
||||
# Every backend words its refusal differently, so this is config, never a
|
||||
# vendor string baked into bridged itself.
|
||||
# env → extra environment for this profile's workers, as a literal key/value map
|
||||
# (CB-511). Use it to give workers a toolchain.
|
||||
#
|
||||
# A worker's environment does NOT come from your shell. bridged hands herdr an
|
||||
# explicit env map and herdr merges it into ITS OWN process env — so before
|
||||
# CB-511 a worker inherited whatever PATH the herdr server happened to be
|
||||
# started with, which on a long-lived herdr can predate your toolchain entirely
|
||||
# and leave workers unable to run `mvn` or `java` at all.
|
||||
# bridged now propagates ITS OWN PATH to every worker by default; set `env:`
|
||||
# only to override that or add more (JAVA_HOME, …). Since the default is the
|
||||
# daemon's PATH, make sure the daemon is started with a good one — see the PATH
|
||||
# lines in deploy/dev.ltms.bridged.plist and deploy/bridged.service.
|
||||
#
|
||||
# Adapter-owned variables always win over `env:`: ANTHROPIC_BASE_URL and the
|
||||
# rest of the ANTHROPIC_*/CLAUDE_* wiring are applied after it, so an `env:`
|
||||
# entry cannot repoint a worker past the SubscriptionGuard — which is checked
|
||||
# against `baseUrl` alone.
|
||||
# Put `defaultMode: "auto"` in each ccs profile so the worker runs autonomously.
|
||||
profiles:
|
||||
gx10: # ccs profile name (NOT a hostname)
|
||||
kind: claude-code # which adapter spawns this profile (default; may omit)
|
||||
baseUrl: http://gx01.gw:8000 # the vLLM host this profile targets (gx00.gw / gx01.gw)
|
||||
model: coder
|
||||
placement: tab
|
||||
workspace: bridged-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
tokenEnv: BRIDGED_WORKER_TOKEN
|
||||
argv: ["ccs", "gx10"]
|
||||
weight: 0.5 # relative selection weight for placement: weighted
|
||||
maxLoad: 2 # max live workers on this profile (omit for unlimited)
|
||||
# gitTokenEnv: GITEA_TOKEN # opt-in: let this profile's workers open their own PR (CB-302)
|
||||
# gitHostEnv: GITEA_HOST # defaults to GITEA_HOST; injected only with gitTokenEnv
|
||||
# exhaustedPattern: "usage limit has been reached" # opt-in: classify a usage-limit refusal (CB-578)
|
||||
# configDir: /Users/me/.ccs/instances/gx10 # CLAUDE_CONFIG_DIR — inherit that profile's skills/MCP
|
||||
# cwd: /Users/me/src/myrepo # pin the working dir; omit to inherit the primary's
|
||||
# parityOverlay: [".claude/settings.local.json", ".env", ".envrc"] # never add .mcp.json — see above
|
||||
gx11: # a second backend, so `placement: weighted` has a choice
|
||||
baseUrl: http://gx01.gw:8000 # self-hosted; ccs handles the model + token
|
||||
placement: tab
|
||||
workspace: bridged-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
argv: ["ccs", "gx11"]
|
||||
weight: 0.5
|
||||
maxLoad: 2
|
||||
# Pin an auto-compact window BELOW the served model's context ceiling. The global
|
||||
# ~/.claude/settings.json value is shared by every ccs instance and the primary, so the
|
||||
# per-profile override belongs here. Equal to the ceiling means auto-compact never fires
|
||||
# before the server rejects the prompt, which kills a worker mid-turn (CB-523).
|
||||
env:
|
||||
CLAUDE_CODE_AUTO_COMPACT_WINDOW: "280000"
|
||||
# CB-402: a second coding-agent kind, proving the PeerLauncher SPI is provider-neutral.
|
||||
# opencode is provider-agnostic and uses NONE of Claude's private seams: no ANTHROPIC_BASE_URL /
|
||||
# SubscriptionGuard (so it needs no `guard` host entry), no --mcp-config / --append-system-prompt.
|
||||
# The bridge MCP + reply charter mount via a generated OPENCODE_CONFIG file, and the model is a
|
||||
# `provider/model` selector. Placement, tabs, cwd, and the readiness gate are shared with Claude.
|
||||
#
|
||||
# Dogfood-verified 2026-07-29 against opencode 1.18.5 (spawn → readiness gate → bridge_send →
|
||||
# structured bridge_reply → teardown). The `opencode/*-free` models run on opencode's own gateway
|
||||
# and need NO credentials — check `opencode models` for the current free list, since the names
|
||||
# change. That also makes the worker off-subscription by construction.
|
||||
# opencode-free:
|
||||
# kind: opencode
|
||||
# model: opencode/north-mini-code-free # `provider/model` selector, injected as `-m`
|
||||
# placement: tab
|
||||
# workspace: bridged-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
#
|
||||
# CB-508: point an opencode profile at your OWN OpenAI-compatible endpoint (local vLLM, llama.cpp,
|
||||
# LM Studio, TGI…) instead of opencode's gateway. Setting `baseUrl` on a `kind: opencode` profile
|
||||
# makes the bridge emit a custom `provider` block into the generated opencode.json — opencode has
|
||||
# no ANTHROPIC_BASE_URL seam, so this is how the endpoint is pinned.
|
||||
# baseUrl → a bare host:port gets `/v1` appended (where these servers mount the API); a URL that
|
||||
# already has a path is used verbatim, so a custom mount point still works.
|
||||
# model → MUST be "<provider>/<model>". The provider half names the generated block; the model
|
||||
# half must match an id the server reports at /v1/models. One field drives both the
|
||||
# declaration and the `-m` flag, so they cannot drift apart. A bare model name with a
|
||||
# baseUrl set is rejected at spawn rather than silently using the default gateway.
|
||||
# tokenEnv → optional; its value becomes the provider apiKey. Most local servers ignore the key,
|
||||
# so a placeholder is used when unset (the AI SDK still requires a non-empty one).
|
||||
# NOTE: no `guard` entry is needed even with a baseUrl set. The SubscriptionGuard exists to stop a
|
||||
# worker borrowing the primary's Anthropic subscription, and an opencode process has no Anthropic
|
||||
# credential path at all.
|
||||
# opencode-local:
|
||||
# kind: opencode
|
||||
# baseUrl: http://127.0.0.1:8000
|
||||
# model: local-vllm/deepseek-v4-flash
|
||||
# placement: tab
|
||||
# workspace: bridged-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
# How an unqualified spawn chooses a profile: fixed (default, reproduces pre-CB-518 behaviour),
|
||||
# round-robin, or weighted. Omitting this key is a strict no-op for existing configs.
|
||||
placement: weighted
|
||||
|
||||
# Re-read this file without restarting the daemon (CB-559). Off unless you add this block, so an
|
||||
# upgraded bridged keeps the old behaviour: the file is read once at boot and never again.
|
||||
# enabled → turn the watch on. bridged checks the file's modified time on a timer and
|
||||
# reloads when it moves.
|
||||
# intervalSeconds → how often to check (default 10). One `stat` per tick, so this is cheap.
|
||||
#
|
||||
# Not every key can move under a running daemon, and the difference is about what already exists
|
||||
# when the reload happens — not about how important the key is:
|
||||
# HOT → takes effect on the next spawn: the whole `fleet:` block (every role pool,
|
||||
# `charters`, and `tabLabel`), `placement:`, and an existing profile's weight / maxLoad. Those are
|
||||
# hot because the placement policy reads them through a supplier — being config is
|
||||
# not by itself enough to make a key hot.
|
||||
# DEFERRED → accepted into the new config, but the wiring built at startup keeps the old value
|
||||
# until you restart: `lifecycle:`, `leadHeartbeat:`, `guard:`, `worktreeRoot:`,
|
||||
# `spawnReadyTimeoutMs` / `spawnReadyPollMs`, ADDING or REMOVING a profile (a new
|
||||
# backend needs its own launcher, and launchers are built once), AND an existing
|
||||
# profile's launch settings — model, baseUrl, argv, env, configDir, mcpUrl, tabLabel.
|
||||
# The launcher takes a copy of `profiles:` at startup and resolves every spawn out of
|
||||
# that copy, so those never reach a launch until you restart. The reload logs them by
|
||||
# name rather than pretending they applied.
|
||||
# COLD → cannot change at all: `bind:`, `herdrSocket:`, `broker:` and `auth:`. The socket is
|
||||
# bound, the broker connection is open, and the auth mode decides who may reach the
|
||||
# port that is already listening.
|
||||
#
|
||||
# A changed COLD key refuses the WHOLE reload — not the hot half applied and the cold half warned
|
||||
# about. A half-applied reload would leave the daemon matching no file on disk, which is the worst
|
||||
# thing a reload can do to an operator debugging one. A file that fails to parse or fails a startup
|
||||
# validator is refused the same way, and the running config stays live.
|
||||
# configReload:
|
||||
# enabled: true
|
||||
# intervalSeconds: 10
|
||||
|
||||
# THE FLEET (CB-557) — who the daemon may run, and under which role. This one block replaced four
|
||||
# older keys: `leaders:`, `members:`, `leadScan:` and `defaultProfile:`.
|
||||
#
|
||||
# A member is anything a lead spawns, and every member has two INDEPENDENT attributes:
|
||||
# role — which contract: architect, dev or reviewer. It picks the launch charter, the role
|
||||
# file, the playbook skill and the authz row.
|
||||
# profile — which backend: one of the `profiles:` keys above (model, CLI adapter, cost).
|
||||
# They vary on their own. A reviewer may run on the same profile as the dev whose diff it reads,
|
||||
# which is why the two cannot be one field.
|
||||
#
|
||||
# The ROLE IS THE CONTAINING KEY, not a `role:` field. That is not only tidier: a misspelled role
|
||||
# used to parse into a member with no contract at all, while a misspelled pool name here simply
|
||||
# declares nothing.
|
||||
#
|
||||
# Each pool lists the profiles that role MAY run on — these are pools, not identities. That is also
|
||||
# what replaced `defaultProfile:`: an unqualified spawn names a role, and that role's pool supplies
|
||||
# the candidates, in definition order. A dev and a reviewer staying anonymous is exactly compatible
|
||||
# with being listed here; the entry key just names the entry.
|
||||
fleet:
|
||||
# Optional launch-charter text, keyed only by the singular role wire names: architect, dev,
|
||||
# reviewer. Changes are HOT and reach the next spawn without a daemon restart. Do not put secrets
|
||||
# here: a later launch step writes this text to a world-readable temp file, and ${ENV} interpolation
|
||||
# is deliberately not supported.
|
||||
charters:
|
||||
architect: |-
|
||||
You are an architect in this fleet. You refine work before anyone builds it:
|
||||
scope, acceptance criteria, risks, and a unit split. You read the repo and
|
||||
write analysis. You never commit production code and never open a PR.
|
||||
A design task is worked by two architects. Design alone first, then exchange
|
||||
and say plainly where you disagree. Do not concede just to agree.
|
||||
dev: |-
|
||||
You implement the one unit you were given, and nothing else. You test it,
|
||||
commit it, and open your own pull request. You never merge.
|
||||
reviewer: |-
|
||||
You review the diff you were given. You report bugs, risks and missing tests.
|
||||
You do not change code.
|
||||
|
||||
# Optional. Template for a member tab's label; {role}, {profile}, {model} and {n} are substituted.
|
||||
# {n} counts per role+profile, so `dev: sonnet #2` really is the second sonnet dev. Because {role}
|
||||
# comes from a closed enum, a generated label can never begin with a lead's tabPrefix.
|
||||
# tabLabel: "{role}: {profile} #{n}"
|
||||
|
||||
# Panes that orchestrate rather than are orchestrated. A lead may now be CREATED as well as
|
||||
# recognised: give it a `profile:` and the daemon launches the shortfall when fewer than
|
||||
# `instances` are live. Omit `profile:` and it is recognise-only, as before.
|
||||
#
|
||||
# `tab:` (CB-579) is REQUIRED and is the only field identity depends on — the exact label of the
|
||||
# tab hosting the lead, matched case-insensitively. Label the tab yourself and put that same
|
||||
# string here, and the pane is recognised on the next rescan. Reopen the tab later, or the session
|
||||
# inside it restarts — the terminal id changes; the tab, and its label, do not, so no config edit
|
||||
# follows a restart.
|
||||
#
|
||||
# A lead the daemon launches is labelled BY the daemon with this same `tab:` value, so it is found
|
||||
# by the same scan. A lead counts as live only when herdr also reports a running agent in that
|
||||
# tab — a label left behind by a session that died does not block the relaunch, and a tab that is
|
||||
# gone entirely drops out of the next scan rather than being remembered forever.
|
||||
#
|
||||
# An auto-launched lead is NOT a member: it gets no worker reply charter, is never registered with
|
||||
# the session lifecycle (the idle reaper would kill your orchestrator), and stays on the
|
||||
# subscription — ANTHROPIC_BASE_URL/AUTH_TOKEN are stripped from its env whatever the profile says.
|
||||
# leaders:
|
||||
# opus-5.0:
|
||||
# profile: opus # omit to never create this lead, only recognise it
|
||||
# instances: 1 # desired live count; only the shortfall is launched. 0 = off
|
||||
# tab: "lead: opus-5.0" # REQUIRED — the exact tab label this lead lives in
|
||||
# tabPrefix: "lead:" # only used to guard against a worker tabLabel colliding with
|
||||
# # this convention at startup; plays no part in matching a lead
|
||||
# scanIntervalSeconds: 10 # rescan cadence, and the worst case before a new tab is seen
|
||||
# workspace: leads # where a launched lead's tab is created (default "leads").
|
||||
# # MUST NOT be a member workspace — those are excluded from the
|
||||
# # scan, so a lead placed in one is never found again.
|
||||
# cwd: /path/to/repo # the launched lead's working directory (default: bridged's own)
|
||||
# kind: claude # descriptive; reported by bridge_whoami
|
||||
# gpt-sol-5.6:
|
||||
# tab: "lead: gpt-sol-5.6"
|
||||
# kind: opencode
|
||||
# model: openai/gpt-5.6-terra
|
||||
|
||||
# architects:
|
||||
# architect-1:
|
||||
# profile: opus # a strong model, on the operator's subscription
|
||||
# architect-2:
|
||||
# profile: sol # a different vendor on purpose — two architects that share a
|
||||
# # model share its blind spots
|
||||
developers:
|
||||
gx10:
|
||||
profile: gx10
|
||||
# reviewers:
|
||||
# gx10:
|
||||
# profile: gx10 # the same backend may serve two roles; that is the point
|
||||
|
||||
# Subscription boundary. A worker's base_url host MUST be one of these; the primary
|
||||
# must carry none. Grounded in ltms-local's real endpoints.
|
||||
# must carry none. Every profile above must have its host listed here.
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
- ollama.ltms.dev
|
||||
- gx01.gw
|
||||
|
||||
# Spawn-readiness gate (CB-306). The launcher blocks until the worker's herdr status is
|
||||
# injectable (IDLE/BLOCKED/DONE) or the timeout elapses. 0 disables the gate.
|
||||
# NOTE: keys are camelCase — config is bound by plain Jackson with no naming strategy and
|
||||
# unknown keys are ignored, so a snake_case key would be silently dropped (default kept).
|
||||
# spawnReadyTimeoutMs: 20000
|
||||
# spawnReadyPollMs: 300
|
||||
|
||||
# Worktree provisioning root (CB-301-ext). Where per-worker git worktrees are checked out so
|
||||
# each worker owns an isolated branch instead of sharing the primary's tree. Omit to default
|
||||
# to a sibling directory of the repo root.
|
||||
# worktreeRoot: /Users/me/src/.bridged-worktrees
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
# the feature disabled. By default the daemon never reaps, caps, or drains sessions.
|
||||
# idleTtlSeconds → reap READY/DONE sessions idle longer than this (never BUSY/SPAWNING)
|
||||
# contextCap → force-release a session after this many delegated turns
|
||||
# drainTimeoutSeconds → seconds to wait for BUSY sessions on shutdown before forced teardown
|
||||
# lifecycle:
|
||||
# idleTtlSeconds: 300
|
||||
# contextCap: 10
|
||||
# drainTimeoutSeconds: 5
|
||||
|
||||
# Durable reply delivery (CB-307 Stage 2). OMIT this block entirely to keep the default
|
||||
# in-memory, soft-state reply inbox (late worker replies are held only until a daemon bounce).
|
||||
# Set a broker uri to swap in the AMQP-backed inbox: worker replies with no open send are held
|
||||
# on a durable per-target queue (agent.<target>.inbox) and survive a restart — the broker
|
||||
# redelivers anything the primary had not yet drained. Production default is LavinMQ; a stock
|
||||
# RabbitMQ speaks the same AMQP 0-9-1, so it is a URI-only swap.
|
||||
# uri → AMQP connection URI. No trailing slash ⇒ the default vhost "/"; an empty path ("/")
|
||||
# is vhost "" and will NOT connect. Encode a named vhost as .../%2Fmyvhost.
|
||||
# broker:
|
||||
# uri: amqp://guest:guest@127.0.0.1:5672
|
||||
|
||||
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open bridge_send,
|
||||
# the ReplyPushLoop injects a *drain nudge* (never the payload) into the primary's own herdr
|
||||
# pane — status-gated (only when injectable, never mid-turn) and bounded. Ack = drain: the loop
|
||||
# stops as soon as the primary's inbox is empty.
|
||||
# terminal → pin the primary's herdr terminal id. Omit to learn it from the connection on
|
||||
# the first orchestration-side MCP call (the normal case). An off-host or
|
||||
# non-herdr primary leaves this unresolved → the loop is a no-op and delivery
|
||||
# degrades to pull; the reply is still never lost.
|
||||
#
|
||||
# REQUIRED (CB-522) if the primary itself runs inside a herdr pane. Caller
|
||||
# identity resolves a loopback PID to its herdr pane, and PaneLocator scans
|
||||
# EVERY pane — not just bridged-spawned ones — so such a primary is otherwise
|
||||
# classified as a WORKER and refused SPAWN/SEND/STOP. That failure is
|
||||
# self-locking: the learned terminal is populated by the very orchestration
|
||||
# calls being refused, so only this pinned value can break the cycle. Read the
|
||||
# id off bridge_whoami (it reports the current terminal even while
|
||||
# misclassified) and re-pin whenever the primary moves panes.
|
||||
# pushReminders → max nudges before giving up (default 5)
|
||||
# pushBackoffMs → delay between nudges in ms (default 15000)
|
||||
# primary:
|
||||
# terminal: term_65619bd6174568
|
||||
# pushReminders: 5
|
||||
# pushBackoffMs: 15000
|
||||
|
||||
@@ -0,0 +1,142 @@
|
||||
# CB-307 — Active push-to-primary + reminder loop (the reliability layer)
|
||||
|
||||
**Status:** design (2026-07-19). Builds directly on the shipped durable landing zone
|
||||
(`AmqpReplyInbox`, main `2bc5f3a`, dogfooded live). gitea #5.
|
||||
|
||||
## Why this exists
|
||||
|
||||
Stage 2 gave a worker→primary reply a **durable place to wait** when no `bridge_send` is
|
||||
open: it lands in `agent.<target>.inbox` on the broker and survives a daemon bounce. But
|
||||
delivery is still **pull** — the primary only sees the reply if it happens to call
|
||||
`bridge_poll(target)` / `GET /sessions/{id}/replies`. A reply can sit indefinitely while
|
||||
the primary works on something else.
|
||||
|
||||
This layer makes delivery **active**: the bridge *pushes* a nudge to the primary the moment
|
||||
a reply lands, and keeps reminding (bounded) until the primary drains it. At-least-once,
|
||||
dedup by `msgId`, and — critically — it never loses the reply even if every push fails,
|
||||
because the durable inbox is the backstop.
|
||||
|
||||
## The hard constraint it works around
|
||||
|
||||
The bridge is an MCP **server**; the primary is an MCP **client**. A server cannot call
|
||||
into a client. So "push to the primary" cannot be an MCP response — it needs a *sideband*
|
||||
channel. The chosen channel: **inject a synthetic user-turn into the primary's own herdr
|
||||
terminal pane** — the same mechanism the bridge already uses to deliver tasks to workers,
|
||||
pointed at the primary's pane instead.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
W["worker"] -->|"bridge_reply (no open send)"| MS["MessageService.reply"]
|
||||
MS -->|"inbox.publish"| INBOX[("agent.<target>.inbox<br/>(durable, LavinMQ)")]
|
||||
MS -->|"notify"| LOOP["ReplyPushLoop"]
|
||||
LOOP -->|"status-gated inject"| PANE["primary's herdr pane"]
|
||||
PANE -->|"primary drains"| DRAIN["bridge_poll(target)<br/>= peek + ack"]
|
||||
DRAIN -->|"inbox now empty"| LOOP
|
||||
LOOP -.->|"still non-empty →<br/>re-inject on backoff"| PANE
|
||||
classDef store fill:#2c5282,stroke:#1a365d,color:#ffffff;
|
||||
class INBOX store
|
||||
```
|
||||
|
||||
*Figure 1 — a reply lands in the durable inbox; the push loop nudges the primary's pane;
|
||||
the primary's drain acks it; a still-full inbox triggers a bounded re-nudge.*
|
||||
|
||||
## Three increments
|
||||
|
||||
### Increment 1 — learn & store the primary's terminal_id
|
||||
|
||||
**Finding (seam map):** `ConnectionIdentity.resolve(remoteAddr, remotePort)` already returns
|
||||
the caller's herdr `terminal_id` for *every* MCP call, via `PaneLocator.terminalForPid`
|
||||
(walks `pane.list`, matches the caller PID to a pane's process tree). It is non-null whenever
|
||||
the caller runs in a herdr pane on this host. Today it's discarded for the primary
|
||||
(`presence.markPresent` is a no-op on it).
|
||||
|
||||
**Plan:** a single-slot `PrimaryRegistry` (thread-safe) holding the primary's `terminal_id`.
|
||||
Populate it from the **orchestration-side** MCP tools — `bridge_send`, `bridge_spawn`,
|
||||
`bridge_poll`, `bridge_list`, `bridge_status`, `bridge_profiles` — capturing
|
||||
`callerTerminal(exchange)` when it is (a) non-null and (b) **not** a registered worker
|
||||
session in `SessionManager`. That caller is, by construction, the primary. Worker-side tools
|
||||
(`bridge_reply`, `bridge_ask`) never set it.
|
||||
|
||||
- **Config override / pin:** a `primary: { terminal: "<id>" }` block in `BridgedConfig`
|
||||
(nested record, same shape as `Broker`). Lets an operator pin it, or supply it when
|
||||
derivation can't (see degrade case).
|
||||
- **Degrade:** if the primary is off-host or in a non-herdr terminal, `terminalForPid`
|
||||
returns null and no override is set → **the registry stays empty → the push loop is a
|
||||
no-op and we fall back to pull** (today's behaviour). The reply is never lost; it's just
|
||||
not actively pushed. This is a safe, explicit degradation, not a failure.
|
||||
|
||||
### Increment 2 — the push loop
|
||||
|
||||
A `ReplyPushLoop` component, notified at the single no-waiter call site
|
||||
(`MessageService.reply` → the `inbox.publish` branch, `MessageService.java:192`).
|
||||
|
||||
- **Inject a nudge, not the payload.** The injected turn tells the primary *to drain*
|
||||
(e.g. "Worker `<target>` returned a reply — run `bridge_poll(target=<target>)` to collect
|
||||
it"), it does **not** carry the reply text. Rationale: replies can be large/multiline and
|
||||
terminal injection would mangle them; the drain response is the clean transport. Keeps the
|
||||
push idempotent — re-nudging is harmless.
|
||||
- **Ack = drain.** The primary draining (`drainReplies` = peek + ack) is the acknowledgement.
|
||||
The loop's **stop condition is `inbox.peek(target).isEmpty()`** — the reply is gone from the
|
||||
inbox because it was acked. No new `bridge_ack` tool needed for v1 (see Increment 3).
|
||||
- **Status-gated injection (mechanism (b), chosen).** A dedicated lightweight scheduled loop,
|
||||
**not** the worker `Injector`. It injects via `AgentControl.send(primaryTerminal, nudge)`
|
||||
(the same herdr `agent.send` = `pane send-text` + submit that delivers to workers) only when
|
||||
`AgentControl.status(primaryTerminal).injectable()` (IDLE/BLOCKED) — never mid-turn. This keeps
|
||||
the primary path fully isolated from `WorkerPresence`/`StatusPoller` (which are worker-scoped),
|
||||
and makes it unit-testable with a fake `AgentControl` + an injected clock (per the CB-306
|
||||
`LongSupplier` clock + `Runnable` sleeper seam). Rejected (a) reuse-the-Injector: it would force
|
||||
the primary terminal into the worker poller set and couple to worker-presence semantics — more
|
||||
integration surface, harder to test, no real gain for a bounded reminder.
|
||||
- **Bounded reminder / backoff.** While `peek(target)` stays non-empty, re-inject on a
|
||||
backoff schedule up to a cap (N reminders or a max duration; config
|
||||
`primary.push_reminders` / `primary.push_backoff_ms`). After the cap, **stop reminding** —
|
||||
the reply remains in the durable inbox and the next natural poll (or a later worker reply's
|
||||
nudge) still surfaces it. Bounded so the bridge never spams the primary.
|
||||
|
||||
### Increment 3 — optional per-`msgId` `bridge_ack` tool (deferred)
|
||||
|
||||
Drain-as-ack is coarse: it clears *all* pending replies for a target at once. If finer
|
||||
control is ever needed (ack one reply, leave others held), add a `bridge_ack(msgId)` tool
|
||||
mapping to `inbox.ack(target, msgId)` — the port already supports per-`msgId` ack. Not built
|
||||
in v1; the stop-on-empty loop is sufficient.
|
||||
|
||||
## The two subtleties (decided here)
|
||||
|
||||
1. **Which caller is "the primary"?** Connection-derived, not self-reported: the caller whose
|
||||
resolved terminal is non-null **and not a registered worker session**, seen on an
|
||||
orchestration-side tool. This never mislabels a worker (workers are in `SessionManager`)
|
||||
and needs no new env var or argument (identity stays connection-derived, per the existing
|
||||
`BridgeMcp` invariant).
|
||||
|
||||
2. **Readiness-gate mismatch → dedicated loop.** The existing `Injector` gates delivery on
|
||||
`ready.test(target)` = `WorkerPresence` (the *worker's* MCP connected). The primary is not
|
||||
in `WorkerPresence`, so reusing `Injector` would mean forcing the primary terminal into the
|
||||
worker `StatusPoller` set and swapping the `ready` predicate — extra integration surface with
|
||||
worker-scoped machinery. Decision: **mechanism (b)** — a small dedicated scheduled loop that
|
||||
calls `AgentControl.status(primaryTerminal).injectable()` then `AgentControl.send(...)`, with
|
||||
an injected clock. Isolated from worker presence, trivially unit-testable, sufficient for a
|
||||
bounded reminder. (Verified live: this primary resolves to `term_656c8cc03e1f0b1`, pane
|
||||
`w2:pY` — the primary genuinely runs in a herdr pane on this host, so the path is exercisable.)
|
||||
|
||||
## Boundary note
|
||||
|
||||
This is the first time the bridge **writes into the primary's pane** — a new direction of
|
||||
control. It stays within the communication-bus identity: the injection is a **nudge** (a
|
||||
synthetic "go drain your replies" turn), **status-gated** so it never interrupts a turn,
|
||||
**bounded** so it never spams, carries **no env** and **never crosses the subscription
|
||||
boundary**. The bridge is signalling the primary that it has mail — not driving its work.
|
||||
|
||||
## Test plan
|
||||
|
||||
- **Unit (hermetic):** `PrimaryRegistry` set/clear/override; the "caller is primary iff
|
||||
non-null terminal AND not a registered session" predicate; the loop's stop-on-empty and
|
||||
bounded-reminder logic with an injected clock + a fake injector (no real herdr).
|
||||
- **Live dogfood (primary-side):** with the daemon on the broker jar + a real worker,
|
||||
delegate a task, let the worker reply after the `bridge_send` window closes, and observe the
|
||||
bridge inject a drain nudge into *this* primary pane; confirm draining stops the reminders;
|
||||
confirm an unreachable primary (registry empty) degrades to pull with no loss.
|
||||
|
||||
## Out of scope
|
||||
|
||||
Multi-host (CB-308) — the push loop is local-only; a remote primary is reached by its own
|
||||
local gateway, not cross-host injection. Federation reuses this loop per-gateway.
|
||||
+144
-5
@@ -6,7 +6,7 @@
|
||||
|
||||
<groupId>dev.ltms</groupId>
|
||||
<artifactId>bridged</artifactId>
|
||||
<version>0.1.0-SNAPSHOT</version>
|
||||
<version>1.0.0</version>
|
||||
<packaging>jar</packaging>
|
||||
|
||||
<name>bridged</name>
|
||||
@@ -17,13 +17,74 @@
|
||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||
<mainClass>dev.ltms.bridged.Bridged</mainClass>
|
||||
|
||||
<jackson.version>2.18.2</jackson.version>
|
||||
<javalin.version>6.3.0</javalin.version>
|
||||
<jackson.version>2.19.0</jackson.version>
|
||||
<javalin.version>6.7.0</javalin.version>
|
||||
<jetty.version>11.0.25</jetty.version>
|
||||
<mcp.version>2.0.0</mcp.version>
|
||||
<slf4j.version>2.0.16</slf4j.version>
|
||||
<logback.version>1.5.15</logback.version>
|
||||
<logback.version>1.5.18</logback.version>
|
||||
<junit.version>5.11.4</junit.version>
|
||||
<amqp.version>5.22.0</amqp.version>
|
||||
<testcontainers.version>1.20.4</testcontainers.version>
|
||||
<commons-compress.version>1.27.1</commons-compress.version>
|
||||
<commons-lang3.version>3.18.0</commons-lang3.version>
|
||||
</properties>
|
||||
|
||||
<!--
|
||||
Dependency security (validate with the JetBrains analyzer's Mend.io check on this pom).
|
||||
Deps are pinned to the latest available versions. Residual advisories with NO upstream fix,
|
||||
accepted for this loopback-bound daemon that processes no untrusted config:
|
||||
- jetty-http 11.0.25 (via Javalin): CVE-2026-2332, CVE-2025-11143 — Jetty 11 is EOL;
|
||||
fixed only in Jetty 12, which needs a Javalin major (6.x rides Jetty 11).
|
||||
- logback-core 1.5.18: CVE-2025-11226, CVE-2026-1225 — both require a MALICIOUS
|
||||
logback.xml (attacker with config write already has code execution); ours is trusted.
|
||||
- jackson-core 2.19.0: WS-2026-0003 — "insufficient information", no fixed version published.
|
||||
- tools.jackson.core (Jackson 3) 3.0.3 via the MCP SDK: CVE-2026-29062 (nesting-depth
|
||||
resource exhaustion). The SDK 2.0.0 is pinned to Jackson 3.0.3 + jackson-annotations
|
||||
3.0-rc5; bumping Jackson 3 to the patched 3.2.x breaks the SDK (annotation mismatch).
|
||||
Only the loopback /mcp endpoint parses this JSON, from trusted local Claude clients.
|
||||
The 11.0.23 -> 11.0.25 bump did clear jetty CVE-2024-8184 (5.9) and CVE-2024-6763.
|
||||
-->
|
||||
|
||||
<!-- Force the latest patched Jetty 11.x across all Javalin-pulled Jetty modules (no version
|
||||
skew). Javalin 6.x rides Jetty 11; a move to Jetty 12 needs a Javalin major. -->
|
||||
<dependencyManagement>
|
||||
<dependencies>
|
||||
<dependency>
|
||||
<groupId>org.eclipse.jetty</groupId>
|
||||
<artifactId>jetty-bom</artifactId>
|
||||
<version>${jetty.version}</version>
|
||||
<type>pom</type>
|
||||
<scope>import</scope>
|
||||
</dependency>
|
||||
<!-- The MCP SDK (Jackson 3) needs jackson-annotations with JsonFormat.Shape.POJO
|
||||
(the 3.0 line); it shares the com.fasterxml.jackson.annotation package with our
|
||||
Jackson 2.19 databind, so both must resolve to the same jar. 3.0 is built to work
|
||||
with Jackson 2.19 databind too — pin it to reconcile the two. -->
|
||||
<dependency>
|
||||
<groupId>com.fasterxml.jackson.core</groupId>
|
||||
<artifactId>jackson-annotations</artifactId>
|
||||
<version>3.0-rc5</version>
|
||||
</dependency>
|
||||
<!-- Testcontainers 1.20.4 pulls commons-compress 1.24.0 (test scope), which carries
|
||||
CVE-2024-25710 (8.1) + CVE-2024-26308 — both fixed in 1.26.0. Pin the patched line.
|
||||
Test-scope only (never shipped in the jar), but bumped per the CVE policy. -->
|
||||
<dependency>
|
||||
<groupId>org.apache.commons</groupId>
|
||||
<artifactId>commons-compress</artifactId>
|
||||
<version>${commons-compress.version}</version>
|
||||
</dependency>
|
||||
<!-- Testcontainers 1.20.4 also pulls commons-lang3 3.16.0 (test scope): CVE-2025-48924
|
||||
(uncontrolled recursion in ClassUtils), fixed in 3.18.0. Pin the patched line.
|
||||
Test-scope only (never shipped in the jar), bumped per the CVE policy. -->
|
||||
<dependency>
|
||||
<groupId>org.apache.commons</groupId>
|
||||
<artifactId>commons-lang3</artifactId>
|
||||
<version>${commons-lang3.version}</version>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
</dependencyManagement>
|
||||
|
||||
<dependencies>
|
||||
<!-- JSON + YAML (config, herdr wire format, REST bodies) -->
|
||||
<dependency>
|
||||
@@ -44,6 +105,24 @@
|
||||
<version>${javalin.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- MCP server: the SERVER face. Streamable-HTTP servlet mounted on Javalin's Jetty at
|
||||
/mcp, exposing bridge_send/bridge_reply/bridge_status as thin adapters over REST. -->
|
||||
<dependency>
|
||||
<groupId>io.modelcontextprotocol.sdk</groupId>
|
||||
<artifactId>mcp</artifactId>
|
||||
<version>${mcp.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Broker client (CB-307 Stage 2): AMQP 0-9-1. Default deploy targets LavinMQ; this same
|
||||
client speaks to RabbitMQ unchanged (URI-only swap), so integration tests run against a
|
||||
stock RabbitMQ container. Only wired when a broker: block is present in config; absent →
|
||||
the in-memory ReplyInbox. -->
|
||||
<dependency>
|
||||
<groupId>com.rabbitmq</groupId>
|
||||
<artifactId>amqp-client</artifactId>
|
||||
<version>${amqp.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Logging -->
|
||||
<dependency>
|
||||
<groupId>org.slf4j</groupId>
|
||||
@@ -63,6 +142,22 @@
|
||||
<version>${junit.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
|
||||
<!-- Testcontainers RabbitMQ: spins a real broker for the @Tag("contract") AMQP integration
|
||||
test only. Excluded from the default build (contract group), so `mvn clean install`
|
||||
stays hermetic and green without Docker; run under -Pcontract with Docker present. -->
|
||||
<dependency>
|
||||
<groupId>org.testcontainers</groupId>
|
||||
<artifactId>rabbitmq</artifactId>
|
||||
<version>${testcontainers.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.testcontainers</groupId>
|
||||
<artifactId>junit-jupiter</artifactId>
|
||||
<version>${testcontainers.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
|
||||
<build>
|
||||
@@ -74,6 +169,30 @@
|
||||
<version>3.14.0</version>
|
||||
</plugin>
|
||||
|
||||
<!--
|
||||
Coverage (CB-509). Build-time tooling only — never a compile/runtime dependency, so
|
||||
it adds nothing to the shipped jar. Report lands at target/site/jacoco/index.html and
|
||||
target/site/jacoco/jacoco.csv. No `check` rule / threshold is wired: a coverage gate
|
||||
rewards writing tests that execute lines, which is the failure mode this project is
|
||||
trying to avoid, not encourage.
|
||||
-->
|
||||
<plugin>
|
||||
<groupId>org.jacoco</groupId>
|
||||
<artifactId>jacoco-maven-plugin</artifactId>
|
||||
<version>0.8.13</version>
|
||||
<executions>
|
||||
<execution>
|
||||
<id>prepare-agent</id>
|
||||
<goals><goal>prepare-agent</goal></goals>
|
||||
</execution>
|
||||
<execution>
|
||||
<id>report</id>
|
||||
<phase>test</phase>
|
||||
<goals><goal>report</goal></goals>
|
||||
</execution>
|
||||
</executions>
|
||||
</plugin>
|
||||
|
||||
<!-- Unit tests run by default; contract tests (live herdr) are tag-excluded. -->
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
@@ -116,7 +235,27 @@
|
||||
</profile>
|
||||
<profile>
|
||||
<id>contract</id>
|
||||
<properties><excludedGroups></excludedGroups></properties>
|
||||
<properties><excludedGroups/></properties>
|
||||
<build>
|
||||
<plugins>
|
||||
<plugin>
|
||||
<artifactId>maven-surefire-plugin</artifactId>
|
||||
<configuration>
|
||||
<!-- Docker-engine compat (see "Running the contract tests" in
|
||||
docs/CB-307-Reliable-Delivery.md): Testcontainers 1.20.4's docker-java
|
||||
client defaults to Docker API 1.32 when no version is set, but modern
|
||||
engines (OrbStack on this dev host, min 1.40) reject that as too old —
|
||||
which surfaces as "Could not find a valid Docker environment". Pinning
|
||||
api.version=1.43 works on OrbStack and Docker 24+, and is overridable
|
||||
per-host via -Dapi.version. Only active under -Pcontract, so the
|
||||
default hermetic build never sets it. -->
|
||||
<systemPropertyVariables>
|
||||
<api.version>1.43</api.version>
|
||||
</systemPropertyVariables>
|
||||
</configuration>
|
||||
</plugin>
|
||||
</plugins>
|
||||
</build>
|
||||
</profile>
|
||||
</profiles>
|
||||
</project>
|
||||
|
||||
@@ -1,19 +1,68 @@
|
||||
package dev.ltms.bridged;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.config.ConfigRef;
|
||||
import dev.ltms.bridged.config.ConfigWatcher;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrClient;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.LeadTabScanner;
|
||||
import dev.ltms.bridged.lead.LeadLauncher;
|
||||
import dev.ltms.bridged.herdr.PaneLocator;
|
||||
import dev.ltms.bridged.herdr.UnixSocketHerdrClient;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.inject.CompletionResolver;
|
||||
import dev.ltms.bridged.inject.ExhaustedPatternLookup;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.inject.StatusPoller;
|
||||
import dev.ltms.bridged.inject.TurnListener;
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import dev.ltms.bridged.auth.MemberRegistry;
|
||||
import dev.ltms.bridged.auth.CallerResolver;
|
||||
import dev.ltms.bridged.mcp.BridgeMcp;
|
||||
import dev.ltms.bridged.mcp.ConnectionIdentity;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.health.FleetHealthMonitor;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.mcp.LsofPeerPidLookup;
|
||||
import dev.ltms.bridged.mcp.LsofProcessCwdLookup;
|
||||
import dev.ltms.bridged.msg.AmqpReplyInbox;
|
||||
import dev.ltms.bridged.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.msg.ReplyInbox;
|
||||
import dev.ltms.bridged.msg.LeadHeartbeatLoop;
|
||||
import dev.ltms.bridged.msg.ReplyPushLoop;
|
||||
import dev.ltms.bridged.rest.BridgedApp;
|
||||
import dev.ltms.bridged.worker.WorkerService;
|
||||
import dev.ltms.bridged.session.GitWorktrees;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.session.SessionReaper;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.bridged.member.CompositePeerLauncher;
|
||||
import dev.ltms.bridged.member.HerdrPeerLauncher;
|
||||
import dev.ltms.bridged.member.OpenCodeLauncher;
|
||||
import io.javalin.Javalin;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* {@code bridged} entry point. Wires the real herdr socket client to the REST app and
|
||||
@@ -24,41 +73,472 @@ public final class Bridged {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(Bridged.class);
|
||||
|
||||
/** How often the injector samples a busy worker's status while it has queued work. */
|
||||
private static final long INJECT_POLL_MILLIS = 250;
|
||||
/** CB-504: how long to wait at startup for herdr's socket before serving degraded. */
|
||||
private static final long HERDR_WAIT_SECONDS = 30;
|
||||
private static final long HERDR_WAIT_POLL_MILLIS = 500;
|
||||
|
||||
static void main(String[] args) {
|
||||
Path configPath = Path.of(args.length > 0 ? args[0] : "bridged.yaml");
|
||||
BridgedConfig cfg = BridgedConfig.load(configPath);
|
||||
// CB-559: `cfg` stays the startup snapshot — every validation and every piece of one-time
|
||||
// wiring below reads it, and must, because those decisions cannot be unmade. `config` is the
|
||||
// live reference the hot paths read per use. Which keys can actually move is ConfigRef's
|
||||
// contract; adding a reader here does not make a key reloadable by itself.
|
||||
ConfigRef config = new ConfigRef(configPath, cfg);
|
||||
|
||||
// The primary/host env that launched bridged must not be tainted.
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
guard.assertPrimaryClean(System.getenv());
|
||||
|
||||
// CB-501: refuse to start if the bind is wider than the auth mode can defend. Under
|
||||
// loopback-trust, "not a known worker" means "the primary" — sound only because the OS
|
||||
// refuses remote connections to a loopback socket. This throws rather than warns so the
|
||||
// dangerous configuration cannot be reached by ignoring a log line.
|
||||
cfg.validateAuthExposure();
|
||||
cfg.validateLeadTabPrefixes();
|
||||
// CB-542: a subscription:true profile whose env: reseats ANTHROPIC_BASE_URL/AUTH_TOKEN would
|
||||
// reach an unguarded endpoint (the launcher skips SubscriptionGuard for it). Refuse at load.
|
||||
cfg.validateSubscriptionProfiles();
|
||||
cfg.validateCharters();
|
||||
// CB-548: every architect slot must name a configured workers: profile — the strong-model
|
||||
// backend the future spawn lifecycle would read. A stale reference dies here, not later.
|
||||
cfg.validateMembers();
|
||||
|
||||
Path socket = cfg.herdrSocket() != null && !cfg.herdrSocket().isBlank()
|
||||
? Path.of(cfg.herdrSocket())
|
||||
: UnixSocketHerdrClient.defaultSocketPath();
|
||||
|
||||
UnixSocketHerdrClient herdr = UnixSocketHerdrClient.connect(socket, new com.fasterxml.jackson.databind.ObjectMapper());
|
||||
Runtime.getRuntime().addShutdownHook(new Thread(herdr::close));
|
||||
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
WorkspaceControl spaces = new WorkspaceControl(herdr);
|
||||
WorkerService workers = new WorkerService(agents, spaces, guard, cfg.worker(), System::getenv);
|
||||
// CB-402: one adapter per configured peer kind, fronted by a composite router. A profile's
|
||||
// `kind:` selects its adapter — claude-code (the default) and opencode partition the profile
|
||||
// set — and the composite dispatches each SPI call to the adapter that owns the profile/pane.
|
||||
Map<String, BridgedConfig.Profile> claudeProfiles = new LinkedHashMap<>();
|
||||
Map<String, BridgedConfig.Profile> opencodeProfiles = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, w) -> {
|
||||
if (w.isOpenCode()) {
|
||||
opencodeProfiles.put(name, w);
|
||||
} else {
|
||||
claudeProfiles.put(name, w);
|
||||
}
|
||||
});
|
||||
List<HerdrPeerLauncher> adapters = new ArrayList<>();
|
||||
// The claude-code adapter is the always-present default; keep it even with no profiles (so a
|
||||
// bridge configured with no workers, or opencode-only, still has a well-defined base adapter)
|
||||
// unless opencode is the only kind configured.
|
||||
if (!claudeProfiles.isEmpty() || opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new ClaudeCodeLauncher(agents, spaces, guard,
|
||||
claudeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet()));
|
||||
}
|
||||
if (!opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new OpenCodeLauncher(agents, spaces,
|
||||
opencodeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet()));
|
||||
}
|
||||
AtomicReference<Function<String, Integer>> liveCountRef = new AtomicReference<>(_ -> 0);
|
||||
PeerLauncher workers = new CompositePeerLauncher(
|
||||
adapters,
|
||||
cfg.effectiveDefaultProfile(),
|
||||
config,
|
||||
profileName -> liveCountRef.get().apply(profileName));
|
||||
// CB-504: under supervision (launchd/systemd) bridged can start before herdr's socket
|
||||
// exists. The client itself is lazy — it connects per call — but the orphan reap below is
|
||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||
// would crash the daemon into a restart loop. Wait, then degrade rather than die: serving
|
||||
// with /healthz reporting "degraded" is strictly more useful than exiting.
|
||||
boolean herdrUp = awaitHerdr(herdr);
|
||||
if (herdrUp) {
|
||||
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
||||
// with the previous process — reap those leaked orphans now, before we start serving.
|
||||
workers.reapOrphanWorkers();
|
||||
} else {
|
||||
log.warn("herdr did not answer within {}s — starting anyway; /healthz will report "
|
||||
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
HERDR_WAIT_SECONDS);
|
||||
}
|
||||
|
||||
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
||||
// CB-301-ext: worktree provisioning seam, optionally rooted at a configured directory.
|
||||
// CB-303 part 2: context cap is opt-in and disabled (0) when absent/null.
|
||||
int contextCap = 0;
|
||||
if (cfg.lifecycle() != null && cfg.lifecycle().contextCap() != null
|
||||
&& cfg.lifecycle().contextCap() > 0) {
|
||||
contextCap = cfg.lifecycle().contextCap();
|
||||
}
|
||||
boolean clearAfterTurn = cfg.lifecycle() != null && cfg.lifecycle().clearAfterTurn();
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(cfg.worktreeRoot()),
|
||||
System::nanoTime, contextCap, clearAfterTurn);
|
||||
liveCountRef.set(profileName -> (int) sessions.roster().stream()
|
||||
.filter(s -> profileName.equals(s.profile()))
|
||||
.count());
|
||||
|
||||
// CB-303 part 1: idle-ttl reaper — only when configured, defaults to disabled.
|
||||
final SessionReaper reaper;
|
||||
if (cfg.lifecycle() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() > 0) {
|
||||
reaper = new SessionReaper(sessions, cfg.lifecycle().idleTtlSeconds());
|
||||
reaper.start();
|
||||
} else {
|
||||
reaper = null;
|
||||
}
|
||||
|
||||
// CB-530: every pane the config names as a lead, merged from `leaders:` and the legacy
|
||||
// singular pin. PrimaryRegistry below still tracks ONE terminal — it addresses the push
|
||||
// loop's nudges, which need a single destination — so it keeps the legacy pin.
|
||||
Map<String, String> leadTerminals = cfg.leaderTerminals();
|
||||
if (leadTerminals.size() > 1) {
|
||||
log.info("leads: {} panes recognised {}", leadTerminals.size(), leadTerminals.values());
|
||||
}
|
||||
// CB-531: on top of the legacy primary.terminal pin, discover leads by the tab labels the
|
||||
// operator writes. CB-557 moved the settings onto the lead they describe, so scanning is on
|
||||
// whenever a `fleet.leaders:` entry exists — with no leads configured the supplier is a
|
||||
// constant and never touches herdr, exactly as a missing `leadScan:` block used to behave.
|
||||
// CB-579: each lead now names its own exact `tab:` label, so one scanner discovers every
|
||||
// configured lead regardless of how differently their tabs are labelled — the old
|
||||
// single-shared-tabPrefix limitation (and its warning) is gone.
|
||||
final Supplier<Map<String, String>> leads;
|
||||
var leaders = cfg.fleet().leaders();
|
||||
if (!leaders.isEmpty()) {
|
||||
Set<String> memberSpaces = cfg.profiles().values().stream()
|
||||
.map(BridgedConfig.Profile::workspace)
|
||||
.filter(Objects::nonNull)
|
||||
.collect(Collectors.toSet());
|
||||
Map<String, String> tabToName = new LinkedHashMap<>();
|
||||
leaders.forEach((name, leader) -> {
|
||||
if (leader != null && leader.tab() != null && !leader.tab().isBlank()) {
|
||||
tabToName.put(leader.tab(), name);
|
||||
}
|
||||
});
|
||||
// One shared rescan cadence: still taken from the first entry, as before — it is an
|
||||
// operational cadence, not identity, so there is no correctness reason to give every
|
||||
// lead its own scanner.
|
||||
int scanIntervalSeconds = leaders.values().iterator().next().scanIntervalSeconds();
|
||||
leads = new LeadTabScanner(herdr, tabToName, memberSpaces,
|
||||
TimeUnit.SECONDS.toNanos(scanIntervalSeconds), System::nanoTime);
|
||||
log.info("lead scan: tabs {} host a lead (rescan every {}s, member spaces {} excluded)",
|
||||
tabToName.keySet(), scanIntervalSeconds, memberSpaces);
|
||||
} else {
|
||||
leads = () -> leadTerminals;
|
||||
}
|
||||
|
||||
// CB-558: start any declared lead that is not already running. After the scanner is built,
|
||||
// because both read the same tab labels and the ordering makes that dependency visible; and
|
||||
// only when herdr answered, because the launcher's whole safety property is that it can
|
||||
// count live leads first — it must never guess and risk a second orchestrator.
|
||||
if (herdrUp && !leaders.isEmpty()) {
|
||||
int launched = new LeadLauncher(agents, spaces, cfg).ensureLeads();
|
||||
if (launched > 0) {
|
||||
log.info("lead auto-launch: {} lead(s) started", launched);
|
||||
}
|
||||
}
|
||||
|
||||
// CB-548: config-declared architect slots. Config supplies only the stable name → profile
|
||||
// map; the terminal → slot binding is owned by the registry and is empty at startup, so no
|
||||
// pane resolves to an architect until the later spawn lifecycle binds one. The registry is
|
||||
// what CallerResolver resolves against and what that lifecycle will read profiles from;
|
||||
// nothing here spawns a slot.
|
||||
MemberRegistry members = new MemberRegistry(cfg.fleet());
|
||||
sessions.setMemberLifecycle(members);
|
||||
if (!members.slots().isEmpty()) {
|
||||
log.info("member slots: {} configured {} — none bound yet (a slot is idle until the "
|
||||
+ "spawn lifecycle binds a live terminal to it)",
|
||||
members.slots().size(), members.slots().keySet());
|
||||
}
|
||||
|
||||
// Status-gated injector (CB-103): the single writer into workers, fed by a poller.
|
||||
// The blocking message endpoint (CB-104) is the producer; the poller is inert until then.
|
||||
Injector injector = new Injector(agents);
|
||||
StatusPoller poller = new StatusPoller(agents, injector, INJECT_POLL_MILLIS);
|
||||
poller.start();
|
||||
Runtime.getRuntime().addShutdownHook(new Thread(poller::stop));
|
||||
// CB-106: a confirmed turn completion resolves a blocked send whose worker never replied.
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
// CB-578 stage A: classify a completion-fallback scrape that matches a profile's configured
|
||||
// usage-limit refusal as BACKEND_EXHAUSTED rather than handing it back as a real answer.
|
||||
// Compiled once at startup, keyed by profile name; a profile with no exhaustedPattern is
|
||||
// simply absent here, so its workers keep today's completion-fallback behaviour unchanged.
|
||||
Map<String, Pattern> exhaustedPatternsByProfile = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (profile.hasExhaustedPattern()) {
|
||||
exhaustedPatternsByProfile.put(name, Pattern.compile(profile.exhaustedPattern()));
|
||||
}
|
||||
});
|
||||
ExhaustedPatternLookup exhaustedPatterns = target -> sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(session -> exhaustedPatternsByProfile.get(session.profile()))
|
||||
.orElse(null);
|
||||
log.info("backend-exhausted classification (CB-578 stage A): {}",
|
||||
CompletionResolver.coverage(cfg.profiles().keySet(), exhaustedPatternsByProfile.keySet()));
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous, exhaustedPatterns);
|
||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||
MemberPresence presence = sessions.asPresence();
|
||||
TurnListener turnListener = new TurnListener() {
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
completion.onTurnComplete(target);
|
||||
sessions.onTurnComplete(target);
|
||||
}
|
||||
|
||||
Javalin app = new BridgedApp(herdr, workers).build();
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
return sessions.hasPostTurnAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
completion.resolveBeforePostAction(target);
|
||||
return sessions.onTurnCompleteWithPostAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target, dev.ltms.bridged.msg.TurnToken token) {
|
||||
completion.onDelivered(target, token);
|
||||
sessions.onDelivered(target, token);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
completion.onTurnFailed(target);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
completion.onTurnFailed(target, reason);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
};
|
||||
Injector injector = new Injector(agents, turnListener, deliverableTo(presence, leads),
|
||||
presence::forget);
|
||||
StatusPoller poller = new StatusPoller(agents, injector, Injector.POLL_INTERVAL_MILLIS);
|
||||
poller.start();
|
||||
|
||||
// CB-307: reply inbox. A broker: block (with a uri) selects the AMQP-backed durable adapter;
|
||||
// absent, bridged stays soft-state on the in-memory inbox. The AMQP inbox owns a broker
|
||||
// connection, so keep the reference to close it in the ordered shutdown hook.
|
||||
final ReplyInbox replyInbox;
|
||||
if (cfg.broker() != null && cfg.broker().isConfigured()) {
|
||||
replyInbox = AmqpReplyInbox.open(cfg.broker().uri());
|
||||
log.info("reply inbox: AMQP broker (durable) at {}", cfg.broker().uri());
|
||||
} else {
|
||||
replyInbox = new InMemoryReplyInbox();
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
}
|
||||
// CB-307: learn the primary's terminal from orchestration tool calls (or pin from config).
|
||||
// The pin also feeds CallerResolver below: a primary running inside a herdr pane would
|
||||
// otherwise resolve as a worker and be refused every orchestration tool.
|
||||
String pinnedPrimaryTerminal = cfg.primary() != null ? cfg.primary().terminal() : null;
|
||||
PrimaryRegistry primaryRegistry = new PrimaryRegistry(pinnedPrimaryTerminal);
|
||||
// CB-532: `primary.terminal` is superseded and no longer needed for either of its jobs —
|
||||
// identity comes from `leaders:`/`leadScan:`, and reply nudges now follow the delegating
|
||||
// lead. Say so once at startup rather than leaving a redundant pin to look load-bearing.
|
||||
if (pinnedPrimaryTerminal != null && !pinnedPrimaryTerminal.isBlank()) {
|
||||
log.warn("primary.terminal is DEPRECATED (CB-532) and can be deleted: identity now comes "
|
||||
+ "from leaders:/leadScan:, and reply nudges follow the lead that delegated. "
|
||||
+ "It still works, and is still the fallback nudge destination when a restart "
|
||||
+ "has lost the delegation map. Its pushReminders/pushBackoffMs stay valid.");
|
||||
}
|
||||
// CB-307: active push-to-primary loop — nudge the primary when replies land without an
|
||||
// open bridge_send. Uses its own lightweight scheduled executor, separate from the injector.
|
||||
int maxReminders = cfg.primary() != null ? cfg.primary().remindersOrDefault() : 5;
|
||||
long backoffMs = cfg.primary() != null ? cfg.primary().backoffMsOrDefault() : 15_000L;
|
||||
var pushScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-push-").unstarted(r));
|
||||
// CB-502: the registry is built before the service and the push loop so send/reply outcomes
|
||||
// are counted at their single funnel rather than at each of the two caller-facing surfaces.
|
||||
// CB-512: the push loop takes it too, so nudge outcomes (delivered|exhausted) are counted.
|
||||
Metrics metrics = BridgedMetrics.create(sessions, replyInbox);
|
||||
var pushLoop = new ReplyPushLoop(primaryRegistry, agents, replyInbox,
|
||||
pushScheduler, maxReminders, backoffMs, metrics);
|
||||
// CB-551: idle-lead heartbeat. Opt-in; absent `leadHeartbeat:` this is never constructed, so
|
||||
// an upgraded daemon cannot silently start spending subscription on nudging an idle lead.
|
||||
// It has its own single-thread scheduler and holds its own scheduler shutdown via close().
|
||||
final LeadHeartbeatLoop heartbeat;
|
||||
var heartbeatScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-heartbeat-").unstarted(r));
|
||||
if (cfg.leadHeartbeat() != null) {
|
||||
var hb = cfg.leadHeartbeat();
|
||||
heartbeat = new LeadHeartbeatLoop(primaryRegistry, agents, replyInbox, sessions::roster,
|
||||
pushLoop, heartbeatScheduler, System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(hb.idleAfterSeconds()), hb.backoffMs(), hb.quietNudgeCap(),
|
||||
metrics);
|
||||
heartbeat.start();
|
||||
} else {
|
||||
heartbeat = null;
|
||||
heartbeatScheduler.shutdownNow();
|
||||
}
|
||||
MessageService messages = new MessageService(agents, injector, rendezvous, replyInbox,
|
||||
pushLoop, metrics);
|
||||
|
||||
// Health is a slow whole-fleet observer. Keep it separate from the 250ms delivery poller.
|
||||
final FleetHealthMonitor healthMonitor;
|
||||
var healthScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-health-").unstarted(r));
|
||||
if (cfg.health() != null && cfg.health().isEnabled()) {
|
||||
// CB-580: a member found GONE/NEVER_READY must fail whatever ticket is waiting on it,
|
||||
// through the same idempotent target-wide operation CB-516 already uses on release.
|
||||
healthMonitor = new FleetHealthMonitor(agents, sessions::roster, messages, healthScheduler,
|
||||
System::nanoTime, cfg.health().intervalOrDefault(), messages::abandon);
|
||||
String coverage = FleetHealthMonitor.coverage(true,
|
||||
cfg.health().notifications() != null && cfg.health().notifications().configured());
|
||||
if ("detection-only".equals(coverage)) {
|
||||
log.warn("fleet health: {} (no notification sink configured)", coverage);
|
||||
} else {
|
||||
log.info("fleet health: {}", coverage);
|
||||
}
|
||||
healthMonitor.start();
|
||||
} else {
|
||||
healthMonitor = null;
|
||||
healthScheduler.shutdownNow();
|
||||
}
|
||||
|
||||
// CB-520: the reply inbox only consumes for agents this gateway owns. own on acquire,
|
||||
// release on teardown. Do this before CB-516 so the inbox is owned before any reply can land.
|
||||
sessions.onAcquire(replyInbox::own);
|
||||
// CB-516: releasing a worker must fail whatever send was waiting on it. Without this a
|
||||
// torn-down delegation kept reporting PENDING until the 30-minute async timeout, and never
|
||||
// reached /metrics — the delegation was unresolvable and nothing said so.
|
||||
sessions.onRelease(terminal -> {
|
||||
messages.abandon(terminal, "the worker session was released before it replied");
|
||||
replyInbox.release(terminal);
|
||||
primaryRegistry.forgetDelegation(terminal); // CB-532: don't leak the lead binding
|
||||
});
|
||||
|
||||
// MCP server face (CB-105): bridge_send/bridge_reply/bridge_status, mounted at /mcp.
|
||||
// Caller identity is resolved from the connection (peer PID → herdr pane), not arguments.
|
||||
ConnectionIdentity identity = new ConnectionIdentity(
|
||||
new PaneLocator(herdr), new LsofPeerPidLookup(), new LsofProcessCwdLookup());
|
||||
|
||||
// CB-501: one resolver behind both entry paths. Worker identity still comes from the
|
||||
// connection and is never token-gated, so enabling token mode cannot lock the fleet out.
|
||||
final CallerResolver callers;
|
||||
if (cfg.auth().tokenMode()) {
|
||||
String token = System.getenv(cfg.auth().tokenEnv());
|
||||
if (token == null || token.isBlank()) {
|
||||
throw new IllegalStateException("auth.mode=token but env var " + cfg.auth().tokenEnv()
|
||||
+ " is unset or empty — export it before starting bridged");
|
||||
}
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, true, token, leads, members);
|
||||
log.info("auth: token mode (bearer required for non-worker callers, env {})",
|
||||
cfg.auth().tokenEnv());
|
||||
} else {
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, false, null, leads, members);
|
||||
log.info("auth: loopback-trust (any loopback non-worker caller is the primary)");
|
||||
}
|
||||
|
||||
BridgeMcp mcp = new BridgeMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, metrics, new BridgeMcp.CapacitySource(profile -> liveCountRef.get().apply(profile),
|
||||
profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.maxLoad();
|
||||
}, () -> config.get().profiles().keySet(), System::nanoTime),
|
||||
new BridgeMcp.HealthCoverageSource(() -> {
|
||||
var health = config.get().health();
|
||||
return FleetHealthMonitor.coverage(health != null && health.isEnabled(),
|
||||
health != null && health.notifications() != null && health.notifications().configured());
|
||||
}));
|
||||
|
||||
// CB-559: opt-in config reload. With no `configReload:` block nothing is constructed, so an
|
||||
// upgraded daemon behaves exactly as before — the file is read once at boot and never again.
|
||||
final ConfigWatcher configWatcher;
|
||||
if (cfg.configReload() != null && cfg.configReload().isEnabled()) {
|
||||
configWatcher = new ConfigWatcher(config, cfg.configReload().intervalSeconds());
|
||||
configWatcher.start();
|
||||
} else {
|
||||
configWatcher = null;
|
||||
}
|
||||
|
||||
// CB-303 part 3: single ordered shutdown hook. Drain sessions first while herdr is still
|
||||
// open (so releases reach the daemon), then stop poller/message/mcp/reaper, and close herdr
|
||||
// last. This replaces the earlier independent hooks that could race and close herdr early.
|
||||
Runtime.getRuntime().addShutdownHook(new Thread(() -> {
|
||||
sessions.close(cfg.lifecycle() != null ? cfg.lifecycle().drainTimeoutSeconds() : null);
|
||||
poller.stop();
|
||||
messages.close();
|
||||
pushLoop.close();
|
||||
if (heartbeat != null) heartbeat.close(); // CB-551: stop the idle-lead heartbeat scheduler
|
||||
if (healthMonitor != null) healthMonitor.stop();
|
||||
if (configWatcher != null) configWatcher.stop(); // CB-559: stop polling the config file
|
||||
mcp.close();
|
||||
if (reaper != null) reaper.stop();
|
||||
// Release the broker connection last among message resources (no-op for the in-memory inbox).
|
||||
if (replyInbox instanceof AutoCloseable closeable) {
|
||||
try {
|
||||
closeable.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("reply inbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
herdr.close();
|
||||
}));
|
||||
|
||||
Javalin app = new BridgedApp(herdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics).build();
|
||||
app.start(cfg.bind().host(), cfg.bind().port());
|
||||
log.info("bridged listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@link Injector}'s readiness gate (CB-534): a target is deliverable if it is a spawned
|
||||
* member whose agent has connected the bridge MCP, <em>or</em> a lead.
|
||||
*
|
||||
* <p>The gate exists for one reason — to hold a delivery out of a <em>spawned</em> member's boot
|
||||
* window, where herdr already reports {@code idle} but the TUI would drop an injected paste. That
|
||||
* hazard is a property of spawning. A lead is never spawned: the operator started it and named it
|
||||
* (or labelled its tab) only once it was up, so there is no boot window to guard.
|
||||
*
|
||||
* <p>A lead is also never enrolled in {@link MemberPresence} — {@code BridgeMcp} marks presence
|
||||
* for every spawned member (worker and architect), deliberately, since that map doubles as the
|
||||
* member roster's availability signal and a lead counted there would show up as an available
|
||||
* member. So without the second disjunct a lead is permanently un-deliverable: every
|
||||
* lead→lead send sat on the gate for {@code READINESS_GRACE_POLLS} (~60s) and then failed
|
||||
* having never been typed into the pane.
|
||||
*
|
||||
* <p>The lead set is read through the supplier on each call rather than snapshotted, so a lead
|
||||
* discovered by {@code leadScan} after startup becomes deliverable without a restart.
|
||||
*/
|
||||
static Predicate<String> deliverableTo(MemberPresence presence, Supplier<Map<String, String>> leads) {
|
||||
return target -> presence.isPresent(target) || leads.get().containsKey(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll herdr's {@code ping} until it answers or {@link #HERDR_WAIT_SECONDS} elapses (CB-504).
|
||||
*
|
||||
* @return true if herdr answered, false if it never did
|
||||
*/
|
||||
private static boolean awaitHerdr(HerdrClient herdr) {
|
||||
long deadline = System.nanoTime() + HERDR_WAIT_SECONDS * 1_000_000_000L;
|
||||
boolean waited = false;
|
||||
while (true) {
|
||||
try {
|
||||
herdr.call("ping");
|
||||
if (waited) {
|
||||
log.info("herdr is up");
|
||||
}
|
||||
return true;
|
||||
} catch (HerdrException e) {
|
||||
if (System.nanoTime() >= deadline) {
|
||||
return false;
|
||||
}
|
||||
if (!waited) {
|
||||
log.info("waiting up to {}s for the herdr socket…", HERDR_WAIT_SECONDS);
|
||||
waited = true;
|
||||
}
|
||||
try {
|
||||
Thread.sleep(HERDR_WAIT_POLL_MILLIS);
|
||||
} catch (InterruptedException ie) {
|
||||
Thread.currentThread().interrupt();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private Bridged() {
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.time.Instant;
|
||||
import java.time.ZoneOffset;
|
||||
import java.time.format.DateTimeFormatter;
|
||||
|
||||
/**
|
||||
* Append-only record of privileged actions (CB-505).
|
||||
*
|
||||
* <p>Writes JSON lines to a dedicated {@code audit} logger — its own appender, separate from the
|
||||
* chatty app log — so the trail stays greppable and can later be shipped without dragging debug
|
||||
* noise along.
|
||||
*
|
||||
* <p><strong>Message content is never recorded.</strong> This bridge carries the user's source
|
||||
* code, diffs, and prompts; an audit trail that quietly accumulated them would be a transcript
|
||||
* archive wearing a security control's clothing. Records carry <em>who / what / against what /
|
||||
* outcome</em> and correlation ids only.
|
||||
*/
|
||||
public final class AuditLog {
|
||||
|
||||
private static final Logger AUDIT = LoggerFactory.getLogger("audit");
|
||||
private static final DateTimeFormatter TS =
|
||||
DateTimeFormatter.ofPattern("yyyy-MM-dd'T'HH:mm:ss.SSSXXX").withZone(ZoneOffset.UTC);
|
||||
|
||||
private AuditLog() {
|
||||
}
|
||||
|
||||
/** Record an allowed action. */
|
||||
public static void allowed(Principal caller, Authz.Action action, String target) {
|
||||
write(caller, action, target, "allowed", null);
|
||||
}
|
||||
|
||||
/** Record a refused action and why. */
|
||||
public static void denied(Principal caller, Authz.Action action, String target, String reason) {
|
||||
write(caller, action, target, "denied", reason);
|
||||
}
|
||||
|
||||
/** Record an action that was authorized but then failed downstream (guard, timeout, herdr). */
|
||||
public static void failed(Principal caller, Authz.Action action, String target, String reason) {
|
||||
write(caller, action, target, "failed", reason);
|
||||
}
|
||||
|
||||
private static void write(Principal caller, Authz.Action action, String target,
|
||||
String outcome, String reason) {
|
||||
Principal c = caller != null ? caller : Principal.anonymous();
|
||||
StringBuilder sb = new StringBuilder(200);
|
||||
// The timestamp is built here rather than by the appender pattern: a pattern that wrapped
|
||||
// literal braces around the message collides with logback's own variable substitution.
|
||||
sb.append("{\"ts\":\"").append(TS.format(Instant.now())).append('"')
|
||||
.append(",\"role\":\"").append(c.role()).append('"')
|
||||
.append(",\"actor\":\"").append(esc(c.describe())).append('"')
|
||||
.append(",\"pid\":").append(c.pid())
|
||||
.append(",\"action\":\"").append(action).append('"')
|
||||
.append(",\"target\":").append(target == null ? "null" : '"' + esc(target) + '"')
|
||||
.append(",\"outcome\":\"").append(outcome).append('"');
|
||||
if (reason != null) {
|
||||
sb.append(",\"reason\":\"").append(esc(reason)).append('"');
|
||||
}
|
||||
sb.append('}');
|
||||
// The appender supplies the timestamp, so it cannot disagree with the app log's clock.
|
||||
AUDIT.info(sb.toString());
|
||||
}
|
||||
|
||||
/** Minimal JSON string escaping — these values are ids and short reasons, never free text. */
|
||||
private static String esc(String s) {
|
||||
return s.replace("\\", "\\\\").replace("\"", "\\\"")
|
||||
.replace("\n", "\\n").replace("\r", "\\r").replace("\t", "\\t");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
/**
|
||||
* The authorization table (CB-505), stated once and enforced on both entry paths.
|
||||
*
|
||||
* <p>Most of these rules are already true de facto — {@code BridgeMcp} derives a worker's identity
|
||||
* from the connection rather than reading it from an argument, so a worker has never been able to
|
||||
* reply <em>as</em> another worker over MCP. What was missing is that the REST surface trusted the
|
||||
* session id in the URL path, and neither surface checked role at all. This class makes the
|
||||
* invariant explicit and testable rather than emergent.
|
||||
*/
|
||||
public final class Authz {
|
||||
|
||||
private Authz() {
|
||||
}
|
||||
|
||||
/** A privileged operation, named for the audit trail. */
|
||||
public enum Action {
|
||||
/** Spawn a worker peer. */
|
||||
SPAWN,
|
||||
/** Tear a worker peer down. */
|
||||
STOP,
|
||||
/** Deliver a turn to a session (or answer a worker's question). */
|
||||
SEND,
|
||||
/** A worker's terminal reply for its own turn. */
|
||||
REPLY,
|
||||
/** A worker's mid-turn question to the primary. */
|
||||
ASK,
|
||||
/** Collect held replies from a session's inbox. */
|
||||
DRAIN,
|
||||
/** Read-only observation: status, roster, profiles, task polling. */
|
||||
READ,
|
||||
/** Scrape the metrics endpoint. */
|
||||
METRICS
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code caller} may perform {@code action} against {@code targetSession}.
|
||||
*
|
||||
* @param targetSession the session id in the request path; only consulted for the worker-scoped
|
||||
* actions ({@code REPLY}, {@code ASK}), ignored otherwise, may be
|
||||
* {@code null}
|
||||
*/
|
||||
public static boolean permits(Principal caller, Action action, String targetSession) {
|
||||
if (caller == null || caller.isAnonymous()) {
|
||||
return false; // authenticated as nothing ⇒ authorized for nothing
|
||||
}
|
||||
return switch (action) {
|
||||
// Fleet lifecycle is the primary's alone — spawn, stop, drain. An architect
|
||||
// deliberately does NOT get these (CB-548), so it cannot tear down or stand up workers
|
||||
// even though it coordinates them; and a worker driving any of these would be a worker
|
||||
// escalating into the orchestrator role.
|
||||
case SPAWN, STOP, DRAIN -> caller.isPrimary();
|
||||
|
||||
// Delivering a turn is open to the primary and the architect: an architect delegates
|
||||
// to workers (that is the role's point) but still has no lifecycle rights. A worker is
|
||||
// excluded — sending would be it escalating.
|
||||
case SEND -> caller.isPrimary() || caller.isArchitect();
|
||||
|
||||
// The load-bearing rule: a caller acts only as the pane it occupies. CB-532 widened who
|
||||
// that can be — a lead answering another lead is replying for its OWN terminal, which
|
||||
// this already permits — while the rule itself is unchanged, and is what stops anyone
|
||||
// forging a reply for a rendezvous someone else is waiting on. An architect's own pane
|
||||
// passes through the same check, so it can answer a funnel that delegated to it. An
|
||||
// unnamed primary (token/loopback, no pane) owns nothing and is still excluded.
|
||||
case REPLY, ASK -> caller.ownsSession(targetSession);
|
||||
|
||||
// Observation is open to every authenticated role: a worker legitimately polls its own
|
||||
// status, and the roster carries no secrets.
|
||||
case READ, METRICS -> caller.isPrimary() || caller.isWorker() || caller.isArchitect();
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Why a request was refused, for the error body. Distinguishes "you are nobody" from "you are
|
||||
* somebody, but not the right somebody" — the first is a credential problem (401), the second
|
||||
* an authorization one (403).
|
||||
*/
|
||||
public static boolean isUnauthenticated(Principal caller) {
|
||||
return caller == null || caller.isAnonymous();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,272 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import dev.ltms.bridged.mcp.ConnectionIdentity;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.security.MessageDigest;
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* Resolves every caller to a {@link Principal}, for both entry paths into the core (CB-501).
|
||||
*
|
||||
* <p>There are two of them and they are not layered the way the docs suggest: {@code BridgeMcp}
|
||||
* calls the service layer directly and is mounted as a raw servlet (so it never passes through a
|
||||
* Javalin filter), while the REST routes historically resolved no identity at all. Both now
|
||||
* delegate here, so the authorization rules are stated once instead of drifting apart.
|
||||
*
|
||||
* <p><strong>Resolution order</strong> — connection identity first, token second, nothing third:
|
||||
* <ol>
|
||||
* <li>A loopback peer PID that maps to a pane named by {@code leaders:}, by the legacy
|
||||
* {@code primary.terminal} pin, or by an operator-labelled lead tab (CB-307, CB-530, CB-531)
|
||||
* ⇒ {@link Role#PRIMARY}, carrying that lead's
|
||||
* name. The pane mapping is as unforgeable as a worker's, and the config explicitly names
|
||||
* that pane as a lead's own — without this rule a lead running <em>inside</em> a herdr pane
|
||||
* is misread as a worker and locked out of orchestration. More than one pane may be named,
|
||||
* so two leads can work as peers rather than one being demoted.</li>
|
||||
* <li>A loopback peer PID that maps to a pane bound to a CB-548 architect slot ⇒
|
||||
* {@link Role#ARCHITECT}, carrying the slot name. Just unforgeable as a worker's, and
|
||||
* resolved from the <em>live</em> terminal→slot binding (never a request argument), before
|
||||
* the generic worker fallback.</li>
|
||||
* <li>A loopback peer PID that maps to any other herdr pane ⇒ {@link Role#WORKER}. This is
|
||||
* unforgeable (the OS reports the PID, herdr owns the PID→pane map) and is honoured
|
||||
* regardless of auth mode, so enabling auth never breaks the fleet.</li>
|
||||
* <li>Otherwise, under {@code token} mode, a valid bearer token ⇒ {@link Role#PRIMARY}.</li>
|
||||
* <li>Otherwise, under {@code loopback-trust}, a loopback caller ⇒ {@link Role#PRIMARY}
|
||||
* (the historical behaviour, now an explicit configured choice).</li>
|
||||
* <li>Otherwise {@link Role#ANONYMOUS}.</li>
|
||||
* </ol>
|
||||
*/
|
||||
public final class CallerResolver {
|
||||
|
||||
private final ConnectionIdentity identity;
|
||||
private final boolean tokenMode;
|
||||
private final byte[] expectedToken; // null unless tokenMode
|
||||
/**
|
||||
* terminal_id → lead name; empty when nothing is pinned. CB-530.
|
||||
*
|
||||
* <p>A supplier rather than a map because the registry is no longer fixed at startup: CB-531
|
||||
* discovers leads by scanning herdr for operator-labelled tabs, so a lead that opens its tab
|
||||
* after the daemon booted must still be recognised. Consulted per resolve; the scanner behind
|
||||
* it is TTL-cached, so this is a map lookup in the common case.
|
||||
*/
|
||||
private final Supplier<Map<String, String>> leadTerminals;
|
||||
/**
|
||||
* terminal_id → architect slot name; empty when nothing is configured. CB-548.
|
||||
*
|
||||
* <p>Like {@link #leadTerminals}, a supplier rather than a fixed map, so a binding injected
|
||||
* after startup — when the later spawn lifecycle establishes a live architect session, or an
|
||||
* operator pins one — takes effect without a restart. Consulted per resolve; today's wiring
|
||||
* in {@code Bridged} reads a constant from config, which is the degenerate live case.
|
||||
*/
|
||||
private final Supplier<Map<String, String>> architectTerminals;
|
||||
private final Function<String, MemberRole> memberSlotRoles;
|
||||
private final Function<String, String> memberSlotNames;
|
||||
|
||||
/** Loopback-trust resolver: no token required, historical behaviour. Test-only. */
|
||||
CallerResolver(ConnectionIdentity identity) {
|
||||
this(identity, false, null, Map.of());
|
||||
}
|
||||
|
||||
/** As {@link #CallerResolver(ConnectionIdentity, boolean, String, Map)} with no leads pinned. Test-only. */
|
||||
CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token) {
|
||||
this(identity, tokenMode, token, Map.of());
|
||||
}
|
||||
|
||||
/**
|
||||
* Single-pin form for the legacy {@code primary.terminal}-only configuration — one lead, named
|
||||
* {@code primary}.
|
||||
*
|
||||
* <p>A static factory rather than a fourth constructor overload on purpose: {@code String} and
|
||||
* {@code Map} overloads are ambiguous for a literal {@code null} argument, which is a compile
|
||||
* error at the call site and exactly the shape "unpinned" is written in.
|
||||
*
|
||||
* @param pinnedPrimaryTerminal the primary's own herdr {@code terminal_id}
|
||||
* ({@code null}/blank = unpinned)
|
||||
*/
|
||||
static CallerResolver pinnedTo(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token, String pinnedPrimaryTerminal) {
|
||||
return new CallerResolver(identity, tokenMode, token,
|
||||
pinnedPrimaryTerminal == null || pinnedPrimaryTerminal.isBlank()
|
||||
? Map.of() : Map.of(pinnedPrimaryTerminal, "primary"));
|
||||
}
|
||||
|
||||
/**
|
||||
* @param identity connection-based worker identification
|
||||
* @param tokenMode when true, a non-worker caller must present a valid bearer token
|
||||
* @param token the expected bearer token; required (non-blank) when {@code tokenMode}
|
||||
* @param leadTerminals herdr {@code terminal_id} → lead name for every configured lead
|
||||
* (CB-530). A caller resolving to one of these panes is that lead — a
|
||||
* {@link Role#PRIMARY} — rather than a worker. Empty = nothing pinned,
|
||||
* so every pane resolves as a worker.
|
||||
*/
|
||||
CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Map<String, String> leadTerminals) {
|
||||
this(identity, tokenMode, token, fixed(leadTerminals), null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Live-registry form: {@code leadTerminals} is consulted on every resolve, so leads discovered
|
||||
* after startup (CB-531's tab scan) take effect without a restart.
|
||||
*
|
||||
* <p>A static factory rather than a fourth constructor overload, for the same reason as
|
||||
* {@link #pinnedTo}: {@code Map} and {@code Supplier} overloads are ambiguous for a literal
|
||||
* {@code null}.
|
||||
*/
|
||||
static CallerResolver withLeads(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token,
|
||||
Supplier<Map<String, String>> leadTerminals) {
|
||||
return new CallerResolver(identity, tokenMode, token, leadTerminals, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Live registry form that can confirm a bound slot is an architect slot.
|
||||
*
|
||||
* <p>This is the only public construction path. It keeps terminal bindings and slot roles in
|
||||
* the same {@link MemberRegistry}, so a configured architect can resolve as an architect.
|
||||
*/
|
||||
public static CallerResolver withLeadsAndMembers(ConnectionIdentity identity,
|
||||
boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
MemberRegistry members) {
|
||||
return new CallerResolver(identity, tokenMode, token, leadTerminals,
|
||||
members == null ? null : members::snapshot,
|
||||
members == null ? null : members::roleForSlot,
|
||||
members == null ? null : members::nameForSlot);
|
||||
}
|
||||
|
||||
private static Supplier<Map<String, String>> fixed(Map<String, String> leadTerminals) {
|
||||
Map<String, String> snapshot = leadTerminals == null ? Map.of() : Map.copyOf(leadTerminals);
|
||||
return () -> snapshot;
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals) {
|
||||
this(identity, tokenMode, token, leadTerminals, architectTerminals, null);
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals,
|
||||
Function<String, MemberRole> memberSlotRoles) {
|
||||
this(identity, tokenMode, token, leadTerminals, architectTerminals, memberSlotRoles, null);
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals,
|
||||
Function<String, MemberRole> memberSlotRoles,
|
||||
Function<String, String> memberSlotNames) {
|
||||
if (tokenMode && (token == null || token.isBlank())) {
|
||||
throw new IllegalArgumentException(
|
||||
"auth.mode=token requires a non-empty token; check that the env var named by "
|
||||
+ "auth.tokenEnv is exported to the daemon's environment");
|
||||
}
|
||||
this.identity = identity;
|
||||
this.tokenMode = tokenMode;
|
||||
this.expectedToken = tokenMode ? token.getBytes(StandardCharsets.UTF_8) : null;
|
||||
this.leadTerminals = leadTerminals == null ? Map::of : leadTerminals;
|
||||
this.architectTerminals = architectTerminals == null ? Map::of : architectTerminals;
|
||||
this.memberSlotRoles = memberSlotRoles == null ? _ -> null : memberSlotRoles;
|
||||
this.memberSlotNames = memberSlotNames == null ? Function.identity() : memberSlotNames;
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-recognised leads, {@code terminal_id → name} (CB-535).
|
||||
*
|
||||
* <p>Deliberately read from the same supplier {@link #resolve} consults, rather than from a
|
||||
* second copy handed to the roster: a lead that is <em>listed</em> but would not <em>resolve</em>
|
||||
* (or the reverse) is an address a peer cannot actually reach, and the two answers drifting apart
|
||||
* is precisely the confusion this exists to end. Live, so a lead discovered by the tab scan after
|
||||
* startup appears without a restart.
|
||||
*/
|
||||
public Map<String, String> leads() {
|
||||
return leadTerminals.get();
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-recognised architect slots, {@code terminal_id → slot name} (CB-548).
|
||||
*
|
||||
* <p>Read from the same supplier {@link #resolve} consults, so a slot that is <em>listed</em>
|
||||
* here but would not <em>resolve</em> (or the reverse) cannot drift apart. Live for the same
|
||||
* reason as {@link #leads()}.
|
||||
*/
|
||||
public Map<String, String> members() {
|
||||
return architectTerminals.get();
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the caller of a request.
|
||||
*
|
||||
* @param remoteAddr the connection's remote address
|
||||
* @param remotePort the connection's remote port (used for the peer-PID lookup)
|
||||
* @param authorizationHeader the raw {@code Authorization} header, or {@code null}
|
||||
*/
|
||||
public Principal resolve(String remoteAddr, int remotePort, String authorizationHeader) {
|
||||
ConnectionIdentity.Caller c = identity.resolve(remoteAddr, remotePort);
|
||||
if (c.terminal() != null) {
|
||||
String lead = leadTerminals.get().get(c.terminal());
|
||||
if (lead != null) {
|
||||
// The config names this pane as a lead's own. The pane mapping is exactly as
|
||||
// unforgeable as a worker's, so it outranks the token path — no credential needed.
|
||||
// Checked before the architect registry so a pane named in BOTH is still the lead
|
||||
// (CB-548 preserves every existing leader behaviour).
|
||||
return Principal.leader(lead, c.terminal(), c.pid());
|
||||
}
|
||||
String slot = architectTerminals.get().get(c.terminal());
|
||||
if (slot != null && memberSlotRoles.apply(slot) == MemberRole.ARCHITECT) {
|
||||
// The config/live binding names this pane as an architect slot's own. Same
|
||||
// unforgeable pane mapping; the live binding, never a request argument, decides.
|
||||
// Check the slot role too: this defence in depth prevents a bad lifecycle bind from
|
||||
// escalating a dev or reviewer into an architect. Checked before the worker fallback.
|
||||
return Principal.architect(memberSlotNames.apply(slot), c.terminal(), c.pid());
|
||||
}
|
||||
return Principal.worker(c.terminal(), c.pid()); // unforgeable; never token-gated
|
||||
}
|
||||
|
||||
if (tokenMode) {
|
||||
return presentedTokenMatches(authorizationHeader)
|
||||
? Principal.primary(c.pid())
|
||||
: Principal.anonymous();
|
||||
}
|
||||
|
||||
// loopback-trust: same-host callers that are not workers are the primary. A non-loopback
|
||||
// caller is anonymous even here — and startup refuses that combination anyway
|
||||
// (BridgedConfig.validateAuthExposure), so this is defence in depth, not the control.
|
||||
return isLoopback(remoteAddr) ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
}
|
||||
|
||||
private boolean presentedTokenMatches(String authorizationHeader) {
|
||||
String presented = bearerValue(authorizationHeader);
|
||||
if (presented == null) {
|
||||
return false;
|
||||
}
|
||||
// Constant-time: MessageDigest.isEqual does not short-circuit on the first differing byte,
|
||||
// so a token cannot be recovered a byte at a time by timing the response.
|
||||
return MessageDigest.isEqual(presented.getBytes(StandardCharsets.UTF_8), expectedToken);
|
||||
}
|
||||
|
||||
/** Extract the credential from {@code Authorization: Bearer <token>}, or {@code null}. */
|
||||
private static String bearerValue(String header) {
|
||||
if (header == null) {
|
||||
return null;
|
||||
}
|
||||
String h = header.trim();
|
||||
if (h.length() < 7 || !h.regionMatches(true, 0, "Bearer ", 0, 7)) {
|
||||
return null;
|
||||
}
|
||||
String token = h.substring(7).trim();
|
||||
return token.isEmpty() ? null : token;
|
||||
}
|
||||
|
||||
private static boolean isLoopback(String remoteAddr) {
|
||||
if (remoteAddr == null) {
|
||||
return false;
|
||||
}
|
||||
return remoteAddr.equals("127.0.0.1") || remoteAddr.equals("::1")
|
||||
|| remoteAddr.equals("0:0:0:0:0:0:0:1") || remoteAddr.startsWith("127.");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
|
||||
/** Optional session lifecycle hook for live member-slot bindings. */
|
||||
public interface MemberLifecycle {
|
||||
|
||||
MemberLifecycle NONE = new MemberLifecycle() {
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
}
|
||||
};
|
||||
|
||||
void acquired(MemberRole role, String profile, String terminal);
|
||||
|
||||
void released(String terminal);
|
||||
}
|
||||
@@ -0,0 +1,232 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* The architect-slot registry (CB-548): every gateway-local architect name and the strong-model
|
||||
* profile it points at, plus the <em>live</em> bindings from a live architect's herdr terminal to
|
||||
* its slot.
|
||||
*
|
||||
* <p>Two halves, split by who owns each:
|
||||
* <ul>
|
||||
* <li><b>slots</b> — configured once, keyed by the gateway-local unique name; each carries the
|
||||
* {@code profile} reference the spawn lifecycle reads when it stands the slot up. A read-only
|
||||
* snapshot taken at construction.</li>
|
||||
* <li><b>terminal bindings</b> — owned by this registry and initially <em>empty</em>. Config
|
||||
* declares no architect terminal, so at startup every slot is idle and nothing resolves to an
|
||||
* architect; a session only becomes one when the spawn lifecycle {@linkplain #bind(String,
|
||||
* String) binds} its terminal to a slot. {@link CallerResolver} reads this through
|
||||
* {@link #snapshot()} to turn a pane into an {@link Role#ARCHITECT}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Spawning/lifecycle is deliberately a separate unit: this class only owns the bindings and
|
||||
* exposes the map the resolver resolves against plus the profile lookup lifecycle will call.
|
||||
* Nothing here creates or manages an architect session.
|
||||
*/
|
||||
public final class MemberRegistry implements MemberLifecycle {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(MemberRegistry.class);
|
||||
|
||||
/**
|
||||
* One flattened {@code fleet:} entry.
|
||||
*
|
||||
* <p>Flattened because a slot name is unique only <em>within</em> its pool — {@code sonnet} may
|
||||
* legitimately be both a developer and a reviewer — while a terminal binds to exactly one thing.
|
||||
* The qualified {@link #key()} is what that binding uses.
|
||||
*
|
||||
* @param name the slot's key inside its pool
|
||||
* @param role the pool it came from
|
||||
* @param profile the backend it runs on
|
||||
*/
|
||||
public record Entry(String name, MemberRole role, String profile) {
|
||||
/** {@code "architect:opus"} — unique across pools, unlike {@link #name()}. */
|
||||
public String key() {
|
||||
return role.wireName() + ":" + name;
|
||||
}
|
||||
}
|
||||
|
||||
private final Map<String, Entry> slots;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code this}. */
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one registry. Leaders are not members. */
|
||||
public MemberRegistry(BridgedConfig.Fleet fleet) {
|
||||
Map<String, Entry> flat = new LinkedHashMap<>();
|
||||
if (fleet != null) {
|
||||
for (MemberRole role : MemberRole.values()) {
|
||||
fleet.pool(role).forEach((name, slot) -> {
|
||||
if (slot != null) {
|
||||
Entry e = new Entry(name, role, slot.profile());
|
||||
flat.put(e.key(), e);
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
this.slots = Collections.unmodifiableMap(flat);
|
||||
}
|
||||
|
||||
/** The configured slots, keyed by qualified {@link Entry#key()}. Unmodifiable snapshot. */
|
||||
public Map<String, Entry> slots() {
|
||||
return slots;
|
||||
}
|
||||
|
||||
/** The slots belonging to {@code role}, in definition order. */
|
||||
public Map<String, Entry> slotsFor(MemberRole role) {
|
||||
Map<String, Entry> out = new LinkedHashMap<>();
|
||||
slots.forEach((key, e) -> {
|
||||
if (e.role() == role) {
|
||||
out.put(key, e);
|
||||
}
|
||||
});
|
||||
return Collections.unmodifiableMap(out);
|
||||
}
|
||||
|
||||
/**
|
||||
* An immutable copy of the live {@code terminal_id → slot name} bindings.
|
||||
*
|
||||
* <p>Passed to {@link CallerResolver} as the source of architect identity, and what
|
||||
* {@code bridge_whoami}/the roster will read to say which slot a pane hosts. Empty until the
|
||||
* spawn lifecycle binds a slot.
|
||||
*/
|
||||
public Map<String, String> snapshot() {
|
||||
synchronized (terminalToSlot) {
|
||||
return Map.copyOf(terminalToSlot);
|
||||
}
|
||||
}
|
||||
|
||||
/** The slot a live terminal is bound to, or {@code null} if it is not an architect slot. */
|
||||
public String slotForTerminal(String terminal) {
|
||||
if (terminal == null) {
|
||||
return null;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
return terminalToSlot.get(terminal);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The strong-model profile a slot runs under — what the spawn lifecycle reads.
|
||||
*
|
||||
* @return the slot's configured {@code profile}, or {@code null} if the slot is unknown or
|
||||
* declares none
|
||||
*/
|
||||
public String profileForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return (e == null || e.profile() == null) ? null : e.profile();
|
||||
}
|
||||
|
||||
/** The role a qualified slot key belongs to, or {@code null} when the key is unknown. */
|
||||
public MemberRole roleForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return e == null ? null : e.role();
|
||||
}
|
||||
|
||||
/** The unqualified configured name for a slot, or {@code null} if it is unknown. */
|
||||
public String nameForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return e == null ? null : e.name();
|
||||
}
|
||||
|
||||
/** True when {@code slotName} is a configured architect slot. */
|
||||
public boolean isSlot(String slotName) {
|
||||
return slots.containsKey(slotName);
|
||||
}
|
||||
|
||||
/**
|
||||
* Bind {@code terminal} to {@code slot} (CB-548).
|
||||
*
|
||||
* <p>The spawn lifecycle calls this when it stands a slot up. The bind is atomic and preserves
|
||||
* the two cardinality invariants: a terminal may occupy at most one slot, and a slot may host at
|
||||
* most one terminal. Binding the same terminal to the same slot again is a harmless no-op.
|
||||
*
|
||||
* @param slot a configured slot name, or the bind is refused
|
||||
* @param terminal the pane that will act as this architect
|
||||
* @return {@code true} if the binding is now {@code terminal → slot}; {@code false} if it was
|
||||
* refused — an unknown slot, a terminal already bound to a different slot, or a slot
|
||||
* already hosting a different terminal
|
||||
*/
|
||||
public boolean bind(String slot, String terminal) {
|
||||
if (slot == null || terminal == null || terminal.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
if (!isSlot(slot)) {
|
||||
return false; // unknown slot — nothing to bind to
|
||||
}
|
||||
String existingSlot = terminalToSlot.get(terminal);
|
||||
if (existingSlot != null) {
|
||||
return slot.equals(existingSlot); // already this slot (idempotent) or a different one
|
||||
}
|
||||
if (terminalToSlot.containsValue(slot)) {
|
||||
return false; // slot already hosts a terminal — no second one
|
||||
}
|
||||
terminalToSlot.put(terminal, slot);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Compare-safe unbind of {@code expectedTerminal} from {@code slot} (CB-548).
|
||||
*
|
||||
* <p>The spawn lifecycle calls this when it tears a slot down. Only the exact binding
|
||||
* {@code expectedTerminal → slot} is removed; if that terminal was since rebound to a different
|
||||
* slot (or the slot to a different terminal), the call is a no-op returning {@code false} — a
|
||||
* stale unbind must never remove a replacement.
|
||||
*
|
||||
* @param slot the slot the caller believes the terminal is bound to
|
||||
* @param expectedTerminal the terminal it expects to be bound there
|
||||
* @return {@code true} if {@code expectedTerminal → slot} was removed; {@code false} if nothing
|
||||
* was (no such binding, or the binding had already moved)
|
||||
*/
|
||||
public boolean unbind(String slot, String expectedTerminal) {
|
||||
if (slot == null || expectedTerminal == null) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
String current = terminalToSlot.get(expectedTerminal);
|
||||
if (current == null || !slot.equals(current)) {
|
||||
return false; // absent, or a replacement/moved binding — leave it in place
|
||||
}
|
||||
terminalToSlot.remove(expectedTerminal);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Bind only architect sessions to a free slot with the resolved profile.
|
||||
*
|
||||
* <p>The role check is lifecycle policy. {@link CallerResolver} repeats it when resolving a
|
||||
* binding, so a later lifecycle regression cannot turn a worker into an architect.
|
||||
*/
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
if (role != MemberRole.ARCHITECT || terminal == null || terminal.isBlank()) {
|
||||
return;
|
||||
}
|
||||
// slotsFor preserves definition order, so duplicate-profile slots use the first free one.
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if (Objects.equals(profile, entry.profile()) && bind(entry.key(), terminal)) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
log.info("member slot: no free architect slot for profile={}; session remains a worker", profile);
|
||||
}
|
||||
|
||||
/** Unbind a released terminal using the compare-safe registry operation. */
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
String slot = slotForTerminal(terminal);
|
||||
if (slot != null) {
|
||||
unbind(slot, terminal);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,127 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
/**
|
||||
* A resolved caller: its {@link Role}, and — for a worker — the herdr {@code terminal_id} that
|
||||
* identifies which worker it is (CB-501).
|
||||
*
|
||||
* @param role what this caller is authorized to act as
|
||||
* @param terminal the herdr {@code terminal_id} of the pane this caller occupies — a worker's, or
|
||||
* (since CB-532) a named lead's; {@code null} for an unnamed primary resolved off a
|
||||
* token or loopback trust, and for {@code ANONYMOUS}
|
||||
* @param pid the connecting process id, or {@code -1} when not resolvable (audit context)
|
||||
* @param name for a lead resolved from the CB-530 {@code leaders:} registry, which lead it is;
|
||||
* for an architect resolved from the CB-548 {@code architects:} registry, which
|
||||
* slot it occupies; {@code null} for every other caller, including an unnamed primary
|
||||
*/
|
||||
public record Principal(Role role, String terminal, long pid, String name) {
|
||||
|
||||
/**
|
||||
* Three-arg form for the callers that have no name to carry (workers, anonymous, and the
|
||||
* token/loopback primary paths). Kept so adding CB-530's {@code name} did not churn every
|
||||
* construction site — and so a reconstructed principal without a stashed name still works.
|
||||
*/
|
||||
public Principal(Role role, String terminal, long pid) {
|
||||
this(role, terminal, pid, null);
|
||||
}
|
||||
|
||||
/** A caller authenticated as nothing — the default when no check establishes anything else. */
|
||||
public static Principal anonymous() {
|
||||
return new Principal(Role.ANONYMOUS, null, -1);
|
||||
}
|
||||
|
||||
/** The orchestrating session, unnamed (token or loopback-trust path). */
|
||||
public static Principal primary(long pid) {
|
||||
return new Principal(Role.PRIMARY, null, pid);
|
||||
}
|
||||
|
||||
/**
|
||||
* A named lead from the {@code leaders:} registry (CB-530).
|
||||
*
|
||||
* <p>Carries {@link Role#PRIMARY}: a lead <em>is</em> a primary as far as authorization goes,
|
||||
* so every existing {@code isPrimary()} gate keeps working unchanged and the role table needed
|
||||
* no new entry. The name is reporting only — it lets {@code bridge_whoami} say <em>which</em>
|
||||
* lead is asking once more than one is configured.
|
||||
*
|
||||
* <p><strong>CB-532: a lead now carries the terminal it was matched by.</strong> Under CB-530 it
|
||||
* deliberately did not, because {@code terminal} meant "which worker pane" everywhere and a
|
||||
* non-null one would have enrolled the lead in the worker presence map. That reading was what
|
||||
* made a lead unaddressable: {@link #ownsSession} could never be true for it, so
|
||||
* {@code bridge_reply} was refused and one lead could send to another but never be answered.
|
||||
* The terminal now means "which pane is this caller", the presence map keys on
|
||||
* {@link #isSpawnedMember()} instead, and a lead is a peer that can both send and receive.
|
||||
*/
|
||||
public static Principal leader(String name, String terminal, long pid) {
|
||||
return new Principal(Role.PRIMARY, terminal, pid, name);
|
||||
}
|
||||
|
||||
/** A worker peer, identified by its herdr pane. */
|
||||
public static Principal worker(String terminal, long pid) {
|
||||
return new Principal(Role.WORKER, terminal, pid);
|
||||
}
|
||||
|
||||
/**
|
||||
* An architect (CB-548), identified by the slot it occupies and the pane bound to it.
|
||||
*
|
||||
* <p>Carries {@link Role#ARCHITECT}. {@code slotName} is reporting only — it lets
|
||||
* {@code bridge_whoami} say <em>which</em> architect slot is asking, and it is the key the
|
||||
* (future) spawn lifecycle reads a profile back from. Identity is the {@code terminal}: like a
|
||||
* worker's it comes from the connection and the live terminal→slot binding, so
|
||||
* {@code ownsSession} works exactly as it does for a worker — an architect acts as its own
|
||||
* pane and no other.
|
||||
*/
|
||||
public static Principal architect(String slotName, String terminal, long pid) {
|
||||
return new Principal(Role.ARCHITECT, terminal, pid, slotName);
|
||||
}
|
||||
|
||||
public boolean isPrimary() {
|
||||
return role == Role.PRIMARY;
|
||||
}
|
||||
|
||||
public boolean isArchitect() {
|
||||
return role == Role.ARCHITECT;
|
||||
}
|
||||
|
||||
public boolean isWorker() {
|
||||
return role == Role.WORKER;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether this caller is a spawned member with its own pane.
|
||||
*
|
||||
* <p>Both workers and architects are spawned members. A lead is excluded because recording it
|
||||
* as present would count it as an available member in the roster.
|
||||
*/
|
||||
public boolean isSpawnedMember() {
|
||||
return role == Role.WORKER || role == Role.ARCHITECT;
|
||||
}
|
||||
|
||||
public boolean isAnonymous() {
|
||||
return role == Role.ANONYMOUS;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether this caller may act <em>as</em> {@code sessionId} — the "own session only" rule that
|
||||
* keeps one peer from replying or asking on another's behalf.
|
||||
*
|
||||
* <p>The rule is about <em>identity</em>, not rank: a caller may act as the pane it demonstrably
|
||||
* occupies, and as no other. CB-532 dropped the extra {@code isWorker()} conjunct that used to
|
||||
* be here. It was not what enforced the rule — {@code terminal.equals(sessionId)} is, and that
|
||||
* terminal comes from the connection, so it cannot be forged either way. All the conjunct did
|
||||
* was make a lead permanently unable to answer anyone, since a lead's terminal was null and a
|
||||
* lead is not a worker. A caller with no terminal at all (an off-host or token-authenticated
|
||||
* primary) still owns nothing, which is the case the null check covers.
|
||||
*/
|
||||
public boolean ownsSession(String sessionId) {
|
||||
return terminal != null && terminal.equals(sessionId);
|
||||
}
|
||||
|
||||
/** Short, non-sensitive description for audit lines and error details. */
|
||||
public String describe() {
|
||||
return switch (role) {
|
||||
case WORKER -> "worker:" + terminal;
|
||||
case ARCHITECT -> "architect:" + name;
|
||||
case PRIMARY -> name == null ? "primary" : "leader:" + name;
|
||||
case ANONYMOUS -> "anonymous";
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
/**
|
||||
* What a caller is allowed to be on the bus (CB-501).
|
||||
*
|
||||
* <p>The ordering matters conceptually: {@link #PRIMARY} is the <em>most</em> privileged role
|
||||
* (it spawns, stops, sends to any session, and drains any inbox), not the least. Before CB-501
|
||||
* the daemon reached {@code PRIMARY} by <em>failing</em> every other check — any caller that did
|
||||
* not resolve to a known worker pane was treated as the primary. That is inverted here:
|
||||
* {@link #ANONYMOUS} is the fallback, and {@code PRIMARY} must be established.
|
||||
*/
|
||||
public enum Role {
|
||||
|
||||
/**
|
||||
* The orchestrating session. Established either by being a loopback caller that is not a
|
||||
* worker pane (under {@code loopback-trust}) or by presenting a valid bearer token (under
|
||||
* {@code token} mode).
|
||||
*/
|
||||
PRIMARY,
|
||||
|
||||
/**
|
||||
* A worker peer, identified by its herdr pane. Unforgeable: derived from the connection's
|
||||
* loopback peer PID via herdr's PID→pane map, never from a request argument.
|
||||
*/
|
||||
WORKER,
|
||||
|
||||
/**
|
||||
* A config-declared architect slot (CB-548): a gateway-local named session on a strong-model
|
||||
* profile that coordinates and delegates turns but does not own the fleet. Unforgeable like a
|
||||
* worker's — derived from the connection's pane and the live terminal→slot binding, never from
|
||||
* a request argument. May {@code SEND} a turn, {@code REPLY}/{@code ASK} only as its own pane,
|
||||
* and {@code READ}/{@code METRICS}; may <em>not</em> {@code SPAWN}/{@code STOP}/{@code DRAIN}
|
||||
* (those stay the primary's, to keep lifecycle in one pair of hands).
|
||||
*/
|
||||
ARCHITECT,
|
||||
|
||||
/** Authenticated as nothing. Authorized for nothing but {@code /healthz}. */
|
||||
ANONYMOUS
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,274 @@
|
||||
package dev.ltms.bridged.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The daemon's live configuration, re-readable without a restart (CB-559).
|
||||
*
|
||||
* <p>Consumers hold this, not a {@link BridgedConfig}, and read through {@link #get()} at the point
|
||||
* of use. A component that captures {@code ref.get()} into a field at construction has opted out of
|
||||
* reload — which is sometimes right (see <em>deferred</em> below), but it must then be a deliberate
|
||||
* choice rather than an accident of where the field was initialised.
|
||||
*
|
||||
* <h2>Not every key can change under a running daemon</h2>
|
||||
* Keys fall into three classes, and the difference is about what already exists when the reload
|
||||
* happens — not about how important the key is.
|
||||
*
|
||||
* <ul>
|
||||
* <li><strong>Hot</strong> — re-read per use, so a reload takes effect on the next spawn:
|
||||
* {@code fleet:} (every role pool, {@code charters}, and {@code tabLabel}),
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Those
|
||||
* three are read through a supplier on {@code CompositePeerLauncher}, which is what makes
|
||||
* them hot — not the fact that they are config. <strong>This does NOT include
|
||||
* {@code fleet.leaders}</strong>: {@code Bridged.main} reads {@code cfg.fleet().leaders()}
|
||||
* once at startup to build the {@code LeadTabScanner} and the {@code LeadLauncher}, and
|
||||
* neither is reconstructed on reload — so a lead added, removed, or re-{@code tab}'d under
|
||||
* {@code fleet.leaders} needs a restart, the same as any deferred key below.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code guard:},
|
||||
* {@code worktreeRoot:}, adding or removing a profile (a new backend needs its own launcher,
|
||||
* which is constructed once), <em>and an existing profile's launch settings</em> —
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl} and the rest.
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} at construction and resolves
|
||||
* each spawn out of that copy, so those never reach a launch until the daemon restarts. A
|
||||
* reload logs these rather than pretending they applied.</li>
|
||||
* <li><strong>Cold</strong> — cannot change at all under a running daemon: {@code bind:},
|
||||
* {@code herdrSocket:}, {@code broker:} and {@code auth:}. The socket is bound, the broker
|
||||
* connection is open, and the auth mode decides who may reach the port that is already
|
||||
* listening.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>A cold change refuses the whole reload.</strong> Not the hot half applied and the cold
|
||||
* half warned about: that would leave the running daemon in a state matching no file on disk, which
|
||||
* is the worst thing a reload can do to an operator debugging one. Refusing keeps the invariant that
|
||||
* the live config is always some version of the file, and the message names the keys that must
|
||||
* change through a restart.
|
||||
*
|
||||
* <p>A reload that fails to parse or fails validation is also refused, and the previous config keeps
|
||||
* running. A config file being edited is normally read once mid-save; degrading a working daemon
|
||||
* because it caught a half-written file would be a bad trade.
|
||||
*/
|
||||
public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigRef.class);
|
||||
|
||||
/** Keys that cannot change under a running daemon — see the class doc. */
|
||||
private static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "broker", "auth");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<BridgedConfig> current;
|
||||
|
||||
public ConfigRef(Path path, BridgedConfig initial) {
|
||||
this.path = path;
|
||||
this.current = new AtomicReference<>(Objects.requireNonNull(initial, "initial config"));
|
||||
}
|
||||
|
||||
/** A fixed reference that never reloads — for tests and for wiring built from a config in code. */
|
||||
public static ConfigRef fixed(BridgedConfig cfg) {
|
||||
return new ConfigRef(null, cfg);
|
||||
}
|
||||
|
||||
/** The live configuration. Read this per use; do not cache it in a field. */
|
||||
@Override
|
||||
public BridgedConfig get() {
|
||||
return current.get();
|
||||
}
|
||||
|
||||
/** The file this ref reloads from, or {@code null} for a {@link #fixed} ref. */
|
||||
public Path path() {
|
||||
return path;
|
||||
}
|
||||
|
||||
/**
|
||||
* What a reload attempt did.
|
||||
*
|
||||
* @param applied true when the new config is now live
|
||||
* @param coldKeys cold keys whose value changed, which is why an unapplied reload was refused
|
||||
* @param deferred keys that changed and were accepted, but whose effect waits for a restart
|
||||
* @param error the parse or validation failure that refused the reload, else {@code null}
|
||||
*/
|
||||
public record Outcome(boolean applied, List<String> coldKeys, List<String> deferred,
|
||||
String error) {
|
||||
|
||||
public Outcome {
|
||||
coldKeys = List.copyOf(coldKeys);
|
||||
deferred = List.copyOf(deferred);
|
||||
}
|
||||
|
||||
static Outcome refusedCold(List<String> keys) {
|
||||
return new Outcome(false, keys, List.of(), null);
|
||||
}
|
||||
|
||||
static Outcome failed(String error) {
|
||||
return new Outcome(false, List.of(), List.of(), error);
|
||||
}
|
||||
|
||||
/** A one-line summary for the operator — the reason, not just the verdict. */
|
||||
public String summary() {
|
||||
if (error != null) {
|
||||
return "config reload refused — " + error;
|
||||
}
|
||||
if (!applied) {
|
||||
return "config reload refused — these keys cannot change under a running daemon: "
|
||||
+ String.join(", ", coldKeys) + ". Restart bridged to apply them.";
|
||||
}
|
||||
if (!deferred.isEmpty()) {
|
||||
return "config reloaded; these changes need a restart to take effect: "
|
||||
+ String.join(", ", deferred);
|
||||
}
|
||||
return "config reloaded";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-read the file, validate it, and swap it in when nothing cold changed.
|
||||
*
|
||||
* <p>Never throws: a reload is a best-effort operation on a daemon that is already serving, and
|
||||
* a bad edit must not take it down. Every failure path leaves the previous config live and is
|
||||
* reported through the returned {@link Outcome}.
|
||||
*/
|
||||
public Outcome reload() {
|
||||
if (path == null) {
|
||||
return Outcome.failed("this config was built in code and has no file to reload from");
|
||||
}
|
||||
BridgedConfig old = current.get();
|
||||
BridgedConfig fresh;
|
||||
try {
|
||||
fresh = BridgedConfig.load(path);
|
||||
// The same gate startup runs. A config that would have refused to boot must not be able
|
||||
// to slip in through a reload — that is how a daemon ends up in a state it could never
|
||||
// have started in, which is the hardest kind to debug.
|
||||
fresh.validateAuthExposure();
|
||||
fresh.validateLeadTabPrefixes();
|
||||
fresh.validateSubscriptionProfiles();
|
||||
fresh.validateCharters();
|
||||
fresh.validateMembers();
|
||||
} catch (RuntimeException e) {
|
||||
String msg = e.getMessage() == null ? e.toString() : e.getMessage();
|
||||
log.warn("config reload from {} refused, keeping the running config: {}", path, msg);
|
||||
return Outcome.failed(msg);
|
||||
}
|
||||
|
||||
List<String> cold = changedColdKeys(old, fresh);
|
||||
if (!cold.isEmpty()) {
|
||||
Outcome out = Outcome.refusedCold(cold);
|
||||
log.warn(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
List<String> deferred = changedDeferredKeys(old, fresh);
|
||||
current.set(fresh);
|
||||
Outcome out = new Outcome(true, List.of(), deferred, null);
|
||||
log.info(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Cold keys whose value differs between the running config and the candidate. */
|
||||
private static List<String> changedColdKeys(BridgedConfig old, BridgedConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.bind(), fresh.bind())) {
|
||||
changed.add("bind");
|
||||
}
|
||||
if (!Objects.equals(old.herdrSocket(), fresh.herdrSocket())) {
|
||||
changed.add("herdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.broker(), fresh.broker())) {
|
||||
changed.add("broker");
|
||||
}
|
||||
if (!Objects.equals(old.auth(), fresh.auth())) {
|
||||
changed.add("auth");
|
||||
}
|
||||
// Kept in step with COLD_KEYS so the doc and the code cannot drift apart silently.
|
||||
assert COLD_KEYS.containsAll(changed) : "a cold key was reported that COLD_KEYS omits";
|
||||
return changed;
|
||||
}
|
||||
|
||||
/** Changed keys that were accepted but whose effect waits for a restart. */
|
||||
private static List<String> changedDeferredKeys(BridgedConfig old, BridgedConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.lifecycle(), fresh.lifecycle())) {
|
||||
changed.add("lifecycle");
|
||||
}
|
||||
if (!Objects.equals(old.leadHeartbeat(), fresh.leadHeartbeat())) {
|
||||
changed.add("leadHeartbeat");
|
||||
}
|
||||
if (!Objects.equals(old.guard(), fresh.guard())) {
|
||||
changed.add("guard");
|
||||
}
|
||||
if (!Objects.equals(old.worktreeRoot(), fresh.worktreeRoot())) {
|
||||
changed.add("worktreeRoot");
|
||||
}
|
||||
if (!Objects.equals(old.spawnReadyTimeoutMs(), fresh.spawnReadyTimeoutMs())
|
||||
|| !Objects.equals(old.spawnReadyPollMs(), fresh.spawnReadyPollMs())) {
|
||||
changed.add("spawnReady*");
|
||||
}
|
||||
Map<String, BridgedConfig.Profile> before =
|
||||
old.profiles() == null ? Map.of() : old.profiles();
|
||||
Map<String, BridgedConfig.Profile> after =
|
||||
fresh.profiles() == null ? Map.of() : fresh.profiles();
|
||||
// Adding or removing a profile is deferred: a new backend needs its own launcher, and
|
||||
// launchers are built once at startup.
|
||||
if (!before.keySet().equals(after.keySet())) {
|
||||
Set<String> diff = new LinkedHashSet<>(before.keySet());
|
||||
diff.addAll(after.keySet());
|
||||
diff.removeIf(p -> before.containsKey(p) && after.containsKey(p));
|
||||
changed.add("profiles (added/removed: " + String.join(", ", diff) + ")");
|
||||
}
|
||||
// An EXISTING profile's launch settings are deferred too, and this is easy to get wrong:
|
||||
// `HerdrPeerLauncher` takes `Map.copyOf(profiles)` at construction and `spawn` resolves the
|
||||
// profile out of that snapshot, so a reloaded model/baseUrl/argv/env never reaches a launch.
|
||||
// Only weight and maxLoad are genuinely hot, because placement reads them through the
|
||||
// supplier on the composite rather than from the adapter's copy. Without this check a
|
||||
// changed model would report "config reloaded" and silently do nothing — the worst outcome
|
||||
// a reload can produce, because the operator has no reason to doubt it.
|
||||
List<String> relaunch = new ArrayList<>();
|
||||
before.forEach((name, was) -> {
|
||||
BridgedConfig.Profile now = after.get(name);
|
||||
if (now != null && !sameLaunchSettings(was, now)) {
|
||||
relaunch.add(name);
|
||||
}
|
||||
});
|
||||
if (!relaunch.isEmpty()) {
|
||||
changed.add("profiles." + String.join("/", relaunch) + " launch settings "
|
||||
+ "(model, baseUrl, argv, env, …) — the launcher holds a startup snapshot");
|
||||
}
|
||||
return changed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically. Compares every component
|
||||
* the launcher reads at spawn; {@code weight} and {@code maxLoad} are excluded because those are
|
||||
* read live by the placement policy and really do take effect on the next spawn.
|
||||
*/
|
||||
private static boolean sameLaunchSettings(BridgedConfig.Profile a, BridgedConfig.Profile b) {
|
||||
return Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
&& Objects.equals(a.model(), b.model())
|
||||
&& Objects.equals(a.configDir(), b.configDir())
|
||||
&& Objects.equals(a.tokenEnv(), b.tokenEnv())
|
||||
&& Objects.equals(a.argv(), b.argv())
|
||||
&& Objects.equals(a.placement(), b.placement())
|
||||
&& Objects.equals(a.workspace(), b.workspace())
|
||||
&& Objects.equals(a.tabLabel(), b.tabLabel())
|
||||
&& Objects.equals(a.mcpUrl(), b.mcpUrl())
|
||||
&& Objects.equals(a.cwd(), b.cwd())
|
||||
&& Objects.equals(a.parityOverlay(), b.parityOverlay())
|
||||
&& Objects.equals(a.gitTokenEnv(), b.gitTokenEnv())
|
||||
&& Objects.equals(a.gitHostEnv(), b.gitHostEnv())
|
||||
&& Objects.equals(a.kind(), b.kind())
|
||||
&& Objects.equals(a.env(), b.env())
|
||||
&& Objects.equals(a.subscription(), b.subscription());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
package dev.ltms.bridged.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Polls {@code bridged.yaml}'s modified time and asks {@link ConfigRef} to reload when it moves
|
||||
* (CB-559). Opt-in through {@code configReload.enabled}.
|
||||
*
|
||||
* <p><strong>Why polling and not a filesystem watch.</strong> {@code WatchService} on macOS has no
|
||||
* native backend — it falls back to polling internally anyway, at an interval this code does not
|
||||
* control — and editors save config files in ways that produce a different event mix per editor
|
||||
* (write-in-place, write-and-rename, write-temp-and-swap). A modified-time check treats all of them
|
||||
* the same and is a single {@code stat} per tick, which at a ten-second cadence costs nothing worth
|
||||
* measuring.
|
||||
*
|
||||
* <p><strong>A missing or unreadable file is not a reason to act.</strong> Many editors briefly
|
||||
* unlink the file during a save. Reloading on "it vanished" would mean reloading from a file that no
|
||||
* longer exists; reporting an error every tick would bury the log. So an unreadable file is skipped
|
||||
* silently and the next tick tries again — the running config stays live, which is the correct
|
||||
* outcome either way.
|
||||
*/
|
||||
public final class ConfigWatcher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigWatcher.class);
|
||||
|
||||
private final ConfigRef ref;
|
||||
private final long intervalSeconds;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
|
||||
private volatile long lastSeenMillis;
|
||||
|
||||
public ConfigWatcher(ConfigRef ref, long intervalSeconds) {
|
||||
this.ref = ref;
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.lastSeenMillis = modifiedMillis(ref.path());
|
||||
this.scheduler = Executors.newSingleThreadScheduledExecutor(r -> {
|
||||
Thread t = new Thread(r, "config-watcher");
|
||||
// A daemon thread: an operator's config watch must never be the reason the JVM refuses
|
||||
// to exit after everything else has shut down.
|
||||
t.setDaemon(true);
|
||||
return t;
|
||||
});
|
||||
}
|
||||
|
||||
/** Begin watching. A ref with no file (a fixed one) is a no-op rather than an error. */
|
||||
public void start() {
|
||||
if (ref.path() == null) {
|
||||
log.debug("config watch not started — this config has no file behind it");
|
||||
return;
|
||||
}
|
||||
scheduler.scheduleWithFixedDelay(this::tick, intervalSeconds, intervalSeconds,
|
||||
TimeUnit.SECONDS);
|
||||
log.info("config watch: {} re-read when it changes (every {}s)", ref.path(), intervalSeconds);
|
||||
}
|
||||
|
||||
/** One poll. Never throws — an exception here would silently cancel the schedule. */
|
||||
void tick() {
|
||||
try {
|
||||
long now = modifiedMillis(ref.path());
|
||||
if (now == 0 || now == lastSeenMillis) {
|
||||
return;
|
||||
}
|
||||
// Stamp BEFORE reloading. A file whose reload is refused (a bad edit, or a cold key)
|
||||
// must not be retried every tick — that would log the same refusal forever. The next
|
||||
// save moves the timestamp again and earns a fresh attempt.
|
||||
lastSeenMillis = now;
|
||||
ref.reload();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("config watch tick failed, still watching: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
private static long modifiedMillis(Path path) {
|
||||
if (path == null) {
|
||||
return 0;
|
||||
}
|
||||
try {
|
||||
return Files.getLastModifiedTime(path).toMillis();
|
||||
} catch (IOException e) {
|
||||
return 0; // mid-save, or gone: say nothing and try again next tick
|
||||
}
|
||||
}
|
||||
|
||||
/** Stop polling. Called from the daemon's ordered shutdown hook, alongside the other loops. */
|
||||
public void stop() {
|
||||
scheduler.shutdownNow();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,40 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
|
||||
/**
|
||||
* Pure classifier. Collection and repair are deliberately outside this package.
|
||||
* {@link HealthState#ERROR_ON_SCREEN} is not decided yet because it needs a bounded pane detection
|
||||
* read and an adapter-specific fatal signature; status facts alone must not guess it.
|
||||
*/
|
||||
public final class FleetHealth {
|
||||
private FleetHealth() { }
|
||||
|
||||
public static HealthDecision decide(HealthSnapshot s, HealthPrior prior, long nowNanos) {
|
||||
if (s.controlLinkDown()) return result(HealthState.CONTROL_LINK_DOWN, false);
|
||||
if (s.targetNotFound()) return result(HealthState.GONE, false);
|
||||
if (s.sessionState() == MemberSession.State.SPAWNING && !s.present() && s.readinessGraceElapsed()) {
|
||||
return result(HealthState.NEVER_READY, false);
|
||||
}
|
||||
if (s.orphanedDelegation()) return result(HealthState.DELEGATION_ORPHANED, false);
|
||||
boolean disagreement = s.sessionState() == MemberSession.State.BUSY && s.acceptedDelivery()
|
||||
&& (s.liveStatus() == AgentStatus.IDLE || s.liveStatus() == AgentStatus.DONE);
|
||||
if (disagreement && prior.busyButDone()) return result(HealthState.TURN_BOUNDARY_LOST, true);
|
||||
if (s.stalled()) return result(HealthState.STALL_SUSPECTED, disagreement);
|
||||
if (s.replyStranded()) return result(HealthState.REPLY_STRANDED, disagreement);
|
||||
if (s.queuedDelivery() || s.inboxMessage()) return result(HealthState.WORK_PENDING, disagreement);
|
||||
if (s.sessionState() == MemberSession.State.SPAWNING) return result(HealthState.STARTING, disagreement);
|
||||
if (s.acceptedDelivery() && s.liveStatus() == AgentStatus.BLOCKED) {
|
||||
return result(HealthState.BLOCKED_AMBIGUOUS, disagreement);
|
||||
}
|
||||
// An accepted delivery remains bridge work even when herdr is late, unknown, or has already
|
||||
// reported DONE once. It cannot be IDLE until the delegation has resolved.
|
||||
if (s.acceptedDelivery()) return result(HealthState.WORKING, disagreement);
|
||||
return result(HealthState.IDLE, disagreement);
|
||||
}
|
||||
|
||||
private static HealthDecision result(HealthState state, boolean disagreement) {
|
||||
return new HealthDecision(state, new HealthPrior(disagreement));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,151 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/** Slow whole-fleet evidence collection. It is deliberately separate from the delivery poller. */
|
||||
public final class FleetHealthMonitor {
|
||||
private static final Logger log = LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
|
||||
/** Bounded attempts to run {@link #failTarget} for one transition. Never retried tick-to-tick (CB-580). */
|
||||
static final int MAX_FAIL_TARGET_ATTEMPTS = 3;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final MessageService messages;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final long intervalSeconds;
|
||||
private final BiConsumer<String, String> failTarget;
|
||||
private final Map<String, HealthPrior> priors = new HashMap<>();
|
||||
private final Map<String, HealthState> states = new HashMap<>();
|
||||
|
||||
// These facts need the evidence publishers introduced by later M4 units. They are not negatives.
|
||||
private static final boolean NOT_YET_OBSERVED = false;
|
||||
|
||||
/**
|
||||
* @param failTarget CB-568's idempotent target-wide failure operation (e.g. {@code messages::abandon}),
|
||||
* invoked once when a member transitions into a terminal health state. Required —
|
||||
* there is deliberately no defaulting overload; a caller that does not want the
|
||||
* fail-tickets-on-terminal-health behavior must pass an explicit inert value (see
|
||||
* {@code TestTurnTokens.inert} / {@code BridgeMcp.CapacitySource.none()} for the pattern).
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, long intervalSeconds,
|
||||
BiConsumer<String, String> failTarget) {
|
||||
this.agents = agents;
|
||||
this.roster = roster;
|
||||
this.messages = messages;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.failTarget = Objects.requireNonNull(failTarget, "failTarget");
|
||||
}
|
||||
|
||||
/** Pure per-member decision seam. */
|
||||
static HealthDecision decide(HealthSnapshot snapshot, HealthPrior prior, long nowNanos) {
|
||||
return FleetHealth.decide(snapshot, prior, nowNanos);
|
||||
}
|
||||
|
||||
public void start() { scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS); }
|
||||
public void stop() { scheduler.shutdownNow(); }
|
||||
|
||||
// Package-private so tests can run one tick without waiting.
|
||||
void tick() {
|
||||
try {
|
||||
List<Agent> agentsNow = agents.list(); // Exactly one list call for this complete observation.
|
||||
List<MemberSession> rosterNow = roster.get(); // One in-memory roster snapshot for this tick.
|
||||
Map<String, Agent> live = new HashMap<>();
|
||||
for (Agent agent : agentsNow) live.put(agent.terminalId(), agent);
|
||||
HashSet<String> current = new HashSet<>();
|
||||
for (MemberSession session : rosterNow) {
|
||||
current.add(session.terminalId());
|
||||
Agent agent = live.get(session.terminalId());
|
||||
AgentStatus status = agent == null ? AgentStatus.UNKNOWN : agent.status();
|
||||
boolean accepted = messages.hasAcceptedDelivery(session.terminalId());
|
||||
HealthSnapshot snapshot = new HealthSnapshot(session.state(), status, accepted, NOT_YET_OBSERVED,
|
||||
messages.hasInboxMessage(session.terminalId()), agent != null, NOT_YET_OBSERVED,
|
||||
NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED);
|
||||
HealthDecision decision = decide(snapshot, priors.getOrDefault(session.terminalId(), HealthPrior.NONE),
|
||||
clock.getAsLong());
|
||||
priors.put(session.terminalId(), decision.prior());
|
||||
reportTransition(session.terminalId(), decision.state());
|
||||
}
|
||||
priors.keySet().retainAll(current);
|
||||
states.keySet().retainAll(current);
|
||||
} catch (Throwable error) {
|
||||
// A list failure is health evidence, and must never kill the monitor's only scheduler task.
|
||||
log.warn("fleet health collection failed; will retry next tick", error);
|
||||
} finally {
|
||||
if (!scheduler.isShutdown()) {
|
||||
scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void reportTransition(String target, HealthState next) {
|
||||
HealthState previous = states.put(target, next);
|
||||
if (previous == next) return;
|
||||
if (fault(next)) {
|
||||
log.warn("fleet health member={} state={} previous={}", target, next, previous);
|
||||
} else if (previous != null && fault(previous)) {
|
||||
log.info("fleet health member={} recovered state={} previous={}", target, next, previous);
|
||||
}
|
||||
// CB-580: a member entering GONE/NEVER_READY must not leave its waiting tickets pending
|
||||
// forever. Fire exactly once per transition — never on a tick where the state is unchanged,
|
||||
// which is what made the rejected commit call abandon() once per tick for as long as a
|
||||
// member stayed terminal.
|
||||
if (terminal(next)) {
|
||||
failTerminalTarget(target, next);
|
||||
}
|
||||
}
|
||||
|
||||
private void failTerminalTarget(String target, HealthState state) {
|
||||
String reason = "fleet health: member reached terminal state " + state.name();
|
||||
RuntimeException last = null;
|
||||
for (int attempt = 1; attempt <= MAX_FAIL_TARGET_ATTEMPTS; attempt++) {
|
||||
try {
|
||||
failTarget.accept(target, reason);
|
||||
return;
|
||||
} catch (RuntimeException error) {
|
||||
last = error;
|
||||
log.warn("fleet health: failTarget attempt {}/{} failed for member={} state={}",
|
||||
attempt, MAX_FAIL_TARGET_ATTEMPTS, target, state, error);
|
||||
}
|
||||
}
|
||||
log.warn("fleet health: giving up on failTarget for member={} state={} after {} attempts",
|
||||
target, state, MAX_FAIL_TARGET_ATTEMPTS, last);
|
||||
}
|
||||
|
||||
private static boolean terminal(HealthState state) {
|
||||
return state == HealthState.GONE || state == HealthState.NEVER_READY;
|
||||
}
|
||||
|
||||
private static boolean fault(HealthState state) {
|
||||
return switch (state) {
|
||||
case NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN -> true;
|
||||
default -> false;
|
||||
};
|
||||
}
|
||||
|
||||
public static String coverage(boolean enabled, boolean notificationConfigured) {
|
||||
return !enabled ? "off" : notificationConfigured ? "full" : "detection-only";
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,4 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
/** Classification plus the private fact that the next pure decision needs. */
|
||||
public record HealthDecision(HealthState state, HealthPrior prior) { }
|
||||
@@ -0,0 +1,6 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
/** Private cross-tick observation. It is deliberately not a reported health value. */
|
||||
public record HealthPrior(boolean busyButDone) {
|
||||
public static final HealthPrior NONE = new HealthPrior(false);
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
|
||||
/** Read-only facts from one fleet collection tick. */
|
||||
public record HealthSnapshot(MemberSession.State sessionState, AgentStatus liveStatus,
|
||||
boolean acceptedDelivery, boolean queuedDelivery, boolean inboxMessage,
|
||||
boolean present, boolean targetNotFound, boolean controlLinkDown,
|
||||
boolean readinessGraceElapsed, boolean orphanedDelegation,
|
||||
boolean replyStranded, boolean stalled) { }
|
||||
@@ -0,0 +1,8 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
/** Health classifications reported for a member. */
|
||||
public enum HealthState {
|
||||
STARTING, IDLE, WORKING, WORK_PENDING, BLOCKED_AMBIGUOUS,
|
||||
NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* Counts turns that ended via the completion fallback instead of {@code bridge_reply}.
|
||||
* MUTE is an observation by target and profile, not a classifier state and never suppresses faults.
|
||||
*/
|
||||
public final class MuteCounter {
|
||||
private final Map<String, Integer> byTarget = new ConcurrentHashMap<>();
|
||||
private final Map<String, Integer> byProfile = new ConcurrentHashMap<>();
|
||||
|
||||
/** Record only fallback completion; a structured reply does not make a member mute. */
|
||||
public void observe(String target, String profile, Rendezvous.Kind kind) {
|
||||
if (kind != Rendezvous.Kind.COMPLETION) return;
|
||||
byTarget.merge(target, 1, Integer::sum);
|
||||
byProfile.merge(profile, 1, Integer::sum);
|
||||
}
|
||||
|
||||
public int forTarget(String target) { return byTarget.getOrDefault(target, 0); }
|
||||
public int forProfile(String profile) { return byProfile.getOrDefault(profile, 0); }
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/** Fixed pane-probe limits. Pane content is never retained here. */
|
||||
public final class PaneBudget {
|
||||
public static final long COOLDOWN_NANOS = 60_000_000_000L;
|
||||
public static final int MAX_PER_TICK = 2;
|
||||
private final Map<String, Long> lastProbe = new HashMap<>();
|
||||
private int cursor;
|
||||
|
||||
public List<String> choose(List<String> candidates, long nowNanos, long configuredCooldownNanos) {
|
||||
long cooldown = Math.max(COOLDOWN_NANOS, configuredCooldownNanos);
|
||||
List<String> out = new ArrayList<>();
|
||||
for (int n = 0; n < candidates.size() && out.size() < MAX_PER_TICK; n++) {
|
||||
String target = candidates.get((cursor + n) % candidates.size());
|
||||
Long last = lastProbe.get(target);
|
||||
if (last == null || nowNanos - last >= cooldown) { out.add(target); lastProbe.put(target, nowNanos); }
|
||||
}
|
||||
if (!candidates.isEmpty()) cursor = (cursor + 1) % candidates.size();
|
||||
return List.copyOf(out);
|
||||
}
|
||||
}
|
||||
@@ -15,6 +15,10 @@ import com.fasterxml.jackson.databind.JsonNode;
|
||||
* @param agentType detected agent kind, e.g. {@code "claude"}, or {@code null} before herdr
|
||||
* has detected it (the start-time shape)
|
||||
* @param status current lifecycle state
|
||||
* @param name the unique label the agent was started with — for a bridge worker this is
|
||||
* {@code claude-<profile>-<nonce>-<seq>} (CB-117 keys orphan reaping on the
|
||||
* nonce); {@code null} for agents the bridge did not start, e.g. a user's own
|
||||
* Claude session
|
||||
*/
|
||||
public record Agent(
|
||||
String terminalId,
|
||||
@@ -23,7 +27,8 @@ public record Agent(
|
||||
String tabId,
|
||||
String sessionId,
|
||||
String agentType,
|
||||
AgentStatus status) {
|
||||
AgentStatus status,
|
||||
String name) {
|
||||
|
||||
/** Project a herdr {@code agent} node. Tolerates the start-time shape (no session yet). */
|
||||
public static Agent from(JsonNode a) {
|
||||
@@ -42,6 +47,7 @@ public record Agent(
|
||||
a.path("tab_id").asText(null),
|
||||
sessionId,
|
||||
type,
|
||||
AgentStatus.fromWire(a.path("agent_status").asText(null)));
|
||||
AgentStatus.fromWire(a.path("agent_status").asText(null)),
|
||||
a.path("name").asText(null));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6,12 +6,18 @@ import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* Domain layer over herdr's native {@code agent.*} namespace — the worker south side.
|
||||
* Chosen in the CB-102 spike over the pane + {@code send_text} fallback because
|
||||
* {@code agent.start} takes a first-class {@code env} map (clean, guard-checked
|
||||
* subscription injection) and herdr tracks each worker's Claude session UUID itself.
|
||||
* Chosen in the CB-102 spike over the pane + {@code send_text} fallback because herdr
|
||||
* tracks each worker's Claude session UUID itself.
|
||||
*
|
||||
* <p>Ported to herdr protocol 19 (herdr 0.8.0, CB-521): {@code agent.start} now starts a
|
||||
* <em>supported</em> agent ({@code kind}) into an <em>existing</em> pane, so the worker's
|
||||
* {@code env}/{@code cwd} move to pane creation ({@code tab.create}/{@code pane.split} — see
|
||||
* {@link WorkspaceControl}), and {@code agent.send} is replaced by {@code agent.prompt}
|
||||
* (which submits in one call) plus {@code agent.send_keys} for the raw Enter nudge.
|
||||
*
|
||||
* <p>Every method is one herdr call through the injected {@link HerdrClient}, so this
|
||||
* layer is unit-testable with a fake and contract-tested against a live daemon.
|
||||
@@ -20,42 +26,100 @@ public final class AgentControl {
|
||||
|
||||
private final HerdrClient herdr;
|
||||
|
||||
/**
|
||||
* Protocol 19 dropped {@code terminal_id} as an {@code agent.*} target — herdr now resolves
|
||||
* targets by pane id or agent name only, while the bridge keys every session on the terminal.
|
||||
* This caches the terminal→pane mapping (stable for a worker's lifetime) so callers keep
|
||||
* addressing agents by terminal; entries are invalidated on {@code agent_not_found}.
|
||||
*/
|
||||
private final Map<String, String> paneByTerminal = new ConcurrentHashMap<>();
|
||||
|
||||
public AgentControl(HerdrClient herdr) {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn an agent. {@code env} is applied to the process environment verbatim — this
|
||||
* is where a worker's {@code ANTHROPIC_BASE_URL} lives, and the ONLY place it should.
|
||||
*
|
||||
* @param name label/kind for herdr status detection (e.g. {@code "claude"})
|
||||
* @param argv launch command, e.g. {@code ["claude"]}
|
||||
* @param env process environment additions ({@code ANTHROPIC_BASE_URL}, token, …)
|
||||
*/
|
||||
public Agent start(String name, List<String> argv, Map<String, String> env) {
|
||||
return start(name, argv, env, null);
|
||||
/** One agent-targeted call, translating a terminal id to its pane id (retrying once fresh). */
|
||||
private JsonNode agentCall(String method, String target, Map<String, Object> extra) {
|
||||
String resolved = resolveTarget(target);
|
||||
try {
|
||||
return herdr.call(method, withTarget(resolved, extra));
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_not_found".equals(e.code()) || resolved.equals(target)) throw e;
|
||||
paneByTerminal.remove(target); // the cached pane went away — re-resolve once
|
||||
String fresh = resolveTarget(target);
|
||||
if (fresh.equals(resolved)) throw e;
|
||||
return herdr.call(method, withTarget(fresh, extra));
|
||||
}
|
||||
}
|
||||
|
||||
private static Map<String, Object> withTarget(String target, Map<String, Object> extra) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("target", target);
|
||||
m.putAll(extra);
|
||||
return m;
|
||||
}
|
||||
|
||||
/** The pane id behind a terminal-id target, or the target verbatim for pane ids / names. */
|
||||
private String resolveTarget(String target) {
|
||||
if (target == null || !target.startsWith("term_")) {
|
||||
return target;
|
||||
}
|
||||
String cached = paneByTerminal.get(target);
|
||||
if (cached != null) {
|
||||
return cached;
|
||||
}
|
||||
for (JsonNode a : herdr.call("agent.list").path("agents")) {
|
||||
if (target.equals(a.path("terminal_id").asText(null))) {
|
||||
String pane = a.path("pane_id").asText(null);
|
||||
if (pane != null) {
|
||||
paneByTerminal.put(target, pane);
|
||||
return pane;
|
||||
}
|
||||
}
|
||||
}
|
||||
return target; // unknown terminal — let herdr report it against the original target
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn an agent into a specific tab. With a non-null {@code tabId} the worker lands
|
||||
* in that tab (the placement policy's dedicated worker tab); with {@code null} herdr
|
||||
* splits the currently-focused tab (legacy pane placement).
|
||||
* Start an agent into {@code paneId}, which must be sitting at its interactive shell prompt —
|
||||
* the seed pane of a freshly-created worker tab, or a fresh split. The pane's shell already
|
||||
* carries the worker's env ({@code ANTHROPIC_BASE_URL}, token, …) and cwd from pane creation;
|
||||
* herdr resolves the executable from {@code kind} and waits (its default timeout) until the
|
||||
* agent is detected and ready for input.
|
||||
*
|
||||
* @param name unique label for this agent ({@code <kind>-<profile>-<nonce>-<seq>})
|
||||
* @param kind supported agent kind and canonical executable, e.g. {@code "claude"},
|
||||
* {@code "opencode"}
|
||||
* @param args extra arguments after the executable, e.g. {@code --mcp-config …}
|
||||
* @param paneId the pane to start the agent in
|
||||
*/
|
||||
public Agent start(String name, List<String> argv, Map<String, String> env, String tabId) {
|
||||
public Agent start(String name, String kind, List<String> args, String paneId) {
|
||||
Map<String, Object> params = new LinkedHashMap<>();
|
||||
params.put("name", name);
|
||||
params.put("argv", argv);
|
||||
params.put("env", env);
|
||||
if (tabId != null) {
|
||||
params.put("tab_id", tabId);
|
||||
}
|
||||
params.put("kind", kind);
|
||||
params.put("pane_id", paneId);
|
||||
params.put("args", args);
|
||||
JsonNode result = herdr.call("agent.start", params);
|
||||
return Agent.from(result.get("agent"));
|
||||
}
|
||||
|
||||
/** Deliver {@code text} to an agent (its next prompt input). */
|
||||
/**
|
||||
* Deliver {@code text} to an agent as its next prompt <em>and submit it</em> — herdr's
|
||||
* {@code agent.prompt} pastes the text (embedded newlines preserved verbatim) and submits it
|
||||
* in the same call, replacing the pre-protocol-19 two-event {@code agent.send} dance.
|
||||
*/
|
||||
public void send(String target, String text) {
|
||||
herdr.call("agent.send", Map.of("target", target, "text", text));
|
||||
agentCall("agent.prompt", target, Map.of("text", text));
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-send the submit keystroke (Enter) to {@code target}. The submit that accompanies a
|
||||
* delivery can race the paste — especially right as the worker's TUI becomes interactive —
|
||||
* leaving the text unsubmitted; the injector nudges it with this until the worker actually
|
||||
* picks up (CB-113).
|
||||
*/
|
||||
public void submit(String target) {
|
||||
agentCall("agent.send_keys", target, Map.of("keys", List.of("enter")));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -64,13 +128,13 @@ public final class AgentControl {
|
||||
* @param source one of {@code visible|recent|recent_unwrapped|detection}
|
||||
*/
|
||||
public String read(String target, String source) {
|
||||
JsonNode result = herdr.call("agent.read", Map.of("target", target, "source", source));
|
||||
JsonNode result = agentCall("agent.read", target, Map.of("source", source));
|
||||
return result.path("read").path("text").asText("");
|
||||
}
|
||||
|
||||
/** Current agent record (status, session UUID, pane). */
|
||||
public Agent get(String target) {
|
||||
return Agent.from(herdr.call("agent.get", Map.of("target", target)).get("agent"));
|
||||
return Agent.from(agentCall("agent.get", target, Map.of()).get("agent"));
|
||||
}
|
||||
|
||||
/** Just the lifecycle status — what the status-gated injector checks before send. */
|
||||
|
||||
@@ -2,28 +2,38 @@ package dev.ltms.bridged.herdr;
|
||||
|
||||
/**
|
||||
* A herdr agent's lifecycle state, as reported by {@code agent_status}. Drives the
|
||||
* status-gated injector: a worker is safe to inject into only when {@link #IDLE} or
|
||||
* {@link #BLOCKED}, never mid-turn ({@link #WORKING}).
|
||||
* status-gated injector: a worker is safe to inject into only when {@link #IDLE},
|
||||
* {@link #BLOCKED}, or {@link #DONE}, never mid-turn ({@link #WORKING}).
|
||||
*/
|
||||
public enum AgentStatus {
|
||||
IDLE,
|
||||
WORKING,
|
||||
BLOCKED,
|
||||
/**
|
||||
* The worker has finished its turn and is settled at an idle prompt. herdr emits this
|
||||
* (observed live alongside {@code idle}) as a turn-complete marker; earlier code mapped the
|
||||
* unrecognized string to {@link #UNKNOWN}, which both wedged delivery (not {@link #injectable})
|
||||
* and mis-fired the CB-109 stall-failure on a worker that had actually answered. It is a
|
||||
* turn-boundary equivalent to {@link #IDLE}: injectable, and a {@code working → done} edge is a
|
||||
* real completion.
|
||||
*/
|
||||
DONE,
|
||||
UNKNOWN;
|
||||
|
||||
/** Map herdr's wire string ({@code idle|working|blocked|unknown}) to the enum. */
|
||||
/** Map herdr's wire string ({@code idle|working|blocked|done|unknown}) to the enum. */
|
||||
public static AgentStatus fromWire(String s) {
|
||||
if (s == null) return UNKNOWN;
|
||||
return switch (s.toLowerCase()) {
|
||||
case "idle" -> IDLE;
|
||||
case "working" -> WORKING;
|
||||
case "blocked" -> BLOCKED;
|
||||
case "done" -> DONE;
|
||||
default -> UNKNOWN;
|
||||
};
|
||||
}
|
||||
|
||||
/** Whether {@code bridged} may inject a message now without stepping on a live turn. */
|
||||
public boolean injectable() {
|
||||
return this == IDLE || this == BLOCKED;
|
||||
return this == IDLE || this == BLOCKED || this == DONE;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,188 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* Discovers which panes host a lead by scanning herdr for tabs the operator labelled by convention
|
||||
* (CB-531), and hands {@link dev.ltms.bridged.auth.CallerResolver} the resulting
|
||||
* {@code terminal_id → lead name} map.
|
||||
*
|
||||
* <p><strong>Why scan at all.</strong> A lead is never spawned — a human opens a tab and starts an
|
||||
* agent in it — so the daemon cannot learn a lead's {@code terminal_id} at creation time the way it
|
||||
* does a worker's. CB-530 solved that by having the operator paste each id into {@code leaders:},
|
||||
* which works but costs a config edit and a daemon restart per lead, and the id is only obtainable
|
||||
* by first starting the session and asking it. Scanning closes that loop: label the tab, and the
|
||||
* pane is recognised on the next resolve.
|
||||
*
|
||||
* <p><strong>CB-579 — matched by name, not prefix.</strong> This used to strip one shared
|
||||
* {@code tabPrefix} off a label to derive the lead's name, and merged a config-supplied
|
||||
* {@code terminal_id} pin over every scan result so the pin could never expire. Both are gone: each
|
||||
* lead now configures its own exact {@code tab} label ({@code fleet.leaders.<name>.tab}), so this
|
||||
* class is handed a {@code tab → name} map up front and matches labels against it exactly
|
||||
* (case-insensitively). There is no merge step — a scan result is the whole answer. That is the
|
||||
* fix for the bug this replaces: a {@code terminal_id} pin surviving in config after the pane it
|
||||
* named was gone, so the daemon kept treating a dead session as a live lead forever.
|
||||
*
|
||||
* <p><strong>Direction of trust.</strong> The label names the lead; it never <em>grants</em>
|
||||
* anything a pane could take for itself. Three properties keep that honest:
|
||||
* <ol>
|
||||
* <li>Worker spaces are excluded wholesale ({@code excludedWorkspaceLabels}), so a worker cannot
|
||||
* become a lead by being placed — as a split, say — inside a matching tab.</li>
|
||||
* <li>A worker cannot rename a tab: {@code tab.rename} is reachable only through
|
||||
* {@link WorkspaceControl}, which no {@code bridge_*} tool exposes. The label is writable by
|
||||
* the human at the terminal and by nobody the bridge is defending against.</li>
|
||||
* <li>The label is a <em>name</em>, not a capability. What a pane may do is decided by
|
||||
* {@code Authz} against the role {@code CallerResolver} returns; a tab that calls itself a
|
||||
* lead still cannot act as one unless the daemon's own registry agrees.</li>
|
||||
* </ol>
|
||||
*
|
||||
* <p><strong>CB-558 — bridged now writes lead labels too.</strong> This class used to be able to say
|
||||
* that bridged never renames a lead tab, so the label was always the human's own writing and there
|
||||
* was no round-trip from the daemon's rename back into its next decision.
|
||||
* {@code dev.ltms.bridged.lead.LeadLauncher} ends that: an auto-launched lead is labelled by the
|
||||
* daemon and found again by this scan. The trust direction above is unaffected — bridged writing a
|
||||
* name for a lead it just started is not a pane promoting itself — but <em>staleness</em> becomes
|
||||
* real: a label left behind by a session that has since died would read as a live lead forever.
|
||||
* This scanner does not solve that (its job is naming, and a stale name costs nothing here); the
|
||||
* launcher does, by requiring a running agent in the tab before it counts the lead as live. If you
|
||||
* ever make a decision that <em>removes</em> something based on this map, add the same check.
|
||||
* The remaining hazard is an <em>operator</em> one — a worker {@code tabLabel} template that
|
||||
* happens to start with the same prefix would promote the whole fleet — and that is refused at
|
||||
* startup by {@code BridgedConfig.validateLeadTabPrefixes} rather than documented here.
|
||||
*
|
||||
* <p><strong>Caching.</strong> {@link #get()} is on the request path (every resolve), so the scan
|
||||
* is TTL-cached and a stale-but-valid map is preferred to a herdr round-trip. A failed scan keeps
|
||||
* the previous answer instead of emptying it — a herdr hiccup must not silently demote a live lead
|
||||
* mid-session.
|
||||
*/
|
||||
public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadTabScanner.class);
|
||||
|
||||
private final HerdrClient herdr;
|
||||
private final Map<String, String> tabToName;
|
||||
private final Set<String> excludedWorkspaceLabels;
|
||||
private final long ttlNanos;
|
||||
private final LongSupplier clock;
|
||||
|
||||
private Map<String, String> cached = Map.of();
|
||||
private long scannedAtNanos;
|
||||
private boolean everScanned;
|
||||
|
||||
/**
|
||||
* @param herdr the herdr client to query ({@code workspace.list},
|
||||
* {@code tab.list}, {@code pane.list} — all read-only)
|
||||
* @param tabToName every configured lead's exact tab label → its name
|
||||
* ({@code fleet.leaders.<name>.tab}), matched case-insensitively
|
||||
* @param excludedWorkspaceLabels workspaces never scanned — the configured worker spaces
|
||||
* @param ttlNanos how long a scan result is reused before the next one
|
||||
* @param clock nanosecond time source ({@code System::nanoTime} in production)
|
||||
*/
|
||||
public LeadTabScanner(HerdrClient herdr, Map<String, String> tabToName,
|
||||
Set<String> excludedWorkspaceLabels, long ttlNanos, LongSupplier clock) {
|
||||
this.herdr = herdr;
|
||||
this.tabToName = normalize(tabToName);
|
||||
this.excludedWorkspaceLabels = excludedWorkspaceLabels == null
|
||||
? Set.of() : Set.copyOf(excludedWorkspaceLabels);
|
||||
this.ttlNanos = ttlNanos;
|
||||
this.clock = clock;
|
||||
}
|
||||
|
||||
/** Keys stripped and lower-cased once, so every lookup is a plain map hit. */
|
||||
private static Map<String, String> normalize(Map<String, String> tabToName) {
|
||||
if (tabToName == null || tabToName.isEmpty()) {
|
||||
return Map.of();
|
||||
}
|
||||
Map<String, String> out = new LinkedHashMap<>();
|
||||
tabToName.forEach((tab, name) -> {
|
||||
if (tab != null && !tab.isBlank() && name != null && !name.isBlank()) {
|
||||
out.put(tab.strip().toLowerCase(Locale.ROOT), name);
|
||||
}
|
||||
});
|
||||
return Collections.unmodifiableMap(out);
|
||||
}
|
||||
|
||||
/**
|
||||
* The current {@code terminal_id → lead name} map, rescanning when the cache has expired.
|
||||
*
|
||||
* <p>Synchronized so a burst of concurrent calls produces one scan rather than one each; a scan
|
||||
* is a handful of RPCs over a Unix socket and is rate-limited to one per TTL.
|
||||
*/
|
||||
@Override
|
||||
public synchronized Map<String, String> get() {
|
||||
long now = clock.getAsLong();
|
||||
if (everScanned && now - scannedAtNanos < ttlNanos) {
|
||||
return cached;
|
||||
}
|
||||
// Stamp before scanning, not after: a herdr that is down must cost one attempt per TTL, not
|
||||
// one per request.
|
||||
scannedAtNanos = now;
|
||||
everScanned = true;
|
||||
try {
|
||||
Map<String, String> fresh = scan();
|
||||
if (!fresh.equals(cached)) {
|
||||
log.info("lead panes: {}", fresh);
|
||||
}
|
||||
cached = fresh;
|
||||
} catch (HerdrException e) {
|
||||
log.warn("lead-tab scan failed, keeping the {} lead(s) already known: {}",
|
||||
cached.size(), e.getMessage());
|
||||
}
|
||||
return cached;
|
||||
}
|
||||
|
||||
/** One full pass: labelled tabs → their panes → those panes' terminals. */
|
||||
private Map<String, String> scan() {
|
||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
||||
for (JsonNode w : herdr.call("workspace.list").path("workspaces")) {
|
||||
Workspace ws = Workspace.from(w);
|
||||
if (ws.workspaceId() == null || excludedWorkspaceLabels.contains(ws.label())) {
|
||||
continue;
|
||||
}
|
||||
for (JsonNode t : herdr.call("tab.list", Map.of("workspace_id", ws.workspaceId())).path("tabs")) {
|
||||
Tab tab = Tab.from(t);
|
||||
String name = leadNameOf(tab.label());
|
||||
if (name != null && tab.tabId() != null) {
|
||||
nameByTab.put(tab.tabId(), name);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Map<String, String> byTerminal = new LinkedHashMap<>();
|
||||
if (!nameByTab.isEmpty()) {
|
||||
// One pane.list for every tab: panes carry tab_id, so the join is local.
|
||||
for (JsonNode p : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String name = nameByTab.get(p.path("tab_id").asText(null));
|
||||
String terminal = p.path("terminal_id").asText(null);
|
||||
if (name != null && terminal != null && !terminal.isBlank()) {
|
||||
byTerminal.put(terminal, name);
|
||||
}
|
||||
}
|
||||
}
|
||||
return Collections.unmodifiableMap(byTerminal);
|
||||
}
|
||||
|
||||
/**
|
||||
* The lead name a tab label declares, or {@code null} if it names none of the configured leads.
|
||||
*
|
||||
* <p>Exact match (case-insensitive, ends stripped) against {@link #tabToName} — no prefix
|
||||
* stripping, so an operator's {@code "lead: something-else"} tab is never mistaken for a
|
||||
* configured lead just because it shares a prefix.
|
||||
*/
|
||||
private String leadNameOf(String label) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
return tabToName.get(label.strip().toLowerCase(Locale.ROOT));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* Resolves which herdr pane a process belongs to — the herdr half of connection-based MCP
|
||||
* identity (CB-105). Given the PID that opened an MCP connection, {@link #terminalForPid} finds
|
||||
* the agent pane whose process tree contains it, so {@code bridged} can tell <em>which worker</em>
|
||||
* is calling without the worker sending anything spoofable.
|
||||
*
|
||||
* <p>herdr owns the PID→pane truth: {@code pane.process_info} reports each pane's {@code shell_pid}
|
||||
* and foreground process PIDs. This scans agent panes; a spawn-time {@code pid→terminal} cache is
|
||||
* the obvious optimization once wired into {@code ClaudeCodeLauncher}.
|
||||
*/
|
||||
public final class PaneLocator {
|
||||
|
||||
private final HerdrClient herdr;
|
||||
|
||||
public PaneLocator(HerdrClient herdr) {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code terminal_id} of the agent pane whose process tree contains {@code pid}, or
|
||||
* {@code null} if no agent pane owns it (e.g. the caller is the primary, or off-host).
|
||||
*/
|
||||
public String terminalForPid(long pid) {
|
||||
if (pid <= 0) {
|
||||
return null;
|
||||
}
|
||||
for (JsonNode pane : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String paneId = pane.path("pane_id").asText(null);
|
||||
if (paneId != null && paneOwnsPid(paneId, pid)) {
|
||||
return pane.path("terminal_id").asText(null);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private boolean paneOwnsPid(String paneId, long pid) {
|
||||
JsonNode info;
|
||||
try {
|
||||
info = herdr.call("pane.process_info", Map.of("pane_id", paneId)).path("process_info");
|
||||
} catch (HerdrException e) {
|
||||
return false; // pane vanished mid-scan — just skip it
|
||||
}
|
||||
if (info.path("shell_pid").asLong(-1) == pid) {
|
||||
return true;
|
||||
}
|
||||
for (JsonNode p : info.path("foreground_processes")) {
|
||||
if (p.path("pid").asLong(-1) == pid) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -23,9 +23,9 @@ public record Tab(String tabId, String workspaceId, String label, int paneCount)
|
||||
}
|
||||
|
||||
/**
|
||||
* A freshly-created tab together with the placeholder shell pane herdr seeds it with.
|
||||
* The caller starts the worker into {@link #tab()} then closes {@link #rootPaneId()} so
|
||||
* only the worker pane remains.
|
||||
* A freshly-created tab together with the shell pane herdr seeds it with. Under protocol 19
|
||||
* the caller starts the worker <em>into</em> {@link #rootPaneId()} — the seed pane's shell
|
||||
* carries the worker's cwd and env from {@code tab.create}, and becomes the worker pane.
|
||||
*/
|
||||
public record Created(Tab tab, String rootPaneId) {
|
||||
/** Project a {@code tab_created} result ({@code {tab, root_pane}}). */
|
||||
|
||||
@@ -5,6 +5,7 @@ import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
@@ -40,6 +41,16 @@ public final class WorkspaceControl {
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Every tab in {@code workspaceId}, in herdr's order. */
|
||||
public List<Tab> listTabs(String workspaceId) {
|
||||
JsonNode result = herdr.call("tab.list", Map.of("workspace_id", workspaceId));
|
||||
List<Tab> out = new ArrayList<>();
|
||||
for (JsonNode t : result.path("tabs")) {
|
||||
out.add(Tab.from(t));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/** The first workspace with this exact label, if any. */
|
||||
public Optional<Workspace> findByLabel(String label) {
|
||||
return listWorkspaces().stream()
|
||||
@@ -65,13 +76,37 @@ public final class WorkspaceControl {
|
||||
}
|
||||
|
||||
/**
|
||||
* A brand-new tab in {@code workspaceId} plus the placeholder shell pane herdr seeds
|
||||
* it with. Start the worker into the tab, then {@code pane.close} the root pane so the
|
||||
* tab holds only the worker.
|
||||
* A brand-new tab in {@code workspaceId} plus the shell pane herdr seeds it with. Under
|
||||
* protocol 19 that seed pane is where the worker <em>starts</em>: its shell carries
|
||||
* {@code cwd} and {@code env} (the worker's {@code ANTHROPIC_BASE_URL} — this is the
|
||||
* subscription-injection seam now), and {@code agent.start} launches the agent into it.
|
||||
*/
|
||||
public Tab.Created createTab(String workspaceId) {
|
||||
JsonNode result = herdr.call("tab.create", Map.of("workspace_id", workspaceId));
|
||||
return Tab.Created.from(result);
|
||||
public Tab.Created createTab(String workspaceId, String cwd, Map<String, String> env) {
|
||||
Map<String, Object> params = new LinkedHashMap<>();
|
||||
params.put("workspace_id", workspaceId);
|
||||
if (cwd != null && !cwd.isBlank()) {
|
||||
params.put("cwd", cwd);
|
||||
}
|
||||
if (env != null && !env.isEmpty()) {
|
||||
params.put("env", env);
|
||||
}
|
||||
return Tab.Created.from(herdr.call("tab.create", params));
|
||||
}
|
||||
|
||||
/**
|
||||
* Split the currently-focused tab and return the new pane's id — the legacy pane placement's
|
||||
* seed pane, carrying {@code cwd} and {@code env} exactly as {@link #createTab}'s does.
|
||||
*/
|
||||
public String splitPane(String cwd, Map<String, String> env) {
|
||||
Map<String, Object> params = new LinkedHashMap<>();
|
||||
params.put("direction", "right");
|
||||
if (cwd != null && !cwd.isBlank()) {
|
||||
params.put("cwd", cwd);
|
||||
}
|
||||
if (env != null && !env.isEmpty()) {
|
||||
params.put("env", env);
|
||||
}
|
||||
return herdr.call("pane.split", params).path("pane").path("pane_id").asText(null);
|
||||
}
|
||||
|
||||
/** Give a worker's tab a human label in the tab bar. */
|
||||
|
||||
@@ -0,0 +1,362 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* The CB-106 completion fallback: bridges the {@link Injector}'s turn-completion signal to the
|
||||
* {@link Rendezvous} so a blocking {@code bridge_send} resolves even when the worker finishes its
|
||||
* task without ever calling {@code bridge_reply} — the common case for a real delegated coding task.
|
||||
*
|
||||
* <p>On a confirmed {@code working → idle} boundary it scrapes the worker's recent transcript and
|
||||
* resolves the awaiting send with that tail (a {@link Rendezvous.Kind#COMPLETION} resolution, so the
|
||||
* caller can tell a scrape from a structured reply). It scrapes only when a send is actually waiting
|
||||
* — a fleet worker's own turns, or a send that already timed out, cost no herdr traffic. An explicit
|
||||
* {@code bridge_reply} that raced in first wins; {@link Rendezvous#resolveCompletion} is then a no-op.
|
||||
*
|
||||
* <p>It also handles the CB-109 stall signal ({@link #onTurnFailed}): a worker that ran a turn then
|
||||
* wedged in an {@code unknown} state resolves the send as a failure (with the error screen as
|
||||
* context) rather than leaving it to time out.
|
||||
*
|
||||
* <p>The scrape is cleaned to the last {@code ⏺} assistant block (stripping TUI chrome) and guarded
|
||||
* against misattribution (CB-115): the pane content is baselined on delivery ({@link #onDelivered}),
|
||||
* and a completion whose scrape is unchanged from that baseline — the previous turn's wind-down
|
||||
* sampled as this turn's boundary on a rapid back-to-back send — is suppressed rather than resolving
|
||||
* the send with a stale answer.
|
||||
*
|
||||
* <p><strong>Waiter-specific resolution (CB-116).</strong> On delivery we also capture the exact
|
||||
* {@link Rendezvous} waiter this turn belongs to, and the completion/failure fallbacks resolve
|
||||
* <em>that</em> waiter — never "whatever send is waiting now". A completion fallback runs on a virtual
|
||||
* thread and can land after the worker's {@code bridge_reply} already resolved the turn and the
|
||||
* <em>next</em> send opened its own waiter on the same session; resolving the current waiter would
|
||||
* then deliver turn N's stale scrape as turn N+1's answer. Targeting the captured waiter makes a late
|
||||
* completion a harmless no-op (its waiter is already done) instead of a cross-turn stale reply.
|
||||
*
|
||||
* <p>Wired as the {@link Injector}'s {@link TurnListener}; the handlers hand off to a virtual thread
|
||||
* so the scrape's herdr round-trip never stalls the status poller. The captured waiter is read on the
|
||||
* poller thread (before any next-turn delivery can overwrite it) and passed into the virtual thread.
|
||||
*/
|
||||
public final class CompletionResolver implements TurnListener {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(CompletionResolver.class);
|
||||
|
||||
/**
|
||||
* herdr {@code agent.read} source for the completion scrape. {@code recent} returns the tail of
|
||||
* the transcript (the worker's last output), which is what a delegator wants when the worker
|
||||
* didn't structure a reply.
|
||||
*/
|
||||
static final String SCRAPE_SOURCE = "recent";
|
||||
|
||||
/** Cap the scraped tail so a long transcript can't return an unbounded blob. */
|
||||
static final int MAX_SCRAPE_CHARS = 4000;
|
||||
|
||||
private static final String CLIPPED_PANE_TAIL_MARKER =
|
||||
"[Pane tail clipped: member did not call bridge_reply.]";
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ExhaustedPatternLookup exhaustedPatterns;
|
||||
|
||||
/**
|
||||
* Per-target record of the turn currently in flight: the exact {@link Rendezvous} waiter its
|
||||
* delivering send opened, plus the assistant block present when it was delivered.
|
||||
*
|
||||
* <p>The {@code waiter} is what makes a late fallback safe (CB-116): we resolve it, not "whoever
|
||||
* is waiting now", so a completion that fires after the next send has opened its own waiter is a
|
||||
* no-op rather than a cross-turn stale reply. The {@code baseline} is the CB-115 staleness
|
||||
* reference: a completion scrape equal to it means the worker produced no new output (the previous
|
||||
* turn's wind-down sampled as this boundary), so it is suppressed. Overwritten on each delivery;
|
||||
* cleared when the turn resolves. Package-private so tests can capture and replay a specific turn.
|
||||
*/
|
||||
record InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline) {
|
||||
}
|
||||
|
||||
private final ConcurrentHashMap<String, InFlight> inFlight = new ConcurrentHashMap<>();
|
||||
|
||||
/**
|
||||
* @param exhaustedPatterns CB-578 stage A: per-target lookup for a profile's configured
|
||||
* usage-limit refusal pattern. Required — there is deliberately no
|
||||
* defaulting overload; a caller that does not want the classification
|
||||
* must pass an explicit inert value ({@link ExhaustedPatternLookup#none()}).
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns) {
|
||||
this.agents = agents;
|
||||
this.rendezvous = rendezvous;
|
||||
this.exhaustedPatterns = Objects.requireNonNull(exhaustedPatterns, "exhaustedPatterns");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target, TurnToken token) {
|
||||
// Capture the exact waiter this turn belongs to (CB-116) and snapshot the pane's pre-turn
|
||||
// content — what it shows *before* the just-delivered turn produces output — as the staleness
|
||||
// reference (CB-115). Done synchronously (like the delivering send itself) so both are in
|
||||
// place before this turn's completion can fire.
|
||||
captureBaseline(target, token);
|
||||
}
|
||||
|
||||
/** Capture the in-flight turn: its waiter and pre-turn baseline (the testable core of {@link #onDelivered}). */
|
||||
void captureBaseline(String target, TurnToken token) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = token.waiter();
|
||||
if (waiter == null) {
|
||||
inFlight.remove(target); // no send is waiting on this delivery — nothing to resolve later
|
||||
return;
|
||||
}
|
||||
String baseline;
|
||||
try {
|
||||
// Clip to the same cap resolve() applies to the tail (line ~134): the CB-115 misattribution
|
||||
// guard compares baseline.equals(tail), so both sides must be the same capped representation.
|
||||
// An unclipped baseline vs a clipped tail would never match for a >MAX_SCRAPE_CHARS block,
|
||||
// defeating the guard and letting a stale completion resolve the send.
|
||||
baseline = clip(lastAssistantBlock(agents.read(target, SCRAPE_SOURCE)));
|
||||
} catch (RuntimeException e) {
|
||||
baseline = null; // fail open: no baseline ⇒ no suppression
|
||||
log.debug("delivery baseline for {} failed: {}", target, e.getMessage());
|
||||
}
|
||||
inFlight.put(target, new InFlight(waiter, baseline));
|
||||
}
|
||||
|
||||
/** The turn currently baselined for {@code target}, or {@code null} — a test hook for the captureBaseline path. */
|
||||
InFlight inFlight(String target) {
|
||||
return inFlight.get(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
// Read the in-flight turn on the poller thread — before any next-turn delivery can overwrite
|
||||
// it — then off-load the scrape (a herdr round-trip we must not block polling on) to a vthread.
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("completion-" + target).start(() -> resolve(target, turn));
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the completed turn before adapter housekeeping can erase its rendered output. This is
|
||||
* intentionally synchronous and used only when a post-turn context reset is enabled; the normal
|
||||
* path remains off-loaded so polling is not blocked by a scrape.
|
||||
*/
|
||||
public void resolveBeforePostAction(String target) {
|
||||
resolve(target, inFlight.get(target));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn, null));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn, reason));
|
||||
}
|
||||
|
||||
/** Synchronous resolve (the unit-testable core of {@link #onTurnComplete}). */
|
||||
void resolve(String target, InFlight turn) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = turn == null ? null : turn.waiter();
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
// Nobody is blocked on THIS turn (it had no send, or its bridge_reply already won). Skip
|
||||
// the scrape; resolving the current waiter here would be the CB-116 cross-turn stale reply.
|
||||
inFlight.remove(target, turn);
|
||||
return;
|
||||
}
|
||||
String tail;
|
||||
String assistantBlock = null;
|
||||
int originalLength = 0;
|
||||
boolean clipped = false;
|
||||
boolean scrapeFailed = false;
|
||||
try {
|
||||
assistantBlock = lastAssistantBlock(agents.read(target, SCRAPE_SOURCE));
|
||||
originalLength = assistantBlock.strip().length();
|
||||
clipped = originalLength > MAX_SCRAPE_CHARS;
|
||||
tail = clip(assistantBlock);
|
||||
} catch (RuntimeException e) {
|
||||
// The worker finished but we couldn't read its screen — still resolve the send so the
|
||||
// caller unblocks; an empty tail beats hanging until the caller's timeout.
|
||||
log.warn("completion scrape for {} failed; resolving with an empty tail: {}",
|
||||
target, e.getMessage());
|
||||
tail = "";
|
||||
scrapeFailed = true;
|
||||
}
|
||||
// Misattribution guard (CB-115): if the scrape is byte-identical to the pane content at
|
||||
// delivery, this turn produced no new output — the boundary belongs to the previous turn's
|
||||
// wind-down (common on rapid back-to-back sends). Suppress rather than resolve the send with
|
||||
// a stale answer; the real bridge_reply (or a later genuine completion) resolves it instead.
|
||||
// A scrape that failed to read is exempt — an empty tail there is "couldn't see", not "no change".
|
||||
String baseline = turn.baseline();
|
||||
if (!scrapeFailed && baseline != null && baseline.equals(tail)) {
|
||||
log.debug("suppressing misattributed completion for {} (no output change since delivery)",
|
||||
target);
|
||||
return; // keep the in-flight record: a later genuine completion still needs it
|
||||
}
|
||||
// CB-578 stage A: a turn that ended with no bridge_reply AND whose scrape matches the
|
||||
// backend's configured usage-limit pattern is a refusal, not an answer. Classify it as
|
||||
// BACKEND_EXHAUSTED rather than handing the caller a scrape that reads like a real reply.
|
||||
if (!scrapeFailed) {
|
||||
Pattern exhausted = exhaustedPatterns.patternFor(target);
|
||||
String matchedLine = exhausted == null ? null : firstMatchingLine(assistantBlock, exhausted);
|
||||
if (matchedLine != null) {
|
||||
String reason = "backend exhausted (usage limit): " + matchedLine;
|
||||
if (rendezvous.resolveExhausted(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("completion for {} classified BACKEND_EXHAUSTED (no bridge_reply; scrape "
|
||||
+ "matched the profile's exhausted pattern): {}", target, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
String completion = clipped ? tail + "\n" + CLIPPED_PANE_TAIL_MARKER : tail;
|
||||
if (rendezvous.resolveCompletion(waiter, completion)) {
|
||||
inFlight.remove(target, turn);
|
||||
if (clipped) {
|
||||
log.warn("completion scrape for {} clipped from {} chars to the {} char cap; "
|
||||
+ "member did not call bridge_reply, so the pane tail is partial",
|
||||
target, originalLength, MAX_SCRAPE_CHARS);
|
||||
}
|
||||
log.debug("resolved send to {} via turn-completion fallback ({} chars scraped)",
|
||||
target, tail.length());
|
||||
}
|
||||
}
|
||||
|
||||
/** Synchronous fail (the unit-testable core of {@link #onTurnFailed}). */
|
||||
void fail(String target, InFlight turn) {
|
||||
fail(target, turn, null);
|
||||
}
|
||||
|
||||
/** Synchronous fail with an optional reason supplied by a dropped worker queue. */
|
||||
void fail(String target, InFlight turn, String explicitReason) {
|
||||
// A never-delivered readiness failure has no in-flight record but still has a blocked send;
|
||||
// fall back to the currently-registered waiter (unambiguous — that send never completed, so
|
||||
// no next turn exists to confuse it with).
|
||||
CompletableFuture<Rendezvous.Resolution> waiter =
|
||||
turn != null ? turn.waiter() : rendezvous.currentWaiter(target);
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
inFlight.remove(target, turn); // nobody blocked on this worker — nothing to fail
|
||||
return;
|
||||
}
|
||||
String reason = explicitReason;
|
||||
if (reason == null || reason.isBlank()) {
|
||||
try {
|
||||
reason = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
} catch (RuntimeException e) {
|
||||
reason = "";
|
||||
}
|
||||
if (reason.isBlank()) {
|
||||
// No screen to scrape — either the worker is stuck (CB-109) or gone (CB-110).
|
||||
reason = "worker did not reply; its turn ended in an unrecoverable state "
|
||||
+ "(worker unreachable or stuck)";
|
||||
}
|
||||
}
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The first line of {@code text} matching {@code pattern}, stripped — the CB-578 stage A
|
||||
* evidence carried in a {@code BACKEND_EXHAUSTED} reason so the operator sees the real refusal
|
||||
* text, never a generic label. {@code null} if no line matches.
|
||||
*/
|
||||
static String firstMatchingLine(String text, Pattern pattern) {
|
||||
if (text == null || text.isEmpty()) return null;
|
||||
for (String line : text.split("\n", -1)) {
|
||||
if (pattern.matcher(line).find()) {
|
||||
return line.strip();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for the CB-578 stage A exhausted-pattern classification, logged at startup
|
||||
* the way {@link dev.ltms.bridged.health.FleetHealthMonitor#coverage} is — so an operator can
|
||||
* see whether the classification is on, and for which profiles, without reading every
|
||||
* profile's config by hand.
|
||||
*
|
||||
* @param allProfiles every configured profile name
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry an exhausted pattern
|
||||
*/
|
||||
public static String coverage(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
if (configuredProfiles.isEmpty()) {
|
||||
return "off (no profile has an exhaustedPattern configured; profiles: " + sorted(allProfiles) + ")";
|
||||
}
|
||||
Set<String> unconfigured = new TreeSet<>(allProfiles);
|
||||
unconfigured.removeAll(configuredProfiles);
|
||||
return unconfigured.isEmpty()
|
||||
? "full (all profiles configured: " + sorted(allProfiles) + ")"
|
||||
: "partial (configured: " + sorted(configuredProfiles) + "; not configured: " + sorted(unconfigured) + ")";
|
||||
}
|
||||
|
||||
private static List<String> sorted(Set<String> names) {
|
||||
return names.stream().sorted().toList();
|
||||
}
|
||||
|
||||
private static String clip(String s) {
|
||||
if (s == null) return "";
|
||||
String trimmed = s.strip();
|
||||
return trimmed.length() <= MAX_SCRAPE_CHARS
|
||||
? trimmed
|
||||
: trimmed.substring(trimmed.length() - MAX_SCRAPE_CHARS);
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract the last assistant message from a raw Claude Code pane scrape (CB-115). Claude Code
|
||||
* prefixes each assistant turn with {@code ⏺}; the delegator wants that answer, not the TUI
|
||||
* chrome around it. Take everything from the final {@code ⏺} onward and stop at the <em>first</em>
|
||||
* hard interface boundary below it — the spinner/status line, input box, {@code ❯} prompt (which
|
||||
* may echo the <em>next</em> turn's text), footer, or tips/warnings. Stopping at the first
|
||||
* boundary (rather than trimming only trailing chrome) is what keeps a following turn's echoed
|
||||
* prompt out of this reply. Blank lines are not boundaries, so a multi-paragraph answer survives;
|
||||
* trailing blanks are trimmed at the end. With no {@code ⏺} marker (an unusual render) the whole
|
||||
* text is scanned the same way, so we never lose the reply.
|
||||
*
|
||||
* <p>Package-private and pure so it is unit-testable without herdr.
|
||||
*/
|
||||
static String lastAssistantBlock(String raw) {
|
||||
if (raw == null || raw.isBlank()) return "";
|
||||
int marker = raw.lastIndexOf('⏺');
|
||||
String block = marker >= 0 ? raw.substring(marker + 1) : raw;
|
||||
StringBuilder out = new StringBuilder();
|
||||
int kept = 0;
|
||||
for (String line : block.split("\n", -1)) {
|
||||
if (isBoundary(line)) break; // first TUI boundary ends the assistant message
|
||||
if (kept++ > 0) out.append('\n');
|
||||
out.append(line);
|
||||
}
|
||||
return out.toString().strip();
|
||||
}
|
||||
|
||||
/**
|
||||
* A hard TUI boundary line that marks the end of an assistant message and the start of interface
|
||||
* chrome (input box, prompt, spinner, footer, tips/warnings). Blank lines are <em>not</em>
|
||||
* boundaries — an answer may contain them — so they are kept and trimmed only if trailing.
|
||||
*/
|
||||
private static boolean isBoundary(String line) {
|
||||
String t = line.strip();
|
||||
if (t.isEmpty()) return false;
|
||||
// A horizontal rule / all box-drawing separators (e.g. "──────").
|
||||
if (t.chars().allMatch(c -> c == '─' || c == '—' || c == '━' || c == '═' || c == '-')) {
|
||||
return true;
|
||||
}
|
||||
String lower = t.toLowerCase();
|
||||
return t.startsWith("╭") || t.startsWith("│") || t.startsWith("╰") || t.startsWith("┌")
|
||||
|| t.startsWith("└") || t.startsWith("❯") || t.startsWith("⏵")
|
||||
|| t.startsWith("⎿") || t.startsWith("⚠")
|
||||
// Status/spinner lines Claude Code renders below a settled or in-flight turn,
|
||||
// e.g. "✻ Baked for 21s", "✶ Forming…".
|
||||
|| t.startsWith("✻") || t.startsWith("✳") || t.startsWith("✽") || t.startsWith("·")
|
||||
|| t.startsWith("●") || t.startsWith("◐") || t.startsWith("✢") || t.startsWith("✶")
|
||||
|| lower.contains("auto mode") || lower.contains("for shortcuts")
|
||||
|| lower.contains("esc to interrupt") || lower.contains("bypass permissions");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Per-target lookup for a profile's configured usage-limit refusal pattern (CB-578 stage A): how
|
||||
* {@link CompletionResolver} tells a backend that refused on a subscription usage limit — the
|
||||
* worker's pane stays healthy, but the account is exhausted — apart from a genuine completion.
|
||||
*
|
||||
* <p>The pattern is always profile config, never a vendor string in Java source: every backend
|
||||
* words its refusal differently, so a hardcoded sentence would only ever match one of them.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface ExhaustedPatternLookup {
|
||||
|
||||
/** The compiled pattern configured for {@code target}'s profile, or {@code null} if none. */
|
||||
Pattern patternFor(String target);
|
||||
|
||||
/**
|
||||
* Inert lookup — no profile has a pattern configured, so the classification never fires and
|
||||
* the completion fallback behaves exactly as before CB-578 stage A. The explicit stand-in a
|
||||
* caller (or a test not exercising this feature) passes instead of a defaulting overload.
|
||||
*/
|
||||
static ExhaustedPatternLookup none() {
|
||||
return target -> null;
|
||||
}
|
||||
}
|
||||
@@ -2,6 +2,7 @@ package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -12,6 +13,8 @@ import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
@@ -32,6 +35,14 @@ import java.util.stream.Collectors;
|
||||
* a transient {@code unknown} — counts as a real pickup, so a detection glitch can't prematurely
|
||||
* release the latch. Perfectly reliable turn boundaries require a herdr {@code events.subscribe}
|
||||
* stream; that is the intended upgrade and would replace only the sampling, not this queue.
|
||||
*
|
||||
* <p><strong>Turn completion (CB-106).</strong> Beyond delivery, the injector reports when a
|
||||
* delegated turn <em>finishes</em>: after a delivery is picked up (a real {@code working} sample),
|
||||
* the next injectable sample is a confirmed {@code working → idle} boundary and fires
|
||||
* {@link TurnListener#onTurnComplete}. Completion is only ever synthesized from a <em>confirmed</em>
|
||||
* turn — the pickup-grace path (a turn too fast to sample) unwedges the queue but does not fire
|
||||
* completion, since without a sampled {@code working} there is no trustworthy "the worker just
|
||||
* finished the task" signal to act on.
|
||||
*/
|
||||
public final class Injector {
|
||||
|
||||
@@ -45,22 +56,96 @@ public final class Injector {
|
||||
*/
|
||||
private static final int PICKUP_GRACE_POLLS = 8;
|
||||
|
||||
/**
|
||||
* How many consecutive {@code unknown} samples while a delegation is outstanding before we
|
||||
* declare it stalled and fire {@link TurnListener#onTurnFailed} (CB-109). A worker wedged in a
|
||||
* state herdr can't classify (e.g. an API-error screen) stays {@code unknown} indefinitely and
|
||||
* would otherwise never resolve; any {@code working}/{@code idle} sample resets the streak, so a
|
||||
* transient detection glitch cannot trip it. At the 250ms poll interval this is ~30s — far longer
|
||||
* than any real detection blip, and still vastly better than the async send's timeout.
|
||||
*/
|
||||
private static final int TURN_STALL_GRACE_POLLS = 120;
|
||||
|
||||
/**
|
||||
* How many consecutive injectable samples a queued-but-undelivered message may wait on the
|
||||
* {@link #ready} gate before we give up and fail it (CB-114). The gate holds a message out of a
|
||||
* worker's boot window (herdr reports {@code idle} while its Claude is still starting), but a
|
||||
* worker whose Claude crashes during boot — or never connects the bridge MCP — stays "idle and
|
||||
* not ready" forever: {@link #ready} never accepts it, the message is never delivered, and the
|
||||
* target would be polled indefinitely with its caller's future never completing. After this
|
||||
* grace the queued messages are failed and the target released. At the 250ms poll interval this
|
||||
* is ~60s — deliberately longer than {@link #TURN_STALL_GRACE_POLLS}, since a first boot (spawn
|
||||
* + model load + MCP connect) legitimately takes longer than an in-turn detection blip.
|
||||
*/
|
||||
private static final int READINESS_GRACE_POLLS = 240;
|
||||
|
||||
/**
|
||||
* The single source for the injector poll cadence — how often the {@link StatusPoller} drives
|
||||
* {@link #onStatus} at. {@code Bridged} passes this to every {@link StatusPoller} it constructs,
|
||||
* and this class reads it to state the readiness grace in seconds on the CB-562 expiry log
|
||||
* instead of hardcoding "60s". One constant, so a cadence change cannot silently desync a log
|
||||
* that claims a grace duration.
|
||||
*/
|
||||
public static final long POLL_INTERVAL_MILLIS = 250;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final TurnListener turnListener;
|
||||
private final Predicate<String> ready; // CB-113: a target is deliverable only when available
|
||||
private final Consumer<String> forget; // CB-114: clear a gone worker's readiness/presence
|
||||
private final ConcurrentHashMap<String, Target> targets = new ConcurrentHashMap<>();
|
||||
|
||||
/** Delivery only; completion signalling is a no-op and every target is treated as available. */
|
||||
public Injector(AgentControl agents) {
|
||||
this(agents, TurnListener.NOOP);
|
||||
}
|
||||
|
||||
/** Delivery plus turn-completion signalling (CB-106); every target is treated as available. */
|
||||
public Injector(AgentControl agents, TurnListener turnListener) {
|
||||
this(agents, turnListener, _ -> true);
|
||||
}
|
||||
|
||||
/**
|
||||
* Delivery, completion signalling (CB-106), and a readiness gate (CB-113): a message is delivered
|
||||
* only when {@code ready} accepts the target — i.e. the worker's Claude has connected the bridge
|
||||
* MCP. This holds the first delivery out of the worker's boot window, where herdr already reports
|
||||
* {@code idle} but the TUI would drop an injected paste.
|
||||
*/
|
||||
public Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready) {
|
||||
this(agents, turnListener, ready, _ -> {
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Delivery, completion signalling (CB-106), a readiness gate (CB-113), and readiness cleanup
|
||||
* (CB-114): {@code forget} is invoked with a target when its worker is gone — dropped
|
||||
* (pane crash) or timed out on the readiness gate — so its stale presence/readiness is cleared
|
||||
* and does not linger past the worker's life.
|
||||
*/
|
||||
public Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||
Consumer<String> forget) {
|
||||
this.agents = agents;
|
||||
this.turnListener = turnListener;
|
||||
this.ready = ready;
|
||||
this.forget = forget;
|
||||
}
|
||||
|
||||
/** A pending message and the future that completes when it has been delivered. */
|
||||
private record Pending(String text, CompletableFuture<Void> delivered) {
|
||||
private record Pending(String text, TurnToken token, CompletableFuture<Void> delivered) {
|
||||
}
|
||||
|
||||
/** Per-worker delivery state, guarded by its own monitor (single writer per worker). */
|
||||
private static final class Target {
|
||||
final Deque<Pending> queue = new ArrayDeque<>();
|
||||
boolean awaitingPickup; // sent a message, waiting for the worker to pick it up
|
||||
boolean awaitingPickup; // sent a message, waiting for the worker to pick it up
|
||||
int injectableSincePickup; // consecutive injectable samples while awaitingPickup
|
||||
boolean awaitingCompletion; // a delivered message's turn is not yet known-complete
|
||||
boolean turnObserved; // saw a real `working` sample since that delivery (turn ran)
|
||||
int unknownSinceTurn; // consecutive `unknown` samples while a delegation is outstanding (CB-109)
|
||||
int notReadySincePoll; // consecutive injectable samples a queued message waited on the readiness gate (CB-114)
|
||||
boolean postTurnPending; // completion observed; adapter housekeeping has not started yet
|
||||
boolean awaitingPostTurnPickup;
|
||||
boolean postTurnObserved;
|
||||
int injectableSincePostTurnPickup;
|
||||
|
||||
synchronized void add(Pending p) {
|
||||
queue.add(p);
|
||||
@@ -75,9 +160,9 @@ public final class Injector {
|
||||
* <p>Uses an atomic map update so a concurrent {@link #drop} cannot slip between "find the
|
||||
* target" and "queue the message" and orphan it in a target it just removed.
|
||||
*/
|
||||
public CompletableFuture<Void> enqueue(String target, String text) {
|
||||
public CompletableFuture<Void> enqueue(String target, String text, TurnToken token) {
|
||||
CompletableFuture<Void> delivered = new CompletableFuture<>();
|
||||
Pending p = new Pending(text, delivered);
|
||||
Pending p = new Pending(text, token, delivered);
|
||||
targets.compute(target, (_, existing) -> {
|
||||
Target t = (existing != null) ? existing : new Target();
|
||||
t.add(p); // synchronized on the Target monitor — atomic with a concurrent drop
|
||||
@@ -99,63 +184,196 @@ public final class Injector {
|
||||
|
||||
Pending sent = null;
|
||||
RuntimeException sendError = null;
|
||||
boolean turnCompleted = false;
|
||||
boolean turnFailed = false;
|
||||
boolean resubmit = false;
|
||||
boolean startPostTurn = false;
|
||||
List<Pending> notReady = null; // queued messages failed because the worker never became ready
|
||||
synchronized (t) {
|
||||
if (status == AgentStatus.WORKING) {
|
||||
// Definitive pickup: the worker is busy on our last message.
|
||||
if (t.awaitingPostTurnPickup) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
t.injectableSincePostTurnPickup = 0;
|
||||
t.postTurnObserved = true;
|
||||
}
|
||||
// Definitive pickup: the worker is busy on our last message, and (if a delivery is
|
||||
// outstanding) a real turn is now confirmed to be running.
|
||||
t.awaitingPickup = false;
|
||||
t.injectableSincePickup = 0;
|
||||
t.unknownSinceTurn = 0;
|
||||
t.notReadySincePoll = 0;
|
||||
if (t.awaitingCompletion) t.turnObserved = true;
|
||||
} else if (status.injectable()) { // IDLE or BLOCKED
|
||||
if (t.awaitingPickup && ++t.injectableSincePickup >= PICKUP_GRACE_POLLS) {
|
||||
// Pickup edge was never sampled (turn faster than the poll, or status lag).
|
||||
// The worker has plainly moved on — release the latch rather than wedge.
|
||||
t.awaitingPickup = false;
|
||||
t.injectableSincePickup = 0;
|
||||
t.unknownSinceTurn = 0;
|
||||
if (t.awaitingPostTurnPickup) {
|
||||
if (++t.injectableSincePostTurnPickup >= PICKUP_GRACE_POLLS) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
t.injectableSincePostTurnPickup = 0;
|
||||
} else {
|
||||
resubmit = true;
|
||||
}
|
||||
} else if (t.postTurnObserved) {
|
||||
t.postTurnObserved = false;
|
||||
}
|
||||
if (t.awaitingPickup) {
|
||||
if (++t.injectableSincePickup >= PICKUP_GRACE_POLLS) {
|
||||
// Pickup edge was never sampled (turn faster than the poll, or status lag).
|
||||
// Release the latch rather than wedge — and give up on synthesizing a
|
||||
// completion for this message, since without a confirmed `working` we cannot
|
||||
// trust that a task-processing turn actually ran.
|
||||
t.awaitingPickup = false;
|
||||
t.injectableSincePickup = 0;
|
||||
t.awaitingCompletion = false;
|
||||
t.turnObserved = false;
|
||||
} else {
|
||||
// Delivered but still idle → the worker hasn't picked it up; the submit
|
||||
// keystroke likely raced the paste (esp. right as the TUI became ready).
|
||||
// Re-nudge Enter (CB-113) until the worker starts (WORKING) or the grace ends.
|
||||
resubmit = true;
|
||||
}
|
||||
}
|
||||
if (!t.awaitingPickup) {
|
||||
Pending p = t.queue.peek();
|
||||
if (p != null) {
|
||||
try {
|
||||
agents.send(target, p.text());
|
||||
t.queue.poll();
|
||||
t.awaitingPickup = true;
|
||||
t.injectableSincePickup = 0;
|
||||
sent = p;
|
||||
} catch (RuntimeException e) {
|
||||
// Delivery failed at herdr; drop the poisoned message and surface it
|
||||
// rather than blocking the queue behind it.
|
||||
t.queue.poll();
|
||||
sent = p;
|
||||
sendError = e;
|
||||
// A confirmed turn (a `working` sample was seen) that has now returned to idle is
|
||||
// a trustworthy `working → idle` completion boundary.
|
||||
if (t.awaitingCompletion && t.turnObserved) {
|
||||
t.awaitingCompletion = false;
|
||||
t.turnObserved = false;
|
||||
turnCompleted = true;
|
||||
if (turnListener.hasPostTurnAction(target)) {
|
||||
t.postTurnPending = true;
|
||||
startPostTurn = true;
|
||||
}
|
||||
}
|
||||
// Deliver the next queued message only once the prior turn is fully settled, so a
|
||||
// completion is never confused with the pickup of the following message — and only
|
||||
// once the worker is available (CB-113), so we never paste into its boot window.
|
||||
if (!t.awaitingCompletion && !t.postTurnPending
|
||||
&& !t.awaitingPostTurnPickup && !t.postTurnObserved) {
|
||||
Pending p = t.queue.peek();
|
||||
if (p != null && ready.test(target)) {
|
||||
t.notReadySincePoll = 0;
|
||||
try {
|
||||
agents.send(target, p.text());
|
||||
t.queue.poll();
|
||||
t.awaitingPickup = true;
|
||||
t.awaitingCompletion = true;
|
||||
t.turnObserved = false;
|
||||
t.injectableSincePickup = 0;
|
||||
sent = p;
|
||||
} catch (RuntimeException e) {
|
||||
// Delivery failed at herdr; drop the poisoned message and surface it
|
||||
// rather than blocking the queue behind it.
|
||||
t.queue.poll();
|
||||
sent = p;
|
||||
sendError = e;
|
||||
}
|
||||
} else if (p != null && ++t.notReadySincePoll >= READINESS_GRACE_POLLS) {
|
||||
// The worker has been idle-but-not-ready for the whole grace: its Claude
|
||||
// never connected the bridge MCP (crashed during boot, or wedged on a
|
||||
// startup prompt). The readiness gate would hold this message forever, so
|
||||
// fail every queued message and release the target (CB-114) instead of
|
||||
// polling it indefinitely with the caller's future never completing.
|
||||
notReady = new ArrayList<>(t.queue);
|
||||
log.warn("readiness grace for {} expired after {} polls ({}s): target never "
|
||||
+ "became deliverable, so failing {} queued message(s) that never "
|
||||
+ "reached its pane",
|
||||
target, READINESS_GRACE_POLLS,
|
||||
READINESS_GRACE_POLLS * POLL_INTERVAL_MILLIS / 1000, notReady.size());
|
||||
t.queue.clear();
|
||||
t.notReadySincePoll = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// UNKNOWN (or any other non-injectable, non-working): not a safe window nor a
|
||||
// reliable pickup signal, so we never deliver or release the pickup latch here. But
|
||||
// an outstanding delegation whose worker has gone unresponsive — stuck in a state
|
||||
// herdr can't classify (CB-109) — will never yield a working→idle boundary. After a
|
||||
// sustained streak, declare it failed so the awaiting send resolves rather than
|
||||
// riding out the async timeout. (This also frees a delivery that wedged before it
|
||||
// was ever picked up, which the injectable-only pickup grace could never release.)
|
||||
if (t.awaitingCompletion && ++t.unknownSinceTurn >= TURN_STALL_GRACE_POLLS) {
|
||||
t.awaitingPickup = false;
|
||||
t.awaitingCompletion = false;
|
||||
t.turnObserved = false;
|
||||
t.unknownSinceTurn = 0;
|
||||
turnFailed = true;
|
||||
}
|
||||
}
|
||||
// UNKNOWN (and any other non-injectable, non-working): do nothing — neither a safe
|
||||
// window nor a reliable pickup signal, so we must not deliver or release the latch.
|
||||
|
||||
// Reclaim the entry once the worker is fully quiescent, so the map cannot grow without
|
||||
// bound across many short-lived workers.
|
||||
if (t.queue.isEmpty() && !t.awaitingPickup) {
|
||||
// Reclaim the entry once the worker is fully quiescent (nothing queued, no pickup or
|
||||
// completion awaited), so the map cannot grow without bound across short-lived workers.
|
||||
if (t.queue.isEmpty() && !t.awaitingPickup && !t.awaitingCompletion
|
||||
&& !t.postTurnPending && !t.awaitingPostTurnPickup && !t.postTurnObserved) {
|
||||
targets.remove(target, t);
|
||||
}
|
||||
}
|
||||
|
||||
// Fire listeners / herdr calls after releasing the monitor so nothing runs on the poller
|
||||
// thread while it holds the target lock.
|
||||
if (resubmit) {
|
||||
try {
|
||||
agents.submit(target); // nudge a raced Enter so the pending paste submits
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("resubmit to {} failed (will retry next poll): {}", target, e.getMessage());
|
||||
}
|
||||
}
|
||||
if (notReady != null) {
|
||||
// Worker never became available: forget its (never-set) readiness, unblock every queued
|
||||
// caller, and route the awaiting send through the same failure path as a stalled turn so
|
||||
// a blocking or async waiter resolves WORKER_FAILED rather than riding out the timeout.
|
||||
forget.accept(target);
|
||||
RuntimeException cause = new IllegalStateException(
|
||||
target + " never became available (no bridge MCP connection within the boot window)");
|
||||
for (Pending p : notReady) {
|
||||
p.delivered().completeExceptionally(cause);
|
||||
}
|
||||
turnListener.onTurnFailed(target);
|
||||
}
|
||||
if (turnCompleted) {
|
||||
if (startPostTurn) {
|
||||
boolean started = turnListener.onTurnCompleteWithPostAction(target);
|
||||
synchronized (t) {
|
||||
t.postTurnPending = false;
|
||||
if (started) {
|
||||
t.awaitingPostTurnPickup = true;
|
||||
t.injectableSincePostTurnPickup = 0;
|
||||
}
|
||||
if (t.queue.isEmpty() && !t.awaitingPostTurnPickup) {
|
||||
targets.remove(target, t);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
turnListener.onTurnComplete(target);
|
||||
}
|
||||
}
|
||||
if (turnFailed) {
|
||||
turnListener.onTurnFailed(target);
|
||||
}
|
||||
if (sent != null) {
|
||||
if (sendError != null) {
|
||||
log.warn("inject to {} failed, dropped message: {}", target, sendError.getMessage());
|
||||
sent.delivered().completeExceptionally(sendError);
|
||||
} else {
|
||||
// Baseline the pane's pre-turn content so a misattributed completion (no new output)
|
||||
// can't resolve this send with the previous turn's stale answer (CB-115).
|
||||
turnListener.onDelivered(target, sent.token());
|
||||
sent.delivered().complete(null);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Targets the poller must keep sampling: those with a queued message or an awaited pickup. */
|
||||
/**
|
||||
* Targets the poller must keep sampling: those with a queued message, an awaited pickup, or an
|
||||
* awaited turn completion (so the {@code working → idle} boundary is observed).
|
||||
*/
|
||||
public Set<String> activeTargets() {
|
||||
return targets.entrySet().stream()
|
||||
.filter(e -> {
|
||||
synchronized (e.getValue()) {
|
||||
return !e.getValue().queue.isEmpty() || e.getValue().awaitingPickup;
|
||||
Target t = e.getValue();
|
||||
return !t.queue.isEmpty() || t.awaitingPickup || t.awaitingCompletion
|
||||
|| t.postTurnPending || t.awaitingPostTurnPickup || t.postTurnObserved;
|
||||
}
|
||||
})
|
||||
.map(java.util.Map.Entry::getKey)
|
||||
@@ -163,20 +381,38 @@ public final class Injector {
|
||||
}
|
||||
|
||||
/**
|
||||
* Forget a target whose worker is gone, failing every still-queued message so awaiting
|
||||
* callers unblock instead of hanging forever. Futures are completed after the monitor is
|
||||
* released.
|
||||
* Forget a target whose worker is gone, failing every still-queued message so awaiting callers
|
||||
* unblock instead of hanging forever. If a message had already been <em>delivered</em> but its
|
||||
* turn was not yet resolved (CB-110 — the worker vanished mid-turn, e.g. its pane crashed), fire
|
||||
* {@link TurnListener#onTurnFailed} for it: a delivered message is no longer in the queue, so
|
||||
* failing queued waiters alone would leave that send's rendezvous hanging until the async
|
||||
* timeout. Futures and listeners are completed after the monitor is released.
|
||||
*/
|
||||
public void drop(String target, Throwable cause) {
|
||||
Target t = targets.remove(target);
|
||||
if (t == null) return;
|
||||
List<Pending> pending;
|
||||
boolean hadDeliveredTurn;
|
||||
synchronized (t) {
|
||||
pending = new ArrayList<>(t.queue);
|
||||
t.queue.clear();
|
||||
hadDeliveredTurn = t.awaitingCompletion;
|
||||
t.awaitingCompletion = false;
|
||||
t.awaitingPickup = false;
|
||||
t.postTurnPending = false;
|
||||
t.awaitingPostTurnPickup = false;
|
||||
t.postTurnObserved = false;
|
||||
}
|
||||
log.warn("{} is gone, dropping its queue: {} message(s) failed{}; cause: {}", target,
|
||||
pending.size(),
|
||||
hadDeliveredTurn ? " (including one turn already in flight whose completion was never confirmed)" : "",
|
||||
cause.getMessage());
|
||||
forget.accept(target); // the worker is gone — clear its readiness/presence too (CB-114)
|
||||
for (Pending p : pending) {
|
||||
p.delivered().completeExceptionally(cause);
|
||||
}
|
||||
// A queued send has no in-flight record, while a delivered turn does. CompletionResolver
|
||||
// handles both forms and resolves its waiter at most once.
|
||||
turnListener.onTurnFailed(target, cause.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* Tracks which workers are <em>available</em> — their Claude has booted and connected its MCP client
|
||||
* to the bridge (CB-113). This is the reliable readiness signal, unlike herdr's {@code agent_status},
|
||||
* which reports {@code idle} for a worker whose Claude is still booting. Delivering into that boot
|
||||
* window pastes into a not-yet-ready TUI (the text is lost) and wedges the worker's delivery state,
|
||||
* so the {@link Injector} holds the first delivery until the worker is present here.
|
||||
*
|
||||
* <p>Populated from the MCP transport: any MCP request whose connection resolves to a worker terminal
|
||||
* marks that worker present (its {@code initialize} is the first such contact). A worker that never
|
||||
* mounts the bridge MCP is never marked present — its sends stay queued until they time out, which is
|
||||
* correct (it could not have replied anyway).
|
||||
*/
|
||||
public class MemberPresence {
|
||||
|
||||
private final Set<String> present = ConcurrentHashMap.newKeySet();
|
||||
|
||||
/** Record that {@code terminal}'s worker has connected its MCP client (is available). */
|
||||
public void markPresent(String terminal) {
|
||||
if (terminal != null && !terminal.isBlank()) {
|
||||
present.add(terminal);
|
||||
}
|
||||
}
|
||||
|
||||
/** Whether {@code terminal}'s worker is available (has been seen on the bridge MCP). */
|
||||
public boolean isPresent(String terminal) {
|
||||
return present.contains(terminal);
|
||||
}
|
||||
|
||||
/** Forget a torn-down worker so its terminal id does not linger as "present". */
|
||||
public void forget(String terminal) {
|
||||
present.remove(terminal);
|
||||
}
|
||||
}
|
||||
@@ -23,13 +23,20 @@ public final class StatusPoller {
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Injector injector;
|
||||
private final StatusRefiner refiner;
|
||||
private final long intervalMillis;
|
||||
private volatile boolean running;
|
||||
private Thread thread;
|
||||
|
||||
public StatusPoller(AgentControl agents, Injector injector, long intervalMillis) {
|
||||
this(agents, injector, new StatusRefiner(agents), intervalMillis);
|
||||
}
|
||||
|
||||
public StatusPoller(AgentControl agents, Injector injector, StatusRefiner refiner,
|
||||
long intervalMillis) {
|
||||
this.agents = agents;
|
||||
this.injector = injector;
|
||||
this.refiner = refiner;
|
||||
this.intervalMillis = intervalMillis;
|
||||
}
|
||||
|
||||
@@ -47,7 +54,9 @@ public final class StatusPoller {
|
||||
for (String target : active) {
|
||||
if (!running) return;
|
||||
try {
|
||||
AgentStatus status = agents.status(target);
|
||||
// herdr's agent_status can misreport a settled worker as `unknown`; refine it
|
||||
// against the pane content before it drives delivery/completion (CB-115).
|
||||
AgentStatus status = refiner.refine(target, agents.status(target));
|
||||
injector.onStatus(target, status);
|
||||
} catch (HerdrException e) {
|
||||
// The worker's agent is gone — stop trying and unblock its waiters.
|
||||
|
||||
@@ -0,0 +1,89 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
/**
|
||||
* Refines an unreliable {@link AgentStatus#UNKNOWN} into a real state by reading the worker's
|
||||
* terminal content (CB-115).
|
||||
*
|
||||
* <p>Some workers' panes are misclassified by herdr as {@code unknown} even when the worker is
|
||||
* plainly settled at an idle prompt (empty {@code ❯}, "auto mode on" footer, a completed
|
||||
* {@code ⏺} answer above). Left as {@code UNKNOWN} that both <em>wedges delivery</em> — the
|
||||
* status-gated {@link Injector} only injects into an {@link AgentStatus#injectable} worker — and
|
||||
* <em>mis-fires the CB-109 stall failure</em> on a worker that has actually answered. herdr's
|
||||
* {@code agent_status} is a heuristic; the pane content is the ground truth.
|
||||
*
|
||||
* <p>The refinement only ever runs on a raw {@code UNKNOWN} sample (every other status is trusted
|
||||
* as-is), so a healthy worker adds zero extra herdr traffic; a persistently-{@code unknown} worker
|
||||
* costs one extra {@code agent.read} per poll while it has work outstanding. Classification is
|
||||
* deliberately conservative — it upgrades {@code UNKNOWN} to {@link AgentStatus#WORKING} or
|
||||
* {@link AgentStatus#IDLE} only on a clear signal, and leaves a genuinely unclassifiable screen
|
||||
* (e.g. a wedged error state) as {@code UNKNOWN} so the CB-109 stall path can still fail it.
|
||||
*/
|
||||
public final class StatusRefiner {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(StatusRefiner.class);
|
||||
|
||||
/**
|
||||
* herdr {@code agent.read} source used to inspect the pane. {@code detection} is the region
|
||||
* herdr itself uses for status detection (the prompt/footer tail), which is exactly what we
|
||||
* need to tell "idle at prompt" from "mid-turn".
|
||||
*/
|
||||
static final String PROBE_SOURCE = "detection";
|
||||
|
||||
private final AgentControl agents;
|
||||
|
||||
public StatusRefiner(AgentControl agents) {
|
||||
this.agents = agents;
|
||||
}
|
||||
|
||||
/**
|
||||
* Return a trustworthy status for {@code target}. Any non-{@code UNKNOWN} {@code raw} is returned
|
||||
* unchanged; an {@code UNKNOWN} triggers a pane read and content classification. A read failure
|
||||
* leaves it {@code UNKNOWN} (the safe default: no delivery, and the stall path still applies).
|
||||
*/
|
||||
public AgentStatus refine(String target, AgentStatus raw) {
|
||||
if (raw != AgentStatus.UNKNOWN) return raw;
|
||||
String pane;
|
||||
try {
|
||||
pane = agents.read(target, PROBE_SOURCE);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("status refine read for {} failed; leaving UNKNOWN: {}", target, e.getMessage());
|
||||
return AgentStatus.UNKNOWN;
|
||||
}
|
||||
AgentStatus refined = classify(pane);
|
||||
if (refined != AgentStatus.UNKNOWN) {
|
||||
log.debug("refined {} from UNKNOWN to {} via pane content", target, refined);
|
||||
}
|
||||
return refined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Classify a Claude Code TUI pane tail. Package-private and pure so it is unit-testable without
|
||||
* herdr.
|
||||
*
|
||||
* <ul>
|
||||
* <li>An active-generation marker ({@code esc to interrupt}) ⇒ {@link AgentStatus#WORKING} —
|
||||
* never inject here.</li>
|
||||
* <li>Otherwise, an interactive input prompt with no active-turn marker ({@code ❯}, the
|
||||
* {@code │ >} input box, or the idle {@code auto mode} / shortcuts footer) ⇒
|
||||
* {@link AgentStatus#IDLE} — settled and safe to inject / a completed turn.</li>
|
||||
* <li>Anything else (blank, or an unrecognizable screen) ⇒ {@link AgentStatus#UNKNOWN}.</li>
|
||||
* </ul>
|
||||
*/
|
||||
static AgentStatus classify(String pane) {
|
||||
if (pane == null || pane.isBlank()) return AgentStatus.UNKNOWN;
|
||||
String lower = pane.toLowerCase();
|
||||
// Claude Code shows "(esc to interrupt)" only while a turn is actively generating.
|
||||
if (lower.contains("esc to interrupt")) return AgentStatus.WORKING;
|
||||
// A settled, ready input prompt with no active-turn marker = idle-at-prompt.
|
||||
boolean readyPrompt = pane.contains("❯")
|
||||
|| pane.contains("│ >")
|
||||
|| lower.contains("auto mode on")
|
||||
|| lower.contains("? for shortcuts");
|
||||
return readyPrompt ? AgentStatus.IDLE : AgentStatus.UNKNOWN;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,68 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
|
||||
/**
|
||||
* Notified when a worker's delegated turn is observed to complete — a confirmed
|
||||
* {@code WORKING → IDLE} transition after a delivery. This is the CB-106 completion signal the
|
||||
* {@code CompletionResolver} uses to resolve a blocked send whose worker never called
|
||||
* {@code bridge_reply}. Kept as a seam so the {@link Injector} needs no dependency on the message
|
||||
* layer and stays unit-testable with a capturing fake.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface TurnListener {
|
||||
|
||||
/** A worker's delegated turn finished (worker returned to idle after visibly working). */
|
||||
void onTurnComplete(String target);
|
||||
|
||||
/**
|
||||
* Whether completion must pause ordinary delivery while adapter-specific housekeeping starts.
|
||||
* This is queried before the injector considers the next queued message, closing the same-tick
|
||||
* ordering gap. Implementations must not mutate state here.
|
||||
*/
|
||||
default boolean hasPostTurnAction(String target) {
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Complete the delegated turn and start its non-turn housekeeping operation.
|
||||
*
|
||||
* @return true when the injector must observe that operation settle before the next delivery
|
||||
*/
|
||||
default boolean onTurnCompleteWithPostAction(String target) {
|
||||
onTurnComplete(target);
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* A worker that visibly ran a delegated turn then wedged in a non-idle, non-working state
|
||||
* (CB-109) — e.g. an error screen herdr classifies as {@code unknown} — so no
|
||||
* {@code working → idle} completion boundary will ever arrive. A default no-op keeps this a
|
||||
* functional interface; the completion resolver overrides it to fail the awaiting send.
|
||||
*/
|
||||
default void onTurnFailed(String target) {
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #onTurnFailed(String)}, carrying the reason a worker became unreachable. The default
|
||||
* keeps existing listeners working while allowing the completion resolver to report a useful cause.
|
||||
*/
|
||||
default void onTurnFailed(String target, String reason) {
|
||||
onTurnFailed(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* A message was just delivered into {@code target}'s pane (CB-115). Fired so the completion
|
||||
* resolver can snapshot the pane's pre-turn content: a later {@link #onTurnComplete} whose
|
||||
* scrape is unchanged from this baseline is a <em>misattributed</em> boundary (e.g. the prior
|
||||
* turn's wind-down sampled as this turn's completion on rapid back-to-back sends) and must not
|
||||
* resolve the send with the previous turn's stale answer. A default no-op keeps the interface
|
||||
* functional for callers that don't scrape.
|
||||
*/
|
||||
default void onDelivered(String target, TurnToken token) {
|
||||
}
|
||||
|
||||
/** No-op default for callers that only need delivery, not completion signalling. */
|
||||
TurnListener NOOP = _ -> {
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,292 @@
|
||||
package dev.ltms.bridged.lead;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.Tab;
|
||||
import dev.ltms.bridged.herdr.Workspace;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* Starts the leads {@code fleet.leaders:} declares, when none is already running (CB-558).
|
||||
*
|
||||
* <p><strong>A lead is not a member, and this class exists to keep it that way.</strong> Every other
|
||||
* spawn path in the daemon goes through {@code HerdrPeerLauncher}, which does three things a lead
|
||||
* must never receive:
|
||||
* <ol>
|
||||
* <li>it appends the <em>reply charter</em> — "you are an off-subscription worker … end every turn
|
||||
* with {@code bridge_reply}". A lead is the orchestrator; telling it that it is a worker is
|
||||
* exactly backwards.</li>
|
||||
* <li>it registers the session with {@code SessionManager}, which subjects it to the idle reaper,
|
||||
* the context cap and the shutdown drain. An idle lead is the normal state of a lead, so the
|
||||
* reaper would kill the orchestrator for doing its job.</li>
|
||||
* <li>it can move a peer off the subscription via {@code ANTHROPIC_BASE_URL}. A lead stays on the
|
||||
* operator's subscription, always.</li>
|
||||
* </ol>
|
||||
* So this launcher talks to {@link AgentControl}/{@link WorkspaceControl} directly. The duplication
|
||||
* with the member launchers (argv and env assembly) is deliberate and is the cheaper half of the
|
||||
* trade: entangling the worker path with a not-a-worker case is how the three rules above get
|
||||
* broken later, quietly.
|
||||
*
|
||||
* <p><strong>Liveness, and the label round-trip.</strong> {@code LeadTabScanner} used to be able to
|
||||
* promise that bridged never writes a lead label. That is no longer true — an auto-launched lead is
|
||||
* labelled by this class, and the scanner reads that label back. The risk this opens is not
|
||||
* privilege escalation (the tab label never granted anything a pane could take for itself; see that
|
||||
* class's javadoc), but <em>staleness</em>: a label left behind by a crashed session would otherwise
|
||||
* read as a live lead forever, and the lead would never be relaunched. So a lead counts as live only
|
||||
* when herdr also reports a <em>running agent</em> in that tab — see {@link #liveLeads}. A labelled
|
||||
* tab with no agent in it is not a lead.
|
||||
*/
|
||||
public final class LeadLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadLauncher.class);
|
||||
|
||||
private final AgentControl agents;
|
||||
private final WorkspaceControl spaces;
|
||||
private final BridgedConfig cfg;
|
||||
|
||||
/**
|
||||
* @param agents herdr agent control (start, list)
|
||||
* @param spaces workspace / tab control (ensure, create, label, list)
|
||||
* @param cfg the loaded config — {@code fleet.leaders}, {@code profiles} and each lead's tab
|
||||
*/
|
||||
public LeadLauncher(AgentControl agents, WorkspaceControl spaces, BridgedConfig cfg) {
|
||||
this.agents = agents;
|
||||
this.spaces = spaces;
|
||||
this.cfg = cfg;
|
||||
}
|
||||
|
||||
/**
|
||||
* Bring every declared lead up to its {@code instances} count, and return how many were started.
|
||||
*
|
||||
* <p>Never throws: a daemon that cannot start a lead must still serve. herdr being unreachable,
|
||||
* a profile that does not exist, a failed {@code agent.start} — each is logged and skipped, and
|
||||
* the remaining leads are still attempted.
|
||||
*/
|
||||
public int ensureLeads() {
|
||||
Map<String, BridgedConfig.Leader> leaders = cfg.fleet().leaders();
|
||||
if (leaders.isEmpty()) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
Map<String, Integer> live;
|
||||
try {
|
||||
live = liveLeads(leaders);
|
||||
} catch (HerdrException e) {
|
||||
// Counting is the whole safety mechanism against double-spawning. If we cannot count, we
|
||||
// must not guess — spawning a second orchestrator is worse than starting none.
|
||||
log.warn("lead auto-launch skipped — cannot tell which leads are live: {}", e.getMessage());
|
||||
return 0;
|
||||
}
|
||||
|
||||
int started = 0;
|
||||
for (Map.Entry<String, BridgedConfig.Leader> e : leaders.entrySet()) {
|
||||
String name = e.getKey();
|
||||
BridgedConfig.Leader lead = e.getValue();
|
||||
int running = live.getOrDefault(name, 0);
|
||||
int wanted = lead.instances();
|
||||
|
||||
if (running >= wanted) {
|
||||
log.info("lead '{}': {} live, {} wanted — nothing to start", name, running, wanted);
|
||||
continue;
|
||||
}
|
||||
if (!lead.isCreatable()) {
|
||||
// A lead with a `tab:` but no `profile:` is recognise-only by design: the operator
|
||||
// opens it by hand. Say so once rather than looking like a silent failure.
|
||||
log.info("lead '{}' is not live, and names no profile — it can be recognised but not "
|
||||
+ "launched. Add `profile:` under fleet.leaders.{} to have bridged start it.",
|
||||
name, name);
|
||||
continue;
|
||||
}
|
||||
|
||||
BridgedConfig.Profile profile = cfg.profiles().get(lead.profile());
|
||||
if (profile == null) {
|
||||
log.warn("lead '{}' names profile '{}', which is not configured — not launching",
|
||||
name, lead.profile());
|
||||
continue;
|
||||
}
|
||||
|
||||
for (int i = running; i < wanted; i++) {
|
||||
if (launch(name, lead, profile)) {
|
||||
started++;
|
||||
}
|
||||
}
|
||||
}
|
||||
return started;
|
||||
}
|
||||
|
||||
/**
|
||||
* How many live leads exist per configured name: a running agent in a tab labelled with that
|
||||
* lead's exact {@code tab} (CB-579). Member workspaces are excluded, exactly as the scanner
|
||||
* excludes them: a member must not be counted as a lead because it happens to sit in a matching
|
||||
* tab.
|
||||
*
|
||||
* <p>There used to be a second path here — a running agent on the terminal a
|
||||
* {@code fleet.leaders.<name>.terminal} pin named, for a lead opened and pinned by hand. That
|
||||
* pin is retired: {@code tab} is now the only field identity depends on, and {@link Agent}
|
||||
* already carries {@link Agent#tabId()} directly, so a hand-opened lead is found the same way an
|
||||
* auto-launched one is — by labelling its tab to match.
|
||||
*/
|
||||
private Map<String, Integer> liveLeads(Map<String, BridgedConfig.Leader> leaders) {
|
||||
Set<String> memberSpaces = cfg.profiles().values().stream()
|
||||
.map(BridgedConfig.Profile::workspace)
|
||||
.filter(w -> w != null && !w.isBlank())
|
||||
.collect(Collectors.toSet());
|
||||
|
||||
// tabId → the lead name its label declares.
|
||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
||||
for (Workspace ws : spaces.listWorkspaces()) {
|
||||
if (ws.workspaceId() == null || memberSpaces.contains(ws.label())) {
|
||||
continue;
|
||||
}
|
||||
for (Tab tab : spaces.listTabs(ws.workspaceId())) {
|
||||
String declared = leadNameOf(tab.label(), leaders);
|
||||
if (declared != null && tab.tabId() != null) {
|
||||
nameByTab.put(tab.tabId(), declared);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Map<String, Integer> counts = new LinkedHashMap<>();
|
||||
for (Agent a : agents.list()) {
|
||||
String name = nameByTab.get(a.tabId());
|
||||
if (name != null) {
|
||||
counts.merge(name, 1, Integer::sum);
|
||||
}
|
||||
}
|
||||
return counts;
|
||||
}
|
||||
|
||||
/**
|
||||
* The configured lead a tab label names, or {@code null} for a label that names none.
|
||||
*
|
||||
* <p>Matched exactly (case-insensitively) against each lead's configured {@code tab}, so an
|
||||
* operator's {@code "lead: something-else"} tab is not mistaken for a configured lead.
|
||||
*/
|
||||
private String leadNameOf(String label, Map<String, BridgedConfig.Leader> leaders) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
String l = label.strip();
|
||||
for (Map.Entry<String, BridgedConfig.Leader> e : leaders.entrySet()) {
|
||||
String tab = e.getValue().tabLabel();
|
||||
if (tab != null && l.equalsIgnoreCase(tab.strip())) {
|
||||
return e.getKey();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Start one lead. Returns false (having logged) rather than throwing on any failure. */
|
||||
private boolean launch(String name, BridgedConfig.Leader lead, BridgedConfig.Profile profile) {
|
||||
String label = lead.tabLabel();
|
||||
String cwd = (lead.cwd() == null || lead.cwd().isBlank())
|
||||
? System.getProperty("user.dir") : lead.cwd();
|
||||
|
||||
Tab.Created tab = null;
|
||||
try {
|
||||
Workspace ws = spaces.ensureWorkspace(lead.workspace());
|
||||
tab = spaces.createTab(ws.workspaceId(), cwd, leadEnv(profile));
|
||||
if (tab.rootPaneId() == null) {
|
||||
throw new IllegalStateException("tab " + tab.tab().tabId()
|
||||
+ " came back with no seed pane — nowhere to start the lead");
|
||||
}
|
||||
// Same shape as the member launchers: herdr resolves the executable from `kind`, so
|
||||
// argv[0] (the configured launcher, e.g. `ccs`) is dropped and only the rest is passed.
|
||||
List<String> argv = leadArgv(profile);
|
||||
Agent started = agents.start("lead-" + name, herdrKind(profile),
|
||||
argv.isEmpty() ? argv : argv.subList(1, argv.size()), tab.rootPaneId());
|
||||
|
||||
// Label AFTER the start succeeds. A label written before would survive a failed start
|
||||
// and then read back as a live lead on the next boot, which is the exact staleness the
|
||||
// agent-liveness check exists to prevent — no need to create the case ourselves.
|
||||
spaces.renameTab(tab.tab().tabId(), label);
|
||||
|
||||
log.info("lead '{}' launched: profile={} tab={} pane={} terminal={} label='{}' cwd={}",
|
||||
name, profile.profile(), tab.tab().tabId(), started.paneId(),
|
||||
started.terminalId(), label, cwd);
|
||||
return true;
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("lead '{}' failed to launch on profile '{}': {}",
|
||||
name, profile.profile(), e.getMessage());
|
||||
if (tab != null && tab.tab() != null && tab.tab().tabId() != null) {
|
||||
try {
|
||||
spaces.closeTab(tab.tab().tabId());
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("could not close the orphaned lead tab {}: {}",
|
||||
tab.tab().tabId(), cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The herdr agent kind for this profile — the same value the matching member adapter passes, so
|
||||
* herdr resolves the same executable for a lead as it does for a member on that backend.
|
||||
*/
|
||||
private static String herdrKind(BridgedConfig.Profile profile) {
|
||||
return profile.isOpenCode() ? "opencode" : "claude";
|
||||
}
|
||||
|
||||
/**
|
||||
* The lead's argv: the profile's own command, the model pin, and the bridge MCP mount.
|
||||
*
|
||||
* <p>No {@code --append-system-prompt}. That flag carries the worker reply charter, and a lead
|
||||
* is not a worker — it reads its orchestration rules from the project's {@code CLAUDE.md} like
|
||||
* any other primary. This is the single most important difference from the member launchers;
|
||||
* do not "unify" it back.
|
||||
*/
|
||||
private List<String> leadArgv(BridgedConfig.Profile profile) {
|
||||
List<String> argv = new ArrayList<>(profile.argv());
|
||||
if (profile.hasMcp()) {
|
||||
argv.add("--mcp-config");
|
||||
argv.add("{\"mcpServers\":{\"bridge\":{\"type\":\"http\",\"url\":\""
|
||||
+ profile.mcpUrl() + "\"}}}");
|
||||
}
|
||||
// Appended last, for the same reason the member launcher does it (CB-533): the argv is
|
||||
// usually a wrapper such as `ccs <profile>`, which exports its own model family over
|
||||
// whatever it inherited, and --model outranks the environment.
|
||||
if (profile.model() != null && !profile.model().isBlank()) {
|
||||
argv.add("--model");
|
||||
argv.add(profile.model());
|
||||
}
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* The lead's environment: the profile's {@code env:} block, and nothing that could move it off
|
||||
* the subscription.
|
||||
*
|
||||
* <p>{@code ANTHROPIC_BASE_URL} and {@code ANTHROPIC_AUTH_TOKEN} are stripped unconditionally —
|
||||
* not defaulted, not guarded, stripped. A lead runs on the operator's subscription by
|
||||
* definition, so there is no configuration under which pointing it elsewhere is correct, and a
|
||||
* profile that carries them (a member profile reused as a lead's backend) must not leak them in.
|
||||
*/
|
||||
private Map<String, String> leadEnv(BridgedConfig.Profile profile) {
|
||||
Map<String, String> out = new LinkedHashMap<>();
|
||||
if (profile.env() != null) {
|
||||
out.putAll(profile.env());
|
||||
}
|
||||
out.remove("ANTHROPIC_BASE_URL");
|
||||
out.remove("ANTHROPIC_AUTH_TOKEN");
|
||||
if (profile.configDir() != null && !profile.configDir().isBlank()) {
|
||||
out.put("CLAUDE_CONFIG_DIR", profile.configDir());
|
||||
}
|
||||
// Deliberately no git token: a lead reviews and merges through the operator's own
|
||||
// credentials, and never needs the scoped write:repository token a member is granted.
|
||||
out.values().removeIf(Objects::isNull);
|
||||
return out;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
package dev.ltms.bridged.logging;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.turbo.TurboFilter;
|
||||
import ch.qos.logback.core.spi.FilterReply;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import org.slf4j.Marker;
|
||||
|
||||
/** Suppresses only the SDK warning for the normal MCP cancellation notification. */
|
||||
public final class McpCancelledNotificationFilter extends TurboFilter {
|
||||
|
||||
static final String LOGGER = "io.modelcontextprotocol.spec.McpStreamableServerSession";
|
||||
static final String UNHANDLED_NOTIFICATION = "No handler registered for notification method: {}";
|
||||
|
||||
@Override
|
||||
public FilterReply decide(Marker marker, Logger logger, Level level, String format, Object[] params,
|
||||
Throwable throwable) {
|
||||
if (level == Level.WARN
|
||||
&& LOGGER.equals(logger.getName())
|
||||
&& UNHANDLED_NOTIFICATION.equals(format)
|
||||
&& params != null
|
||||
&& params.length == 1
|
||||
&& params[0] instanceof McpSchema.JSONRPCNotification notification
|
||||
&& "notifications/cancelled".equals(notification.method())) {
|
||||
return FilterReply.DENY;
|
||||
}
|
||||
return FilterReply.NEUTRAL;
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,66 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
import dev.ltms.bridged.herdr.PaneLocator;
|
||||
|
||||
/**
|
||||
* Resolves <em>who is calling</em> an MCP tool from the connection alone — the anti-spoofing
|
||||
* identity model of the MCP contract. It ties the connection's loopback peer PID (from the OS)
|
||||
* to a herdr agent pane (from herdr), yielding the caller's worker {@code terminal_id}. A caller
|
||||
* that maps to no worker pane — the primary, or an off-host client — resolves to {@code null}.
|
||||
*
|
||||
* <p>Both sources are authoritative and unforgeable: the OS reports the real connecting PID, and
|
||||
* herdr owns the PID→pane mapping. A worker cannot claim to be another worker, nor the primary.
|
||||
* Single-host only (the herd shares the {@code bridged} host); the token path is the split-host
|
||||
* fallback.
|
||||
*/
|
||||
public final class ConnectionIdentity {
|
||||
|
||||
private final PaneLocator panes;
|
||||
private final PeerPidLookup pids;
|
||||
private final ProcessCwdLookup cwds;
|
||||
|
||||
/** Identity only (no cwd resolution — {@link #cwdForPid} returns {@code null}). */
|
||||
public ConnectionIdentity(PaneLocator panes, PeerPidLookup pids) {
|
||||
this(panes, pids, _ -> null);
|
||||
}
|
||||
|
||||
/** Identity plus cwd resolution (CB-112 — inherit the primary's directory on spawn). */
|
||||
public ConnectionIdentity(PaneLocator panes, PeerPidLookup pids, ProcessCwdLookup cwds) {
|
||||
this.panes = panes;
|
||||
this.pids = pids;
|
||||
this.cwds = cwds;
|
||||
}
|
||||
|
||||
/**
|
||||
* The caller resolved from the connection: its worker {@code terminal} (or {@code null} for the
|
||||
* primary / an off-host client) and its {@code pid} (or {@code -1} if not resolvable).
|
||||
*/
|
||||
public record Caller(String terminal, long pid) {
|
||||
}
|
||||
|
||||
/** Resolve the caller's terminal and PID from one peer-PID lookup. */
|
||||
public Caller resolve(String remoteAddr, int remotePort) {
|
||||
if (!isLoopback(remoteAddr)) {
|
||||
return new Caller(null, -1); // only same-host callers can be workers
|
||||
}
|
||||
long pid = pids.pidForLocalPort(remotePort);
|
||||
return new Caller(panes.terminalForPid(pid), pid);
|
||||
}
|
||||
|
||||
/**
|
||||
* The calling worker's {@code terminal_id}, or {@code null} if the caller is not a known
|
||||
* on-host worker (treat as the primary).
|
||||
*/
|
||||
public String callerTerminal(String remoteAddr, int remotePort) {
|
||||
return resolve(remoteAddr, remotePort).terminal();
|
||||
}
|
||||
|
||||
/** The working directory of {@code pid} (the primary's cwd on an MCP spawn), or {@code null}. */
|
||||
public String cwdForPid(long pid) {
|
||||
return pid > 0 ? cwds.cwdForPid(pid) : null;
|
||||
}
|
||||
|
||||
private static boolean isLoopback(String addr) {
|
||||
return "127.0.0.1".equals(addr) || "::1".equals(addr) || "0:0:0:0:0:0:0:1".equals(addr);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.InputStreamReader;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* {@link PeerPidLookup} via {@code lsof} (present on macOS and Linux). For a loopback TCP source
|
||||
* {@code port}, both the client and this daemon appear on that port — so we exclude our own PID
|
||||
* and take the other end, which is the calling process.
|
||||
*/
|
||||
public final class LsofPeerPidLookup implements PeerPidLookup {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LsofPeerPidLookup.class);
|
||||
|
||||
private final long selfPid = ProcessHandle.current().pid();
|
||||
|
||||
@Override
|
||||
public long pidForLocalPort(int port) {
|
||||
try {
|
||||
Process p = new ProcessBuilder("lsof", "-nP", "-FpP", "-iTCP:" + port)
|
||||
.redirectErrorStream(true).start();
|
||||
long found = -1;
|
||||
try (BufferedReader r = new BufferedReader(
|
||||
new InputStreamReader(p.getInputStream(), StandardCharsets.UTF_8))) {
|
||||
long current = -1;
|
||||
String line;
|
||||
// -Fp emits records: a 'p<pid>' line, then the ports/files under that pid.
|
||||
while ((line = r.readLine()) != null) {
|
||||
if (line.startsWith("p")) {
|
||||
current = parse(line.substring(1));
|
||||
} else if (current > 0 && current != selfPid) {
|
||||
found = current; // first process on this port that isn't us = the client
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!p.waitFor(2, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
}
|
||||
return found;
|
||||
} catch (Exception e) {
|
||||
log.debug("lsof peer-pid lookup for port {} failed: {}", port, e.getMessage());
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
|
||||
private static long parse(String s) {
|
||||
try {
|
||||
return Long.parseLong(s.trim());
|
||||
} catch (NumberFormatException e) {
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.InputStreamReader;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* {@link ProcessCwdLookup} via {@code lsof} (present on macOS and Linux): {@code lsof -a -p <pid>
|
||||
* -d cwd -Fn} prints the process's cwd on the {@code n…} line. Used to inherit the primary's
|
||||
* working directory for a spawned worker (CB-112).
|
||||
*/
|
||||
public final class LsofProcessCwdLookup implements ProcessCwdLookup {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LsofProcessCwdLookup.class);
|
||||
|
||||
@Override
|
||||
public String cwdForPid(long pid) {
|
||||
if (pid <= 0) {
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
Process p = new ProcessBuilder("lsof", "-a", "-p", Long.toString(pid), "-d", "cwd", "-Fn")
|
||||
.redirectErrorStream(true).start();
|
||||
String cwd = null;
|
||||
try (BufferedReader r = new BufferedReader(
|
||||
new InputStreamReader(p.getInputStream(), StandardCharsets.UTF_8))) {
|
||||
String line;
|
||||
while ((line = r.readLine()) != null) {
|
||||
if (line.startsWith("n")) { // 'n<path>' is the file-name field for the cwd fd
|
||||
cwd = line.substring(1);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!p.waitFor(2, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
}
|
||||
return (cwd == null || cwd.isBlank()) ? null : cwd;
|
||||
} catch (Exception e) {
|
||||
log.debug("lsof cwd lookup for pid {} failed: {}", pid, e.getMessage());
|
||||
return null;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
/**
|
||||
* Resolves the OS PID that owns a loopback TCP source port — the OS half of connection-based MCP
|
||||
* identity. Java exposes no peer PID for a TCP socket, so this shells out. Injectable so
|
||||
* {@link ConnectionIdentity} is testable without a real connection.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface PeerPidLookup {
|
||||
|
||||
/** The PID whose socket has local (source) {@code port} on loopback, or {@code -1} if unknown. */
|
||||
long pidForLocalPort(int port);
|
||||
}
|
||||
@@ -0,0 +1,117 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Optional;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
/**
|
||||
* Single-slot, thread-safe registry for the primary's herdr {@code terminal_id}.
|
||||
*
|
||||
* <p>Populated from the caller terminal of orchestration-side MCP tools
|
||||
* ({@code bridge_send}, {@code bridge_spawn}) — tools that only the primary calls.
|
||||
* A pinned terminal (from config) seeds the registry at construction and makes
|
||||
* subsequent {@link #record(String)} calls no-ops.
|
||||
*
|
||||
* <p>The push loop ({@code ReplyPushLoop}) uses {@link #isKnown()} to decide
|
||||
* whether active nudging is possible; an empty registry means the primary is
|
||||
* off-host or non-herdr and delivery falls back to pull.
|
||||
*/
|
||||
public final class PrimaryRegistry {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(PrimaryRegistry.class);
|
||||
|
||||
private final AtomicReference<String> terminal = new AtomicReference<>();
|
||||
private final boolean pinned;
|
||||
|
||||
/**
|
||||
* CB-532: worker terminal → the lead that delegated to it. The single slot above answers "who is
|
||||
* THE primary", a question with no correct answer once two leads orchestrate the same fleet:
|
||||
* whichever called {@code bridge_send} first captured every nudge, including nudges for the
|
||||
* other lead's delegations. This map answers the question that actually matters — "who is
|
||||
* waiting on THIS worker" — and is what lets {@code primary.terminal} be retired.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, String> leadByTarget = new ConcurrentHashMap<>();
|
||||
|
||||
/**
|
||||
* @param pinnedTerminal an optional pinned terminal from config ({@code null}/blank = unpinned)
|
||||
*/
|
||||
public PrimaryRegistry(String pinnedTerminal) {
|
||||
if (pinnedTerminal != null && !pinnedTerminal.isBlank()) {
|
||||
this.terminal.set(pinnedTerminal);
|
||||
this.pinned = true;
|
||||
log.info("primary terminal pinned: {}", pinnedTerminal);
|
||||
} else {
|
||||
this.pinned = false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Record a terminal_id. No-op when:
|
||||
* <ul>
|
||||
* <li>the registry is pinned (config override),
|
||||
* <li>{@code terminalId} is {@code null} or blank (non-herdr caller).
|
||||
* </ul>
|
||||
*/
|
||||
public void record(String terminalId) {
|
||||
if (pinned) return;
|
||||
if (terminalId == null || terminalId.isBlank()) return;
|
||||
String prev = terminal.getAndSet(terminalId);
|
||||
if (prev == null) {
|
||||
log.debug("primary terminal learned: {}", terminalId);
|
||||
} else if (!prev.equals(terminalId)) {
|
||||
log.debug("primary terminal changed: {} -> {}", prev, terminalId);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Record that {@code leadTerminal} owns the accepted delegation of worker {@code target} (CB-532).
|
||||
*
|
||||
* <p>Called from the {@code MessageService} accepted-delivery hook — only after a send has won
|
||||
* the session's send lock and queued delivery — where both halves are known (CB-548). It is
|
||||
* deliberately <em>not</em> called at {@code bridge_send} request time: a concurrent sender that
|
||||
* times out {@code BUSY} must not steal a live delegation's reply routing without ever owning
|
||||
* the turn. Last writer wins — if a second lead's later send is accepted, replies follow the
|
||||
* lead that most recently delegated to it, which is the one waiting.
|
||||
*/
|
||||
public void recordDelegation(String target, String leadTerminal) {
|
||||
if (target == null || target.isBlank() || leadTerminal == null || leadTerminal.isBlank()) {
|
||||
return;
|
||||
}
|
||||
leadByTarget.put(target, leadTerminal);
|
||||
}
|
||||
|
||||
/** Forget a worker's delegating lead — call on release, so a torn-down session leaks nothing. */
|
||||
public void forgetDelegation(String target) {
|
||||
if (target != null) {
|
||||
leadByTarget.remove(target);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Where a nudge about {@code target}'s reply should go: the lead that delegated to it, falling
|
||||
* back to the single known primary.
|
||||
*
|
||||
* <p>The fallback matters after a daemon restart, which loses the map while the durable inbox
|
||||
* keeps the reply. With one lead the fallback is unambiguous and correct. With several and no
|
||||
* recorded delegation there is no right answer, so this returns empty rather than guessing —
|
||||
* delivery degrades to pull, which is exactly what the durable inbox is for, instead of
|
||||
* interrupting the wrong lead with someone else's result.
|
||||
*/
|
||||
public Optional<String> nudgeTargetFor(String target) {
|
||||
String lead = target == null ? null : leadByTarget.get(target);
|
||||
return lead != null ? Optional.of(lead) : Optional.ofNullable(terminal.get());
|
||||
}
|
||||
|
||||
/** The known primary terminal, or empty if not yet learned (and not pinned). */
|
||||
public Optional<String> primaryTerminal() {
|
||||
return Optional.ofNullable(terminal.get());
|
||||
}
|
||||
|
||||
/** {@code true} once a terminal has been recorded (or was pinned at construction). */
|
||||
public boolean isKnown() {
|
||||
return terminal.get() != null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
/**
|
||||
* Resolves a process's current working directory from its PID — the OS half of CB-112's
|
||||
* "a worker inherits the primary's directory." Injectable so {@link ConnectionIdentity} stays
|
||||
* testable without shelling out.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface ProcessCwdLookup {
|
||||
|
||||
/** The working directory of {@code pid}, or {@code null} if unknown. */
|
||||
String cwdForPid(long pid);
|
||||
}
|
||||
@@ -0,0 +1,325 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The {@link HerdrPeerLauncher} adapter for <strong>Claude Code</strong> — the safe path from a
|
||||
* delegation request to a running off-subscription Claude.
|
||||
*
|
||||
* <p>Everything transport-related (tab/pane placement, the CB-306 spawn-readiness gate, unique
|
||||
* naming, CB-117 orphan reap, teardown, listing, cwd resolution) lives in the base. This class
|
||||
* supplies only the two Claude-specific seams:
|
||||
* <ul>
|
||||
* <li>the {@code claude} name prefix (so reap matches {@code claude-*} panes, never another
|
||||
* adapter's), and</li>
|
||||
* <li>{@link #buildLaunch}, which encodes the subscription boundary: build the worker env with
|
||||
* {@code ANTHROPIC_BASE_URL}, assert that host is on the allowlist <em>before</em> touching
|
||||
* herdr, and mount the bridge MCP + reply charter as inline launch flags. A worker's base_url
|
||||
* lives in the env map handed to herdr and nowhere else; {@code bridged}'s own environment is
|
||||
* never mutated, and nothing is written to the worker's profile.</li>
|
||||
* </ul>
|
||||
*/
|
||||
public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
/** Label prefix for this adapter's herdr agent names (drives naming + orphan reap). */
|
||||
private static final String NAME_PREFIX = "claude";
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ClaudeCodeLauncher.class);
|
||||
|
||||
private final SubscriptionGuard guard;
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so
|
||||
* existing deployments and tests keep the legacy non-blocking spawn semantics.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env, 0,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(300));
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with spawn-ready gate enabled. The gate polls {@code agents.status()}
|
||||
* until the pane reports an injectable state or {@code spawnReadyTimeoutMs} elapses.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, spawnReadyPollMs, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor carrying the fleet-wide tab-label template (CB-557). The template comes
|
||||
* from {@code fleet.tabLabel}, which a profile cannot know because it names the member's
|
||||
* <em>role</em>; a profile may still override it with its own {@code tabLabel}.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
fleet);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor. Every injectable collaborator is explicit so unit tests supply
|
||||
* fakes for the clock ({@code nowMillis}) and poll-loop wait ({@code sleeper}). The
|
||||
* {@code sleeper} is never called when the gate is disabled ({@code spawnReadyTimeoutMs == 0}).
|
||||
*
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param guard subscription-boundary guard (checked before spawning)
|
||||
* @param profiles configured worker profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (e.g. {@code () -> Thread.sleep(pollMs)}); it
|
||||
* already encodes the poll interval, so the 8th positional argument
|
||||
* (poll ms) is accepted for API symmetry but otherwise unused here
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the fleet-wide tab-label template (CB-557).
|
||||
*
|
||||
* @param fleet live fleet config, read once for each spawn
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>The spawn sequence encodes the subscription boundary: assert the profile's base_url is on
|
||||
* the allowlist <em>before</em> any herdr call, then build the worker env with
|
||||
* {@code ANTHROPIC_*}, the parity-neutral git-forge grant, and the bridge MCP + reply charter
|
||||
* mounted as inline launch flags. When the request carries session identity (CB-547a) it is
|
||||
* applied here — see {@link #applySessionIdentity}.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg, LaunchSpec spec) {
|
||||
// CB-539: a profile may deliberately opt into the subscription (subscription: true) when no
|
||||
// off-subscription endpoint exists for it — e.g. `sonnet` on `ccs`. That profile gets no
|
||||
// ANTHROPIC_BASE_URL/AUTH_TOKEN (there is nothing to point them at) and the guard's base_url
|
||||
// requirement is skipped FOR IT ONLY. Every other profile keeps the hard boundary below.
|
||||
boolean onSubscription = cfg.isSubscription();
|
||||
String baseUrl = cfg.baseUrl();
|
||||
|
||||
if (onSubscription) {
|
||||
// NO SILENT CONTRADICTION: subscription:true + a baseUrl state opposite intents; refuse
|
||||
// loudly rather than pick a winner.
|
||||
if (baseUrl != null && !baseUrl.isBlank()) {
|
||||
throw new IllegalStateException("profile '" + cfg.profile()
|
||||
+ "' sets both subscription: true and a baseUrl ('" + baseUrl + "') — the two "
|
||||
+ "are contradictory: a subscription profile must not point at an endpoint. "
|
||||
+ "Drop baseUrl, or drop subscription: true.");
|
||||
}
|
||||
// Visible without anyone going looking for it: this worker bills the subscription.
|
||||
log.warn("spawning profile '{}' on the Claude subscription (subscription: true) — this "
|
||||
+ "worker WILL bill the operator's subscription", cfg.profile());
|
||||
} else {
|
||||
guard.assertWorker(baseUrl); // hard stop before we spawn anything
|
||||
}
|
||||
|
||||
Map<String, String> workerEnv = baseEnv(cfg);
|
||||
if (onSubscription) {
|
||||
// CB-542 belt-and-braces: on the subscription path no guard vets these two keys, and the
|
||||
// profile's env: is layered in by baseEnv — so strip any that rode in there. Config load
|
||||
// already rejects this (loudly, naming the profile); this makes the boundary hold even
|
||||
// for a profile built in code that never passed through that validation.
|
||||
workerEnv.remove("ANTHROPIC_BASE_URL");
|
||||
workerEnv.remove("ANTHROPIC_AUTH_TOKEN");
|
||||
} else {
|
||||
workerEnv.put("ANTHROPIC_BASE_URL", baseUrl);
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", env.apply(cfg.tokenEnv()));
|
||||
}
|
||||
putIfPresent(workerEnv, "ANTHROPIC_MODEL", cfg.model());
|
||||
putIfPresent(workerEnv, "CLAUDE_CONFIG_DIR", cfg.configDir());
|
||||
applyGitToken(workerEnv, cfg);
|
||||
|
||||
// CB-547a: Claude Code can MINT its own session id, so bridged chooses it — a fresh spawn
|
||||
// gets a UUID we pass as --session-id and return from agentSessionId(), so the resume
|
||||
// handle is known BEFORE the agent has written anything; a resume spawn adopts its prior
|
||||
// id via -r and passes no --session-id (the two conflict). Both are injected before the
|
||||
// model flag so --model keeps outranking the operator's own argv.
|
||||
// mutableArgv: argvWithBridge may hand back the profile's own (immutable) List.of when it
|
||||
// has neither MCP nor a charter — session flags must be added into a list we own.
|
||||
List<String> argv = mutableArgv(argvWithBridge(cfg, spec.charter()));
|
||||
String agentSessionId = applySessionIdentity(argv, spec.sessionName(), spec.resumeSessionId());
|
||||
return new Launch(workerEnv, argvWithModel(argv, cfg), agentSessionId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Add the Claude-specific session-identity flags to {@code argv} and return the peer's OWN
|
||||
* session id — the resume handle. A resume request passes the prior id via {@code -r} and
|
||||
* returns that id; a fresh named session mints a new UUID, passes it via {@code --session-id},
|
||||
* and returns the mint. The bridge's logical name rides along as {@code -n} when present. When
|
||||
* <em>no</em> identity is requested (sessionName and resumeSessionId both blank) this adds
|
||||
* nothing and returns {@code null}, keeping the legacy no-identity launch byte-identical.
|
||||
*/
|
||||
private static String applySessionIdentity(List<String> argv, String sessionName, String resumeSessionId) {
|
||||
boolean resuming = resumeSessionId != null && !resumeSessionId.isBlank();
|
||||
boolean named = sessionName != null && !sessionName.isBlank();
|
||||
if (!resuming && !named) {
|
||||
return null; // no identity requested — keep the legacy launch byte-identical
|
||||
}
|
||||
if (named) {
|
||||
argv.add("-n");
|
||||
argv.add(sessionName);
|
||||
}
|
||||
if (resuming) {
|
||||
argv.add("-r");
|
||||
argv.add(resumeSessionId);
|
||||
return resumeSessionId;
|
||||
}
|
||||
String minted = UUID.randomUUID().toString();
|
||||
argv.add("--session-id");
|
||||
argv.add(minted);
|
||||
return minted;
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv, plus an inline {@code --mcp-config} when {@code worker.mcpUrl} is set and
|
||||
* {@code --append-system-prompt} when the base composed a charter. Neither touches the profile's
|
||||
* config; both are pure command-line flags. This inline-flag mount is Claude Code specific —
|
||||
* other adapters mount MCP and instructions their own way.
|
||||
*/
|
||||
private List<String> argvWithBridge(BridgedConfig.Profile cfg, String charter) {
|
||||
if (!cfg.hasMcp() && charter == null) {
|
||||
return cfg.argv();
|
||||
}
|
||||
List<String> argv = mutableArgv(cfg.argv());
|
||||
if (cfg.hasMcp()) {
|
||||
String mcpJson = "{\"mcpServers\":{\"bridge\":{\"type\":\"http\",\"url\":\""
|
||||
+ cfg.mcpUrl() + "\"}}}";
|
||||
argv.add("--mcp-config");
|
||||
argv.add(mcpJson);
|
||||
}
|
||||
if (charter != null) {
|
||||
argv.add("--append-system-prompt");
|
||||
argv.add(charter);
|
||||
}
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pin the model on the command line as well as in {@code ANTHROPIC_MODEL} (CB-533).
|
||||
*
|
||||
* <p>The env var alone is not a reliable pin for this adapter, because the argv is usually a
|
||||
* launcher rather than {@code claude} itself — {@code ["ccs", "<profile>"]} — and {@code ccs}
|
||||
* exports its profile's own model family ({@code ANTHROPIC_MODEL}, {@code DEFAULT_OPUS/SONNET/
|
||||
* HAIKU}, {@code CLAUDE_CODE_SUBAGENT_MODEL}) over whatever it inherited. A worker profile that
|
||||
* set {@code model:} therefore got silently overruled by its own launcher. Claude Code's
|
||||
* {@code --model} flag outranks the environment, and {@code ccs <profile> [claude-args...]}
|
||||
* passes trailing arguments through, so the flag survives the wrapper.
|
||||
*
|
||||
* <p>Appended last so it also outranks anything in the operator's own {@code argv}. Profiles
|
||||
* that deliberately leave {@code model:} unset (letting {@code ccs} own model selection, as
|
||||
* {@code gx10} does) are untouched — this adds nothing when there is nothing to add. This is
|
||||
* the {@code kind: claude} counterpart of the opencode adapter's {@code -m provider/model}.
|
||||
*/
|
||||
private static List<String> argvWithModel(List<String> argv, BridgedConfig.Profile cfg) {
|
||||
if (cfg.model() == null || cfg.model().isBlank()) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withModel = mutableArgv(argv);
|
||||
withModel.add("--model");
|
||||
withModel.add(cfg.model());
|
||||
return withModel;
|
||||
}
|
||||
|
||||
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
|
||||
|
||||
/** Spawn a worker for the default profile in the resolved default cwd. */
|
||||
public Agent spawn() {
|
||||
return spawnInternal(null, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile (null → default) in the resolved default cwd. */
|
||||
public Agent spawn(String profileName) {
|
||||
return spawnInternal(profileName, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile with an explicit requested/caller cwd (CB-112). */
|
||||
public Agent spawn(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd);
|
||||
}
|
||||
|
||||
// --- capabilities --------------------------------------------------------------------------
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
Set<Capability> caps = EnumSet.of(Capability.MID_TURN_ASK, Capability.WORKTREE,
|
||||
Capability.CONTEXT_RESET, Capability.ORPHAN_REAP,
|
||||
Capability.SESSION_NAME, Capability.SESSION_RESUME);
|
||||
if (hasGitTokenProfile()) {
|
||||
caps.add(Capability.SELF_PR);
|
||||
}
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
String target = agentTarget(id);
|
||||
if (target == null) {
|
||||
return false;
|
||||
}
|
||||
// This deliberately bypasses Injector: /clear is housekeeping, not a delegated turn.
|
||||
agents().send(target, "/clear");
|
||||
return true;
|
||||
}
|
||||
|
||||
/** Whether any configured profile opts into a git-forge token (required for {@link Capability#SELF_PR}). */
|
||||
private boolean hasGitTokenProfile() {
|
||||
return profileConfigs().stream().anyMatch(BridgedConfig.Profile::hasGitToken);
|
||||
}
|
||||
|
||||
// --- CB-117 reap predicate (Claude prefix), kept for direct unit testing -------------------
|
||||
|
||||
/**
|
||||
* Whether {@code name} is a Claude Code bridge worker started by a <em>different</em> process
|
||||
* than {@code currentNonce}. A thin {@code claude}-prefix binding of
|
||||
* {@link HerdrPeerLauncher#isForeignWorker(String, String, String)}.
|
||||
*/
|
||||
static boolean isForeignWorker(String name, String currentNonce) {
|
||||
return HerdrPeerLauncher.isForeignWorker(NAME_PREFIX, name, currentNonce);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,422 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import dev.ltms.bridged.placement.PlacementCandidate;
|
||||
import dev.ltms.bridged.placement.PlacementContext;
|
||||
import dev.ltms.bridged.placement.PlacementException;
|
||||
import dev.ltms.bridged.placement.PlacementPolicies;
|
||||
import dev.ltms.bridged.placement.PlacementPolicy;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.EnumSet;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The {@link PeerLauncher} the core actually holds when more than one adapter is configured — a thin
|
||||
* router in front of one {@link HerdrPeerLauncher} per peer {@code kind} (Claude Code, opencode, …).
|
||||
* It owns no transport of its own; it dispatches each SPI call to the delegate that owns the profile
|
||||
* involved, and fans the fleet-wide queries (list/reap/caps/profiles) across all delegates.
|
||||
*
|
||||
* <p>Routing rules:
|
||||
* <ul>
|
||||
* <li><strong>By profile</strong> — {@link #spawn}, {@link #effectiveCwd}, {@link #parityOverlay}
|
||||
* resolve the profile (a null/blank name → the global {@link #defaultProfile}) and delegate to
|
||||
* the single adapter that declares it. Profiles partition cleanly across adapters: the
|
||||
* constructor rejects a name claimed by two.</li>
|
||||
* <li><strong>By pane id</strong> — {@link #stop} routes to the adapter that spawned that pane
|
||||
* (recorded at spawn time). A pane the composite never spawned (only real for a caller that
|
||||
* hand-rolls an id) falls back to the first delegate; teardown is pane-id addressed and
|
||||
* tab cleanup is single-occupant guarded, so it is safe either way.</li>
|
||||
* <li><strong>Fleet-wide</strong> — {@link #reapOrphanWorkers} and {@link #capabilities} fan out
|
||||
* and combine. {@link #list} is deduplicated by pane id because every herdr-backed delegate
|
||||
* shares one herdr connection and so reports the same global agent set.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>CB-518: an unqualified spawn is routed through a {@link PlacementPolicy}. The default
|
||||
* {@code fixed} policy reproduces the historical default-profile behaviour; {@code weighted} uses
|
||||
* smooth weighted round-robin with {@code maxLoad} gating. If a chosen profile fails with
|
||||
* {@link PeerUnreachableException}, the composite advances to the next available candidate and
|
||||
* retries, bounded by the number of candidates.
|
||||
*/
|
||||
public final class CompositePeerLauncher implements PeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(CompositePeerLauncher.class);
|
||||
|
||||
private final List<HerdrPeerLauncher> delegates;
|
||||
private final Map<String, HerdrPeerLauncher> byProfile;
|
||||
private final String defaultProfile;
|
||||
|
||||
/** paneId → the delegate that spawned it, so {@link #stop} tears down through the right adapter. */
|
||||
private final Map<String, HerdrPeerLauncher> spawnedBy = new ConcurrentHashMap<>();
|
||||
|
||||
private final Function<String, Integer> liveCount;
|
||||
|
||||
/**
|
||||
* CB-559: the placement inputs are read <em>per spawn</em>, not captured at construction, so a
|
||||
* config reload changes where the next member lands without a restart. These are the hot keys —
|
||||
* role pools, an existing profile's weight/maxLoad, and the placement policy. What cannot change
|
||||
* this way is the set of adapters ({@link #byProfile}), because a new backend needs a launcher
|
||||
* and launchers are built once; {@code ConfigRef} classifies that as deferred and says so.
|
||||
*/
|
||||
private final Supplier<Map<String, BridgedConfig.Profile>> profileConfigs;
|
||||
private final Supplier<PlacementPolicy> placementPolicy;
|
||||
|
||||
/**
|
||||
* CB-557: the role pools an unqualified spawn draws its candidates from. A supplier that yields
|
||||
* {@code null}, and an empty pool for a role, both fall back to every configured profile — the
|
||||
* pre-CB-557 behaviour.
|
||||
*/
|
||||
private final Supplier<BridgedConfig.Fleet> fleet;
|
||||
|
||||
/**
|
||||
* Backward-compatible constructor: fixed placement, no live-counting. Use this for tests and
|
||||
* simple wiring; it preserves the pre-CB-518 behaviour exactly.
|
||||
*
|
||||
* @param delegates one adapter per configured peer kind; must be non-empty and declare
|
||||
* disjoint profile-name sets
|
||||
* @param defaultProfile the profile a no-argument spawn resolves to (may be null)
|
||||
* @throws IllegalArgumentException if {@code delegates} is empty or two adapters claim one profile
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates, String defaultProfile) {
|
||||
this(delegates, defaultProfile, Map.of(), PlacementPolicies.fixed(), _ -> 0);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with a placement policy and live-worker counter.
|
||||
*
|
||||
* @param delegates one adapter per configured peer kind; must be non-empty and declare
|
||||
* disjoint profile-name sets
|
||||
* @param defaultProfile the profile a no-argument spawn resolves to under {@code fixed} policy
|
||||
* @param profileConfigs all configured worker profiles (used for candidate weights/caps)
|
||||
* @param placementPolicy which policy governs unqualified spawns
|
||||
* @param liveCount live worker count per profile (must never return {@code null})
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, BridgedConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with role pools (CB-557). An unqualified spawn draws its candidates from
|
||||
* {@code fleet.<role>} instead of from every configured profile, so a reviewer is placed on a
|
||||
* reviewer backend and never on, say, the architect-only one.
|
||||
*
|
||||
* @param fleet the configured role pools; {@code null} ⇒ every profile is a candidate for every
|
||||
* role, which is the pre-CB-557 behaviour
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, BridgedConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
BridgedConfig.Fleet fleet) {
|
||||
// LinkedHashMap, not Map.copyOf: candidates() promises definition order and the weighted
|
||||
// policy breaks exact-weight ties on it, so a salted iteration order would make placement
|
||||
// differ from one JVM run to the next.
|
||||
this(delegates, defaultProfile,
|
||||
constant(Collections.unmodifiableMap(new LinkedHashMap<>(profileConfigs))),
|
||||
constant(placementPolicy), liveCount, constant(fleet));
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor that re-reads its placement inputs per spawn (CB-559), so a config
|
||||
* reload retargets the next member without a restart.
|
||||
*
|
||||
* @param config the live configuration — read at every spawn, never captured
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Supplier<BridgedConfig> config,
|
||||
Function<String, Integer> liveCount) {
|
||||
this(delegates, defaultProfile,
|
||||
() -> config.get().profiles(),
|
||||
() -> PlacementPolicies.fromName(config.get().placement()),
|
||||
liveCount,
|
||||
() -> config.get().fleet());
|
||||
}
|
||||
|
||||
/** The all-suppliers form every other constructor funnels into. */
|
||||
private CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Supplier<Map<String, BridgedConfig.Profile>> profileConfigs,
|
||||
Supplier<PlacementPolicy> placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
this.fleet = fleet;
|
||||
if (delegates.isEmpty()) {
|
||||
throw new IllegalArgumentException("at least one peer adapter must be configured");
|
||||
}
|
||||
this.delegates = List.copyOf(delegates);
|
||||
this.defaultProfile = defaultProfile;
|
||||
this.profileConfigs = profileConfigs;
|
||||
this.placementPolicy = placementPolicy;
|
||||
this.liveCount = liveCount;
|
||||
Map<String, HerdrPeerLauncher> index = new LinkedHashMap<>();
|
||||
for (HerdrPeerLauncher d : this.delegates) {
|
||||
for (String profile : d.profiles()) {
|
||||
HerdrPeerLauncher prev = index.putIfAbsent(profile, d);
|
||||
if (prev != null) {
|
||||
throw new IllegalArgumentException(
|
||||
"worker profile '" + profile + "' is claimed by two peer adapters");
|
||||
}
|
||||
}
|
||||
}
|
||||
// Order-preserving for the same reason, and because profiles() is user-visible (bridge_profiles).
|
||||
this.byProfile = Collections.unmodifiableMap(index);
|
||||
}
|
||||
|
||||
/** A supplier of a value fixed at construction — how the non-reloading constructors funnel in. */
|
||||
private static <T> Supplier<T> constant(T value) {
|
||||
return () -> value;
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-configured profiles, never null.
|
||||
*
|
||||
* <p>Read fresh on every call so a reload is visible; a caller that needs two consistent reads
|
||||
* takes one local, as {@link #poolFor} does.
|
||||
*/
|
||||
private Map<String, BridgedConfig.Profile> profiles0() {
|
||||
Map<String, BridgedConfig.Profile> m = profileConfigs.get();
|
||||
return m == null ? Map.of() : m;
|
||||
}
|
||||
|
||||
/** The adapter owning {@code profileName} (null/blank → the default). Throws on an unknown profile. */
|
||||
private HerdrPeerLauncher route(String profileName) {
|
||||
String resolved = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (resolved == null) {
|
||||
// No profile and no default configured — hand to the first delegate so it raises the
|
||||
// same "no default" error it would on its own; keeps the SPI contract single-sourced.
|
||||
return delegates.getFirst();
|
||||
}
|
||||
HerdrPeerLauncher d = byProfile.get(resolved);
|
||||
if (d == null) {
|
||||
throw new IllegalArgumentException("unknown worker profile: " + resolved);
|
||||
}
|
||||
return d;
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
String requestedProfile = req.profileName();
|
||||
if (requestedProfile != null && !requestedProfile.isBlank()) {
|
||||
// An explicit profile bypasses the placement policy, but not the capacity cap: maxLoad
|
||||
// is documented as an unconditional limit on this profile (BridgedConfig.Profile), and
|
||||
// the charter makes explicit-profile spawns the normal path — so skipping the check
|
||||
// here would leave the cap dead config in real operation.
|
||||
HerdrPeerLauncher d = route(requestedProfile);
|
||||
enforceMaxLoad(requestedProfile);
|
||||
PeerHandle handle = d.spawn(req);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
}
|
||||
|
||||
// CB-557: an unqualified spawn is placed inside the pool of the role it asked for, not across
|
||||
// the whole profile list. An EXPLICIT profile (above) is left alone on purpose — it is the
|
||||
// operator overriding, and refusing it would break `bridge_spawn{profile:"opus"}`, which
|
||||
// carries no role and so would be judged against the dev pool it was never meant for.
|
||||
List<PlacementCandidate> candidates = candidates(req.role());
|
||||
String roleDefault = defaultProfileFor(req.role());
|
||||
Set<String> unreachable = new HashSet<>();
|
||||
PlacementContext ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable);
|
||||
|
||||
int maxAttempts = candidates.isEmpty() ? 1 : candidates.size();
|
||||
for (int attempt = 0; attempt < maxAttempts; attempt++) {
|
||||
// Deliberately uncaught: when no candidate is left (all at cap, or all unreachable) the
|
||||
// policy already throws a clear message. Catching it to rethrow a generic
|
||||
// PeerUnreachableException would replace a precise diagnosis with a vague one.
|
||||
PlacementCandidate chosen = placementPolicy.get().select(ctx);
|
||||
|
||||
HerdrPeerLauncher d = byProfile.get(chosen.profile());
|
||||
if (d == null) {
|
||||
// A configured profile with no adapter is a wiring bug; fail fast.
|
||||
unreachable.add(chosen.profile());
|
||||
continue;
|
||||
}
|
||||
|
||||
// CB-547a: route the chosen profile but keep the caller's session identity — dropping it
|
||||
// here would silently sever the resume handle on every policy-routed spawn. CB-557: the
|
||||
// role rides along for the same reason, or a routed spawn would be labelled as a dev.
|
||||
SpawnRequest routedReq = new SpawnRequest(chosen.profile(), req.requestedCwd(), req.callerCwd(),
|
||||
req.sessionName(), req.resumeSessionId(), req.role());
|
||||
try {
|
||||
PeerHandle handle = d.spawn(routedReq);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
} catch (PeerUnreachableException e) {
|
||||
log.warn("spawn on profile {} unreachable, will retry next candidate if any: {}",
|
||||
chosen.profile(), e.getMessage());
|
||||
unreachable.add(chosen.profile());
|
||||
// Update the context for the next selection so the policy excludes this profile.
|
||||
ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable);
|
||||
}
|
||||
}
|
||||
|
||||
throw new PeerUnreachableException(
|
||||
"no reachable worker profile available after trying " + unreachable.size()
|
||||
+ " candidate(s): " + String.join(", ", unreachable));
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse an explicit-profile spawn when the profile is at its {@code maxLoad} cap.
|
||||
*
|
||||
* <p>maxLoad is a documented, unconditional capacity limit (see {@code BridgedConfig.Profile#maxLoad}),
|
||||
* and the charter makes explicit-profile spawns the normal path — so enforcing it only in placement
|
||||
* ({@code PlacementPolicyUtil}, package-private, hence not linked) would leave the cap dead config
|
||||
* on every call that names a profile. Same rule as placement: {@code live >= cap} is at capacity.
|
||||
*
|
||||
* <p>Deliberately no fallback to another profile: the caller named {@code profile} for a cost/model
|
||||
* reason, and silently re-routing a paid-tier (subscription) request elsewhere is worse than
|
||||
* refusing it. A caller that wants placement should omit the profile and let the policy pick.
|
||||
*
|
||||
* <p>Known TOCTOU limitation — documented, not fixed. {@link #liveCount} is read outside any lock and
|
||||
* {@code SessionManager} registers a session only after {@code launcher.spawn} returns, so two
|
||||
* genuinely concurrent spawns can both pass this check. The race already exists on the placement
|
||||
* path. Closing it needs slot reservation in the registry; serializing spawn here would block on
|
||||
* the readiness gate and is a far worse trade.
|
||||
*
|
||||
* @param profile the profile the caller explicitly named
|
||||
* @throws PlacementException when the profile is at capacity
|
||||
*/
|
||||
private void enforceMaxLoad(String profile) {
|
||||
// Absent config, or a config whose maxLoad normalized to null (non-positive ⇒ unlimited at
|
||||
// load), means no cap — never cap what wasn't configured.
|
||||
BridgedConfig.Profile cfg = profiles0().get(profile);
|
||||
Integer cap = (cfg == null) ? null : cfg.maxLoad();
|
||||
if (cap == null) {
|
||||
return;
|
||||
}
|
||||
int live = liveCount.apply(profile);
|
||||
if (live >= cap) {
|
||||
throw new PlacementException("worker profile '" + profile + "' is at maxLoad: " + live
|
||||
+ " live >= " + cap + " cap; refusing spawn — no fallback to another profile");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The profile names {@code role} may be placed on, in definition order.
|
||||
*
|
||||
* <p>An empty or absent pool means "unconstrained", not "nothing allowed": a config that declares
|
||||
* no pool for a role must keep spawning, so it falls back to every configured profile. Names in a
|
||||
* pool that no adapter declares are dropped here rather than thrown — config load already rejects
|
||||
* a pool entry with no profile, so a survivor is a profile this particular composite does not own.
|
||||
*/
|
||||
private List<String> poolFor(MemberRole role) {
|
||||
Map<String, BridgedConfig.Profile> configured = profiles0();
|
||||
BridgedConfig.Fleet f = fleet.get();
|
||||
List<String> pool = (f == null) ? List.of() : f.profilesFor(role);
|
||||
List<String> known = pool.stream().filter(configured::containsKey).toList();
|
||||
return known.isEmpty() ? List.copyOf(configured.keySet()) : known;
|
||||
}
|
||||
|
||||
/** The profile an unqualified spawn for {@code role} falls back to under {@code fixed} placement. */
|
||||
private String defaultProfileFor(MemberRole role) {
|
||||
List<String> pool = poolFor(role);
|
||||
return pool.isEmpty() ? defaultProfile : pool.getFirst();
|
||||
}
|
||||
|
||||
/** Build the candidate list from {@code role}'s pool, in definition order. */
|
||||
private List<PlacementCandidate> candidates(MemberRole role) {
|
||||
List<PlacementCandidate> out = new ArrayList<>();
|
||||
for (String name : poolFor(role)) {
|
||||
BridgedConfig.Profile w = profiles0().get(name);
|
||||
if (w != null) {
|
||||
out.add(new PlacementCandidate(name, null, w.weight(), w.maxLoad()));
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
return route(req.profileName()).effectiveCwd(req);
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
return route(profileName).parityOverlay(profileName);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) {
|
||||
HerdrPeerLauncher d = spawnedBy.remove(id);
|
||||
if (d == null) {
|
||||
log.debug("stop({}) — no recorded owner, routing to the first adapter (pane-addressed)", id);
|
||||
d = delegates.getFirst();
|
||||
}
|
||||
d.stop(id);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
HerdrPeerLauncher delegate = spawnedBy.get(id);
|
||||
if (delegate == null) {
|
||||
log.debug("clearContext({}) ignored — no recorded owning adapter", id);
|
||||
return false;
|
||||
}
|
||||
return delegate.clearContext(id);
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return byProfile.keySet();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return defaultProfile;
|
||||
}
|
||||
|
||||
/** Every herdr agent, deduplicated by pane id (all delegates share one herdr and list globally). */
|
||||
@Override
|
||||
public List<Agent> list() {
|
||||
Map<String, Agent> byPane = new LinkedHashMap<>();
|
||||
for (HerdrPeerLauncher d : delegates) {
|
||||
for (Agent a : d.list()) {
|
||||
if (a.paneId() != null) {
|
||||
byPane.putIfAbsent(a.paneId(), a);
|
||||
}
|
||||
}
|
||||
}
|
||||
return List.copyOf(byPane.values());
|
||||
}
|
||||
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
int reaped = 0;
|
||||
for (HerdrPeerLauncher d : delegates) {
|
||||
reaped += d.reapOrphanWorkers();
|
||||
}
|
||||
return reaped;
|
||||
}
|
||||
|
||||
/** The union of every adapter's capabilities — a capability any adapter offers, the fleet offers. */
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
EnumSet<Capability> caps = EnumSet.noneOf(Capability.class);
|
||||
for (HerdrPeerLauncher d : delegates) {
|
||||
caps.addAll(d.capabilities());
|
||||
}
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,740 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.Tab;
|
||||
import dev.ltms.bridged.herdr.Workspace;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.security.SecureRandom;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collection;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ConcurrentMap;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Abstract base for {@link PeerLauncher} adapters that materialize a peer as a <em>herdr</em>
|
||||
* agent (a CLI coding agent running in a herdr tab/pane). It owns everything that is the same
|
||||
* regardless of <em>which</em> coding agent runs: tab/pane placement, the CB-306 spawn-readiness
|
||||
* gate, unique naming, CB-117 orphan reap, teardown, {@link #list() listing}, and cwd resolution.
|
||||
*
|
||||
* <p>Two seams are peer-specific and supplied by the concrete adapter:
|
||||
* <ul>
|
||||
* <li>{@code namePrefix} (constructor arg) — the label prefix ({@code claude}, {@code opencode})
|
||||
* that drives both unique naming and the orphan-reap pattern, so each adapter reaps only its
|
||||
* own kind of pane and never another's.</li>
|
||||
* <li>{@link #buildLaunch(BridgedConfig.Profile, LaunchSpec)} — the peer-specific env map + argv, including any
|
||||
* subscription/guard check, MCP mount, and instruction injection. The base never sees how the
|
||||
* peer is configured; it only places and starts the returned {@link Launch}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Placement: in the default {@code tab} policy a peer lands in its own tab inside a dedicated
|
||||
* worker space (found-or-created once, then shared), so peers never split or clutter the user's
|
||||
* real work spaces. Teardown removes the peer's pane <em>and</em> its now-empty tab, tolerating an
|
||||
* already-gone peer so a repeated DELETE is harmless.
|
||||
*/
|
||||
public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
|
||||
/** herdr rejects a duplicate agent {@code name}; we retry a bumped name this many times. */
|
||||
private static final int NAME_RETRIES = 8;
|
||||
|
||||
/**
|
||||
* Retries for {@code agent.start} against a seed pane whose shell has not reached its prompt
|
||||
* yet — {@code tab.create}/{@code pane.split} return as soon as the pane exists, and herdr
|
||||
* refuses to start an agent in a pane that is not "an available shell" ({@code agent_pane_busy}).
|
||||
*/
|
||||
private static final int SHELL_READY_RETRIES = 20;
|
||||
|
||||
private final String namePrefix; // label prefix: naming + reap scheme
|
||||
private final AgentControl agents;
|
||||
private final WorkspaceControl spaces;
|
||||
private final Map<String, BridgedConfig.Profile> profiles; // profile name → spawn settings
|
||||
private final String defaultProfile; // profile a no-arg spawn uses (nullable)
|
||||
|
||||
/** Host env lookup (injectable for tests); adapters read it in {@link #buildLaunch}. */
|
||||
protected final Function<String, String> env;
|
||||
|
||||
private final AtomicLong nameSeq = new AtomicLong(); // per-peer counter (herdr agent names only)
|
||||
|
||||
/**
|
||||
* Live fleet config, read once per spawn. A null supplier or value leaves tab labels at their
|
||||
* default and supplies no role charter. A profile's own {@code tabLabel} still overrides it.
|
||||
*
|
||||
* <p>CB-559: a supplier rather than a snapshot, so a config reload affects the next launch
|
||||
* without a restart. Existing tabs keep the label they were given.
|
||||
*/
|
||||
private final Supplier<BridgedConfig.Fleet> fleet;
|
||||
|
||||
/** The final instruction always requires a bridge reply when the bridge MCP is mounted. */
|
||||
protected static final String REPLY_CHARTER =
|
||||
"You are a spawned member in the claude-bridge fleet. Every message you receive arrives "
|
||||
+ "through the bridge, and the ONLY channel back to the sender is the bridge_reply MCP tool. "
|
||||
+ "Text you write in your terminal is NOT sent anywhere — the sender cannot see your screen, "
|
||||
+ "so an in-terminal answer is silently discarded. Therefore you MUST end EVERY turn by calling "
|
||||
+ "bridge_reply with `content` set to your complete response. This holds for every message without "
|
||||
+ "exception — tasks, questions, clarifications, acknowledgements, and ordinary back-and-forth "
|
||||
+ "conversation. Call bridge_reply exactly once, as the final action of your turn, with your full "
|
||||
+ "answer in `content`; never wait for confirmation first. If you end a turn without calling "
|
||||
+ "bridge_reply, the sender receives nothing and the exchange stalls.";
|
||||
|
||||
/**
|
||||
* Tab numbers, counted per {@code role/profile} pair (CB-557).
|
||||
*
|
||||
* <p>Deliberately not {@link #nameSeq}. That counter is shared by every profile this launcher
|
||||
* serves, because its job is to make herdr <em>agent names</em> unique. Reusing it for the tab
|
||||
* label made the numbers global, so sibling tabs read {@code #4}, {@code #9}, {@code #17} — gaps
|
||||
* that look like a member died. Counting per role+profile makes {@code dev: sonnet #2} mean the
|
||||
* second sonnet dev, which is what a reader assumes it means.
|
||||
*
|
||||
* <p>Resets when the daemon restarts, and that is fine: the label is a human-facing hint, not an
|
||||
* identity. Identity is {@link PeerHandle#id()}.
|
||||
*/
|
||||
private final ConcurrentMap<String, AtomicLong> labelSeq = new ConcurrentHashMap<>();
|
||||
|
||||
private final long spawnReadyTimeoutMs; // 0 = disable gate (legacy non-blocking spawn)
|
||||
private final LongSupplier nowMillis; // monotonic clock (injectable for tests)
|
||||
private final Runnable sleeper; // sleep/wait hook (injectable for tests; never real-sleep in unit tests)
|
||||
|
||||
// Per-process token mixed into each peer name so a fresh process (nameSeq back at 0) cannot
|
||||
// collide with same-profile peers that outlived a restart. See startUniquelyNamed.
|
||||
private final String nameNonce = String.format("%06x", new SecureRandom().nextInt(1 << 24));
|
||||
|
||||
// CB-519: PeerHandle.id() is a host-unique opaque UUID, decoupled from the herdr pane id. The
|
||||
// routing/registry key is the UUID; the herdr pane id is a launcher-private placement/teardown
|
||||
// coordinate. This map bridges the two so stop(id) can resolve a host-unique key back to the
|
||||
// exact pane it must tear down. The pane id is launcher-private (never the routing key) — see
|
||||
// PeerHandle.id().
|
||||
private final ConcurrentMap<String, String> paneByAgentId = new ConcurrentHashMap<>();
|
||||
private final AtomicBoolean resetUnsupportedLogged = new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* @param namePrefix label prefix for this peer kind (drives naming and reap)
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param profiles configured peer profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (never called when the gate is disabled); the poll
|
||||
* interval is baked into this hook, so the base needs no poll field
|
||||
*/
|
||||
protected HerdrPeerLauncher(String namePrefix, AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper) {
|
||||
this(namePrefix, agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
nowMillis, sleeper, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, plus the live {@code fleet} config (CB-557).
|
||||
*
|
||||
* @param fleet live fleet config, read once per spawn; {@code null} ⇒ default tab label and no
|
||||
* role charter. A separate constructor rather than a new parameter on the one
|
||||
* above, so every existing call site keeps the default without an edit.
|
||||
*/
|
||||
protected HerdrPeerLauncher(String namePrefix, AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
this.fleet = fleet;
|
||||
this.namePrefix = namePrefix;
|
||||
this.agents = agents;
|
||||
this.spaces = spaces;
|
||||
this.profiles = Map.copyOf(profiles);
|
||||
this.defaultProfile = defaultProfile;
|
||||
this.env = env;
|
||||
this.spawnReadyTimeoutMs = spawnReadyTimeoutMs;
|
||||
this.nowMillis = nowMillis;
|
||||
this.sleeper = sleeper;
|
||||
}
|
||||
|
||||
// --- adapter seams -------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Build the peer-specific launch for {@code cfg}: the environment map and argv handed to herdr.
|
||||
* Any subscription/guard check, MCP mount, and instruction injection happen here. The env map
|
||||
* and argv are adapter-private; the base only places and starts what is returned.
|
||||
*/
|
||||
protected abstract Launch buildLaunch(BridgedConfig.Profile cfg, LaunchSpec spec);
|
||||
|
||||
/** Direct transport access for peer-specific, non-turn control operations. */
|
||||
protected final AgentControl agents() {
|
||||
return agents;
|
||||
}
|
||||
|
||||
/** Resolve the public peer id to the launcher's private herdr target. */
|
||||
protected final String agentTarget(String id) {
|
||||
return paneByAgentId.get(id);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
if (resetUnsupportedLogged.compareAndSet(false, true)) {
|
||||
log.warn("context reset is unsupported for peer kind {}; clearAfterTurn is a no-op",
|
||||
namePrefix);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* A peer-specific launch: the herdr {@code env} map and {@code argv}, plus — for an adapter
|
||||
* that carries durable session identity (CB-547a) — the peer's OWN session id
|
||||
* ({@link PeerHandle#agentSessionId()}), known before the peer has written anything. Null for
|
||||
* a launch that carries no identity.
|
||||
*/
|
||||
protected record Launch(Map<String, String> env, List<String> argv, String agentSessionId) {
|
||||
|
||||
/** A launch without a discoverable agent session id (an adapter that carries none). */
|
||||
Launch(Map<String, String> env, List<String> argv) {
|
||||
this(env, argv, null);
|
||||
}
|
||||
}
|
||||
|
||||
/** All per-spawn values adapters may need, including the base-composed effective charter. */
|
||||
protected record LaunchSpec(String sessionName, String resumeSessionId, MemberRole role, String charter) {
|
||||
}
|
||||
|
||||
// --- profile surface -----------------------------------------------------------------------
|
||||
|
||||
/** The configured peer profile names (what {@code spawn(profile)} accepts). */
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return profiles.keySet();
|
||||
}
|
||||
|
||||
/** The parity-overlay file list for {@code profileName} (default list when unset). */
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
String name = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (name == null || name.isBlank()) {
|
||||
return List.of();
|
||||
}
|
||||
BridgedConfig.Profile cfg = profiles.get(name);
|
||||
return cfg == null ? List.of() : cfg.parityOverlay();
|
||||
}
|
||||
|
||||
/** The profile a no-argument spawn uses, or {@code null} if none is configured. */
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return defaultProfile;
|
||||
}
|
||||
|
||||
/** The configured profiles, for adapter capability decisions (e.g. any git-token grant). */
|
||||
protected Collection<BridgedConfig.Profile> profileConfigs() {
|
||||
return profiles.values();
|
||||
}
|
||||
|
||||
/** Resolve {@code profileName} (null/blank → default) to its config, or throw with the options. */
|
||||
protected BridgedConfig.Profile requireProfile(String profileName) {
|
||||
String name = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (name == null || name.isBlank()) {
|
||||
throw new IllegalArgumentException("no default worker profile is configured — "
|
||||
+ "pass a profile; configured: " + profiles.keySet());
|
||||
}
|
||||
BridgedConfig.Profile cfg = profiles.get(name);
|
||||
if (cfg == null) {
|
||||
throw new IllegalArgumentException("unknown worker profile '" + name
|
||||
+ "' — configured: " + profiles.keySet());
|
||||
}
|
||||
return cfg;
|
||||
}
|
||||
|
||||
// --- spawn ---------------------------------------------------------------------------------
|
||||
|
||||
/** A started peer plus the launch's agent-session id (the resume handle, or null). */
|
||||
private record Spawned(Agent agent, String agentSessionId) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a peer. {@code profileName} null/blank → the default profile. The working directory
|
||||
* (CB-112) is resolved by {@link #resolveCwd}: an explicit {@code requestedCwd}, else the
|
||||
* profile's configured {@code cwd}, else {@code callerCwd} (the primary's cwd, when the spawn
|
||||
* came from the primary over MCP), else the daemon's cwd — never assumed to be {@code $HOME}.
|
||||
* The adapter's {@link #buildLaunch} runs before any herdr call.
|
||||
*/
|
||||
protected Agent spawnInternal(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd, null, null).agent();
|
||||
}
|
||||
|
||||
/** Pre-CB-557 shape: no explicit role, so the tab is labelled as a {@code dev}. */
|
||||
protected Spawned spawnInternal(String profileName, String requestedCwd, String callerCwd,
|
||||
String sessionName, String resumeSessionId) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd, sessionName, resumeSessionId,
|
||||
MemberRole.DEV);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a peer with session identity (CB-547a). The session values, role, and charter are
|
||||
* threaded from the {@link SpawnRequest} into {@link #buildLaunch(BridgedConfig.Profile,
|
||||
* LaunchSpec)}, and the launch's resolved agent-session id is returned alongside the agent so
|
||||
* the caller can put it on the {@link PeerHandle}.
|
||||
*/
|
||||
protected Spawned spawnInternal(String profileName, String requestedCwd, String callerCwd,
|
||||
String sessionName, String resumeSessionId, MemberRole role) {
|
||||
BridgedConfig.Profile cfg = requireProfile(profileName);
|
||||
BridgedConfig.Fleet liveFleet = fleet == null ? null : fleet.get();
|
||||
String roleCharter = liveFleet == null ? null : liveFleet.charterFor(role);
|
||||
String replyCharter = cfg.hasMcp() ? REPLY_CHARTER : null;
|
||||
String charter = roleCharter == null ? replyCharter
|
||||
: replyCharter == null ? roleCharter : roleCharter + "\n\n" + replyCharter;
|
||||
Launch launch = buildLaunch(cfg, new LaunchSpec(sessionName, resumeSessionId, role, charter));
|
||||
String cwd = resolveCwd(requestedCwd, cfg, callerCwd);
|
||||
Agent agent = cfg.tabPlacement()
|
||||
? spawnInTab(cfg, launch.env(), launch.argv(), cwd, role, liveFleet)
|
||||
: spawnAsPane(cfg, launch.env(), launch.argv(), cwd);
|
||||
return new Spawned(agent, launch.agentSessionId());
|
||||
}
|
||||
|
||||
/**
|
||||
* The next tab number for {@code role} on {@code profile}, starting at 1.
|
||||
*
|
||||
* <p>Starts at 1 rather than 0 because the number is read by a person: {@code "dev: sonnet #1"}
|
||||
* is the first one, and {@code #0} invites the question of where {@code #1} went.
|
||||
*/
|
||||
private long nextLabelSeq(MemberRole role, String profile) {
|
||||
String key = (role == null ? "" : role.wireName()) + "/" + profile;
|
||||
return labelSeq.computeIfAbsent(key, _ -> new AtomicLong()).incrementAndGet();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Delegates to {@link #spawnInternal} and wraps the resulting herdr {@link Agent} in a
|
||||
* {@link WorkerHandle} whose {@link PeerHandle#id()} is a fresh <em>host-unique</em> opaque
|
||||
* UUID (CB-519), deliberately decoupled from the herdr pane id: the id is the registry/routing
|
||||
* key and must never collide across daemon processes on the same host, while the herdr pane id
|
||||
* stays a launcher-private placement/teardown coordinate, remembered here so {@link #stop}
|
||||
* can resolve the host-unique key back to its pane. When {@code spawnReadyTimeoutMs > 0},
|
||||
* blocks until the peer's herdr status is injectable or the timeout elapses; on timeout the
|
||||
* pane is closed (no orphan) and a {@link PeerUnreachableException} is thrown.
|
||||
*/
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
Spawned spawned = spawnInternal(req.profileName(), req.requestedCwd(), req.callerCwd(),
|
||||
req.sessionName(), req.resumeSessionId(), req.role());
|
||||
Agent agent = spawned.agent();
|
||||
String paneId = agent.paneId();
|
||||
if (spawnReadyTimeoutMs > 0) {
|
||||
waitUntilInjectableOrThrow(paneId);
|
||||
}
|
||||
// CB-519: the handle id is a host-unique UUID; the herdr pane it maps to stays internal.
|
||||
String id = UUID.randomUUID().toString();
|
||||
paneByAgentId.put(id, paneId);
|
||||
return new WorkerHandle(id, agent.terminalId(), requireProfile(req.profileName()).profile(),
|
||||
req.sessionName(), spawned.agentSessionId());
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
return effectiveCwd(req.profileName(), req.requestedCwd(), req.callerCwd());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-301: the effective working directory a spawn for {@code profileName} would use, without
|
||||
* actually spawning.
|
||||
*/
|
||||
private String effectiveCwd(String profileName, String requestedCwd, String callerCwd) {
|
||||
return resolveCwd(requestedCwd, requireProfile(profileName), callerCwd);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-112 cwd resolution: spawn arg → profile config → the primary's cwd → the daemon's cwd.
|
||||
* Never returns {@code null}/blank: {@code "."} (the daemon's own working directory) is the
|
||||
* guaranteed last resort so a pathological environment with an unset {@code user.dir} still
|
||||
* honours the "never assume {@code $HOME}" contract rather than letting herdr default the pane.
|
||||
*/
|
||||
private static String resolveCwd(String requestedCwd, BridgedConfig.Profile cfg, String callerCwd) {
|
||||
return firstNonBlank(requestedCwd, cfg.cwd(), callerCwd, System.getProperty("user.dir"), ".");
|
||||
}
|
||||
|
||||
private static String firstNonBlank(String... values) {
|
||||
for (String v : values) {
|
||||
if (v != null && !v.isBlank()) return v;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Dedicated worker space → own tab (carrying cwd+env) → start the peer into the seed pane. */
|
||||
private Agent spawnInTab(BridgedConfig.Profile cfg, Map<String, String> workerEnv,
|
||||
List<String> argv, String cwd, MemberRole role, BridgedConfig.Fleet liveFleet) {
|
||||
Workspace space = spaces.ensureWorkspace(cfg.workspace());
|
||||
Tab.Created tab = spaces.createTab(space.workspaceId(), cwd, workerEnv);
|
||||
log.info("spawning {} profile={} space={} tab={} cwd={}",
|
||||
namePrefix, cfg.profile(), space.workspaceId(), tab.tab().tabId(), cwd);
|
||||
|
||||
Started started;
|
||||
try {
|
||||
if (tab.rootPaneId() == null) {
|
||||
// Protocol 19 starts the agent INTO the seed pane — without one there is nowhere
|
||||
// to start, and a partial tab would be left behind.
|
||||
throw new IllegalStateException("tab " + tab.tab().tabId()
|
||||
+ " had no seed pane in the create response — cannot start a peer in it");
|
||||
}
|
||||
started = startUniquelyNamed(cfg, argv, tab.rootPaneId());
|
||||
} catch (RuntimeException e) {
|
||||
// The peer never started — don't leave the tab we just created orphaned.
|
||||
// Best-effort cleanup; never let it mask the real spawn failure.
|
||||
try {
|
||||
spaces.closeTab(tab.tab().tabId());
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to close orphaned tab {} after spawn error: {}",
|
||||
tab.tab().tabId(), cleanup.getMessage());
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
|
||||
// The peer is LIVE now, in the seed pane itself (no shell pane to drop — protocol 19).
|
||||
// Labelling is cosmetic: it must not fail the spawn or orphan the running peer — on error
|
||||
// we log and still return it so the caller gets its paneId and can tear it down.
|
||||
tidy("label tab " + tab.tab().tabId(),
|
||||
() -> spaces.renameTab(tab.tab().tabId(),
|
||||
cfg.renderTabLabel(
|
||||
liveFleet == null ? null : liveFleet.tabLabel(),
|
||||
role, nextLabelSeq(role, cfg.profile()))));
|
||||
log.info("{} started pane={} tab={} terminal={}",
|
||||
namePrefix, started.agent().paneId(), started.agent().tabId(), started.agent().terminalId());
|
||||
return started.agent();
|
||||
}
|
||||
|
||||
/** Run a best-effort post-start cleanup step, logging (not throwing) on failure. */
|
||||
private void tidy(String what, Runnable step) {
|
||||
try {
|
||||
step.run();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("post-start step failed ({}) — peer is running regardless: {}", what, e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/** Legacy placement: split the currently-focused tab; the peer still starts in {@code cwd}. */
|
||||
private Agent spawnAsPane(BridgedConfig.Profile cfg, Map<String, String> workerEnv,
|
||||
List<String> argv, String cwd) {
|
||||
log.info("spawning {} (pane placement) profile={} cwd={} argv={}",
|
||||
namePrefix, cfg.profile(), cwd, argv);
|
||||
String paneId = spaces.splitPane(cwd, workerEnv);
|
||||
if (paneId == null) {
|
||||
throw new IllegalStateException("pane.split returned no pane — cannot start a peer");
|
||||
}
|
||||
Agent peer = startUniquelyNamed(cfg, argv, paneId).agent();
|
||||
log.info("{} started pane={} terminal={}", namePrefix, peer.paneId(), peer.terminalId());
|
||||
return peer;
|
||||
}
|
||||
|
||||
/** A started peer together with the sequence its unique name/label used. */
|
||||
private record Started(Agent agent, long seq) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Start the peer under a unique herdr agent name. herdr requires each running agent's
|
||||
* {@code name} to be distinct (a 2nd identical {@code name} fails {@code agent_name_taken}) —
|
||||
* the exact case that makes multiple peers useful. The name is
|
||||
* {@code <prefix>-<profile>-<nonce>-<seq>}: {@code seq} distinguishes peers within this process,
|
||||
* and the per-process {@code nonce} keeps a fresh process (whose {@code seq} restarts at 0) from
|
||||
* colliding with same-profile peers that outlived a restart. The retry is a belt-and-braces
|
||||
* backstop for the astronomically unlikely nonce+seq clash; the name is a label only — herdr
|
||||
* detects kind and status from terminal output, not from it.
|
||||
*/
|
||||
private Started startUniquelyNamed(BridgedConfig.Profile cfg, List<String> argv, String paneId) {
|
||||
// Protocol 19 resolves the executable from the agent kind (== namePrefix here), so
|
||||
// argv[0] — the configured executable — is dropped and only the extra args are passed.
|
||||
List<String> args = argv.isEmpty() ? argv : argv.subList(1, argv.size());
|
||||
HerdrException last = null;
|
||||
for (int attempt = 0; attempt < NAME_RETRIES; attempt++) {
|
||||
long seq = nameSeq.incrementAndGet();
|
||||
String name = namePrefix + "-" + cfg.profile() + "-" + nameNonce + "-" + seq;
|
||||
try {
|
||||
return new Started(startAwaitingShellPrompt(name, args, paneId), seq);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_name_taken".equals(e.code())) throw e;
|
||||
log.debug("peer name '{}' taken, retrying", name);
|
||||
last = e;
|
||||
}
|
||||
}
|
||||
throw last;
|
||||
}
|
||||
|
||||
/** Start the agent into {@code paneId}, waiting out the seed shell's boot with the sleeper. */
|
||||
private Agent startAwaitingShellPrompt(String name, List<String> args, String paneId) {
|
||||
HerdrException busy = null;
|
||||
for (int attempt = 0; attempt < SHELL_READY_RETRIES; attempt++) {
|
||||
try {
|
||||
return agents.start(name, namePrefix, args, paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_pane_busy".equals(e.code())) throw e;
|
||||
log.debug("pane {} not at its shell prompt yet, retrying agent.start", paneId);
|
||||
busy = e;
|
||||
sleeper.run();
|
||||
}
|
||||
}
|
||||
throw busy;
|
||||
}
|
||||
|
||||
// --- discovery + reap ----------------------------------------------------------------------
|
||||
|
||||
/** All herdr-tracked agents — discovery for "what peers exist". */
|
||||
@Override
|
||||
public List<Agent> list() {
|
||||
return agents.list();
|
||||
}
|
||||
|
||||
/**
|
||||
* Reap peer panes left behind by an earlier daemon process (CB-117). herdr keeps a peer's pane
|
||||
* alive across a daemon restart <em>by design</em>, and that pane's id is held only by its
|
||||
* spawner — so a peer whose owning process exited before issuing the matching teardown leaks
|
||||
* with nothing tracking it. On boot we scan herdr for agents whose name matches our
|
||||
* {@code <prefix>-<profile>-<nonce>-<seq>} scheme with a nonce <em>other</em> than this
|
||||
* process's {@link #nameNonce}, and tear each one down (its pane and, via {@link #stop}, its
|
||||
* now-empty dedicated tab). A current-nonce peer is ours and live, so it is left running; a
|
||||
* user's own session carries no such name and is never touched. A peer from a <em>different</em>
|
||||
* adapter (different prefix) is likewise never touched. Best-effort: a failed listing, or a
|
||||
* failure to stop any one peer, is logged and never aborts startup.
|
||||
*
|
||||
* @return the number of orphaned peers reaped
|
||||
*/
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
List<Agent> all;
|
||||
try {
|
||||
all = agents.list();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("orphan-peer reap skipped — agent.list failed: {}", e.getMessage());
|
||||
return 0;
|
||||
}
|
||||
int reaped = 0;
|
||||
for (Agent a : all) {
|
||||
if (!isForeignWorker(namePrefix, a.name(), nameNonce)) continue;
|
||||
try {
|
||||
stop(a.paneId());
|
||||
reaped++;
|
||||
log.info("reaped orphan {} {} (pane={} tab={}) left by a prior daemon",
|
||||
namePrefix, a.name(), a.paneId(), a.tabId());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("could not reap orphan {} {} (pane={}): {}",
|
||||
namePrefix, a.name(), a.paneId(), e.getMessage());
|
||||
}
|
||||
}
|
||||
if (reaped > 0) {
|
||||
log.info("orphan-peer reap complete — {} stale {} peer(s) removed at startup", reaped, namePrefix);
|
||||
}
|
||||
return reaped;
|
||||
}
|
||||
|
||||
/** The {@code <prefix>-<profile>-<nonce>-<seq>} name pattern; group 1 captures the 6-hex nonce. */
|
||||
static Pattern workerNamePattern(String prefix) {
|
||||
return Pattern.compile(prefix + "-.*-([0-9a-f]{6})-\\d+");
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code name} is a peer of kind {@code prefix} started by a <em>different</em> process
|
||||
* than {@code currentNonce} — the reap predicate (CB-117). True only for the prefix's naming
|
||||
* scheme with a foreign nonce: a non-peer name, a different adapter's name, or our own live
|
||||
* nonce is excluded. Pure and package-private so the decision is unit-testable without herdr.
|
||||
*/
|
||||
static boolean isForeignWorker(String prefix, String name, String currentNonce) {
|
||||
String nonce = workerNonce(prefix, name);
|
||||
return nonce != null && !nonce.equals(currentNonce);
|
||||
}
|
||||
|
||||
/** The 6-hex nonce embedded in a {@code prefix} peer name, or {@code null} if not one. */
|
||||
static String workerNonce(String prefix, String name) {
|
||||
if (name == null) return null;
|
||||
Matcher m = workerNamePattern(prefix).matcher(name);
|
||||
return m.matches() ? m.group(1) : null;
|
||||
}
|
||||
|
||||
/** This process's peer-name nonce (a label component only; exposed for reaper tests). */
|
||||
String nameNonce() {
|
||||
return nameNonce;
|
||||
}
|
||||
|
||||
// --- teardown ------------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Tear a peer down: close the pane, and close its tab <em>only</em> when the peer is that tab's
|
||||
* sole occupant. The single-pane check is what makes this safe regardless of how the peer was
|
||||
* placed (or a placement-config change across a restart): a pane-placement peer sitting in one
|
||||
* of the user's shared tabs has siblings, so its tab is never closed — we only ever remove a
|
||||
* tab we created to hold one peer.
|
||||
*
|
||||
* <p>{@code idOrPane} is the {@link PeerHandle#id()} of a peer this launcher spawned (CB-519's
|
||||
* host-unique opaque UUID), resolved through {@link #paneByAgentId} to the pane it must tear
|
||||
* down. An argument that is not one of our ids is treated as a raw herdr pane id — the
|
||||
* {@link #reapOrphanWorkers() orphan-reap} and spawn-gate-timeout paths, plus any caller that
|
||||
* passes a pane directly, keep working without an owning id.
|
||||
*
|
||||
* <p>Resolves the tab from the pane <em>before</em> closing it. An already-gone pane/tab
|
||||
* (repeated DELETE, crashed peer) is treated as success; any other failure propagates so a
|
||||
* genuinely failed teardown is not reported as done.
|
||||
*/
|
||||
@Override
|
||||
public void stop(String idOrPane) {
|
||||
// Teardown knows only the pane, not which profile spawned it. Attempt tab cleanup when any
|
||||
// profile uses tab placement (so the bridge may have created a dedicated peer tab); the
|
||||
// single-occupant check below is what actually protects the user's shared tabs.
|
||||
String paneId = paneByAgentId.remove(idOrPane);
|
||||
if (paneId == null) {
|
||||
paneId = idOrPane; // raw-pane fallback (reap, gate timeout, pane-addressed callers)
|
||||
}
|
||||
WorkspaceControl.PaneLocation loc = usesTabPlacement() ? spaces.locatePane(paneId) : null;
|
||||
try {
|
||||
agents.close(paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!isAlreadyGone(e)) throw e;
|
||||
log.debug("pane.close({}) ignored — already gone: {}", paneId, e.getMessage());
|
||||
}
|
||||
if (loc != null && loc.tabPaneCount() == 1) {
|
||||
spaces.closeTab(loc.tabId());
|
||||
} else if (loc != null) {
|
||||
log.debug("not closing tab {} — it holds {} panes (not a dedicated peer tab)",
|
||||
loc.tabId(), loc.tabPaneCount());
|
||||
}
|
||||
}
|
||||
|
||||
/** Whether any configured profile places peers in their own tab (so tabs may need cleanup). */
|
||||
private boolean usesTabPlacement() {
|
||||
return profiles.values().stream().anyMatch(BridgedConfig.Profile::tabPlacement);
|
||||
}
|
||||
|
||||
/** True when a herdr error means the target is already gone (safe to treat as done). */
|
||||
private static boolean isAlreadyGone(HerdrException e) {
|
||||
return e.code() != null && e.code().endsWith("_not_found");
|
||||
}
|
||||
|
||||
// --- spawn-readiness gate (CB-306) ---------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Poll {@link AgentControl#status} until the pane reports an injectable state or the configured
|
||||
* timeout elapses. On timeout, close the pane (self-reap) and throw.
|
||||
*/
|
||||
private void waitUntilInjectableOrThrow(String paneId) {
|
||||
long deadline = nowMillis.getAsLong() + spawnReadyTimeoutMs;
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
if (agents.status(paneId).injectable()) {
|
||||
log.debug("peer pane={} reached injectable state", paneId);
|
||||
return;
|
||||
}
|
||||
sleeper.run();
|
||||
}
|
||||
log.warn("peer pane={} did not become injectable within {}ms — closing", paneId, spawnReadyTimeoutMs);
|
||||
stop(paneId);
|
||||
throw new PeerUnreachableException(
|
||||
"worker pane " + paneId + " did not reach injectable state within "
|
||||
+ spawnReadyTimeoutMs + "ms");
|
||||
}
|
||||
|
||||
/**
|
||||
* A concrete {@link PeerHandle} wrapping herdr agent coordinates, the profile that spawned it,
|
||||
* and the session identity the launch resolved (CB-547a): the bridge's logical name and the
|
||||
* peer's own session id, both null when the spawn carried no identity.
|
||||
*/
|
||||
private record WorkerHandle(String id, String terminalId, String profile,
|
||||
String sessionName, String agentSessionId) implements PeerHandle {
|
||||
}
|
||||
|
||||
// --- shared helpers ------------------------------------------------------------------------
|
||||
|
||||
/** Put {@code k → v} only when {@code v} is present (non-null, non-blank). */
|
||||
protected static void putIfPresent(Map<String, String> m, String k, String v) {
|
||||
if (v != null && !v.isBlank()) {
|
||||
m.put(k, v);
|
||||
}
|
||||
}
|
||||
|
||||
/** Host env lookup that tolerates an unconfigured (null/blank) var name — returns null then. */
|
||||
protected String resolveEnv(String name) {
|
||||
return (name == null || name.isBlank()) ? null : env.apply(name);
|
||||
}
|
||||
|
||||
/**
|
||||
* The parity-neutral git-forge token grant (CB-302): when {@code cfg} opts in via
|
||||
* {@code gitTokenEnv} and the token resolves, inject {@code GITEA_TOKEN} plus its paired
|
||||
* {@code GITEA_HOST}. Push over SSH is unaffected; the only incremental grant is PR-create.
|
||||
* Peer-neutral, so every herdr adapter reuses it unchanged.
|
||||
*/
|
||||
protected void applyGitToken(Map<String, String> workerEnv, BridgedConfig.Profile cfg) {
|
||||
if (!cfg.hasGitToken()) {
|
||||
return;
|
||||
}
|
||||
String gitToken = resolveEnv(cfg.gitTokenEnv());
|
||||
if (gitToken != null) {
|
||||
workerEnv.put("GITEA_TOKEN", gitToken);
|
||||
putIfPresent(workerEnv, "GITEA_HOST", resolveEnv(cfg.gitHostEnv()));
|
||||
}
|
||||
}
|
||||
|
||||
/** A fresh mutable env map — the conventional starting point for {@link #buildLaunch}. */
|
||||
/**
|
||||
* Seed a worker's environment (CB-511): the daemon's own {@code PATH}, then the profile's
|
||||
* {@code env:} entries.
|
||||
*
|
||||
* <p>Why this exists: bridged passes herdr an explicit env map, and herdr merges it into
|
||||
* <em>its own</em> process environment. So before this, a worker inherited whatever PATH the
|
||||
* herdr server happened to be started with — on this host, one from weeks earlier with no JDK
|
||||
* and no Maven, which left workers unable to run the build they were being asked to run. The
|
||||
* worker's toolchain must follow from configuration, not from how a long-lived daemon was
|
||||
* launched.
|
||||
*
|
||||
* <p>Adapter-specific variables are layered on top of this by {@code buildLaunch} and therefore
|
||||
* win. That ordering is deliberate and load-bearing: it stops a profile's {@code env:} from
|
||||
* overriding {@code ANTHROPIC_BASE_URL} and slipping past {@link
|
||||
* dev.ltms.bridged.guard.SubscriptionGuard}, which is checked against the profile's
|
||||
* {@code baseUrl} and nothing else.
|
||||
*/
|
||||
protected Map<String, String> baseEnv(BridgedConfig.Profile cfg) {
|
||||
Map<String, String> workerEnv = new LinkedHashMap<>();
|
||||
String path = env.apply("PATH");
|
||||
if (path != null && !path.isBlank()) {
|
||||
workerEnv.put("PATH", path);
|
||||
}
|
||||
if (cfg != null && cfg.env() != null) {
|
||||
workerEnv.putAll(cfg.env());
|
||||
}
|
||||
return workerEnv;
|
||||
}
|
||||
|
||||
/** Defensive copy of {@code argv} plus room to append launch flags. */
|
||||
protected static List<String> mutableArgv(List<String> argv) {
|
||||
return new ArrayList<>(argv);
|
||||
}
|
||||
|
||||
/**
|
||||
* Uninterruptible sleep — the production {@link #sleeper}. Tests supply their own no-op /
|
||||
* fast-faking sleeper so they never real-sleep.
|
||||
*/
|
||||
protected static void sleepUninterruptibly(long ms) {
|
||||
try {
|
||||
Thread.sleep(ms);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
// preserve the interrupt flag but continue — poll loops should not be aborted by an
|
||||
// interrupt that was not meant for them.
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,459 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.fasterxml.jackson.databind.node.ObjectNode;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The {@link HerdrPeerLauncher} adapter for <strong>opencode</strong> — an open-source,
|
||||
* provider-agnostic terminal coding agent. Its whole reason for existing is to prove the
|
||||
* {@code PeerLauncher} SPI is genuinely provider-neutral: opencode shares none of Claude Code's
|
||||
* private launch seams, yet reuses every line of shared transport in the base (tab/pane placement,
|
||||
* the CB-306 readiness gate, unique naming + CB-117 reap, teardown, listing, cwd).
|
||||
*
|
||||
* <p>The divergences from {@link ClaudeCodeLauncher}, all confined to {@link #buildLaunch}:
|
||||
* <ul>
|
||||
* <li><strong>No subscription boundary.</strong> opencode carries no {@code ANTHROPIC_BASE_URL}
|
||||
* and there is no {@link dev.ltms.bridged.guard.SubscriptionGuard} — the guard is a
|
||||
* Claude-private concern, not part of the SPI. opencode reads the operator's own provider
|
||||
* credentials from its global {@code auth.json}; the bridge injects none.</li>
|
||||
* <li><strong>File-based MCP mount + instructions.</strong> opencode has no inline
|
||||
* {@code --mcp-config}/{@code --append-system-prompt}. Instead the bridge writes an ephemeral
|
||||
* {@code opencode.json} that declares the bridge as a {@code remote} MCP server and lists a
|
||||
* member-charter file under {@code instructions}, then points the worker at it with
|
||||
* {@code OPENCODE_CONFIG}. This is the one place the launcher touches disk — Claude never did.</li>
|
||||
* <li><strong>Model as a flag.</strong> the {@code provider/model} selector is passed as
|
||||
* {@code -m}, not an env var.</li>
|
||||
* <li><strong>{@code opencode} name prefix</strong> so reap matches {@code opencode-*} panes and
|
||||
* never another adapter's.</li>
|
||||
* </ul>
|
||||
*/
|
||||
public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
/** Label prefix for this adapter's herdr agent names (drives naming + orphan reap). */
|
||||
private static final String NAME_PREFIX = "opencode";
|
||||
|
||||
/** Writer for the generated {@code opencode.json}. */
|
||||
private static final ObjectMapper JSON = new ObjectMapper();
|
||||
|
||||
/** Root under which per-spawn opencode config dirs are created (injectable for tests). */
|
||||
private final Path configRoot;
|
||||
|
||||
/**
|
||||
* Session discovery against opencode's on-disk storage ({@link OpenCodeSessionDiscovery}) —
|
||||
* the one seam that knows opencode's private session-file layout. Its root is injectable for
|
||||
* tests so they never touch the operator's real {@code ~/.local/share/opencode}.
|
||||
*/
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so it
|
||||
* matches the legacy non-blocking spawn semantics. Config dirs are created under the JVM temp dir.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, 0,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(300),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot());
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with the spawn-ready gate enabled. Polls {@code agents.status()} until
|
||||
* the pane reports an injectable state or {@code spawnReadyTimeoutMs} elapses.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs) {
|
||||
this(agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, spawnReadyPollMs, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor carrying the fleet-wide tab-label template (CB-557). The template comes
|
||||
* from {@code fleet.tabLabel}, which a profile cannot know because it names the member's
|
||||
* <em>role</em>; a profile may still override it with its own {@code tabLabel}.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor. Every injectable collaborator is explicit so unit tests supply a
|
||||
* fake clock ({@code nowMillis}), poll-loop wait ({@code sleeper}), and a temp {@code configRoot}
|
||||
* they can inspect the generated {@code opencode.json}/charter under.
|
||||
*
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param profiles configured worker profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (encodes the poll interval; never called when the
|
||||
* gate is disabled)
|
||||
* @param configRoot existing directory under which per-spawn config dirs are created
|
||||
* @param discoveryRoot opencode's on-disk storage root to scan for session records
|
||||
* (injectable for tests; opencode's layout is matched at
|
||||
* {@link OpenCodeSessionDiscovery})
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
nowMillis, sleeper, configRoot, discoveryRoot, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the fleet-wide tab-label template (CB-557).
|
||||
*
|
||||
* @param fleet live fleet config, read once for each spawn
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
}
|
||||
|
||||
private static Path defaultConfigRoot() {
|
||||
return Path.of(System.getProperty("java.io.tmpdir"));
|
||||
}
|
||||
|
||||
/** The default opencode storage root: {@code ~/.local/share/opencode} (the XDG data dir). */
|
||||
private static Path defaultDiscoveryRoot() {
|
||||
return Path.of(System.getProperty("user.home"), ".local", "share", "opencode");
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Builds the opencode launch: no {@code ANTHROPIC_*} and no guard (opencode reads its own
|
||||
* provider credentials); when the profile mounts the bridge MCP or has a member charter,
|
||||
* generate an ephemeral {@code opencode.json} (remote MCP server + member-charter instructions)
|
||||
* and point the worker at it via {@code OPENCODE_CONFIG}; carry the parity-neutral git-forge
|
||||
* grant; and select the model with {@code -m}.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg, LaunchSpec spec) {
|
||||
Map<String, String> workerEnv = baseEnv(cfg);
|
||||
// A config file is needed for the bridge MCP mount, a member charter, or a pinned endpoint (CB-508).
|
||||
if (cfg.hasMcp() || spec.charter() != null || hasCustomProvider(cfg)) {
|
||||
workerEnv.put("OPENCODE_CONFIG", writeConfig(cfg, spec.charter()).toString());
|
||||
}
|
||||
applyGitToken(workerEnv, cfg);
|
||||
return new Launch(workerEnv,
|
||||
argvWithResume(argvWithModel(argvWithAuto(cfg), cfg), spec.resumeSessionId()));
|
||||
}
|
||||
|
||||
/**
|
||||
* True when this profile pins its own OpenAI-compatible endpoint (CB-508) rather than using
|
||||
* whatever provider opencode resolves by default.
|
||||
*
|
||||
* <p>Note this reuses {@code baseUrl}, the same field the Claude adapter injects as
|
||||
* {@code ANTHROPIC_BASE_URL} — but it does <em>not</em> go through {@code SubscriptionGuard}.
|
||||
* That asymmetry is deliberate and safe: the guard exists to stop a worker borrowing the
|
||||
* primary's Anthropic subscription, and an opencode process has no Anthropic credential path
|
||||
* at all. Pointing it at a local vLLM cannot leak the subscription.
|
||||
*/
|
||||
private static boolean hasCustomProvider(BridgedConfig.Profile cfg) {
|
||||
return cfg.baseUrl() != null && !cfg.baseUrl().isBlank();
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv plus the unconditional {@code --auto} flag, which auto-approves the
|
||||
* permissions opencode does not explicitly deny. It is unconditional, not a preference: a
|
||||
* spawned peer has no human at its pane — the bridge spawned it — so one that stops at an
|
||||
* approval prompt is a wedged agent, indistinguishable from a legitimate mid-turn wait and
|
||||
* unable to end its turn with {@code bridge_reply}. opencode's own help calls this
|
||||
* "dangerous!", but the blast radius here is already bounded by design: a worker runs in its
|
||||
* own git worktree on its own branch, is off-subscription, and cannot merge — the lead is the
|
||||
* gate.
|
||||
*/
|
||||
private List<String> argvWithAuto(BridgedConfig.Profile cfg) {
|
||||
List<String> argv = mutableArgv(cfg.argv());
|
||||
argv.add("--auto");
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv plus, on a resumed spawn, opencode's {@code -s <id>} flag to continue a prior
|
||||
* conversation by its session id. {@code -s, --session <id>} resumes an existing session; on a
|
||||
* fresh spawn (no resume target) no flag is added, letting opencode start a brand-new session.
|
||||
* The id comes from the base launch spec.
|
||||
*/
|
||||
private List<String> argvWithResume(List<String> argv, String id) {
|
||||
if (id == null || id.isBlank()) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withResume = mutableArgv(argv);
|
||||
withResume.add("-s");
|
||||
withResume.add(id);
|
||||
return withResume;
|
||||
}
|
||||
|
||||
/** The launch argv plus, when a model is configured, the opencode {@code -m provider/model} flag. */
|
||||
private List<String> argvWithModel(List<String> argv, BridgedConfig.Profile cfg) {
|
||||
if (cfg.model() != null && !cfg.model().isBlank()) {
|
||||
argv.add("-m");
|
||||
argv.add(cfg.model());
|
||||
}
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* Write an ephemeral {@code opencode.json} (and the member-charter file it references) into a
|
||||
* fresh per-spawn directory under {@link #configRoot}, and return the config file's path for
|
||||
* {@code OPENCODE_CONFIG}. The dir is unique per spawn so concurrent workers never race on it;
|
||||
* it is best-effort cleaned on JVM exit (worker config is disposable — regenerated every spawn).
|
||||
*/
|
||||
private Path writeConfig(BridgedConfig.Profile cfg, String charterText) {
|
||||
try {
|
||||
Path dir = Files.createTempDirectory(configRoot, "bridged-opencode-");
|
||||
dir.toFile().deleteOnExit();
|
||||
|
||||
ObjectNode root = JSON.createObjectNode();
|
||||
root.put("$schema", "https://opencode.ai/config.json");
|
||||
// CB-523, opencode side: a worker that runs out of context dies mid-turn, and its reply
|
||||
// — the entire point of the turn — is lost with it. Auto-compaction is therefore not an
|
||||
// operator preference for a bridged worker, it is a condition of the turn contract.
|
||||
//
|
||||
// Stated deliberately even though it is redundant today: OPENCODE_CONFIG is MERGED over
|
||||
// ~/.config/opencode/config.json rather than replacing it, so a worker already inherits
|
||||
// an `auto: true` set at home. We do not want that inheritance to be what the guarantee
|
||||
// rests on — the home file is outside this repo, differs per machine, and is not ours.
|
||||
//
|
||||
// Know the cost before removing it: this key WINS over the home config (verified — an
|
||||
// OPENCODE_CONFIG value overrides the home value, it does not defer to it), so an
|
||||
// operator who sets `compaction.auto: false` at home cannot turn it off for bridged
|
||||
// workers. That is the intended trade for peers we spawn and whose turns we must land;
|
||||
// if per-profile control is ever wanted, add a profile knob rather than dropping this.
|
||||
root.putObject("compaction").put("auto", true);
|
||||
|
||||
if (charterText != null) {
|
||||
Path charter = dir.resolve("member-charter.md");
|
||||
Files.writeString(charter, charterText);
|
||||
charter.toFile().deleteOnExit();
|
||||
|
||||
root.putArray("instructions").add(charter.toAbsolutePath().toString());
|
||||
}
|
||||
|
||||
if (cfg.hasMcp()) {
|
||||
ObjectNode bridge = root.putObject("mcp").putObject("bridge");
|
||||
bridge.put("type", "remote");
|
||||
bridge.put("url", cfg.mcpUrl());
|
||||
bridge.put("enabled", true);
|
||||
}
|
||||
if (hasCustomProvider(cfg)) {
|
||||
addCustomProvider(root, cfg);
|
||||
}
|
||||
|
||||
Path cfgFile = dir.resolve("opencode.json");
|
||||
// Built with Jackson rather than string concatenation: the provider block is nested and
|
||||
// carries operator-supplied values (URL, model id, api key), so escaping must be real.
|
||||
Files.writeString(cfgFile, JSON.writerWithDefaultPrettyPrinter().writeValueAsString(root));
|
||||
cfgFile.toFile().deleteOnExit();
|
||||
return cfgFile;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException(
|
||||
"cannot write opencode config for profile " + cfg.profile(), e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Declare a custom OpenAI-compatible provider so the worker talks to a pinned endpoint (a local
|
||||
* vLLM, say) instead of opencode's default gateway (CB-508).
|
||||
*
|
||||
* <p>The provider id comes from the {@code provider/model} selector in {@code model:}, so one
|
||||
* field drives both the declaration and the {@code -m} flag and they cannot drift apart.
|
||||
*/
|
||||
private void addCustomProvider(ObjectNode root, BridgedConfig.Profile cfg) {
|
||||
String[] parts = splitModelSelector(cfg);
|
||||
String providerId = parts[0];
|
||||
String modelId = parts[1];
|
||||
|
||||
ObjectNode provider = root.putObject("provider").putObject(providerId);
|
||||
provider.put("npm", "@ai-sdk/openai-compatible");
|
||||
provider.put("name", providerId + " (bridged)");
|
||||
|
||||
ObjectNode options = provider.putObject("options");
|
||||
options.put("baseURL", openAiBaseUrl(cfg.baseUrl()));
|
||||
// vLLM and friends usually ignore the key, but the AI SDK still requires a non-empty one.
|
||||
String token = resolveEnv(cfg.tokenEnv());
|
||||
options.put("apiKey", (token == null || token.isBlank()) ? "bridged-local-noauth" : token);
|
||||
|
||||
provider.putObject("models").putObject(modelId).put("name", modelId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Split {@code model:} into its {@code provider} and {@code model} halves. A pinned endpoint
|
||||
* needs both, so a bare model name is rejected loudly rather than silently falling back to the
|
||||
* default gateway — a worker quietly talking to the wrong endpoint is the failure this avoids.
|
||||
*/
|
||||
private static String[] splitModelSelector(BridgedConfig.Profile cfg) {
|
||||
String model = cfg.model();
|
||||
int slash = model == null ? -1 : model.indexOf('/');
|
||||
if (model == null || model.isBlank() || slash <= 0 || slash == model.length() - 1) {
|
||||
throw new IllegalArgumentException(
|
||||
"profile " + cfg.profile() + " sets baseUrl (a pinned opencode endpoint) so"
|
||||
+ " model: must be \"<provider>/<model>\", e.g."
|
||||
+ " \"local-vllm/deepseek-v4-flash\"; got "
|
||||
+ (model == null ? "null" : '"' + model + '"'));
|
||||
}
|
||||
return new String[]{model.substring(0, slash), model.substring(slash + 1)};
|
||||
}
|
||||
|
||||
/**
|
||||
* The OpenAI-compatible base URL for {@code baseUrl}. A bare {@code host:port} gets {@code /v1}
|
||||
* appended (where these servers put the API); a URL that already carries a path is taken as-is,
|
||||
* so an endpoint mounted somewhere unusual is still reachable.
|
||||
*/
|
||||
private static String openAiBaseUrl(String baseUrl) {
|
||||
String trimmed = baseUrl.trim();
|
||||
while (trimmed.endsWith("/")) {
|
||||
trimmed = trimmed.substring(0, trimmed.length() - 1);
|
||||
}
|
||||
int schemeEnd = trimmed.indexOf("://");
|
||||
String afterScheme = schemeEnd < 0 ? trimmed : trimmed.substring(schemeEnd + 3);
|
||||
return afterScheme.contains("/") ? trimmed : trimmed + "/v1";
|
||||
}
|
||||
|
||||
/** Add lazy on-disk session discovery to the base handle. */
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
PeerHandle inner = super.spawn(req);
|
||||
return new SessionAwareHandle(inner, discovery, effectiveCwd(req));
|
||||
}
|
||||
|
||||
/**
|
||||
* A {@link PeerHandle} that delegates everything to the base's worker handle but resolves
|
||||
* {@link #agentSessionId()} lazily through opencode session discovery. Delegate-only, so the
|
||||
* base's id/terminalId/profile semantics (CB-519's host-unique routing key, herdr coordinates)
|
||||
* are untouched — only the opencode-specific identity answer is added. {@code sessionName()}
|
||||
* stays null: opencode has no display-name seam, so the logical name lives only in the bridge's
|
||||
* roster (see the SESSION_NAME capability).
|
||||
*/
|
||||
private static final class SessionAwareHandle implements PeerHandle {
|
||||
private final PeerHandle delegate;
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
private final String cwd;
|
||||
|
||||
SessionAwareHandle(PeerHandle delegate, OpenCodeSessionDiscovery discovery, String cwd) {
|
||||
this.delegate = delegate;
|
||||
this.discovery = discovery;
|
||||
this.cwd = cwd;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String id() {
|
||||
return delegate.id();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String terminalId() {
|
||||
return delegate.terminalId();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String profile() {
|
||||
return delegate.profile();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String sessionName() {
|
||||
return delegate.sessionName();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String agentSessionId() {
|
||||
// Lazy + retried, never a spawn-time blocker: opencode writes the session record only
|
||||
// when the session is first persisted, so null here is the correct interim answer and
|
||||
// the caller re-calls later (each call re-scans, picking up a record that has since
|
||||
// appeared).
|
||||
return discovery.sessionIdForDirectory(cwd);
|
||||
}
|
||||
}
|
||||
|
||||
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
|
||||
|
||||
/** Spawn a worker for the default profile in the resolved default cwd. */
|
||||
public Agent spawn() {
|
||||
return spawnInternal(null, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile (null → default) in the resolved default cwd. */
|
||||
public Agent spawn(String profileName) {
|
||||
return spawnInternal(profileName, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile with an explicit requested/caller cwd (CB-112). */
|
||||
public Agent spawn(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd);
|
||||
}
|
||||
|
||||
// --- capabilities --------------------------------------------------------------------------
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
Set<Capability> caps = EnumSet.of(Capability.MID_TURN_ASK, Capability.WORKTREE,
|
||||
Capability.ORPHAN_REAP, Capability.SESSION_RESUME);
|
||||
if (hasGitTokenProfile()) {
|
||||
caps.add(Capability.SELF_PR);
|
||||
}
|
||||
// Deliberately NOT SESSION_NAME: opencode has no display-name flag, so the bridge's logical
|
||||
// name can't surface in the peer's own UI — declaring the capability would hide that
|
||||
// asymmetry rather than make it honest. For opencode the name lives only in the bridge's
|
||||
// roster (see PeerHandle.sessionName() returning null).
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
|
||||
/** Whether any configured profile opts into a git-forge token (required for {@link Capability#SELF_PR}). */
|
||||
private boolean hasGitTokenProfile() {
|
||||
return profileConfigs().stream().anyMatch(BridgedConfig.Profile::hasGitToken);
|
||||
}
|
||||
|
||||
// --- CB-117 reap predicate (opencode prefix), kept for direct unit testing -----------------
|
||||
|
||||
/**
|
||||
* Whether {@code name} is an opencode bridge worker started by a <em>different</em> process than
|
||||
* {@code currentNonce}. A thin {@code opencode}-prefix binding of
|
||||
* {@link HerdrPeerLauncher#isForeignWorker(String, String, String)}.
|
||||
*/
|
||||
static boolean isForeignWorker(String name, String currentNonce) {
|
||||
return HerdrPeerLauncher.isForeignWorker(NAME_PREFIX, name, currentNonce);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
/**
|
||||
* Resolves the opencode session id for a bridged worker from opencode's on-disk storage — the
|
||||
* only place this adapter touches opencode's private layout, and deliberately the <em>only</em>
|
||||
* class that does.
|
||||
*
|
||||
* <p><strong>Why this is isolated behind one seam.</strong> The layout is version-coupled and not a
|
||||
* stable contract: opencode writes one JSON file per session under
|
||||
* {@code <storageRoot>/session/<projectID>/<ses_*.json>}, and each record carries a
|
||||
* {@code "version"} field (e.g. {@code "1.1.31"}), so the exact directory shape, file naming, and
|
||||
* field names can move between opencode releases. opencode also ships a headless HTTP server that
|
||||
* may supersede file scanning entirely. Everything this adapter knows about that private storage —
|
||||
* its shape, naming, and field names — lives here, so a layout change, or a switch to the HTTP
|
||||
* server, changes exactly one class and nothing in {@link OpenCodeLauncher}.
|
||||
*
|
||||
* <p>The determinism that makes this useful is structural, not a guess: every bridged worker runs
|
||||
* in its own unique git worktree, so the record's {@code directory} (its project root) equals the
|
||||
* worker's cwd identifies <em>its</em> session unambiguously. We match on {@code directory} rather
|
||||
* than diffing {@code opencode session list} before/after — that races under concurrent spawns, and
|
||||
* the CLI listing does not even show the directory.
|
||||
*
|
||||
* <p>All reads are best-effort and never throw: a missing or unreadable storage root, a record that
|
||||
* fails to parse, or a directory with no record yet all yield {@code null}, and the caller (the
|
||||
* session handle) treats that as "identity not resolved yet" and retries later.
|
||||
*/
|
||||
final class OpenCodeSessionDiscovery {
|
||||
|
||||
private final Path storageRoot; // e.g. ~/.local/share/opencode (injectable for tests)
|
||||
private final ObjectMapper json;
|
||||
|
||||
OpenCodeSessionDiscovery(Path storageRoot) {
|
||||
this.storageRoot = storageRoot;
|
||||
this.json = new ObjectMapper();
|
||||
}
|
||||
|
||||
/**
|
||||
* The opencode session id whose record references {@code directory} (the worker's cwd), or
|
||||
* {@code null} when no record matches yet. When several records share the directory — e.g.
|
||||
* repeated spawns into the same worktree — the <em>most recently modified</em> one wins: it is
|
||||
* the session the pane most likely corresponds to.
|
||||
*
|
||||
* <p>Never throws: a missing {@code storageRoot}, an unreadable/malformed record, or a
|
||||
* directory that has not been persisted yet all resolve to {@code null} rather than failing a
|
||||
* spawn. A bridged worker's session record is written lazily (when the session is first
|
||||
* persisted), so {@code null} here is the normal answer right after the pane is ready, and the
|
||||
* caller retries later.
|
||||
*
|
||||
* @param directory the worker's cwd, as resolved for this spawn
|
||||
* @return the matching session id, or {@code null} if none is known yet
|
||||
*/
|
||||
String sessionIdForDirectory(String directory) {
|
||||
if (directory == null || directory.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
Path sessionRoot = storageRoot.resolve("session");
|
||||
if (!Files.isDirectory(sessionRoot)) {
|
||||
return null;
|
||||
}
|
||||
String best = null;
|
||||
long bestMtime = Long.MIN_VALUE;
|
||||
try (Stream<Path> projectDirs = Files.list(sessionRoot)) {
|
||||
for (Path projectDir : projectDirs.filter(Files::isDirectory).toList()) {
|
||||
try (Stream<Path> records = Files.list(projectDir)) {
|
||||
for (Path record : records.toList()) {
|
||||
String id = matchId(record, directory);
|
||||
if (id == null) {
|
||||
continue;
|
||||
}
|
||||
long mtime = lastModifiedEpochMillis(record);
|
||||
if (mtime > bestMtime) {
|
||||
bestMtime = mtime;
|
||||
best = id;
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// one project dir unreadable — skip it; another may still match
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// storage root vanished or became unreadable — "no session known yet"
|
||||
return null;
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
/**
|
||||
* The record's session id when it references {@code directory}, else {@code null}. A record
|
||||
* that is not JSON, lacks {@code id}/{@code directory}, or points at a different directory is
|
||||
* simply not our session; a malformed one is skipped, never fatal.
|
||||
*/
|
||||
private String matchId(Path record, String directory) {
|
||||
try {
|
||||
JsonNode node = json.readTree(record.toFile());
|
||||
JsonNode id = node == null ? null : node.get("id");
|
||||
JsonNode dir = node == null ? null : node.get("directory");
|
||||
if (id == null || dir == null || !directory.equals(dir.asText())) {
|
||||
return null;
|
||||
}
|
||||
return id.asText();
|
||||
} catch (IOException e) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** The record's last-modified epoch ms, or {@code Long.MIN_VALUE} if unreadable (never wins). */
|
||||
private static long lastModifiedEpochMillis(Path record) {
|
||||
try {
|
||||
return Files.getLastModifiedTime(record).toMillis();
|
||||
} catch (IOException e) {
|
||||
return Long.MIN_VALUE;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,101 @@
|
||||
package dev.ltms.bridged.metrics;
|
||||
|
||||
import dev.ltms.bridged.msg.ReplyInbox;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* The daemon's metric definitions (CB-502) — one place where every series is named, described, and
|
||||
* (for gauges) bound to live state.
|
||||
*
|
||||
* <p>The set is deliberately small: each series maps to a failure mode this project has actually
|
||||
* hit, not to whatever was easy to count. The two worth watching in practice are
|
||||
* {@code bridged_sends_total{outcome="completion_fallback"}} — a rising share means turn detection
|
||||
* is degrading, the CB-115/116/118 failure family — and
|
||||
* {@code bridged_push_nudges_total{outcome="exhausted"}}, which means the primary stopped draining
|
||||
* its inbox and CB-307's active push gave up.
|
||||
*/
|
||||
public final class BridgedMetrics {
|
||||
|
||||
/** Counter: delegated sends by terminal outcome. */
|
||||
public static final String SENDS = "bridged_sends_total";
|
||||
/** Counter: worker replies by the path that carried them (rendezvous vs stranded-to-inbox). */
|
||||
public static final String REPLIES = "bridged_replies_total";
|
||||
/** Counter: push-loop nudges to the primary, by outcome. */
|
||||
public static final String PUSH_NUDGES = "bridged_push_nudges_total";
|
||||
/** Counter: idle-lead heartbeat nudges to the lead, by outcome (CB-551). */
|
||||
public static final String HEARTBEAT_NUDGES = "bridged_lead_heartbeat_nudges_total";
|
||||
/** Counter: spawn attempts by peer kind and outcome. */
|
||||
public static final String SPAWNS = "bridged_spawns_total";
|
||||
/** Counter: herdr socket calls by method and outcome. */
|
||||
public static final String HERDR_CALLS = "bridged_herdr_calls_total";
|
||||
/** Counter: rejected requests by reason (CB-501). */
|
||||
public static final String AUTH_FAILURES = "bridged_auth_failures_total";
|
||||
/** Gauge: session census by lifecycle state. */
|
||||
public static final String SESSIONS = "bridged_sessions";
|
||||
/** Gauge: undrained replies held per target. */
|
||||
public static final String INBOX_DEPTH = "bridged_inbox_depth";
|
||||
|
||||
private BridgedMetrics() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the registry with its help text and live gauges bound.
|
||||
*
|
||||
* @param sessions the authoritative session registry (census gauge)
|
||||
* @param inbox the reply inbox; only used for a depth gauge when it can be inspected
|
||||
*/
|
||||
public static Metrics create(SessionManager sessions, ReplyInbox inbox) {
|
||||
Metrics m = new Metrics();
|
||||
|
||||
m.describe(SENDS, "counter",
|
||||
"Delegated sends by terminal outcome (replied|completion_fallback|timeout|failed).");
|
||||
m.describe(REPLIES, "counter",
|
||||
"Worker replies by delivery path (rendezvous=resolved an open send, inbox=stranded and held).");
|
||||
m.describe(PUSH_NUDGES, "counter",
|
||||
"CB-307 push-loop nudges to the primary (delivered|exhausted).");
|
||||
m.describe(HEARTBEAT_NUDGES, "counter",
|
||||
"CB-551 idle-lead heartbeat nudges (delivered|failed|exhausted). Quiet-cap exhaustion "
|
||||
+ "means the lead idled with nothing pending and was told to stand down.");
|
||||
m.describe(SPAWNS, "counter",
|
||||
"Worker spawn attempts by peer kind and outcome (ready|timeout|guard_rejected).");
|
||||
m.describe(HERDR_CALLS, "counter",
|
||||
"herdr socket calls by method and outcome — the dependency everything else rests on.");
|
||||
m.describe(AUTH_FAILURES, "counter",
|
||||
"Requests refused by CB-501/505 (unauthenticated|forbidden).");
|
||||
m.describe(SESSIONS, "gauge",
|
||||
"Registered worker sessions by lifecycle state.");
|
||||
m.describe(INBOX_DEPTH, "gauge",
|
||||
"Replies held for a target that the primary has not drained. Steady state is 0; "
|
||||
+ "a target stuck above 0 means CB-307 delivery is not completing.");
|
||||
|
||||
// One gauge per state so a scrape shows the whole census even when a state is empty —
|
||||
// an absent series and a zero series read very differently on a dashboard.
|
||||
for (MemberSession.State state : MemberSession.State.values()) {
|
||||
String label = state.name().toLowerCase();
|
||||
m.gauge(SESSIONS, () -> countIn(sessions, state), "state", label);
|
||||
}
|
||||
|
||||
// Depth is per live session, so the label set is only known at scrape time. peek() is the
|
||||
// port's non-destructive read — scraping metrics must never ack a reply out of the inbox.
|
||||
m.collector(INBOX_DEPTH, "target", () -> {
|
||||
Map<String, Number> depths = new LinkedHashMap<>();
|
||||
for (MemberSession s : sessions.roster()) {
|
||||
String target = s.terminalId();
|
||||
if (target == null) {
|
||||
continue;
|
||||
}
|
||||
depths.put(target, inbox.peek(target).size());
|
||||
}
|
||||
return depths;
|
||||
});
|
||||
return m;
|
||||
}
|
||||
|
||||
private static long countIn(SessionManager sessions, MemberSession.State state) {
|
||||
return sessions.roster().stream().filter(s -> s.state() == state).count();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,181 @@
|
||||
package dev.ltms.bridged.metrics;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.NavigableMap;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ConcurrentSkipListMap;
|
||||
import java.util.concurrent.atomic.LongAdder;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The daemon's metric registry and Prometheus text renderer (CB-502).
|
||||
*
|
||||
* <p>Deliberately dependency-free. The roadmap's tech-stack table specified Micrometer, but this
|
||||
* pom already carries an unusual reconciliation burden (a hand-pinned {@code jackson-annotations}
|
||||
* to make the MCP SDK's Jackson 3 coexist with our Jackson 2, a Jetty BOM import to stop version
|
||||
* skew, and four documented accepted-CVE advisories), and the dependency CVE gate this project
|
||||
* mandates could not be run when this landed. The metric set is small and fully known, and
|
||||
* Prometheus text exposition is a stable, well-specified format — so the registry is ~100 lines
|
||||
* here instead of a new transitive tree. {@code GET /metrics} is the swap seam if Micrometer's
|
||||
* ecosystem is ever wanted.
|
||||
*
|
||||
* <p>Thread-safe: counters are {@link LongAdder} (built for contended increment), gauges are
|
||||
* supplier-backed so they read live state at scrape time rather than needing to be pushed.
|
||||
*/
|
||||
public final class Metrics {
|
||||
|
||||
/** Counter series, keyed by the fully-rendered {@code name{labels}} sample id. */
|
||||
private final NavigableMap<String, LongAdder> counters = new ConcurrentSkipListMap<>();
|
||||
/** Gauge series, evaluated at scrape time. */
|
||||
private final NavigableMap<String, Supplier<Number>> gauges = new ConcurrentSkipListMap<>();
|
||||
/** Gauge families whose label set is only known at scrape time, keyed by metric name. */
|
||||
private final NavigableMap<String, Collector> collectors = new ConcurrentSkipListMap<>();
|
||||
/** HELP/TYPE metadata, keyed by bare metric name. */
|
||||
private final Map<String, String[]> meta = new ConcurrentHashMap<>();
|
||||
|
||||
/** A gauge family whose series are discovered per scrape (one label, many values). */
|
||||
private record Collector(String labelName, Supplier<Map<String, Number>> samples) {
|
||||
}
|
||||
|
||||
/** Declare a metric's help text and type once, so the exposition carries HELP/TYPE lines. */
|
||||
public Metrics describe(String name, String type, String help) {
|
||||
meta.put(name, new String[]{type, help});
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Increment a counter by one. */
|
||||
public void inc(String name, String... labelPairs) {
|
||||
add(name, 1, labelPairs);
|
||||
}
|
||||
|
||||
/** Increment a counter by {@code delta}. */
|
||||
public void add(String name, long delta, String... labelPairs) {
|
||||
counters.computeIfAbsent(sample(name, labelPairs), _ -> new LongAdder()).add(delta);
|
||||
}
|
||||
|
||||
/**
|
||||
* Register a live gauge. The supplier is called at scrape time, so it reflects current state
|
||||
* (session census, inbox depth) without anything having to remember to update it.
|
||||
*/
|
||||
public void gauge(String name, Supplier<Number> value, String... labelPairs) {
|
||||
gauges.put(sample(name, labelPairs), value);
|
||||
}
|
||||
|
||||
/**
|
||||
* Register a gauge family whose label values are not known up front — inbox depth per target,
|
||||
* for instance, where the set of targets changes as workers come and go. The supplier returns
|
||||
* {@code labelValue → value} and is evaluated once per scrape.
|
||||
*/
|
||||
public void collector(String name, String labelName, Supplier<Map<String, Number>> samples) {
|
||||
collectors.put(name, new Collector(labelName, samples));
|
||||
}
|
||||
|
||||
/** Current value of a counter series — for assertions in tests. */
|
||||
public long count(String name, String... labelPairs) {
|
||||
LongAdder a = counters.get(sample(name, labelPairs));
|
||||
return a == null ? 0 : a.sum();
|
||||
}
|
||||
|
||||
/**
|
||||
* Render the Prometheus text exposition format (version 0.0.4): optional {@code # HELP} and
|
||||
* {@code # TYPE} lines per metric family, then one line per sample.
|
||||
*/
|
||||
public String render() {
|
||||
StringBuilder out = new StringBuilder(1024);
|
||||
String lastFamily = null;
|
||||
for (Map.Entry<String, LongAdder> e : counters.entrySet()) {
|
||||
lastFamily = emitHeader(out, e.getKey(), lastFamily);
|
||||
out.append(e.getKey()).append(' ').append(e.getValue().sum()).append('\n');
|
||||
}
|
||||
for (Map.Entry<String, Supplier<Number>> e : gauges.entrySet()) {
|
||||
lastFamily = emitHeader(out, e.getKey(), lastFamily);
|
||||
Number v;
|
||||
try {
|
||||
v = e.getValue().get();
|
||||
} catch (RuntimeException ex) {
|
||||
continue; // a broken gauge must never break the whole scrape
|
||||
}
|
||||
if (v == null) {
|
||||
continue;
|
||||
}
|
||||
out.append(e.getKey()).append(' ').append(format(v)).append('\n');
|
||||
}
|
||||
for (Map.Entry<String, Collector> e : collectors.entrySet()) {
|
||||
Map<String, Number> samples;
|
||||
try {
|
||||
samples = e.getValue().samples().get();
|
||||
} catch (RuntimeException ex) {
|
||||
continue; // a broken collector must never break the whole scrape
|
||||
}
|
||||
if (samples == null || samples.isEmpty()) {
|
||||
continue;
|
||||
}
|
||||
lastFamily = emitHeader(out, e.getKey(), lastFamily);
|
||||
// Sort so repeated scrapes are byte-stable and diffable.
|
||||
new java.util.TreeMap<>(samples).forEach((label, v) -> {
|
||||
if (v != null) {
|
||||
out.append(sample(e.getKey(), e.getValue().labelName(), label))
|
||||
.append(' ').append(format(v)).append('\n');
|
||||
}
|
||||
});
|
||||
}
|
||||
return out.toString();
|
||||
}
|
||||
|
||||
/** Emit HELP/TYPE when the sample starts a new metric family; returns the current family. */
|
||||
private String emitHeader(StringBuilder out, String sampleId, String lastFamily) {
|
||||
String family = familyOf(sampleId);
|
||||
if (family.equals(lastFamily)) {
|
||||
return lastFamily;
|
||||
}
|
||||
String[] m = meta.get(family);
|
||||
if (m != null) {
|
||||
out.append("# HELP ").append(family).append(' ').append(m[1]).append('\n');
|
||||
out.append("# TYPE ").append(family).append(' ').append(m[0]).append('\n');
|
||||
}
|
||||
return family;
|
||||
}
|
||||
|
||||
private static String familyOf(String sampleId) {
|
||||
int brace = sampleId.indexOf('{');
|
||||
return brace < 0 ? sampleId : sampleId.substring(0, brace);
|
||||
}
|
||||
|
||||
/** Whole numbers render without a decimal point; everything else as-is. */
|
||||
private static String format(Number v) {
|
||||
double d = v.doubleValue();
|
||||
return (d == Math.rint(d) && !Double.isInfinite(d))
|
||||
? Long.toString((long) d)
|
||||
: Double.toString(d);
|
||||
}
|
||||
|
||||
/** Build the {@code name{k="v",k2="v2"}} sample id; labels are sorted for stable output. */
|
||||
private static String sample(String name, String... labelPairs) {
|
||||
if (labelPairs == null || labelPairs.length == 0) {
|
||||
return name;
|
||||
}
|
||||
if (labelPairs.length % 2 != 0) {
|
||||
throw new IllegalArgumentException("labels must be key/value pairs, got " + labelPairs.length);
|
||||
}
|
||||
NavigableMap<String, String> sorted = new java.util.TreeMap<>();
|
||||
for (int i = 0; i < labelPairs.length; i += 2) {
|
||||
sorted.put(labelPairs[i], labelPairs[i + 1] == null ? "" : labelPairs[i + 1]);
|
||||
}
|
||||
StringBuilder sb = new StringBuilder(name.length() + 16 * sorted.size());
|
||||
sb.append(name).append('{');
|
||||
boolean first = true;
|
||||
for (Map.Entry<String, String> e : sorted.entrySet()) {
|
||||
if (!first) {
|
||||
sb.append(',');
|
||||
}
|
||||
first = false;
|
||||
sb.append(e.getKey()).append("=\"").append(escapeLabel(e.getValue())).append('"');
|
||||
}
|
||||
return sb.append('}').toString();
|
||||
}
|
||||
|
||||
/** Label values are escaped per the exposition format: backslash, quote, newline. */
|
||||
private static String escapeLabel(String v) {
|
||||
return v.replace("\\", "\\\\").replace("\"", "\\\"").replace("\n", "\\n");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,242 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import com.rabbitmq.client.AMQP;
|
||||
import com.rabbitmq.client.Channel;
|
||||
import com.rabbitmq.client.Connection;
|
||||
import com.rabbitmq.client.ConnectionFactory;
|
||||
import com.rabbitmq.client.DeliverCallback;
|
||||
import com.rabbitmq.client.Recoverable;
|
||||
import com.rabbitmq.client.RecoveryListener;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* AMQP-backed {@link ReplyInbox} (CB-307 Stage 2): genuine cross-restart durability behind the same
|
||||
* port {@link InMemoryReplyInbox} implements as soft state.
|
||||
*
|
||||
* <p><strong>Mapping — consume-and-hold with deferred manual ack.</strong> Each target has a durable
|
||||
* queue {@code agent.<target>.inbox}. The gateway that owns the target starts a manual-ack consumer
|
||||
* ({@link #own}) that pulls persistent messages off that queue into an in-memory <em>held</em> map
|
||||
* (keyed by {@code msgId}) but does <em>not</em> ack them. {@link #peek} returns that snapshot;
|
||||
* {@link #ack} acks the broker delivery-tag and drops the entry. Because messages stay unacked until
|
||||
* the owning gateway actually drains them, a crash (or a {@code java -jar} bounce) before caller-ack
|
||||
* leaves them on the broker — it redelivers on reconnect. That is the durability the in-memory
|
||||
* adapter cannot give, with the port contract preserved.
|
||||
*
|
||||
* <p><strong>Ownership is explicit.</strong> {@link #own} declares the queue and starts the consumer;
|
||||
* {@link #release} cancels it. {@link #publish} sends to the queue but does <em>not</em> imply ownership
|
||||
* and does not attach a consumer. This split is required by CB-308 federation, where one gateway may
|
||||
* publish to an agent owned by another gateway; in that case the publisher must not compete for
|
||||
* deliveries.
|
||||
*
|
||||
* <p><strong>Dedup.</strong> The consumer keys the held map by {@code msgId}; a redelivered duplicate
|
||||
* (at-least-once, or a producer double-publish) is acked-and-dropped on arrival, so it never
|
||||
* double-queues.
|
||||
*
|
||||
* <p><strong>Visibility.</strong> Unlike the in-memory adapter, publish → broker → consumer is
|
||||
* asynchronous, so a {@link #peek} immediately after {@link #publish} may not yet see the message
|
||||
* (broker delivery latency). Callers that need the reply drained poll (as the primary already does);
|
||||
* the contract test waits for visibility. This is inherent to broker-backed delivery, not a defect.
|
||||
*
|
||||
* <p>The default deploy targets LavinMQ; a stock RabbitMQ speaks the same AMQP 0-9-1 (URI-only swap),
|
||||
* so the {@code @Tag("contract")} integration test runs against a RabbitMQ container.
|
||||
*/
|
||||
public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(AmqpReplyInbox.class);
|
||||
|
||||
private static final String QUEUE_PREFIX = "agent.";
|
||||
private static final String QUEUE_SUFFIX = ".inbox";
|
||||
|
||||
private final Connection connection;
|
||||
private final Channel channel;
|
||||
/** All channel operations (publish/declare/ack/cancel) serialize on this — a Channel is not thread-safe. */
|
||||
private final Object channelLock = new Object();
|
||||
/** target → (msgId → held delivery). Per-target map is guarded by synchronizing on itself. */
|
||||
private final ConcurrentHashMap<String, LinkedHashMap<String, Held>> held = new ConcurrentHashMap<>();
|
||||
/** Targets whose queue is declared and consumer is running, mapped to their broker consumer tag. */
|
||||
private final ConcurrentHashMap<String, String> consumerTags = new ConcurrentHashMap<>();
|
||||
|
||||
/** A message pulled off the broker but not yet acked: its delivery-tag plus the port payload. */
|
||||
private record Held(long deliveryTag, InboxMessage message) {}
|
||||
|
||||
/** Connect to {@code uri} (e.g. {@code amqp://guest:guest@127.0.0.1:5672/}) and open the inbox. */
|
||||
public static AmqpReplyInbox open(String uri) {
|
||||
try {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares queues and re-attaches consumers.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
return new AmqpReplyInbox(factory.newConnection("bridged-reply-inbox"));
|
||||
} catch (Exception e) {
|
||||
throw new IllegalStateException("cannot connect to AMQP broker at " + uri, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** Wrap an already-open connection (injection seam for the contract test). */
|
||||
AmqpReplyInbox(Connection connection) {
|
||||
this.connection = connection;
|
||||
try {
|
||||
this.channel = connection.createChannel();
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot open AMQP channel", e);
|
||||
}
|
||||
// On automatic recovery the broker redelivers unacked messages with FRESH delivery-tags; the
|
||||
// tags we were holding are now stale. Drop the held snapshot so the re-attached consumer
|
||||
// repopulates it with valid tags (dedup by msgId still prevents any double-queue).
|
||||
if (connection instanceof Recoverable recoverable) {
|
||||
recoverable.addRecoveryListener(new RecoveryListener() {
|
||||
@Override
|
||||
public void handleRecovery(Recoverable recoverable) {
|
||||
held.clear();
|
||||
log.info("AMQP connection recovered; cleared held replies for fresh redelivery");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void handleRecoveryStarted(Recoverable recoverable) {
|
||||
// no-op: we act once recovery completes
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void own(String target) {
|
||||
synchronized (channelLock) {
|
||||
if (consumerTags.containsKey(target)) {
|
||||
return; // already owning this target
|
||||
}
|
||||
String queue = queueName(target);
|
||||
try {
|
||||
channel.queueDeclare(queue, true, false, false, null); // durable, non-exclusive, keep on idle
|
||||
String tag = channel.basicConsume(queue, false, deliverCallback(target), _ -> { });
|
||||
consumerTags.put(target, tag);
|
||||
log.debug("AMQP inbox owns queue {} for target {}", queue, target);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot own queue " + queue, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
synchronized (channelLock) {
|
||||
String tag = consumerTags.remove(target);
|
||||
held.remove(target); // stale delivery tags must not survive release
|
||||
if (tag == null) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
channel.basicCancel(tag);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot cancel consumer for " + target, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
AMQP.BasicProperties props = new AMQP.BasicProperties.Builder()
|
||||
.messageId(msgId)
|
||||
.deliveryMode(2) // persistent — survives a broker restart
|
||||
.contentType("text/plain")
|
||||
.build();
|
||||
try {
|
||||
synchronized (channelLock) {
|
||||
channel.basicPublish("", queueName(target), props, content.getBytes(StandardCharsets.UTF_8));
|
||||
}
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot publish reply to " + queueName(target), e);
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
return List.of();
|
||||
}
|
||||
synchronized (perTarget) {
|
||||
return perTarget.values().stream().map(Held::message).toList();
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
return;
|
||||
}
|
||||
Held h;
|
||||
synchronized (perTarget) {
|
||||
h = perTarget.remove(msgId);
|
||||
}
|
||||
if (h == null) {
|
||||
return; // never held (or already acked) — no-op
|
||||
}
|
||||
try {
|
||||
synchronized (channelLock) {
|
||||
channel.basicAck(h.deliveryTag(), false);
|
||||
}
|
||||
} catch (IOException e) {
|
||||
// Ack didn't reach the broker: restore the entry so a later ack (or a redelivery after
|
||||
// reconnect) can retry. Keeps the at-least-once contract — a reply is never silently lost.
|
||||
synchronized (perTarget) {
|
||||
perTarget.putIfAbsent(msgId, h);
|
||||
}
|
||||
throw new IllegalStateException("cannot ack reply " + msgId + " on " + queueName(target), e);
|
||||
}
|
||||
}
|
||||
|
||||
private DeliverCallback deliverCallback(String target) {
|
||||
return (_, delivery) -> {
|
||||
String msgId = delivery.getProperties().getMessageId();
|
||||
long tag = delivery.getEnvelope().getDeliveryTag();
|
||||
if (msgId == null || msgId.isBlank()) {
|
||||
msgId = Long.toHexString(tag); // synthesize an id so dedup still has a key
|
||||
}
|
||||
String content = new String(delivery.getBody(), StandardCharsets.UTF_8);
|
||||
var perTarget = held.computeIfAbsent(target, _ -> new LinkedHashMap<>());
|
||||
boolean duplicate;
|
||||
synchronized (perTarget) {
|
||||
if (perTarget.containsKey(msgId)) {
|
||||
duplicate = true;
|
||||
} else {
|
||||
perTarget.put(msgId, new Held(tag, new InboxMessage(msgId, target, content)));
|
||||
duplicate = false;
|
||||
}
|
||||
}
|
||||
if (duplicate) {
|
||||
// Redelivered duplicate: ack the new tag and drop it so the broker stops resending.
|
||||
synchronized (channelLock) {
|
||||
channel.basicAck(tag, false);
|
||||
}
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
private static String queueName(String target) {
|
||||
return QUEUE_PREFIX + target + QUEUE_SUFFIX;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
try {
|
||||
channel.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("AMQP channel close: {}", e.toString());
|
||||
}
|
||||
try {
|
||||
connection.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("AMQP connection close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* Soft-state {@link ReplyInbox} backed by a {@link ConcurrentHashMap} keyed by target session.
|
||||
* Per-target FIFO ordering (insertion order via {@link LinkedHashMap}). Dedup by {@code msgId}
|
||||
* within a target. Thread-safe for concurrent publish vs. drain.
|
||||
*
|
||||
* <p><strong>Ownership is explicit.</strong> {@link #own} marks a target as locally owned so that
|
||||
* {@link #peek} and {@link #ack} operate on it; {@link #publish} works whether or not the target is
|
||||
* owned. {@link #release} clears the local snapshot. This mirrors the AMQP adapter's contract so the
|
||||
* non-broker path stays interchangeable.
|
||||
*
|
||||
* <p><strong>This is soft-state, NOT persistence.</strong> Lost on a {@code java -jar} bounce — that
|
||||
* is correct and consistent with "bridged stays soft-state." The Stage-2 AMQP adapter replaces this.
|
||||
*/
|
||||
public final class InMemoryReplyInbox implements ReplyInbox {
|
||||
|
||||
private final ConcurrentHashMap<String, LinkedHashMap<String, InboxMessage>> store = new ConcurrentHashMap<>();
|
||||
private final Set<String> owned = ConcurrentHashMap.newKeySet();
|
||||
|
||||
@Override
|
||||
public void own(String target) {
|
||||
owned.add(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
owned.remove(target);
|
||||
store.remove(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
var perTarget = store.computeIfAbsent(target, _ -> new LinkedHashMap<>());
|
||||
//noinspection SynchronizationOnLocalVariableOrMethodParameter
|
||||
synchronized (perTarget) {
|
||||
perTarget.putIfAbsent(msgId, new InboxMessage(msgId, target, content));
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
if (!owned.contains(target)) {
|
||||
return List.of();
|
||||
}
|
||||
var perTarget = store.get(target);
|
||||
if (perTarget == null) {
|
||||
return List.of();
|
||||
}
|
||||
//noinspection SynchronizationOnLocalVariableOrMethodParameter
|
||||
synchronized (perTarget) {
|
||||
return List.copyOf(perTarget.values());
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
if (!owned.contains(target)) {
|
||||
return;
|
||||
}
|
||||
var perTarget = store.get(target);
|
||||
if (perTarget != null) {
|
||||
//noinspection SynchronizationOnLocalVariableOrMethodParameter
|
||||
synchronized (perTarget) {
|
||||
perTarget.remove(msgId);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,351 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* CB-551: an opt-in heartbeat that nudges the single idle lead back to work once it has been
|
||||
* continuously idle past a quiet period with no open {@code bridge_send} driving it.
|
||||
*
|
||||
* <p>Why this exists: the fleet is ONE lead + architects + workers, so an idle, stalled lead is a
|
||||
* single point of failure for the fleet's progress. {@link ReplyPushLoop} nudges the lead only when
|
||||
* a worker reply lands; this loop is the timer that catches the gap where nothing lands and the
|
||||
* lead simply sits idle with nobody to prompt it onward.
|
||||
*
|
||||
* <p>Mechanism mirrors {@link ReplyPushLoop}: a pure, unit-testable decision function
|
||||
* ({@link #decide(long, Long, int, AgentStatus, boolean, boolean, FleetState)}) plus a thin
|
||||
* scheduler around it. The loop is <em>entirely opt-in</em> ({@code leadHeartbeat:} in config); with
|
||||
* no such block it is never constructed, so upgrading the daemon cannot silently acquire a behaviour
|
||||
* that spends the operator's model subscription on its own initiative (constraint 1).
|
||||
*
|
||||
* <p>Four invariants keep it from becoming a runaway subscription burner:
|
||||
* <ol>
|
||||
* <li><b>Status-gated</b> — a {@code WORKING} lead is making progress and is never touched; only an
|
||||
* injectable (idle/done/blocked) lead is even considered (constraint 2).</li>
|
||||
* <li><b>Debounced</b> — a nudge only fires once the lead has been <em>continuously</em> idle past
|
||||
* {@code idleAfterSeconds}, so a lead that just finished a turn is not re-prompted into every
|
||||
* natural pause (constraint 3).</li>
|
||||
* <li><b>Bounded when quiet</b> — consecutive nudges that find no pending fleet state are capped at
|
||||
* {@code quietNudgeCap}, then the loop stops nudging until real state returns (constraint 4).</li>
|
||||
* <li><b>Never races {@link ReplyPushLoop}</b> — while that loop is actively nudging any target this
|
||||
* loop stands down, so two competing injections never start two turns in the same pane
|
||||
* (constraint 6).</li>
|
||||
* </ol>
|
||||
*/
|
||||
public final class LeadHeartbeatLoop {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadHeartbeatLoop.class);
|
||||
|
||||
/** Sentinel for {@link #idleSinceNanos}: the lead is not currently in an idle stretch. */
|
||||
private static final long NOT_IDLE = Long.MIN_VALUE;
|
||||
|
||||
private final PrimaryRegistry primaryRegistry;
|
||||
private final AgentControl agents;
|
||||
private final ReplyInbox inbox;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final ReplyPushLoop pushLoop;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final long idleAfterNanos;
|
||||
private final long backoffMs;
|
||||
private final int quietNudgeCap;
|
||||
private final Metrics metrics; // CB-512 pattern: nullable — no registry in unit tests
|
||||
|
||||
/** When the current idle stretch began (nanos), or {@link #NOT_IDLE}. Single scheduler thread only. */
|
||||
private long idleSinceNanos = NOT_IDLE;
|
||||
/** Consecutive nudges that found no pending fleet state. Single scheduler thread only. */
|
||||
private int quietCount = 0;
|
||||
|
||||
/** Constructor with an injectable clock and no metric registry (unit tests, or wiring that opts out). */
|
||||
public LeadHeartbeatLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
Supplier<List<MemberSession>> roster, ReplyPushLoop pushLoop,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock,
|
||||
long idleAfterNanos, long backoffMs, int quietNudgeCap) {
|
||||
this(primaryRegistry, agents, inbox, roster, pushLoop, scheduler, clock,
|
||||
idleAfterNanos, backoffMs, quietNudgeCap, null);
|
||||
}
|
||||
|
||||
/** As above, with a metric registry (the CB-512 pattern) so nudge outcomes are counted. */
|
||||
public LeadHeartbeatLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
Supplier<List<MemberSession>> roster, ReplyPushLoop pushLoop,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock,
|
||||
long idleAfterNanos, long backoffMs, int quietNudgeCap, Metrics metrics) {
|
||||
this.primaryRegistry = primaryRegistry;
|
||||
this.agents = agents;
|
||||
this.inbox = inbox;
|
||||
this.roster = roster;
|
||||
this.pushLoop = pushLoop;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.idleAfterNanos = idleAfterNanos;
|
||||
this.backoffMs = backoffMs;
|
||||
this.quietNudgeCap = quietNudgeCap;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Count one nudge outcome when a registry is wired; a no-op in unit tests. */
|
||||
private void countNudge(String outcome) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(BridgedMetrics.HEARTBEAT_NUDGES, "outcome", outcome);
|
||||
}
|
||||
}
|
||||
|
||||
// --- decision logic (package-private for unit-testing) --------------------------------------
|
||||
|
||||
/** The action the loop should take on one tick. */
|
||||
enum Action {
|
||||
/** Send a heartbeat nudge to the lead now. */
|
||||
INJECT,
|
||||
/** Lead is idle but not yet past the quiet period — keep waiting, do nothing. */
|
||||
WAIT_IDLE,
|
||||
/** Lead is making progress (or its status cannot be read) — reset the idle window and re-arm. */
|
||||
LEAD_BUSY,
|
||||
/** Idle past the quiet period with nothing pending and the cap exhausted — stop until new state appears. */
|
||||
QUIET_DONE,
|
||||
/** {@link ReplyPushLoop} is actively nudging — stand aside rather than start a competing turn. */
|
||||
STAND_DOWN
|
||||
}
|
||||
|
||||
/** The outcome of one decision: the action plus the state to persist for the next tick. */
|
||||
record Decision(Action action, Long idleSinceNanos, int quietCount) {}
|
||||
|
||||
/**
|
||||
* Pure decision function: given the current loop state and fleet/lead facts, return what to do
|
||||
* next and the exact state to carry forward. Pure means no I/O and no mutation — the caller
|
||||
* ({@link #tick}) applies {@link Decision} to its fields. This is what makes the loop unit-testable
|
||||
* with a fake clock and no sleeping.
|
||||
*
|
||||
* @param nowNanos the current time (an injected clock in tests, {@link System#nanoTime()} live)
|
||||
* @param idleSinceNanos when the current idle stretch began, or {@code null} if the lead is not idle
|
||||
* @param quietCount consecutive nudges so far that found no pending fleet state
|
||||
* @param status the lead's herdr status
|
||||
* @param pushLoopActive whether {@link ReplyPushLoop} is currently nudging some target (constraint 6)
|
||||
* @param leadKnown whether a lead terminal is known to nudge at all
|
||||
* @param fleet a snapshot of the pending fleet state (constraint 5)
|
||||
* @return the action to take and the state to persist
|
||||
*/
|
||||
Decision decide(long nowNanos, Long idleSinceNanos, int quietCount, AgentStatus status,
|
||||
boolean pushLoopActive, boolean leadKnown, FleetState fleet) {
|
||||
// Constraint 6: while ReplyPushLoop is actively nudging the lead, injecting a second,
|
||||
// competing prompt into the same pane would start a second turn — racing loops multiply
|
||||
// turns and context burn. Stand aside, and treat the active push as real state (re-arm the
|
||||
// quiet counter), because the reply that drove it is exactly the kind of new state that
|
||||
// should reset the cap.
|
||||
if (pushLoopActive) {
|
||||
return new Decision(Action.STAND_DOWN, idleSinceNanos, 0);
|
||||
}
|
||||
// Constraint 2: a WORKING lead is making progress and must NOT be touched; an unreadable
|
||||
// status (read failure, or the agent is gone) is safest treated the same way — never inject
|
||||
// into a state we cannot read. Either way, reset the idle window and the quiet counter: the
|
||||
// lead was / may be active, so the next idle stretch must count its own quiet period fresh.
|
||||
if (status == null || !status.injectable()) {
|
||||
return new Decision(Action.LEAD_BUSY, null, 0);
|
||||
}
|
||||
if (idleSinceNanos == null) {
|
||||
// The lead just became injectable — record the start of an idle stretch and wait out the
|
||||
// debounce quiet period before ever nudging (constraint 3).
|
||||
return new Decision(Action.WAIT_IDLE, nowNanos, quietCount);
|
||||
}
|
||||
if (nowNanos - idleSinceNanos < idleAfterNanos) {
|
||||
// Still within the quiet period: the lead that just finished a turn sits momentarily idle
|
||||
// and must not be re-prompted into every natural pause.
|
||||
return new Decision(Action.WAIT_IDLE, idleSinceNanos, quietCount);
|
||||
}
|
||||
// Past the quiet period with an injectable lead: it is a genuine candidate for a nudge. Two
|
||||
// gating facts decide whether and how:
|
||||
if (!leadKnown) {
|
||||
// No lead terminal is known yet (e.g. the registry has not learned one) — there is nobody
|
||||
// to nudge. Keep waiting; the window stays open so discovery re-arms it without a fresh
|
||||
// quiet period.
|
||||
return new Decision(Action.WAIT_IDLE, idleSinceNanos, quietCount);
|
||||
}
|
||||
if (fleet.hasPending()) {
|
||||
// Real fleet state is waiting — a worker reply or a DONE session. This is new state, so
|
||||
// it resets the quiet counter (constraint 4) and the lead is nudged to go collect it.
|
||||
return new Decision(Action.INJECT, idleSinceNanos, 0);
|
||||
}
|
||||
if (quietCount < quietNudgeCap) {
|
||||
// Nothing is pending, but the cap is not exhausted: nudge anyway, telling the lead
|
||||
// exactly that nothing is waiting so it can choose to stand down rather than hunt
|
||||
// (constraint 5). Count it toward the consecutive-quiet cap.
|
||||
return new Decision(Action.INJECT, idleSinceNanos, quietCount + 1);
|
||||
}
|
||||
// Nothing pending and the cap is exhausted: stop nudging until real state appears again
|
||||
// (constraint 4). The loop still ticks on backoff so a genuinely new reply or session change
|
||||
// re-arms it — QUIET_DONE stops injection, not observation.
|
||||
return new Decision(Action.QUIET_DONE, idleSinceNanos, quietCount);
|
||||
}
|
||||
|
||||
// --- loop ----------------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Start the heartbeat. Schedules the first tick after one backoff so a freshly-booting daemon
|
||||
* does not evaluate the lead's idle state before the fleet has settled.
|
||||
*/
|
||||
public void start() {
|
||||
log.info("idle-lead heartbeat: on — nudge lead after {}s idle (recheck {}ms, quiet cap {})",
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleAfterNanos), backoffMs, quietNudgeCap);
|
||||
scheduler.schedule(this::tick, backoffMs, TimeUnit.MILLISECONDS);
|
||||
}
|
||||
|
||||
/** One loop tick, every {@link #backoffMs} — the thin scheduler around {@link #decide}. */
|
||||
private void tick() {
|
||||
boolean leadKnown = primaryRegistry.primaryTerminal().isPresent();
|
||||
FleetState fleet = snapshot(inbox, roster);
|
||||
AgentStatus status = AgentStatus.UNKNOWN;
|
||||
if (leadKnown) {
|
||||
try {
|
||||
status = agents.status(primaryRegistry.primaryTerminal().orElseThrow());
|
||||
} catch (RuntimeException e) {
|
||||
// A failed status read degrades to "unknown" — decide() treats that like a busy lead
|
||||
// and never injects into a state it cannot read. Retry on the next backoff.
|
||||
log.debug("idle-heartbeat: status check failed for lead, will retry: {}", e.toString());
|
||||
}
|
||||
}
|
||||
|
||||
Decision d = decide(clock.getAsLong(),
|
||||
idleSinceNanos == NOT_IDLE ? null : idleSinceNanos,
|
||||
quietCount, status, pushLoop.isActive(), leadKnown, fleet);
|
||||
applyDecision(d);
|
||||
switch (d.action()) {
|
||||
case INJECT -> injectNudge(fleet);
|
||||
case QUIET_DONE -> countNudge("exhausted");
|
||||
case WAIT_IDLE, LEAD_BUSY, STAND_DOWN -> { /* nothing to inject, nothing to count */ }
|
||||
}
|
||||
scheduleNext();
|
||||
}
|
||||
|
||||
/** Persist the state a decision returned, so the next tick starts from it. */
|
||||
private void applyDecision(Decision d) {
|
||||
idleSinceNanos = d.idleSinceNanos() == null ? NOT_IDLE : d.idleSinceNanos();
|
||||
quietCount = d.quietCount();
|
||||
}
|
||||
|
||||
/** Send the nudge to the known lead. */
|
||||
private void injectNudge(FleetState fleet) {
|
||||
var lead = primaryRegistry.primaryTerminal();
|
||||
if (lead.isEmpty()) {
|
||||
return; // the lead disappeared between the decision and the injection
|
||||
}
|
||||
String leadTerminal = lead.get();
|
||||
try {
|
||||
agents.send(leadTerminal, fleet.nudgeText());
|
||||
log.debug("idle-heartbeat: nudge sent to lead {} (quiet nudges so far in this stretch: {})",
|
||||
leadTerminal, quietCount);
|
||||
countNudge("delivered");
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("idle-heartbeat: failed to nudge lead {}: {}", leadTerminal, e.toString());
|
||||
countNudge("failed");
|
||||
}
|
||||
}
|
||||
|
||||
/** Schedule the next tick on the scheduler thread pool. */
|
||||
private void scheduleNext() {
|
||||
scheduler.schedule(this::tick, backoffMs, TimeUnit.MILLISECONDS);
|
||||
}
|
||||
|
||||
// --- lifecycle -----------------------------------------------------------------------------
|
||||
|
||||
/** Shut down the scheduler; outstanding ticks are cancelled. */
|
||||
public void stop() {
|
||||
scheduler.shutdownNow();
|
||||
}
|
||||
|
||||
/** @see #stop() */
|
||||
public void close() {
|
||||
stop();
|
||||
}
|
||||
|
||||
/**
|
||||
* A read-only snapshot of the fleet facts the heartbeat reports to an idle lead. Carries exact
|
||||
* counts and names the targets that hold replies, so the nudge is actionable rather than a
|
||||
* content-free "keep working" that would cost the lead a turn just to discover what there is to
|
||||
* do (constraint 5).
|
||||
*
|
||||
* @param pendingReplies total undrained worker replies across owned targets
|
||||
* @param doneSessions session count in the DONE state — finished turns awaiting teardown
|
||||
* @param liveWorkers sessions still registered (acquired minus released)
|
||||
* @param replyTargets the terminals whose inboxes currently hold at least one reply
|
||||
*/
|
||||
record FleetState(int pendingReplies, int doneSessions, int liveWorkers, List<String> replyTargets) {
|
||||
|
||||
/** True when the lead has something real to collect — a reply or a session awaiting teardown. */
|
||||
boolean hasPending() {
|
||||
return pendingReplies > 0 || doneSessions > 0;
|
||||
}
|
||||
|
||||
/** The nudge body, phrased for the two cases the heartbeat distinguishes. */
|
||||
String nudgeText() {
|
||||
StringBuilder sb = new StringBuilder(
|
||||
"Heartbeat: you are idle and no bridge_send is waiting on you.");
|
||||
if (hasPending()) {
|
||||
sb.append(" The fleet has state to collect: ").append(pendingDetail());
|
||||
} else {
|
||||
sb.append(" Fleet is quiet — nothing pending to collect (")
|
||||
.append(pendingReplies).append(" worker replies, ")
|
||||
.append(doneSessions).append(" DONE sessions); ")
|
||||
.append(liveWorkers).append(" workers live. You may stand down until new state re-arms me.");
|
||||
}
|
||||
return sb.toString();
|
||||
}
|
||||
|
||||
private String pendingDetail() {
|
||||
StringBuilder sb = new StringBuilder();
|
||||
sb.append(pendingReplies).append(" worker ")
|
||||
.append(pendingReplies == 1 ? "reply" : "replies")
|
||||
.append(" pending collection");
|
||||
if (!replyTargets.isEmpty()) {
|
||||
// Render each as the exact command so the lead can act without parsing: the nearest
|
||||
// analogue to ReplyPushLoop's bridge_poll(target=...) nudge.
|
||||
sb.append(" (")
|
||||
.append(String.join(", ",
|
||||
replyTargets.stream().map(t -> "bridge_poll(target=" + t + ")").toList()))
|
||||
.append(")");
|
||||
}
|
||||
sb.append(", ").append(doneSessions).append(" DONE session").append(doneSessions == 1 ? "" : "s")
|
||||
.append(" awaiting teardown")
|
||||
.append(", ").append(liveWorkers).append(" worker").append(liveWorkers == 1 ? "" : "s")
|
||||
.append(" live");
|
||||
return sb.toString();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a {@link FleetState} snapshot over the live roster and the reply inbox — the same
|
||||
* sources the metrics collector uses, so the heartbeat reports facts the operator can cross-check
|
||||
* against {@code /metrics}. Package-private and pure (reads only) so it is unit-testable.
|
||||
*/
|
||||
static FleetState snapshot(ReplyInbox inbox, Supplier<List<MemberSession>> roster) {
|
||||
List<MemberSession> sessions = roster.get();
|
||||
int replies = 0;
|
||||
int done = 0;
|
||||
List<String> replyTargets = new ArrayList<>();
|
||||
for (MemberSession s : sessions) {
|
||||
if (s.terminalId() == null) {
|
||||
continue;
|
||||
}
|
||||
int depth = inbox.peek(s.terminalId()).size();
|
||||
if (depth > 0) {
|
||||
replies += depth;
|
||||
replyTargets.add(s.terminalId());
|
||||
}
|
||||
if (s.state() == MemberSession.State.DONE) {
|
||||
done++;
|
||||
}
|
||||
}
|
||||
return new FleetState(replies, done, sessions.size(), List.copyOf(replyTargets));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,700 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.CompletionException;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ExecutionException;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.TimeoutException;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.locks.ReentrantLock;
|
||||
|
||||
/**
|
||||
* The blocking delegation feature (CB-104): deliver {@code content} into a worker and block until
|
||||
* the worker returns a <em>structured reply</em> via {@code bridge_reply} (the {@link Rendezvous}),
|
||||
* then hand that reply back. Delivery is the {@link Injector}'s job (the background poller sends it
|
||||
* when the worker is injectable); this service never drives the injector or scrapes the terminal —
|
||||
* completion is the worker's explicit reply, not a guess about {@code agent_status}.
|
||||
*
|
||||
* <p>Sends are serialized per session so exactly one reply can be outstanding per worker, which is
|
||||
* what lets a reply map unambiguously to its send (no cross-talk between concurrent callers).
|
||||
*
|
||||
* <p>If the worker never replies within the timeout, the caller gets a typed "still working" /
|
||||
* "queued" outcome — the message may still be mid-flight. A finished-but-unreplied turn is caught
|
||||
* by the CB-106 completion fallback (see {@link Rendezvous#resolveCompletion}).
|
||||
*
|
||||
* <p><strong>Async fire-and-poll (CB-107).</strong> A caller's MCP client caps a blocking call at
|
||||
* ~60s, but a real delegated task runs for minutes. {@link #sendAsync} therefore runs the same
|
||||
* blocking {@link #send} on a background virtual thread and hands back a <em>ticket</em> the caller
|
||||
* polls with {@link #poll}. The blocking and async paths share one code path (and the same per-target
|
||||
* serialization), so async inherits the reply + completion resolution behaviour for free.
|
||||
*/
|
||||
public final class MessageService {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(MessageService.class);
|
||||
|
||||
/**
|
||||
* The window a fire-and-poll send waits for resolution — generous, since no caller is blocked on
|
||||
* it; a real delegated task resolves (reply or completion) well within this, and only a genuinely
|
||||
* hung worker rides it out.
|
||||
*/
|
||||
private static final long ASYNC_TIMEOUT_MS = 30 * 60 * 1_000L;
|
||||
|
||||
/** How long a finished (terminal) ticket is retained for polling before it is pruned. */
|
||||
private static final long TICKET_TTL_NANOS = 10 * 60 * 1_000_000_000L;
|
||||
|
||||
/** Outcome of a blocking send. */
|
||||
public enum Outcome {
|
||||
/** The worker called {@code bridge_reply}; {@code text} holds the structured answer. */
|
||||
REPLIED,
|
||||
/**
|
||||
* The worker's delegated turn finished without a {@code bridge_reply} (CB-106 fallback);
|
||||
* {@code text} is the scraped transcript tail rather than a structured answer.
|
||||
*/
|
||||
COMPLETED_UNREPLIED,
|
||||
/**
|
||||
* The worker ran the turn then wedged in an unrecoverable state (CB-109); {@code text} is the
|
||||
* failure context (e.g. the error screen). Terminal, but not a successful completion.
|
||||
*/
|
||||
WORKER_FAILED,
|
||||
/**
|
||||
* The turn finished without a {@code bridge_reply} and the scrape matched the backend's
|
||||
* configured usage-limit refusal pattern (CB-578 stage A); {@code text} is the reason,
|
||||
* carrying the matched line. The worker's pane is healthy — only its account is refusing —
|
||||
* so this is never reported as a completed reply, and is kept distinct from
|
||||
* {@link #WORKER_FAILED} (a wedged worker) and a session simply going {@code GONE}.
|
||||
*/
|
||||
BACKEND_EXHAUSTED,
|
||||
/**
|
||||
* The worker paused mid-turn to ask the primary a question (CB-205); {@code text} is the
|
||||
* question and {@code turnId} correlates the answer. Not terminal — the primary answers with
|
||||
* {@link #answer(String, String, long)} and the turn resumes.
|
||||
*/
|
||||
QUESTION,
|
||||
/** Timed out after the message was delivered — the worker is still working. */
|
||||
TIMED_OUT_WORKING,
|
||||
/** Timed out before delivery — the message is still queued for the worker. */
|
||||
TIMED_OUT_QUEUED,
|
||||
/** Another send to this session was in flight for the whole window. */
|
||||
BUSY,
|
||||
/**
|
||||
* An answer ({@link #answer(String, String, long)}) referenced a {@code turnId} that is no
|
||||
* longer open — the worker's {@code bridge_ask} already timed out or was answered.
|
||||
*/
|
||||
STALE_TURN
|
||||
}
|
||||
|
||||
/**
|
||||
* @param outcome how the send ended (or paused)
|
||||
* @param text the worker's answer when {@link #completed()} (a structured {@code bridge_reply}
|
||||
* for {@link Outcome#REPLIED}, a scraped transcript tail for
|
||||
* {@link Outcome#COMPLETED_UNREPLIED}), or the question for {@link Outcome#QUESTION},
|
||||
* else {@code null}
|
||||
* @param turnId correlation id for a {@link Outcome#QUESTION} (answered via
|
||||
* {@link #answer(String, String, long)}), else {@code null}
|
||||
*/
|
||||
public record Reply(Outcome outcome, String text, String turnId) {
|
||||
/** A reply with no correlation id (the common terminal outcomes). */
|
||||
public Reply(Outcome outcome, String text) {
|
||||
this(outcome, text, null);
|
||||
}
|
||||
|
||||
/** Whether the worker's turn actually finished with an answer (replied or scraped). */
|
||||
public boolean completed() {
|
||||
return outcome == Outcome.REPLIED || outcome == Outcome.COMPLETED_UNREPLIED;
|
||||
}
|
||||
}
|
||||
|
||||
/** How a worker's {@code bridge_ask} (CB-205) resolved. */
|
||||
public enum AskOutcome {
|
||||
/** The primary answered; {@link AskResult#answer} carries it. */
|
||||
ANSWERED,
|
||||
/** No delegation was open to surface the question to — the worker has no one to ask. */
|
||||
NO_WAITER,
|
||||
/** The primary did not answer within the window. */
|
||||
TIMED_OUT
|
||||
}
|
||||
|
||||
/** The outcome of a worker's {@code bridge_ask}: how it resolved and (if answered) the answer. */
|
||||
public record AskResult(AskOutcome outcome, String answer) {
|
||||
}
|
||||
|
||||
/** Lifecycle phase of an async delegation ticket. */
|
||||
public enum Phase {
|
||||
/** Delegated and in flight — queued for the worker or being worked. */
|
||||
PENDING,
|
||||
/** The worker is paused in {@code bridge_ask}; {@link TaskView#reply} and {@link TaskView#turnId} identify it. */
|
||||
ASKING,
|
||||
/** The worker's turn finished; {@link TaskView#reply} holds the answer. */
|
||||
DONE,
|
||||
/** The delegation could not complete (timed out, worker gone, or busy). */
|
||||
FAILED
|
||||
}
|
||||
|
||||
/**
|
||||
* A poll snapshot of an async delegation.
|
||||
*
|
||||
* @param reply the answer when {@link #phase} is {@link Phase#DONE}, or the question when
|
||||
* {@link #phase} is {@link Phase#ASKING}; otherwise {@code null}
|
||||
* @param replySource {@code "reply"} (structured {@code bridge_reply}) or {@code "transcript"}
|
||||
* (completion scrape) when {@link Phase#DONE}, else {@code null}
|
||||
* @param detail a human note (live worker status while pending, ask state, or failure reason)
|
||||
* @param turnId correlation id for an {@link Phase#ASKING} ticket, else {@code null}
|
||||
*/
|
||||
public record TaskView(String ticket, Phase phase, String reply, String replySource, String detail,
|
||||
String turnId) {
|
||||
}
|
||||
|
||||
/** An in-flight or finished async delegation, keyed by its ticket. */
|
||||
private static final class Task {
|
||||
private final String target;
|
||||
private final CompletableFuture<Reply> future = new CompletableFuture<>();
|
||||
private final long createdNanos = System.nanoTime();
|
||||
private volatile Reply question;
|
||||
private volatile String turnId;
|
||||
|
||||
private Task(String target) {
|
||||
this.target = target;
|
||||
}
|
||||
}
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Injector injector;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ReplyInbox inbox;
|
||||
private final ReplyPushLoop pushLoop;
|
||||
private final Metrics metrics; // CB-502: nullable — no registry in unit tests
|
||||
private final ConcurrentHashMap<String, ReentrantLock> sessionLocks = new ConcurrentHashMap<>();
|
||||
private final ConcurrentHashMap<String, Task> tasks = new ConcurrentHashMap<>();
|
||||
/** Async task that owns each exact forward rendezvous waiter. */
|
||||
private final ConcurrentHashMap<CompletableFuture<Rendezvous.Resolution>, Task> asyncTasksByWaiter =
|
||||
new ConcurrentHashMap<>();
|
||||
/** Async tickets paused on a specific {@code bridge_ask} turn. */
|
||||
private final ConcurrentHashMap<String, Task> asyncTasksByTurn = new ConcurrentHashMap<>();
|
||||
private final AtomicLong ticketSeq = new AtomicLong();
|
||||
private final ExecutorService asyncExecutor = Executors.newThreadPerTaskExecutor(
|
||||
Thread.ofVirtual().name("bridge-async-", 0).factory());
|
||||
|
||||
/**
|
||||
* Create with an explicit {@link ReplyInbox} and optional {@link ReplyPushLoop}.
|
||||
*
|
||||
* @param pushLoop nullable — when non-null, the push loop is notified on the no-waiter reply
|
||||
* branch ({@link #reply}) so it can nudge the primary to drain the inbox
|
||||
*/
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous,
|
||||
ReplyInbox inbox, ReplyPushLoop pushLoop) {
|
||||
this(agents, injector, rendezvous, inbox, pushLoop, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, with a metric registry (CB-502). Instrumenting here rather than at the REST and MCP
|
||||
* edges means both surfaces are counted by one piece of code and cannot drift.
|
||||
*
|
||||
* @param metrics nullable — when null, nothing is recorded
|
||||
*/
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous,
|
||||
ReplyInbox inbox, ReplyPushLoop pushLoop, Metrics metrics) {
|
||||
this.agents = agents;
|
||||
this.injector = injector;
|
||||
this.rendezvous = rendezvous;
|
||||
this.inbox = inbox;
|
||||
this.pushLoop = pushLoop;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Create with an explicit {@link ReplyInbox} and no push loop. */
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous, ReplyInbox inbox) {
|
||||
this(agents, injector, rendezvous, inbox, null);
|
||||
}
|
||||
|
||||
/** Backward-compatible constructor that uses a default {@link InMemoryReplyInbox}. */
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous) {
|
||||
this(agents, injector, rendezvous, new InMemoryReplyInbox());
|
||||
}
|
||||
|
||||
/** Current lifecycle status of a worker (the {@code GET /sessions/{id}/status} surface). */
|
||||
public AgentStatus status(String target) {
|
||||
return agents.status(target);
|
||||
}
|
||||
|
||||
/** Read-only delegation fact for fleet views. */
|
||||
public boolean hasAcceptedDelivery(String target) {
|
||||
return rendezvous.isWaiting(target);
|
||||
}
|
||||
|
||||
/** Read-only inbox fact for fleet views. */
|
||||
public boolean hasInboxMessage(String target) {
|
||||
return !inbox.peek(target).isEmpty();
|
||||
}
|
||||
|
||||
/**
|
||||
* Route a worker's explicit {@code bridge_reply}: resolve an open send, or queue it in the
|
||||
* inbox if no send is currently open. Unlike the bare {@link Rendezvous#resolve}, a no-waiter
|
||||
* result is <em>not</em> a failure — the reply is held for later drain.
|
||||
*
|
||||
* <p><strong>Do NOT use this for mid-turn questions.</strong> {@code bridge_ask} /
|
||||
* {@link Rendezvous#resolveQuestion} must keep today's {@code NO_WAITER} behaviour — questions
|
||||
* are interactive and must never be queued.
|
||||
*
|
||||
* @return always {@code true} — the reply either resolved a live send or was queued
|
||||
*/
|
||||
public boolean reply(String session, String content) {
|
||||
if (rendezvous.resolve(session, content)) {
|
||||
count(BridgedMetrics.REPLIES, "path", "rendezvous");
|
||||
return true; // a live send took it — unchanged fast path
|
||||
}
|
||||
inbox.publish(session, UUID.randomUUID().toString(), content);
|
||||
// A rising inbox share is the signal CB-307 exists to make visible: the worker finished but
|
||||
// nobody was waiting, so delivery now depends on the push loop and a drain.
|
||||
count(BridgedMetrics.REPLIES, "path", "inbox");
|
||||
if (pushLoop != null) {
|
||||
pushLoop.onReplyQueued(session);
|
||||
}
|
||||
return true; // held, not lost
|
||||
}
|
||||
|
||||
/** Record a counter sample when a registry is wired; a no-op in unit tests. */
|
||||
private void count(String name, String... labels) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(name, labels);
|
||||
}
|
||||
}
|
||||
|
||||
/** Count a send's terminal outcome and pass the reply through unchanged. */
|
||||
private Reply recorded(Reply r) {
|
||||
String label = sendOutcomeLabel(r.outcome());
|
||||
if (label != null) {
|
||||
count(BridgedMetrics.SENDS, "outcome", label);
|
||||
}
|
||||
return r;
|
||||
}
|
||||
|
||||
/** Map a terminal send outcome to its metric label, or {@code null} for non-terminal ones. */
|
||||
private static String sendOutcomeLabel(Outcome o) {
|
||||
return switch (o) {
|
||||
case REPLIED -> "replied";
|
||||
case COMPLETED_UNREPLIED -> "completion_fallback";
|
||||
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, BUSY -> "timeout";
|
||||
case WORKER_FAILED -> "failed";
|
||||
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
||||
case STALE_TURN, QUESTION -> null; // not a completed delegation
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Abandon any send still waiting on {@code target} because its session has gone away (CB-516).
|
||||
*
|
||||
* <p>Without this, tearing a worker down left its rendezvous waiter open: a blocking
|
||||
* {@code bridge_send} kept blocking, and an async one kept reporting {@code PENDING} until
|
||||
* {@link #ASYNC_TIMEOUT_MS} — thirty minutes — even though the worker provably no longer
|
||||
* existed and the delegation could never complete. Worse, {@code poll} already had the evidence
|
||||
* (it calls {@code liveStatus} to build its detail string and gets back {@code "unknown"}) and
|
||||
* reported {@code PENDING} anyway.
|
||||
*
|
||||
* <p>Resolving the waiter as a failure — rather than letting it time out — also means the
|
||||
* outcome is counted, so a torn-down delegation stops being invisible to {@code /metrics}.
|
||||
*
|
||||
* @return true if a live waiter was failed
|
||||
*/
|
||||
public boolean abandon(String target, String reason) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(target);
|
||||
boolean failed = waiter != null && !waiter.isDone() && rendezvous.resolveFailure(waiter, reason);
|
||||
boolean asyncFailed = false;
|
||||
for (Task task : tasks.values()) {
|
||||
if (target.equals(task.target) && task.question == null
|
||||
&& task.future.complete(new Reply(Outcome.WORKER_FAILED, reason))) {
|
||||
asyncFailed = true;
|
||||
}
|
||||
}
|
||||
if (failed) {
|
||||
log.warn("abandoning the blocked send to {}: {}", target, reason);
|
||||
}
|
||||
return failed || asyncFailed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Acknowledge a specific reply by {@code msgId} for {@code target}. Removes it from the inbox
|
||||
* so that a subsequent drain or peek no longer returns it.
|
||||
*/
|
||||
public void ackReply(String target, String msgId) {
|
||||
inbox.ack(target, msgId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Drain (peek + ack) all pending inbox replies for {@code target}. At-least-once: returns the
|
||||
* messages and acknowledges them; an in-flight failure between returning and the caller
|
||||
* processing them re-surfaces them on a subsequent drain (the ack is local).
|
||||
*
|
||||
* @return the drained messages, newest last (FIFO); empty list if none
|
||||
*/
|
||||
public List<ReplyInbox.InboxMessage> drainReplies(String target) {
|
||||
var messages = inbox.peek(target);
|
||||
for (var msg : messages) {
|
||||
inbox.ack(target, msg.msgId());
|
||||
}
|
||||
return messages;
|
||||
}
|
||||
|
||||
/**
|
||||
* Deliver {@code content} to {@code target} (a herdr {@code terminal_id}) and block until the
|
||||
* worker replies via {@link Rendezvous} or {@code timeoutMillis} elapses.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis) {
|
||||
return send(target, content, timeoutMillis, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #send(String, String, long)}, but with an accepted-delivery hook.
|
||||
*
|
||||
* <p>{@code onAccepted} is invoked exactly once, once this send has won {@code target}'s send
|
||||
* lock and so become the <em>accepted target turn</em> — it runs <em>before</em> delivery is
|
||||
* queued, so a throwing hook fails the send cleanly (the waiter it already opened is closed and
|
||||
* nothing is left queued). It is <em>not</em> invoked when the send is {@link Outcome#BUSY}
|
||||
* (lock never taken). A caller uses this to record that <em>it</em> now owns the delegation's
|
||||
* reply routing (CB-548: {@code PrimaryRegistry} delegator ownership) — recording only on
|
||||
* acceptance means a concurrent sender that times out {@code BUSY} can never steal ownership it
|
||||
* never earned. {@code null} disables the hook.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis, Runnable onAccepted) {
|
||||
return send(target, content, timeoutMillis, onAccepted, null);
|
||||
}
|
||||
|
||||
/** Run a send, optionally stopping an async task that teardown already failed before acceptance. */
|
||||
private Reply send(String target, String content, long timeoutMillis, Runnable onAccepted, Task task) {
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(target, _ -> new ReentrantLock());
|
||||
|
||||
if (!tryLock(lock, remainingMillis(deadlineNanos))) {
|
||||
return new Reply(Outcome.BUSY, null); // another send held the session the whole window
|
||||
}
|
||||
try {
|
||||
if (task != null && task.future.isDone()) {
|
||||
return task.future.getNow(null);
|
||||
}
|
||||
if (hasAsyncQuestion(target)) {
|
||||
return new Reply(Outcome.BUSY, null); // the worker's current turn is paused for its lead
|
||||
}
|
||||
// Open the waiter BEFORE queueing delivery (CB-548). A fast reply — the worker already
|
||||
// injectable the instant we enqueue — otherwise arrives before the waiter is registered
|
||||
// and orphans into the inbox while this send blocks to the timeout (the enqueue-before-
|
||||
// open race). Opening first also means a throwing onAccepted (fired before enqueue) or an
|
||||
// enqueue failure is safely closed by the finally below: nothing is left queued, and the
|
||||
// failed send leaves no stale waiter behind.
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(target);
|
||||
try {
|
||||
if (task != null) {
|
||||
asyncTasksByWaiter.put(reply, task);
|
||||
}
|
||||
TurnToken token = new TurnToken(target, reply);
|
||||
// The send has won the lock; the accepted-delivery hook records delegator ownership
|
||||
// here (CB-548). It runs BEFORE enqueue so a throwing hook — onAccepted is now a
|
||||
// public callback — fails the send without queuing a message that would orphan.
|
||||
if (onAccepted != null) {
|
||||
onAccepted.run();
|
||||
}
|
||||
CompletableFuture<Void> delivered = injector.enqueue(target, content, token);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return recorded(new Reply(outcomeOf(r.kind()), r.text(), r.turnId()));
|
||||
} catch (TimeoutException e) {
|
||||
boolean wasDelivered = delivered.isDone() && !delivered.isCompletedExceptionally();
|
||||
log.debug("send to {} timed out (delivered={})", target, wasDelivered);
|
||||
return recorded(new Reply(
|
||||
wasDelivered ? Outcome.TIMED_OUT_WORKING : Outcome.TIMED_OUT_QUEUED, null));
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + target, e);
|
||||
}
|
||||
} finally {
|
||||
asyncTasksByWaiter.remove(reply);
|
||||
rendezvous.close(target, reply);
|
||||
}
|
||||
} finally {
|
||||
lock.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A worker's mid-turn question (CB-205 reverse rendezvous): surface {@code question} to the
|
||||
* primary by resolving its open blocking {@code bridge_send}, then block this (worker) call until
|
||||
* the primary answers via {@link #answer} or {@code timeoutMillis} elapses. Identity is the
|
||||
* worker's own session — it does not address the primary.
|
||||
*
|
||||
* <p>Returns {@link AskOutcome#NO_WAITER} when no delegation is open to surface the question to
|
||||
* (nothing to answer it), {@link AskOutcome#ANSWERED} with the primary's answer, or
|
||||
* {@link AskOutcome#TIMED_OUT} if the primary stayed silent. The worker resumes its turn either
|
||||
* way — an answered ask hands back the answer; an unanswered one leaves it to proceed alone.
|
||||
*/
|
||||
public AskResult ask(String workerSession, String question, long timeoutMillis) {
|
||||
Rendezvous.AskTicket ticket = rendezvous.openAsk(workerSession);
|
||||
// Only the freshly-opening caller surfaces the question; a coalesced duplicate simply blocks on
|
||||
// the shared answer future that the fresh owner is already responsible for.
|
||||
if (ticket.fresh()) {
|
||||
// Register the reverse waiter first, then surface the question — so the answer, which can
|
||||
// arrive the instant the primary reacts, always finds an open waiter to resolve.
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(workerSession);
|
||||
Task task = markAsyncQuestion(waiter, question, ticket.turnId());
|
||||
if (!rendezvous.resolveQuestion(workerSession, question, ticket.turnId())) {
|
||||
if (task != null) {
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
}
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
return new AskResult(AskOutcome.NO_WAITER, null); // no primary is blocked on this worker
|
||||
}
|
||||
}
|
||||
try {
|
||||
String answer = ticket.answer().get(timeoutMillis, TimeUnit.MILLISECONDS);
|
||||
return new AskResult(AskOutcome.ANSWERED, answer);
|
||||
} catch (TimeoutException e) {
|
||||
log.debug("bridge_ask from {} went unanswered in {}ms", workerSession, timeoutMillis);
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
return new AskResult(AskOutcome.TIMED_OUT, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting the primary's answer for " + workerSession, e);
|
||||
} finally {
|
||||
// Only the fresh owner tears down the shared turn; a duplicate must leave it open.
|
||||
if (ticket.fresh()) {
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The primary's answer to a worker's {@code bridge_ask} (CB-205): resolve the worker's blocked
|
||||
* question identified by {@code turnId}, then — like a fresh {@link #send} — block for the worker's
|
||||
* eventual {@code bridge_reply} as it finishes the resumed turn. The worker session is derived from
|
||||
* {@code turnId}, never a caller argument.
|
||||
*
|
||||
* <p>Unlike {@link #send} this does not re-inject through the {@link Injector}: the worker is
|
||||
* mid-turn (already picked up), so the answer flows back through its own open {@code bridge_ask}
|
||||
* call, not a new status-gated delivery. The forward waiter is opened <em>before</em> the worker
|
||||
* is unblocked so a reply that lands the instant it resumes is not lost.
|
||||
*/
|
||||
public Reply answer(String turnId, String content, long timeoutMillis) {
|
||||
String workerSession = rendezvous.askSession(turnId);
|
||||
if (workerSession == null) {
|
||||
return new Reply(Outcome.STALE_TURN, null); // the ask lapsed (timed out or already answered)
|
||||
}
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(workerSession, _ -> new ReentrantLock());
|
||||
if (!tryLock(lock, remainingMillis(deadlineNanos))) {
|
||||
return new Reply(Outcome.BUSY, null);
|
||||
}
|
||||
try {
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession);
|
||||
if (!rendezvous.answerAsk(turnId, content)) {
|
||||
rendezvous.close(workerSession, reply);
|
||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||
}
|
||||
clearAsyncQuestion(turnId, false);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
Reply result = new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
finishAsyncTask(turnId, result);
|
||||
return result;
|
||||
} catch (TimeoutException e) {
|
||||
// The worker resumed but hasn't replied yet — no completion fallback arms an answered
|
||||
// turn (it never re-entered the injector), so a silent worker rides out the window.
|
||||
return new Reply(Outcome.TIMED_OUT_WORKING, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + workerSession, e);
|
||||
} finally {
|
||||
rendezvous.close(workerSession, reply);
|
||||
}
|
||||
} finally {
|
||||
lock.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fire-and-poll variant of {@link #send}: deliver {@code content} to {@code target} on a
|
||||
* background virtual thread and return immediately with a ticket to {@link #poll}. This is how a
|
||||
* long task is delegated without tripping the caller's MCP client call timeout.
|
||||
*
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content) {
|
||||
return sendAsync(target, content, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #sendAsync(String, String)}, with the accepted-delivery hook of
|
||||
* {@link #send(String, String, long, Runnable)} — the running {@code send} invokes {@code onAccepted}
|
||||
* the moment it becomes the accepted target turn, so async flooding records delegator ownership
|
||||
* exactly as the blocking path does (CB-548).
|
||||
*
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content, Runnable onAccepted) {
|
||||
String ticket = "task-" + ticketSeq.incrementAndGet();
|
||||
Task task = new Task(target);
|
||||
tasks.put(ticket, task);
|
||||
asyncExecutor.submit(() -> {
|
||||
try {
|
||||
Reply result = send(target, content, ASYNC_TIMEOUT_MS, onAccepted, task);
|
||||
if (result.outcome() == Outcome.QUESTION) {
|
||||
// Keep the accepted owner until answer() finishes it. markAsyncQuestion may run
|
||||
// just after resolveQuestion wakes this thread.
|
||||
} else {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
} catch (Throwable t) {
|
||||
task.future.completeExceptionally(t);
|
||||
}
|
||||
});
|
||||
pruneTerminalTickets();
|
||||
log.debug("async send {} -> {}", ticket, target);
|
||||
return ticket;
|
||||
}
|
||||
|
||||
/**
|
||||
* Snapshot the state of an async delegation. Returns {@code null} for an unknown/expired ticket;
|
||||
* otherwise a {@link Phase#PENDING} view (with the live worker status as detail), a
|
||||
* {@link Phase#DONE} view carrying the reply, or a {@link Phase#FAILED} view with the reason.
|
||||
*/
|
||||
public TaskView poll(String ticket) {
|
||||
Task task = tasks.get(ticket);
|
||||
if (task == null) {
|
||||
return null;
|
||||
}
|
||||
CompletableFuture<Reply> f = task.future;
|
||||
if (!f.isDone()) {
|
||||
Reply question = task.question;
|
||||
if (question != null) {
|
||||
return new TaskView(ticket, Phase.ASKING, question.text(), null,
|
||||
"worker is waiting for your answer", question.turnId());
|
||||
}
|
||||
return new TaskView(ticket, Phase.PENDING, null, null, "worker " + liveStatus(task.target), null);
|
||||
}
|
||||
Reply r;
|
||||
try {
|
||||
r = f.getNow(null);
|
||||
} catch (CompletionException | java.util.concurrent.CancellationException e) {
|
||||
Throwable cause = (e instanceof CompletionException ce && ce.getCause() != null) ? ce.getCause() : e;
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, cause.getMessage(), null);
|
||||
}
|
||||
if (r.completed()) {
|
||||
String source = r.outcome() == Outcome.REPLIED ? "reply" : "transcript";
|
||||
return new TaskView(ticket, Phase.DONE, r.text(), source, null, null);
|
||||
}
|
||||
// A wedged worker (CB-109) or a backend-exhausted classification (CB-578 stage A) carries
|
||||
// the real cause as its reason; the timeout/busy outcomes carry none, so fall back to the
|
||||
// outcome name.
|
||||
boolean carriesReason = r.outcome() == Outcome.WORKER_FAILED || r.outcome() == Outcome.BACKEND_EXHAUSTED;
|
||||
String detail = carriesReason && r.text() != null
|
||||
? r.text()
|
||||
: "no reply — " + r.outcome().name().toLowerCase();
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, detail, null);
|
||||
}
|
||||
|
||||
/** Best-effort live worker status for a pending poll; never throws (a lookup error is just noise). */
|
||||
private String liveStatus(String target) {
|
||||
try {
|
||||
return agents.status(target).name().toLowerCase();
|
||||
} catch (RuntimeException e) {
|
||||
return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
/** Drop finished tickets older than the TTL so the registry cannot grow without bound. */
|
||||
private void pruneTerminalTickets() {
|
||||
long cutoff = System.nanoTime() - TICKET_TTL_NANOS;
|
||||
tasks.values().removeIf(t -> t.future.isDone() && t.createdNanos < cutoff);
|
||||
}
|
||||
|
||||
/** Record the active question for an async ticket; blocking sends have no entry and stay unchanged. */
|
||||
private Task markAsyncQuestion(CompletableFuture<Rendezvous.Resolution> waiter, String text, String turnId) {
|
||||
Task task = waiter == null ? null : asyncTasksByWaiter.get(waiter);
|
||||
if (task != null) {
|
||||
task.question = new Reply(Outcome.QUESTION, text, turnId);
|
||||
task.turnId = turnId;
|
||||
asyncTasksByTurn.put(turnId, task);
|
||||
}
|
||||
return task;
|
||||
}
|
||||
|
||||
/** Clear an answered or lapsed question, but only when it matches the ticket's current turn. */
|
||||
private void clearAsyncQuestion(String turnId, boolean forgetTurn) {
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null && turnId.equals(task.turnId)) {
|
||||
task.question = null;
|
||||
if (forgetTurn) {
|
||||
asyncTasksByTurn.remove(turnId, task);
|
||||
task.turnId = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete and detach an async ticket after its worker's actual terminal reply. */
|
||||
private void finishAsyncTask(Task task, Reply result) {
|
||||
task.future.complete(result);
|
||||
if (task.turnId != null) {
|
||||
asyncTasksByTurn.remove(task.turnId, task);
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete the async ticket correlated to a specific answered turn. */
|
||||
private void finishAsyncTask(String turnId, Reply result) {
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null) {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
}
|
||||
|
||||
/** A new send must not open a waiter while an async ticket owns this worker's paused turn. */
|
||||
private boolean hasAsyncQuestion(String target) {
|
||||
return asyncTasksByTurn.values().stream().anyMatch(task -> target.equals(task.target));
|
||||
}
|
||||
|
||||
/** Release the async executor. */
|
||||
public void close() {
|
||||
asyncExecutor.shutdown();
|
||||
}
|
||||
|
||||
/** Map a rendezvous {@link Rendezvous.Kind} onto its send {@link Outcome} (shared by send/answer). */
|
||||
private static Outcome outcomeOf(Rendezvous.Kind kind) {
|
||||
return switch (kind) {
|
||||
case REPLY -> Outcome.REPLIED;
|
||||
case COMPLETION -> Outcome.COMPLETED_UNREPLIED;
|
||||
case FAILED -> Outcome.WORKER_FAILED;
|
||||
case BACKEND_EXHAUSTED -> Outcome.BACKEND_EXHAUSTED;
|
||||
case QUESTION -> Outcome.QUESTION;
|
||||
};
|
||||
}
|
||||
|
||||
private static boolean tryLock(ReentrantLock lock, long millis) {
|
||||
try {
|
||||
return lock.tryLock(Math.max(0, millis), TimeUnit.MILLISECONDS);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting the session send lock", e);
|
||||
}
|
||||
}
|
||||
|
||||
private static long remainingMillis(long deadlineNanos) {
|
||||
return (deadlineNanos - System.nanoTime()) / 1_000_000L;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,251 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
|
||||
/**
|
||||
* The reply rendezvous: where a blocking {@code bridge_send} awaits how the worker's delegated turn
|
||||
* ends. The sending (primary) request thread {@link #open}s a waiter; it is resolved either by the
|
||||
* worker's explicit {@code bridge_reply} ({@link #resolve}, arriving on a different thread via
|
||||
* {@code POST /sessions/{id}/reply}) or — the CB-106 fallback — by the injector observing the
|
||||
* worker's delegated turn return to idle without a reply ({@link #resolveCompletion}).
|
||||
*
|
||||
* <p>At most one waiter per session — {@link MessageService} serializes sends per session, so a
|
||||
* resolution maps unambiguously to the one outstanding send and cannot be captured by another.
|
||||
*
|
||||
* <p><strong>Waiter identity (CB-116).</strong> The completion/failure fallbacks run asynchronously
|
||||
* and can fire <em>after</em> the turn they belong to has already been resolved by an explicit reply
|
||||
* and a <em>next</em> send has opened its own waiter on the same session. Resolving "whatever waiter
|
||||
* is registered now" would then land turn N's stale scrape on turn N+1's send. So those fallbacks
|
||||
* resolve a <em>specific</em> {@link CompletableFuture} captured when their turn was delivered
|
||||
* ({@link #resolveCompletion(CompletableFuture, String)} /
|
||||
* {@link #resolveFailure(CompletableFuture, String)}): a no-op if that waiter was already resolved,
|
||||
* and it can never touch a later send's waiter.
|
||||
*/
|
||||
public final class Rendezvous {
|
||||
|
||||
/** How a delegated turn ended (or paused). */
|
||||
public enum Kind {
|
||||
/** The worker called {@code bridge_reply} with a structured answer. */
|
||||
REPLY,
|
||||
/** The worker's turn finished without a {@code bridge_reply}; {@code text} is a scrape. */
|
||||
COMPLETION,
|
||||
/** The worker ran the turn then wedged (CB-109); {@code text} is the failure context. */
|
||||
FAILED,
|
||||
/**
|
||||
* The turn finished without a {@code bridge_reply}, and the scrape matched the backend's
|
||||
* configured usage-limit refusal pattern (CB-578 stage A); {@code text} is the reason,
|
||||
* carrying the matched line. The pane is healthy — only the account is refusing — so this
|
||||
* is kept separate from a session simply going {@code GONE}.
|
||||
*/
|
||||
BACKEND_EXHAUSTED,
|
||||
/**
|
||||
* The worker paused mid-turn to ask the primary a question (CB-205 reverse rendezvous);
|
||||
* {@code text} is the question and {@code turnId} correlates the primary's answer back to
|
||||
* the worker's blocked {@code bridge_ask}. Not terminal — the turn resumes after the answer.
|
||||
*/
|
||||
QUESTION
|
||||
}
|
||||
|
||||
/**
|
||||
* The resolved outcome of a send: its {@link Kind}, the associated text, and — only for
|
||||
* {@link Kind#QUESTION} — the {@code turnId} the primary answers with (else {@code null}).
|
||||
*/
|
||||
public record Resolution(Kind kind, String text, String turnId) {
|
||||
/** A terminal resolution (reply / completion / failure) with no correlation id. */
|
||||
public Resolution(Kind kind, String text) {
|
||||
this(kind, text, null);
|
||||
}
|
||||
}
|
||||
|
||||
/** A worker's open mid-turn question: the worker session it belongs to and the answer future. */
|
||||
private record AskWaiter(String session, CompletableFuture<String> answer) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Handle to a reverse-rendezvous turn: the {@code turnId}, its answer future, and whether this
|
||||
* call freshly opened it (versus coalescing onto an already-open ask).
|
||||
*/
|
||||
public record AskTicket(String turnId, CompletableFuture<String> answer, boolean fresh) {
|
||||
}
|
||||
|
||||
private final ConcurrentHashMap<String, CompletableFuture<Resolution>> waiters = new ConcurrentHashMap<>();
|
||||
|
||||
/** Reverse rendezvous (CB-205): worker questions awaiting the primary's answer, keyed by {@code turnId}. */
|
||||
private final ConcurrentHashMap<String, AskWaiter> asks = new ConcurrentHashMap<>();
|
||||
private final AtomicLong askSeq = new AtomicLong();
|
||||
/** Per-session index of the currently-open ask, so duplicate bridge_ask calls coalesce onto one turn. */
|
||||
private final ConcurrentHashMap<String, String> openAsksBySession = new ConcurrentHashMap<>();
|
||||
|
||||
/**
|
||||
* Register a waiter for {@code session} — the await side of the public {@code resolve*} methods.
|
||||
* The caller must hold that session's send lock.
|
||||
*
|
||||
* <p>Atomic fail-if-present (CB-548): if a waiter is already registered for {@code session}, an
|
||||
* {@link IllegalStateException} is thrown rather than replacing the first — so any future
|
||||
* invariant violation fails loudly instead of silently swapping the waiter another send is
|
||||
* blocked on. {@code MessageService} serializes sends per session (the send lock), so in correct
|
||||
* code a double open is impossible; this is a tripwire for the day that no longer holds.
|
||||
*/
|
||||
public CompletableFuture<Resolution> open(String session) {
|
||||
CompletableFuture<Resolution> waiter = new CompletableFuture<>();
|
||||
CompletableFuture<Resolution> existing = waiters.putIfAbsent(session, waiter);
|
||||
if (existing != null) {
|
||||
throw new IllegalStateException(
|
||||
"rendezvous double-open for session " + session + " — a waiter is already registered");
|
||||
}
|
||||
return waiter;
|
||||
}
|
||||
|
||||
/**
|
||||
* Remove {@code waiter} for {@code session}, only if it is still the registered one. The
|
||||
* symmetric complement of {@link #open}: a terminal send deregisters its waiter so the next
|
||||
* send on the session may {@link #open} a fresh one (CB-548 makes double-open an error, so a
|
||||
* successful {@code open} after a finished turn requires this close to have happened first).
|
||||
*/
|
||||
public void close(String session, CompletableFuture<Resolution> waiter) {
|
||||
waiters.remove(session, waiter);
|
||||
}
|
||||
|
||||
/** Whether a send is currently awaiting a resolution for {@code session}. */
|
||||
public boolean isWaiting(String session) {
|
||||
return waiters.containsKey(session);
|
||||
}
|
||||
|
||||
/**
|
||||
* The waiter currently registered for {@code session}, or {@code null} if none is waiting. The
|
||||
* completion/failure fallbacks capture this at delivery time so they can later resolve that exact
|
||||
* send (see the CB-116 note above) rather than whichever send happens to be waiting when they fire.
|
||||
*/
|
||||
public CompletableFuture<Resolution> currentWaiter(String session) {
|
||||
return waiters.get(session);
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the send awaiting on {@code session} with the worker's explicit reply {@code content}.
|
||||
*
|
||||
* @return {@code true} if a waiter was resolved; {@code false} if none was waiting (a late or
|
||||
* spurious reply — e.g. the send already timed out)
|
||||
*/
|
||||
public boolean resolve(String session, String content) {
|
||||
return complete(session, new Resolution(Kind.REPLY, content));
|
||||
}
|
||||
|
||||
// --- reverse rendezvous (CB-205 bridge_ask) ------------------------------------------------
|
||||
|
||||
/**
|
||||
* Open a reverse-rendezvous waiter for a worker's mid-turn question. If {@code session} already has
|
||||
* an open ask, coalesce onto it (same {@code turnId}, same answer future). Otherwise atomically mint
|
||||
* a fresh {@code turnId}, register it in both the per-turn and per-session indexes, and hand it back
|
||||
* marked fresh. The caller then {@link #resolveQuestion surfaces the question} to the primary and
|
||||
* blocks on the returned future until the primary {@link #answerAsk answers}.
|
||||
*/
|
||||
public AskTicket openAsk(String session) {
|
||||
while (true) {
|
||||
AskWaiter[] minted = { null };
|
||||
String turnId = openAsksBySession.computeIfAbsent(session, _ -> {
|
||||
String newTurnId = session + "#" + askSeq.incrementAndGet();
|
||||
CompletableFuture<String> answer = new CompletableFuture<>();
|
||||
AskWaiter waiter = new AskWaiter(session, answer);
|
||||
asks.put(newTurnId, waiter);
|
||||
minted[0] = waiter;
|
||||
return newTurnId;
|
||||
});
|
||||
if (minted[0] != null) {
|
||||
return new AskTicket(turnId, minted[0].answer(), true);
|
||||
}
|
||||
AskWaiter existing = asks.get(turnId);
|
||||
if (existing != null) {
|
||||
return new AskTicket(turnId, existing.answer(), false);
|
||||
}
|
||||
// A close raced and removed the waiter after we read the turnId; clear the stale index entry
|
||||
// and retry so a fresh ask is always backed by a registered waiter.
|
||||
openAsksBySession.remove(session, turnId);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Surface a worker's mid-turn {@code question} by resolving the primary's open {@code bridge_send}
|
||||
* with a {@link Kind#QUESTION} carrying {@code turnId}. Same session-keyed semantics as
|
||||
* {@link #resolve}: the one outstanding send for {@code session} unblocks with the question.
|
||||
*
|
||||
* @return {@code true} if a send was awaiting (the question reached the primary); {@code false}
|
||||
* if none was (no delegation is open to answer it)
|
||||
*/
|
||||
public boolean resolveQuestion(String session, String question, String turnId) {
|
||||
return complete(session, new Resolution(Kind.QUESTION, question, turnId));
|
||||
}
|
||||
|
||||
/** The worker session an outstanding ask {@code turnId} belongs to, or {@code null} if unknown/lapsed. */
|
||||
public String askSession(String turnId) {
|
||||
AskWaiter w = asks.get(turnId);
|
||||
return w == null ? null : w.session();
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a worker's blocked {@code bridge_ask} with the primary's {@code answer}, unblocking it
|
||||
* to resume its turn.
|
||||
*
|
||||
* @return {@code true} if the ask was still open and got the answer; {@code false} if the
|
||||
* {@code turnId} is unknown or the ask already lapsed (timed out / was answered)
|
||||
*/
|
||||
public boolean answerAsk(String turnId, String answer) {
|
||||
AskWaiter w = asks.get(turnId);
|
||||
return w != null && w.answer().complete(answer);
|
||||
}
|
||||
|
||||
/** Drop a reverse-rendezvous turn once its {@code bridge_ask} has resolved (answered or lapsed). */
|
||||
public void closeAsk(String turnId) {
|
||||
AskWaiter w = asks.get(turnId);
|
||||
if (w == null) {
|
||||
return;
|
||||
}
|
||||
// Remove the session index first and only if it still points to this turn, so a concurrent
|
||||
// fresh ask cannot inherit a waiter we are about to drop.
|
||||
openAsksBySession.remove(w.session(), turnId);
|
||||
asks.remove(turnId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a specific captured {@code waiter} as a completion (the delegated turn finished with no
|
||||
* {@code bridge_reply}); {@code text} is the scraped transcript tail. The waiter is the one
|
||||
* captured when this turn was delivered, so a late completion for turn N cannot land on turn N+1's
|
||||
* send (CB-116). A no-op if that waiter was already resolved — a raced {@code bridge_reply} wins.
|
||||
*
|
||||
* @return {@code true} if this call resolved the waiter, {@code false} if it was null or already resolved
|
||||
*/
|
||||
public boolean resolveCompletion(CompletableFuture<Resolution> waiter, String text) {
|
||||
return waiter != null && waiter.complete(new Resolution(Kind.COMPLETION, text));
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a specific captured {@code waiter} as a failure — the worker ran the turn but wedged in
|
||||
* an unrecoverable state (CB-109); {@code reason} is the failure context (e.g. the error screen).
|
||||
* Like {@link #resolveCompletion(CompletableFuture, String)} it targets the exact captured send
|
||||
* (CB-116). A no-op if that waiter was already resolved — first resolution wins.
|
||||
*
|
||||
* @return {@code true} if this call resolved the waiter, {@code false} if it was null or already resolved
|
||||
*/
|
||||
public boolean resolveFailure(CompletableFuture<Resolution> waiter, String reason) {
|
||||
return waiter != null && waiter.complete(new Resolution(Kind.FAILED, reason));
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a specific captured {@code waiter} as {@link Kind#BACKEND_EXHAUSTED} (CB-578 stage A):
|
||||
* the turn finished with no {@code bridge_reply} and the scrape matched the backend's configured
|
||||
* usage-limit pattern; {@code reason} carries the matched line. Like
|
||||
* {@link #resolveCompletion(CompletableFuture, String)} it targets the exact captured send
|
||||
* (CB-116). A no-op if that waiter was already resolved — first resolution wins.
|
||||
*
|
||||
* @return {@code true} if this call resolved the waiter, {@code false} if it was null or already resolved
|
||||
*/
|
||||
public boolean resolveExhausted(CompletableFuture<Resolution> waiter, String reason) {
|
||||
return waiter != null && waiter.complete(new Resolution(Kind.BACKEND_EXHAUSTED, reason));
|
||||
}
|
||||
|
||||
private boolean complete(String session, Resolution resolution) {
|
||||
CompletableFuture<Resolution> waiter = waiters.get(session);
|
||||
return waiter != null && waiter.complete(resolution);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,51 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Holds terminal worker→primary replies that arrive with no live send to resolve, keyed by worker
|
||||
* session (target), until the primary drains them. Soft-state in Stage 1 (in-memory, lost on restart);
|
||||
* the Stage 2 AMQP adapter implements the same contract with cross-restart durability.
|
||||
*
|
||||
* <p><strong>This interface is the port.</strong> {@link InMemoryReplyInbox} is the Stage-1 adapter;
|
||||
* an AMQP-backed adapter (Stage 2) must implement the same contract (idempotent publish, FIFO peek,
|
||||
* at-least-once ack).
|
||||
*
|
||||
* <p><strong>Ownership is explicit.</strong> A gateway {@link #own owns} the inbox for each agent it
|
||||
* spawned; only the owner consumes and drains it. {@link #publish} sends a reply to the target's
|
||||
* inbox but does <em>not</em> imply ownership or start a consumer. This separation is required by
|
||||
* CB-308 federation, where one gateway may publish to an agent owned by another gateway.
|
||||
*/
|
||||
public interface ReplyInbox {
|
||||
|
||||
/** A queued reply: an idempotency id, the worker session it came from, and the reply text. */
|
||||
record InboxMessage(String msgId, String target, String content) {}
|
||||
|
||||
/**
|
||||
* Start owning (consuming) the inbox for {@code target}. Idempotent: multiple calls for the same
|
||||
* target are no-ops. The owner is the only gateway that may {@link #peek} and {@link #ack} replies
|
||||
* for this target.
|
||||
*/
|
||||
void own(String target);
|
||||
|
||||
/**
|
||||
* Stop owning (consuming) the inbox for {@code target}. Idempotent. Any replies held locally but
|
||||
* not yet acked are dropped from the local snapshot; the underlying durable queue keeps
|
||||
* unacked messages for redelivery when the target is re-owned.
|
||||
*/
|
||||
void release(String target);
|
||||
|
||||
/**
|
||||
* Queue {@code content} from worker {@code target} under {@code msgId}. Idempotent: publishing an
|
||||
* already-present {@code msgId} for {@code target} is a no-op (dedup), so an at-least-once Stage-2
|
||||
* redelivery cannot double-queue. Publishing does <em>not</em> imply ownership and must not start a
|
||||
* consumer.
|
||||
*/
|
||||
void publish(String target, String msgId, String content);
|
||||
|
||||
/** Non-destructive snapshot of pending replies for {@code target} (FIFO), empty list if none. */
|
||||
List<InboxMessage> peek(String target);
|
||||
|
||||
/** Remove the reply {@code msgId} for {@code target} once the primary has taken it. No-op if absent. */
|
||||
void ack(String target, String msgId);
|
||||
}
|
||||
@@ -0,0 +1,199 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Mechanism (b) of CB-307: a dedicated, status-gated push loop that nudges the primary's own
|
||||
* herdr pane when a worker reply lands with no live {@code bridge_send} to resolve it.
|
||||
*
|
||||
* <p>The loop is triggered by {@link #onReplyQueued(String)} (called from
|
||||
* {@link MessageService#reply} after the durable inbox publish). It checks four conditions
|
||||
* at each tick via {@link #decide(String, int)}, then either injects a drain nudge,
|
||||
* waits for the primary to become injectable, or stops reminding.
|
||||
*
|
||||
* <p>Bounded: at most {@link #maxReminders} nudges per target, with a configurable backoff
|
||||
* between them. The reply is never lost — the durable inbox is the backstop.
|
||||
*/
|
||||
public final class ReplyPushLoop {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ReplyPushLoop.class);
|
||||
static final String NUDGE_FORMAT = "Worker %s returned a reply — run bridge_poll(target=%s) to collect it";
|
||||
|
||||
private final PrimaryRegistry primaryRegistry;
|
||||
private final AgentControl agents;
|
||||
private final ReplyInbox inbox;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final int maxReminders;
|
||||
private final long backoffMs;
|
||||
private final Metrics metrics; // CB-512: nullable — no registry in unit tests
|
||||
|
||||
/** Track targets that have an active schedule. */
|
||||
private final ConcurrentHashMap<String, Boolean> activeTargets = new ConcurrentHashMap<>();
|
||||
|
||||
public ReplyPushLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
ScheduledExecutorService scheduler,
|
||||
int maxReminders, long backoffMs) {
|
||||
this(primaryRegistry, agents, inbox, scheduler, maxReminders, backoffMs, null);
|
||||
}
|
||||
|
||||
/** As above, with a metric registry (CB-512) so push outcomes are counted. */
|
||||
public ReplyPushLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
ScheduledExecutorService scheduler,
|
||||
int maxReminders, long backoffMs, Metrics metrics) {
|
||||
this.primaryRegistry = primaryRegistry;
|
||||
this.agents = agents;
|
||||
this.inbox = inbox;
|
||||
this.scheduler = scheduler;
|
||||
this.maxReminders = maxReminders;
|
||||
this.backoffMs = backoffMs;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Count one nudge outcome when a registry is wired; a no-op in unit tests. */
|
||||
private void countNudge(String outcome) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(BridgedMetrics.PUSH_NUDGES, "outcome", outcome);
|
||||
}
|
||||
}
|
||||
|
||||
// --- decision logic (package-private for unit-testing) -------------------------------------
|
||||
|
||||
/** The action the loop should take for a target at the given reminder count. */
|
||||
enum Action { INJECT, WAIT_BUSY, STOP }
|
||||
|
||||
/**
|
||||
* Pure decision function: examine the current state and return what the loop should do.
|
||||
*
|
||||
* @param target the worker session (target terminal id)
|
||||
* @param reminderCount how many nudges have been sent so far for this target
|
||||
* @return the action the caller should take
|
||||
*/
|
||||
Action decide(String target, int reminderCount) {
|
||||
// CB-532: the destination is per-delegation — the lead that sent this worker its work, not
|
||||
// "the primary". With two leads orchestrating one fleet the singular question has no right
|
||||
// answer, and answering it anyway interrupted whichever lead happened to call bridge_send
|
||||
// first with results it never asked for.
|
||||
var nudgeTarget = primaryRegistry.nudgeTargetFor(target);
|
||||
if (nudgeTarget.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on {}, stopping reminder", target);
|
||||
return Action.STOP;
|
||||
}
|
||||
if (inbox.peek(target).isEmpty()) {
|
||||
log.debug("push: inbox empty for {}, stopping reminder", target);
|
||||
return Action.STOP;
|
||||
}
|
||||
if (reminderCount >= maxReminders) {
|
||||
log.debug("push: reminder cap ({}) reached for {}, stopping", maxReminders, target);
|
||||
countNudge("exhausted");
|
||||
return Action.STOP;
|
||||
}
|
||||
String leadTerminal = nudgeTarget.get();
|
||||
AgentStatus status;
|
||||
try {
|
||||
status = agents.status(leadTerminal);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("push: status check failed for lead {}, will retry", leadTerminal, e);
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
if (status.injectable()) {
|
||||
return Action.INJECT;
|
||||
}
|
||||
log.debug("push: lead {} is {} (not injectable), waiting", leadTerminal, status);
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
|
||||
// --- public entrypoint ---------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Called when a reply is queued for {@code target}. Idempotent per target: a second call while
|
||||
* a schedule is active is a no-op. The schedule nudges the primary, then schedules a follow-up
|
||||
* check (reminder on backoff, or re-check on WAIT_BUSY), until the inbox is empty or the cap
|
||||
* is reached.
|
||||
*/
|
||||
public void onReplyQueued(String target) {
|
||||
if (activeTargets.putIfAbsent(target, Boolean.TRUE) != null) {
|
||||
log.debug("push: already active for {}, ignoring duplicate trigger", target);
|
||||
return; // already scheduled
|
||||
}
|
||||
log.debug("push: starting reminder loop for {}", target);
|
||||
scheduleNext(target, 0);
|
||||
}
|
||||
|
||||
/** Execute one loop tick — called on the scheduler thread. */
|
||||
private void tick(String target, int reminderCount) {
|
||||
var action = decide(target, reminderCount);
|
||||
switch (action) {
|
||||
case INJECT -> {
|
||||
injectNudge(target, reminderCount);
|
||||
scheduleNext(target, reminderCount + 1);
|
||||
}
|
||||
// Re-check after the configured backoff; the primary may become injectable soon.
|
||||
case WAIT_BUSY -> scheduleNext(target, reminderCount);
|
||||
case STOP -> {
|
||||
activeTargets.remove(target);
|
||||
log.debug("push: reminder loop ended for {}", target);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Send the nudge and log the event. */
|
||||
private void injectNudge(String target, int reminderCount) {
|
||||
// Re-read rather than threading it down from decide(): the delegating lead can change
|
||||
// between the decision and the injection, and the nudge should follow the current one.
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: lead for {} disappeared before the nudge could be sent", target);
|
||||
return;
|
||||
}
|
||||
String leadTerminal = lead.get();
|
||||
String nudge = NUDGE_FORMAT.formatted(target, target);
|
||||
try {
|
||||
agents.send(leadTerminal, nudge);
|
||||
log.debug("push: nudge {}/{} sent to lead {} for target {}",
|
||||
reminderCount + 1, maxReminders, leadTerminal, target);
|
||||
countNudge("delivered");
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("push: failed to nudge lead {} for target {} (reminder {}/{}): {}",
|
||||
leadTerminal, target, reminderCount + 1, maxReminders, e.toString());
|
||||
}
|
||||
}
|
||||
|
||||
/** Schedule the next tick on the scheduler thread pool. */
|
||||
private void scheduleNext(String target, int nextReminderCount) {
|
||||
scheduler.schedule(() -> tick(target, nextReminderCount), backoffMs, TimeUnit.MILLISECONDS);
|
||||
}
|
||||
|
||||
// --- lifecycle -----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Whether any reminder loop is currently active for some target (CB-551). The idle-lead heartbeat
|
||||
* uses this to stand aside: while the push loop is actively nudging the lead, a concurrent
|
||||
* heartbeat injection would start a second competing turn in the same pane — racing loops multiply
|
||||
* turns and context burn. "Active" means a schedule exists in {@link #activeTargets}; the set is
|
||||
* bounded by what has been triggered, not by any persistent state.
|
||||
*/
|
||||
public boolean isActive() {
|
||||
return !activeTargets.isEmpty();
|
||||
}
|
||||
|
||||
/** Shut down the scheduler. Outstanding reminders are cancelled. */
|
||||
public void stop() {
|
||||
scheduler.shutdownNow();
|
||||
activeTargets.clear();
|
||||
}
|
||||
|
||||
/** @see #stop() */
|
||||
public void close() {
|
||||
stop();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
|
||||
/**
|
||||
* Identity for one accepted send. The session turn is deliberately absent: CompletionResolver's
|
||||
* delivery callback runs before SessionManager.onDelivered, so binding it needs a later ordering design.
|
||||
*/
|
||||
public final class TurnToken {
|
||||
private final String target;
|
||||
private final CompletableFuture<Rendezvous.Resolution> waiter;
|
||||
|
||||
public TurnToken(String target, CompletableFuture<Rendezvous.Resolution> waiter) {
|
||||
this.target = target;
|
||||
this.waiter = waiter;
|
||||
}
|
||||
|
||||
public String target() { return target; }
|
||||
public CompletableFuture<Rendezvous.Resolution> waiter() { return waiter; }
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
package dev.ltms.bridged.peer;
|
||||
|
||||
/**
|
||||
* Declared capabilities of a {@link PeerLauncher}. The protocol is the union across all
|
||||
* configured launchers; a verb invoked against a peer that lacks the capability returns a clean
|
||||
* "unsupported for this peer" rather than a crash. Capabilities keep the protocol honest as peers
|
||||
* diversify and prevent the core from assuming "every peer is a Claude in a worktree."
|
||||
*/
|
||||
public enum Capability {
|
||||
|
||||
/**
|
||||
* The peer supports {@code bridge_ask} rendezvous — pausing its delegated turn to ask
|
||||
* the primary a question, then resuming once answered. All Claude Code peers support this.
|
||||
*/
|
||||
MID_TURN_ASK,
|
||||
|
||||
/**
|
||||
* The peer can open its own PR at the end of an implementation turn (CB-302). Opt-in per
|
||||
* profile: granted only when the profile carries a git-forge token ({@code gitTokenEnv}).
|
||||
*/
|
||||
SELF_PR,
|
||||
|
||||
/**
|
||||
* The peer can run inside a provisioned isolated git worktree. All CLI-based peers support
|
||||
* this since their cwd is set at spawn time.
|
||||
*/
|
||||
WORKTREE,
|
||||
|
||||
/**
|
||||
* The peer can discard its current conversation context without starting a delegated bridge
|
||||
* turn. The command or API used to do that is adapter-specific.
|
||||
*/
|
||||
CONTEXT_RESET,
|
||||
|
||||
/**
|
||||
* The spawner can reconcile orphaned peers on boot — workers that outlived a prior daemon
|
||||
* process and whose pane ids died with it (CB-117). Claude Code over herdr supports this
|
||||
* via name-based matching against the herdr agent list.
|
||||
*/
|
||||
ORPHAN_REAP,
|
||||
|
||||
/**
|
||||
* The peer surfaces the bridge's logical name ({@link SpawnRequest#sessionName()}) in its own
|
||||
* UI at spawn — e.g. Claude Code's {@code -n} display name, which shows in the prompt box, the
|
||||
* {@code /resume} picker, and the terminal title. This is what lets an operator tell the
|
||||
* bridge's worker apart from a user's own session in the same terminal after a restart.
|
||||
*/
|
||||
SESSION_NAME,
|
||||
|
||||
/**
|
||||
* The peer can be relaunched onto a prior conversation via that conversation's own session id
|
||||
* ({@link SpawnRequest#resumeSessionId()}) — e.g. Claude Code's {@code -r}, which adopts the id
|
||||
* as the agent's own identity rather than starting a new conversation. A launcher that declares
|
||||
* this mints/resolves that id at spawn and exposes it on the returned
|
||||
* {@link PeerHandle#agentSessionId()}, so a later resume addresses the same conversation.
|
||||
*/
|
||||
SESSION_RESUME
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
package dev.ltms.bridged.peer;
|
||||
|
||||
import java.util.Locale;
|
||||
|
||||
/**
|
||||
* What a member is <em>for</em> — the contract it runs under.
|
||||
*
|
||||
* <p>A member is anything a lead spawns. Every member carries two independent attributes:
|
||||
*
|
||||
* <ul>
|
||||
* <li><b>role</b> (this enum) — <em>which contract</em>: the launch charter it is given, the role
|
||||
* file it reads, the playbook skill it loads, and its authorization row.</li>
|
||||
* <li><b>profile</b> (a {@code profiles:} key) — <em>which backend</em>: model, CLI adapter,
|
||||
* credentials, cost.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>These are separate axes on purpose. A {@code REVIEWER} may run on the same profile as the
|
||||
* {@code DEV} whose diff it reviews — same backend, different contract. That case is what proves
|
||||
* role and profile cannot be collapsed into one field.
|
||||
*
|
||||
* <p>A lead is deliberately <em>not</em> a role here. A lead is not spawned: it is a pre-existing,
|
||||
* human-facing session that config recognises. Only spawned peers have a member role.
|
||||
*/
|
||||
public enum MemberRole {
|
||||
|
||||
/**
|
||||
* Refines a ticket before anyone implements it: scope, acceptance criteria, risks, unit split.
|
||||
*
|
||||
* <p>Reads the repo and writes analysis. Never commits code and never opens a pull request —
|
||||
* an architect that starts implementing has stopped doing the job that makes it useful.
|
||||
*
|
||||
* <p>Architects are the one member kind declared in config, because a lead addresses the same
|
||||
* slots across many tickets and needs a stable name for them.
|
||||
*/
|
||||
ARCHITECT,
|
||||
|
||||
/**
|
||||
* Implements one unit of work: provisions a worktree, writes the code, commits, pushes, and
|
||||
* opens its own pull request.
|
||||
*
|
||||
* <p>Never merges. The lead is the gate, and a member that merged its own work would remove the
|
||||
* only independent check in the flow.
|
||||
*/
|
||||
DEV,
|
||||
|
||||
/**
|
||||
* Reviews a diff it did not write and reports one structured finding.
|
||||
*
|
||||
* <p>Never commits, never merges, and is never the member that wrote the scope under review.
|
||||
* Briefed from the diff rather than from the implementer's rationale, because that rationale
|
||||
* carries the same blind spot that produced the bug.
|
||||
*/
|
||||
REVIEWER;
|
||||
|
||||
/** The lowercase spelling used in config and on the wire ({@code architect}, {@code dev}, …). */
|
||||
public String wireName() {
|
||||
return name().toLowerCase(Locale.ROOT);
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code fleet:} block that holds this role's pool — {@code architects},
|
||||
* {@code developers}, {@code reviewers}.
|
||||
*
|
||||
* <p>Plural, and not always the wire name: the pool of things a {@code dev} may run on reads
|
||||
* naturally as {@code developers:}. The wire name stays the singular {@code dev}, because that
|
||||
* is what a tab label and a roster row say.
|
||||
*/
|
||||
public String configKey() {
|
||||
return switch (this) {
|
||||
case ARCHITECT -> "architects";
|
||||
case DEV -> "developers";
|
||||
case REVIEWER -> "reviewers";
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* The role owning the {@code fleet:} pool named {@code key}, or {@code null} when the key is not
|
||||
* a role pool ({@code leaders}, {@code tabLabel}, …). Null rather than a throw: callers use this
|
||||
* to sort a fleet block's children, where a non-pool key is normal rather than an error.
|
||||
*/
|
||||
public static MemberRole fromConfigKey(String key) {
|
||||
if (key == null) {
|
||||
return null;
|
||||
}
|
||||
String k = key.trim().toLowerCase(Locale.ROOT);
|
||||
for (MemberRole r : values()) {
|
||||
if (r.configKey().equals(k)) {
|
||||
return r;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a config/wire spelling, case-insensitively.
|
||||
*
|
||||
* @param s the spelling to parse
|
||||
* @return the matching role
|
||||
* @throws IllegalArgumentException when {@code s} is null, blank, or not a known role — the
|
||||
* message lists the valid spellings, because a typo'd role in
|
||||
* config should fail at startup with the fix in the error
|
||||
*/
|
||||
public static MemberRole parse(String s) {
|
||||
if (s != null && !s.isBlank()) {
|
||||
String t = s.trim().toLowerCase(Locale.ROOT);
|
||||
for (MemberRole r : values()) {
|
||||
if (r.wireName().equals(t)) {
|
||||
return r;
|
||||
}
|
||||
}
|
||||
}
|
||||
StringBuilder valid = new StringBuilder();
|
||||
for (MemberRole r : values()) {
|
||||
if (!valid.isEmpty()) {
|
||||
valid.append(", ");
|
||||
}
|
||||
valid.append(r.wireName());
|
||||
}
|
||||
throw new IllegalArgumentException(
|
||||
"unknown member role '" + s + "'; valid roles are: " + valid);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,70 @@
|
||||
package dev.ltms.bridged.peer;
|
||||
|
||||
/**
|
||||
* An opaque handle returned by {@link PeerLauncher#spawn(SpawnRequest)}. The core routes on
|
||||
* {@link #id()} (the registry/routing key) and uses {@link #terminalId()} for session tracking;
|
||||
* launcher-private coordinates beyond these are reachable through the concrete implementation.
|
||||
*
|
||||
* <p>A {@link PeerHandle} is returned <em>after</em> the peer process is live — the launcher
|
||||
* has already completed subscription-guarded env/vfs setup, process start, and placement. The
|
||||
* handle is a ticket the core exchanges for the running peer, not a lazy/delayed reference.
|
||||
*/
|
||||
public interface PeerHandle {
|
||||
|
||||
/**
|
||||
* The registry/routing key — an opaque, launcher-assigned identifier (CB-519). Multiple
|
||||
* daemon processes may run on one host, so the contract is <em>host-unique</em>, not merely
|
||||
* process-unique: the herdr-backed launcher mints a fresh UUID per spawn, and a non-herdr
|
||||
* launcher is likewise expected to return an identifier that cannot collide across processes
|
||||
* on the same host. This id is the routing key and is deliberately decoupled from any launcher
|
||||
* transport coordinate (e.g. a herdr pane id), which stays launcher-private. Guaranteed to be
|
||||
* non-null and unique among live peers on the host.
|
||||
*/
|
||||
String id();
|
||||
|
||||
/**
|
||||
* The transport-level session identifier used for message routing and presence tracking.
|
||||
* For the herdr launcher this is the herdr terminal UUID. A non-herdr launcher may return
|
||||
* its own analogous identifier, or {@code null} if the concept does not apply.
|
||||
*/
|
||||
default String terminalId() {
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The worker profile that spawned this peer, if the launcher resolved one. A launcher that
|
||||
* performs dynamic profile selection (e.g. CB-518 weighted placement) sets this so the
|
||||
* session registry records the actual profile rather than the requested/default one.
|
||||
*
|
||||
* @return the profile name, or {@code null} when the launcher leaves it unspecified
|
||||
*/
|
||||
default String profile() {
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The bridge's logical name for this session, as assigned at spawn
|
||||
* ({@link SpawnRequest#sessionName()}). Stable across restarts and meaningful to an operator,
|
||||
* unlike the transport identifiers above; a launch with no name leaves the peer's display
|
||||
* identity to the launcher to derive.
|
||||
*
|
||||
* @return the bridge-assigned logical session name, or {@code null} if none was assigned
|
||||
*/
|
||||
default String sessionName() {
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The peer's OWN session id — the handle that resumes this conversation later (the id a
|
||||
* later {@link Capability#SESSION_RESUME resume} spawn would pass back). Null when the adapter
|
||||
* cannot determine it — the contract for an adapter that declines
|
||||
* {@link Capability#SESSION_RESUME}; an adapter that declares that capability returns this
|
||||
* non-null for a spawn that requested session identity, because it knows the id before the
|
||||
* peer has written anything.
|
||||
*
|
||||
* @return the peer's own session id, or {@code null} when not determinable
|
||||
*/
|
||||
default String agentSessionId() {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
package dev.ltms.bridged.peer;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* SPI for materializing a connected peer — the only way the bridge core creates or tears down
|
||||
* a peer process. Every launcher is a first-party, in-tree adapter selected by (future) profile
|
||||
* config; today's single adapter is the {@code ClaudeCodeLauncher} / Claude Code over herdr.
|
||||
*
|
||||
* <p>The core delegates spawn and teardown to this interface without knowing how the peer is set
|
||||
* up. Environment variables, CLI flags, subscription guards, transport (herdr tab/pane) layout,
|
||||
* and naming conventions are all adapter-private — the core sees only the returned
|
||||
* {@link PeerHandle} whose {@code id()} is the registry/routing key.
|
||||
*
|
||||
* <p>The interface is a superset of what {@code SessionManager} and {@code Bridged.main} call
|
||||
* on the concrete launcher today.
|
||||
*/
|
||||
public interface PeerLauncher {
|
||||
|
||||
/**
|
||||
* The set of {@link Capability capabilities} this launcher declares. A peer whose profile
|
||||
* opts into a git-forge token should include {@link Capability#SELF_PR}; the base set for
|
||||
* the Claude Code herdr adapter is always {@code MID_TURN_ASK, WORKTREE, ORPHAN_REAP}.
|
||||
*/
|
||||
Set<Capability> capabilities();
|
||||
|
||||
/**
|
||||
* {@code profileName}/requestedCwd null/blank → default resolution. Returns after the peer
|
||||
* process is live (env + argv + placement complete). Never returns {@code null}.
|
||||
*
|
||||
* @param req the spawn parameters (profile, requested cwd, caller cwd)
|
||||
* @return a handle whose {@link PeerHandle#id()} is the registry/routing key
|
||||
* @throws IllegalArgumentException if the profile is unknown and no default is configured
|
||||
*/
|
||||
PeerHandle spawn(SpawnRequest req);
|
||||
|
||||
/**
|
||||
* The configured worker profile names — the set of names {@code spawn(profileName)} accepts.
|
||||
*/
|
||||
Set<String> profiles();
|
||||
|
||||
/**
|
||||
* The profile a no-argument {@link #spawn(SpawnRequest)} uses, or {@code null} if none is configured.
|
||||
*/
|
||||
String defaultProfile();
|
||||
|
||||
/**
|
||||
* Resolve the effective working directory for a spawn {@code req} without actually spawning.
|
||||
* Resolution order: requestedCwd → profile cwd → callerCwd → daemon cwd.
|
||||
*
|
||||
* @return the resolved absolute path, never null/blank
|
||||
*/
|
||||
String effectiveCwd(SpawnRequest req);
|
||||
|
||||
/**
|
||||
* The parity-overlay file list for {@code profileName} (default list when unset). Used by
|
||||
* worktree provisioning to copy config files into the isolated checkout before spawning.
|
||||
*/
|
||||
List<String> parityOverlay(String profileName);
|
||||
|
||||
/**
|
||||
* The set of all agents this launcher currently tracks, transport-specific. Each element
|
||||
* exposes at minimum a pane-like {@code id()} matching this launcher's {@link PeerHandle}
|
||||
* scheme, plus transport-level status. Callers merge this set with the session registry to
|
||||
* build a live roster view.
|
||||
*/
|
||||
List<?> list();
|
||||
|
||||
/**
|
||||
* Reap orphaned peers left behind by a prior daemon process. Only peers whose naming scheme
|
||||
* matches this launcher's and whose nonce differs from the current process are eligible.
|
||||
* Best-effort: a failure to list or to stop any one peer is logged and never aborts startup.
|
||||
*
|
||||
* @return the number of orphaned peers reaped
|
||||
*/
|
||||
int reapOrphanWorkers();
|
||||
|
||||
/**
|
||||
* Tear a peer down by its registry/routing key ({@link PeerHandle#id()}). Tolerates an
|
||||
* already-gone peer. Also cleans up launcher-private resources (e.g. empty dedicated tabs)
|
||||
* when safe to do so.
|
||||
*/
|
||||
void stop(String id);
|
||||
|
||||
/**
|
||||
* Discard the context of the peer identified by {@code id}. Implementations must bypass normal
|
||||
* bridge delivery/turn accounting. Unsupported peer kinds return {@code false} without sending
|
||||
* a guessed command.
|
||||
*
|
||||
* @return {@code true} when a reset was sent and its status transition must settle before reuse
|
||||
*/
|
||||
boolean clearContext(String id);
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
package dev.ltms.bridged.peer;
|
||||
|
||||
/**
|
||||
* Thrown when a {@link PeerLauncher} starts a peer process but the peer
|
||||
* does not reach an injectable (ready-to-receive) state within the configured
|
||||
* timeout. The launcher MUST clean up any resources it created (pane, tab)
|
||||
* before throwing — no orphaned peer or pane is left behind.
|
||||
*
|
||||
* <p>This is a spawn-time failure, distinct from a post-spawn disconnect.
|
||||
* Callers treat this as a clean spawn error (the peer never materialized
|
||||
* into a usable session), not a mid-life session fault.
|
||||
*/
|
||||
public final class PeerUnreachableException extends RuntimeException {
|
||||
|
||||
public PeerUnreachableException(String message) {
|
||||
super(message);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,32 @@
|
||||
package dev.ltms.bridged.peer;
|
||||
|
||||
/**
|
||||
* What a caller asks for when spawning a peer.
|
||||
*
|
||||
* @param profileName the {@code profiles:} entry to spawn on; {@code null}/blank ⇒ the caller
|
||||
* did not choose, and the role's pool supplies the candidates
|
||||
* @param requestedCwd working directory asked for by the caller ({@code null} ⇒ unset)
|
||||
* @param callerCwd the caller's own working directory, used when nothing else pins one
|
||||
* @param sessionName agent session name (CB-547a); {@code null} ⇒ the launcher mints one
|
||||
* @param resumeSessionId prior agent session to resume; {@code null} ⇒ a fresh session
|
||||
* @param role the contract this member runs under (CB-557). Picks the tab label and,
|
||||
* with the profile, the counter its tab number comes from. {@code null} is
|
||||
* read as {@link MemberRole#DEV} — an unqualified spawn is a unit of work
|
||||
*/
|
||||
public record SpawnRequest(String profileName, String requestedCwd, String callerCwd,
|
||||
String sessionName, String resumeSessionId, MemberRole role) {
|
||||
|
||||
public SpawnRequest {
|
||||
role = (role == null) ? MemberRole.DEV : role;
|
||||
}
|
||||
|
||||
public SpawnRequest(String profileName, String requestedCwd, String callerCwd) {
|
||||
this(profileName, requestedCwd, callerCwd, null, null, null);
|
||||
}
|
||||
|
||||
/** Pre-CB-557 shape: session identity without an explicit role (defaults to {@code dev}). */
|
||||
public SpawnRequest(String profileName, String requestedCwd, String callerCwd,
|
||||
String sessionName, String resumeSessionId) {
|
||||
this(profileName, requestedCwd, callerCwd, sessionName, resumeSessionId, null);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
/**
|
||||
* Backward-compatible placement: an unqualified spawn always resolves to the configured default
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. This ignores caps and
|
||||
* reachability so that a pre-existing config behaves identically after upgrade.
|
||||
*/
|
||||
final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
|
||||
@Override
|
||||
public PlacementCandidate select(PlacementContext ctx) {
|
||||
String d = ctx.defaultProfile();
|
||||
if (d != null && !d.isBlank()) {
|
||||
return new PlacementCandidate(d, null, 1.0f, null);
|
||||
}
|
||||
if (!ctx.candidates().isEmpty()) {
|
||||
PlacementCandidate first = ctx.candidates().getFirst();
|
||||
return new PlacementCandidate(first.profile(), null, first.weight(), first.maxLoad());
|
||||
}
|
||||
throw new PlacementException("no worker profiles configured");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
/**
|
||||
* A profile (and, in CB-308, a host) that can be chosen by a {@link PlacementPolicy}.
|
||||
*
|
||||
* <p>Keeping this as a small descriptor rather than a bare profile name lets CB-308 widen
|
||||
* selection to {@code (host, profile)} pairs without changing the policy interface.
|
||||
*/
|
||||
public record PlacementCandidate(String profile, String host, float weight, Integer maxLoad) {
|
||||
|
||||
/** A candidate with no explicit host (the single-host default) and the given weight/cap. */
|
||||
public static PlacementCandidate profile(String profile, float weight, Integer maxLoad) {
|
||||
return new PlacementCandidate(profile, null, weight, maxLoad);
|
||||
}
|
||||
|
||||
/** A candidate with no explicit host, unit weight, and no cap. */
|
||||
public static PlacementCandidate profile(String profile) {
|
||||
return new PlacementCandidate(profile, null, 1.0f, null);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,19 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
|
||||
/**
|
||||
* Everything a {@link PlacementPolicy} needs to make one selection.
|
||||
*
|
||||
* @param defaultProfile profile a {@code fixed} policy should return (may be {@code null})
|
||||
* @param candidates every configured candidate; the policy filters out those at cap or unreachable
|
||||
* @param liveCount current live worker count per profile (from the session registry)
|
||||
* @param unreachable profiles already known to have failed in this spawn attempt
|
||||
*/
|
||||
public record PlacementContext(String defaultProfile,
|
||||
List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount,
|
||||
Set<String> unreachable) {
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
/**
|
||||
* Thrown when a {@link PlacementPolicy} has no candidate available. Kept as a distinct type so
|
||||
* callers can distinguish "no capacity" from a spawn-time transport failure.
|
||||
*/
|
||||
public final class PlacementException extends IllegalStateException {
|
||||
|
||||
public PlacementException(String message) {
|
||||
super(message);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,41 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
/**
|
||||
* Factory for the built-in placement policies.
|
||||
*/
|
||||
public final class PlacementPolicies {
|
||||
|
||||
private PlacementPolicies() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a policy name from config. Absent/blank values and {@code "fixed"} return the
|
||||
* backward-compatible fixed policy; unknown names throw.
|
||||
*/
|
||||
public static PlacementPolicy fromName(String name) {
|
||||
String n = (name == null) ? "" : name.toLowerCase();
|
||||
if (n.isBlank() || "fixed".equals(n)) {
|
||||
return fixed();
|
||||
}
|
||||
if ("weighted".equals(n)) {
|
||||
return weighted();
|
||||
}
|
||||
if ("round-robin".equals(n)) {
|
||||
return roundRobin();
|
||||
}
|
||||
throw new IllegalArgumentException("unknown placement policy '" + name
|
||||
+ "' — must be one of: fixed, round-robin, weighted");
|
||||
}
|
||||
|
||||
public static PlacementPolicy fixed() {
|
||||
return new FixedPlacementPolicy();
|
||||
}
|
||||
|
||||
public static PlacementPolicy weighted() {
|
||||
return new WeightedRoundRobinPolicy();
|
||||
}
|
||||
|
||||
public static PlacementPolicy roundRobin() {
|
||||
return new RoundRobinPlacementPolicy();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
/**
|
||||
* How {@code bridged} chooses a worker profile when a spawn names none. Implementations are
|
||||
* deterministic and unit-testable; the caller (the composite launcher) handles failover retries.
|
||||
*/
|
||||
public interface PlacementPolicy {
|
||||
|
||||
/**
|
||||
* Pick one candidate from the configured set.
|
||||
*
|
||||
* @throws java.lang.IllegalStateException when no candidate is available, with a message naming
|
||||
* whether every profile is at capacity or unreachable
|
||||
*/
|
||||
PlacementCandidate select(PlacementContext ctx);
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Shared filtering and empty-set reporting used by the built-in placement policies.
|
||||
*/
|
||||
final class PlacementPolicyUtil {
|
||||
|
||||
private PlacementPolicyUtil() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Candidates that are not known-unreachable and have not reached their maxLoad.
|
||||
* A {@code null} maxLoad means unlimited.
|
||||
*/
|
||||
static List<PlacementCandidate> available(PlacementContext ctx) {
|
||||
List<PlacementCandidate> out = new ArrayList<>();
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (ctx.unreachable().contains(c.profile())) {
|
||||
continue;
|
||||
}
|
||||
Integer cap = c.maxLoad();
|
||||
if (cap != null) {
|
||||
int live = ctx.liveCount().apply(c.profile());
|
||||
if (live >= cap) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
out.add(c);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a clear exception describing why every candidate was dropped: all at capacity,
|
||||
* all unreachable, or a mix.
|
||||
*/
|
||||
static PlacementException emptyException(PlacementContext ctx) {
|
||||
int atCap = 0;
|
||||
int unreachable = 0;
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
Integer cap = c.maxLoad();
|
||||
if (ctx.unreachable().contains(c.profile())) {
|
||||
unreachable++;
|
||||
} else if (cap != null && ctx.liveCount().apply(c.profile()) >= cap) {
|
||||
atCap++;
|
||||
}
|
||||
}
|
||||
|
||||
int total = ctx.candidates().size();
|
||||
if (total == 0) {
|
||||
return new PlacementException("no worker profiles configured");
|
||||
}
|
||||
if (atCap == total) {
|
||||
return new PlacementException("all worker profiles are at maxLoad");
|
||||
}
|
||||
if (unreachable == total) {
|
||||
return new PlacementException("all worker profiles are unreachable");
|
||||
}
|
||||
return new PlacementException("no worker profile available: " + atCap + " at maxLoad, "
|
||||
+ unreachable + " unreachable, " + (total - atCap - unreachable) + " remaining");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
|
||||
/**
|
||||
* Deterministic round-robin over the profiles that still have capacity and are not known to be
|
||||
* unreachable. The index advances only on successful selections so the distribution stays even
|
||||
* across spawns.
|
||||
*/
|
||||
final class RoundRobinPlacementPolicy implements PlacementPolicy {
|
||||
|
||||
private final AtomicInteger index = new AtomicInteger(0);
|
||||
|
||||
@Override
|
||||
public synchronized PlacementCandidate select(PlacementContext ctx) {
|
||||
List<PlacementCandidate> available = PlacementPolicyUtil.available(ctx);
|
||||
if (available.isEmpty()) {
|
||||
throw PlacementPolicyUtil.emptyException(ctx);
|
||||
}
|
||||
return available.get(index.getAndIncrement() % available.size());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* Smooth weighted round-robin (nginx-style): for each selection, add the candidate's weight to
|
||||
* its current score, pick the highest score, then subtract the total weight of all available
|
||||
* candidates from the winner. Weights are not required to sum to 1.0; only their ratios matter.
|
||||
*
|
||||
* <p>The state is per-policy instance and protected by {@code synchronized} so concurrent spawns
|
||||
* see a consistent, deterministic sequence rather than interleaving updates.
|
||||
*/
|
||||
final class WeightedRoundRobinPolicy implements PlacementPolicy {
|
||||
|
||||
private final Map<String, Double> current = new ConcurrentHashMap<>();
|
||||
|
||||
@Override
|
||||
public synchronized PlacementCandidate select(PlacementContext ctx) {
|
||||
List<PlacementCandidate> available = PlacementPolicyUtil.available(ctx);
|
||||
if (available.isEmpty()) {
|
||||
throw PlacementPolicyUtil.emptyException(ctx);
|
||||
}
|
||||
|
||||
double total = 0.0;
|
||||
for (PlacementCandidate c : available) {
|
||||
total += c.weight();
|
||||
}
|
||||
if (total <= 0.0) {
|
||||
throw new PlacementException("all available profiles have non-positive weight");
|
||||
}
|
||||
|
||||
PlacementCandidate best = null;
|
||||
double bestScore = Double.NEGATIVE_INFINITY;
|
||||
for (PlacementCandidate c : available) {
|
||||
double score = current.merge(c.profile(), (double) c.weight(), (old, add) -> old + add);
|
||||
if (score > bestScore) {
|
||||
bestScore = score;
|
||||
best = c;
|
||||
}
|
||||
}
|
||||
if (best == null) {
|
||||
throw new PlacementException("no placement candidate could be selected");
|
||||
}
|
||||
|
||||
current.put(best.profile(), current.get(best.profile()) - total);
|
||||
return best;
|
||||
}
|
||||
}
|
||||
@@ -1,18 +1,35 @@
|
||||
package dev.ltms.bridged.rest;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.bridged.auth.AuditLog;
|
||||
import dev.ltms.bridged.auth.Authz;
|
||||
import dev.ltms.bridged.auth.CallerResolver;
|
||||
import dev.ltms.bridged.auth.Principal;
|
||||
import dev.ltms.bridged.guard.GuardException;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.HerdrClient;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.worker.WorkerService;
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import dev.ltms.bridged.session.WorktreeRequest;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import io.javalin.Javalin;
|
||||
import io.javalin.http.Context;
|
||||
import jakarta.servlet.http.HttpServlet;
|
||||
import org.eclipse.jetty.servlet.ServletHolder;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* The REST surface — {@code bridged}'s contract and its testability seam. Every
|
||||
@@ -25,25 +42,141 @@ import java.util.Map;
|
||||
*/
|
||||
public final class BridgedApp {
|
||||
|
||||
private final HerdrClient herdr;
|
||||
private final WorkerService workers;
|
||||
/** Default blocking window for a message; kept under typical HTTP idle timeouts. */
|
||||
private static final long DEFAULT_MESSAGE_TIMEOUT_MS = 25_000;
|
||||
private static final long MAX_MESSAGE_TIMEOUT_MS = 120_000;
|
||||
/** Blocking window for a worker's bridge_ask (CB-205); the worker's MCP client caps its own call. */
|
||||
private static final long DEFAULT_ASK_TIMEOUT_MS = 55_000;
|
||||
private static final long MAX_ASK_TIMEOUT_MS = 115_000;
|
||||
|
||||
public BridgedApp(HerdrClient herdr, WorkerService workers) {
|
||||
/** Context attribute under which the resolved caller is stashed by the auth filter. */
|
||||
private static final String CALLER = "bridged.caller";
|
||||
|
||||
private final HerdrClient herdr;
|
||||
private final PeerLauncher workers;
|
||||
private final SessionManager sessions; // CB-301: authoritative session registry
|
||||
private final MessageService messages;
|
||||
private final MemberPresence presence; // CB-113: which workers are MCP-connected (available)
|
||||
private final HttpServlet mcpServlet; // MCP Streamable-HTTP endpoint, mounted at /mcp (nullable)
|
||||
private final CallerResolver auth; // CB-501: null → authz not enforced (legacy behaviour)
|
||||
private final Metrics metrics; // CB-502: null → /metrics not exposed
|
||||
private final ObjectMapper mapper = new ObjectMapper();
|
||||
|
||||
/**
|
||||
* Legacy constructor — no identity resolution and no authorization, exactly as the REST surface
|
||||
* behaved before CB-501. Retained so existing acceptance tests keep exercising handler
|
||||
* behaviour without each needing an auth fixture.
|
||||
*/
|
||||
public BridgedApp(HerdrClient herdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet) {
|
||||
this(herdr, workers, sessions, messages, presence, mcpServlet, null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param auth resolves each request's {@link Principal}; {@code null} disables authorization
|
||||
* entirely (legacy). {@code main} always supplies one.
|
||||
* @param metrics registry to instrument and expose at {@code GET /metrics}; {@code null} omits
|
||||
* the endpoint
|
||||
*/
|
||||
public BridgedApp(HerdrClient herdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics) {
|
||||
this.herdr = herdr;
|
||||
this.workers = workers;
|
||||
this.sessions = sessions;
|
||||
this.messages = messages;
|
||||
this.presence = presence;
|
||||
this.mcpServlet = mcpServlet;
|
||||
this.auth = auth;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Wire routes onto a fresh, unstarted Javalin instance. Caller starts it. */
|
||||
public Javalin build() {
|
||||
Javalin app = Javalin.create(cfg -> cfg.showJavalinBanner = false);
|
||||
Javalin app = Javalin.create(cfg -> {
|
||||
cfg.showJavalinBanner = false;
|
||||
if (mcpServlet != null) {
|
||||
// The MCP server shares the daemon's port; Jetty routes /mcp to its servlet.
|
||||
cfg.jetty.modifyServletContextHandler(h ->
|
||||
h.addServlet(new ServletHolder(mcpServlet), "/mcp"));
|
||||
}
|
||||
});
|
||||
// CB-501: resolve identity once per request, before any handler. /mcp does NOT pass through
|
||||
// here — it is a raw servlet on Jetty's context handler — so BridgeMcp enforces separately
|
||||
// against the same CallerResolver. Any check that lives in only one place is not a control.
|
||||
if (auth != null) {
|
||||
app.before(ctx -> ctx.attribute(CALLER,
|
||||
auth.resolve(ctx.req().getRemoteAddr(), ctx.req().getRemotePort(),
|
||||
ctx.header("Authorization"))));
|
||||
}
|
||||
app.get("/healthz", this::healthz);
|
||||
if (metrics != null) {
|
||||
app.get("/metrics", this::metrics);
|
||||
}
|
||||
app.get("/sessions", this::sessions);
|
||||
app.get("/agents", this::agents);
|
||||
app.post("/workers", this::spawnWorker);
|
||||
app.delete("/workers/{paneId}", this::stopWorker);
|
||||
app.get("/members", this::listMembers); // CB-304: registry roster + live herdr status
|
||||
app.get("/profiles", this::profiles); // configured backend profiles
|
||||
app.post("/members", this::spawnMember); // optional ?role=&profile= or {"role":…,"profile":…}
|
||||
app.delete("/members/{paneId}", this::stopMember);
|
||||
app.post("/sessions/{id}/message", this::sendMessage); // bridge_send (primary; blocking, wait:false, or answer via turnId)
|
||||
app.post("/sessions/{id}/reply", this::replyMessage); // bridge_reply (worker)
|
||||
app.get("/sessions/{id}/replies", this::drainReplies); // drain reply inbox (CB-307)
|
||||
app.post("/sessions/{id}/ask", this::askMessage); // bridge_ask (worker → primary, CB-205)
|
||||
app.get("/sessions/{id}/status", this::sessionStatus); // bridge_status
|
||||
app.get("/tasks/{ticket}", this::taskStatus); // poll an async (wait:false) send
|
||||
return app;
|
||||
}
|
||||
|
||||
/**
|
||||
* Gate a handler on the CB-505 authorization table. Returns {@code true} when the request may
|
||||
* proceed; otherwise writes the error response and returns {@code false}.
|
||||
*
|
||||
* <p>401 vs 403 is a real distinction here: 401 means "you presented no usable identity" (a
|
||||
* credential problem the caller can fix), 403 means "you are authenticated, but this is not
|
||||
* yours" (a worker reaching for another worker's session, or for orchestration).
|
||||
*/
|
||||
private boolean allow(Context ctx, Authz.Action action, String target) {
|
||||
if (auth == null) {
|
||||
return true; // legacy: authorization not enforced
|
||||
}
|
||||
Principal caller = ctx.attribute(CALLER);
|
||||
if (Authz.permits(caller, action, target)) {
|
||||
if (action != Authz.Action.READ && action != Authz.Action.METRICS) {
|
||||
AuditLog.allowed(caller, action, target); // reads would drown the trail
|
||||
}
|
||||
return true;
|
||||
}
|
||||
if (Authz.isUnauthenticated(caller)) {
|
||||
AuditLog.denied(caller, action, target, "unauthenticated");
|
||||
countAuthFailure("unauthenticated");
|
||||
ctx.status(401).json(Map.of("error", "unauthenticated",
|
||||
"detail", "present Authorization: Bearer <token>"));
|
||||
} else {
|
||||
AuditLog.denied(caller, action, target, "forbidden");
|
||||
countAuthFailure("forbidden");
|
||||
ctx.status(403).json(Map.of("error", "forbidden",
|
||||
"detail", caller.describe() + " may not " + action + " on "
|
||||
+ (target == null ? "this resource" : target)));
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
private void countAuthFailure(String reason) {
|
||||
if (metrics != null) {
|
||||
metrics.inc("bridged_auth_failures_total", "reason", reason);
|
||||
}
|
||||
}
|
||||
|
||||
/** Prometheus scrape endpoint (CB-502). */
|
||||
private void metrics(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.METRICS, null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).contentType("text/plain; version=0.0.4; charset=utf-8").result(metrics.render());
|
||||
}
|
||||
|
||||
/** Liveness + herdr reachability. 200 when herdr answers ping, 503 otherwise. */
|
||||
private void healthz(Context ctx) {
|
||||
try {
|
||||
@@ -63,6 +196,9 @@ public final class BridgedApp {
|
||||
|
||||
/** Sessions view derived from herdr {@code workspace.list} (one workspace → one row). */
|
||||
private void sessions(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
return;
|
||||
}
|
||||
JsonNode result = herdr.call("workspace.list");
|
||||
List<Map<String, Object>> out = new ArrayList<>();
|
||||
for (JsonNode w : result.path("workspaces")) {
|
||||
@@ -78,25 +214,336 @@ public final class BridgedApp {
|
||||
|
||||
/** Discovery: every agent herdr tracks, keyed by its Claude session UUID. */
|
||||
private void agents(Context ctx) {
|
||||
ctx.status(200).json(Map.of("agents", workers.list().stream().map(BridgedApp::view).toList()));
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).json(Map.of("agents",
|
||||
workers.list().stream().map(Agent.class::cast).map(BridgedApp::view).toList()));
|
||||
}
|
||||
|
||||
/** Spawn a guard-checked worker. 403 if the base_url would breach the subscription boundary. */
|
||||
private void spawnWorker(Context ctx) {
|
||||
/** CB-304: bridge-owned roster merged with live herdr status by paneId. */
|
||||
private void listMembers(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
return;
|
||||
}
|
||||
// CB-519: the registry key is a host-unique id, not the pane coordinate — join on terminal.
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.map(Agent.class::cast)
|
||||
.filter(a -> a.terminalId() != null)
|
||||
.collect(Collectors.toMap(Agent::terminalId, Function.identity(), (_, b) -> b));
|
||||
List<Map<String, Object>> out = sessions.roster().stream()
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.terminalId())))
|
||||
.toList();
|
||||
ctx.status(200).json(Map.of("workers", out));
|
||||
}
|
||||
|
||||
/** The configured worker profiles and which one a no-argument spawn uses. */
|
||||
private void profiles(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).json(Map.of(
|
||||
"profiles", workers.profiles(),
|
||||
"default", workers.defaultProfile() == null ? "" : workers.defaultProfile()));
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a guard-checked worker. An optional {@code profile} (query param or {@code {"profile":…}}
|
||||
* body) picks which configured profile; omitted → the default. 403 if the base_url would breach
|
||||
* the subscription boundary, 400 for an unknown profile.
|
||||
*/
|
||||
private void spawnMember(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.SPAWN, null)) {
|
||||
return;
|
||||
}
|
||||
String role = ctx.queryParam("role");
|
||||
String profile = ctx.queryParam("profile");
|
||||
String cwd = ctx.queryParam("cwd");
|
||||
String worktree = ctx.queryParam("worktree");
|
||||
String ticket = ctx.queryParam("ticket");
|
||||
if (profile == null || profile.isBlank() || cwd == null || cwd.isBlank()
|
||||
|| worktree == null || worktree.isBlank()) {
|
||||
try {
|
||||
String body = ctx.body();
|
||||
if (!body.isBlank()) {
|
||||
JsonNode b = mapper.readTree(body);
|
||||
if (role == null || role.isBlank()) role = b.path("role").asText(null);
|
||||
if (profile == null || profile.isBlank()) profile = b.path("profile").asText(null);
|
||||
if (cwd == null || cwd.isBlank()) cwd = b.path("cwd").asText(null);
|
||||
if (worktree == null || worktree.isBlank()) worktree = b.path("worktree").asText(null);
|
||||
if (ticket == null || ticket.isBlank()) ticket = b.path("ticket").asText(null);
|
||||
}
|
||||
} catch (Exception ignored) {
|
||||
// A malformed/empty body just means "no overrides" → fall through to defaults.
|
||||
}
|
||||
}
|
||||
WorktreeRequest wt = worktreeRequest(worktree, ticket);
|
||||
MemberRole memberRole;
|
||||
try {
|
||||
Agent worker = workers.spawn();
|
||||
ctx.status(201).json(view(worker));
|
||||
memberRole = (role == null || role.isBlank()) ? MemberRole.DEV : MemberRole.parse(role);
|
||||
} catch (IllegalArgumentException e) {
|
||||
ctx.status(400).json(Map.of("error", "unknown_role", "detail", e.getMessage()));
|
||||
return;
|
||||
}
|
||||
try {
|
||||
// No MCP caller over REST, so callerCwd and ownerTerminal are null.
|
||||
MemberSession member = sessions.acquire(blankToNull(profile), memberRole,
|
||||
blankToNull(cwd), null, null, wt);
|
||||
ctx.status(201).json(view(member));
|
||||
} catch (GuardException e) {
|
||||
ctx.status(403).json(Map.of("error", "subscription_boundary", "detail", e.getMessage()));
|
||||
} catch (IllegalArgumentException e) {
|
||||
ctx.status(400).json(Map.of("error", "unknown_profile", "detail", e.getMessage()));
|
||||
} catch (PeerUnreachableException e) {
|
||||
ctx.status(502).json(Map.of("error", "spawn_timeout", "detail", e.getMessage()));
|
||||
}
|
||||
}
|
||||
|
||||
private static WorktreeRequest worktreeRequest(String worktree, String ticket) {
|
||||
if (worktree == null || worktree.isBlank() || "false".equalsIgnoreCase(worktree)) {
|
||||
return null;
|
||||
}
|
||||
if ("true".equalsIgnoreCase(worktree)) {
|
||||
if (ticket == null || ticket.isBlank()) {
|
||||
throw new IllegalArgumentException("worktree=true requires a ticket slug");
|
||||
}
|
||||
return new WorktreeRequest(ticket, null);
|
||||
}
|
||||
return new WorktreeRequest(worktree, null);
|
||||
}
|
||||
|
||||
private static String blankToNull(String s) {
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
}
|
||||
|
||||
/** Tear a worker down by pane id. */
|
||||
private void stopWorker(Context ctx) {
|
||||
workers.stop(ctx.pathParam("paneId"));
|
||||
private void stopMember(Context ctx) {
|
||||
String paneId = ctx.pathParam("paneId");
|
||||
if (!allow(ctx, Authz.Action.STOP, paneId)) {
|
||||
return;
|
||||
}
|
||||
sessions.release(paneId);
|
||||
ctx.status(204);
|
||||
}
|
||||
|
||||
/**
|
||||
* The blocking delegation call (CB-104): inject {@code content} into the worker via the
|
||||
* status-gated injector and block until the worker returns a structured {@code bridge_reply}.
|
||||
* Times out with a typed 202 (working / queued / busy) rather than an error — the message may
|
||||
* still land.
|
||||
*/
|
||||
private void sendMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.SEND, id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
String turnId;
|
||||
long timeout;
|
||||
boolean wait;
|
||||
try {
|
||||
JsonNode body = mapper.readTree(ctx.body());
|
||||
content = body.path("content").asText("");
|
||||
turnId = body.path("turnId").asText(null);
|
||||
timeout = body.path("timeoutMs").asLong(DEFAULT_MESSAGE_TIMEOUT_MS);
|
||||
wait = body.path("wait").asBoolean(true); // default: block for the reply (CB-104)
|
||||
} catch (Exception e) {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "body must be JSON"));
|
||||
return;
|
||||
}
|
||||
if (content.isBlank()) {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "content is required"));
|
||||
return;
|
||||
}
|
||||
timeout = Math.clamp(timeout, 1, MAX_MESSAGE_TIMEOUT_MS);
|
||||
|
||||
// Answering a worker's bridge_ask (CB-205): always blocks, and derives the worker from turnId.
|
||||
if (turnId != null && !turnId.isBlank()) {
|
||||
writeReply(ctx, id, messages.answer(turnId, content, timeout), timeout);
|
||||
return;
|
||||
}
|
||||
|
||||
if (!wait) {
|
||||
// Fire-and-poll (CB-107): return a ticket immediately; the caller polls GET /tasks/{ticket}.
|
||||
String ticket = messages.sendAsync(id, content);
|
||||
ctx.status(202).json(Map.of("sessionId", id, "ticket", ticket, "status", "accepted"));
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
writeReply(ctx, id, messages.send(id, content, timeout), timeout);
|
||||
} catch (HerdrException e) {
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Render a {@link MessageService.Reply} onto the response — shared by a normal send and a
|
||||
* bridge_ask answer. A structured/scraped completion is 200; a worker's mid-turn question a 202
|
||||
* (with its {@code turnId}); a stale answer a 409; every other non-terminal outcome a typed 202.
|
||||
*/
|
||||
private void writeReply(Context ctx, String id, MessageService.Reply reply, long timeout) {
|
||||
switch (reply.outcome()) {
|
||||
case QUESTION -> ctx.status(202).json(Map.of(
|
||||
"sessionId", id, "status", "question",
|
||||
"question", reply.text(), "turnId", reply.turnId()));
|
||||
case STALE_TURN -> ctx.status(409).json(Map.of(
|
||||
"sessionId", id, "error", "stale_turn",
|
||||
"detail", "that question is no longer open (timed out or already answered)"));
|
||||
case REPLIED, COMPLETED_UNREPLIED -> {
|
||||
// replySource distinguishes a structured bridge_reply from the CB-106 completion
|
||||
// fallback (a scrape of the worker's transcript when it finished without replying).
|
||||
String source = reply.outcome() == MessageService.Outcome.REPLIED ? "reply" : "transcript";
|
||||
ctx.status(200).json(Map.of("sessionId", id, "reply", reply.text(), "replySource", source));
|
||||
}
|
||||
default -> ctx.status(202).json(Map.of(
|
||||
"sessionId", id,
|
||||
"status", switch (reply.outcome()) {
|
||||
case TIMED_OUT_WORKING -> "working";
|
||||
case TIMED_OUT_QUEUED -> "queued";
|
||||
case BUSY -> "busy";
|
||||
case WORKER_FAILED -> "failed";
|
||||
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
||||
default -> "done"; // unreachable (terminal outcomes handled above)
|
||||
},
|
||||
"detail", (reply.outcome() == MessageService.Outcome.WORKER_FAILED
|
||||
|| reply.outcome() == MessageService.Outcome.BACKEND_EXHAUSTED)
|
||||
&& reply.text() != null
|
||||
? reply.text()
|
||||
: "no reply within " + timeout + "ms; poll status or retry"));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A worker's mid-turn question ({@code bridge_ask}, CB-205) — surfaces to the primary's open
|
||||
* blocking send and blocks until it answers. 200 with the answer, 409 if no delegation is open,
|
||||
* 202 if the primary stayed silent.
|
||||
*/
|
||||
private void askMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.ASK, id)) {
|
||||
return;
|
||||
}
|
||||
String question;
|
||||
long timeout;
|
||||
try {
|
||||
JsonNode body = mapper.readTree(ctx.body());
|
||||
question = body.path("question").asText("");
|
||||
timeout = body.path("timeoutMs").asLong(DEFAULT_ASK_TIMEOUT_MS);
|
||||
} catch (Exception e) {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "body must be JSON"));
|
||||
return;
|
||||
}
|
||||
if (question.isBlank()) {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "question is required"));
|
||||
return;
|
||||
}
|
||||
timeout = Math.clamp(timeout, 1, MAX_ASK_TIMEOUT_MS);
|
||||
MessageService.AskResult r = messages.ask(id, question, timeout);
|
||||
switch (r.outcome()) {
|
||||
case ANSWERED -> ctx.status(200).json(Map.of("sessionId", id, "answered", true, "answer", r.answer()));
|
||||
case NO_WAITER -> ctx.status(409).json(Map.of(
|
||||
"sessionId", id, "error", "no_pending_send",
|
||||
"detail", "no primary is awaiting this turn to answer a question"));
|
||||
case TIMED_OUT -> ctx.status(202).json(Map.of(
|
||||
"sessionId", id, "status", "no_answer",
|
||||
"detail", "the primary did not answer within " + timeout + "ms"));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The worker's structured reply ({@code bridge_reply}) — resolves the blocking send awaiting
|
||||
* on this session, or queues the reply in the inbox when no send is open (CB-307).
|
||||
*/
|
||||
private void replyMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
// The rule that matters: a worker may reply only as itself. Over MCP this was already true
|
||||
// structurally (identity comes from the connection, never an argument); over REST the path
|
||||
// id was simply trusted, so this is where the invariant actually gets enforced.
|
||||
if (!allow(ctx, Authz.Action.REPLY, id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
try {
|
||||
content = mapper.readTree(ctx.body()).path("content").asText("");
|
||||
} catch (Exception e) {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "body must be JSON"));
|
||||
return;
|
||||
}
|
||||
messages.reply(id, content);
|
||||
ctx.status(200).json(Map.of("sessionId", id, "delivered", true));
|
||||
}
|
||||
|
||||
/**
|
||||
* Drain the reply inbox for a worker session — peek + ack any replies that arrived when no send
|
||||
* was open. At-least-once: draining removes them from the inbox so a subsequent read returns
|
||||
* nothing; an in-flight failure between the drain and the caller's processing re-surfaces them.
|
||||
*/
|
||||
private void drainReplies(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.DRAIN, id)) {
|
||||
return;
|
||||
}
|
||||
var replies = messages.drainReplies(id);
|
||||
ctx.status(200).json(Map.of("sessionId", id, "replies",
|
||||
replies.stream().map(m -> Map.of(
|
||||
"msgId", m.msgId(),
|
||||
"content", m.content())).toList()));
|
||||
}
|
||||
|
||||
/**
|
||||
* Live lifecycle status of a worker (MCP `bridge_status` wraps this in CB-105), plus its
|
||||
* <em>readiness</em> (CB-113): {@code ready} is true once the worker's Claude has connected the
|
||||
* bridge MCP — the reliable "available to receive a task" signal, unlike bare {@code idle}, which
|
||||
* is also true during boot.
|
||||
*/
|
||||
private void sessionStatus(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.READ, id)) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
ctx.status(200).json(Map.of(
|
||||
"sessionId", id,
|
||||
"status", messages.status(id).name().toLowerCase(),
|
||||
"ready", presence.isPresent(id)));
|
||||
} catch (HerdrException e) {
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** Poll an async (wait:false) delegation by ticket. 404 for an unknown/expired ticket. */
|
||||
private void taskStatus(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
return;
|
||||
}
|
||||
MessageService.TaskView v = messages.poll(ctx.pathParam("ticket"));
|
||||
if (v == null) {
|
||||
ctx.status(404).json(Map.of("error", "unknown_ticket", "detail", "no such task (or it has expired)"));
|
||||
return;
|
||||
}
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
body.put("ticket", v.ticket());
|
||||
body.put("phase", v.phase().name().toLowerCase());
|
||||
if (v.reply() != null) {
|
||||
body.put("reply", v.reply());
|
||||
body.put("replySource", v.replySource());
|
||||
}
|
||||
if (v.detail() != null) {
|
||||
body.put("detail", v.detail());
|
||||
}
|
||||
ctx.status(200).json(body);
|
||||
}
|
||||
|
||||
/** Map a herdr failure: unknown target → 404, anything else → 502 (herdr is upstream). */
|
||||
private static void herdrError(Context ctx, HerdrException e) {
|
||||
if (e.code() != null && e.code().endsWith("_not_found")) {
|
||||
ctx.status(404).json(Map.of("error", "session_not_found", "detail", e.getMessage()));
|
||||
} else {
|
||||
ctx.status(502).json(Map.of("error", "herdr_error", "detail", e.getMessage()));
|
||||
}
|
||||
}
|
||||
|
||||
/** Stable JSON projection of an agent (null-safe for the start-time shape). */
|
||||
private static Map<String, Object> view(Agent a) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
@@ -109,4 +556,22 @@ public final class BridgedApp {
|
||||
m.put("status", a.status().name().toLowerCase());
|
||||
return m;
|
||||
}
|
||||
|
||||
/** CB-301 projection of an authoritative bridge-owned session. */
|
||||
private static Map<String, Object> view(MemberSession s) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("terminalId", s.terminalId());
|
||||
m.put("paneId", s.paneId());
|
||||
m.put("profile", s.profile());
|
||||
m.put("cwd", s.cwd());
|
||||
m.put("ownerTerminal", s.ownerTerminal());
|
||||
m.put("state", s.state().name().toLowerCase());
|
||||
if (s.worktree() != null) {
|
||||
m.put("worktree", s.worktree());
|
||||
}
|
||||
if (s.branch() != null) {
|
||||
m.put("branch", s.branch());
|
||||
}
|
||||
return m;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,293 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStreamReader;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.security.SecureRandom;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* Production {@link Worktrees} implementation that shells {@code git} via {@link ProcessBuilder}.
|
||||
* Non-zero exits become {@link WorktreeException}. Worktree directories live under a configurable
|
||||
* root (default: a sibling {@code .bridged-worktrees} of the repo root) so they are never nested
|
||||
* inside the primary working tree.
|
||||
*/
|
||||
public final class GitWorktrees implements Worktrees {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(GitWorktrees.class);
|
||||
|
||||
/** Project-level MCP config. Present in the repo, so every worktree would otherwise inherit the
|
||||
* primary's IDE server mounts (a CB-523 worker edited the primary checkout; see the isolation
|
||||
* javadoc). Neutralized unconditionally. */
|
||||
private static final String MCP_CONFIG = ".mcp.json";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code .mcp.json}: a valid, explicitly empty server map. */
|
||||
private static final String NEUTRAL_MCP_CONFIG = "{\n \"mcpServers\": {}\n}\n";
|
||||
|
||||
/** OpenCode's repo-level config. Tracked here, so it lands in every worktree, and it mounts the
|
||||
* primary's gitea and context7 servers with the primary's credentials. Neutralized so the worker
|
||||
* gets only the config its launcher writes via {@code OPENCODE_CONFIG}.
|
||||
*
|
||||
* <p>The reason has changed shape and is now stronger. It used to be a crash: the file carried
|
||||
* {@code {file:.secrets/...}} references to gitignored files that never reached a worktree, and
|
||||
* opencode refuses to start on a dangling reference (CB-543). Those credentials now live in one
|
||||
* shell-level store and the file reads them as {@code {env:...}}, so in a worktree the reference
|
||||
* resolves instead of failing. That is worse, not better: a member would silently inherit the
|
||||
* primary's admin-scoped {@code GITEA_ACCESS_TOKEN}. A loud crash became a quiet privilege leak,
|
||||
* so this entry protects a boundary now rather than papering over a startup error. */
|
||||
private static final String OPENCODE_CONFIG = "opencode.json";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code opencode.json}: a valid, empty JSON object. */
|
||||
private static final String NEUTRAL_OPENCODE_CONFIG = "{}\n";
|
||||
|
||||
/** Autoenv's repo-level config. Not tracked today, but re-landing it must stay safe: autoenv
|
||||
* authorizes by path, so a fresh worktree path is always unauthorized and its interactive prompt
|
||||
* would block every spawn — neutralize it so it can never be committed. */
|
||||
private static final String AUTOENV_CONFIG = ".autoenv";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code .autoenv}: a valid, empty env file. */
|
||||
private static final String NEUTRAL_AUTOENV_CONFIG = "";
|
||||
|
||||
/**
|
||||
* A tracked project config that is hostile in a provisioned worktree, and what to replace it
|
||||
* with. {@link #file} is the repo-relative path; {@link #stub} is a neutral but VALID payload for
|
||||
* that file's format — a malformed stub would only trade one crash for another;
|
||||
* {@link #createIfAbsent} keeps {@code .mcp.json}'s long-standing behaviour of writing its stub
|
||||
* even when the repo carries no such file, whereas the others are only touched when present.
|
||||
*/
|
||||
private record WorktreeHostileConfig(String file, String stub, boolean createIfAbsent) {}
|
||||
|
||||
/** The worktree-hostile configs neutralized in every provisioned worktree, in order. */
|
||||
private static final List<WorktreeHostileConfig> WORKTREE_HOSTILE_CONFIGS = List.of(
|
||||
new WorktreeHostileConfig(MCP_CONFIG, NEUTRAL_MCP_CONFIG, true),
|
||||
new WorktreeHostileConfig(OPENCODE_CONFIG, NEUTRAL_OPENCODE_CONFIG, false),
|
||||
new WorktreeHostileConfig(AUTOENV_CONFIG, NEUTRAL_AUTOENV_CONFIG, false)
|
||||
);
|
||||
|
||||
private final String configuredRoot;
|
||||
private final SecureRandom random = new SecureRandom();
|
||||
private final AtomicLong seq = new AtomicLong();
|
||||
|
||||
/** Default constructor: worktree root is derived per-repo as {@code <repoRoot>/../.bridged-worktrees}. */
|
||||
public GitWorktrees() {
|
||||
this(null);
|
||||
}
|
||||
|
||||
/** @param configuredRoot nullable absolute or relative path; null/blank derives a sibling of the repo root. */
|
||||
public GitWorktrees(String configuredRoot) {
|
||||
this.configuredRoot = configuredRoot;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String add(String repoRoot, String branch, String baseRef) {
|
||||
String base = (baseRef == null || baseRef.isBlank()) ? "HEAD" : baseRef;
|
||||
String nonce = nonce();
|
||||
Path root = resolveRoot(repoRoot);
|
||||
Path path = root.resolve(nonce);
|
||||
try {
|
||||
Files.createDirectories(root);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot create worktree root " + root + ": " + e.getMessage(), e);
|
||||
}
|
||||
String wt = path.toAbsolutePath().toString();
|
||||
log.info("adding worktree branch={} path={} base={}", branch, wt, base);
|
||||
exec("git", "-C", repoRoot, "worktree", "add", wt, "-b", branch, base);
|
||||
isolateToolSurface(wt);
|
||||
return wt;
|
||||
}
|
||||
|
||||
/**
|
||||
* Neutralize the worktree's worktree-hostile project configs so a worker inherits only the tools
|
||||
* and environment its launcher mounts (the bridge via {@code --mcp-config}, the opencode config
|
||||
* via {@code OPENCODE_CONFIG}) — never the primary's.
|
||||
*
|
||||
* <p>This is unconditional, and it is not the same job as the parity overlay. The repo's own
|
||||
* committed {@code .mcp.json} declares the primary's IDE servers, so a fresh checkout mounts them
|
||||
* whether or not the overlay copies anything; a worker that inherits them navigates and edits
|
||||
* through tools bound to the <em>primary's</em> IntelliJ project, which silently hands it absolute
|
||||
* paths outside its own worktree. That is not hypothetical: a CB-523 worker made all 59 of its
|
||||
* edits in the primary checkout while compiling its worktree, so every build it ran was of code
|
||||
* that did not contain its changes. {@code opencode.json} is the same trap one tool over — tracked,
|
||||
* so it lands in every worktree, and it mounts gitea and context7 with the primary's own
|
||||
* credentials, which a member must never hold. {@code .autoenv} extends the principle to a
|
||||
* config that is not tracked today: autoenv authorizes by path, so a fresh worktree path is always
|
||||
* unauthorized and its interactive prompt would block every spawn, so re-landing one must be safe.
|
||||
*
|
||||
* <p>Where a config exists it is replaced by a valid neutral stub (an explicitly empty
|
||||
* map/object, or an empty env file — never a deletion, which would still let a later
|
||||
* {@code git checkout} restore the hostile copy). The {@code --skip-worktree} bit keeps the
|
||||
* neutralized copy from ever showing up as a local modification the worker might commit. A config
|
||||
* the repo does not carry is skipped silently — no stub is invented for a file the repo does not
|
||||
* have, and one missing file must never fail provisioning.
|
||||
*/
|
||||
private void isolateToolSurface(String worktreePath) {
|
||||
Path root = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
for (WorktreeHostileConfig cfg : WORKTREE_HOSTILE_CONFIGS) {
|
||||
neutralize(root, worktreePath, cfg);
|
||||
}
|
||||
}
|
||||
|
||||
private void neutralize(Path root, String worktreePath, WorktreeHostileConfig cfg) {
|
||||
Path target = root.resolve(cfg.file());
|
||||
if (!Files.exists(target) && !cfg.createIfAbsent()) {
|
||||
log.debug("{} absent in the worktree — skipping (repo does not carry it)", cfg.file());
|
||||
return;
|
||||
}
|
||||
try {
|
||||
Files.writeString(target, cfg.stub());
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot neutralize " + cfg.file() + " in the worktree: "
|
||||
+ e.getMessage(), e);
|
||||
}
|
||||
if (isTracked(root, cfg.file())) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", cfg.file());
|
||||
}
|
||||
log.debug("neutralized {} — worker tool surface is launcher-mounted only", cfg.file());
|
||||
}
|
||||
|
||||
@Override
|
||||
public void remove(String repoRoot, String worktreePath) {
|
||||
Path p = Path.of(worktreePath);
|
||||
if (!Files.exists(p)) {
|
||||
log.debug("worktree {} already gone — nothing to remove", worktreePath);
|
||||
return;
|
||||
}
|
||||
log.info("removing worktree {}", worktreePath);
|
||||
exec("git", "-C", repoRoot, "worktree", "remove", "--force", worktreePath);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasUncommitted(String worktreePath) {
|
||||
// A worktree that is already gone holds no work to lose, and it must not break teardown:
|
||||
// git -C <missing-dir> status exits non-zero and would throw where release() is mid-way
|
||||
// through stopping a pane. Mirror remove()'s already-gone tolerance by treating it as clean.
|
||||
Path p = Path.of(worktreePath);
|
||||
if (!Files.exists(p)) {
|
||||
log.debug("worktree {} already gone — nothing can be uncommitted", worktreePath);
|
||||
return false;
|
||||
}
|
||||
// No --untracked-files=no: the exact shape of the work lost in CB-576 was a new file
|
||||
// that was never added, so an untracked-only worktree is still dirty.
|
||||
String out = exec("git", "-C", worktreePath, "status", "--porcelain");
|
||||
return !out.isBlank();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void overlayParity(String repoRoot, String worktreePath, List<String> overlay) {
|
||||
if (overlay == null || overlay.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
Path srcRoot = Path.of(repoRoot).toAbsolutePath().normalize();
|
||||
Path dstRoot = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
for (String rel : overlay) {
|
||||
Path src = srcRoot.resolve(rel).normalize();
|
||||
if (!Files.exists(src)) {
|
||||
log.debug("parity overlay source missing — skipping {}", rel);
|
||||
continue;
|
||||
}
|
||||
Path dst = dstRoot.resolve(rel).normalize();
|
||||
try {
|
||||
Files.createDirectories(dst.getParent());
|
||||
Files.copy(src, dst, StandardCopyOption.REPLACE_EXISTING, StandardCopyOption.COPY_ATTRIBUTES);
|
||||
log.debug("copied parity overlay {}", rel);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot copy overlay " + rel + ": " + e.getMessage(), e);
|
||||
}
|
||||
if (isTracked(dstRoot, rel)) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", rel);
|
||||
log.debug("marked overlay --skip-worktree {}", rel);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public String repoRoot(String cwd) {
|
||||
String out = exec("git", "-C", cwd, "rev-parse", "--show-toplevel");
|
||||
return Path.of(out.trim()).toAbsolutePath().normalize().toString();
|
||||
}
|
||||
|
||||
/** Resolve the directory that will hold per-session worktree checkouts. */
|
||||
private Path resolveRoot(String repoRoot) {
|
||||
if (configuredRoot != null && !configuredRoot.isBlank()) {
|
||||
return Path.of(configuredRoot).toAbsolutePath().normalize();
|
||||
}
|
||||
Path repo = Path.of(repoRoot).toAbsolutePath().normalize();
|
||||
return repo.resolveSibling(".bridged-worktrees");
|
||||
}
|
||||
|
||||
private String nonce() {
|
||||
return String.format("%06x", random.nextInt(1 << 24)) + "-" + seq.incrementAndGet();
|
||||
}
|
||||
|
||||
private boolean isTracked(Path worktreeRoot, String rel) {
|
||||
return exitCode("git", "-C", worktreeRoot.toString(), "ls-files", "--error-unmatch", rel) == 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Run a command and return its stdout. Non-zero exit → {@link WorktreeException} with both
|
||||
* stdout and stderr (merged by redirectErrorStream).
|
||||
*/
|
||||
private String exec(String... command) {
|
||||
String out;
|
||||
int code;
|
||||
Process p;
|
||||
try {
|
||||
p = new ProcessBuilder(command).redirectErrorStream(true).start();
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("failed to start " + command[0] + ": " + e.getMessage(), e);
|
||||
}
|
||||
try (BufferedReader r = new BufferedReader(new InputStreamReader(p.getInputStream(), StandardCharsets.UTF_8))) {
|
||||
out = r.lines().collect(Collectors.joining("\n"));
|
||||
} catch (IOException e) {
|
||||
p.destroyForcibly();
|
||||
throw new UncheckedIOException(e);
|
||||
}
|
||||
try {
|
||||
if (!p.waitFor(30, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
throw new WorktreeException("command timed out: " + String.join(" ", command) + "\n" + out);
|
||||
}
|
||||
code = p.exitValue();
|
||||
} catch (InterruptedException e) {
|
||||
p.destroyForcibly();
|
||||
Thread.currentThread().interrupt();
|
||||
throw new WorktreeException("interrupted waiting for command: " + String.join(" ", command), e);
|
||||
}
|
||||
if (code != 0) {
|
||||
throw new WorktreeException("exit " + code + " for: " + String.join(" ", command)
|
||||
+ (out.isBlank() ? "" : "\n" + out));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
private int exitCode(String... command) {
|
||||
Process p;
|
||||
try {
|
||||
p = new ProcessBuilder(command).redirectErrorStream(true).start();
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("failed to start " + command[0] + ": " + e.getMessage(), e);
|
||||
}
|
||||
try {
|
||||
if (!p.waitFor(30, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
throw new WorktreeException("command timed out: " + String.join(" ", command));
|
||||
}
|
||||
return p.exitValue();
|
||||
} catch (InterruptedException e) {
|
||||
p.destroyForcibly();
|
||||
Thread.currentThread().interrupt();
|
||||
throw new WorktreeException("interrupted waiting for command: " + String.join(" ", command), e);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,68 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
|
||||
/**
|
||||
* A bridge-owned worker session — the authoritative in-daemon record of a worker this
|
||||
* process spawned. Immutable; state transitions are performed by replacing the record in
|
||||
* {@link SessionManager}'s registry.
|
||||
*
|
||||
* @param paneId the host-unique opaque id (CB-519) — the registry key and the argument to
|
||||
* teardown. Despite the historical name this is the {@link
|
||||
* dev.ltms.bridged.peer.PeerHandle#id()}, a UUID, and is distinct from the
|
||||
* launcher-private herdr pane coordinate.
|
||||
* @param terminalId herdr terminal handle — the {@code target} for send/read/status
|
||||
* @param profile the profile name that spawned this session — WHICH BACKEND
|
||||
* @param role the contract this member runs under — WHAT IT IS FOR. Independent of
|
||||
* {@code profile}: a reviewer may run on the same profile as the dev
|
||||
* whose diff it reads. {@code null} only for a session recorded before
|
||||
* the role was known
|
||||
* @param cwd the resolved working directory the worker started in
|
||||
* @param ownerTerminal the caller that requested this worker ({@code null} = daemon/anon)
|
||||
* @param spawnedAtNanos {@link System#nanoTime()} when the session was registered
|
||||
* @param lastActivityAtNanos {@link System#nanoTime()} of the most recent lifecycle event
|
||||
* @param turnCount number of delegated turns that have been delivered to this session
|
||||
* @param state current lifecycle state in the one-shot FSM
|
||||
*/
|
||||
public record MemberSession(
|
||||
String paneId,
|
||||
String terminalId,
|
||||
String profile,
|
||||
MemberRole role,
|
||||
String cwd,
|
||||
String ownerTerminal,
|
||||
long spawnedAtNanos,
|
||||
long lastActivityAtNanos,
|
||||
int turnCount,
|
||||
State state,
|
||||
String worktree,
|
||||
String branch) {
|
||||
|
||||
/** One-shot worker lifecycle states. */
|
||||
public enum State {
|
||||
SPAWNING,
|
||||
READY,
|
||||
BUSY,
|
||||
DONE,
|
||||
FAILED,
|
||||
RELEASED
|
||||
}
|
||||
|
||||
/** Return a copy of this session in {@code state}. */
|
||||
public MemberSession withState(State state) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch);
|
||||
}
|
||||
|
||||
/** Return a copy with {@code lastActivityAtNanos} updated to {@code nowNanos}. */
|
||||
public MemberSession withActivity(long nowNanos) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount, state, worktree, branch);
|
||||
}
|
||||
|
||||
/** Return a copy with the turn count incremented and activity timestamped at {@code nowNanos}. */
|
||||
public MemberSession bumpTurn(long nowNanos) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount + 1, state, worktree, branch);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,640 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import dev.ltms.bridged.auth.MemberLifecycle;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.inject.TurnListener;
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.security.SecureRandom;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
* Authoritative in-daemon registry of the worker sessions this {@code bridged} process spawned.
|
||||
* Delegates spawn/teardown to a {@link PeerLauncher} (which performs subscription-guarded env
|
||||
* setup and process/materialization) and adds lifecycle tracking, ownership, and deterministic
|
||||
* teardown on top.
|
||||
*
|
||||
* <p>The state machine is intentionally one-shot / no-reuse: every acquired worker is fresh,
|
||||
* and a finished or released worker is torn down, never pooled or reused.
|
||||
*
|
||||
* <p>The manager implements {@link TurnListener} so the injector's turn boundaries drive
|
||||
* {@code READY → BUSY → DONE} (or {@code FAILED}). It exposes a {@link MemberPresence} view via
|
||||
* {@link #asPresence()}: any MCP contact from a worker marks it present and simultaneously
|
||||
* transitions the session {@code SPAWNING → READY}.
|
||||
*/
|
||||
public final class SessionManager implements TurnListener {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(SessionManager.class);
|
||||
|
||||
private final PeerLauncher launcher;
|
||||
private final Worktrees worktrees;
|
||||
private final ConcurrentHashMap<String /*paneId*/, MemberSession> registry = new ConcurrentHashMap<>();
|
||||
private final MemberPresence presence;
|
||||
private final SecureRandom nonceRandom = new SecureRandom();
|
||||
private final AtomicLong nonceSeq = new AtomicLong();
|
||||
private final LongSupplier nowNanos;
|
||||
private final int contextCap;
|
||||
private final boolean clearAfterTurn;
|
||||
private volatile MemberLifecycle memberLifecycle = MemberLifecycle.NONE;
|
||||
|
||||
/** CB-520: notified with a terminalId on every acquire; no-op until wired. */
|
||||
private final List<Consumer<String>> acquireListeners = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
/** CB-516: notified with a terminalId on every release; no-op until wired. */
|
||||
private final List<Consumer<String>> releaseListeners = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
|
||||
/** Backward-compatible constructor: shared-tree sessions, production git seam. */
|
||||
public SessionManager(PeerLauncher launcher) {
|
||||
this(launcher, new GitWorktrees(), System::nanoTime, 0, false);
|
||||
}
|
||||
|
||||
/** Backward-compatible constructor with an injectable worktree seam. */
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees) {
|
||||
this(launcher, worktrees, System::nanoTime, 0, false);
|
||||
}
|
||||
|
||||
/** Test constructor with an injectable clock. */
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos) {
|
||||
this(launcher, worktrees, nowNanos, 0, false);
|
||||
}
|
||||
|
||||
/** Production constructor with a configured context turn cap. */
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, int contextCap) {
|
||||
this(launcher, worktrees, System::nanoTime, contextCap, false);
|
||||
}
|
||||
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap) {
|
||||
this(launcher, worktrees, nowNanos, contextCap, false);
|
||||
}
|
||||
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap, boolean clearAfterTurn) {
|
||||
this.launcher = launcher;
|
||||
this.worktrees = worktrees;
|
||||
this.presence = new PresenceBridge(this);
|
||||
this.nowNanos = nowNanos;
|
||||
this.contextCap = contextCap;
|
||||
this.clearAfterTurn = clearAfterTurn;
|
||||
}
|
||||
|
||||
/**
|
||||
* The single {@link MemberPresence} view of this manager: it records availability and forwards
|
||||
* the signal to the {@code SPAWNING → READY} transition. Pass this to the {@code Injector} and
|
||||
* {@code BridgeMcp} where they previously accepted a plain {@link MemberPresence}. The same
|
||||
* instance is returned every call — presence is shared state, so a fresh bridge per call would
|
||||
* fragment the {@code present} set and lose signals across callers.
|
||||
*/
|
||||
public MemberPresence asPresence() {
|
||||
return presence;
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a worker and register it as {@link MemberSession.State#SPAWNING}. The caller's
|
||||
* identity is recorded as {@code ownerTerminal} ({@code null} for daemon/anon callers).
|
||||
*/
|
||||
public MemberSession acquire(String profile, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal) {
|
||||
return acquire(profile, requestedCwd, callerCwd, ownerTerminal, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a member, optionally inside a fresh git worktree, defaulting the role to
|
||||
* {@link MemberRole#DEV}.
|
||||
*
|
||||
* <p>{@code DEV} is the right default because it is exactly what the old "worker" meant: an
|
||||
* unqualified spawn is a unit of implementation work. An architect or a reviewer is always
|
||||
* asked for on purpose, so neither is ever what a caller silently gets.
|
||||
*/
|
||||
public MemberSession acquire(String profile, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt) {
|
||||
return acquire(profile, MemberRole.DEV, requestedCwd, callerCwd, ownerTerminal, wt);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a member, optionally inside a fresh git worktree. When {@code wt} is non-null the
|
||||
* worktree is provisioned, parity-overlaid, and its path becomes the member's cwd. On any
|
||||
* failure before registration the worktree is removed so no dangling checkout is left.
|
||||
*
|
||||
* @param profile which backend to run on — a {@code profiles:} key
|
||||
* @param role which contract the member runs under; never {@code null}
|
||||
*/
|
||||
public MemberSession acquire(String profile, MemberRole role, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt) {
|
||||
MemberRole memberRole = (role == null) ? MemberRole.DEV : role;
|
||||
if (wt == null) {
|
||||
// CB-557: the role must ride on the SpawnRequest, not stay a local. The launcher needs it
|
||||
// to pick the profile out of that role's pool and to label the tab; a role kept only on
|
||||
// the MemberSession is recorded after the spawn it was supposed to steer.
|
||||
SpawnRequest req = new SpawnRequest(profile, requestedCwd, callerCwd, null, null, memberRole);
|
||||
PeerHandle handle;
|
||||
try {
|
||||
handle = launcher.spawn(req);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("spawn failed for profile={} role={}: {}", profile, memberRole, e.getMessage());
|
||||
throw e;
|
||||
}
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, requestedCwd, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
memberRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
now,
|
||||
0,
|
||||
MemberSession.State.SPAWNING,
|
||||
null,
|
||||
null);
|
||||
registry.put(handle.id(), session);
|
||||
memberLifecycle.acquired(session.role(), session.profile(), session.terminalId());
|
||||
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
return acquireWithWorktree(profile, memberRole, requestedCwd, callerCwd, ownerTerminal, wt);
|
||||
}
|
||||
|
||||
/** Tear a worker down by pane id and remove it from the registry. Idempotent. */
|
||||
public void release(String paneId) {
|
||||
release(paneId, ReleaseCause.COMPLETED);
|
||||
}
|
||||
|
||||
/**
|
||||
* Core teardown: always stops the worker pane and deregisters the session; whether the worker's
|
||||
* git worktree is also removed depends on {@code cause}.
|
||||
*
|
||||
* <p>CB-544: these are two concerns that used to be fused. Stopping the pane is correct on every
|
||||
* teardown — the worker process must end. Removing the worktree is a destructive act that is only
|
||||
* correct for a deliberately-finished teardown (an explicit stop of a completed session, the
|
||||
* reaper releasing a genuinely idle one, or a context-capped session). A shutdown drain
|
||||
* must stop panes but preserve worktrees: a worker's uncommitted work exists in exactly one
|
||||
* place — its worktree — so deleting it while the daemon simply goes down is silent data loss,
|
||||
* with no copy and no error. Do NOT fuse these back together; the cost of an orphaned worktree
|
||||
* is a logged path an operator can reclaim, the cost of a deleted one is unrecoverable work.
|
||||
*/
|
||||
private void release(String paneId, ReleaseCause cause) {
|
||||
MemberSession removed = registry.remove(paneId);
|
||||
boolean preserveWorktree = cause == ReleaseCause.SHUTDOWN;
|
||||
if (removed != null) {
|
||||
memberLifecycle.released(removed.terminalId());
|
||||
log.debug("releasing session pane={} terminal={} state={} cause={}",
|
||||
removed.paneId(), removed.terminalId(), removed.state(), cause);
|
||||
if (preserveWorktree && removed.worktree() != null) {
|
||||
logPreservedForShutdown(removed);
|
||||
} else if (removed.worktree() != null && worktrees.hasUncommitted(removed.worktree())) {
|
||||
// CB-576: a release that would otherwise remove the worktree finds it holding
|
||||
// uncommitted work the bridge cannot see. A worker that ends a turn without
|
||||
// committing (normally because it stopped to ask a question or refused the turn)
|
||||
// has its only copy of that work in the worktree. Remove would --force-delete it,
|
||||
// so preserve the directory and tell an operator where to find it.
|
||||
preserveWorktree = true;
|
||||
log.warn("release {} preserves dirty worktree {} for pane={} terminal={}: "
|
||||
+ "the worktree holds uncommitted changes that --force remove would destroy",
|
||||
cause, removed.worktree(), removed.paneId(), removed.terminalId());
|
||||
}
|
||||
// CB-516: a send still waiting on this worker can never be answered now. Tell the
|
||||
// listener BEFORE the pane is torn down, so a blocked caller fails fast with a real
|
||||
// reason instead of sitting on a rendezvous nothing will ever resolve.
|
||||
notifyReleased(removed.terminalId());
|
||||
}
|
||||
launcher.stop(paneId);
|
||||
if (removed != null && !preserveWorktree && removed.worktree() != null) {
|
||||
worktrees.remove(worktrees.repoRoot(removed.cwd()), removed.worktree());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Why a session is being released — governs whether its worktree is preserved or removed.
|
||||
* Worktree removal is reserved for the one case that is genuinely finished; everything else
|
||||
* must keep the worker's only copy of its work.
|
||||
*/
|
||||
public enum ReleaseCause {
|
||||
/** Deliberate teardown of a finished session. Stops the pane and removes the worktree. */
|
||||
COMPLETED,
|
||||
/** Daemon shutdown drain. Stops the pane but PRESERVES the worktree. */
|
||||
SHUTDOWN
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-544 shutdown drain log for a worktree we deliberately kept. A session still {@code BUSY}
|
||||
* when the drain timeout expired was abandoned mid-turn — that work may be uncommitted and is
|
||||
* the only copy — so the message is loud and points at the path an operator needs to reclaim.
|
||||
*/
|
||||
private void logPreservedForShutdown(MemberSession session) {
|
||||
if (session.state() == MemberSession.State.BUSY) {
|
||||
log.warn("shutdown drain abandoned BUSY session pane={} terminal={} mid-turn; "
|
||||
+ "worktree preserved at {}", session.paneId(), session.terminalId(),
|
||||
session.worktree());
|
||||
} else {
|
||||
log.info("shutdown drain preserved worktree at {} for pane={}",
|
||||
session.worktree(), session.paneId());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Register a callback invoked with a session's {@code terminalId} whenever it is acquired
|
||||
* (CB-520). This is the hook that lets the reply inbox {@code own} a target's queue.
|
||||
*/
|
||||
public void onAcquire(Consumer<String> listener) {
|
||||
if (listener != null) {
|
||||
acquireListeners.add(listener);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Register a callback invoked with a session's {@code terminalId} whenever it is released
|
||||
* (CB-516). Every teardown path funnels through {@link #release}, so one hook covers the REST
|
||||
* and MCP stop tools, the idle-TTL reaper, and shutdown drain alike.
|
||||
*
|
||||
* <p>Added rather than injected because {@code MessageService} — one intended listener — is
|
||||
* constructed after this manager (it needs the injector and rendezvous, which need the session
|
||||
* presence view this manager exposes). Wiring it at construction would require breaking that
|
||||
* cycle for one callback.
|
||||
*/
|
||||
public void onRelease(Consumer<String> listener) {
|
||||
if (listener != null) {
|
||||
releaseListeners.add(listener);
|
||||
}
|
||||
}
|
||||
|
||||
/** Inject the optional member-slot lifecycle after construction without changing constructors. */
|
||||
public void setMemberLifecycle(MemberLifecycle memberLifecycle) {
|
||||
this.memberLifecycle = memberLifecycle == null ? MemberLifecycle.NONE : memberLifecycle;
|
||||
}
|
||||
|
||||
/** A listener failure must never prevent the acquisition it is reacting to. */
|
||||
private void notifyAcquired(String terminalId) {
|
||||
if (terminalId == null) {
|
||||
return;
|
||||
}
|
||||
for (Consumer<String> listener : acquireListeners) {
|
||||
try {
|
||||
listener.accept(terminalId);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("acquire listener failed for terminal {}: {}", terminalId, e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** A listener failure must never prevent the teardown it is reacting to. */
|
||||
private void notifyReleased(String terminalId) {
|
||||
if (terminalId == null) {
|
||||
return;
|
||||
}
|
||||
for (Consumer<String> listener : releaseListeners) {
|
||||
try {
|
||||
listener.accept(terminalId);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("release listener failed for terminal {}: {}", terminalId, e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private MemberSession acquireWithWorktree(String profile, MemberRole memberRole, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt) {
|
||||
String preResolvedProfile = (profile == null || profile.isBlank())
|
||||
? launcher.defaultProfile() : profile;
|
||||
// CB-507: resolve through the launcher's CB-112 chain (requested → profile cwd → caller →
|
||||
// daemon cwd → "."), never the raw args. A plain REST spawn supplies neither a requested
|
||||
// nor a caller cwd, so taking the first non-blank of those two yielded null and put
|
||||
// `git -C null` on the command line — an NPE out of ProcessBuilder, surfacing as HTTP 500.
|
||||
// The non-worktree path always used this chain; only this branch was missed.
|
||||
String repoRoot = worktrees.repoRoot(
|
||||
launcher.effectiveCwd(new SpawnRequest(preResolvedProfile, requestedCwd, callerCwd)));
|
||||
String branch = "worker/" + slug(wt.ticketSlug()) + "-" + nonce();
|
||||
String path = null;
|
||||
PeerHandle handle;
|
||||
try {
|
||||
path = worktrees.add(repoRoot, branch, wt.baseRef());
|
||||
worktrees.overlayParity(repoRoot, path, launcher.parityOverlay(preResolvedProfile));
|
||||
handle = launcher.spawn(new SpawnRequest(profile, path, callerCwd, null, null, memberRole));
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("spawn failed for profile={} role={} branch={} path={}: {}",
|
||||
preResolvedProfile, memberRole, branch, path, e.getMessage());
|
||||
if (path != null) {
|
||||
try {
|
||||
worktrees.remove(repoRoot, path);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to clean up worktree {} after spawn error: {}", path, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, path, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
memberRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
now,
|
||||
0,
|
||||
MemberSession.State.SPAWNING,
|
||||
path,
|
||||
branch);
|
||||
registry.put(handle.id(), session);
|
||||
memberLifecycle.acquired(session.role(), session.profile(), session.terminalId());
|
||||
log.debug("acquired worktree session id={} terminal={} profile={} branch={} path={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.branch(), session.worktree());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
|
||||
private String slug(String raw) {
|
||||
return raw == null ? "ticket" : raw.toLowerCase().replaceAll("[^a-z0-9]+", "-").replaceAll("^-+|-+$", "");
|
||||
}
|
||||
|
||||
private String nonce() {
|
||||
return String.format("%06x", nonceRandom.nextInt(1 << 24)) + "-" + nonceSeq.incrementAndGet();
|
||||
}
|
||||
|
||||
/**
|
||||
* The profile to record for a session. A launcher that performed dynamic selection tells us
|
||||
* the actual profile via {@link PeerHandle#profile()}; otherwise fall back to what the caller
|
||||
* requested (or the launcher's default for a no-profile spawn).
|
||||
*/
|
||||
private String resolveProfile(PeerHandle handle, String requestedProfile) {
|
||||
String fromHandle = handle.profile();
|
||||
if (fromHandle != null && !fromHandle.isBlank()) {
|
||||
return fromHandle;
|
||||
}
|
||||
if (requestedProfile != null && !requestedProfile.isBlank()) {
|
||||
return requestedProfile;
|
||||
}
|
||||
return launcher.defaultProfile();
|
||||
}
|
||||
|
||||
/** The session for {@code paneId}, if it is still registered and not released. */
|
||||
public Optional<MemberSession> get(String paneId) {
|
||||
return Optional.ofNullable(registry.get(paneId));
|
||||
}
|
||||
|
||||
/** Bridge-owned roster: all registered sessions (acquired minus released). */
|
||||
public List<MemberSession> roster() {
|
||||
return List.copyOf(registry.values());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-304 merged roster+live view. The registry is authoritative for worktree, branch,
|
||||
* profile, owner, and state; the optional live agent supplies the herdr-reported status.
|
||||
*/
|
||||
public static Map<String, Object> rosterView(MemberSession session, Agent live) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("sessionId", session.terminalId());
|
||||
m.put("paneId", session.paneId());
|
||||
// Both axes, always: profile says which backend this member runs on, role says what it is
|
||||
// for. A lead reading the roster needs both — two rows may share a profile and still be
|
||||
// allowed to do entirely different things.
|
||||
m.put("profile", session.profile());
|
||||
m.put("role", session.role() == null ? "dev" : session.role().wireName());
|
||||
m.put("state", session.state().name().toLowerCase());
|
||||
if (session.worktree() != null) {
|
||||
m.put("worktree", session.worktree());
|
||||
}
|
||||
if (session.branch() != null) {
|
||||
m.put("branch", session.branch());
|
||||
}
|
||||
if (session.ownerTerminal() != null) {
|
||||
m.put("owner", session.ownerTerminal());
|
||||
}
|
||||
m.put("liveStatus", live == null ? "unknown" : live.status().name().toLowerCase());
|
||||
return m;
|
||||
}
|
||||
|
||||
/** Lifecycle hook: worker became available on the bridge MCP. */
|
||||
void onReady(String terminalId) {
|
||||
transitionByTerminal(terminalId, MemberSession.State.SPAWNING, MemberSession.State.READY);
|
||||
}
|
||||
|
||||
/**
|
||||
* Lifecycle hook: a message was delivered into the worker — it is now busy on a turn.
|
||||
* The turn count is bumped and the activity timestamp is refreshed. A {@code DONE} session
|
||||
* can be re-delivered for multi-turn reuse until it is released.
|
||||
*/
|
||||
@Override
|
||||
public void onDelivered(String target, TurnToken token) {
|
||||
MemberSession current = findByTerminal(target);
|
||||
if (current == null) return;
|
||||
if (current.state() != MemberSession.State.READY && current.state() != MemberSession.State.DONE) {
|
||||
return;
|
||||
}
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession updated = current.withState(MemberSession.State.BUSY).bumpTurn(now);
|
||||
if (replace(current, updated)) {
|
||||
log.debug("session transitioned terminal={} pane={} {} -> BUSY turn={}",
|
||||
target, current.paneId(), current.state(), updated.turnCount());
|
||||
}
|
||||
}
|
||||
|
||||
/** Lifecycle hook: the worker's delegated turn completed successfully. */
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
completeTurn(target, false);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
if (!clearAfterTurn) return false;
|
||||
MemberSession current = findByTerminal(target);
|
||||
return current != null && current.state() == MemberSession.State.BUSY
|
||||
&& (contextCap <= 0 || current.turnCount() < contextCap);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
return completeTurn(target, true);
|
||||
}
|
||||
|
||||
private boolean completeTurn(String target, boolean startContextReset) {
|
||||
MemberSession current = findByTerminal(target);
|
||||
if (current == null || current.state() != MemberSession.State.BUSY) return false;
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession updated = current.withState(MemberSession.State.DONE).withActivity(now);
|
||||
if (replace(current, updated)) {
|
||||
log.debug("session transitioned terminal={} pane={} BUSY -> DONE turn={}",
|
||||
target, current.paneId(), updated.turnCount());
|
||||
}
|
||||
if (contextCap > 0 && updated.turnCount() >= contextCap) {
|
||||
release(current.paneId());
|
||||
return false;
|
||||
}
|
||||
if (!startContextReset || !clearAfterTurn) return false;
|
||||
try {
|
||||
return launcher.clearContext(current.paneId());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("context reset failed for terminal={} pane={}; continuing without reset: {}",
|
||||
target, current.paneId(), e.getMessage());
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/** Lifecycle hook: the worker's delegated turn failed. */
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
onFailed(target);
|
||||
}
|
||||
|
||||
/** Lifecycle hook: the worker vanished or was dropped mid-life. */
|
||||
void onFailed(String target) {
|
||||
MemberSession current = findByTerminal(target);
|
||||
if (current == null) return;
|
||||
if (current.state() == MemberSession.State.RELEASED) return;
|
||||
MemberSession.State priorState = current.state();
|
||||
if (replace(current, current.withState(MemberSession.State.FAILED))) {
|
||||
log.warn("member terminal={} pane={} can no longer be delegated to: its turn never resolved "
|
||||
+ "(was {} when it failed)", target, current.paneId(), priorState);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort reap of sessions that have been idle longer than {@code idleTtlNanos}. Only
|
||||
* {@code READY} and {@code DONE} sessions are eligible — never a {@code SPAWNING} or
|
||||
* {@code BUSY} worker. Returns the number of sessions released.
|
||||
*/
|
||||
int reapIdle(long idleTtlNanos) {
|
||||
long now = nowNanos.getAsLong();
|
||||
int reaped = 0;
|
||||
for (MemberSession s : roster()) {
|
||||
if (s.state() != MemberSession.State.READY && s.state() != MemberSession.State.DONE) {
|
||||
continue;
|
||||
}
|
||||
long idleNanos = now - s.lastActivityAtNanos();
|
||||
if (idleNanos > idleTtlNanos) {
|
||||
log.debug("reaping idle session terminal={} pane={}: idle {}s exceeds the {}s ttl",
|
||||
s.terminalId(), s.paneId(), TimeUnit.NANOSECONDS.toSeconds(idleNanos),
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos));
|
||||
release(s.paneId());
|
||||
reaped++;
|
||||
}
|
||||
}
|
||||
return reaped;
|
||||
}
|
||||
|
||||
/**
|
||||
* Gracefully drain all registered sessions on daemon shutdown. For each session that is
|
||||
* {@code BUSY}, poll up to {@code timeoutNanos} for it to leave {@code BUSY}, then release it
|
||||
* regardless. Non-busy sessions are released immediately. A failure releasing one session is
|
||||
* logged and does not abort the rest.
|
||||
*
|
||||
* <p>CB-544: this is a {@link ReleaseCause#SHUTDOWN} release — the worker's pane is stopped
|
||||
* (the process must end) but its worktree is preserved and its path logged. Shutdown is never
|
||||
* a reason to delete a worker's only copy of its uncommitted work. A session still {@code BUSY}
|
||||
* when the timeout expired is abandoned mid-turn and logged loudly so an operator can find its
|
||||
* kept worktree.
|
||||
*/
|
||||
void drainAll(long timeoutNanos) {
|
||||
long deadline = System.nanoTime() + timeoutNanos;
|
||||
for (MemberSession s : roster()) {
|
||||
try {
|
||||
if (s.state() == MemberSession.State.BUSY) {
|
||||
while (System.nanoTime() < deadline) {
|
||||
MemberSession current = registry.get(s.paneId());
|
||||
if (current == null || current.state() != MemberSession.State.BUSY) {
|
||||
break;
|
||||
}
|
||||
try {
|
||||
long remaining = deadline - System.nanoTime();
|
||||
Thread.sleep(Math.min(TimeUnit.NANOSECONDS.toMillis(remaining), 50));
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
release(s.paneId(), ReleaseCause.SHUTDOWN);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("drain failed for pane={}; continuing with remaining sessions", s.paneId(), e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Close this manager by draining all sessions. The timeout comes from configuration when set,
|
||||
* otherwise a sensible default.
|
||||
*/
|
||||
public void close(Integer drainTimeoutSeconds) {
|
||||
int seconds = (drainTimeoutSeconds != null && drainTimeoutSeconds > 0) ? drainTimeoutSeconds : 5;
|
||||
drainAll(TimeUnit.SECONDS.toNanos(seconds));
|
||||
}
|
||||
|
||||
/** Number of sessions currently registered. */
|
||||
public int size() {
|
||||
return registry.size();
|
||||
}
|
||||
|
||||
/**
|
||||
* The registered session owning {@code terminalId}, or {@code null} if none does.
|
||||
*
|
||||
* <p>A null {@code terminalId} is a normal input, not a caller bug: every lifecycle hook here is
|
||||
* fed from the MCP transport, where the <em>primary</em> resolves to a {@link
|
||||
* dev.ltms.bridged.auth.Principal} with no terminal. {@code BridgeMcp} documents that contact as
|
||||
* a no-op, and {@link dev.ltms.bridged.inject.MemberPresence#markPresent} honours it — but
|
||||
* {@code PresenceBridge} then forwards the same null here. Matching on a null id can never
|
||||
* succeed anyway (a registered session always has a terminal), so answer "no match" rather than
|
||||
* throwing: an NPE on this path takes down an unrelated tool call for the primary.
|
||||
*/
|
||||
private MemberSession findByTerminal(String terminalId) {
|
||||
if (terminalId == null) return null;
|
||||
for (MemberSession s : registry.values()) {
|
||||
if (terminalId.equals(s.terminalId())) return s;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private void transitionByTerminal(String terminalId, MemberSession.State from,
|
||||
MemberSession.State to) {
|
||||
MemberSession current = findByTerminal(terminalId);
|
||||
if (current == null || current.state() != from) return;
|
||||
long now = nowNanos.getAsLong();
|
||||
if (replace(current, current.withState(to).withActivity(now))) {
|
||||
log.debug("session transitioned terminal={} pane={} {} -> {}",
|
||||
terminalId, current.paneId(), from, to);
|
||||
}
|
||||
}
|
||||
|
||||
private boolean replace(MemberSession expected, MemberSession updated) {
|
||||
return registry.replace(expected.paneId(), expected, updated);
|
||||
}
|
||||
|
||||
/** MemberPresence bridge that also drives the manager's READY transition. */
|
||||
private static final class PresenceBridge extends MemberPresence {
|
||||
private final SessionManager sessions;
|
||||
|
||||
PresenceBridge(SessionManager sessions) {
|
||||
this.sessions = sessions;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void markPresent(String terminal) {
|
||||
if (terminal == null || terminal.isBlank()) {
|
||||
return; // the primary's contact carries no worker terminal — not a readiness signal
|
||||
}
|
||||
super.markPresent(terminal);
|
||||
sessions.onReady(terminal);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,70 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Periodic virtual-thread reaper that tears down {@code READY}/{@code DONE} sessions which have
|
||||
* exceeded their idle TTL. Modeled on {@link dev.ltms.bridged.inject.StatusPoller}: a single
|
||||
* virtual-thread loop, idempotent start/stop, and no {@code ScheduledExecutorService}.
|
||||
*/
|
||||
public final class SessionReaper {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(SessionReaper.class);
|
||||
private static final long DEFAULT_INTERVAL_MILLIS = 5000;
|
||||
|
||||
private final SessionManager sessions;
|
||||
private final long idleTtlNanos;
|
||||
private final long intervalMillis;
|
||||
private volatile boolean running;
|
||||
private Thread thread;
|
||||
|
||||
/** Construct a reaper with the default 5-second polling interval. */
|
||||
public SessionReaper(SessionManager sessions, long idleTtlSeconds) {
|
||||
this(sessions, idleTtlSeconds, DEFAULT_INTERVAL_MILLIS);
|
||||
}
|
||||
|
||||
/** Construct a reaper with an explicit polling interval (useful for tests). */
|
||||
public SessionReaper(SessionManager sessions, long idleTtlSeconds, long intervalMillis) {
|
||||
this.sessions = sessions;
|
||||
this.idleTtlNanos = TimeUnit.SECONDS.toNanos(idleTtlSeconds);
|
||||
this.intervalMillis = intervalMillis;
|
||||
}
|
||||
|
||||
/** Start the reaper loop on a virtual thread. Idempotent. */
|
||||
public synchronized void start() {
|
||||
if (running) return;
|
||||
running = true;
|
||||
thread = Thread.ofVirtual().name("session-reaper").start(this::loop);
|
||||
log.info("session reaper started (idle ttl {}s, interval {}ms)",
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos), intervalMillis);
|
||||
}
|
||||
|
||||
private void loop() {
|
||||
while (running) {
|
||||
try {
|
||||
sessions.reapIdle(idleTtlNanos);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("session reaper iteration failed; continuing", e);
|
||||
}
|
||||
sleep();
|
||||
}
|
||||
}
|
||||
|
||||
private void sleep() {
|
||||
try {
|
||||
Thread.sleep(intervalMillis);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
running = false;
|
||||
}
|
||||
}
|
||||
|
||||
/** Stop the reaper loop. Idempotent. */
|
||||
public synchronized void stop() {
|
||||
running = false;
|
||||
if (thread != null) thread.interrupt();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
/** Non-zero exit or I/O failure from a git worktree operation. */
|
||||
public final class WorktreeException extends RuntimeException {
|
||||
public WorktreeException(String message) {
|
||||
super(message);
|
||||
}
|
||||
|
||||
public WorktreeException(String message, Throwable cause) {
|
||||
super(message, cause);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,6 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
/** Ask {@link SessionManager#acquire} to provision an isolated worktree. null ⇒ run in the shared primary tree. */
|
||||
public record WorktreeRequest(String ticketSlug, String baseRef) {
|
||||
// ticketSlug seeds the branch name; baseRef null/blank ⇒ current HEAD of the repo.
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** Seam between {@link SessionManager} and git worktree operations. Tests use a recording fake. */
|
||||
public interface Worktrees {
|
||||
/** git -C <repoRoot> worktree add <path> -b <branch> <baseRef|HEAD>. Returns the worktree path. */
|
||||
String add(String repoRoot, String branch, String baseRef);
|
||||
|
||||
/** git -C <repoRoot> worktree remove --force <path>. Idempotent (already-gone tolerated). */
|
||||
void remove(String repoRoot, String worktreePath);
|
||||
|
||||
/**
|
||||
* True when the worktree holds uncommitted changes the bridge cannot see: tracked
|
||||
* modifications, staged files, or untracked files. {@code git status --porcelain} is the
|
||||
* test; an empty result means clean. Callers use this to decide whether removing the
|
||||
* worktree would silently destroy a worker's only copy of its work.
|
||||
*
|
||||
* <p>An already-gone worktree is reported as clean (no throw), matching {@link #remove}'s
|
||||
* idempotent contract: a path that does not exist holds no work to lose, and must not break
|
||||
* a teardown that is mid-way through stopping the pane.
|
||||
*/
|
||||
boolean hasUncommitted(String worktreePath);
|
||||
|
||||
/** Copy each existing overlay path repoRoot→worktree; mark tracked ones --skip-worktree. */
|
||||
void overlayParity(String repoRoot, String worktreePath, List<String> overlay);
|
||||
|
||||
/** git -C <cwd> rev-parse --show-toplevel — the repo root that owns cwd. */
|
||||
String repoRoot(String cwd);
|
||||
}
|
||||
@@ -1,206 +0,0 @@
|
||||
package dev.ltms.bridged.worker;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.Tab;
|
||||
import dev.ltms.bridged.herdr.Workspace;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.security.SecureRandom;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Function;
|
||||
|
||||
/**
|
||||
* Spawns and lists worker sessions — the safe path from a delegation request to a
|
||||
* running off-subscription Claude.
|
||||
*
|
||||
* <p>The spawn sequence encodes the subscription boundary: build the worker env with
|
||||
* {@code ANTHROPIC_BASE_URL}, assert that host is on the allowlist <em>before</em>
|
||||
* touching herdr, and only then {@code agent.start}. A worker's base_url lives in the
|
||||
* env map handed to herdr and nowhere else; {@code bridged}'s own environment is never
|
||||
* mutated.
|
||||
*
|
||||
* <p>Placement: in the default {@code tab} policy a worker lands in its own tab inside a
|
||||
* dedicated worker space (found-or-created once, then shared), so workers never split or
|
||||
* clutter the user's real work spaces. Teardown removes the worker's pane <em>and</em> its
|
||||
* now-empty tab, tolerating an already-gone worker so a repeated DELETE is harmless.
|
||||
*/
|
||||
public final class WorkerService {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(WorkerService.class);
|
||||
|
||||
/** herdr rejects a duplicate agent {@code name}; we retry a bumped name this many times. */
|
||||
private static final int NAME_RETRIES = 8;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final WorkspaceControl spaces;
|
||||
private final SubscriptionGuard guard;
|
||||
private final BridgedConfig.Worker cfg;
|
||||
private final Function<String, String> env; // host env lookup (injectable for tests)
|
||||
private final AtomicLong nameSeq = new AtomicLong(); // per-worker counter (also the tab #)
|
||||
// Per-process token mixed into each worker name so a fresh process (nameSeq back at 0)
|
||||
// cannot collide with same-profile workers that outlived a restart. See startUniquelyNamed.
|
||||
private final String nameNonce = String.format("%06x", new SecureRandom().nextInt(1 << 24));
|
||||
|
||||
public WorkerService(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
BridgedConfig.Worker cfg, Function<String, String> env) {
|
||||
this.agents = agents;
|
||||
this.spaces = spaces;
|
||||
this.guard = guard;
|
||||
this.cfg = cfg;
|
||||
this.env = env;
|
||||
}
|
||||
|
||||
/** Spawn a worker for the configured profile. Guard runs before any herdr call. */
|
||||
public Agent spawn() {
|
||||
String baseUrl = cfg.baseUrl();
|
||||
guard.assertWorker(baseUrl); // hard stop before we spawn anything
|
||||
|
||||
Map<String, String> workerEnv = new LinkedHashMap<>();
|
||||
workerEnv.put("ANTHROPIC_BASE_URL", baseUrl);
|
||||
putIfPresent(workerEnv, "ANTHROPIC_MODEL", cfg.model());
|
||||
putIfPresent(workerEnv, "CLAUDE_CONFIG_DIR", cfg.configDir());
|
||||
String token = env.apply(cfg.tokenEnv());
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", token);
|
||||
|
||||
return cfg.tabPlacement() ? spawnInTab(workerEnv) : spawnAsPane(workerEnv);
|
||||
}
|
||||
|
||||
/** Dedicated worker space → own tab → drop the placeholder shell so only the worker remains. */
|
||||
private Agent spawnInTab(Map<String, String> workerEnv) {
|
||||
Workspace space = spaces.ensureWorkspace(cfg.workspace());
|
||||
Tab.Created tab = spaces.createTab(space.workspaceId());
|
||||
log.info("spawning worker profile={} base_url={} space={} tab={}",
|
||||
cfg.profile(), cfg.baseUrl(), space.workspaceId(), tab.tab().tabId());
|
||||
|
||||
Started started;
|
||||
try {
|
||||
started = startUniquelyNamed(workerEnv, tab.tab().tabId());
|
||||
} catch (RuntimeException e) {
|
||||
// The worker never started — don't leave the tab we just created orphaned.
|
||||
// Best-effort cleanup; never let it mask the real spawn failure.
|
||||
try {
|
||||
spaces.closeTab(tab.tab().tabId());
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to close orphaned tab {} after spawn error: {}",
|
||||
tab.tab().tabId(), cleanup.getMessage());
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
|
||||
// The worker is LIVE now. The remaining steps are cosmetic (drop herdr's seed shell
|
||||
// so the tab holds only the worker; label the tab). They must not fail the spawn or
|
||||
// orphan the running worker — on error we log and still return it so the caller gets
|
||||
// its paneId and can tear it down.
|
||||
if (tab.rootPaneId() != null) {
|
||||
tidy("close seed pane " + tab.rootPaneId(), () -> agents.close(tab.rootPaneId()));
|
||||
} else {
|
||||
log.warn("tab {} had no seed pane in the create response; worker tab may hold an extra pane",
|
||||
tab.tab().tabId());
|
||||
}
|
||||
tidy("label tab " + tab.tab().tabId(),
|
||||
() -> spaces.renameTab(tab.tab().tabId(), cfg.renderTabLabel(started.seq())));
|
||||
log.info("worker started pane={} tab={} terminal={}",
|
||||
started.agent().paneId(), started.agent().tabId(), started.agent().terminalId());
|
||||
return started.agent();
|
||||
}
|
||||
|
||||
/** Run a best-effort post-start cleanup step, logging (not throwing) on failure. */
|
||||
private void tidy(String what, Runnable step) {
|
||||
try {
|
||||
step.run();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("post-start step failed ({}) — worker is running regardless: {}", what, e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/** Legacy placement: herdr splits the currently-focused tab. */
|
||||
private Agent spawnAsPane(Map<String, String> workerEnv) {
|
||||
log.info("spawning worker (pane placement) profile={} base_url={} argv={}",
|
||||
cfg.profile(), cfg.baseUrl(), cfg.argv());
|
||||
Agent worker = startUniquelyNamed(workerEnv, null).agent();
|
||||
log.info("worker started pane={} terminal={}", worker.paneId(), worker.terminalId());
|
||||
return worker;
|
||||
}
|
||||
|
||||
/** A started worker together with the sequence its unique name/label used. */
|
||||
private record Started(Agent agent, long seq) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Start the worker under a unique herdr agent name. herdr requires each running
|
||||
* agent's {@code name} to be distinct (a 2nd {@code name:"claude"} fails
|
||||
* {@code agent_name_taken}) — the exact case that makes multiple workers useful. The name
|
||||
* is {@code claude-<profile>-<nonce>-<seq>}: {@code seq} distinguishes workers within this
|
||||
* process, and the per-process {@code nonce} keeps a fresh process (whose {@code seq}
|
||||
* restarts at 0) from colliding with same-profile workers that outlived a restart. The
|
||||
* retry is a belt-and-braces backstop for the astronomically unlikely nonce+seq clash;
|
||||
* the name is a label only — herdr detects kind and status from terminal output, not it.
|
||||
*/
|
||||
private Started startUniquelyNamed(Map<String, String> workerEnv, String tabId) {
|
||||
HerdrException last = null;
|
||||
for (int attempt = 0; attempt < NAME_RETRIES; attempt++) {
|
||||
long seq = nameSeq.incrementAndGet();
|
||||
String name = "claude-" + cfg.profile() + "-" + nameNonce + "-" + seq;
|
||||
try {
|
||||
return new Started(agents.start(name, cfg.argv(), workerEnv, tabId), seq);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_name_taken".equals(e.code())) throw e;
|
||||
log.debug("worker name '{}' taken, retrying", name);
|
||||
last = e;
|
||||
}
|
||||
}
|
||||
throw last;
|
||||
}
|
||||
|
||||
/** All herdr-tracked agents — discovery for "what workers exist". */
|
||||
public List<Agent> list() {
|
||||
return agents.list();
|
||||
}
|
||||
|
||||
/**
|
||||
* Tear a worker down by pane id: close the pane, and close its tab <em>only</em> when the
|
||||
* worker is that tab's sole occupant. The single-pane check is what makes this safe
|
||||
* regardless of how the worker was placed (or a placement-config change across a restart):
|
||||
* a pane-placement worker sitting in one of the user's shared tabs has siblings, so its
|
||||
* tab is never closed — we only ever remove a tab we created to hold one worker.
|
||||
*
|
||||
* <p>Resolves the tab from the pane <em>before</em> closing it. An already-gone pane/tab
|
||||
* (repeated DELETE, crashed worker) is treated as success; any other failure propagates so
|
||||
* a genuinely failed teardown is not reported as done.
|
||||
*/
|
||||
public void stop(String paneId) {
|
||||
WorkspaceControl.PaneLocation loc = cfg.tabPlacement() ? spaces.locatePane(paneId) : null;
|
||||
try {
|
||||
agents.close(paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!isAlreadyGone(e)) throw e;
|
||||
log.debug("pane.close({}) ignored — already gone: {}", paneId, e.getMessage());
|
||||
}
|
||||
if (loc != null && loc.tabPaneCount() == 1) {
|
||||
spaces.closeTab(loc.tabId());
|
||||
} else if (loc != null) {
|
||||
log.debug("not closing tab {} — it holds {} panes (not a dedicated worker tab)",
|
||||
loc.tabId(), loc.tabPaneCount());
|
||||
}
|
||||
}
|
||||
|
||||
/** True when a herdr error means the target is already gone (safe to treat as done). */
|
||||
private static boolean isAlreadyGone(HerdrException e) {
|
||||
return e.code() != null && e.code().endsWith("_not_found");
|
||||
}
|
||||
|
||||
private static void putIfPresent(Map<String, String> m, String k, String v) {
|
||||
if (v != null && !v.isBlank()) {
|
||||
m.put(k, v);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,14 +1,52 @@
|
||||
<configuration>
|
||||
<!--
|
||||
CB-575: MCP SDK 2.0.0 has no public notification registration API. It registers only
|
||||
notifications/initialized and notifications/roots/list_changed, so other client notifications
|
||||
still warn when unhandled. Clients may legitimately send notifications/cancelled; suppress only
|
||||
that SDK WARN because it would devalue the action-needed WARN level used by M4 fleet health.
|
||||
If a later SDK handles cancellation, this filter simply stops matching and can be removed.
|
||||
-->
|
||||
<turboFilter class="dev.ltms.bridged.logging.McpCancelledNotificationFilter"/>
|
||||
|
||||
<appender name="STDOUT" class="ch.qos.logback.core.ConsoleAppender">
|
||||
<encoder>
|
||||
<pattern>%d{HH:mm:ss.SSS} %-5level [%thread] %logger{28} - %msg%n</pattern>
|
||||
</encoder>
|
||||
</appender>
|
||||
|
||||
<!--
|
||||
CB-505 audit trail. Its own file, deliberately not the app log: privileged actions
|
||||
(spawn/stop/send/reply/drain) must stay greppable and shippable without dragging DEBUG noise
|
||||
along. AuditLog emits a complete JSON object including its own ISO-8601 "ts" field, so the
|
||||
pattern is a bare %msg — a pattern that spliced literal braces around the message would
|
||||
collide with logback's own variable substitution. Rolls daily, 30 days retained, 100MB cap.
|
||||
|
||||
NOTE: records carry who/what/target/outcome only. Message CONTENT is never written here —
|
||||
this bridge carries source code and prompts, and an audit log that accumulated them would be
|
||||
a transcript archive rather than a control.
|
||||
-->
|
||||
<appender name="AUDIT" class="ch.qos.logback.core.rolling.RollingFileAppender">
|
||||
<file>logs/audit.log</file>
|
||||
<rollingPolicy class="ch.qos.logback.core.rolling.SizeAndTimeBasedRollingPolicy">
|
||||
<fileNamePattern>logs/audit.%d{yyyy-MM-dd}.%i.log</fileNamePattern>
|
||||
<maxFileSize>10MB</maxFileSize>
|
||||
<maxHistory>30</maxHistory>
|
||||
<totalSizeCap>100MB</totalSizeCap>
|
||||
</rollingPolicy>
|
||||
<encoder>
|
||||
<pattern>%msg%n</pattern>
|
||||
</encoder>
|
||||
</appender>
|
||||
|
||||
<logger name="dev.ltms.bridged" level="DEBUG"/>
|
||||
<logger name="io.javalin" level="INFO"/>
|
||||
<logger name="org.eclipse.jetty" level="WARN"/>
|
||||
|
||||
<!-- additivity=false keeps the audit stream out of stdout; it is its own record. -->
|
||||
<logger name="audit" level="INFO" additivity="false">
|
||||
<appender-ref ref="AUDIT"/>
|
||||
</logger>
|
||||
|
||||
<root level="INFO">
|
||||
<appender-ref ref="STDOUT"/>
|
||||
</root>
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
package dev.ltms.bridged;
|
||||
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.Map;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-534: the injector's readiness gate must open for a lead as well as for a present worker.
|
||||
*
|
||||
* <p>The bug these cover was silent and slow: a lead was never marked present (only workers are), so
|
||||
* every lead→lead delivery sat on the gate for the full readiness grace and failed ~60s later without
|
||||
* a keystroke ever reaching the pane.
|
||||
*/
|
||||
class BridgedDeliverabilityTest {
|
||||
|
||||
private static Supplier<Map<String, String>> leads(Map<String, String> m) {
|
||||
return () -> m;
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a worker that has connected its MCP is deliverable")
|
||||
void presentWorkerIsDeliverable() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
presence.markPresent("term_worker");
|
||||
|
||||
assertTrue(Bridged.deliverableTo(presence, leads(Map.of())).test("term_worker"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a worker still in its boot window is held back")
|
||||
void absentWorkerIsNotDeliverable() {
|
||||
assertFalse(Bridged.deliverableTo(new MemberPresence(), leads(Map.of())).test("term_booting"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a lead is deliverable without ever being marked present")
|
||||
void leadIsDeliverableWithoutPresence() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
Predicate<String> deliverable =
|
||||
Bridged.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")));
|
||||
|
||||
assertFalse(presence.isPresent("term_lead"), "a lead is never enrolled in worker presence");
|
||||
assertTrue(deliverable.test("term_lead"), "…and must be deliverable anyway");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an unknown terminal is deliverable to neither")
|
||||
void strangerIsNotDeliverable() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
presence.markPresent("term_worker");
|
||||
|
||||
assertFalse(Bridged.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")))
|
||||
.test("term_stranger"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a lead discovered after startup becomes deliverable with no restart")
|
||||
void leadSetIsReadThroughOnEveryCall() {
|
||||
Map<String, String> discovered = new HashMap<>();
|
||||
Predicate<String> deliverable = Bridged.deliverableTo(new MemberPresence(), leads(discovered));
|
||||
|
||||
assertFalse(deliverable.test("term_late"));
|
||||
discovered.put("term_late", "gpt-sol-5.6"); // leadScan picks up a newly labelled tab
|
||||
assertTrue(deliverable.test("term_late"), "the supplier must be re-read, not snapshotted");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("forgetting a torn-down worker does not strip a lead of its deliverability")
|
||||
void forgetDoesNotDisarmALead() {
|
||||
MemberPresence presence = new MemberPresence();
|
||||
Predicate<String> deliverable =
|
||||
Bridged.deliverableTo(presence, leads(Map.of("term_lead", "opus-5.0")));
|
||||
|
||||
presence.forget("term_lead"); // the injector's cleanup path runs against every target
|
||||
|
||||
assertTrue(deliverable.test("term_lead"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,100 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.LoggerContext;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-505 — the audit record's shape.
|
||||
*
|
||||
* <p>These exist because the first cut of this feature emitted lines that were <em>not</em> valid
|
||||
* JSON: the timestamp was spliced on by a logback pattern whose literal braces collided with
|
||||
* logback's variable substitution. The appender failed to parse, and nothing in the build noticed.
|
||||
* An audit trail that silently stops being machine-readable is worse than none.
|
||||
*/
|
||||
class AuditLogTest {
|
||||
|
||||
private final ObjectMapper mapper = new ObjectMapper();
|
||||
private ListAppender<ILoggingEvent> appender;
|
||||
private ch.qos.logback.classic.Logger auditLogger;
|
||||
|
||||
@BeforeEach
|
||||
void attach() {
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
auditLogger = ctx.getLogger("audit");
|
||||
appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
auditLogger.addAppender(appender);
|
||||
auditLogger.setLevel(Level.INFO);
|
||||
}
|
||||
|
||||
@AfterEach
|
||||
void detach() {
|
||||
auditLogger.detachAppender(appender);
|
||||
}
|
||||
|
||||
private JsonNode onlyRecord() throws Exception {
|
||||
assertEquals(1, appender.list.size(), "exactly one audit line expected");
|
||||
String line = appender.list.getFirst().getFormattedMessage();
|
||||
return mapper.readTree(line); // throws if the line is not valid JSON
|
||||
}
|
||||
|
||||
@Test
|
||||
void anAllowedActionIsRecordedAsValidJson() throws Exception {
|
||||
AuditLog.allowed(Principal.primary(4242), Authz.Action.SPAWN, "term_a");
|
||||
|
||||
JsonNode r = onlyRecord();
|
||||
assertEquals("PRIMARY", r.path("role").asText());
|
||||
assertEquals("primary", r.path("actor").asText());
|
||||
assertEquals(4242, r.path("pid").asLong());
|
||||
assertEquals("SPAWN", r.path("action").asText());
|
||||
assertEquals("term_a", r.path("target").asText());
|
||||
assertEquals("allowed", r.path("outcome").asText());
|
||||
assertFalse(r.path("ts").asText().isBlank(), "every record carries its own timestamp");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aDenialRecordsTheReason() throws Exception {
|
||||
AuditLog.denied(Principal.worker("term_b", 7), Authz.Action.REPLY, "term_a", "forbidden");
|
||||
|
||||
JsonNode r = onlyRecord();
|
||||
assertEquals("WORKER", r.path("role").asText());
|
||||
assertEquals("worker:term_b", r.path("actor").asText());
|
||||
assertEquals("denied", r.path("outcome").asText());
|
||||
assertEquals("forbidden", r.path("reason").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNullCallerIsRecordedAsAnonymousRatherThanCrashing() throws Exception {
|
||||
AuditLog.failed(null, Authz.Action.SEND, null, "herdr unreachable");
|
||||
|
||||
JsonNode r = onlyRecord();
|
||||
assertEquals("ANONYMOUS", r.path("role").asText());
|
||||
assertTrue(r.path("target").isNull(), "an absent target is JSON null, not the string \"null\"");
|
||||
assertEquals("failed", r.path("outcome").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
void hostileValuesAreEscapedAndCannotForgeAnExtraRecord() throws Exception {
|
||||
// A target id containing a quote and a newline must not be able to terminate the JSON
|
||||
// object early and inject a second, attacker-shaped audit line.
|
||||
AuditLog.denied(Principal.worker("term_a", 1), Authz.Action.REPLY,
|
||||
"evil\",\"outcome\":\"allowed\"}\n{\"forged\":true", "forbidden");
|
||||
|
||||
JsonNode r = onlyRecord();
|
||||
assertEquals("denied", r.path("outcome").asText(),
|
||||
"the injected outcome must not override the real one");
|
||||
assertTrue(r.path("target").asText().contains("forged"),
|
||||
"the hostile text survives as inert data inside the target field");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,124 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static dev.ltms.bridged.auth.Authz.Action.*;
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/** CB-505 — the authorization table, pinned so it cannot drift silently. */
|
||||
class AuthzTest {
|
||||
|
||||
private static final Principal PRIMARY = Principal.primary(100);
|
||||
private static final Principal WORKER_A = Principal.worker("term_a", 200);
|
||||
private static final Principal WORKER_B = Principal.worker("term_b", 300);
|
||||
private static final Principal ANON = Principal.anonymous();
|
||||
private static final Principal ARCH_DESIGN = Principal.architect("lead-designer", "term_design", 400);
|
||||
private static final Principal ARCH_OTHER = Principal.architect("reviewer", "term_review", 500);
|
||||
|
||||
@Test
|
||||
void anonymousIsAuthorizedForNothing() {
|
||||
for (Authz.Action a : Authz.Action.values()) {
|
||||
assertFalse(Authz.permits(ANON, a, "term_a"),
|
||||
a + " must be refused to an unauthenticated caller");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNullCallerIsTreatedAsAnonymous() {
|
||||
assertFalse(Authz.permits(null, READ, null));
|
||||
assertTrue(Authz.isUnauthenticated(null));
|
||||
}
|
||||
|
||||
@Test
|
||||
void orchestrationBelongsToThePrimaryAlone() {
|
||||
for (Authz.Action a : new Authz.Action[]{SPAWN, STOP, SEND, DRAIN}) {
|
||||
assertTrue(Authz.permits(PRIMARY, a, "term_a"), "the primary orchestrates: " + a);
|
||||
assertFalse(Authz.permits(WORKER_A, a, "term_a"),
|
||||
"a worker performing " + a + " would be escalating into the orchestrator role");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerMayReplyAndAskOnlyAsItself() {
|
||||
assertTrue(Authz.permits(WORKER_A, REPLY, "term_a"));
|
||||
assertTrue(Authz.permits(WORKER_A, ASK, "term_a"));
|
||||
|
||||
assertFalse(Authz.permits(WORKER_A, REPLY, "term_b"),
|
||||
"worker A must not be able to reply on worker B's session");
|
||||
assertFalse(Authz.permits(WORKER_B, ASK, "term_a"),
|
||||
"worker B must not be able to ask as worker A");
|
||||
}
|
||||
|
||||
@Test
|
||||
void thePrimaryMayNotForgeAWorkersReply() {
|
||||
// Not a hypothetical nicety: a forged reply would resolve the rendezvous the primary is
|
||||
// itself blocked on, corrupting the correlation between a turn and its answer.
|
||||
assertFalse(Authz.permits(PRIMARY, REPLY, "term_a"));
|
||||
assertFalse(Authz.permits(PRIMARY, ASK, "term_a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerWithNoTargetCannotReply() {
|
||||
assertFalse(Authz.permits(WORKER_A, REPLY, null),
|
||||
"an absent session id must not satisfy the own-session rule");
|
||||
}
|
||||
|
||||
// ── CB-548: the architect matrix ───────────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void anArchitectMaySendButNotSpawnStopOrDrain() {
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, SEND, "term_worker"),
|
||||
"delegating a turn to a worker IS the architect's job");
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, SEND, null));
|
||||
|
||||
for (Authz.Action a : new Authz.Action[]{SPAWN, STOP, DRAIN}) {
|
||||
assertFalse(Authz.permits(ARCH_DESIGN, a, null),
|
||||
"an architect must not " + a + " — fleet lifecycle is the primary's alone, so "
|
||||
+ "a coordinator cannot also stand up or tear down the fleet");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectMayReplyAndAskOnlyAsItsOwnPane() {
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, REPLY, "term_design"), "its own pane is its own");
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, ASK, "term_design"));
|
||||
|
||||
assertFalse(Authz.permits(ARCH_DESIGN, REPLY, "term_review"),
|
||||
"architect 'lead-designer' must not reply on reviewer's pane");
|
||||
assertFalse(Authz.permits(ARCH_OTHER, ASK, "term_design"),
|
||||
"reviewer must not ask as lead-designer — no terminal is another's");
|
||||
assertFalse(Authz.permits(ARCH_DESIGN, REPLY, null),
|
||||
"an absent target must not pass the own-session rule");
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectMayReadAndScrapeMetrics() {
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, READ, null));
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, METRICS, null));
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectIsNotCountedAsPrimaryOrWorker() {
|
||||
assertFalse(Authz.permits(ARCH_DESIGN, SPAWN, null), "not a primary — no lifecycle");
|
||||
assertFalse(ARCH_DESIGN.isPrimary());
|
||||
assertFalse(ARCH_DESIGN.isWorker(), "an architect is its own role, not a widened worker");
|
||||
assertTrue(ARCH_DESIGN.isArchitect());
|
||||
}
|
||||
|
||||
@Test
|
||||
void observationIsOpenToBothAuthenticatedRoles() {
|
||||
assertTrue(Authz.permits(PRIMARY, READ, null));
|
||||
assertTrue(Authz.permits(WORKER_A, READ, null));
|
||||
assertTrue(Authz.permits(PRIMARY, METRICS, null));
|
||||
assertTrue(Authz.permits(WORKER_A, METRICS, null));
|
||||
}
|
||||
|
||||
@Test
|
||||
void unauthenticatedIsDistinguishedFromMerelyForbidden() {
|
||||
// Drives the 401-vs-403 split: a missing credential is fixable by the caller, a wrong role
|
||||
// is not.
|
||||
assertTrue(Authz.isUnauthenticated(ANON));
|
||||
assertFalse(Authz.isUnauthenticated(WORKER_A));
|
||||
assertFalse(Authz.isUnauthenticated(PRIMARY));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,416 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.PaneLocator;
|
||||
import dev.ltms.bridged.mcp.ConnectionIdentity;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-501. The behaviour under test is the inversion of the pre-CB-501 default: failing every
|
||||
* identity check must yield {@link Role#ANONYMOUS}, not {@code PRIMARY}.
|
||||
*/
|
||||
class CallerResolverTest {
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
|
||||
/** Identity resolving the canned worker pane, keyed off a faked peer-PID lookup. */
|
||||
private ConnectionIdentity identity(long pid) {
|
||||
return new ConnectionIdentity(new PaneLocator(herdr), _ -> pid);
|
||||
}
|
||||
|
||||
/** A PID that owns a worker pane in the fake. */
|
||||
private ConnectionIdentity workerIdentity() {
|
||||
return identity(FakeHerdr.WORKER_PID);
|
||||
}
|
||||
|
||||
/** A PID that owns no pane — i.e. the primary, or any other local process. */
|
||||
private ConnectionIdentity nonWorkerIdentity() {
|
||||
return identity(999_999);
|
||||
}
|
||||
|
||||
private static MemberRegistry boundMembers(String slot, MemberRole role) {
|
||||
Map<String, BridgedConfig.Slot> architects = role == MemberRole.ARCHITECT
|
||||
? Map.of("lead-designer", new BridgedConfig.Slot("sonnet")) : Map.of();
|
||||
Map<String, BridgedConfig.Slot> devs = role == MemberRole.DEV
|
||||
? Map.of("builder", new BridgedConfig.Slot("sonnet")) : Map.of();
|
||||
MemberRegistry members = new MemberRegistry(
|
||||
new BridgedConfig.Fleet(Map.of(), architects, devs, Map.of(), null));
|
||||
assertTrue(members.bind(slot, "term_a"));
|
||||
return members;
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLoopbackWorkerPaneResolvesToWorkerRegardlessOfAuthMode() {
|
||||
Principal underTrust = new CallerResolver(workerIdentity()).resolve("127.0.0.1", 42, null);
|
||||
Principal underToken = new CallerResolver(workerIdentity(), true, "s3cret")
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, underTrust.role());
|
||||
assertEquals("term_a", underTrust.terminal());
|
||||
assertEquals(Role.WORKER, underToken.role(),
|
||||
"worker identity is unforgeable and must never be token-gated — otherwise enabling "
|
||||
+ "auth would lock the whole fleet out of bridge_reply");
|
||||
assertEquals("term_a", underToken.terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPinnedPrimaryTerminalResolvesToPrimaryNotWorker() {
|
||||
// The primary's own session lives in a herdr pane (term_a here). Without the pin the pane
|
||||
// match wins and the primary is locked out of spawn/send/stop as a misread worker.
|
||||
Principal p = CallerResolver.pinnedTo(workerIdentity(), false, null, "term_a")
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPinnedPrimaryTerminalNeedsNoTokenEvenInTokenMode() {
|
||||
Principal p = CallerResolver.pinnedTo(workerIdentity(), true, "s3cret", "term_a")
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role(),
|
||||
"the pane mapping is as unforgeable as a worker's — the pin outranks the token path");
|
||||
}
|
||||
|
||||
@Test
|
||||
void otherPanesRemainWorkersWhenAPinIsSet() {
|
||||
Principal p = CallerResolver.pinnedTo(workerIdentity(), false, null, "term_someone_else")
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role());
|
||||
assertEquals("term_a", p.terminal());
|
||||
}
|
||||
|
||||
/** The pin is optional config, so an absent or whitespace one must change nothing at all. */
|
||||
@Test
|
||||
void aBlankPinLeavesWorkerResolutionUntouched() {
|
||||
assertEquals(Role.WORKER,
|
||||
CallerResolver.pinnedTo(workerIdentity(), false, null, " ").resolve("127.0.0.1", 42, null).role());
|
||||
assertEquals(Role.WORKER,
|
||||
CallerResolver.pinnedTo(workerIdentity(), false, null, null).resolve("127.0.0.1", 42, null).role());
|
||||
}
|
||||
|
||||
@Test
|
||||
void loopbackTrustTreatsANonWorkerLoopbackCallerAsThePrimary() {
|
||||
Principal p = new CallerResolver(nonWorkerIdentity()).resolve("127.0.0.1", 99, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role(), "the historical behaviour, now an explicit choice");
|
||||
}
|
||||
|
||||
@Test
|
||||
void tokenModeRefusesANonWorkerCallerThatPresentsNoToken() {
|
||||
Principal p = new CallerResolver(nonWorkerIdentity(), true, "s3cret")
|
||||
.resolve("127.0.0.1", 99, null);
|
||||
|
||||
assertEquals(Role.ANONYMOUS, p.role(),
|
||||
"no credential must mean NOTHING, not the most privileged role on the bus");
|
||||
}
|
||||
|
||||
@Test
|
||||
void tokenModeAcceptsAValidBearerTokenAsThePrimary() {
|
||||
Principal p = new CallerResolver(nonWorkerIdentity(), true, "s3cret")
|
||||
.resolve("127.0.0.1", 99, "Bearer s3cret");
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
}
|
||||
|
||||
@Test
|
||||
void tokenModeRejectsAWrongOrMalformedCredential() {
|
||||
CallerResolver r = new CallerResolver(nonWorkerIdentity(), true, "s3cret");
|
||||
|
||||
assertEquals(Role.ANONYMOUS, r.resolve("127.0.0.1", 99, "Bearer wrong").role());
|
||||
assertEquals(Role.ANONYMOUS, r.resolve("127.0.0.1", 99, "s3cret").role(), "scheme required");
|
||||
assertEquals(Role.ANONYMOUS, r.resolve("127.0.0.1", 99, "Bearer ").role(), "empty credential");
|
||||
assertEquals(Role.ANONYMOUS, r.resolve("127.0.0.1", 99, "Basic s3cret").role(), "wrong scheme");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theBearerSchemeIsCaseInsensitivePerRfc7235() {
|
||||
CallerResolver r = new CallerResolver(nonWorkerIdentity(), true, "s3cret");
|
||||
|
||||
assertEquals(Role.PRIMARY, r.resolve("127.0.0.1", 99, "bearer s3cret").role());
|
||||
assertEquals(Role.PRIMARY, r.resolve("127.0.0.1", 99, "BEARER s3cret").role());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonLoopbackCallerIsNeverThePrimaryUnderLoopbackTrust() {
|
||||
// Defence in depth: startup already refuses this pairing (validateAuthExposure), but if a
|
||||
// proxy ever forwards a remote peer onto the loopback listener, the resolver must not
|
||||
// hand it the primary role.
|
||||
Principal p = new CallerResolver(nonWorkerIdentity()).resolve("10.0.0.7", 99, null);
|
||||
|
||||
assertEquals(Role.ANONYMOUS, p.role());
|
||||
}
|
||||
|
||||
// ── CB-530: the leaders registry ────────────────────────────────────────────────────────────
|
||||
// The fake resolves exactly one pane (term_a) from a PID, so "two leads both resolve" is
|
||||
// asserted at the config layer (BridgedConfigTest#leaderTerminals…). What matters here is that
|
||||
// resolution is a REGISTRY LOOKUP rather than a single equality test against one pin.
|
||||
|
||||
@Test
|
||||
void aRegisteredLeadPaneResolvesToPrimaryCarryingItsName() {
|
||||
Principal p = new CallerResolver(workerIdentity(), false, null, Map.of("term_a", "opus-5.0"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
assertEquals("opus-5.0", p.name(), "whoami must be able to say WHICH lead is asking");
|
||||
assertEquals("term_a", p.terminal(),
|
||||
"CB-532: a lead carries the pane it was matched by. Without it ownsSession() can "
|
||||
+ "never be true for a lead, so it can send to a peer but never answer one");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aSecondLeadIsRecognisedRatherThanSilentlyDemoted() {
|
||||
// The regression this feature exists for: with a singular pin, whichever lead was not the
|
||||
// pin resolved as a worker and was refused every orchestration call.
|
||||
Principal p = new CallerResolver(workerIdentity(), false, null,
|
||||
Map.of("term_elsewhere", "gpt-sol-5.6", "term_a", "opus-5.0"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
assertEquals("opus-5.0", p.name());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPaneAbsentFromTheRegistryIsStillAWorker() {
|
||||
Principal p = new CallerResolver(workerIdentity(), false, null,
|
||||
Map.of("term_elsewhere", "gpt-sol-5.6"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role());
|
||||
assertEquals("term_a", p.terminal());
|
||||
assertNull(p.name());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aRegisteredLeadNeedsNoTokenEvenInTokenMode() {
|
||||
Principal p = new CallerResolver(workerIdentity(), true, "s3cret", Map.of("term_a", "opus"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role(),
|
||||
"the pane mapping is as unforgeable as a worker's — it outranks the token path");
|
||||
assertEquals("opus", p.name());
|
||||
}
|
||||
|
||||
/** The pre-CB-530 spelling must keep working, exactly, including for configs that never migrate. */
|
||||
@Test
|
||||
void theLegacySinglePinBehavesAsALeadNamedPrimary() {
|
||||
Principal p = CallerResolver.pinnedTo(workerIdentity(), false, null, "term_a")
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
assertEquals("primary", p.name());
|
||||
}
|
||||
|
||||
@Test
|
||||
void anEmptyRegistryLeavesEveryPaneAWorker() {
|
||||
Map<String, String> noLeads = null;
|
||||
assertEquals(Role.WORKER,
|
||||
new CallerResolver(workerIdentity(), false, null, Map.of())
|
||||
.resolve("127.0.0.1", 42, null).role());
|
||||
assertEquals(Role.WORKER,
|
||||
new CallerResolver(workerIdentity(), false, null, noLeads)
|
||||
.resolve("127.0.0.1", 42, null).role());
|
||||
// CB-531: and the same for the live-registry form, whose supplier may also be absent.
|
||||
assertEquals(Role.WORKER,
|
||||
CallerResolver.withLeads(workerIdentity(), false, null, null)
|
||||
.resolve("127.0.0.1", 42, null).role());
|
||||
}
|
||||
|
||||
/** The audit line must distinguish leads once several exist, or a log says nothing useful. */
|
||||
@Test
|
||||
void describeNamesTheLeadButStillReadsPrimaryWhenUnnamed() {
|
||||
assertEquals("leader:opus-5.0", Principal.leader("opus-5.0", "term_a", 1).describe());
|
||||
assertEquals("primary", Principal.primary(1).describe());
|
||||
assertEquals("worker:term_a", Principal.worker("term_a", 1).describe());
|
||||
}
|
||||
|
||||
// ── CB-532: a lead is an addressable peer, not only a sender ────────────────────────────────
|
||||
|
||||
/**
|
||||
* The regression this ticket exists for: two leads could both be recognised (CB-530/531) and
|
||||
* still not converse, because REPLY is gated on ownsSession() and a lead owned nothing.
|
||||
*/
|
||||
@Test
|
||||
void aLeadOwnsItsOwnPaneSoItMayAnswerAPeer() {
|
||||
Principal lead = new CallerResolver(workerIdentity(), false, null, Map.of("term_a", "opus-5.0"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertTrue(lead.ownsSession("term_a"));
|
||||
assertTrue(Authz.permits(lead, Authz.Action.REPLY, "term_a"),
|
||||
"a lead answering a peer replies for its OWN terminal — the rendezvous the sender "
|
||||
+ "opened is keyed on exactly that");
|
||||
assertTrue(Authz.permits(lead, Authz.Action.ASK, "term_a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLeadStillCannotActAsAnyoneElse() {
|
||||
Principal lead = new CallerResolver(workerIdentity(), false, null, Map.of("term_a", "opus-5.0"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertFalse(lead.ownsSession("term_someone_else"));
|
||||
assertFalse(Authz.permits(lead, Authz.Action.REPLY, "term_someone_else"),
|
||||
"widening WHO may reply must not widen WHAT they may reply as");
|
||||
}
|
||||
|
||||
/** A primary with no pane — token mode, or off-host — owns nothing and must stay a sender only. */
|
||||
@Test
|
||||
void anUnnamedPrimaryWithNoPaneOwnsNothing() {
|
||||
Principal p = new CallerResolver(nonWorkerIdentity(), true, "s3cret")
|
||||
.resolve("127.0.0.1", 99, "Bearer s3cret");
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
assertNull(p.terminal());
|
||||
assertFalse(p.ownsSession(null), "a null terminal must never match a null session id");
|
||||
assertFalse(Authz.permits(p, Authz.Action.REPLY, null));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLeadKeepsEveryOrchestrationRightItAlreadyHad() {
|
||||
Principal lead = new CallerResolver(workerIdentity(), false, null, Map.of("term_a", "opus-5.0"))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertTrue(Authz.permits(lead, Authz.Action.SPAWN, null));
|
||||
assertTrue(Authz.permits(lead, Authz.Action.SEND, "term_worker"));
|
||||
assertTrue(Authz.permits(lead, Authz.Action.STOP, null));
|
||||
assertTrue(Authz.permits(lead, Authz.Action.DRAIN, null));
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-531: the registry is read per resolve, not snapshotted at construction — a lead that
|
||||
* labels its tab after the daemon booted is recognised without a restart.
|
||||
*/
|
||||
@Test
|
||||
void aLeadRegisteredAfterConstructionIsHonouredWithoutRebuildingTheResolver() {
|
||||
Map<String, String> live = new java.util.HashMap<>();
|
||||
CallerResolver r = CallerResolver.withLeads(workerIdentity(), false, null, () -> live);
|
||||
|
||||
assertEquals(Role.WORKER, r.resolve("127.0.0.1", 42, null).role());
|
||||
|
||||
live.put("term_a", "gpt-sol-5.6"); // the scanner sees a newly-labelled tab
|
||||
|
||||
Principal p = r.resolve("127.0.0.1", 42, null);
|
||||
assertEquals(Role.PRIMARY, p.role());
|
||||
assertEquals("gpt-sol-5.6", p.name());
|
||||
}
|
||||
|
||||
/** The map form must stay a snapshot: a caller handing over a map is not offering live state. */
|
||||
@Test
|
||||
void theMapFormIsCopiedSoLaterMutationCannotGrantLeadership() {
|
||||
Map<String, String> mutable = new java.util.HashMap<>();
|
||||
CallerResolver r = new CallerResolver(workerIdentity(), false, null, mutable);
|
||||
|
||||
mutable.put("term_a", "sneaky");
|
||||
|
||||
assertEquals(Role.WORKER, r.resolve("127.0.0.1", 42, null).role());
|
||||
}
|
||||
|
||||
// ── CB-548: architect slots ─────────────────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void aBoundArchitectPaneResolvesToArchitectBeforeTheWorkerFallback() {
|
||||
// This is the production construction path used by Bridged.
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, boundMembers("architect:lead-designer", MemberRole.ARCHITECT))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.ARCHITECT, p.role(),
|
||||
"a terminal bound to an architect slot is an architect, NOT the generic worker it "
|
||||
+ "would otherwise resolve to");
|
||||
assertEquals("lead-designer", p.name(), "whoami must say WHICH slot is asking");
|
||||
assertEquals("term_a", p.terminal(), "the pane identity is carried so ownsSession works");
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectNeedsNoTokenEvenInTokenMode() {
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), true, "s3cret",
|
||||
Map::of, boundMembers("architect:lead-designer", MemberRole.ARCHITECT))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.ARCHITECT, p.role(),
|
||||
"the pane mapping is as unforgeable as a worker's — it outranks the token path");
|
||||
}
|
||||
|
||||
@Test
|
||||
void anUnboundPaneStillResolvesAsAWorker() {
|
||||
MemberRegistry members = new MemberRegistry(new BridgedConfig.Fleet(Map.of(),
|
||||
Map.of("lead-designer", new BridgedConfig.Slot("sonnet")), Map.of(), Map.of(), null));
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, members).resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role());
|
||||
assertNull(p.name());
|
||||
}
|
||||
|
||||
/** CB-548 precedence: lead > architect > worker, so a pane named in BOTH is still a lead. */
|
||||
@Test
|
||||
void aLeadWinsOverAnArchitectBindingForTheSamePane() {
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
() -> Map.of("term_a", "opus-5.0"),
|
||||
boundMembers("architect:lead-designer", MemberRole.ARCHITECT))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.PRIMARY, p.role(),
|
||||
"a pane the config calls a lead must keep resolving as a lead — no behaviour change "
|
||||
+ "when an architect binding is added to an existing fleet");
|
||||
assertEquals("opus-5.0", p.name());
|
||||
}
|
||||
|
||||
/** The registry is live, like leads: a binding injected after construction is honoured. */
|
||||
@Test
|
||||
void anArchitectBoundAfterConstructionIsHonouredWithoutRebuildingTheResolver() {
|
||||
MemberRegistry members = new MemberRegistry(new BridgedConfig.Fleet(Map.of(),
|
||||
Map.of("lead-designer", new BridgedConfig.Slot("sonnet")), Map.of(), Map.of(), null));
|
||||
CallerResolver r = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, members);
|
||||
|
||||
assertEquals(Role.WORKER, r.resolve("127.0.0.1", 42, null).role());
|
||||
|
||||
assertTrue(members.bind("architect:lead-designer", "term_a")); // the later lifecycle binds the slot
|
||||
|
||||
assertEquals(Role.ARCHITECT, r.resolve("127.0.0.1", 42, null).role());
|
||||
assertEquals("architect:lead-designer", r.members().get("term_a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void describeNamesTheArchitectSlot() {
|
||||
assertEquals("architect:lead-designer",
|
||||
Principal.architect("lead-designer", "term_a", 1).describe());
|
||||
}
|
||||
|
||||
/** An architect acts only as its own pane — the same ownsSession rule as a worker or lead. */
|
||||
@Test
|
||||
void anArchitectOwnsItsOwnPaneAndNoOther() {
|
||||
Principal arch = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, boundMembers("architect:lead-designer", MemberRole.ARCHITECT))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertTrue(arch.ownsSession("term_a"));
|
||||
assertTrue(Authz.permits(arch, Authz.Action.REPLY, "term_a"));
|
||||
assertFalse(arch.ownsSession("term_b"));
|
||||
assertFalse(Authz.permits(arch, Authz.Action.REPLY, "term_b"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBoundNonArchitectSlotStillResolvesAsAWorker() {
|
||||
Principal p = CallerResolver.withLeadsAndMembers(workerIdentity(), false, null,
|
||||
Map::of, boundMembers("dev:builder", MemberRole.DEV))
|
||||
.resolve("127.0.0.1", 42, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role(), "a dev binding must never grant architect rights");
|
||||
}
|
||||
|
||||
@Test
|
||||
void tokenModeRequiresANonEmptyConfiguredToken() {
|
||||
ConnectionIdentity id = nonWorkerIdentity();
|
||||
|
||||
assertThrows(IllegalArgumentException.class, () -> new CallerResolver(id, true, null));
|
||||
assertThrows(IllegalArgumentException.class, () -> new CallerResolver(id, true, " "));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,269 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.Future;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-548 — the architect-slot registry: the config snapshot of slot → profile, and the live
|
||||
* terminal → slot bindings it owns. The role a binding produces is asserted in
|
||||
* {@link CallerResolverTest}; this pins the registry object itself — its invariants and their
|
||||
* thread-safety.
|
||||
*/
|
||||
class MemberRegistryTest {
|
||||
|
||||
/**
|
||||
* Slot keys are qualified by role (CB-557): a bare name is unique only within its pool, so
|
||||
* {@code sonnet} can be both a developer and a reviewer, while a terminal binds to exactly one.
|
||||
*/
|
||||
private static final String DESIGNER = "architect:lead-designer";
|
||||
private static final String REVIEWER = "architect:code-reviewer";
|
||||
|
||||
private static BridgedConfig.Fleet fleetWith(Map<String, BridgedConfig.Slot> architects) {
|
||||
return new BridgedConfig.Fleet(Map.of(), architects, Map.of(), Map.of(), null);
|
||||
}
|
||||
|
||||
private static Map<String, BridgedConfig.Slot> architects() {
|
||||
Map<String, BridgedConfig.Slot> pool = new LinkedHashMap<>();
|
||||
pool.put("lead-designer", new BridgedConfig.Slot("sonnet"));
|
||||
pool.put("code-reviewer", new BridgedConfig.Slot("gx10"));
|
||||
return pool;
|
||||
}
|
||||
|
||||
private final MemberRegistry registry = new MemberRegistry(fleetWith(architects()));
|
||||
|
||||
@Test
|
||||
void exposesTheConfiguredSlotsQualifiedByRole() {
|
||||
assertEquals(Set.of(DESIGNER, REVIEWER), registry.slots().keySet());
|
||||
assertTrue(registry.isSlot(REVIEWER));
|
||||
assertFalse(registry.isSlot("nope"));
|
||||
assertFalse(registry.isSlot("lead-designer"),
|
||||
"the bare name is not the key — it is unique only inside its pool");
|
||||
}
|
||||
|
||||
/** The case the pool shape exists for: one profile serving two roles is not a duplicate. */
|
||||
@Test
|
||||
void oneProfileMayServeTwoRolesUnderTheSameSlotName() {
|
||||
Map<String, BridgedConfig.Slot> devs = Map.of("sonnet", new BridgedConfig.Slot("sonnet"));
|
||||
Map<String, BridgedConfig.Slot> revs = Map.of("sonnet", new BridgedConfig.Slot("sonnet"));
|
||||
MemberRegistry r = new MemberRegistry(
|
||||
new BridgedConfig.Fleet(Map.of(), Map.of(), devs, revs, null));
|
||||
|
||||
assertEquals(Set.of("dev:sonnet", "reviewer:sonnet"), r.slots().keySet());
|
||||
assertEquals(MemberRole.DEV, r.roleForSlot("dev:sonnet"));
|
||||
assertEquals(MemberRole.REVIEWER, r.roleForSlot("reviewer:sonnet"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void theSpawnLifecycleReadsTheProfileBackFromASlot() {
|
||||
assertEquals("sonnet", registry.profileForSlot(DESIGNER));
|
||||
assertEquals("gx10", registry.profileForSlot(REVIEWER));
|
||||
assertNull(registry.profileForSlot("unknown"), "an unknown slot has no profile");
|
||||
}
|
||||
|
||||
@Test
|
||||
void startsEmptySoNoTerminalResolvesToAnArchitect() {
|
||||
assertTrue(registry.snapshot().isEmpty());
|
||||
assertNull(registry.slotForTerminal("term_design"),
|
||||
"config declares no architect terminal — nothing is recognised until a bind");
|
||||
assertNull(registry.slotForTerminal(null), "no terminal ⇒ no slot");
|
||||
}
|
||||
|
||||
// ── bind ──────────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void bindResolvesTheTerminalToTheSlot() {
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
assertEquals(DESIGNER, registry.slotForTerminal("term_design"));
|
||||
assertEquals(Map.of("term_design", DESIGNER), registry.snapshot());
|
||||
}
|
||||
|
||||
@Test
|
||||
void bindRefusesAnUnknownSlot() {
|
||||
assertFalse(registry.bind("nope", "term_x"),
|
||||
"a slot that is not configured must be refused — bind is not a way to invent one");
|
||||
assertNull(registry.slotForTerminal("term_x"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void bindRefusesATerminalInTwoSlots() {
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
assertFalse(registry.bind(REVIEWER, "term_design"),
|
||||
"a terminal may occupy at most one slot");
|
||||
assertEquals(DESIGNER, registry.slotForTerminal("term_design"),
|
||||
"the first binding survives the refused second");
|
||||
}
|
||||
|
||||
@Test
|
||||
void bindRefusesASlotWithTwoTerminals() {
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
assertFalse(registry.bind(DESIGNER, "term_other"),
|
||||
"a slot may host at most one terminal");
|
||||
assertEquals(DESIGNER, registry.slotForTerminal("term_design"),
|
||||
"the first binding survives the refused second");
|
||||
assertNull(registry.slotForTerminal("term_other"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void rebindingTheSamePairIsAnIdempotentNoOp() {
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"),
|
||||
"the same terminal → slot is harmless to repeat");
|
||||
assertEquals(1, registry.snapshot().size());
|
||||
}
|
||||
|
||||
// ── unbind ────────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void unbindRemovesTheExactBinding() {
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
assertTrue(registry.unbind(DESIGNER, "term_design"));
|
||||
assertNull(registry.slotForTerminal("term_design"));
|
||||
assertTrue(registry.snapshot().isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aStaleUnbindDoesNotRemoveAReplacement() {
|
||||
// Bind, tear down, and stand the slot back up with a NEW terminal.
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
registry.unbind(DESIGNER, "term_design");
|
||||
assertTrue(registry.bind(DESIGNER, "term_new"));
|
||||
|
||||
// A late unbind naming the OLD terminal must not remove the replacement binding.
|
||||
assertFalse(registry.unbind(DESIGNER, "term_design"));
|
||||
assertEquals(DESIGNER, registry.slotForTerminal("term_new"),
|
||||
"the replacement terminal stays bound");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aStaleUnbindForATerminalThatMovedSlotsDoesNothing() {
|
||||
// term_design starts in lead-designer, is torn down, and stands back up in a FREE slot.
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
registry.unbind(DESIGNER, "term_design");
|
||||
assertTrue(registry.bind(REVIEWER, "term_design"));
|
||||
|
||||
// Unbinding against the slot it no longer occupies is refused; the new binding is intact.
|
||||
assertFalse(registry.unbind(DESIGNER, "term_design"),
|
||||
"the old slot must not unbind a terminal that moved elsewhere");
|
||||
assertEquals(REVIEWER, registry.slotForTerminal("term_design"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void unbindOfNothingIsAFalseNoOp() {
|
||||
assertFalse(registry.unbind(DESIGNER, "term_design"),
|
||||
"nothing was bound, so nothing is removed");
|
||||
}
|
||||
|
||||
// ── snapshot ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void theSnapshotIsAnImmutableCopyNotAliveState() {
|
||||
assertTrue(registry.bind(DESIGNER, "term_design"));
|
||||
Map<String, String> snap = registry.snapshot();
|
||||
|
||||
assertThrows(UnsupportedOperationException.class, () -> snap.put("x", "y"),
|
||||
"a handed-out snapshot cannot be mutated in place");
|
||||
|
||||
// Later binds must not leak into an earlier snapshot.
|
||||
assertTrue(registry.bind(REVIEWER, "term_review"));
|
||||
assertFalse(snap.containsKey("term_review"),
|
||||
"a snapshot is a point-in-time copy, not a live view");
|
||||
}
|
||||
|
||||
// ── concurrency (CB-548 invariants hold under contention) ─────────────────────────────────
|
||||
|
||||
@Test
|
||||
void concurrentBindsNeverGiveASlotTwoTerminals() throws Exception {
|
||||
int n = 16;
|
||||
ExecutorService pool = Executors.newFixedThreadPool(n);
|
||||
try {
|
||||
CountDownLatch go = new CountDownLatch(1);
|
||||
List<Future<Boolean>> results = new ArrayList<>();
|
||||
for (int i = 0; i < n; i++) {
|
||||
final String term = "term_" + i; // every thread races for the SAME slot
|
||||
results.add(pool.submit(() -> {
|
||||
go.await();
|
||||
return registry.bind(DESIGNER, term);
|
||||
}));
|
||||
}
|
||||
go.countDown();
|
||||
|
||||
int won = 0;
|
||||
for (Future<Boolean> r : results) {
|
||||
if (r.get()) {
|
||||
won++;
|
||||
}
|
||||
}
|
||||
assertEquals(1, won, "exactly one terminal may win the sole slot, got " + won);
|
||||
assertEquals(1, registry.snapshot().size(),
|
||||
"the slot hosts at most one terminal after the race");
|
||||
} finally {
|
||||
pool.shutdownNow();
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void concurrentBindsNeverPutOneTerminalInTwoSlots() throws Exception {
|
||||
int n = 16;
|
||||
ExecutorService pool = Executors.newFixedThreadPool(n);
|
||||
try {
|
||||
CountDownLatch go = new CountDownLatch(1);
|
||||
List<Future<String>> results = new ArrayList<>();
|
||||
for (int i = 0; i < n; i++) {
|
||||
final String slot = (i % 2 == 0) ? DESIGNER : REVIEWER; // all race for ONE terminal
|
||||
results.add(pool.submit(() -> {
|
||||
go.await();
|
||||
return registry.bind(slot, "shared_term")
|
||||
? registry.slotForTerminal("shared_term") : null;
|
||||
}));
|
||||
}
|
||||
go.countDown();
|
||||
|
||||
// Rebinding the same terminal to the same slot is a harmless idempotent true, so count
|
||||
// winners is not the assertion — agreement is: every thread that reported success must
|
||||
// have seen the terminal in the SAME slot, never in two at once.
|
||||
String bound = null;
|
||||
boolean conflict = false;
|
||||
for (Future<String> r : results) {
|
||||
String s = r.get();
|
||||
if (s != null) {
|
||||
if (bound == null) {
|
||||
bound = s;
|
||||
} else if (!bound.equals(s)) {
|
||||
conflict = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
assertFalse(conflict, "a terminal was observed in two slots at once");
|
||||
assertNotNull(bound, "at least one thread bound the terminal");
|
||||
assertEquals(1, registry.snapshot().size(),
|
||||
"the terminal occupies exactly one slot in the final snapshot");
|
||||
assertEquals(bound, registry.slotForTerminal("shared_term"));
|
||||
} finally {
|
||||
pool.shutdownNow();
|
||||
}
|
||||
}
|
||||
|
||||
/** A handed-over slot map is snapshotted at construction, not offered as live state. */
|
||||
@Test
|
||||
void theSlotSnapshotIsFixedByConstruction() {
|
||||
Map<String, BridgedConfig.Slot> mutable = architects();
|
||||
MemberRegistry r = new MemberRegistry(fleetWith(mutable));
|
||||
|
||||
mutable.put("hijack", new BridgedConfig.Slot("gx10"));
|
||||
|
||||
assertFalse(r.isSlot("architect:hijack"), "a handed-over map is not offered as live state");
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user