Compare commits
21 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 5f5d16fbd4 | |||
| 724b35b46e | |||
| 2eb2d6112e | |||
| 7c458e8bf2 | |||
| 133f03e428 | |||
| 804279175d | |||
| 9dea289975 | |||
| 28b45d97e5 | |||
| 1a397e962e | |||
| 209e1231ea | |||
| 5d8b9d365c | |||
| a6aeda39e7 | |||
| 31b3c24caa | |||
| d105da978d | |||
| 5051a06443 | |||
| 6f275227d2 | |||
| 283ccf8423 | |||
| b96fba4a03 | |||
| 6d97d210b4 | |||
| 37b23cd704 | |||
| 367facf6a6 |
@@ -59,7 +59,7 @@ as `matches HEAD`, `drift`, or `unknown`; do not turn an unclear timestamp into
|
||||
Report the process identifier (PID) and uptime too:
|
||||
|
||||
```bash
|
||||
PIDS="$(pgrep -f 'run/fleetd.jar' || true)"
|
||||
PIDS="$(pgrep -f 'fleetd.jar' || true)"
|
||||
if [ -z "$PIDS" ]; then
|
||||
printf '%s\n' 'fleetd: not running'
|
||||
else
|
||||
|
||||
@@ -164,6 +164,13 @@ fails.
|
||||
- **`accepted` does not mean your pane has been cleared.** It means every gate passed and the roll
|
||||
is scheduled to run once your current turn ends. Say your goodbye in the same turn — you will not
|
||||
get another one.
|
||||
- **If you are still running after that turn, the roll did not happen.** A roll that works clears
|
||||
you, so surviving your own goodbye is itself the signal that it refused. Check with
|
||||
`fleet_handover{action: "status", token}`, using the token you confirmed. `TURN_NEVER_SETTLED`
|
||||
means your turn ran past `leadRollover.turnSettleSeconds` and **no `/clear` was ever sent**: your
|
||||
context is intact and nothing was lost. Open a fresh request and retry. Never assume the roll
|
||||
succeeded because `confirm` answered `accepted` — by the time it refuses, there is no caller left
|
||||
to tell, so this check is the only thing that closes that gap.
|
||||
- **There is no terminal or session parameter, on purpose.** The pane is always your own, resolved
|
||||
from your connection, so you can only ever roll yourself.
|
||||
- **`operatorConfirmed` is your report of what a human told you.** Do not pass `true` because you
|
||||
|
||||
@@ -20,16 +20,22 @@ public final class Authz {
|
||||
SPAWN,
|
||||
/** Tear a worker peer down. */
|
||||
STOP,
|
||||
/** Deliver a turn to a session (or answer a worker's question). */
|
||||
/** Deliver a turn to a local session, addressed by {@code sessionId}. */
|
||||
SEND,
|
||||
/** Resolve a worker's blocked question and resume its turn, addressed by {@code turnId}. */
|
||||
ANSWER,
|
||||
/** Address a peer lead on another daemon over the coordination broker, by {@code coordId}. */
|
||||
COORD_SEND,
|
||||
/** A worker's terminal reply for its own turn. */
|
||||
REPLY,
|
||||
/** A worker's mid-turn question to the primary. */
|
||||
ASK,
|
||||
/** Collect held replies from a session's inbox. */
|
||||
DRAIN,
|
||||
/** Read-only observation: status, roster, profiles, task polling. */
|
||||
/** Read-only roster, profile, and identity observation: no ticket, task, or turn state. */
|
||||
READ,
|
||||
/** Poll a ticket, or read a session's status. */
|
||||
TASK_READ,
|
||||
/**
|
||||
* Read (never ack) this daemon's own held lead-to-lead coordination mail (fleetd #421).
|
||||
*
|
||||
@@ -71,11 +77,19 @@ public final class Authz {
|
||||
// escalating into the orchestrator role.
|
||||
case SPAWN, STOP, DRAIN, HANDOVER -> caller.isPrimary();
|
||||
|
||||
// Delivering a turn is open to the primary and the architect: an architect delegates
|
||||
// to workers (that is the role's point) but still has no lifecycle rights. A worker is
|
||||
// excluded — sending would be it escalating.
|
||||
// Delivering a turn to a local session is open to the primary and the architect: an
|
||||
// architect delegates to workers (that is the role's point) but still has no lifecycle
|
||||
// rights. A worker is excluded — sending would be it escalating.
|
||||
case SEND -> caller.isPrimary() || caller.isArchitect();
|
||||
|
||||
// Same grant as SEND. Resolving a worker's blocked question is part of delegating to
|
||||
// it, not a separate capability.
|
||||
case ANSWER -> caller.isPrimary() || caller.isArchitect();
|
||||
|
||||
// Same grant as SEND. This leaves the daemon over the coordination broker rather than
|
||||
// addressing a local session, but the caller who may do one may do the other.
|
||||
case COORD_SEND -> caller.isPrimary() || caller.isArchitect();
|
||||
|
||||
// The load-bearing rule: a caller acts only as the pane it occupies. CB-532 widened who
|
||||
// that can be — a lead answering another lead is replying for its OWN terminal, which
|
||||
// this already permits — while the rule itself is unchanged, and is what stops anyone
|
||||
@@ -84,10 +98,17 @@ public final class Authz {
|
||||
// unnamed primary (token/loopback, no pane) owns nothing and is still excluded.
|
||||
case REPLY, ASK -> caller.ownsSession(targetSession);
|
||||
|
||||
// Observation is open to every authenticated role: a worker legitimately polls its own
|
||||
// status, and the roster carries no secrets.
|
||||
// READ is roster, profile, and identity observation — fleet_list, fleet_profiles, and
|
||||
// fleet_whoami — and carries no secrets: no ticket reply, no pending question, and no
|
||||
// other session's turn state. Those live under TASK_READ. METRICS is the separate
|
||||
// Prometheus scrape. Both stay open to every authenticated role.
|
||||
case READ, METRICS -> caller.isPrimary() || caller.isWorker() || caller.isArchitect();
|
||||
|
||||
// Ticket polling and session status, open to every authenticated role the same as READ.
|
||||
// Unlike READ, a holder may poll a ticket it did not create, or read another session's
|
||||
// pending question and the turnId that answers it.
|
||||
case TASK_READ -> caller.isPrimary() || caller.isWorker() || caller.isArchitect();
|
||||
|
||||
// fleetd #421: reading held lead-to-lead mail is the primary's alone. An architect
|
||||
// holds READ today (CB-548), so "not primary" must mean not-architect here too — this
|
||||
// is coordination between leads, not observation of the roster.
|
||||
|
||||
@@ -1399,11 +1399,13 @@ public record FleetConfig(
|
||||
* called FROM the calling lead's own turn, so its pane is still {@code WORKING} the instant
|
||||
* {@code confirm()} validates every gate and schedules the roll. {@code
|
||||
* dev.ltms.fleet.lead.LeadRollover}'s deferred continuation waits up to this many seconds for
|
||||
* that SAME pane to report an injectable state again — i.e. for the calling turn to actually
|
||||
* end — before it sends {@code /clear} at all. If that wait times out, no {@code /clear} is
|
||||
* that SAME pane to report {@code IDLE} or {@code DONE} — i.e. for the calling turn to actually
|
||||
* end — before it sends {@code /clear} at all. {@code BLOCKED} does not count: that is a live
|
||||
* turn merely paused, not one that has finished. If that wait times out, no {@code /clear} is
|
||||
* ever sent: a lead that never goes idle is still doing real work, and clearing it would
|
||||
* destroy live context. This is a separate wait from {@code clearSettleSeconds} below, which
|
||||
* bounds the SECOND wait, for the pane to re-settle AFTER {@code /clear} has already gone out.
|
||||
* bounds the SECOND wait, for the pane to reach {@code IDLE} or {@code DONE} again AFTER
|
||||
* {@code /clear} has already gone out.
|
||||
*
|
||||
* @param handoverPath required when this block is present — where the handover file a fresh
|
||||
* lead session reads must live. There is no sane non-null default for an
|
||||
@@ -1422,12 +1424,13 @@ public record FleetConfig(
|
||||
* @param maxDocAgeSeconds default 3600 — refuse a handover file whose modified time is older
|
||||
* than this many seconds, so a stale leftover from an earlier rollover
|
||||
* attempt can never be mistaken for a fresh one.
|
||||
* @param turnSettleSeconds default 20 — bound on how long the deferred roll waits for the
|
||||
* CALLING lead's own turn to end (its pane to report injectable again)
|
||||
* before sending {@code /clear} at all. See the paragraph above.
|
||||
* @param turnSettleSeconds default 300 — bound on how long the deferred roll waits for the
|
||||
* CALLING lead's own turn to end (its pane to report {@code IDLE} or
|
||||
* {@code DONE}) before sending {@code /clear} at all. See the paragraph
|
||||
* above.
|
||||
* @param clearSettleSeconds default 20 — bound on how long to wait for the lead's pane to
|
||||
* report an injectable state again after {@code /clear} before giving up. A
|
||||
* roll that times out here never sends {@code bootstrapText}.
|
||||
* report {@code IDLE} or {@code DONE} again after {@code /clear} before
|
||||
* giving up. A roll that times out here never sends {@code bootstrapText}.
|
||||
* @param bootstrapText default a sentence naming the RESOLVED handover path — sent to the
|
||||
* lead's pane once it settles after {@code /clear}, telling the fresh
|
||||
* session where to read the handover and carry on. Left {@code null} here
|
||||
@@ -1444,7 +1447,7 @@ public record FleetConfig(
|
||||
public LeadRollover {
|
||||
requireOperatorConfirm = requireOperatorConfirm == null || requireOperatorConfirm;
|
||||
maxDocAgeSeconds = (maxDocAgeSeconds == null || maxDocAgeSeconds <= 0) ? 3600 : maxDocAgeSeconds;
|
||||
turnSettleSeconds = (turnSettleSeconds == null || turnSettleSeconds <= 0) ? 20 : turnSettleSeconds;
|
||||
turnSettleSeconds = (turnSettleSeconds == null || turnSettleSeconds <= 0) ? 300 : turnSettleSeconds;
|
||||
clearSettleSeconds = (clearSettleSeconds == null || clearSettleSeconds <= 0) ? 20 : clearSettleSeconds;
|
||||
bootstrapText = (bootstrapText == null || bootstrapText.isBlank()) ? null : bootstrapText;
|
||||
}
|
||||
|
||||
@@ -690,7 +690,7 @@ public final class FleetMcp {
|
||||
return null; // AuthorizationMode.UNENFORCED: authorization not enforced (fleetd #518)
|
||||
}
|
||||
if (Authz.permits(caller, action, target)) {
|
||||
if (action != Authz.Action.READ) {
|
||||
if (action != Authz.Action.READ && action != Authz.Action.TASK_READ) {
|
||||
AuditLog.allowed(caller, action, target); // reads would drown the trail
|
||||
}
|
||||
return null;
|
||||
@@ -1035,31 +1035,21 @@ public final class FleetMcp {
|
||||
}
|
||||
|
||||
/**
|
||||
* Which authorization action a {@code fleet_poll} call needs, decided by its arguments
|
||||
* (fleetd #272, widened by fleetd #421).
|
||||
* Which authorization action a {@code fleet_poll} call needs, decided by its arguments.
|
||||
*
|
||||
* <p>{@code fleet_poll} is now <strong>three operations behind one tool name</strong>. With
|
||||
* <p>{@code fleet_poll} is <strong>three operations behind one tool name</strong>. With
|
||||
* {@code ticket} it observes an async delegation and changes nothing, which is a {@link
|
||||
* Authz.Action#READ}. With {@code target} it calls {@link MessageService#drainReplies} on that
|
||||
* session -- the replies are removed from the inbox and a second call returns nothing -- so it
|
||||
* is a {@link Authz.Action#DRAIN}, the same gate {@code fleet_ack} already uses for removing a
|
||||
* single message, and the same one the REST path uses at {@code FleetApp.drainReplies}. With
|
||||
* {@code coordId} it reads (never acks) this daemon's own held lead-to-lead mail, which is a
|
||||
* {@link Authz.Action#COORD_READ} -- <strong>not</strong> {@code READ}, even though nothing is
|
||||
* consumed: {@code READ}'s grant is open to every authenticated role on the premise that the
|
||||
* roster carries no secrets, and a lead-to-lead body is not the roster. Mapping a non-destructive
|
||||
* peer-mail read to {@code READ} would let any worker read every peer lead's mail in full.
|
||||
*
|
||||
* <p>Before this method existed (fleetd #272) the handler passed a constant {@code READ} for
|
||||
* both of the original branches. {@code READ} is open to every authenticated role, so any
|
||||
* worker could read a peer's id out of {@code fleet_list} and destroy the replies that peer had
|
||||
* queued for the primary. The gate failed open, and it did so because the required action is a
|
||||
* function of the arguments while the handler chose it before looking at them.
|
||||
* Authz.Action#TASK_READ}. With {@code target} it calls {@link MessageService#drainReplies} on
|
||||
* that session -- the replies are removed from the inbox and a second call returns nothing --
|
||||
* so it is a {@link Authz.Action#DRAIN}, the same gate {@code fleet_ack} already uses for
|
||||
* removing a single message, and the same one the REST path uses at
|
||||
* {@code FleetApp.drainReplies}. With {@code coordId} it reads (never acks) this daemon's own
|
||||
* held lead-to-lead mail, which is a {@link Authz.Action#COORD_READ} -- <strong>not</strong>
|
||||
* {@code TASK_READ} or {@code READ}: a lead-to-lead body is a different inbox from either, and
|
||||
* folding it into either would let any worker or architect read every peer lead's mail in full.
|
||||
*
|
||||
* <p>The choice lives in this method, and not inline in the handler, so that a test can assert
|
||||
* the mapping the handler actually uses. {@code FleetMcpAuthzTest} already checked every
|
||||
* {@link Authz.Action} against every {@link Role} and passed throughout -- it tested the policy
|
||||
* table, which was correct, while the defect was in which action the caller handed it.
|
||||
* the mapping the handler actually uses.
|
||||
*
|
||||
* <p>Checked first, and exclusively of {@code target}: a call naming {@code coordId} is reading
|
||||
* a different inbox entirely (this daemon's own lead channel, never a worker's), so it takes
|
||||
@@ -1072,7 +1062,30 @@ public final class FleetMcp {
|
||||
if (!isBlank(coordId)) {
|
||||
return Authz.Action.COORD_READ;
|
||||
}
|
||||
return isBlank(target) ? Authz.Action.READ : Authz.Action.DRAIN;
|
||||
return isBlank(target) ? Authz.Action.TASK_READ : Authz.Action.DRAIN;
|
||||
}
|
||||
|
||||
/**
|
||||
* Which authorization action a {@code fleet_send} call needs, decided by its arguments.
|
||||
*
|
||||
* <p>{@code fleet_send} is three call shapes behind one tool name, mirroring {@link
|
||||
* #pollAction}. With {@code coordId} it addresses a peer lead on another daemon over the
|
||||
* coordination broker, which is {@link Authz.Action#COORD_SEND}. With {@code turnId} it
|
||||
* resolves a worker's blocked {@code fleet_ask} and resumes that turn, which is {@link
|
||||
* Authz.Action#ANSWER}. Otherwise it delivers to a local session by {@code sessionId}, which is
|
||||
* the plain {@link Authz.Action#SEND}.
|
||||
*
|
||||
* <p>Checked in the same order the handler branches: {@code coordId} first and exclusively of
|
||||
* {@code turnId}, matching {@link #sendToLead}'s own mutual-exclusion check.
|
||||
*
|
||||
* @param coordId the {@code coordId} argument of the call, or {@code null}/blank when absent
|
||||
* @param turnId the {@code turnId} argument of the call, or {@code null}/blank when absent
|
||||
*/
|
||||
static Authz.Action sendAction(String coordId, String turnId) {
|
||||
if (!isBlank(coordId)) {
|
||||
return Authz.Action.COORD_SEND;
|
||||
}
|
||||
return isBlank(turnId) ? Authz.Action.SEND : Authz.Action.ANSWER;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1097,10 +1110,11 @@ public final class FleetMcp {
|
||||
*/
|
||||
private static Authz.Action authzAction(FleetTool tool, Map<String, Object> arguments) {
|
||||
return switch (tool) {
|
||||
case SEND -> Authz.Action.SEND;
|
||||
case SEND -> sendAction(str(arguments, "coordId"), str(arguments, "turnId"));
|
||||
case REPLY -> Authz.Action.REPLY;
|
||||
case ASK -> Authz.Action.ASK;
|
||||
case STATUS, LIST, PROFILES, WHOAMI -> Authz.Action.READ;
|
||||
case STATUS -> Authz.Action.TASK_READ;
|
||||
case LIST, PROFILES, WHOAMI -> Authz.Action.READ;
|
||||
case POLL -> pollAction(str(arguments, "target"), str(arguments, "coordId"));
|
||||
case ACK -> Authz.Action.DRAIN;
|
||||
case SPAWN -> Authz.Action.SPAWN;
|
||||
|
||||
@@ -49,18 +49,36 @@ import java.util.stream.Collectors;
|
||||
*/
|
||||
public final class FleetApp {
|
||||
|
||||
/** The authorization action the matching route handler hands to {@link #allow}. */
|
||||
/**
|
||||
* The authorization action the matching route handler hands to {@link #allow}, for a route
|
||||
* whose action does not depend on the request body.
|
||||
*/
|
||||
static Authz.Action routeAction(String route) {
|
||||
return routeAction(route, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, plus the one route whose action depends on the body: {@code POST
|
||||
* /sessions/{id}/message} carries a {@code turnId} (the answer-a-blocked-worker shape) or not
|
||||
* (a plain delivery), mirroring {@code FleetMcp#sendAction}'s split of the same two call
|
||||
* shapes over MCP. {@code turnId} is ignored by every other route.
|
||||
*
|
||||
* @param turnId the request body's {@code turnId}, or {@code null}/blank when absent or not
|
||||
* applicable to this route
|
||||
*/
|
||||
static Authz.Action routeAction(String route, String turnId) {
|
||||
return switch (route) {
|
||||
case "GET /metrics" -> Authz.Action.METRICS;
|
||||
case "POST /members" -> Authz.Action.SPAWN;
|
||||
case "DELETE /members/{paneId}" -> Authz.Action.STOP;
|
||||
case "POST /sessions/{id}/message" -> Authz.Action.SEND;
|
||||
case "POST /sessions/{id}/message" -> turnId == null || turnId.isBlank()
|
||||
? Authz.Action.SEND : Authz.Action.ANSWER;
|
||||
case "POST /sessions/{id}/reply" -> Authz.Action.REPLY;
|
||||
case "GET /sessions/{id}/replies" -> Authz.Action.DRAIN;
|
||||
case "POST /sessions/{id}/ask" -> Authz.Action.ASK;
|
||||
case "GET /sessions", "GET /agents", "GET /members", "GET /profiles",
|
||||
"GET /member-credentials", "GET /sessions/{id}/status", "GET /tasks/{ticket}" -> Authz.Action.READ;
|
||||
"GET /member-credentials" -> Authz.Action.READ;
|
||||
case "GET /sessions/{id}/status", "GET /tasks/{ticket}" -> Authz.Action.TASK_READ;
|
||||
default -> throw new IllegalArgumentException("route has no authorization gate: " + route);
|
||||
};
|
||||
}
|
||||
@@ -250,7 +268,8 @@ public final class FleetApp {
|
||||
}
|
||||
Principal caller = ctx.attribute(CALLER);
|
||||
if (Authz.permits(caller, action, target)) {
|
||||
if (action != Authz.Action.READ && action != Authz.Action.METRICS) {
|
||||
if (action != Authz.Action.READ && action != Authz.Action.METRICS
|
||||
&& action != Authz.Action.TASK_READ) {
|
||||
AuditLog.allowed(caller, action, target); // reads would drown the trail
|
||||
}
|
||||
return true;
|
||||
@@ -603,26 +622,33 @@ public final class FleetApp {
|
||||
* status-gated injector and block until the worker returns a structured {@code fleet_reply}.
|
||||
* Times out with a typed 202 (working / queued / busy) rather than an error — the message may
|
||||
* still land.
|
||||
*
|
||||
* <p>Two call shapes share this route, exactly as {@code fleet_send} does over MCP (see
|
||||
* {@code FleetMcp#sendAction}): a plain delivery to {@code id}, and -- when the body carries
|
||||
* {@code turnId} -- resolving a worker's blocked question. The body is parsed before the
|
||||
* authorization check so the right one of {@link Authz.Action#SEND}/{@link Authz.Action#ANSWER}
|
||||
* reaches the gate; a body that fails to parse is treated as the plain shape for that check
|
||||
* alone, and is rejected afterward exactly as before.
|
||||
*/
|
||||
private void sendMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/message"), id)) {
|
||||
JsonNode body;
|
||||
try {
|
||||
body = mapper.readTree(ctx.body());
|
||||
} catch (Exception e) {
|
||||
body = null;
|
||||
}
|
||||
String turnId = body == null ? null : body.path("turnId").asText(null);
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/message", turnId), id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
String turnId;
|
||||
long timeout;
|
||||
boolean wait;
|
||||
try {
|
||||
JsonNode body = mapper.readTree(ctx.body());
|
||||
content = body.path("content").asText("");
|
||||
turnId = body.path("turnId").asText(null);
|
||||
timeout = body.path("timeoutMs").asLong(DEFAULT_MESSAGE_TIMEOUT_MS);
|
||||
wait = body.path("wait").asBoolean(true); // default: block for the reply (CB-104)
|
||||
} catch (Exception e) {
|
||||
if (body == null) {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "body must be JSON"));
|
||||
return;
|
||||
}
|
||||
String content = body.path("content").asText("");
|
||||
long timeout = body.path("timeoutMs").asLong(DEFAULT_MESSAGE_TIMEOUT_MS);
|
||||
boolean wait = body.path("wait").asBoolean(true); // default: block for the reply (CB-104)
|
||||
if (content.isBlank()) {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "content is required"));
|
||||
return;
|
||||
|
||||
@@ -0,0 +1,166 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.auth.Authz;
|
||||
import dev.ltms.fleet.auth.Principal;
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Method;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* Asserts that the {@link FleetMcp} built by {@link FleetdAssembly#assembleAndStart} applies the
|
||||
* authorization table: a worker is refused {@code SPAWN}, and the primary is allowed it.
|
||||
*/
|
||||
class FleetdAssemblyAuthorizationModeTest {
|
||||
|
||||
private static final class TestResourcePorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> new ReplyInbox() {
|
||||
@Override public void own(String target) { }
|
||||
@Override public void release(String target) { }
|
||||
@Override public void publish(String target, String msgId, String content) { }
|
||||
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||
@Override public boolean ack(String target, String msgId) { return false; }
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException(
|
||||
"leadMailboxOpener must not be called — no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Binding a real port would clash with any daemon already listening on it.
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("FakeHerdr is healthy; no poll wait is expected");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private TestResourcePorts ports;
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
if (ports != null && ports.shutdownHook != null) {
|
||||
ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
health:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-test-broker/vh"
|
||||
""");
|
||||
return FleetConfig.load(file);
|
||||
}
|
||||
|
||||
private FleetMcp assemble(Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ports = new TestResourcePorts();
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg,
|
||||
new ConfigRef(dir.resolve("fleetd.yaml"), cfg), new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||
return runtime.mcp();
|
||||
}
|
||||
|
||||
/**
|
||||
* Invokes {@code FleetMcp#denyFor}, which is package-private to {@code dev.ltms.fleet.mcp}
|
||||
* while this test is in {@code dev.ltms.fleet}. Nothing here catches a missing method: if
|
||||
* {@code denyFor} is renamed or removed, {@link NoSuchMethodException} propagates and the
|
||||
* test fails.
|
||||
*/
|
||||
private static McpSchema.CallToolResult denyFor(FleetMcp mcp, Principal caller, Authz.Action action,
|
||||
String target) throws Exception {
|
||||
Method m = FleetMcp.class.getDeclaredMethod("denyFor", Principal.class, Authz.Action.class, String.class);
|
||||
m.setAccessible(true);
|
||||
return (McpSchema.CallToolResult) m.invoke(mcp, caller, action, target);
|
||||
}
|
||||
|
||||
@Test
|
||||
void productionBootPathRefusesAnUnauthorizedCallerThroughTheAssembledFleetMcp(@TempDir Path dir)
|
||||
throws Exception {
|
||||
FleetMcp mcp = assemble(dir);
|
||||
|
||||
McpSchema.CallToolResult deniedForWorker = denyFor(mcp, Principal.worker("term_a", 200),
|
||||
Authz.Action.SPAWN, "term_a");
|
||||
assertNotNull(deniedForWorker,
|
||||
"a worker must not be able to fleet_spawn through the assembled FleetMcp");
|
||||
assertTrue(deniedForWorker.isError(), "a refusal is returned as an MCP tool error");
|
||||
|
||||
McpSchema.CallToolResult allowedForPrimary = denyFor(mcp, Principal.primary(100),
|
||||
Authz.Action.SPAWN, "term_a");
|
||||
assertNull(allowedForPrimary,
|
||||
"control: the primary must still be allowed to fleet_spawn — otherwise the worker "
|
||||
+ "refusal above would pass even with the gate wired backwards");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,190 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.inject.StatusPoller;
|
||||
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||
import dev.ltms.fleet.msg.LeadCoordLoop;
|
||||
import dev.ltms.fleet.msg.LeadMessage;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
|
||||
/**
|
||||
* Asserts that the assembled loops use the production reminder, coordination, and delivery timing
|
||||
* defaults when no {@code primary:} block configures the reply-push values.
|
||||
*/
|
||||
class FleetdAssemblyTimingDefaultsTest {
|
||||
|
||||
private static final class FakeLeadChannel implements LeadChannelHandle {
|
||||
@Override
|
||||
public void publish(String toCoordId, LeadMessage message) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<LeadMessage> peek() {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String msgId) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public String selfCoordId() {
|
||||
return "test-lead";
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean heldDurable() {
|
||||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public MailboxState inspect(String coordId) {
|
||||
return MailboxState.unknown(coordId);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
private static final class TestResourcePorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> new ReplyInbox() {
|
||||
@Override public void own(String target) { }
|
||||
@Override public void release(String target) { }
|
||||
@Override public void publish(String target, String msgId, String content) { }
|
||||
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||
@Override public boolean ack(String target, String msgId) { return false; }
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> new FakeLeadChannel();
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("FakeHerdr is healthy; no poll wait is expected");
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
private TestResourcePorts ports;
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
if (ports != null && ports.shutdownHook != null) {
|
||||
ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
coordinator:
|
||||
uri: "amqp://fake-lead-broker/vh"
|
||||
selfId: "test-lead"
|
||||
""");
|
||||
return FleetConfig.load(file);
|
||||
}
|
||||
|
||||
private FleetdRuntime assemble(Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ports = new TestResourcePorts();
|
||||
return FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg,
|
||||
new ConfigRef(dir.resolve("fleetd.yaml"), cfg), new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||
}
|
||||
|
||||
private static long longField(Object target, String name) throws Exception {
|
||||
Field field = target.getClass().getDeclaredField(name);
|
||||
field.setAccessible(true);
|
||||
return field.getLong(target);
|
||||
}
|
||||
|
||||
@Test
|
||||
void productionBootPathUsesTheExpectedLoopTimingDefaults(@TempDir Path dir) throws Exception {
|
||||
FleetdRuntime runtime = assemble(dir);
|
||||
|
||||
ReplyPushLoop pushLoop = runtime.pushLoop();
|
||||
assertEquals(5, longField(pushLoop, "maxReminders"),
|
||||
"without primary:, ReplyPushLoop must stop after five reminder attempts");
|
||||
assertEquals(15_000L, longField(pushLoop, "backoffMs"),
|
||||
"without primary:, ReplyPushLoop must wait fifteen seconds before the next reminder");
|
||||
|
||||
LeadCoordLoop leadCoordLoop = runtime.leadCoordLoop();
|
||||
assertNotNull(leadCoordLoop, "control: coordinator: must build LeadCoordLoop");
|
||||
assertEquals(3_000L, longField(leadCoordLoop, "intervalMs"),
|
||||
"LeadCoordLoop must poll for peer-lead mail every three seconds");
|
||||
|
||||
StatusPoller poller = runtime.poller();
|
||||
assertEquals(Injector.POLL_INTERVAL_MILLIS, longField(poller, "intervalMillis"),
|
||||
"StatusPoller must use Injector's delivery poll interval");
|
||||
}
|
||||
}
|
||||
@@ -38,6 +38,37 @@ class AuthzTest {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code fleet_send} is three call shapes behind one action name until {@code
|
||||
* FleetMcp#sendAction} picks one: a plain local {@link Authz.Action#SEND}, the {@code coordId}
|
||||
* route ({@link Authz.Action#COORD_SEND}), and the {@code turnId} answer form ({@link
|
||||
* Authz.Action#ANSWER}). All three carry the same grant as the undivided action did — a worker
|
||||
* is excluded from every one, exactly as it was excluded from the one combined action before.
|
||||
*/
|
||||
@Test
|
||||
void theThreeSendShapesCarryTheSameGrantAsTheOldUndividedAction() {
|
||||
for (Authz.Action a : new Authz.Action[]{SEND, COORD_SEND, ANSWER}) {
|
||||
assertTrue(Authz.permits(PRIMARY, a, "term_a"), "the primary may " + a);
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, a, "term_a"), "an architect may " + a);
|
||||
assertFalse(Authz.permits(WORKER_A, a, "term_a"),
|
||||
"a worker performing " + a + " would be escalating into the orchestrator role");
|
||||
assertFalse(Authz.permits(ANON, a, "term_a"));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code fleet_poll{ticket}} and {@code fleet_status} are {@link Authz.Action#TASK_READ}, split
|
||||
* out of the roster-only {@link Authz.Action#READ} (fleetd #678). The grant is unchanged from
|
||||
* what the undivided {@code READ} action gave every one of these callers.
|
||||
*/
|
||||
@Test
|
||||
void taskReadCarriesTheSameGrantReadDidBeforeTheSplit() {
|
||||
assertTrue(Authz.permits(PRIMARY, TASK_READ, null));
|
||||
assertTrue(Authz.permits(WORKER_A, TASK_READ, null));
|
||||
assertTrue(Authz.permits(ARCH_DESIGN, TASK_READ, null));
|
||||
assertFalse(Authz.permits(ANON, TASK_READ, null));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerMayReplyAndAskOnlyAsItself() {
|
||||
assertTrue(Authz.permits(WORKER_A, REPLY, "term_a"));
|
||||
|
||||
@@ -3175,4 +3175,41 @@ class FleetConfigTest {
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertTrue(cfg.models().offIds().isEmpty());
|
||||
}
|
||||
|
||||
// ── fleetd #651: leadRollover.turnSettleSeconds default resolution ─────────────────────────
|
||||
|
||||
@Test
|
||||
void turnSettleSecondsDefaultsTo300WhenUnset(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bare-rollover.yaml");
|
||||
Files.writeString(f, "bind:\n port: 8080\nleadRollover: {}\n");
|
||||
|
||||
FleetConfig.LeadRollover rollover = FleetConfig.load(f).leadRollover();
|
||||
assertNotNull(rollover);
|
||||
assertEquals(300, rollover.turnSettleSeconds());
|
||||
}
|
||||
|
||||
@Test
|
||||
void turnSettleSecondsUsesAnExplicitPositiveValue(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("rollover.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
port: 8080
|
||||
leadRollover:
|
||||
turnSettleSeconds: 45
|
||||
""");
|
||||
|
||||
FleetConfig.LeadRollover rollover = FleetConfig.load(f).leadRollover();
|
||||
assertEquals(45, rollover.turnSettleSeconds());
|
||||
}
|
||||
|
||||
@Test
|
||||
void turnSettleSecondsFallsBackTo300WhenZeroOrNegative(@TempDir Path dir) throws Exception {
|
||||
Path zero = dir.resolve("zero.yaml");
|
||||
Files.writeString(zero, "bind:\n port: 8080\nleadRollover:\n turnSettleSeconds: 0\n");
|
||||
assertEquals(300, FleetConfig.load(zero).leadRollover().turnSettleSeconds());
|
||||
|
||||
Path negative = dir.resolve("negative.yaml");
|
||||
Files.writeString(negative, "bind:\n port: 8080\nleadRollover:\n turnSettleSeconds: -5\n");
|
||||
assertEquals(300, FleetConfig.load(negative).leadRollover().turnSettleSeconds());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -44,11 +44,11 @@ import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
* seventh validator just to exercise the claim.</li>
|
||||
* <li>{@link #validateAllReachesEveryOneOfTodaysRealValidators()} proves {@link
|
||||
* FleetConfig#validateAll()} itself is wired to that same generic mechanism and genuinely
|
||||
* reaches seven of today's eight real validators — reusing the exact minimal failing
|
||||
* configurations {@code FleetConfigTest} already established for each one directly, so a
|
||||
* single call to {@code validateAll()} is shown to reproduce every one of those seven
|
||||
* failures. The eighth, {@link FleetConfig#validateLeadRollover()}, has no case here yet —
|
||||
* a pre-existing gap tracked as fleetd #668.</li>
|
||||
* reaches every one of today's real validators — reusing the exact minimal failing
|
||||
* configurations {@code FleetConfigTest} already established for each one directly, plus a
|
||||
* dedicated fixture for {@link FleetConfig#validateLeadRollover()}, which no other test
|
||||
* drives through {@code validateAll()} — so a single call to {@code validateAll()} is shown
|
||||
* to reproduce every one of those failures.</li>
|
||||
* </ol>
|
||||
*
|
||||
* <p>Together with the direct-{@code Fleetd.main}-invocation tests in {@code
|
||||
@@ -58,10 +58,10 @@ import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
* a test in this module.
|
||||
*
|
||||
* <p><b>What is NOT pinned, measured rather than assumed.</b> Reverting {@link
|
||||
* FleetConfig#validateAll()} to a hardcoded list of today's six method calls leaves the whole
|
||||
* suite green (measured at review: 1491 tests, 0 failures). Nothing ties {@code validateAll()} to
|
||||
* FleetConfig#validateAll()} to a hardcoded list of today's method calls leaves the whole
|
||||
* suite green. Nothing ties {@code validateAll()} to
|
||||
* the generic sweep — claim 1 proves {@link FleetConfig#invokeAllValidators} is generic, and claim
|
||||
* 2 proves {@code validateAll()} reaches today's six, and a hardcoded list satisfies both. So the
|
||||
* 2 proves {@code validateAll()} reaches today's validators, and a hardcoded list satisfies both. So the
|
||||
* reflective sweep is a convenience, not the guarantee. The guarantee is {@link
|
||||
* #fleetConfigDeclaresExactlyTheseValidatorsToday()}: it fails the moment any validator is added
|
||||
* or removed, which forces whoever changes the set to look at this file.
|
||||
@@ -210,7 +210,7 @@ class FleetConfigValidateAllTest {
|
||||
+ "name) must all be skipped");
|
||||
}
|
||||
|
||||
// ── Claim 2: FleetConfig.validateAll() is wired to that mechanism and reaches seven of eight today ──
|
||||
// ── Claim 2: FleetConfig.validateAll() is wired to that mechanism and reaches every validator ──
|
||||
|
||||
/**
|
||||
* Reflectively enumerates {@link FleetConfig}'s own public, no-arg, void {@code validateXxx()}
|
||||
@@ -220,6 +220,11 @@ class FleetConfigValidateAllTest {
|
||||
* the {@code Set.of} below, so a reader adding or removing one sees this assertion name the new
|
||||
* count rather than a silent pass at the old one. The count lives only in that set, not in this
|
||||
* method's name, so the two cannot drift apart.
|
||||
*
|
||||
* <p>This assertion alone proves only that the validator exists with the right shape — it
|
||||
* cannot prove {@code validateAll()} actually reaches it. Only {@link
|
||||
* #validateAllReachesEveryOneOfTodaysRealValidators()} proves reachability, which is why this
|
||||
* method's failure message sends the reader there too.
|
||||
*/
|
||||
@Test
|
||||
void fleetConfigDeclaresExactlyTheseValidatorsToday() {
|
||||
@@ -237,11 +242,16 @@ class FleetConfigValidateAllTest {
|
||||
"validateSubscriptionProfiles", "validateCharters", "validateMembers",
|
||||
"validateModels", "validateLeadRollover", "validatePanePlacementAgainstLeadTabs")),
|
||||
names,
|
||||
"FleetConfig's public validate*() methods changed. Do TWO things, in this "
|
||||
"FleetConfig's public validate*() methods changed. Do THREE things, in this "
|
||||
+ "order. First confirm validateAll() still delegates to "
|
||||
+ "invokeAllValidators(this) — a hardcoded list there passes every other "
|
||||
+ "test in this class, so this assertion is the only place that will ever "
|
||||
+ "make you check. Only then update the expected set to match.");
|
||||
+ "make you check. Second, update the expected set below to match. Third, "
|
||||
+ "add or remove a case for that validator in "
|
||||
+ "validateAllReachesEveryOneOfTodaysRealValidators() below — this "
|
||||
+ "assertion proves only that the validator exists with the right shape, "
|
||||
+ "never that validateAll() reaches it; that enumeration is the test that "
|
||||
+ "does.");
|
||||
}
|
||||
|
||||
/** A minimal, otherwise-valid file — same shape FleetConfigTest and ConfigRefTest use. */
|
||||
@@ -265,14 +275,18 @@ class FleetConfigValidateAllTest {
|
||||
}
|
||||
|
||||
/**
|
||||
* The heart of claim 2: for seven of today's eight real validators, a minimal file that fails
|
||||
* ONLY that one — the exact fixtures {@code FleetConfigTest} uses to test each validator
|
||||
* directly — must also fail through {@link FleetConfig#validateAll()}. If a future edit to
|
||||
* {@code validateAll()} silently dropped one of these seven from the sweep (e.g. a typo'd name
|
||||
* filter), exactly one of them would start passing when it must not.
|
||||
* The heart of claim 2: for every one of today's real validators, a minimal file that
|
||||
* fails ONLY that one — the exact fixtures {@code FleetConfigTest} uses to test each validator
|
||||
* directly, or a dedicated minimal fixture where no other test drives that validator through
|
||||
* {@code validateAll()} — must also fail through {@link FleetConfig#validateAll()}. If a
|
||||
* future edit to {@code validateAll()} silently dropped one of these from the sweep
|
||||
* (e.g. a typo'd name filter), exactly one of them would start passing when it must not.
|
||||
*
|
||||
* <p>The eighth, {@link FleetConfig#validateLeadRollover()}, has no case here — a pre-existing
|
||||
* gap tracked as fleetd #668, not fixed by this change.
|
||||
* <p>This is the single place that proves {@code validateAll()} reaches a given validator.
|
||||
* Adding or removing a validator on {@link FleetConfig} must add or remove a case here, not
|
||||
* only an updated name in {@link #fleetConfigDeclaresExactlyTheseValidatorsToday()}'s expected
|
||||
* set — that assertion proves the validator's shape, never that {@code validateAll()} reaches
|
||||
* it.
|
||||
*/
|
||||
@Test
|
||||
void validateAllReachesEveryOneOfTodaysRealValidators(@TempDir Path dir) throws Exception {
|
||||
@@ -362,6 +376,15 @@ class FleetConfigValidateAllTest {
|
||||
opus:
|
||||
tab: "lead: opus"
|
||||
""", "gx10");
|
||||
|
||||
// validateLeadRollover: a leadRollover: block present with no handoverPath.
|
||||
assertValidateAllRefuses(dir, "lead-rollover.yaml", """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
leadRollover:
|
||||
requireOperatorConfirm: false
|
||||
""", "handoverPath");
|
||||
}
|
||||
|
||||
private static void assertValidateAllRefuses(Path dir, String fileName, String yaml,
|
||||
|
||||
@@ -156,6 +156,45 @@ class FleetMcpAuthzTest {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #669 Unit A: {@code SEND} is split into three actions ({@link Authz.Action#SEND},
|
||||
* {@link Authz.Action#COORD_SEND}, {@link Authz.Action#ANSWER}), each carrying the same grant
|
||||
* the one undivided action gave. An architect holds all three, exactly as it held the one.
|
||||
*/
|
||||
@Test
|
||||
void anArchitectMayUseAllThreeSendShapesOverMcp() {
|
||||
FleetMcp m = mcp(true);
|
||||
for (Authz.Action a : new Authz.Action[]{Authz.Action.SEND, Authz.Action.COORD_SEND,
|
||||
Authz.Action.ANSWER}) {
|
||||
assertNull(m.denyFor(ARCH_DESIGN, a, "term_a"),
|
||||
a + " carries the same grant the undivided SEND action gave an architect");
|
||||
}
|
||||
}
|
||||
|
||||
/** The other half of the same split: a worker is excluded from all three, as it was from one. */
|
||||
@Test
|
||||
void aWorkerMayNotUseAnySendShapeOverMcp() {
|
||||
FleetMcp m = mcp(true);
|
||||
for (Authz.Action a : new Authz.Action[]{Authz.Action.SEND, Authz.Action.COORD_SEND,
|
||||
Authz.Action.ANSWER}) {
|
||||
McpSchema.CallToolResult denied = m.denyFor(WORKER_A, a, "term_a");
|
||||
assertNotNull(denied, a + " must stay refused to a worker");
|
||||
assertTrue(denied.isError(), "a refusal is returned as an MCP tool error");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #669 Unit A / #678: {@code TASK_READ} (ticket polling, session status) is split out of
|
||||
* the roster-only {@code READ}, carrying forward the grant the undivided action gave. A worker
|
||||
* still has both — it never gained or lost anything by the split.
|
||||
*/
|
||||
@Test
|
||||
void aWorkerKeepsBothReadActionsAfterTheSplit() {
|
||||
FleetMcp m = mcp(true);
|
||||
assertNull(m.denyFor(WORKER_A, Authz.Action.READ, null));
|
||||
assertNull(m.denyFor(WORKER_A, Authz.Action.TASK_READ, null));
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectMayReplyAndAskOnlyAsItsOwnPaneOverMcp() {
|
||||
FleetMcp m = mcp(true);
|
||||
@@ -297,35 +336,53 @@ class FleetMcpAuthzTest {
|
||||
* the whole time the defect was live -- the table was right, the action fed to it was wrong.
|
||||
*/
|
||||
@Test
|
||||
void pollingByTargetIsADrainAndPollingByTicketIsARead() {
|
||||
void pollingByTargetIsADrainAndPollingByTicketIsATaskRead() {
|
||||
assertEquals(Authz.Action.DRAIN, FleetMcp.pollAction("term_b", null),
|
||||
"poll by target removes the replies — that is a drain, not an observation");
|
||||
assertEquals(Authz.Action.READ, FleetMcp.pollAction(null, null),
|
||||
"poll by ticket changes nothing");
|
||||
assertEquals(Authz.Action.READ, FleetMcp.pollAction(" ", null),
|
||||
assertEquals(Authz.Action.TASK_READ, FleetMcp.pollAction(null, null),
|
||||
"poll by ticket changes nothing, but is not the roster-only READ action");
|
||||
assertEquals(Authz.Action.TASK_READ, FleetMcp.pollAction(" ", null),
|
||||
"a blank target is an absent target");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #421: a coordId branch is a THIRD operation behind fleet_poll's one name, and it must
|
||||
* map to {@link Authz.Action#COORD_READ} — never {@link Authz.Action#READ}, even though this
|
||||
* branch also consumes nothing. READ's grant is open to every authenticated role on the premise
|
||||
* that the roster carries no secrets; a lead-to-lead body is not the roster, so folding this
|
||||
* branch into READ would let any worker read every peer lead's mail in full. coordId also takes
|
||||
* priority over target when both happen to be present — it addresses a different inbox entirely.
|
||||
* map to {@link Authz.Action#COORD_READ} — never {@link Authz.Action#READ} or {@link
|
||||
* Authz.Action#TASK_READ}, even though this branch also consumes nothing. A lead-to-lead body
|
||||
* is not the roster and not a ticket/status read, so folding this branch into either would let
|
||||
* any worker or architect read every peer lead's mail in full. coordId also takes priority over
|
||||
* target when both happen to be present — it addresses a different inbox entirely.
|
||||
*/
|
||||
@Test
|
||||
void pollingByCoordIdIsACoordReadNeverAPlainRead() {
|
||||
void pollingByCoordIdIsACoordReadNeverAPlainOrTaskRead() {
|
||||
assertEquals(Authz.Action.COORD_READ, FleetMcp.pollAction(null, "mac-opus"),
|
||||
"reading held peer mail must not be mapped to the everyone-readable READ action");
|
||||
"reading held peer mail must not be mapped to a widely-readable action");
|
||||
assertEquals(Authz.Action.COORD_READ, FleetMcp.pollAction(" ", "mac-opus"),
|
||||
"a blank target must not fall through to READ/DRAIN when coordId is present");
|
||||
assertEquals(Authz.Action.READ, FleetMcp.pollAction(null, " "),
|
||||
assertEquals(Authz.Action.TASK_READ, FleetMcp.pollAction(null, " "),
|
||||
"a blank coordId is an absent coordId, same as target/ticket");
|
||||
assertEquals(Authz.Action.COORD_READ, FleetMcp.pollAction("term_b", "mac-opus"),
|
||||
"coordId takes priority over target — this is a different inbox, not a drain");
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code fleet_send} is three call shapes behind one tool name, exactly as {@code fleet_poll}
|
||||
* is (fleetd #669 Unit A). {@link FleetMcp#sendAction} picks the action from the arguments, not
|
||||
* the handler, for the same reason {@link FleetMcp#pollAction} does: a test can assert the
|
||||
* mapping the handler actually uses.
|
||||
*/
|
||||
@Test
|
||||
void sendMapsToThreeDifferentActionsByItsArguments() {
|
||||
assertEquals(Authz.Action.SEND, FleetMcp.sendAction(null, null),
|
||||
"a plain delivery, with neither coordId nor turnId, is a local SEND");
|
||||
assertEquals(Authz.Action.COORD_SEND, FleetMcp.sendAction("mac-opus", null),
|
||||
"coordId addresses a peer lead over the coordination broker");
|
||||
assertEquals(Authz.Action.ANSWER, FleetMcp.sendAction(null, "turn-1"),
|
||||
"turnId resolves a worker's blocked question");
|
||||
assertEquals(Authz.Action.COORD_SEND, FleetMcp.sendAction("mac-opus", "turn-1"),
|
||||
"coordId takes priority over turnId, mirroring sendToLead's own mutual-exclusion check");
|
||||
}
|
||||
|
||||
@Test
|
||||
void everyRegisteredToolHasItsHandlerActionPinned() {
|
||||
// fleetd #469: this used to scrape FleetMcp.java's tool("…") calls for the registered set —
|
||||
@@ -344,16 +401,22 @@ class FleetMcpAuthzTest {
|
||||
() -> tool + " is registered but has no pinned authorization action"));
|
||||
|
||||
assertEquals(Authz.Action.SEND, FleetMcp.toolAction("fleet_send", Map.of()));
|
||||
assertEquals(Authz.Action.SEND,
|
||||
FleetMcp.toolAction("fleet_send", Map.of("sessionId", "term_a", "content", "hi")));
|
||||
assertEquals(Authz.Action.COORD_SEND,
|
||||
FleetMcp.toolAction("fleet_send", Map.of("coordId", "mac-opus", "content", "hi")));
|
||||
assertEquals(Authz.Action.ANSWER,
|
||||
FleetMcp.toolAction("fleet_send", Map.of("turnId", "turn-1", "content", "hi")));
|
||||
assertEquals(Authz.Action.REPLY, FleetMcp.toolAction("fleet_reply", Map.of()));
|
||||
assertEquals(Authz.Action.ASK, FleetMcp.toolAction("fleet_ask", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_status", Map.of()));
|
||||
assertEquals(Authz.Action.TASK_READ, FleetMcp.toolAction("fleet_status", Map.of()));
|
||||
assertEquals(Authz.Action.DRAIN, FleetMcp.toolAction("fleet_ack", Map.of()));
|
||||
assertEquals(Authz.Action.SPAWN, FleetMcp.toolAction("fleet_spawn", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_list", Map.of()));
|
||||
assertEquals(Authz.Action.STOP, FleetMcp.toolAction("fleet_stop", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_profiles", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_whoami", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_poll", Map.of("ticket", "task")));
|
||||
assertEquals(Authz.Action.TASK_READ, FleetMcp.toolAction("fleet_poll", Map.of("ticket", "task")));
|
||||
assertEquals(Authz.Action.DRAIN, FleetMcp.toolAction("fleet_poll", Map.of("target", "term_b")));
|
||||
assertEquals(Authz.Action.COORD_READ,
|
||||
FleetMcp.toolAction("fleet_poll", Map.of("coordId", "mac-opus")));
|
||||
|
||||
@@ -60,13 +60,25 @@ class MessageServiceTest {
|
||||
inbox.own(T);
|
||||
}
|
||||
|
||||
/**
|
||||
* A send budget large enough that a test's own setup — {@link #awaitWaiting()} plus whatever
|
||||
* status transitions it drives afterward — can never compete with it for the same clock. A test
|
||||
* that needs {@code send.get(...)}'s own window to be the only timing bound it depends on uses
|
||||
* {@link #sendAsync(String, long)} with this value instead of the default 5000 ms.
|
||||
*/
|
||||
private static final long GENEROUS_SEND_BUDGET_MILLIS = 30_000;
|
||||
|
||||
/** Run {@code send} on a background thread; the current thread drives the worker's turn. */
|
||||
private CompletableFuture<MessageService.Reply> sendAsync() {
|
||||
return sendAsync("do the task");
|
||||
}
|
||||
|
||||
private CompletableFuture<MessageService.Reply> sendAsync(String content) {
|
||||
return CompletableFuture.supplyAsync(() -> messages.send(T, content, 5000));
|
||||
return sendAsync(content, 5000);
|
||||
}
|
||||
|
||||
private CompletableFuture<MessageService.Reply> sendAsync(String content, long timeoutMillis) {
|
||||
return CompletableFuture.supplyAsync(() -> messages.send(T, content, timeoutMillis));
|
||||
}
|
||||
|
||||
private void awaitWaiting() throws InterruptedException {
|
||||
@@ -80,7 +92,7 @@ class MessageServiceTest {
|
||||
|
||||
@Test
|
||||
void completionFallbackResolvesATurnThatNeverCalledFleetReply() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync("do the task", GENEROUS_SEND_BUDGET_MILLIS);
|
||||
awaitWaiting();
|
||||
|
||||
herdr.readText("$ prompt"); // pre-turn pane: no answer yet (baseline reference)
|
||||
@@ -96,6 +108,32 @@ class MessageServiceTest {
|
||||
assertTrue(reply.completed(), "a scraped completion still counts as completed");
|
||||
}
|
||||
|
||||
/**
|
||||
* Pins {@link #GENEROUS_SEND_BUDGET_MILLIS} as the budget {@link
|
||||
* #completionFallbackResolvesATurnThatNeverCalledFleetReply} depends on. A 5500 ms delay between
|
||||
* {@link #awaitWaiting()} and the status transitions that drive completion stands in for a loaded
|
||||
* machine's setup overhead — comfortably past the 5000 ms budget this send no longer uses, and
|
||||
* still well inside this method's own 30 000 ms budget. The only clock this test depends on is
|
||||
* {@code send.get}'s own 10 s window.
|
||||
*/
|
||||
@Test
|
||||
void completionFallbackSurvivesASlowHarnessBecauseItsSendBudgetIsNotTheBindingClock() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync("do the task", GENEROUS_SEND_BUDGET_MILLIS);
|
||||
awaitWaiting();
|
||||
|
||||
Thread.sleep(5500);
|
||||
|
||||
herdr.readText("$ prompt");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
herdr.readText("BUILD GREEN: 391 files");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
|
||||
MessageService.Reply reply = send.get(10, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.COMPLETED_UNREPLIED, reply.outcome(),
|
||||
"a slow harness must not be mistaken for a timed-out delivery");
|
||||
}
|
||||
|
||||
@Test
|
||||
void completionFallbackReplacesAnEchoedInjectedBriefWithNoReportOutcome() throws Exception {
|
||||
String brief = "Implement the requested change. ".repeat(20);
|
||||
|
||||
@@ -122,12 +122,28 @@ class FleetAppAuthTest {
|
||||
assertEquals(Authz.Action.DRAIN, FleetApp.routeAction("GET /sessions/{id}/replies"));
|
||||
assertEquals(Authz.Action.ASK, FleetApp.routeAction("POST /sessions/{id}/ask"));
|
||||
for (String route : Set.of("GET /sessions", "GET /agents", "GET /members", "GET /profiles",
|
||||
"GET /member-credentials", "GET /sessions/{id}/status", "GET /tasks/{ticket}")) {
|
||||
"GET /member-credentials")) {
|
||||
assertEquals(Authz.Action.READ, FleetApp.routeAction(route), route);
|
||||
}
|
||||
for (String route : Set.of("GET /sessions/{id}/status", "GET /tasks/{ticket}")) {
|
||||
assertEquals(Authz.Action.TASK_READ, FleetApp.routeAction(route), route);
|
||||
}
|
||||
assertThrows(IllegalArgumentException.class, () -> FleetApp.routeAction("GET /healthz"));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #669 Unit A: {@code POST /sessions/{id}/message} is two call shapes behind one route,
|
||||
* mirroring {@code fleet_send}'s MCP-side split into {@link Authz.Action#SEND} and {@link
|
||||
* Authz.Action#ANSWER} ({@code FleetMcp#sendAction}). The route never carries a {@code coordId}
|
||||
* shape — that peer-lead route is MCP-only — so only these two apply here.
|
||||
*/
|
||||
@Test
|
||||
void theMessageRouteIsASendWithNoTurnIdAndAnAnswerWithOne() {
|
||||
assertEquals(Authz.Action.SEND, FleetApp.routeAction("POST /sessions/{id}/message", null));
|
||||
assertEquals(Authz.Action.SEND, FleetApp.routeAction("POST /sessions/{id}/message", " "));
|
||||
assertEquals(Authz.Action.ANSWER, FleetApp.routeAction("POST /sessions/{id}/message", "turn-1"));
|
||||
}
|
||||
|
||||
private static Set<String> routesTheServerRegisters() {
|
||||
try {
|
||||
String source = Files.readString(REST_SOURCE).lines()
|
||||
@@ -192,6 +208,21 @@ class FleetAppAuthTest {
|
||||
"draining an inbox is the primary's collection step");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #669 Unit A: the {@code turnId} shape of {@code POST /sessions/{id}/message} maps to
|
||||
* {@link Authz.Action#ANSWER}, not the plain {@link Authz.Action#SEND} the test above drives —
|
||||
* a worker must stay refused on this shape too, exactly as it was refused on the one undivided
|
||||
* action before the split.
|
||||
*/
|
||||
@Test
|
||||
void aWorkerMayNotAnswerAnotherSessionsBlockedQuestionOverRest() throws Exception {
|
||||
int port = start(FakeHerdr.WORKER_PID, false, null);
|
||||
|
||||
assertEquals(403, send(port, "POST", "/sessions/term_b/message",
|
||||
"{\"turnId\":\"turn-1\",\"content\":\"hi\"}", null).statusCode(),
|
||||
"resolving another session's blocked question would be a worker escalating too");
|
||||
}
|
||||
|
||||
// --- token mode ---------------------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
|
||||
+10
-4
@@ -580,6 +580,12 @@ install_candidate() {
|
||||
|
||||
# -------------------------------------------------------------------------------- the report path
|
||||
#
|
||||
# Masks basic-auth userinfo (scheme://user:pass@host) in a daemon verdict line before it reaches
|
||||
# the terminal.
|
||||
mask_verdict_userinfo() {
|
||||
printf '%s\n' "$1" | sed -E 's#://[^@/[:space:]]*@#://<redacted>@#g'
|
||||
}
|
||||
|
||||
# Prints the literal command the operator (or a test) can run to restore the backup by hand — the
|
||||
# absolute path to THIS script plus the overrides actually in force, so it works from any cwd.
|
||||
restore_command_line() {
|
||||
@@ -599,8 +605,8 @@ restore_and_confirm() {
|
||||
ok "restored from $backup"
|
||||
if wait_for_verdict "$LOG" "$mark2" "$WAIT_SECONDS"; then
|
||||
case "$VERDICT_KIND" in
|
||||
refused) warn "the RESTORE was also refused by the daemon: $VERDICT_LINE" ;;
|
||||
*) ok "restore confirmed: $VERDICT_LINE" ;;
|
||||
refused) warn "the RESTORE was also refused by the daemon: $(mask_verdict_userinfo "$VERDICT_LINE")" ;;
|
||||
*) ok "restore confirmed: $(mask_verdict_userinfo "$VERDICT_LINE")" ;;
|
||||
esac
|
||||
else
|
||||
warn "the restore is on disk, but no confirming verdict line appeared within ${WAIT_SECONDS}s"
|
||||
@@ -616,7 +622,7 @@ report_outcome() {
|
||||
|
||||
say "waiting for the daemon's verdict (up to ${WAIT_SECONDS}s)"
|
||||
if wait_for_verdict "$LOG" "$mark" "$WAIT_SECONDS"; then
|
||||
kind="$VERDICT_KIND"; line="$VERDICT_LINE"
|
||||
kind="$VERDICT_KIND"; line="$(mask_verdict_userinfo "$VERDICT_LINE")"
|
||||
else
|
||||
kind="none"
|
||||
fi
|
||||
@@ -673,7 +679,7 @@ check_mode() {
|
||||
local verdict
|
||||
verdict="$(last_verdict_line "$LOG")"
|
||||
if [ -n "$verdict" ]; then
|
||||
ok "last verdict in log: $verdict"
|
||||
ok "last verdict in log: $(mask_verdict_userinfo "$verdict")"
|
||||
else
|
||||
warn "no reload verdict line found in $LOG"
|
||||
fi
|
||||
|
||||
@@ -81,11 +81,12 @@ MODULE="$REPO/fleetd"
|
||||
BUILD_JAR="$MODULE/target/fleetd.jar"
|
||||
JAR="$MODULE/run/fleetd.jar"
|
||||
OUT="$MODULE/fleetd.out"
|
||||
# Matches BOTH the absolute form and the relative `java -jar run/fleetd.jar` a hand-start
|
||||
# produces from inside fleetd/. Anchoring on the absolute path alone was a real bug: the daemon
|
||||
# restarted correctly and the script still reported "no process appeared", because it launched with
|
||||
# a relative path and then looked for an absolute one.
|
||||
PATTERN='run/fleetd.jar'
|
||||
# Matches a fleetd daemon's command line wherever its jar sits — absolute or relative, under
|
||||
# run/, under target/, or anywhere else a build or a hand-start might point it. Detecting a
|
||||
# daemon this script did not start, including one running from a jar outside $JAR's own
|
||||
# directory, is this pattern's whole job; running_pid()'s `comm = java` allowlist below is what
|
||||
# keeps that breadth from counting a shell that merely types the pattern as literal text.
|
||||
PATTERN='fleetd.jar'
|
||||
HEALTH='http://127.0.0.1:8765/healthz'
|
||||
STOP_WAIT=30 # seconds to wait for a clean exit before reporting failure
|
||||
HEALTH_WAIT=60 # seconds to wait for /healthz to answer after start — fleetd #603: also the pid-
|
||||
@@ -231,7 +232,7 @@ report_jar_state() {
|
||||
# launched as `java -jar ...` — a native image, a renamed launcher — `running_pid()` silently
|
||||
# returns nothing and `assert_single_daemon` stops noticing a second daemon at all. For a guard,
|
||||
# that false-negative direction is the worse one to be wrong in. This is not a new assumption,
|
||||
# though: `PATTERN='run/fleetd.jar'` two lines up already assumes the daemon is a jar, which
|
||||
# though: `PATTERN='fleetd.jar'` two lines up already assumes the daemon is a jar, which
|
||||
# is only ever run by `java`. If that launch method changes, `PATTERN` stops matching anything
|
||||
# before this allowlist would ever get the chance to be wrong — the allowlist rides on the same
|
||||
# assumption that is already load-bearing, it does not add a new one. Whoever changes the launch
|
||||
@@ -627,6 +628,45 @@ check_log_path_matches_plist() {
|
||||
ok "log path check: script and plist agree ($resolved_out)"
|
||||
}
|
||||
|
||||
# Reads the launchd plist's ProgramArguments for the argument that follows "-jar", resolves it
|
||||
# alongside $jar_path, and dies when the two differ. Call it only when the agent is loaded; it
|
||||
# never touches launchd or the daemon itself.
|
||||
check_jar_path_matches_plist() {
|
||||
local jar_path="$1" plist_path="$2"
|
||||
local plist_args plist_jar resolved_jar resolved_plist_jar
|
||||
if ! plist_args="$(/usr/libexec/PlistBuddy -c 'Print :ProgramArguments' "$plist_path" 2>/dev/null)"; then
|
||||
die "launchd agent is loaded but PlistBuddy could not read ProgramArguments from
|
||||
$plist_path
|
||||
— cannot verify which jar the supervised daemon launches. Fix the plist before
|
||||
redeploying supervised."
|
||||
fi
|
||||
plist_jar="$(printf '%s\n' "$plist_args" | awk '
|
||||
{ gsub(/^[ \t]+|[ \t]+$/, "") }
|
||||
prev == "-jar" { print; exit }
|
||||
{ prev = $0 }
|
||||
')"
|
||||
if [ -z "$plist_jar" ]; then
|
||||
die "launchd agent is loaded but its ProgramArguments at
|
||||
$plist_path
|
||||
do not contain a '-jar <path>' pair — cannot verify which jar the supervised daemon
|
||||
launches. Fix the plist before redeploying supervised."
|
||||
fi
|
||||
resolved_jar="$(cd "$(dirname "$jar_path")" 2>/dev/null && pwd -P)/$(basename "$jar_path")" || true
|
||||
resolved_plist_jar="$(cd "$(dirname "$plist_jar")" 2>/dev/null && pwd -P)/$(basename "$plist_jar")" || true
|
||||
if [ -z "$resolved_jar" ] || [ -z "$resolved_plist_jar" ] || [ "$resolved_jar" != "$resolved_plist_jar" ]; then
|
||||
die "jar path mismatch — this script deploys to
|
||||
$jar_path (resolved: ${resolved_jar:-<directory does not exist>})
|
||||
but the loaded plist's ProgramArguments names
|
||||
$plist_jar (resolved: ${resolved_plist_jar:-<directory does not exist>})
|
||||
The swap renames the built jar into place, so the old path stops existing after a redeploy;
|
||||
a launchd-initiated start from this plist (a reboot, or KeepAlive after a crash) would then
|
||||
run java against a missing file. Reinstall the plist at
|
||||
$plist_path
|
||||
so its ProgramArguments names $jar_path before redeploying supervised."
|
||||
fi
|
||||
ok "jar path check: script and plist agree ($resolved_jar)"
|
||||
}
|
||||
|
||||
# fleetd #552: the post-restart fresh-log capture, pulled out of the main flow so it is testable by
|
||||
# sourcing (the same reason systemd_installed/systemd_loaded above guard their OWN mktemp inline
|
||||
# instead of leaving it bare) even though its only caller sits below the SOURCED guard. By the time
|
||||
@@ -960,6 +1000,7 @@ report_supervisor_state() {
|
||||
SUPERVISED=1
|
||||
ok "launchd agent loaded ($LAUNCHD_LABEL) — launchd supervises this daemon"
|
||||
check_log_path_matches_plist "$OUT" "$LAUNCHD_PLIST"
|
||||
check_jar_path_matches_plist "$JAR" "$LAUNCHD_PLIST"
|
||||
;;
|
||||
systemd)
|
||||
SUPERVISED=1
|
||||
|
||||
@@ -629,6 +629,65 @@ test_refusal_shape_from_parse_failure_wording_is_recognised() {
|
||||
assert_equals 4 "$RUN_RC" "the parse-failure refusal shape must also exit 4, not be read as silence"
|
||||
}
|
||||
|
||||
# A verdict line carrying a credentialed URI has its userinfo masked, with a positive control
|
||||
# proving the rest of the line still reaches the output unchanged.
|
||||
test_verdict_userinfo_is_masked_with_positive_control() {
|
||||
local dir
|
||||
dir="$(new_fixture)"
|
||||
|
||||
start_run "$dir" 5 --set '.profiles.sonnet.weight=4'
|
||||
sleep 1
|
||||
printf 'config reload from %s refused, keeping the running config: refusing to start: malformed pattern — profiles.local.errorPattern ("amqp://user:hunter2@host/vhost"): Unclosed character class near index 8\n' \
|
||||
"$dir/fleetd.yaml" >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
|
||||
assert_equals 4 "$RUN_RC" "refusal-with-userinfo exit code"
|
||||
assert_not_contains "user:hunter2" "$RUN_OUTPUT" "the userinfo must never reach the output"
|
||||
assert_contains "amqp://<redacted>@host/vhost" "$RUN_OUTPUT" \
|
||||
"the userinfo must be MASKED, not deleted — the rest of the quoted value must survive"
|
||||
# Positive control: the diagnostic prose on both sides of the userinfo must still reach the
|
||||
# output. Without this, a mutant that drops the whole verdict line would pass identically.
|
||||
assert_contains "malformed pattern" "$RUN_OUTPUT" "prose BEFORE the userinfo must still reach the output"
|
||||
assert_contains "Unclosed character class near index 8" "$RUN_OUTPUT" \
|
||||
"prose AFTER the userinfo must still reach the output"
|
||||
}
|
||||
|
||||
# An ordinary refusal line quotes the offending pattern, not a credential, and must survive byte
|
||||
# for byte: the rewrite is scoped to userinfo only, and the quoted pattern is the detail an
|
||||
# operator needs to fix the refusal.
|
||||
test_ordinary_refusal_line_passes_through_unchanged() {
|
||||
local dir real_line
|
||||
dir="$(new_fixture)"
|
||||
real_line="config reload from $dir/fleetd.yaml refused, keeping the running config: refusing to start: malformed pattern — profiles.local.errorPattern (\"[unclosed\"): Unclosed character class near index 8"
|
||||
|
||||
start_run "$dir" 5 --set '.profiles.sonnet.weight=4'
|
||||
sleep 1
|
||||
printf '%s\n' "$real_line" >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
|
||||
assert_equals 4 "$RUN_RC" "ordinary refusal exit code"
|
||||
assert_contains "$real_line" "$RUN_OUTPUT" \
|
||||
"an ordinary refusal with no userinfo must pass through byte for byte, unchanged"
|
||||
}
|
||||
|
||||
# A verdict line can hold a URI with NO userinfo and a later, unrelated @ further on in the same
|
||||
# line (an email address in diagnostic prose, for example). The rewrite must stop at the end of
|
||||
# the URI and must not treat the later @ as a second userinfo delimiter.
|
||||
test_uri_without_userinfo_survives_a_later_at_sign() {
|
||||
local dir real_line
|
||||
dir="$(new_fixture)"
|
||||
real_line="config reload from $dir/fleetd.yaml refused, keeping the running config: broker.uri amqp://broker.local/vhost unreachable, contact ops@example.com"
|
||||
|
||||
start_run "$dir" 5 --set '.profiles.sonnet.weight=4'
|
||||
sleep 1
|
||||
printf '%s\n' "$real_line" >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
|
||||
assert_equals 4 "$RUN_RC" "no-userinfo-with-later-at-sign exit code"
|
||||
assert_contains "$real_line" "$RUN_OUTPUT" \
|
||||
"a URI with no userinfo plus a later @ in the same line must pass through byte for byte"
|
||||
}
|
||||
|
||||
# --set runs yq over the whole candidate. It warns when that changes more lines than the requested
|
||||
# pairs, but a simple file with only the intended changed line must stay quiet.
|
||||
new_fixture_reformat_sensitive() {
|
||||
@@ -720,6 +779,12 @@ echo "== extra: --check is read-only and always exits 0 =="
|
||||
test_check_is_read_only_and_exits_zero
|
||||
echo "== extra: the parse-failure refusal shape is also recognised =="
|
||||
test_refusal_shape_from_parse_failure_wording_is_recognised
|
||||
echo "== verdict-redaction criteria 2+3: verdict userinfo is masked, rest of line survives =="
|
||||
test_verdict_userinfo_is_masked_with_positive_control
|
||||
echo "== verdict-redaction criterion 4: an ordinary refusal passes through unchanged =="
|
||||
test_ordinary_refusal_line_passes_through_unchanged
|
||||
echo "== fleetd #638: a URI with no userinfo survives a later @ in the same line =="
|
||||
test_uri_without_userinfo_survives_a_later_at_sign
|
||||
echo "== acceptance criterion 17: --set warns about yq formatting churn =="
|
||||
test_set_warns_when_yq_reformats_extra_lines
|
||||
echo "== acceptance criterion 18: --set stays quiet without formatting churn =="
|
||||
|
||||
@@ -492,6 +492,38 @@ test_running_pid_finds_a_real_java_named_second_process() {
|
||||
|| fail "running_pid() did not find a real second process (pid $standin_pid, comm forced to 'java' via exec -a) whose own argv holds the pattern: before=[$before] after=[$after]"
|
||||
}
|
||||
|
||||
# PATTERN matches a fleetd jar in either build layout, not only the run/ one: a process whose
|
||||
# argv names a jar under target/ must be found too, the same way the run/ case above is.
|
||||
test_running_pid_finds_a_real_java_named_process_from_target_dir() {
|
||||
local before after standin_pid
|
||||
before="$(running_pid)"
|
||||
( exec -a java sh -c 'echo "target/fleetd.jar" >/dev/null; sleep 20' ) &
|
||||
standin_pid=$!
|
||||
sleep 0.3
|
||||
after="$(running_pid)"
|
||||
kill "$standin_pid" 2>/dev/null || true
|
||||
wait "$standin_pid" 2>/dev/null || true
|
||||
printf '%s\n' "$after" | grep -qxF "$standin_pid" \
|
||||
|| fail "running_pid() did not find a real second process (pid $standin_pid, comm forced to 'java' via exec -a) naming a jar under target/: before=[$before] after=[$after]"
|
||||
}
|
||||
|
||||
# Broadening PATTERN to match both build layouts must not also broaden it into matching a
|
||||
# non-exec'ing shell that merely holds the target/ text as a literal argument, the same
|
||||
# self-matching shape test_running_pid_excludes_self_matching_wrapper_shell above excludes for
|
||||
# the run/ text.
|
||||
test_running_pid_excludes_self_matching_wrapper_shell_naming_target_dir() {
|
||||
local before after wrapper_pid
|
||||
before="$(running_pid)"
|
||||
sh -c 'echo "target/fleetd.jar" >/dev/null; sleep 20' &
|
||||
wrapper_pid=$!
|
||||
sleep 0.3
|
||||
after="$(running_pid)"
|
||||
kill "$wrapper_pid" 2>/dev/null || true
|
||||
wait "$wrapper_pid" 2>/dev/null || true
|
||||
[ "$after" = "$before" ] \
|
||||
|| fail "running_pid() counted a self-matching wrapper shell (pid $wrapper_pid, holding 'target/fleetd.jar' as literal text in its own argv, not the daemon): before=[$before] after=[$after]"
|
||||
}
|
||||
|
||||
# fleetd #593 CORRECTION 1, hole 2 — the round-1 filter denied known shell names (sh/bash/zsh/
|
||||
# dash/ksh) and counted everything else. `ssh`, `perl`, `python3`, `ruby`, `tail` — anything not on
|
||||
# that list, carrying the pattern in its own argv — was still counted right alongside the real
|
||||
@@ -702,6 +734,19 @@ test_report_jar_state_both_absent_is_not_a_mismatch() {
|
||||
fi
|
||||
}
|
||||
|
||||
# Reads JAR and BUILD_JAR exactly as the script sources them, with nothing here assigning
|
||||
# either first. JAR must resolve outside $MODULE/target/, and JAR must differ from BUILD_JAR:
|
||||
# the daemon's live path and Maven's own build output are never the same file.
|
||||
test_jar_and_build_jar_are_sourced_outside_target_and_differ() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
case "$JAR" in
|
||||
"$MODULE"/target/*)
|
||||
fail "\$JAR must not live under \$MODULE/target/ — got $JAR" ;;
|
||||
esac
|
||||
[ "$JAR" != "$BUILD_JAR" ] \
|
||||
|| fail "\$JAR and \$BUILD_JAR must not be the same path — got $JAR"
|
||||
}
|
||||
|
||||
# fleetd #493/#664 — never build into the path a running process holds. swap_staged_jar is
|
||||
# exercised directly against real files on disk (not stubs), because the whole point is file
|
||||
# behavior (does the content move, does the source disappear, does a failure leave both sides
|
||||
@@ -1305,12 +1350,71 @@ test_run_drain_gate_declined_reply_refuses() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
# Writes a launchd-plist fixture naming jar_path as the ProgramArguments entry after "-jar", so
|
||||
# check_jar_path_matches_plist has something real to read back.
|
||||
write_launchd_plist_fixture() {
|
||||
local path="$1" jar_path="$2"
|
||||
cat > "$path" <<PLIST
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
|
||||
<plist version="1.0">
|
||||
<dict>
|
||||
<key>Label</key>
|
||||
<string>test.fixture</string>
|
||||
<key>ProgramArguments</key>
|
||||
<array>
|
||||
<string>/usr/bin/java</string>
|
||||
<string>-jar</string>
|
||||
<string>$jar_path</string>
|
||||
<string>fleetd.yaml</string>
|
||||
</array>
|
||||
</dict>
|
||||
</plist>
|
||||
PLIST
|
||||
}
|
||||
|
||||
# Agreeing case: a plist whose ProgramArguments names the same jar, resolved, must proceed and
|
||||
# say so, never die.
|
||||
test_check_jar_path_matches_plist_agrees_ok() {
|
||||
local dir jar plist output rc=0
|
||||
dir="$TMP/jar-path-agree"; mkdir -p "$dir/run"
|
||||
jar="$dir/run/fleetd.jar"
|
||||
plist="$dir/agree.plist"
|
||||
write_launchd_plist_fixture "$plist" "$jar"
|
||||
output="$(check_jar_path_matches_plist "$jar" "$plist" 2>&1)" || rc=$?
|
||||
[ "$rc" -eq 0 ] \
|
||||
|| fail "check_jar_path_matches_plist must succeed when the plist names the same jar: $output"
|
||||
printf '%s' "$output" | grep -qF 'jar path check' \
|
||||
|| fail "check_jar_path_matches_plist did not print the agreement line: $output"
|
||||
}
|
||||
|
||||
# The disagreeing case, and the positive control this check exists for: a plist naming a
|
||||
# different jar must die, naming both paths.
|
||||
test_check_jar_path_matches_plist_mismatch_dies() {
|
||||
local dir jar other_jar plist output rc=0
|
||||
dir="$TMP/jar-path-mismatch"; mkdir -p "$dir/run" "$dir/target"
|
||||
jar="$dir/run/fleetd.jar"
|
||||
other_jar="$dir/target/fleetd.jar"
|
||||
plist="$dir/mismatch.plist"
|
||||
write_launchd_plist_fixture "$plist" "$other_jar"
|
||||
output="$(check_jar_path_matches_plist "$jar" "$plist" 2>&1)" || rc=$?
|
||||
[ "$rc" -ne 0 ] \
|
||||
|| fail "check_jar_path_matches_plist must die when the plist names a different jar"
|
||||
printf '%s' "$output" | grep -qF "$jar" \
|
||||
|| fail "die message does not name this script's jar path: $output"
|
||||
printf '%s' "$output" | grep -qF "$other_jar" \
|
||||
|| fail "die message does not name the plist's jar path: $output"
|
||||
}
|
||||
|
||||
# fleetd #555 item 4 — the report-state dispatch on $SUPERVISOR_KIND. Inverting this used to report
|
||||
# the wrong supervisor and, for the launchd arm specifically, skip check_log_path_matches_plist.
|
||||
CHECK_LOG_PATH_CALLED=0
|
||||
CHECK_JAR_PATH_CALLED=0
|
||||
stub_check_log_path_recorder() {
|
||||
CHECK_LOG_PATH_CALLED=0
|
||||
CHECK_JAR_PATH_CALLED=0
|
||||
check_log_path_matches_plist() { CHECK_LOG_PATH_CALLED=1; }
|
||||
check_jar_path_matches_plist() { CHECK_JAR_PATH_CALLED=1; }
|
||||
}
|
||||
|
||||
test_report_supervisor_state_launchd_sets_supervised_and_checks_log_path() {
|
||||
@@ -1321,6 +1425,8 @@ test_report_supervisor_state_launchd_sets_supervised_and_checks_log_path() {
|
||||
[ "$SUPERVISED" = 1 ] || fail "report_supervisor_state launchd must set SUPERVISED=1"
|
||||
[ "$CHECK_LOG_PATH_CALLED" = 1 ] \
|
||||
|| fail "report_supervisor_state launchd must call check_log_path_matches_plist"
|
||||
[ "$CHECK_JAR_PATH_CALLED" = 1 ] \
|
||||
|| fail "report_supervisor_state launchd must call check_jar_path_matches_plist"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
@@ -1332,6 +1438,8 @@ test_report_supervisor_state_systemd_sets_supervised_without_log_path_check() {
|
||||
[ "$SUPERVISED" = 1 ] || fail "report_supervisor_state systemd must set SUPERVISED=1"
|
||||
[ "$CHECK_LOG_PATH_CALLED" = 0 ] \
|
||||
|| fail "report_supervisor_state systemd must NOT call check_log_path_matches_plist"
|
||||
[ "$CHECK_JAR_PATH_CALLED" = 0 ] \
|
||||
|| fail "report_supervisor_state systemd must NOT call check_jar_path_matches_plist"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
@@ -2189,6 +2297,8 @@ test_assert_single_daemon_accepts_one_pid
|
||||
test_assert_single_daemon_rejects_two_pids
|
||||
test_running_pid_excludes_self_matching_wrapper_shell
|
||||
test_running_pid_finds_a_real_java_named_second_process
|
||||
test_running_pid_finds_a_real_java_named_process_from_target_dir
|
||||
test_running_pid_excludes_self_matching_wrapper_shell_naming_target_dir
|
||||
test_running_pid_drops_a_pid_whose_comm_is_not_java
|
||||
test_running_pid_drops_a_pid_that_exited_before_the_comm_lookup
|
||||
test_running_pid_counts_a_pid_whose_comm_is_java
|
||||
@@ -2200,6 +2310,7 @@ test_jar_id_reports_unhashable_when_no_hasher_on_path
|
||||
test_report_jar_state_agrees_when_hashes_match
|
||||
test_report_jar_state_warns_when_hashes_differ
|
||||
test_report_jar_state_both_absent_is_not_a_mismatch
|
||||
test_jar_and_build_jar_are_sourced_outside_target_and_differ
|
||||
test_swap_staged_jar_moves_staged_onto_live
|
||||
test_swap_staged_jar_dies_without_staged_file
|
||||
test_swap_staged_jar_dies_when_mv_fails
|
||||
@@ -2241,6 +2352,8 @@ test_drain_confirmed_false_on_anything_else
|
||||
test_run_drain_gate_skips_prompt_when_not_required
|
||||
test_run_drain_gate_confirmed_reply_does_not_refuse
|
||||
test_run_drain_gate_declined_reply_refuses
|
||||
test_check_jar_path_matches_plist_agrees_ok
|
||||
test_check_jar_path_matches_plist_mismatch_dies
|
||||
test_report_supervisor_state_launchd_sets_supervised_and_checks_log_path
|
||||
test_report_supervisor_state_systemd_sets_supervised_without_log_path_check
|
||||
test_report_supervisor_state_none_leaves_supervised_zero
|
||||
|
||||
Reference in New Issue
Block a user