Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 9e4e423ad6 | |||
| d4a2cd720c |
+15
-15
@@ -452,15 +452,6 @@ placement: weighted
|
||||
# seconds, before a spawn may land on it again. Applies to every profile's effective credential
|
||||
# (its own name, or its credentialId if set above) — there is no per-profile override. Default
|
||||
# 1800 (30 minutes) when omitted or non-positive.
|
||||
#
|
||||
# fleetd #466: this is now only the BASE of an escalating backoff, not a flat retry rate. A
|
||||
# credential quarantined again within one base cooldown of the previous quarantine ending (still
|
||||
# reporting exhausted — e.g. a weekly subscription limit that hasn't reset) backs off further:
|
||||
# cooldown doubles each such time, capped at 12x this value (~6 hours at the 1800s default). A
|
||||
# quarantine that starts after a base-cooldown's worth of quiet resets back to this value. Not
|
||||
# configurable per se — the multiplier and ceiling are constants in BackendQuarantine, not new
|
||||
# YAML keys; see its class doc for the exact formula and why there is no automatic probe to clear
|
||||
# it early (the operator's own design constraint — a probe spends the quota it's measuring).
|
||||
# DEFERRED: baked once into the BackendQuarantine built at startup — a running quarantine keeps
|
||||
# its original cooldown regardless; a new value only applies to a quarantine that starts after a
|
||||
# restart. Editing this needs a daemon restart to take effect.
|
||||
@@ -802,12 +793,21 @@ guard:
|
||||
# .claude/skills/, so a member spawned against ANY repo — not only one that already ships its own
|
||||
# copy — can load a bridge skill (e.g. implementer). Unset (the default): no worktree is touched
|
||||
# beyond today's behaviour. A skill folder the target repo already carries under
|
||||
# .claude/skills/<name> is never overwritten — the repo's own copy always wins. Claude Code
|
||||
# members only; an opencode member reads a different path (.opencode/agent) this key does not
|
||||
# touch. Best-effort like worktreeGroup above: a missing/unreadable directory here is logged and
|
||||
# skipped, never a failed spawn. Every non-hidden subdirectory of this directory is copied
|
||||
# wholesale, with no per-file allowlist — don't park scratch files or drafts alongside the real
|
||||
# skill folders, they will be copied into every provisioned worktree too.
|
||||
# .claude/skills/<name> is never overwritten — the repo's own copy always wins. Best-effort like
|
||||
# worktreeGroup above: a missing/unreadable directory here is logged and skipped, never a failed
|
||||
# spawn. Every non-hidden subdirectory of this directory is copied wholesale, with no per-file
|
||||
# allowlist — don't park scratch files or drafts alongside the real skill folders, they will be
|
||||
# copied into every provisioned worktree too.
|
||||
#
|
||||
# fleetd #393: which member KINDS actually consume this once it is copied. kind: claude-code —
|
||||
# the Claude Code CLI discovers .claude/skills/ on its own; nothing else is needed. kind: opencode
|
||||
# — opencode has no such discovery, so OpenCodeLauncher reads whatever landed under
|
||||
# .claude/skills/ and appends each seeded skill's SKILL.md to the generated instructions[] file
|
||||
# (opencode's only channel for static guidance text; unlike Claude Code's Skill tool, the content
|
||||
# is always part of the system prompt, not loaded on demand). Both kinds are covered as of #393 —
|
||||
# earlier builds copied the files for every kind but only claude-code could read them, and the
|
||||
# seeding log said "N of M" regardless. Check the per-spawn launcher log (not just the seeding
|
||||
# log) to see what a given member actually got.
|
||||
# memberSkills: /path/to/fleetd/checkout/.claude/skills
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
|
||||
@@ -222,13 +222,7 @@ public final class Fleetd {
|
||||
// (checked at spawn) and the exhaustion sink wired in below (written on BACKEND_EXHAUSTED).
|
||||
// The cooldown is deferred (see FleetConfig#quarantineCooldownSeconds): it is read once
|
||||
// here, at startup, and a config reload only changes it for a daemon restart.
|
||||
// fleetd #466: escalating, not flat — a credential that keeps reporting exhaustion (e.g. a
|
||||
// weekly subscription limit, which would otherwise be retried on every ~30-minute cooldown,
|
||||
// about 336 times across the week) backs off further each consecutive time, capped at
|
||||
// BackendQuarantine.DEFAULT_MAX_COOLDOWN_MULTIPLE x the base cooldown. See BackendQuarantine's
|
||||
// class doc for the mechanism, why this never fires on cooling-off (a separate, unescalated
|
||||
// mechanism — BackendOutagePolicy below), and the reset.
|
||||
BackendQuarantine quarantine = BackendQuarantine.withEscalation(System::nanoTime,
|
||||
BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));
|
||||
// fleetd #201 Unit 5: one outage-cool-off tracker for the whole daemon, shared between the
|
||||
// launcher (checked at spawn, like `quarantine` above) and the backend-error sink wired in
|
||||
|
||||
@@ -70,14 +70,12 @@ import java.util.regex.PatternSyntaxException;
|
||||
* {@code fixed} (default), {@code round-robin}, or {@code weighted}
|
||||
* @param auth API authentication mode ({@code null} → {@code loopback-trust}, the
|
||||
* historical behaviour), CB-501
|
||||
* @param quarantineCooldownSeconds the BASE cooldown a credential is quarantined for after a
|
||||
* @param quarantineCooldownSeconds how long a credential stays quarantined after a
|
||||
* {@code BACKEND_EXHAUSTED} classification (CB-578 stage B); {@code null}/{@code
|
||||
* <=0} → {@link #DEFAULT_QUARANTINE_COOLDOWN_SECONDS}. Since fleetd #466 this is
|
||||
* only the first occurrence's length — a credential quarantined again shortly
|
||||
* after this cooldown ends backs off further, up to a ceiling; see {@code
|
||||
* BackendQuarantine}'s class doc. Baked once into the {@code BackendQuarantine}
|
||||
* built at startup, so it is DEFERRED: changing it needs a restart, and a
|
||||
* quarantine already running keeps whatever cooldown was live when it started.
|
||||
* <=0} → {@link #DEFAULT_QUARANTINE_COOLDOWN_SECONDS}. Baked once into the
|
||||
* {@code BackendQuarantine} built at startup, so it is DEFERRED: changing it
|
||||
* needs a restart, and a quarantine already running keeps whatever cooldown was
|
||||
* live when it started.
|
||||
* @param memberCredentials deny-by-default policy (CB-596) for which of the operator's own host
|
||||
* credentials a spawned member's pane inherits. {@code null} (the block
|
||||
* omitted) blocks nothing — see {@link MemberCredentials}.
|
||||
|
||||
@@ -1269,13 +1269,6 @@ public final class FleetMcp {
|
||||
* (the backend text that triggered the most recent quarantine of that credential, omitted when
|
||||
* none is known) — so a lead can see WHICH model to turn off and WHY, without reading the
|
||||
* daemon log.
|
||||
*
|
||||
* <p>fleetd #466 scope item 2: each {@code quarantined} row also names {@code
|
||||
* quarantineAttempt} — 1 for a first-time exhaustion, 2 for the second in a row, and so on — so
|
||||
* an operator sees "this is the 5th time" instead of inferring it from {@code
|
||||
* quarantinedForSeconds} alone. Read off {@link BackendQuarantine#status(String)}, the same
|
||||
* one-call accessor {@code quarantinedForSeconds} itself comes from here (see its doc) — never a
|
||||
* separately derived count.
|
||||
*/
|
||||
public static Map<String, Object> profilesView(PeerLauncher workers, QuarantineSource quarantine, OutageSource outage) {
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
@@ -1288,14 +1281,10 @@ public final class FleetMcp {
|
||||
exhaustionDetectionArmed.put(profile, quarantine.exhaustedPatternArmed().apply(profile));
|
||||
String credentialId = quarantine.credentialIdFor().apply(profile);
|
||||
if (credentialId != null) {
|
||||
// fleetd #466 scope item 2: quarantinedForSeconds and quarantineAttempt come off the
|
||||
// ONE BackendQuarantine#status(credentialId) call — never a second, independent read
|
||||
// for the attempt count — so the two can never disagree about which streak this is.
|
||||
quarantine.quarantine().status(credentialId).ifPresent(status -> {
|
||||
quarantine.quarantine().remainingSeconds(credentialId).ifPresent(remaining -> {
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("credentialId", credentialId);
|
||||
row.put("quarantinedForSeconds", status.remainingSeconds());
|
||||
row.put("quarantineAttempt", status.repeatCount());
|
||||
row.put("quarantinedForSeconds", remaining);
|
||||
String model = quarantine.modelFor().apply(profile);
|
||||
if (model != null && !model.isBlank()) {
|
||||
row.put("model", model);
|
||||
@@ -1735,13 +1724,10 @@ public final class FleetMcp {
|
||||
}
|
||||
String credentialId = quarantine.credentialIdFor().apply(profile);
|
||||
if (credentialId != null) {
|
||||
// fleetd #466 scope item 2: same one-call status() read as profilesView above — see its
|
||||
// comment for why this must not become two separate lookups.
|
||||
quarantine.quarantine().status(credentialId).ifPresent(status -> {
|
||||
quarantine.quarantine().remainingSeconds(credentialId).ifPresent(remaining -> {
|
||||
row.put("free", 0);
|
||||
row.put("credentialId", credentialId);
|
||||
row.put("quarantinedForSeconds", status.remainingSeconds());
|
||||
row.put("quarantineAttempt", status.repeatCount());
|
||||
row.put("quarantinedForSeconds", remaining);
|
||||
});
|
||||
}
|
||||
String outageCredentialId = outage.credentialIdFor().apply(profile);
|
||||
|
||||
@@ -331,11 +331,22 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
+ "form — opencode's per-model context limit could not be applied for this profile",
|
||||
cfg.profile(), cfg.model());
|
||||
}
|
||||
// fleetd #393: memberSkills seeding (GitWorktrees#seedSkills) copies skill folders into
|
||||
// EVERY provisioned worktree's .claude/skills/ regardless of which kind ultimately spawns
|
||||
// into it — that copy step cannot know the kind, only the caller of GitWorktrees#add does
|
||||
// (see that method's own javadoc). .claude/skills/ is a Claude Code CLI convention the CLI
|
||||
// discovers on its own; opencode has no such discovery, so without this, a seeded skill
|
||||
// never reaches an opencode member even though GitWorktrees logged it as seeded. Read
|
||||
// whatever landed under <cwd>/.claude/skills/ here — the one place in this launcher that
|
||||
// knows both the kind (opencode, by construction: this IS OpenCodeLauncher) and the cwd.
|
||||
List<Path> skillInstructionFiles = skillInstructionFiles(spec.cwd());
|
||||
// A config file is needed for the bridge MCP mount, a member charter, the IDE MCP (+ its
|
||||
// guidance overlay, CB-634), a pinned endpoint (CB-508), or a resolvable autoCompactWindow.
|
||||
// guidance overlay, CB-634), a pinned endpoint (CB-508), a resolvable autoCompactWindow, or
|
||||
// at least one seeded skill to deliver via instructions[] (fleetd #393).
|
||||
if (cfg.hasMcp() || cfg.hasIdeMcp() || spec.charter() != null || hasCustomProvider(cfg)
|
||||
|| wantsContextLimit) {
|
||||
workerEnv.put("OPENCODE_CONFIG", writeConfig(cfg, spec.charter(), spec.cwd()).toString());
|
||||
|| wantsContextLimit || !skillInstructionFiles.isEmpty()) {
|
||||
workerEnv.put("OPENCODE_CONFIG",
|
||||
writeConfig(cfg, spec.charter(), spec.cwd(), skillInstructionFiles).toString());
|
||||
}
|
||||
applyGitToken(workerEnv, cfg);
|
||||
List<String> argv = argvWithResume(argvWithModel(argvWithAuto(cfg), cfg), spec.resumeSessionId());
|
||||
@@ -358,6 +369,65 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
return withAgent;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #393: the {@code SKILL.md} paths under {@code <cwd>/.claude/skills/} this launcher can
|
||||
* turn into {@code instructions[]} entries, plus the honest log this ticket asks for — emitted
|
||||
* here, at the one point a skill's fate for THIS spawn is actually known, rather than trusting
|
||||
* {@code GitWorktrees#seedSkills}'s kind-blind "N of M" line to mean "and it will be read."
|
||||
*
|
||||
* <p>Every non-hidden subdirectory of {@code .claude/skills/} is a candidate, whether it got
|
||||
* there via {@code memberSkills:} seeding or because the target repo ships its own copy — this
|
||||
* launcher does not care which; it only cares what it can find at spawn time. A candidate with
|
||||
* a {@code SKILL.md} at its top level (the same shape {@link #writeConfig} already requires for
|
||||
* the charter and IDE-rules instructions entries) is delivered; anything else is a directory
|
||||
* this launcher cannot turn into a flat instructions entry, named explicitly in the log rather
|
||||
* than silently dropped, so a caller sees a real "cannot consume" reason and not just a smaller
|
||||
* number than {@code GitWorktrees}' own count.
|
||||
*
|
||||
* <p>No candidates at all (directory absent or empty) logs nothing — the same
|
||||
* no-log-when-nothing-to-say shape {@link #hasCustomProvider} and friends already follow, and
|
||||
* the shape {@code GitWorktrees#seedSkills} itself uses when {@code memberSkills:} is unset.
|
||||
* A failure to even list the directory is logged and treated as "nothing delivered" — best
|
||||
* effort, must never fail the spawn, matching {@code GitWorktrees#seedSkills}'s own contract.
|
||||
*/
|
||||
private List<Path> skillInstructionFiles(String cwd) {
|
||||
if (cwd == null || cwd.isBlank()) {
|
||||
return List.of();
|
||||
}
|
||||
Path skillsDir = Path.of(cwd, ".claude", "skills");
|
||||
if (!Files.isDirectory(skillsDir)) {
|
||||
return List.of();
|
||||
}
|
||||
List<Path> candidates;
|
||||
try (var listing = Files.list(skillsDir)) {
|
||||
candidates = listing.filter(Files::isDirectory)
|
||||
.filter(p -> !p.getFileName().toString().startsWith("."))
|
||||
.sorted()
|
||||
.toList();
|
||||
} catch (IOException e) {
|
||||
log.warn("could not scan {} for skill folders to deliver to this opencode member: {}",
|
||||
skillsDir, e.getMessage());
|
||||
return List.of();
|
||||
}
|
||||
if (candidates.isEmpty()) {
|
||||
return List.of();
|
||||
}
|
||||
List<Path> delivered = candidates.stream()
|
||||
.map(dir -> dir.resolve("SKILL.md"))
|
||||
.filter(Files::isRegularFile)
|
||||
.toList();
|
||||
List<String> undeliverable = candidates.stream()
|
||||
.filter(dir -> !Files.isRegularFile(dir.resolve("SKILL.md")))
|
||||
.map(dir -> dir.getFileName().toString())
|
||||
.toList();
|
||||
log.info("skill delivery: {} of {} skill folder(s) under {} reached this opencode member via "
|
||||
+ "instructions[] (opencode does not read .claude/skills/ natively, unlike "
|
||||
+ "Claude Code){}",
|
||||
delivered.size(), candidates.size(), skillsDir,
|
||||
undeliverable.isEmpty() ? "" : "; no SKILL.md, could not be delivered: " + undeliverable);
|
||||
return delivered;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when this profile pins its own OpenAI-compatible endpoint (CB-508) rather than using
|
||||
* whatever provider opencode resolves by default.
|
||||
@@ -418,8 +488,15 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
* fresh per-spawn directory under {@link #configRoot}, and return the config file's path for
|
||||
* {@code OPENCODE_CONFIG}. The dir is unique per spawn so concurrent workers never race on it;
|
||||
* it is best-effort cleaned on JVM exit (worker config is disposable — regenerated every spawn).
|
||||
*
|
||||
* @param skillInstructionFiles fleetd #393: absolute {@code SKILL.md} paths from
|
||||
* {@link #skillInstructionFiles(String)}, appended to
|
||||
* {@code instructions[]} so a {@code memberSkills:}-seeded skill
|
||||
* reaches this opencode member the same way the charter and IDE
|
||||
* rules already do.
|
||||
*/
|
||||
private Path writeConfig(FleetConfig.Profile cfg, String charterText, String cwd) {
|
||||
private Path writeConfig(FleetConfig.Profile cfg, String charterText, String cwd,
|
||||
List<Path> skillInstructionFiles) {
|
||||
try {
|
||||
Path dir = Files.createTempDirectory(configParentDir(), "fleetd-opencode-");
|
||||
dir.toFile().deleteOnExit();
|
||||
@@ -447,7 +524,28 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
Files.writeString(charter, charterText);
|
||||
charter.toFile().deleteOnExit();
|
||||
|
||||
root.putArray("instructions").add(charter.toAbsolutePath().toString());
|
||||
// fleetd #393 follow-up: withArray, not putArray. putArray REPLACES whatever node
|
||||
// is already at "instructions" — harmless only as long as this block runs first
|
||||
// against a still-empty root, which is an ordering constraint nothing declared or
|
||||
// tested. The skills writer just below, and the IDE-rules writer further down,
|
||||
// both already use withArray (get-or-create) for exactly this reason; this was the
|
||||
// one straggler. Proven load-bearing on the fleetd #393 merge: flipping this one
|
||||
// call back to putArray left the whole suite green while silently deleting the
|
||||
// charter entry whenever skills or IDE rules ran after it — an opencode member
|
||||
// would launch with no role contract at all, worse than the bug #393 fixed, and
|
||||
// nothing caught it. See OpenCodeLauncherTest's
|
||||
// instructionsArrayHoldsCharterThenIdeRulesInOrder and
|
||||
// instructionsArrayHoldsCharterThenSkillsThenIdeRulesInOrder.
|
||||
root.withArray("instructions").add(charter.toAbsolutePath().toString());
|
||||
}
|
||||
|
||||
// fleetd #393: each seeded skill's SKILL.md, delivered as a plain instructions[] entry
|
||||
// — the only mechanism opencode has for static guidance text. Unlike Claude Code's
|
||||
// Skill tool, opencode cannot load one of these on demand by name; the content is just
|
||||
// always part of the system prompt from spawn. That is a real difference in HOW the
|
||||
// content reaches the member, not a reason to withhold it.
|
||||
for (Path skillFile : skillInstructionFiles) {
|
||||
root.withArray("instructions").add(skillFile.toAbsolutePath().toString());
|
||||
}
|
||||
|
||||
if (cfg.hasMcp() || cfg.hasIdeMcp()) {
|
||||
|
||||
@@ -3,7 +3,6 @@ package dev.ltms.fleet.placement;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.LongSupplier;
|
||||
@@ -22,158 +21,46 @@ import java.util.function.LongSupplier;
|
||||
* <p>The clock is injected ({@link LongSupplier}, conventionally {@code System::nanoTime} like
|
||||
* {@code FleetHealthMonitor}), never read inline, so a quarantine's expiry is testable without a
|
||||
* real sleep.
|
||||
*
|
||||
* <h2>Escalation (fleetd #466)</h2>
|
||||
* A flat cooldown does not fit every exhaustion. A backend that reports "out of capacity for the
|
||||
* rest of the hour" recovers in one cooldown; a weekly subscription limit does not — it keeps
|
||||
* reporting exhausted on every attempt made before the window resets, so a flat 30-minute cooldown
|
||||
* (the default {@code cooldownNanos}) means roughly 336 pointless spawn attempts across a week, one
|
||||
* every cooldown.
|
||||
*
|
||||
* <p><strong>This class only ever sees the exhaustion signal.</strong> Its only production caller is
|
||||
* {@code Fleetd.exhaustionSink}, wired to fire on a {@code BACKEND_EXHAUSTED} classification alone.
|
||||
* The daemon's other outage state — a credential "cooling off" after repeated non-exhaustion
|
||||
* backend errors (an HTTP 5xx storm, say) — is a separate mechanism, {@code BackendOutagePolicy},
|
||||
* with its own short fixed 60s cooldown and no repeat tracking. The two are never merged: escalating
|
||||
* on a cooling-off signal would turn a transient 5xx storm into a multi-hour backoff, which is
|
||||
* exactly the failure this ticket is not asking for. Confirmed by reading every call site of
|
||||
* {@link #quarantine} — {@code BackendOutagePolicy} has its own {@code coolOff} method and never
|
||||
* calls this one.
|
||||
*
|
||||
* <p><strong>Mechanism</strong> — the {@link #withEscalation} constructors track, per credential, how
|
||||
* many times in a row {@link #quarantine} has been called without an intervening "quiet" gap.
|
||||
* Each call computes {@code cooldownNanos * backoffMultiplier ^ (repeatCount - 1)}, capped at
|
||||
* {@code maxCooldownNanos}. A call counts as a continuation of the same streak — {@code repeatCount}
|
||||
* increments — when it arrives no more than one base {@code cooldownNanos} after the previous
|
||||
* quarantine's deadline (this covers both "still quarantined" and "quarantine just expired and it
|
||||
* was exhausted again immediately"); otherwise the streak resets and this call is treated as a fresh
|
||||
* first occurrence at the base cooldown.
|
||||
*
|
||||
* <p><strong>Reset, honestly stated.</strong> The ideal reset signal is "the cooldown expired and the
|
||||
* next attempt succeeded" — but nothing in this codebase reports a spawn success back to this class
|
||||
* (checked: {@code SessionManager} and {@code CompositePeerLauncher} never call any method here
|
||||
* except {@link #quarantine}/{@link #isQuarantined}/{@link #remainingSeconds}, none of which is a
|
||||
* success hook). Lacking that signal, the reset used here is a time-based proxy: a base-cooldown's
|
||||
* worth of quiet — no exhaustion report for that credential — since the last quarantine ended. It is
|
||||
* not proof the credential started working again, only the best available evidence without adding an
|
||||
* active probe, which is out of scope by the operator's own design constraint (no automatic probing
|
||||
* of a limited backend).
|
||||
*
|
||||
* <p><strong>Ceiling.</strong> {@code maxCooldownNanos} bounds the growth — an unbounded backoff is a
|
||||
* permanent, unrecoverable-without-a-restart outage, which would be worse than the flat-rate bug this
|
||||
* escalation fixes. {@link #withEscalation(LongSupplier, long)} defaults the ceiling to
|
||||
* {@value #DEFAULT_MAX_COOLDOWN_MULTIPLE}x the base cooldown (12x the 1800s default ≈ 6 hours), so a
|
||||
* chronically exhausted credential still gets re-tried roughly every 6 hours instead of every 30
|
||||
* minutes — about a dozen attempts a week instead of ~336.
|
||||
*
|
||||
* <p><strong>Backward compatibility.</strong> The original two-argument {@link #BackendQuarantine(
|
||||
* LongSupplier, long)} constructor is unchanged in behaviour: it is exactly {@code
|
||||
* withEscalation}'s mechanism with {@code backoffMultiplier = 1.0} and {@code maxCooldownNanos =
|
||||
* cooldownNanos}, which collapses the formula back to the original flat {@code now + cooldownNanos}
|
||||
* on every call regardless of history. Every existing call site (roughly 20 across the test suite,
|
||||
* plus {@link #none()}) keeps its current shape and behaviour unchanged.
|
||||
*/
|
||||
public final class BackendQuarantine {
|
||||
|
||||
/** Default growth per consecutive exhaustion streak — see the class doc's Mechanism section. */
|
||||
static final double DEFAULT_BACKOFF_MULTIPLIER = 2.0;
|
||||
/** Default ceiling, expressed as a multiple of the base cooldown — see the class doc's Ceiling section. */
|
||||
static final long DEFAULT_MAX_COOLDOWN_MULTIPLE = 12;
|
||||
|
||||
private final ConcurrentHashMap<String, QuarantineState> quarantines = new ConcurrentHashMap<>();
|
||||
private final ConcurrentHashMap<String, Long> quarantinedUntilNanos = new ConcurrentHashMap<>();
|
||||
private final LongSupplier nowNanos;
|
||||
private final long cooldownNanos;
|
||||
private final double backoffMultiplier;
|
||||
private final long maxCooldownNanos;
|
||||
/** True only for {@link #none()}. See {@link #quarantine} for why this exists. */
|
||||
private final boolean inert;
|
||||
|
||||
/** How many consecutive exhaustion reports a credential is on, and when the resulting cooldown ends. */
|
||||
private record QuarantineState(int repeatCount, long deadlineNanos) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Flat cooldown, unchanged from before fleetd #466 — every {@link #quarantine} call blocks the
|
||||
* credential for exactly {@code cooldownNanos}, regardless of how many times it was called
|
||||
* before. Equivalent to {@link #withEscalation} with no growth ({@code backoffMultiplier = 1.0})
|
||||
* and a ceiling equal to the base cooldown, so it degrades to the identical {@code now +
|
||||
* cooldownNanos} formula every call. Kept for the existing call sites that want a fixed cooldown
|
||||
* (and for tests exercising the fixed-cooldown shape in isolation); production wiring uses
|
||||
* {@link #withEscalation} instead.
|
||||
*
|
||||
* @param nowNanos monotonic clock, injected for testability
|
||||
* @param cooldownNanos how long a fresh {@link #quarantine} call blocks the credential for;
|
||||
* must be positive
|
||||
*/
|
||||
public BackendQuarantine(LongSupplier nowNanos, long cooldownNanos) {
|
||||
this(nowNanos, cooldownNanos, 1.0, cooldownNanos, false);
|
||||
this(nowNanos, cooldownNanos, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* Escalating cooldown (fleetd #466) — see the class doc's Mechanism/Reset/Ceiling sections.
|
||||
*
|
||||
* @param nowNanos monotonic clock, injected for testability
|
||||
* @param cooldownNanos base cooldown, applied to a fresh (non-streak) exhaustion; must be
|
||||
* positive
|
||||
* @param backoffMultiplier growth per consecutive exhaustion; must be {@code >= 1.0} ({@code 1.0}
|
||||
* disables growth and is exactly the flat two-argument constructor)
|
||||
* @param maxCooldownNanos ceiling on the escalated cooldown; must be {@code >= cooldownNanos}
|
||||
*/
|
||||
public BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, double backoffMultiplier,
|
||||
long maxCooldownNanos) {
|
||||
this(nowNanos, cooldownNanos, backoffMultiplier, maxCooldownNanos, false);
|
||||
}
|
||||
|
||||
private BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, double backoffMultiplier,
|
||||
long maxCooldownNanos, boolean inert) {
|
||||
private BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, boolean inert) {
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
if (cooldownNanos <= 0) {
|
||||
throw new IllegalArgumentException("cooldownNanos must be positive: " + cooldownNanos);
|
||||
}
|
||||
if (backoffMultiplier < 1.0) {
|
||||
throw new IllegalArgumentException("backoffMultiplier must be >= 1.0: " + backoffMultiplier);
|
||||
}
|
||||
if (maxCooldownNanos < cooldownNanos) {
|
||||
throw new IllegalArgumentException(
|
||||
"maxCooldownNanos must be >= cooldownNanos: " + maxCooldownNanos + " < " + cooldownNanos);
|
||||
}
|
||||
this.cooldownNanos = cooldownNanos;
|
||||
this.backoffMultiplier = backoffMultiplier;
|
||||
this.maxCooldownNanos = maxCooldownNanos;
|
||||
this.inert = inert;
|
||||
}
|
||||
|
||||
/**
|
||||
* Escalating cooldown with the fleetd #466 default shape: cooldown doubles
|
||||
* ({@value #DEFAULT_BACKOFF_MULTIPLIER}x) per consecutive exhaustion streak, capped at
|
||||
* {@value #DEFAULT_MAX_COOLDOWN_MULTIPLE}x the base cooldown. This is what production wiring
|
||||
* ({@code Fleetd.main}) uses.
|
||||
*
|
||||
* @param nowNanos monotonic clock, injected for testability
|
||||
* @param cooldownNanos base cooldown, applied to a fresh (non-streak) exhaustion; must be positive
|
||||
*/
|
||||
public static BackendQuarantine withEscalation(LongSupplier nowNanos, long cooldownNanos) {
|
||||
return new BackendQuarantine(nowNanos, cooldownNanos, DEFAULT_BACKOFF_MULTIPLIER,
|
||||
cooldownNanos * DEFAULT_MAX_COOLDOWN_MULTIPLE, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* Inert quarantine — {@link #quarantine} does nothing on this instance, so nothing is ever
|
||||
* quarantined. The explicit stand-in a caller (or a test not exercising this feature) passes
|
||||
* instead of a defaulting overload, exactly like {@code ExhaustedPatternLookup.none()}.
|
||||
*/
|
||||
public static BackendQuarantine none() {
|
||||
return new BackendQuarantine(() -> 0L, 1, 1.0, 1, true);
|
||||
return new BackendQuarantine(() -> 0L, 1, true);
|
||||
}
|
||||
|
||||
/**
|
||||
* Quarantine {@code credentialId} starting now. On a flat instance (the two-argument
|
||||
* constructor) this always blocks for exactly {@code cooldownNanos}, restarting the cooldown at
|
||||
* full length on every call — a fresh refusal is fresh evidence the account is still exhausted,
|
||||
* not a reason to let an earlier, shorter wait stand. On an escalating instance ({@link
|
||||
* #withEscalation}) the cooldown grows with each call that arrives within one base cooldown of
|
||||
* the previous deadline, and resets to the base cooldown once a call arrives after a longer gap
|
||||
* — see the class doc.
|
||||
* Quarantine {@code credentialId} for the configured cooldown, starting now. A repeat call while
|
||||
* already quarantined restarts the cooldown at full length — a fresh refusal is fresh evidence the
|
||||
* account is still exhausted, not a reason to let an earlier, shorter wait stand.
|
||||
*
|
||||
* <p>On {@link #none()} this is a no-op. It has to be: that instance holds a clock frozen at 0,
|
||||
* so recording a deadline would produce a quarantine that never expires — a credential locked out
|
||||
@@ -186,13 +73,7 @@ public final class BackendQuarantine {
|
||||
if (inert) {
|
||||
return;
|
||||
}
|
||||
long now = nowNanos.getAsLong();
|
||||
quarantines.compute(credentialId, (id, prev) -> {
|
||||
int repeatCount = (prev == null || now - prev.deadlineNanos() > cooldownNanos)
|
||||
? 1
|
||||
: prev.repeatCount() + 1;
|
||||
return new QuarantineState(repeatCount, now + escalatedCooldownNanos(repeatCount));
|
||||
});
|
||||
quarantinedUntilNanos.put(credentialId, nowNanos.getAsLong() + cooldownNanos);
|
||||
}
|
||||
|
||||
/** Whether {@code credentialId} is quarantined right now. */
|
||||
@@ -206,49 +87,6 @@ public final class BackendQuarantine {
|
||||
return remaining > 0 ? OptionalLong.of(toSecondsRoundedUp(remaining)) : OptionalLong.empty();
|
||||
}
|
||||
|
||||
/**
|
||||
* Remaining seconds together with which consecutive exhaustion this is (fleetd #466 scope item
|
||||
* 2) — {@code repeatCount} 1 for a first occurrence, 2 for the second in a row, and so on; see
|
||||
* {@link #quarantine}'s class-doc Mechanism section for exactly when a call continues a streak
|
||||
* versus starts a fresh one.
|
||||
*
|
||||
* <p><strong>Read together, off the one {@link QuarantineState} entry {@link #quarantine} itself
|
||||
* wrote</strong> — a single {@code quarantines.get(credentialId)}, never a separate lookup or a
|
||||
* value re-derived from {@code remainingSeconds} (e.g. inverting {@link
|
||||
* #escalatedCooldownNanos}). That inversion is not just extra work to avoid: once a streak has
|
||||
* hit {@code maxCooldownNanos}, every further consecutive exhaustion reports the identical
|
||||
* cooldown, so a derivation that starts from the cooldown value cannot tell the 4th repeat from
|
||||
* the 9th — only the stored {@code repeatCount} can. This is the same rule {@code
|
||||
* CompositePeerLauncher.modelGateState()} documents for its own gate/report pair: the report
|
||||
* reads the exact accessor the behaviour reads, so it can never disagree with what actually
|
||||
* happened (the fleetd #404/#422 lesson). {@code fleet_profiles}/{@code fleet_list}/{@code GET
|
||||
* /profiles} all call this — never {@link #remainingSeconds} plus a second, independent count —
|
||||
* for exactly that reason.
|
||||
*
|
||||
* @return empty when {@code credentialId} is not currently quarantined (including on {@link
|
||||
* #none()}, which quarantines nothing)
|
||||
*/
|
||||
public Optional<Status> status(String credentialId) {
|
||||
QuarantineState state = quarantines.get(credentialId);
|
||||
if (state == null) {
|
||||
return Optional.empty();
|
||||
}
|
||||
long remaining = state.deadlineNanos() - nowNanos.getAsLong();
|
||||
return remaining > 0
|
||||
? Optional.of(new Status(toSecondsRoundedUp(remaining), state.repeatCount()))
|
||||
: Optional.empty();
|
||||
}
|
||||
|
||||
/**
|
||||
* @param remainingSeconds seconds left on the quarantine, identical to {@link
|
||||
* #remainingSeconds(String)}'s answer for the same credential at the
|
||||
* same instant
|
||||
* @param repeatCount 1 for a first occurrence, 2 for the second consecutive one, etc. —
|
||||
* see {@link #status(String)}
|
||||
*/
|
||||
public record Status(long remainingSeconds, int repeatCount) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Every currently-quarantined credential id and its remaining seconds (CB-578 stage B fleet
|
||||
* reporting) — expired entries are never included. Not pruned from the backing map here: it stays
|
||||
@@ -257,8 +95,8 @@ public final class BackendQuarantine {
|
||||
*/
|
||||
public Map<String, Long> activeRemainingSeconds() {
|
||||
Map<String, Long> out = new LinkedHashMap<>();
|
||||
quarantines.forEach((credentialId, state) -> {
|
||||
long remaining = state.deadlineNanos() - nowNanos.getAsLong();
|
||||
quarantinedUntilNanos.forEach((credentialId, deadline) -> {
|
||||
long remaining = deadline - nowNanos.getAsLong();
|
||||
if (remaining > 0) {
|
||||
out.put(credentialId, toSecondsRoundedUp(remaining));
|
||||
}
|
||||
@@ -267,14 +105,8 @@ public final class BackendQuarantine {
|
||||
}
|
||||
|
||||
private long remainingNanos(String credentialId) {
|
||||
QuarantineState state = quarantines.get(credentialId);
|
||||
return state == null ? 0L : state.deadlineNanos() - nowNanos.getAsLong();
|
||||
}
|
||||
|
||||
/** {@code cooldownNanos * backoffMultiplier ^ (repeatCount - 1)}, capped at {@code maxCooldownNanos}. */
|
||||
private long escalatedCooldownNanos(int repeatCount) {
|
||||
double raw = cooldownNanos * Math.pow(backoffMultiplier, repeatCount - 1);
|
||||
return raw >= (double) maxCooldownNanos ? maxCooldownNanos : (long) raw;
|
||||
Long deadline = quarantinedUntilNanos.get(credentialId);
|
||||
return deadline == null ? 0L : deadline - nowNanos.getAsLong();
|
||||
}
|
||||
|
||||
private static long toSecondsRoundedUp(long nanos) {
|
||||
|
||||
@@ -690,14 +690,18 @@ public final class GitWorktrees implements Worktrees {
|
||||
* {@code fleet.seededSkillsNote}, readable with {@code git config --worktree --get-all
|
||||
* fleet.seededSkills}.
|
||||
*
|
||||
* <p><b>Claude Code specific by construction, not by a backend check here.</b> Only {@code
|
||||
* .claude/skills/<name>/SKILL.md} is a path any launcher reads today (opencode's equivalent is a
|
||||
* different shape under {@code .opencode/agent}, out of scope — see issue #362). This method
|
||||
* only copies files; like {@link #isolateToolSurface} — which neutralizes BOTH {@code .mcp.json}
|
||||
* and {@code opencode.json} unconditionally — it runs the same for every worktree regardless of
|
||||
* which backend ultimately spawns into it, because the backend is not yet chosen at {@link #add}
|
||||
* time. A seeded {@code .claude/skills/} directory in an opencode member's worktree is simply
|
||||
* never read by that launcher.
|
||||
* <p><b>Kind-blind by construction, not by a backend check here — this used to be a real gap
|
||||
* (fleetd #393).</b> This method only copies files; like {@link #isolateToolSurface} — which
|
||||
* neutralizes BOTH {@code .mcp.json} and {@code opencode.json} unconditionally — it runs the
|
||||
* same for every worktree regardless of which backend ultimately spawns into it, because no
|
||||
* caller of {@link #add} hands this class a kind to consult. Before fleetd #393, that made the
|
||||
* log line below a false claim of success for a {@code kind: opencode} member: opencode has no
|
||||
* built-in discovery of {@code .claude/skills/}, unlike the Claude Code CLI, so a seeded skill
|
||||
* never reached one. It now does — {@code OpenCodeLauncher#skillInstructionFiles} reads
|
||||
* whatever this method copied into {@code .claude/skills/} and appends each {@code SKILL.md} to
|
||||
* the generated {@code instructions[]} — but that delivery, and the log line that honestly
|
||||
* claims it (kind-aware, unlike this one), happens at the launcher, once the kind is actually
|
||||
* known, not here.
|
||||
*/
|
||||
private void seedSkills(String worktreePath) {
|
||||
if (memberSkillsSource == null) {
|
||||
@@ -739,7 +743,15 @@ public final class GitWorktrees implements Worktrees {
|
||||
if (detail.isEmpty()) {
|
||||
detail = "no skill folders found under " + source;
|
||||
}
|
||||
log.info("skill seeding: {} of {} candidate(s) from {} into {}/.claude/skills — {}",
|
||||
// fleetd #393: this only claims the copy step, deliberately — it cannot know the member
|
||||
// kind that will spawn into this worktree (see this method's own javadoc), so it must not
|
||||
// read as "and the member will act on it." Whether that is true depends on the kind: the
|
||||
// Claude Code CLI discovers .claude/skills/ on its own; OpenCodeLauncher logs its own
|
||||
// "skill delivery" line, once the kind is known, naming what it could and could not turn
|
||||
// into instructions[].
|
||||
log.info("skill seeding: {} of {} candidate(s) from {} into {}/.claude/skills — {} "
|
||||
+ "(whether the spawned member can act on this depends on its kind — see "
|
||||
+ "the launcher's own log for that)",
|
||||
seeded.size(), seeded.size() + kept.size(), source, worktreePath, detail);
|
||||
if (seeded.isEmpty()) {
|
||||
return;
|
||||
|
||||
@@ -1,65 +0,0 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #466 follow-up: {@code Fleetd.main} builds the daemon's one {@code BackendQuarantine}
|
||||
* from {@link dev.ltms.fleet.placement.BackendQuarantine#withEscalation(java.util.function.LongSupplier,
|
||||
* long)} — the escalating factory — rather than the plain two-argument constructor, which is still a
|
||||
* flat cooldown (kept for backward compatibility, see that class's doc). {@code
|
||||
* BackendQuarantineTest} proves {@code withEscalation} itself escalates, is ceilinged, and resets;
|
||||
* it says nothing about which one {@code main} actually calls.
|
||||
*
|
||||
* <p>Measured directly: reverting {@code main} to {@code new BackendQuarantine(System::nanoTime,
|
||||
* TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()))} — the pre-#466 flat call — compiles
|
||||
* with 0 errors and leaves the entire 1608-test suite (including every {@code BackendQuarantineTest}
|
||||
* case) green, because no other test constructs its {@code BackendQuarantine} through {@code main};
|
||||
* every one of them builds its own instance directly. That silent regression is exactly the shape
|
||||
* {@link FleetdLeadSeatWiringTest} and {@link FleetdCompletionResolverWiringTest} already guard
|
||||
* against for their own constructor arguments — this is the same class of gap for fleetd #466's
|
||||
* factory choice, following their approach.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||
* BackendQuarantine} and never runs {@code main} — a green result here proves only that the exact
|
||||
* text {@code main} calls {@code BackendQuarantine.withEscalation(...)} rather than the flat
|
||||
* constructor. It does not prove that call actually executes at startup (no test here starts the
|
||||
* daemon), and it does not prove the escalation reaches a real backend or credential — only
|
||||
* {@code BackendQuarantineTest} proves the factory's own behaviour, and only a live daemon proves
|
||||
* the wiring runs.
|
||||
*/
|
||||
class FleetdBackendQuarantineWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] main's BackendQuarantine local is still built from BackendQuarantine.withEscalation(...)")
|
||||
void mainStillWiresTheEscalatingQuarantineFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"BackendQuarantine quarantine = BackendQuarantine.withEscalation(System::nanoTime,\n"
|
||||
+ " TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));"),
|
||||
"Fleetd.main's BackendQuarantine local must still be built from "
|
||||
+ "BackendQuarantine.withEscalation(System::nanoTime, "
|
||||
+ "TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds())). Reverting to the flat "
|
||||
+ "two-argument constructor (fleetd #466's measured regression) compiles with 0 errors "
|
||||
+ "and leaves the whole suite green, including every BackendQuarantineTest case that "
|
||||
+ "proves the escalation itself works — this source check is what must go red instead. "
|
||||
+ "A reverted daemon would go back to retrying a weekly subscription limit on every "
|
||||
+ "flat ~30-minute cooldown, about 336 times across the week.");
|
||||
|
||||
// Negative form of the same check: the pre-#466 flat call, if it ever reappears at this
|
||||
// declaration, must not be mistaken for the escalating one by a looser positive-only check.
|
||||
assertFalse(source.contains(
|
||||
"BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,\n"
|
||||
+ " TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));"),
|
||||
"main's BackendQuarantine local must never regress to the flat two-argument constructor");
|
||||
}
|
||||
}
|
||||
@@ -574,38 +574,6 @@ class FleetMcpTest {
|
||||
assertTrue(out.contains("\"quarantinedForSeconds\":1800"), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #466 scope item 2: {@code fleet_profiles} must carry the repeat count beside the
|
||||
* remaining seconds, and the two must come off the one {@link BackendQuarantine#status} call so
|
||||
* they can never disagree about which streak this is (see {@code profilesView}'s javadoc).
|
||||
*/
|
||||
@Test
|
||||
void profilesReportsQuarantineAttemptBesideRemainingSeconds() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
java.util.concurrent.atomic.AtomicLong now = new java.util.concurrent.atomic.AtomicLong(0L);
|
||||
BackendQuarantine quarantine = BackendQuarantine.withEscalation(now::get, TimeUnit.MINUTES.toNanos(30));
|
||||
quarantine.quarantine("shared-openai"); // attempt 1: 1800s
|
||||
now.set(TimeUnit.MINUTES.toNanos(30));
|
||||
quarantine.quarantine("shared-openai"); // attempt 2: 3600s
|
||||
FleetMcp.QuarantineSource source = new FleetMcp.QuarantineSource(
|
||||
profile -> "ltms-local".equals(profile) ? "shared-openai" : null, quarantine);
|
||||
McpSchema.CallToolResult res = FleetMcp.profiles(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), source);
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"quarantinedForSeconds\":3600"), out);
|
||||
assertTrue(out.contains("\"quarantineAttempt\":2"), out);
|
||||
}
|
||||
|
||||
/** A never-quarantined profile must not carry {@code quarantineAttempt} either. */
|
||||
@Test
|
||||
void profilesOmitsQuarantineAttemptWhenNotQuarantined() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = FleetMcp.profiles(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), FleetMcp.QuarantineSource.none());
|
||||
String out = textOf(res);
|
||||
assertFalse(out.contains("quarantineAttempt"), out);
|
||||
}
|
||||
|
||||
/** fleetd #201 Unit 5: {@code coolingOff} is a SEPARATE map from {@code quarantined}. */
|
||||
@Test
|
||||
void profilesReportsACoolingOffCredentialInASeparateMap() {
|
||||
@@ -1153,35 +1121,6 @@ class FleetMcpTest {
|
||||
assertTrue(out.contains("\"quarantinedForSeconds\":1200"), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #466 scope item 2: {@code fleet_list}'s capacity rows (CB-583: they reuse quarantine)
|
||||
* must carry {@code quarantineAttempt} beside {@code quarantinedForSeconds}, off the same
|
||||
* {@link BackendQuarantine#status} call as {@code fleet_profiles} -- see {@code capacityView}'s
|
||||
* comment pointing back to {@code profilesView}.
|
||||
*/
|
||||
@Test
|
||||
void capacityRowReportsQuarantineAttemptBesideRemainingSeconds() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
java.util.concurrent.atomic.AtomicLong now = new java.util.concurrent.atomic.AtomicLong(0L);
|
||||
BackendQuarantine quarantine = BackendQuarantine.withEscalation(now::get, TimeUnit.MINUTES.toNanos(20));
|
||||
quarantine.quarantine("shared-openai"); // attempt 1
|
||||
now.set(TimeUnit.MINUTES.toNanos(20));
|
||||
quarantine.quarantine("shared-openai"); // attempt 2
|
||||
now.set(TimeUnit.MINUTES.toNanos(60));
|
||||
quarantine.quarantine("shared-openai"); // attempt 3
|
||||
FleetMcp.QuarantineSource source = new FleetMcp.QuarantineSource(
|
||||
profile -> "terra".equals(profile) ? "shared-openai" : null, quarantine);
|
||||
|
||||
String out = textOf(FleetMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
sessions, null, new FleetMcp.CapacitySource(profile -> 0, profile -> 2,
|
||||
() -> Set.of("terra"), () -> 0), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
source, Map.of(), ""));
|
||||
|
||||
assertTrue(out.contains("\"credentialId\":\"shared-openai\""), out);
|
||||
assertTrue(out.contains("\"quarantineAttempt\":3"), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void everyProfileSharingTheQuarantinedCredentialReportsZeroFree() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
|
||||
@@ -762,6 +762,250 @@ class OpenCodeLauncherTest {
|
||||
"no IDE server when ideMcpUrl is unset");
|
||||
}
|
||||
|
||||
// --- fleetd #393: memberSkills seeding must actually reach an opencode member -------------------
|
||||
//
|
||||
// Before this fix, GitWorktrees#seedSkills copied skill folders into EVERY provisioned
|
||||
// worktree's .claude/skills/ and logged "skill seeding: N of M" regardless of which kind ended
|
||||
// up spawning into that worktree — a claim that held for kind: claude-code (the CLI discovers
|
||||
// that directory on its own) but was a guaranteed no-op for kind: opencode, which has no such
|
||||
// discovery. These tests drive the REAL GitWorktrees#add seeding path (not a hand-built
|
||||
// .claude/skills/ fixture), then spawn an opencode-kind member against the seeded worktree and
|
||||
// assert on what the member can actually consume — an instructions[] entry — not on the
|
||||
// seeding log alone. A minimal, non-hermetic git repo is enough here: unlike
|
||||
// GitWorktreesTest's own seeding tests, nothing in this file cares about core.excludesFile
|
||||
// composition, only about what lands in .claude/skills/ and whether OpenCodeLauncher reads it.
|
||||
|
||||
private static void git(Path cwd, String... args) throws Exception {
|
||||
Process p = new ProcessBuilder(prepend("git", args)).directory(cwd.toFile())
|
||||
.redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git timed out: git " + String.join(" ", args));
|
||||
assertEquals(0, p.exitValue(), "git " + String.join(" ", args) + " failed:\n" + out);
|
||||
}
|
||||
|
||||
private static List<String> prepend(String head, String... rest) {
|
||||
List<String> cmd = new ArrayList<>();
|
||||
cmd.add(head);
|
||||
cmd.addAll(List.of(rest));
|
||||
return cmd;
|
||||
}
|
||||
|
||||
private static Path initRepo(Path dir) throws Exception {
|
||||
Files.createDirectories(dir);
|
||||
git(dir, "init", "-q", "-b", "main");
|
||||
git(dir, "config", "user.email", "test@example.invalid");
|
||||
git(dir, "config", "user.name", "Test");
|
||||
Files.writeString(dir.resolve("README.md"), "seed\n");
|
||||
git(dir, "add", "README.md");
|
||||
git(dir, "commit", "-q", "-m", "seed");
|
||||
return dir;
|
||||
}
|
||||
|
||||
@Test
|
||||
void aSeededSkillReachesTheOpencodeMembersInstructionsArray(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Path skillsSource = tmp.resolve("skills-src");
|
||||
Path skillFile = skillsSource.resolve("implementer").resolve("SKILL.md");
|
||||
Files.createDirectories(skillFile.getParent());
|
||||
Files.writeString(skillFile, "IMPLEMENTER PROCEDURE\n");
|
||||
|
||||
// The real seeding path (fleetd #362), not a hand-built .claude/skills/ fixture — proves
|
||||
// OpenCodeLauncher reads what GitWorktrees#add actually produced.
|
||||
dev.ltms.fleet.session.GitWorktrees worktrees =
|
||||
new dev.ltms.fleet.session.GitWorktrees(tmp.resolve("wts").toString(), null, skillsSource.toString());
|
||||
String wt = worktrees.add(repo.toString(), "cb-393-opencode", "HEAD");
|
||||
Path seededSkillMd = Path.of(wt, ".claude", "skills", "implementer", "SKILL.md");
|
||||
assertTrue(Files.exists(seededSkillMd),
|
||||
"sanity: the real seeding step must have copied the skill into the worktree");
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(OpenCodeLauncher.class);
|
||||
Level original = logger.getLevel();
|
||||
logger.setLevel(Level.INFO);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
|
||||
String cfgPath;
|
||||
try {
|
||||
// No mcpUrl, no ideUrl, no fleet (no charter): the seeded skill alone must be enough to
|
||||
// trigger OPENCODE_CONFIG — proves the gate itself was updated, not only writeConfig's body.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Path configRoot = Files.createDirectory(tmp.resolve("configs"));
|
||||
service(herdr, configRoot, opencodeIdeCfg(null, null, wt)).spawn();
|
||||
cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(original);
|
||||
}
|
||||
|
||||
assertNotNull(cfgPath, "a seeded skill with nothing else configured must still write a config");
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
List<String> instructions = new ArrayList<>();
|
||||
json.path("instructions").forEach(n -> instructions.add(n.asText()));
|
||||
assertTrue(instructions.contains(seededSkillMd.toAbsolutePath().toString()),
|
||||
"the seeded skill's SKILL.md must be an instructions[] entry — got: " + instructions);
|
||||
|
||||
List<String> infos = appender.list.stream()
|
||||
.filter(e -> e.getLevel() == Level.INFO)
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.toList();
|
||||
assertTrue(infos.stream().anyMatch(m -> m.contains("skill delivery") && m.contains("1 of 1")),
|
||||
"the launcher must log, kind-aware, that it delivered the skill — got:\n" + infos);
|
||||
}
|
||||
|
||||
@Test
|
||||
void aSkillFolderWithoutSkillMdIsNeverDeliveredAndTheLogNamesIt(@TempDir Path tmp) throws Exception {
|
||||
Path wt = Files.createDirectories(tmp.resolve("wt"));
|
||||
Path goodSkill = wt.resolve(".claude").resolve("skills").resolve("implementer");
|
||||
Files.createDirectories(goodSkill);
|
||||
Files.writeString(goodSkill.resolve("SKILL.md"), "GOOD\n");
|
||||
Path halfShipped = wt.resolve(".claude").resolve("skills").resolve("half-shipped");
|
||||
Files.createDirectories(halfShipped);
|
||||
Files.writeString(halfShipped.resolve("README.md"), "no SKILL.md here\n");
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(OpenCodeLauncher.class);
|
||||
Level original = logger.getLevel();
|
||||
logger.setLevel(Level.INFO);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
|
||||
String cfgPath;
|
||||
try {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Path configRoot = Files.createDirectory(tmp.resolve("configs"));
|
||||
service(herdr, configRoot, opencodeIdeCfg(null, null, wt.toString())).spawn();
|
||||
cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(original);
|
||||
}
|
||||
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
List<String> instructions = new ArrayList<>();
|
||||
json.path("instructions").forEach(n -> instructions.add(n.asText()));
|
||||
assertTrue(instructions.contains(goodSkill.resolve("SKILL.md").toAbsolutePath().toString()),
|
||||
"the well-formed skill is still delivered alongside the malformed one");
|
||||
assertFalse(instructions.stream().anyMatch(i -> i.contains("half-shipped")),
|
||||
"a skill folder with no SKILL.md can never become an instructions[] entry");
|
||||
|
||||
List<String> infos = appender.list.stream()
|
||||
.filter(e -> e.getLevel() == Level.INFO)
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.toList();
|
||||
assertTrue(infos.stream().anyMatch(m -> m.contains("skill delivery") && m.contains("1 of 2")
|
||||
&& m.contains("half-shipped") && m.contains("could not be delivered")),
|
||||
"the log must say plainly which folder could not be consumed and why — got:\n" + infos);
|
||||
}
|
||||
|
||||
// --- fleetd #393 follow-up: instructions[] has three writers (charter, seeded skills, IDE
|
||||
// rules), and no test above ever exercises more than one or two of them together. A writer
|
||||
// that flips from withArray (get-or-create) to putArray (create-or-REPLACE) silently deletes
|
||||
// every entry written before it — proven live on this branch's merge: switching just the
|
||||
// skills writer to putArray left the entire suite (1603 tests) green while deleting the
|
||||
// charter entry an opencode member needs for its role contract. That hazard was found by the
|
||||
// fleet01 lead and independently verified against this branch; it is not a defect in the
|
||||
// skills-delivery or logging tests above, which both hold up under their own mutations — the
|
||||
// gap is that none of them combine all three writers in one config.
|
||||
//
|
||||
// These three tests assert instructions[] CONTENT as an exact, ordered list, not a size or a
|
||||
// "contains" check: a putArray mutation can replace N entries with a different N entries of
|
||||
// the same count, so only a content comparison can tell "all three paths present" apart from
|
||||
// "two paths present that replaced the earlier ones".
|
||||
|
||||
@Test
|
||||
void instructionsArrayHoldsExactlyTheCharterWhenNothingElseWritesToIt(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Fleet fleet = new FleetConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(),
|
||||
Map.of("dev", "role rule"), null);
|
||||
Path cwd = Files.createDirectory(root.resolve("checkout"));
|
||||
service(herdr, root, opencodeIdeCfg(null, null, cwd.toString()), () -> fleet).spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "a role charter alone still writes a config");
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
Path charter = Path.of(cfgPath).resolveSibling("member-charter.md");
|
||||
assertTrue(Files.exists(charter), "the charter file was written");
|
||||
|
||||
List<String> instructions = new ArrayList<>();
|
||||
json.path("instructions").forEach(n -> instructions.add(n.asText()));
|
||||
assertEquals(List.of(charter.toAbsolutePath().toString()), instructions,
|
||||
"with only the charter writer active, instructions[] holds exactly one entry: the "
|
||||
+ "charter — got: " + instructions);
|
||||
}
|
||||
|
||||
@Test
|
||||
void instructionsArrayHoldsCharterThenIdeRulesInOrder(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Fleet fleet = new FleetConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(),
|
||||
Map.of("dev", "role rule"), null);
|
||||
Path cwd = Files.createDirectory(root.resolve("checkout"));
|
||||
service(herdr, root, opencodeIdeCfg(null,
|
||||
"http://127.0.0.1:29170/index-mcp/streamable-http", cwd.toString()), () -> fleet).spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "charter + IDE rules still writes a config");
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
Path charter = Path.of(cfgPath).resolveSibling("member-charter.md");
|
||||
Path rules = Path.of(cfgPath).resolveSibling("ide-rules.md");
|
||||
assertTrue(Files.exists(charter), "the charter file was written");
|
||||
assertTrue(Files.exists(rules), "the ide-rules file was written");
|
||||
|
||||
List<String> instructions = new ArrayList<>();
|
||||
json.path("instructions").forEach(n -> instructions.add(n.asText()));
|
||||
assertEquals(List.of(charter.toAbsolutePath().toString(), rules.toAbsolutePath().toString()),
|
||||
instructions,
|
||||
"with charter + IDE-rules writers active, instructions[] holds both, charter first — "
|
||||
+ "got: " + instructions);
|
||||
}
|
||||
|
||||
@Test
|
||||
void instructionsArrayHoldsCharterThenSkillsThenIdeRulesInOrder(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Path skillsSource = tmp.resolve("skills-src");
|
||||
Path skillFile = skillsSource.resolve("implementer").resolve("SKILL.md");
|
||||
Files.createDirectories(skillFile.getParent());
|
||||
Files.writeString(skillFile, "IMPLEMENTER PROCEDURE\n");
|
||||
|
||||
dev.ltms.fleet.session.GitWorktrees worktrees = new dev.ltms.fleet.session.GitWorktrees(
|
||||
tmp.resolve("wts").toString(), null, skillsSource.toString());
|
||||
String wt = worktrees.add(repo.toString(), "cb-393-follow-up", "HEAD");
|
||||
Path seededSkillMd = Path.of(wt, ".claude", "skills", "implementer", "SKILL.md");
|
||||
assertTrue(Files.exists(seededSkillMd), "sanity: the real seeding step copied the skill");
|
||||
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Fleet fleet = new FleetConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(),
|
||||
Map.of("dev", "role rule"), null);
|
||||
Path configRoot = Files.createDirectory(tmp.resolve("configs"));
|
||||
service(herdr, configRoot, opencodeIdeCfg(null,
|
||||
"http://127.0.0.1:29170/index-mcp/streamable-http", wt), () -> fleet).spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "charter + skills + IDE rules still writes a config");
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
Path charter = Path.of(cfgPath).resolveSibling("member-charter.md");
|
||||
Path rules = Path.of(cfgPath).resolveSibling("ide-rules.md");
|
||||
assertTrue(Files.exists(charter), "the charter file was written");
|
||||
assertTrue(Files.exists(rules), "the ide-rules file was written");
|
||||
|
||||
List<String> instructions = new ArrayList<>();
|
||||
json.path("instructions").forEach(n -> instructions.add(n.asText()));
|
||||
// Exact ordered list, not size or "contains": a putArray mutation on any writer after the
|
||||
// charter replaces every entry written before it, and the replacement can still be a
|
||||
// plausible-looking array of a different shape. This is the one combination all three
|
||||
// writers are active for — and per fleet01 the realistic shape on a host where weighted
|
||||
// placement makes opencode the default for most members.
|
||||
assertEquals(List.of(charter.toAbsolutePath().toString(),
|
||||
seededSkillMd.toAbsolutePath().toString(),
|
||||
rules.toAbsolutePath().toString()),
|
||||
instructions,
|
||||
"with all three writers active, instructions[] must hold charter, then the seeded "
|
||||
+ "skill, then IDE rules — in that order and with nothing replaced. A "
|
||||
+ "putArray mutation on any writer after the charter would silently drop "
|
||||
+ "earlier entries here while still producing a same-shaped array — got: "
|
||||
+ instructions);
|
||||
}
|
||||
|
||||
// --- fleetd #219: config root + discovery root under memberHerdrSocket ------------------------
|
||||
|
||||
/** A config with {@code memberHerdrSocket:} set, and optionally {@code worktreeRoot:}/{@code worktreeGroup:}. */
|
||||
|
||||
@@ -116,262 +116,4 @@ class BackendQuarantineTest {
|
||||
assertThrows(IllegalArgumentException.class, () -> new BackendQuarantine(() -> 0L, 0L));
|
||||
assertThrows(IllegalArgumentException.class, () -> new BackendQuarantine(() -> 0L, -1L));
|
||||
}
|
||||
|
||||
// --- fleetd #466: escalating cooldown -----------------------------------------------------
|
||||
//
|
||||
// Base cooldown 600s (10 min), multiplier 2.0, ceiling 2400s (4x base) — small round numbers
|
||||
// chosen so every deadline is an exact assertion, not just "greater than before". Each call
|
||||
// below lands at or before the previous deadline (a zero or negative gap), which is always
|
||||
// "no more than one base cooldown after the previous deadline" — i.e. every call continues the
|
||||
// same streak, matching a credential that keeps reporting exhausted with no lull.
|
||||
|
||||
@Test
|
||||
void anInvalidBackoffMultiplierIsRejected() {
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30), 0.5, TimeUnit.HOURS.toNanos(6)));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aCeilingBelowTheBaseCooldownIsRejected() {
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30), 2.0, TimeUnit.MINUTES.toNanos(10)));
|
||||
}
|
||||
|
||||
@Test
|
||||
void repeatedExhaustionEscalatesTheCooldownByExactAmounts() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
|
||||
TimeUnit.SECONDS.toNanos(2400));
|
||||
|
||||
q.quarantine("shared-openai"); // 1st: base cooldown
|
||||
assertEquals(OptionalLong.of(600L), q.remainingSeconds("shared-openai"));
|
||||
|
||||
now.set(TimeUnit.SECONDS.toNanos(100)); // still inside the 1st quarantine (deadline 600s)
|
||||
q.quarantine("shared-openai"); // 2nd: 600 * 2^1 = 1200
|
||||
assertEquals(OptionalLong.of(1200L), q.remainingSeconds("shared-openai"),
|
||||
"a second consecutive exhaustion must double the cooldown, not just increase it");
|
||||
|
||||
now.set(TimeUnit.SECONDS.toNanos(1300)); // exactly the 2nd deadline (100 + 1200)
|
||||
q.quarantine("shared-openai"); // 3rd: 600 * 2^2 = 2400 (exactly at the ceiling)
|
||||
assertEquals(OptionalLong.of(2400L), q.remainingSeconds("shared-openai"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void escalationStopsAtTheCeiling() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
|
||||
TimeUnit.SECONDS.toNanos(2400));
|
||||
|
||||
q.quarantine("shared-openai"); // 1st: 600
|
||||
now.set(TimeUnit.SECONDS.toNanos(600));
|
||||
q.quarantine("shared-openai"); // 2nd: 1200, deadline 1800
|
||||
now.set(TimeUnit.SECONDS.toNanos(1800));
|
||||
q.quarantine("shared-openai"); // 3rd: 600 * 4 = 2400, at the ceiling, deadline 4200
|
||||
now.set(TimeUnit.SECONDS.toNanos(4200));
|
||||
q.quarantine("shared-openai"); // 4th: 600 * 8 = 4800 uncapped, must stay capped at 2400
|
||||
assertEquals(OptionalLong.of(2400L), q.remainingSeconds("shared-openai"),
|
||||
"the cooldown must never exceed the configured ceiling, however long the streak gets");
|
||||
|
||||
now.set(TimeUnit.SECONDS.toNanos(6600)); // 4th deadline
|
||||
q.quarantine("shared-openai"); // 5th: still capped
|
||||
assertEquals(OptionalLong.of(2400L), q.remainingSeconds("shared-openai"),
|
||||
"pushing well past the ceiling must not budge it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aQuietGapLongerThanTheBaseCooldownResetsToTheBaseCooldown() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
|
||||
TimeUnit.SECONDS.toNanos(2400));
|
||||
|
||||
q.quarantine("shared-openai"); // 1st: 600, deadline 600
|
||||
now.set(TimeUnit.SECONDS.toNanos(600));
|
||||
q.quarantine("shared-openai"); // 2nd: 1200, deadline 1800
|
||||
now.set(TimeUnit.SECONDS.toNanos(1800));
|
||||
q.quarantine("shared-openai"); // 3rd: 2400, deadline 4200
|
||||
assertEquals(OptionalLong.of(2400L), q.remainingSeconds("shared-openai"));
|
||||
|
||||
// Quiet for well over one base cooldown (600s) past the 3rd deadline (4200s).
|
||||
now.set(TimeUnit.SECONDS.toNanos(20_000));
|
||||
q.quarantine("shared-openai"); // treated as a fresh occurrence
|
||||
assertEquals(OptionalLong.of(600L), q.remainingSeconds("shared-openai"),
|
||||
"a long quiet gap must reset the streak back to the base cooldown");
|
||||
}
|
||||
|
||||
@Test
|
||||
void escalatingOneCredentialDoesNotSlowAnother() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
|
||||
TimeUnit.SECONDS.toNanos(2400));
|
||||
|
||||
q.quarantine("shared-openai"); // 1st: 600
|
||||
now.set(TimeUnit.SECONDS.toNanos(600));
|
||||
q.quarantine("shared-openai"); // 2nd: 1200
|
||||
now.set(TimeUnit.SECONDS.toNanos(1800));
|
||||
q.quarantine("shared-openai"); // 3rd: 2400 — three-in-a-row streak on this credential only
|
||||
|
||||
q.quarantine("another-credential"); // its first and only exhaustion
|
||||
assertEquals(OptionalLong.of(600L), q.remainingSeconds("another-credential"),
|
||||
"an unrelated credential's cooldown must stay at the base rate, unaffected by a sibling's streak");
|
||||
}
|
||||
|
||||
@Test
|
||||
void withEscalationDefaultsToDoublingCappedAtTwelveTimesTheBase() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendQuarantine q = BackendQuarantine.withEscalation(now::get, TimeUnit.MINUTES.toNanos(30));
|
||||
|
||||
q.quarantine("shared-openai");
|
||||
assertEquals(OptionalLong.of(1800L), q.remainingSeconds("shared-openai"),
|
||||
"the first occurrence must still use the base cooldown");
|
||||
|
||||
now.set(TimeUnit.MINUTES.toNanos(30));
|
||||
q.quarantine("shared-openai");
|
||||
assertEquals(OptionalLong.of(3600L), q.remainingSeconds("shared-openai"),
|
||||
"the default multiplier must be 2.0");
|
||||
}
|
||||
|
||||
// --- fleetd #466 scope item 2: status() reports repeatCount beside remainingSeconds ------------
|
||||
|
||||
@Test
|
||||
void statusReportsAttemptOneForAFirstOccurrenceNeverAbsentOrZero() {
|
||||
BackendQuarantine q = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30), 2.0,
|
||||
TimeUnit.HOURS.toNanos(6));
|
||||
q.quarantine("shared-openai");
|
||||
|
||||
BackendQuarantine.Status status = q.status("shared-openai").orElseThrow();
|
||||
assertEquals(1800L, status.remainingSeconds());
|
||||
assertEquals(1, status.repeatCount(),
|
||||
"a first-ever occurrence must report attempt 1, not 0 or absent -- 1 means unambiguously "
|
||||
+ "'the first time', where 0 would be indistinguishable from a bug that forgot to count");
|
||||
}
|
||||
|
||||
@Test
|
||||
void statusIsAbsentWhenTheCredentialIsNotQuarantined() {
|
||||
BackendQuarantine q = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30), 2.0,
|
||||
TimeUnit.HOURS.toNanos(6));
|
||||
assertTrue(q.status("shared-openai").isEmpty());
|
||||
}
|
||||
|
||||
/**
|
||||
* The acceptance criterion's strong form: the reported {@code repeatCount} must match the exact
|
||||
* step the cooldown's own growth implies, read off {@link BackendQuarantine#status}'s single
|
||||
* call -- not two independent reads that happen to agree in this easy case.
|
||||
*/
|
||||
@Test
|
||||
void statusReportsTheGrowingAttemptCountAlongsideTheEscalatingCooldown() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
|
||||
TimeUnit.SECONDS.toNanos(2400));
|
||||
|
||||
q.quarantine("shared-openai"); // 1st: 600s, attempt 1
|
||||
BackendQuarantine.Status first = q.status("shared-openai").orElseThrow();
|
||||
assertEquals(600L, first.remainingSeconds());
|
||||
assertEquals(1, first.repeatCount());
|
||||
|
||||
now.set(TimeUnit.SECONDS.toNanos(600));
|
||||
q.quarantine("shared-openai"); // 2nd: 1200s, attempt 2
|
||||
BackendQuarantine.Status second = q.status("shared-openai").orElseThrow();
|
||||
assertEquals(1200L, second.remainingSeconds());
|
||||
assertEquals(2, second.repeatCount());
|
||||
|
||||
now.set(TimeUnit.SECONDS.toNanos(1800));
|
||||
q.quarantine("shared-openai"); // 3rd: 2400s (at the ceiling), attempt 3
|
||||
BackendQuarantine.Status third = q.status("shared-openai").orElseThrow();
|
||||
assertEquals(2400L, third.remainingSeconds());
|
||||
assertEquals(3, third.repeatCount());
|
||||
}
|
||||
|
||||
/**
|
||||
* Once the cooldown hits its ceiling, every further consecutive exhaustion reports the SAME
|
||||
* {@code remainingSeconds} -- so a {@code repeatCount} re-derived from the cooldown value (e.g.
|
||||
* inverting {@code cooldownNanos * multiplier^(n-1)}) could not tell attempt 4 from attempt 9;
|
||||
* only the stored counter can. This is the scenario that makes "read the count off a second,
|
||||
* independent computation" provably wrong rather than just risky.
|
||||
*/
|
||||
@Test
|
||||
void repeatCountKeepsGrowingPastTheCeilingEvenThoughTheCooldownStaysFlat() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
|
||||
TimeUnit.SECONDS.toNanos(2400));
|
||||
|
||||
long t = 0L;
|
||||
for (int attempt = 1; attempt <= 5; attempt++) {
|
||||
now.set(t);
|
||||
q.quarantine("shared-openai");
|
||||
BackendQuarantine.Status status = q.status("shared-openai").orElseThrow();
|
||||
assertEquals(attempt, status.repeatCount(),
|
||||
"attempt " + attempt + " must be reported as exactly " + attempt
|
||||
+ ", not collapsed to whatever attempt first reached the ceiling");
|
||||
if (attempt >= 3) {
|
||||
assertEquals(2400L, status.remainingSeconds(), "attempt " + attempt + " must be capped");
|
||||
}
|
||||
t += status.remainingSeconds(); // land exactly on the next deadline: still the same streak
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aQuietGapResetsTheReportedAttemptCountToOneToo() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
|
||||
TimeUnit.SECONDS.toNanos(2400));
|
||||
|
||||
q.quarantine("shared-openai");
|
||||
now.set(TimeUnit.SECONDS.toNanos(600));
|
||||
q.quarantine("shared-openai");
|
||||
now.set(TimeUnit.SECONDS.toNanos(1800));
|
||||
q.quarantine("shared-openai"); // attempt 3
|
||||
assertEquals(3, q.status("shared-openai").orElseThrow().repeatCount());
|
||||
|
||||
now.set(TimeUnit.SECONDS.toNanos(20_000)); // long quiet gap
|
||||
q.quarantine("shared-openai");
|
||||
assertEquals(1, q.status("shared-openai").orElseThrow().repeatCount(),
|
||||
"a reset streak must report attempt 1 again, matching the reset base cooldown");
|
||||
}
|
||||
|
||||
@Test
|
||||
void escalatingOneCredentialsAttemptCountDoesNotAffectAnother() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
|
||||
TimeUnit.SECONDS.toNanos(2400));
|
||||
|
||||
q.quarantine("shared-openai");
|
||||
now.set(TimeUnit.SECONDS.toNanos(600));
|
||||
q.quarantine("shared-openai");
|
||||
now.set(TimeUnit.SECONDS.toNanos(1800));
|
||||
q.quarantine("shared-openai"); // 3-in-a-row streak on this credential only
|
||||
|
||||
q.quarantine("another-credential");
|
||||
assertEquals(1, q.status("another-credential").orElseThrow().repeatCount(),
|
||||
"an unrelated credential's attempt count must stay at 1, unaffected by a sibling's streak");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noneReportsNoStatusForAnything() {
|
||||
BackendQuarantine q = BackendQuarantine.none();
|
||||
q.quarantine("shared-openai"); // no-op on none(), same as every other mutator
|
||||
|
||||
assertTrue(q.status("shared-openai").isEmpty(),
|
||||
"none() quarantines nothing, so it must report no status at all -- never a fabricated "
|
||||
+ "attempt count for a credential that was never actually quarantined");
|
||||
}
|
||||
|
||||
/** The flat (non-escalating) two-argument constructor must still report a real, growing count. */
|
||||
@Test
|
||||
void aFlatTwoArgumentInstanceStillReportsAGrowingAttemptCountEvenThoughTheCooldownStaysFlat() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600));
|
||||
|
||||
q.quarantine("shared-openai");
|
||||
BackendQuarantine.Status first = q.status("shared-openai").orElseThrow();
|
||||
assertEquals(600L, first.remainingSeconds());
|
||||
assertEquals(1, first.repeatCount());
|
||||
|
||||
now.set(TimeUnit.SECONDS.toNanos(100)); // still inside the 1st quarantine
|
||||
q.quarantine("shared-openai");
|
||||
BackendQuarantine.Status second = q.status("shared-openai").orElseThrow();
|
||||
assertEquals(600L, second.remainingSeconds(),
|
||||
"the flat constructor's cooldown must stay exactly the base length regardless of the streak");
|
||||
assertEquals(2, second.repeatCount(),
|
||||
"the flat constructor still counts the real streak -- it just does not scale the cooldown by it");
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user