diff --git a/fleetd/fleetd.example.yaml b/fleetd/fleetd.example.yaml index 08e97f9..cbf3240 100644 --- a/fleetd/fleetd.example.yaml +++ b/fleetd/fleetd.example.yaml @@ -452,6 +452,15 @@ placement: weighted # seconds, before a spawn may land on it again. Applies to every profile's effective credential # (its own name, or its credentialId if set above) — there is no per-profile override. Default # 1800 (30 minutes) when omitted or non-positive. +# +# fleetd #466: this is now only the BASE of an escalating backoff, not a flat retry rate. A +# credential quarantined again within one base cooldown of the previous quarantine ending (still +# reporting exhausted — e.g. a weekly subscription limit that hasn't reset) backs off further: +# cooldown doubles each such time, capped at 12x this value (~6 hours at the 1800s default). A +# quarantine that starts after a base-cooldown's worth of quiet resets back to this value. Not +# configurable per se — the multiplier and ceiling are constants in BackendQuarantine, not new +# YAML keys; see its class doc for the exact formula and why there is no automatic probe to clear +# it early (the operator's own design constraint — a probe spends the quota it's measuring). # DEFERRED: baked once into the BackendQuarantine built at startup — a running quarantine keeps # its original cooldown regardless; a new value only applies to a quarantine that starts after a # restart. Editing this needs a daemon restart to take effect. diff --git a/fleetd/src/main/java/dev/ltms/fleet/Fleetd.java b/fleetd/src/main/java/dev/ltms/fleet/Fleetd.java index 8cb9852..14420b9 100644 --- a/fleetd/src/main/java/dev/ltms/fleet/Fleetd.java +++ b/fleetd/src/main/java/dev/ltms/fleet/Fleetd.java @@ -222,7 +222,13 @@ public final class Fleetd { // (checked at spawn) and the exhaustion sink wired in below (written on BACKEND_EXHAUSTED). // The cooldown is deferred (see FleetConfig#quarantineCooldownSeconds): it is read once // here, at startup, and a config reload only changes it for a daemon restart. - BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime, + // fleetd #466: escalating, not flat — a credential that keeps reporting exhaustion (e.g. a + // weekly subscription limit, which would otherwise be retried on every ~30-minute cooldown, + // about 336 times across the week) backs off further each consecutive time, capped at + // BackendQuarantine.DEFAULT_MAX_COOLDOWN_MULTIPLE x the base cooldown. See BackendQuarantine's + // class doc for the mechanism, why this never fires on cooling-off (a separate, unescalated + // mechanism — BackendOutagePolicy below), and the reset. + BackendQuarantine quarantine = BackendQuarantine.withEscalation(System::nanoTime, TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds())); // fleetd #201 Unit 5: one outage-cool-off tracker for the whole daemon, shared between the // launcher (checked at spawn, like `quarantine` above) and the backend-error sink wired in diff --git a/fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java b/fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java index de64b31..9b9504d 100644 --- a/fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java +++ b/fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java @@ -70,12 +70,14 @@ import java.util.regex.PatternSyntaxException; * {@code fixed} (default), {@code round-robin}, or {@code weighted} * @param auth API authentication mode ({@code null} → {@code loopback-trust}, the * historical behaviour), CB-501 - * @param quarantineCooldownSeconds how long a credential stays quarantined after a + * @param quarantineCooldownSeconds the BASE cooldown a credential is quarantined for after a * {@code BACKEND_EXHAUSTED} classification (CB-578 stage B); {@code null}/{@code - * <=0} → {@link #DEFAULT_QUARANTINE_COOLDOWN_SECONDS}. Baked once into the - * {@code BackendQuarantine} built at startup, so it is DEFERRED: changing it - * needs a restart, and a quarantine already running keeps whatever cooldown was - * live when it started. + * <=0} → {@link #DEFAULT_QUARANTINE_COOLDOWN_SECONDS}. Since fleetd #466 this is + * only the first occurrence's length — a credential quarantined again shortly + * after this cooldown ends backs off further, up to a ceiling; see {@code + * BackendQuarantine}'s class doc. Baked once into the {@code BackendQuarantine} + * built at startup, so it is DEFERRED: changing it needs a restart, and a + * quarantine already running keeps whatever cooldown was live when it started. * @param memberCredentials deny-by-default policy (CB-596) for which of the operator's own host * credentials a spawned member's pane inherits. {@code null} (the block * omitted) blocks nothing — see {@link MemberCredentials}. diff --git a/fleetd/src/main/java/dev/ltms/fleet/placement/BackendQuarantine.java b/fleetd/src/main/java/dev/ltms/fleet/placement/BackendQuarantine.java index 3fb437a..48b406b 100644 --- a/fleetd/src/main/java/dev/ltms/fleet/placement/BackendQuarantine.java +++ b/fleetd/src/main/java/dev/ltms/fleet/placement/BackendQuarantine.java @@ -21,46 +21,158 @@ import java.util.function.LongSupplier; *
The clock is injected ({@link LongSupplier}, conventionally {@code System::nanoTime} like * {@code FleetHealthMonitor}), never read inline, so a quarantine's expiry is testable without a * real sleep. + * + *
This class only ever sees the exhaustion signal. Its only production caller is + * {@code Fleetd.exhaustionSink}, wired to fire on a {@code BACKEND_EXHAUSTED} classification alone. + * The daemon's other outage state — a credential "cooling off" after repeated non-exhaustion + * backend errors (an HTTP 5xx storm, say) — is a separate mechanism, {@code BackendOutagePolicy}, + * with its own short fixed 60s cooldown and no repeat tracking. The two are never merged: escalating + * on a cooling-off signal would turn a transient 5xx storm into a multi-hour backoff, which is + * exactly the failure this ticket is not asking for. Confirmed by reading every call site of + * {@link #quarantine} — {@code BackendOutagePolicy} has its own {@code coolOff} method and never + * calls this one. + * + *
Mechanism — the {@link #withEscalation} constructors track, per credential, how + * many times in a row {@link #quarantine} has been called without an intervening "quiet" gap. + * Each call computes {@code cooldownNanos * backoffMultiplier ^ (repeatCount - 1)}, capped at + * {@code maxCooldownNanos}. A call counts as a continuation of the same streak — {@code repeatCount} + * increments — when it arrives no more than one base {@code cooldownNanos} after the previous + * quarantine's deadline (this covers both "still quarantined" and "quarantine just expired and it + * was exhausted again immediately"); otherwise the streak resets and this call is treated as a fresh + * first occurrence at the base cooldown. + * + *
Reset, honestly stated. The ideal reset signal is "the cooldown expired and the + * next attempt succeeded" — but nothing in this codebase reports a spawn success back to this class + * (checked: {@code SessionManager} and {@code CompositePeerLauncher} never call any method here + * except {@link #quarantine}/{@link #isQuarantined}/{@link #remainingSeconds}, none of which is a + * success hook). Lacking that signal, the reset used here is a time-based proxy: a base-cooldown's + * worth of quiet — no exhaustion report for that credential — since the last quarantine ended. It is + * not proof the credential started working again, only the best available evidence without adding an + * active probe, which is out of scope by the operator's own design constraint (no automatic probing + * of a limited backend). + * + *
Ceiling. {@code maxCooldownNanos} bounds the growth — an unbounded backoff is a + * permanent, unrecoverable-without-a-restart outage, which would be worse than the flat-rate bug this + * escalation fixes. {@link #withEscalation(LongSupplier, long)} defaults the ceiling to + * {@value #DEFAULT_MAX_COOLDOWN_MULTIPLE}x the base cooldown (12x the 1800s default ≈ 6 hours), so a + * chronically exhausted credential still gets re-tried roughly every 6 hours instead of every 30 + * minutes — about a dozen attempts a week instead of ~336. + * + *
Backward compatibility. The original two-argument {@link #BackendQuarantine(
+ * LongSupplier, long)} constructor is unchanged in behaviour: it is exactly {@code
+ * withEscalation}'s mechanism with {@code backoffMultiplier = 1.0} and {@code maxCooldownNanos =
+ * cooldownNanos}, which collapses the formula back to the original flat {@code now + cooldownNanos}
+ * on every call regardless of history. Every existing call site (roughly 20 across the test suite,
+ * plus {@link #none()}) keeps its current shape and behaviour unchanged.
*/
public final class BackendQuarantine {
- private final ConcurrentHashMap On {@link #none()} this is a no-op. It has to be: that instance holds a clock frozen at 0,
* so recording a deadline would produce a quarantine that never expires — a credential locked out
@@ -73,7 +185,13 @@ public final class BackendQuarantine {
if (inert) {
return;
}
- quarantinedUntilNanos.put(credentialId, nowNanos.getAsLong() + cooldownNanos);
+ long now = nowNanos.getAsLong();
+ quarantines.compute(credentialId, (id, prev) -> {
+ int repeatCount = (prev == null || now - prev.deadlineNanos() > cooldownNanos)
+ ? 1
+ : prev.repeatCount() + 1;
+ return new QuarantineState(repeatCount, now + escalatedCooldownNanos(repeatCount));
+ });
}
/** Whether {@code credentialId} is quarantined right now. */
@@ -95,8 +213,8 @@ public final class BackendQuarantine {
*/
public Map Measured directly: reverting {@code main} to {@code new BackendQuarantine(System::nanoTime,
+ * TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()))} — the pre-#466 flat call — compiles
+ * with 0 errors and leaves the entire 1608-test suite (including every {@code BackendQuarantineTest}
+ * case) green, because no other test constructs its {@code BackendQuarantine} through {@code main};
+ * every one of them builds its own instance directly. That silent regression is exactly the shape
+ * {@link FleetdLeadSeatWiringTest} and {@link FleetdCompletionResolverWiringTest} already guard
+ * against for their own constructor arguments — this is the same class of gap for fleetd #466's
+ * factory choice, following their approach.
+ *
+ * This test checks source text, not runtime behaviour. It never constructs a {@code
+ * BackendQuarantine} and never runs {@code main} — a green result here proves only that the exact
+ * text {@code main} calls {@code BackendQuarantine.withEscalation(...)} rather than the flat
+ * constructor. It does not prove that call actually executes at startup (no test here starts the
+ * daemon), and it does not prove the escalation reaches a real backend or credential — only
+ * {@code BackendQuarantineTest} proves the factory's own behaviour, and only a live daemon proves
+ * the wiring runs.
+ */
+class FleetdBackendQuarantineWiringTest {
+
+ private static String fleetdSource() throws Exception {
+ return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
+ }
+
+ @Test
+ @DisplayName("[SOURCE TEXT] main's BackendQuarantine local is still built from BackendQuarantine.withEscalation(...)")
+ void mainStillWiresTheEscalatingQuarantineFactory() throws Exception {
+ String source = fleetdSource();
+ assertTrue(source.contains(
+ "BackendQuarantine quarantine = BackendQuarantine.withEscalation(System::nanoTime,\n"
+ + " TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));"),
+ "Fleetd.main's BackendQuarantine local must still be built from "
+ + "BackendQuarantine.withEscalation(System::nanoTime, "
+ + "TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds())). Reverting to the flat "
+ + "two-argument constructor (fleetd #466's measured regression) compiles with 0 errors "
+ + "and leaves the whole suite green, including every BackendQuarantineTest case that "
+ + "proves the escalation itself works — this source check is what must go red instead. "
+ + "A reverted daemon would go back to retrying a weekly subscription limit on every "
+ + "flat ~30-minute cooldown, about 336 times across the week.");
+
+ // Negative form of the same check: the pre-#466 flat call, if it ever reappears at this
+ // declaration, must not be mistaken for the escalating one by a looser positive-only check.
+ assertFalse(source.contains(
+ "BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,\n"
+ + " TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));"),
+ "main's BackendQuarantine local must never regress to the flat two-argument constructor");
+ }
+}
diff --git a/fleetd/src/test/java/dev/ltms/fleet/placement/BackendQuarantineTest.java b/fleetd/src/test/java/dev/ltms/fleet/placement/BackendQuarantineTest.java
index 4a38d01..450ddcb 100644
--- a/fleetd/src/test/java/dev/ltms/fleet/placement/BackendQuarantineTest.java
+++ b/fleetd/src/test/java/dev/ltms/fleet/placement/BackendQuarantineTest.java
@@ -116,4 +116,117 @@ class BackendQuarantineTest {
assertThrows(IllegalArgumentException.class, () -> new BackendQuarantine(() -> 0L, 0L));
assertThrows(IllegalArgumentException.class, () -> new BackendQuarantine(() -> 0L, -1L));
}
+
+ // --- fleetd #466: escalating cooldown -----------------------------------------------------
+ //
+ // Base cooldown 600s (10 min), multiplier 2.0, ceiling 2400s (4x base) — small round numbers
+ // chosen so every deadline is an exact assertion, not just "greater than before". Each call
+ // below lands at or before the previous deadline (a zero or negative gap), which is always
+ // "no more than one base cooldown after the previous deadline" — i.e. every call continues the
+ // same streak, matching a credential that keeps reporting exhausted with no lull.
+
+ @Test
+ void anInvalidBackoffMultiplierIsRejected() {
+ assertThrows(IllegalArgumentException.class,
+ () -> new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30), 0.5, TimeUnit.HOURS.toNanos(6)));
+ }
+
+ @Test
+ void aCeilingBelowTheBaseCooldownIsRejected() {
+ assertThrows(IllegalArgumentException.class,
+ () -> new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30), 2.0, TimeUnit.MINUTES.toNanos(10)));
+ }
+
+ @Test
+ void repeatedExhaustionEscalatesTheCooldownByExactAmounts() {
+ AtomicLong now = new AtomicLong(0L);
+ BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
+ TimeUnit.SECONDS.toNanos(2400));
+
+ q.quarantine("shared-openai"); // 1st: base cooldown
+ assertEquals(OptionalLong.of(600L), q.remainingSeconds("shared-openai"));
+
+ now.set(TimeUnit.SECONDS.toNanos(100)); // still inside the 1st quarantine (deadline 600s)
+ q.quarantine("shared-openai"); // 2nd: 600 * 2^1 = 1200
+ assertEquals(OptionalLong.of(1200L), q.remainingSeconds("shared-openai"),
+ "a second consecutive exhaustion must double the cooldown, not just increase it");
+
+ now.set(TimeUnit.SECONDS.toNanos(1300)); // exactly the 2nd deadline (100 + 1200)
+ q.quarantine("shared-openai"); // 3rd: 600 * 2^2 = 2400 (exactly at the ceiling)
+ assertEquals(OptionalLong.of(2400L), q.remainingSeconds("shared-openai"));
+ }
+
+ @Test
+ void escalationStopsAtTheCeiling() {
+ AtomicLong now = new AtomicLong(0L);
+ BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
+ TimeUnit.SECONDS.toNanos(2400));
+
+ q.quarantine("shared-openai"); // 1st: 600
+ now.set(TimeUnit.SECONDS.toNanos(600));
+ q.quarantine("shared-openai"); // 2nd: 1200, deadline 1800
+ now.set(TimeUnit.SECONDS.toNanos(1800));
+ q.quarantine("shared-openai"); // 3rd: 600 * 4 = 2400, at the ceiling, deadline 4200
+ now.set(TimeUnit.SECONDS.toNanos(4200));
+ q.quarantine("shared-openai"); // 4th: 600 * 8 = 4800 uncapped, must stay capped at 2400
+ assertEquals(OptionalLong.of(2400L), q.remainingSeconds("shared-openai"),
+ "the cooldown must never exceed the configured ceiling, however long the streak gets");
+
+ now.set(TimeUnit.SECONDS.toNanos(6600)); // 4th deadline
+ q.quarantine("shared-openai"); // 5th: still capped
+ assertEquals(OptionalLong.of(2400L), q.remainingSeconds("shared-openai"),
+ "pushing well past the ceiling must not budge it");
+ }
+
+ @Test
+ void aQuietGapLongerThanTheBaseCooldownResetsToTheBaseCooldown() {
+ AtomicLong now = new AtomicLong(0L);
+ BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
+ TimeUnit.SECONDS.toNanos(2400));
+
+ q.quarantine("shared-openai"); // 1st: 600, deadline 600
+ now.set(TimeUnit.SECONDS.toNanos(600));
+ q.quarantine("shared-openai"); // 2nd: 1200, deadline 1800
+ now.set(TimeUnit.SECONDS.toNanos(1800));
+ q.quarantine("shared-openai"); // 3rd: 2400, deadline 4200
+ assertEquals(OptionalLong.of(2400L), q.remainingSeconds("shared-openai"));
+
+ // Quiet for well over one base cooldown (600s) past the 3rd deadline (4200s).
+ now.set(TimeUnit.SECONDS.toNanos(20_000));
+ q.quarantine("shared-openai"); // treated as a fresh occurrence
+ assertEquals(OptionalLong.of(600L), q.remainingSeconds("shared-openai"),
+ "a long quiet gap must reset the streak back to the base cooldown");
+ }
+
+ @Test
+ void escalatingOneCredentialDoesNotSlowAnother() {
+ AtomicLong now = new AtomicLong(0L);
+ BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
+ TimeUnit.SECONDS.toNanos(2400));
+
+ q.quarantine("shared-openai"); // 1st: 600
+ now.set(TimeUnit.SECONDS.toNanos(600));
+ q.quarantine("shared-openai"); // 2nd: 1200
+ now.set(TimeUnit.SECONDS.toNanos(1800));
+ q.quarantine("shared-openai"); // 3rd: 2400 — three-in-a-row streak on this credential only
+
+ q.quarantine("another-credential"); // its first and only exhaustion
+ assertEquals(OptionalLong.of(600L), q.remainingSeconds("another-credential"),
+ "an unrelated credential's cooldown must stay at the base rate, unaffected by a sibling's streak");
+ }
+
+ @Test
+ void withEscalationDefaultsToDoublingCappedAtTwelveTimesTheBase() {
+ AtomicLong now = new AtomicLong(0L);
+ BackendQuarantine q = BackendQuarantine.withEscalation(now::get, TimeUnit.MINUTES.toNanos(30));
+
+ q.quarantine("shared-openai");
+ assertEquals(OptionalLong.of(1800L), q.remainingSeconds("shared-openai"),
+ "the first occurrence must still use the base cooldown");
+
+ now.set(TimeUnit.MINUTES.toNanos(30));
+ q.quarantine("shared-openai");
+ assertEquals(OptionalLong.of(3600L), q.remainingSeconds("shared-openai"),
+ "the default multiplier must be 2.0");
+ }
}