diff --git a/fleetd/fleetd.example.yaml b/fleetd/fleetd.example.yaml
index 08e97f9..cbf3240 100644
--- a/fleetd/fleetd.example.yaml
+++ b/fleetd/fleetd.example.yaml
@@ -452,6 +452,15 @@ placement: weighted
# seconds, before a spawn may land on it again. Applies to every profile's effective credential
# (its own name, or its credentialId if set above) — there is no per-profile override. Default
# 1800 (30 minutes) when omitted or non-positive.
+#
+# fleetd #466: this is now only the BASE of an escalating backoff, not a flat retry rate. A
+# credential quarantined again within one base cooldown of the previous quarantine ending (still
+# reporting exhausted — e.g. a weekly subscription limit that hasn't reset) backs off further:
+# cooldown doubles each such time, capped at 12x this value (~6 hours at the 1800s default). A
+# quarantine that starts after a base-cooldown's worth of quiet resets back to this value. Not
+# configurable per se — the multiplier and ceiling are constants in BackendQuarantine, not new
+# YAML keys; see its class doc for the exact formula and why there is no automatic probe to clear
+# it early (the operator's own design constraint — a probe spends the quota it's measuring).
# DEFERRED: baked once into the BackendQuarantine built at startup — a running quarantine keeps
# its original cooldown regardless; a new value only applies to a quarantine that starts after a
# restart. Editing this needs a daemon restart to take effect.
diff --git a/fleetd/src/main/java/dev/ltms/fleet/Fleetd.java b/fleetd/src/main/java/dev/ltms/fleet/Fleetd.java
index 8cb9852..14420b9 100644
--- a/fleetd/src/main/java/dev/ltms/fleet/Fleetd.java
+++ b/fleetd/src/main/java/dev/ltms/fleet/Fleetd.java
@@ -222,7 +222,13 @@ public final class Fleetd {
// (checked at spawn) and the exhaustion sink wired in below (written on BACKEND_EXHAUSTED).
// The cooldown is deferred (see FleetConfig#quarantineCooldownSeconds): it is read once
// here, at startup, and a config reload only changes it for a daemon restart.
- BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,
+ // fleetd #466: escalating, not flat — a credential that keeps reporting exhaustion (e.g. a
+ // weekly subscription limit, which would otherwise be retried on every ~30-minute cooldown,
+ // about 336 times across the week) backs off further each consecutive time, capped at
+ // BackendQuarantine.DEFAULT_MAX_COOLDOWN_MULTIPLE x the base cooldown. See BackendQuarantine's
+ // class doc for the mechanism, why this never fires on cooling-off (a separate, unescalated
+ // mechanism — BackendOutagePolicy below), and the reset.
+ BackendQuarantine quarantine = BackendQuarantine.withEscalation(System::nanoTime,
TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));
// fleetd #201 Unit 5: one outage-cool-off tracker for the whole daemon, shared between the
// launcher (checked at spawn, like `quarantine` above) and the backend-error sink wired in
diff --git a/fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java b/fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java
index de64b31..9b9504d 100644
--- a/fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java
+++ b/fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java
@@ -70,12 +70,14 @@ import java.util.regex.PatternSyntaxException;
* {@code fixed} (default), {@code round-robin}, or {@code weighted}
* @param auth API authentication mode ({@code null} → {@code loopback-trust}, the
* historical behaviour), CB-501
- * @param quarantineCooldownSeconds how long a credential stays quarantined after a
+ * @param quarantineCooldownSeconds the BASE cooldown a credential is quarantined for after a
* {@code BACKEND_EXHAUSTED} classification (CB-578 stage B); {@code null}/{@code
- * <=0} → {@link #DEFAULT_QUARANTINE_COOLDOWN_SECONDS}. Baked once into the
- * {@code BackendQuarantine} built at startup, so it is DEFERRED: changing it
- * needs a restart, and a quarantine already running keeps whatever cooldown was
- * live when it started.
+ * <=0} → {@link #DEFAULT_QUARANTINE_COOLDOWN_SECONDS}. Since fleetd #466 this is
+ * only the first occurrence's length — a credential quarantined again shortly
+ * after this cooldown ends backs off further, up to a ceiling; see {@code
+ * BackendQuarantine}'s class doc. Baked once into the {@code BackendQuarantine}
+ * built at startup, so it is DEFERRED: changing it needs a restart, and a
+ * quarantine already running keeps whatever cooldown was live when it started.
* @param memberCredentials deny-by-default policy (CB-596) for which of the operator's own host
* credentials a spawned member's pane inherits. {@code null} (the block
* omitted) blocks nothing — see {@link MemberCredentials}.
diff --git a/fleetd/src/main/java/dev/ltms/fleet/placement/BackendQuarantine.java b/fleetd/src/main/java/dev/ltms/fleet/placement/BackendQuarantine.java
index 3fb437a..48b406b 100644
--- a/fleetd/src/main/java/dev/ltms/fleet/placement/BackendQuarantine.java
+++ b/fleetd/src/main/java/dev/ltms/fleet/placement/BackendQuarantine.java
@@ -21,46 +21,158 @@ import java.util.function.LongSupplier;
*
The clock is injected ({@link LongSupplier}, conventionally {@code System::nanoTime} like
* {@code FleetHealthMonitor}), never read inline, so a quarantine's expiry is testable without a
* real sleep.
+ *
+ *
Escalation (fleetd #466)
+ * A flat cooldown does not fit every exhaustion. A backend that reports "out of capacity for the
+ * rest of the hour" recovers in one cooldown; a weekly subscription limit does not — it keeps
+ * reporting exhausted on every attempt made before the window resets, so a flat 30-minute cooldown
+ * (the default {@code cooldownNanos}) means roughly 336 pointless spawn attempts across a week, one
+ * every cooldown.
+ *
+ * This class only ever sees the exhaustion signal. Its only production caller is
+ * {@code Fleetd.exhaustionSink}, wired to fire on a {@code BACKEND_EXHAUSTED} classification alone.
+ * The daemon's other outage state — a credential "cooling off" after repeated non-exhaustion
+ * backend errors (an HTTP 5xx storm, say) — is a separate mechanism, {@code BackendOutagePolicy},
+ * with its own short fixed 60s cooldown and no repeat tracking. The two are never merged: escalating
+ * on a cooling-off signal would turn a transient 5xx storm into a multi-hour backoff, which is
+ * exactly the failure this ticket is not asking for. Confirmed by reading every call site of
+ * {@link #quarantine} — {@code BackendOutagePolicy} has its own {@code coolOff} method and never
+ * calls this one.
+ *
+ *
Mechanism — the {@link #withEscalation} constructors track, per credential, how
+ * many times in a row {@link #quarantine} has been called without an intervening "quiet" gap.
+ * Each call computes {@code cooldownNanos * backoffMultiplier ^ (repeatCount - 1)}, capped at
+ * {@code maxCooldownNanos}. A call counts as a continuation of the same streak — {@code repeatCount}
+ * increments — when it arrives no more than one base {@code cooldownNanos} after the previous
+ * quarantine's deadline (this covers both "still quarantined" and "quarantine just expired and it
+ * was exhausted again immediately"); otherwise the streak resets and this call is treated as a fresh
+ * first occurrence at the base cooldown.
+ *
+ *
Reset, honestly stated. The ideal reset signal is "the cooldown expired and the
+ * next attempt succeeded" — but nothing in this codebase reports a spawn success back to this class
+ * (checked: {@code SessionManager} and {@code CompositePeerLauncher} never call any method here
+ * except {@link #quarantine}/{@link #isQuarantined}/{@link #remainingSeconds}, none of which is a
+ * success hook). Lacking that signal, the reset used here is a time-based proxy: a base-cooldown's
+ * worth of quiet — no exhaustion report for that credential — since the last quarantine ended. It is
+ * not proof the credential started working again, only the best available evidence without adding an
+ * active probe, which is out of scope by the operator's own design constraint (no automatic probing
+ * of a limited backend).
+ *
+ *
Ceiling. {@code maxCooldownNanos} bounds the growth — an unbounded backoff is a
+ * permanent, unrecoverable-without-a-restart outage, which would be worse than the flat-rate bug this
+ * escalation fixes. {@link #withEscalation(LongSupplier, long)} defaults the ceiling to
+ * {@value #DEFAULT_MAX_COOLDOWN_MULTIPLE}x the base cooldown (12x the 1800s default ≈ 6 hours), so a
+ * chronically exhausted credential still gets re-tried roughly every 6 hours instead of every 30
+ * minutes — about a dozen attempts a week instead of ~336.
+ *
+ *
Backward compatibility. The original two-argument {@link #BackendQuarantine(
+ * LongSupplier, long)} constructor is unchanged in behaviour: it is exactly {@code
+ * withEscalation}'s mechanism with {@code backoffMultiplier = 1.0} and {@code maxCooldownNanos =
+ * cooldownNanos}, which collapses the formula back to the original flat {@code now + cooldownNanos}
+ * on every call regardless of history. Every existing call site (roughly 20 across the test suite,
+ * plus {@link #none()}) keeps its current shape and behaviour unchanged.
*/
public final class BackendQuarantine {
- private final ConcurrentHashMap quarantinedUntilNanos = new ConcurrentHashMap<>();
+ /** Default growth per consecutive exhaustion streak — see the class doc's Mechanism section. */
+ static final double DEFAULT_BACKOFF_MULTIPLIER = 2.0;
+ /** Default ceiling, expressed as a multiple of the base cooldown — see the class doc's Ceiling section. */
+ static final long DEFAULT_MAX_COOLDOWN_MULTIPLE = 12;
+
+ private final ConcurrentHashMap quarantines = new ConcurrentHashMap<>();
private final LongSupplier nowNanos;
private final long cooldownNanos;
+ private final double backoffMultiplier;
+ private final long maxCooldownNanos;
/** True only for {@link #none()}. See {@link #quarantine} for why this exists. */
private final boolean inert;
+ /** How many consecutive exhaustion reports a credential is on, and when the resulting cooldown ends. */
+ private record QuarantineState(int repeatCount, long deadlineNanos) {
+ }
+
/**
+ * Flat cooldown, unchanged from before fleetd #466 — every {@link #quarantine} call blocks the
+ * credential for exactly {@code cooldownNanos}, regardless of how many times it was called
+ * before. Equivalent to {@link #withEscalation} with no growth ({@code backoffMultiplier = 1.0})
+ * and a ceiling equal to the base cooldown, so it degrades to the identical {@code now +
+ * cooldownNanos} formula every call. Kept for the existing call sites that want a fixed cooldown
+ * (and for tests exercising the fixed-cooldown shape in isolation); production wiring uses
+ * {@link #withEscalation} instead.
+ *
* @param nowNanos monotonic clock, injected for testability
* @param cooldownNanos how long a fresh {@link #quarantine} call blocks the credential for;
* must be positive
*/
public BackendQuarantine(LongSupplier nowNanos, long cooldownNanos) {
- this(nowNanos, cooldownNanos, false);
+ this(nowNanos, cooldownNanos, 1.0, cooldownNanos, false);
}
- private BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, boolean inert) {
+ /**
+ * Escalating cooldown (fleetd #466) — see the class doc's Mechanism/Reset/Ceiling sections.
+ *
+ * @param nowNanos monotonic clock, injected for testability
+ * @param cooldownNanos base cooldown, applied to a fresh (non-streak) exhaustion; must be
+ * positive
+ * @param backoffMultiplier growth per consecutive exhaustion; must be {@code >= 1.0} ({@code 1.0}
+ * disables growth and is exactly the flat two-argument constructor)
+ * @param maxCooldownNanos ceiling on the escalated cooldown; must be {@code >= cooldownNanos}
+ */
+ public BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, double backoffMultiplier,
+ long maxCooldownNanos) {
+ this(nowNanos, cooldownNanos, backoffMultiplier, maxCooldownNanos, false);
+ }
+
+ private BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, double backoffMultiplier,
+ long maxCooldownNanos, boolean inert) {
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
if (cooldownNanos <= 0) {
throw new IllegalArgumentException("cooldownNanos must be positive: " + cooldownNanos);
}
+ if (backoffMultiplier < 1.0) {
+ throw new IllegalArgumentException("backoffMultiplier must be >= 1.0: " + backoffMultiplier);
+ }
+ if (maxCooldownNanos < cooldownNanos) {
+ throw new IllegalArgumentException(
+ "maxCooldownNanos must be >= cooldownNanos: " + maxCooldownNanos + " < " + cooldownNanos);
+ }
this.cooldownNanos = cooldownNanos;
+ this.backoffMultiplier = backoffMultiplier;
+ this.maxCooldownNanos = maxCooldownNanos;
this.inert = inert;
}
+ /**
+ * Escalating cooldown with the fleetd #466 default shape: cooldown doubles
+ * ({@value #DEFAULT_BACKOFF_MULTIPLIER}x) per consecutive exhaustion streak, capped at
+ * {@value #DEFAULT_MAX_COOLDOWN_MULTIPLE}x the base cooldown. This is what production wiring
+ * ({@code Fleetd.main}) uses.
+ *
+ * @param nowNanos monotonic clock, injected for testability
+ * @param cooldownNanos base cooldown, applied to a fresh (non-streak) exhaustion; must be positive
+ */
+ public static BackendQuarantine withEscalation(LongSupplier nowNanos, long cooldownNanos) {
+ return new BackendQuarantine(nowNanos, cooldownNanos, DEFAULT_BACKOFF_MULTIPLIER,
+ cooldownNanos * DEFAULT_MAX_COOLDOWN_MULTIPLE, false);
+ }
+
/**
* Inert quarantine — {@link #quarantine} does nothing on this instance, so nothing is ever
* quarantined. The explicit stand-in a caller (or a test not exercising this feature) passes
* instead of a defaulting overload, exactly like {@code ExhaustedPatternLookup.none()}.
*/
public static BackendQuarantine none() {
- return new BackendQuarantine(() -> 0L, 1, true);
+ return new BackendQuarantine(() -> 0L, 1, 1.0, 1, true);
}
/**
- * Quarantine {@code credentialId} for the configured cooldown, starting now. A repeat call while
- * already quarantined restarts the cooldown at full length — a fresh refusal is fresh evidence the
- * account is still exhausted, not a reason to let an earlier, shorter wait stand.
+ * Quarantine {@code credentialId} starting now. On a flat instance (the two-argument
+ * constructor) this always blocks for exactly {@code cooldownNanos}, restarting the cooldown at
+ * full length on every call — a fresh refusal is fresh evidence the account is still exhausted,
+ * not a reason to let an earlier, shorter wait stand. On an escalating instance ({@link
+ * #withEscalation}) the cooldown grows with each call that arrives within one base cooldown of
+ * the previous deadline, and resets to the base cooldown once a call arrives after a longer gap
+ * — see the class doc.
*
* On {@link #none()} this is a no-op. It has to be: that instance holds a clock frozen at 0,
* so recording a deadline would produce a quarantine that never expires — a credential locked out
@@ -73,7 +185,13 @@ public final class BackendQuarantine {
if (inert) {
return;
}
- quarantinedUntilNanos.put(credentialId, nowNanos.getAsLong() + cooldownNanos);
+ long now = nowNanos.getAsLong();
+ quarantines.compute(credentialId, (id, prev) -> {
+ int repeatCount = (prev == null || now - prev.deadlineNanos() > cooldownNanos)
+ ? 1
+ : prev.repeatCount() + 1;
+ return new QuarantineState(repeatCount, now + escalatedCooldownNanos(repeatCount));
+ });
}
/** Whether {@code credentialId} is quarantined right now. */
@@ -95,8 +213,8 @@ public final class BackendQuarantine {
*/
public Map activeRemainingSeconds() {
Map out = new LinkedHashMap<>();
- quarantinedUntilNanos.forEach((credentialId, deadline) -> {
- long remaining = deadline - nowNanos.getAsLong();
+ quarantines.forEach((credentialId, state) -> {
+ long remaining = state.deadlineNanos() - nowNanos.getAsLong();
if (remaining > 0) {
out.put(credentialId, toSecondsRoundedUp(remaining));
}
@@ -105,8 +223,14 @@ public final class BackendQuarantine {
}
private long remainingNanos(String credentialId) {
- Long deadline = quarantinedUntilNanos.get(credentialId);
- return deadline == null ? 0L : deadline - nowNanos.getAsLong();
+ QuarantineState state = quarantines.get(credentialId);
+ return state == null ? 0L : state.deadlineNanos() - nowNanos.getAsLong();
+ }
+
+ /** {@code cooldownNanos * backoffMultiplier ^ (repeatCount - 1)}, capped at {@code maxCooldownNanos}. */
+ private long escalatedCooldownNanos(int repeatCount) {
+ double raw = cooldownNanos * Math.pow(backoffMultiplier, repeatCount - 1);
+ return raw >= (double) maxCooldownNanos ? maxCooldownNanos : (long) raw;
}
private static long toSecondsRoundedUp(long nanos) {
diff --git a/fleetd/src/test/java/dev/ltms/fleet/placement/BackendQuarantineTest.java b/fleetd/src/test/java/dev/ltms/fleet/placement/BackendQuarantineTest.java
index 4a38d01..450ddcb 100644
--- a/fleetd/src/test/java/dev/ltms/fleet/placement/BackendQuarantineTest.java
+++ b/fleetd/src/test/java/dev/ltms/fleet/placement/BackendQuarantineTest.java
@@ -116,4 +116,117 @@ class BackendQuarantineTest {
assertThrows(IllegalArgumentException.class, () -> new BackendQuarantine(() -> 0L, 0L));
assertThrows(IllegalArgumentException.class, () -> new BackendQuarantine(() -> 0L, -1L));
}
+
+ // --- fleetd #466: escalating cooldown -----------------------------------------------------
+ //
+ // Base cooldown 600s (10 min), multiplier 2.0, ceiling 2400s (4x base) — small round numbers
+ // chosen so every deadline is an exact assertion, not just "greater than before". Each call
+ // below lands at or before the previous deadline (a zero or negative gap), which is always
+ // "no more than one base cooldown after the previous deadline" — i.e. every call continues the
+ // same streak, matching a credential that keeps reporting exhausted with no lull.
+
+ @Test
+ void anInvalidBackoffMultiplierIsRejected() {
+ assertThrows(IllegalArgumentException.class,
+ () -> new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30), 0.5, TimeUnit.HOURS.toNanos(6)));
+ }
+
+ @Test
+ void aCeilingBelowTheBaseCooldownIsRejected() {
+ assertThrows(IllegalArgumentException.class,
+ () -> new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30), 2.0, TimeUnit.MINUTES.toNanos(10)));
+ }
+
+ @Test
+ void repeatedExhaustionEscalatesTheCooldownByExactAmounts() {
+ AtomicLong now = new AtomicLong(0L);
+ BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
+ TimeUnit.SECONDS.toNanos(2400));
+
+ q.quarantine("shared-openai"); // 1st: base cooldown
+ assertEquals(OptionalLong.of(600L), q.remainingSeconds("shared-openai"));
+
+ now.set(TimeUnit.SECONDS.toNanos(100)); // still inside the 1st quarantine (deadline 600s)
+ q.quarantine("shared-openai"); // 2nd: 600 * 2^1 = 1200
+ assertEquals(OptionalLong.of(1200L), q.remainingSeconds("shared-openai"),
+ "a second consecutive exhaustion must double the cooldown, not just increase it");
+
+ now.set(TimeUnit.SECONDS.toNanos(1300)); // exactly the 2nd deadline (100 + 1200)
+ q.quarantine("shared-openai"); // 3rd: 600 * 2^2 = 2400 (exactly at the ceiling)
+ assertEquals(OptionalLong.of(2400L), q.remainingSeconds("shared-openai"));
+ }
+
+ @Test
+ void escalationStopsAtTheCeiling() {
+ AtomicLong now = new AtomicLong(0L);
+ BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
+ TimeUnit.SECONDS.toNanos(2400));
+
+ q.quarantine("shared-openai"); // 1st: 600
+ now.set(TimeUnit.SECONDS.toNanos(600));
+ q.quarantine("shared-openai"); // 2nd: 1200, deadline 1800
+ now.set(TimeUnit.SECONDS.toNanos(1800));
+ q.quarantine("shared-openai"); // 3rd: 600 * 4 = 2400, at the ceiling, deadline 4200
+ now.set(TimeUnit.SECONDS.toNanos(4200));
+ q.quarantine("shared-openai"); // 4th: 600 * 8 = 4800 uncapped, must stay capped at 2400
+ assertEquals(OptionalLong.of(2400L), q.remainingSeconds("shared-openai"),
+ "the cooldown must never exceed the configured ceiling, however long the streak gets");
+
+ now.set(TimeUnit.SECONDS.toNanos(6600)); // 4th deadline
+ q.quarantine("shared-openai"); // 5th: still capped
+ assertEquals(OptionalLong.of(2400L), q.remainingSeconds("shared-openai"),
+ "pushing well past the ceiling must not budge it");
+ }
+
+ @Test
+ void aQuietGapLongerThanTheBaseCooldownResetsToTheBaseCooldown() {
+ AtomicLong now = new AtomicLong(0L);
+ BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
+ TimeUnit.SECONDS.toNanos(2400));
+
+ q.quarantine("shared-openai"); // 1st: 600, deadline 600
+ now.set(TimeUnit.SECONDS.toNanos(600));
+ q.quarantine("shared-openai"); // 2nd: 1200, deadline 1800
+ now.set(TimeUnit.SECONDS.toNanos(1800));
+ q.quarantine("shared-openai"); // 3rd: 2400, deadline 4200
+ assertEquals(OptionalLong.of(2400L), q.remainingSeconds("shared-openai"));
+
+ // Quiet for well over one base cooldown (600s) past the 3rd deadline (4200s).
+ now.set(TimeUnit.SECONDS.toNanos(20_000));
+ q.quarantine("shared-openai"); // treated as a fresh occurrence
+ assertEquals(OptionalLong.of(600L), q.remainingSeconds("shared-openai"),
+ "a long quiet gap must reset the streak back to the base cooldown");
+ }
+
+ @Test
+ void escalatingOneCredentialDoesNotSlowAnother() {
+ AtomicLong now = new AtomicLong(0L);
+ BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.SECONDS.toNanos(600), 2.0,
+ TimeUnit.SECONDS.toNanos(2400));
+
+ q.quarantine("shared-openai"); // 1st: 600
+ now.set(TimeUnit.SECONDS.toNanos(600));
+ q.quarantine("shared-openai"); // 2nd: 1200
+ now.set(TimeUnit.SECONDS.toNanos(1800));
+ q.quarantine("shared-openai"); // 3rd: 2400 — three-in-a-row streak on this credential only
+
+ q.quarantine("another-credential"); // its first and only exhaustion
+ assertEquals(OptionalLong.of(600L), q.remainingSeconds("another-credential"),
+ "an unrelated credential's cooldown must stay at the base rate, unaffected by a sibling's streak");
+ }
+
+ @Test
+ void withEscalationDefaultsToDoublingCappedAtTwelveTimesTheBase() {
+ AtomicLong now = new AtomicLong(0L);
+ BackendQuarantine q = BackendQuarantine.withEscalation(now::get, TimeUnit.MINUTES.toNanos(30));
+
+ q.quarantine("shared-openai");
+ assertEquals(OptionalLong.of(1800L), q.remainingSeconds("shared-openai"),
+ "the first occurrence must still use the base cooldown");
+
+ now.set(TimeUnit.MINUTES.toNanos(30));
+ q.quarantine("shared-openai");
+ assertEquals(OptionalLong.of(3600L), q.remainingSeconds("shared-openai"),
+ "the default multiplier must be 2.0");
+ }
}