Compare commits

..

2 Commits

Author SHA1 Message Date
Dai Ha 7840e9adf6 CB-201: make unmapped target notice truthful
CI / contract (pull_request) Successful in 1m13s
CI / build (pull_request) Successful in 1m15s
2026-09-03 10:44:18 +07:00
Dai Ha ee932fd85b CB-201: nudge leads about backend outages
CI / contract (pull_request) Successful in 1m7s
CI / build (pull_request) Successful in 1m55s
2026-09-03 10:28:51 +07:00
6 changed files with 340 additions and 394 deletions
@@ -1,35 +0,0 @@
package dev.ltms.fleet.inject;
import java.util.regex.Pattern;
/**
* Per-target lookup for a profile's configured backend-error pattern (fleetd#201 / #227): how
* {@link CompletionResolver} recognizes a backend that failed a turn outright (a credential
* outage, a provider 5xx) from whatever ends up on the member's pane, apart from a genuine
* completion.
*
* <p>The pattern is always profile config, never a vendor string in Java source: every backend
* words its failure differently, so a hardcoded sentence would only ever match one of them. A
* target with no configured pattern is not "off" the way {@link ExhaustedPatternLookup#none()}
* is — {@link CompletionResolver} falls back to its narrow {@code (?i)\bAPI Error\s*:}
* compatibility pattern instead, so classification still happens, just without a profile-specific
* match. {@link #legacy()} is the explicit stand-in existing (pre-fleetd#201) callers pass to get
* exactly that fallback-only behavior until a caller wires a real, profile-driven lookup
* (fleetd#201 Unit 5).
*/
@FunctionalInterface
public interface BackendErrorPatternLookup {
/** The compiled pattern configured for {@code target}'s profile, or {@code null} if none. */
Pattern patternFor(String target);
/**
* Legacy lookup — no profile has a configured pattern, so {@link CompletionResolver} classifies
* every target using only its built-in {@code (?i)\bAPI Error\s*:} compatibility pattern. The
* explicit stand-in existing constructors pass so their behavior is unchanged until a caller
* wires a real, profile-driven lookup.
*/
static BackendErrorPatternLookup legacy() {
return target -> null;
}
}
@@ -1,35 +0,0 @@
package dev.ltms.fleet.inject;
/**
* Notified when {@link CompletionResolver} actually delivers a typed backend-error classification
* to a waiting send (fleetd#201 / #227) — never on a race that lost. {@link CompletionResolver}
* calls this only after {@code Rendezvous.resolveFailure} returns {@code true} for that exact
* waiter, mirroring the win-only race rule {@link ExhaustionSink} already uses.
*
* <p>The public send result is unchanged by this classification — it is still a failed send
* ({@code Rendezvous.Kind#FAILED}); this sink is the internal seam a later stage (fleetd#201 Unit
* 5, the policy that cools a credential off after two such failures in 60 seconds and tells the
* lead) consumes. {@link CompletionResolver} knows only {@code target} (a herdr terminal id); it
* has no notion of profiles or credentials, so mapping {@code target} to whatever should be
* quarantined is entirely the sink's job — exactly like {@link ExhaustionSink}.
*/
@FunctionalInterface
public interface BackendErrorSink {
/**
* @param target the herdr terminal id whose turn was classified as a backend error
* @param matchedLine the single pane line that matched the backend-error pattern
* @param reason the full failure reason carried by the classification (the matched line
* plus whatever pane context the classifying path attaches)
*/
void onBackendError(String target, String matchedLine, String reason);
/**
* Inert sink — nothing happens on a backend-error classification. The explicit stand-in a
* caller (or a test not exercising this feature) passes instead of a defaulting overload,
* exactly like {@link ExhaustionSink#none()}.
*/
static BackendErrorSink none() {
return (target, matchedLine, reason) -> { };
}
}
@@ -89,8 +89,6 @@ public final class CompletionResolver implements TurnListener {
private final Rendezvous rendezvous;
private final ExhaustedPatternLookup exhaustedPatterns;
private final ExhaustionSink exhaustionSink;
private final BackendErrorPatternLookup backendErrorPatterns;
private final BackendErrorSink backendErrorSink;
private final LongSupplier nowNanos;
/**
@@ -131,48 +129,10 @@ public final class CompletionResolver implements TurnListener {
* classification actually resolves a waiter. Required for the same
* reason as {@code exhaustedPatterns} — pass {@link ExhaustionSink#none()}
* to opt out.
*
* <p>Transition constructor (fleetd#201 Unit 1): delegates to the full constructor below with
* {@link BackendErrorPatternLookup#legacy()} and {@link BackendErrorSink#none()}, so every
* existing caller keeps today's behavior — the narrow built-in {@code API Error:} match, no
* sink notified — until a caller wires a real, profile-driven backend-error lookup and sink
* (fleetd#201 Unit 5).
*/
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
ExhaustionSink exhaustionSink) {
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), System::nanoTime);
}
/**
* Transition constructor (fleetd#201 Unit 1): same legacy backend-error defaults as the 4-arg
* constructor above, but with the injectable clock. Kept so existing fleetd#164 timing tests
* (and any other caller that wants a controllable clock but not the backend-error feature)
* keep compiling unchanged.
*/
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
ExhaustionSink exhaustionSink, LongSupplier nowNanos) {
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), nowNanos);
}
/**
* Production constructor (fleetd#201 Unit 1): adds the target-keyed backend-error pattern
* lookup and sink alongside the existing exhaustion pair. Required, like {@code exhaustedPatterns}
* and {@code exhaustionSink} — pass {@link BackendErrorPatternLookup#legacy()} and
* {@link BackendErrorSink#none()} to opt out.
*
* @param backendErrorPatterns fleetd#201: per-target lookup for a profile's configured
* backend-error pattern; a target with none configured is matched
* against the built-in compatibility pattern instead (never "off").
* @param backendErrorSink fleetd#201: notified when a backend-error classification actually
* resolves a waiter — never on a race that lost.
*/
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
BackendErrorSink backendErrorSink) {
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, backendErrorPatterns, backendErrorSink,
System::nanoTime);
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, System::nanoTime);
}
/**
@@ -185,14 +145,11 @@ public final class CompletionResolver implements TurnListener {
* package.
*/
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
BackendErrorSink backendErrorSink, LongSupplier nowNanos) {
ExhaustionSink exhaustionSink, LongSupplier nowNanos) {
this.agents = agents;
this.rendezvous = rendezvous;
this.exhaustedPatterns = Objects.requireNonNull(exhaustedPatterns, "exhaustedPatterns");
this.exhaustionSink = Objects.requireNonNull(exhaustionSink, "exhaustionSink");
this.backendErrorPatterns = Objects.requireNonNull(backendErrorPatterns, "backendErrorPatterns");
this.backendErrorSink = Objects.requireNonNull(backendErrorSink, "backendErrorSink");
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
}
@@ -275,7 +232,7 @@ public final class CompletionResolver implements TurnListener {
// screen, since that is usually the backend's own error.
long elapsedNanos = nowNanos.getAsLong() - turn.deliveredAtNanos();
if (elapsedNanos < MIN_TURN_NANOS) {
failTooFast(target, turn, waiter, elapsedNanos);
fail(target, turn, tooFastReason(target, elapsedNanos));
return;
}
String tail;
@@ -345,27 +302,18 @@ public final class CompletionResolver implements TurnListener {
}
return;
}
// fleetd#164 (part 2) / fleetd#201: a scrape that read cleanly and produced content still
// isn't a real reply when that content is the backend's own rejection (e.g. an HTTP 400
// before the worker did any work). Classify it as a failure naming the member, rather than
// handing the caller a scrape that reads like a completed answer, and — only on the
// resolution that actually wins the race, mirroring the exhaustion sink above — notify the
// typed backend-error sink so a later stage can act on repeated failures.
String backendError = firstMatchingLine(assistantBlock, backendErrorPatternOrFallback(target));
// fleetd#164 (part 2): a scrape that read cleanly and produced content still isn't a real
// reply when that content is the backend's own rejection (e.g. an HTTP 400 before the worker
// did any work). Classify it as a failure naming the member, rather than handing the caller a
// scrape that reads like a completed answer.
String backendError = firstMatchingLine(assistantBlock, BACKEND_ERROR);
if (backendError != null) {
// Carry the whole scrape, not just the matched line. The pattern is a heuristic: a member
// that forgot fleet_reply while reporting *about* a backend error matches it too. Failing
// is still right — the caller must not read a scrape as an answer — but dropping the rest
// of the pane would destroy the report, which is the same defect fleetd#164 is about.
String reason = "member " + target + " ended on a backend error: " + backendError
+ "\n--- pane tail ---\n" + tail;
if (rendezvous.resolveFailure(waiter, reason)) {
inFlight.remove(target, turn);
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
// fleetd#201 Unit 1: only on the resolution that actually won the race — a late
// duplicate must never double-count one backend failure.
backendErrorSink.onBackendError(target, backendError, reason);
}
fail(target, turn, "member " + target + " ended on a backend error: " + backendError
+ "\n--- pane tail ---\n" + tail);
return;
}
String completion = clipped ? tail + "\n" + CLIPPED_PANE_TAIL_MARKER : tail;
@@ -405,20 +353,14 @@ public final class CompletionResolver implements TurnListener {
}
return true;
}
String backendError = firstMatchingLine(raw, backendErrorPatternOrFallback(target));
String backendError = firstMatchingLine(raw, BACKEND_ERROR);
if (backendError != null) {
// Carry the pane, not just the matched line — the same fleetd#164 rule the normal path
// above applies. Here it matters more, not less: the trimmed block was empty, so the raw
// scrape is the ONLY copy of whatever the member managed to say. Clipped to the same cap
// the normal path uses, since a raw screen has no boundary trimming to bound it.
String reason = "member " + target + " ended on a backend error: " + backendError
+ "\n--- pane tail ---\n" + clip(raw);
if (rendezvous.resolveFailure(waiter, reason)) {
inFlight.remove(target, turn);
log.warn("failing send to {} via turn-stall fallback from the raw scrape: {}", target, reason);
// fleetd#201 Unit 1: only on the resolution that actually won the race.
backendErrorSink.onBackendError(target, backendError, reason);
}
fail(target, turn, "member " + target + " ended on a backend error: " + backendError
+ "\n--- pane tail ---\n" + clip(raw));
return true;
}
return false;
@@ -460,53 +402,22 @@ public final class CompletionResolver implements TurnListener {
}
/**
* fleetd#164 (floor) / fleetd#201 (classification): fail a turn that settled inside
* {@link #MIN_TURN_NANOS} — a crash signature (e.g. a backend HTTP 400 before the worker did
* anything) that a bare {@code BUSY -> DONE} transition cannot be told apart from a genuinely
* fast completion. Runs the same backend-error classification the normal and raw-scrape paths
* apply, against whatever is on screen right now: a match is a typed failure that notifies
* {@link #backendErrorSink} (only on the resolution that wins the race); a non-match stays the
* original generic too-fast failure, naming the member and both timings, with whatever the pane
* shows appended so the caller sees the cause, not just "it failed".
* fleetd#164: the failure reason for a turn that settled inside {@link #MIN_TURN_NANOS} — names
* the member and both timings, and appends whatever the pane shows (usually the backend's own
* error) so the caller sees the cause, not just "it failed".
*/
private void failTooFast(String target, InFlight turn, CompletableFuture<Rendezvous.Resolution> waiter,
long elapsedNanos) {
private String tooFastReason(String target, long elapsedNanos) {
String scrape;
try {
scrape = agents.read(target, SCRAPE_SOURCE);
scrape = clip(agents.read(target, SCRAPE_SOURCE));
} catch (RuntimeException e) {
scrape = "";
}
String clippedScrape = clip(scrape);
String baseReason = String.format(
String reason = String.format(
"member %s went BUSY -> DONE in %dms (floor %dms) — too fast to be real work, most "
+ "likely a backend error before any work started",
target, elapsedNanos / 1_000_000, MIN_TURN_NANOS / 1_000_000);
String backendError = firstMatchingLine(scrape, backendErrorPatternOrFallback(target));
if (backendError != null) {
String reason = baseReason + ": " + clippedScrape;
if (rendezvous.resolveFailure(waiter, reason)) {
inFlight.remove(target, turn);
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
// fleetd#201 Unit 1: only on the resolution that actually won the race.
backendErrorSink.onBackendError(target, backendError, reason);
}
return;
}
fail(target, turn, clippedScrape.isBlank() ? baseReason : baseReason + ": " + clippedScrape);
}
/**
* fleetd#201: the pattern to classify a backend error against for {@code target} — its
* profile's configured {@link BackendErrorPatternLookup} entry when there is one, else the
* built-in {@link #BACKEND_ERROR} compatibility pattern. A target with no configured pattern is
* never "off": it always falls back to this narrow default, exactly like the classification
* behaved before fleetd#201 (Unit 5 reports a target relying on this fallback separately, as
* legacy-default coverage rather than full per-profile coverage).
*/
private Pattern backendErrorPatternOrFallback(String target) {
Pattern configured = backendErrorPatterns.patternFor(target);
return configured != null ? configured : BACKEND_ERROR;
return scrape.isBlank() ? reason : reason + ": " + scrape;
}
/**
@@ -9,8 +9,10 @@ import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
import java.util.ArrayList;
import java.util.Collection;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.concurrent.ConcurrentHashMap;
import java.util.concurrent.ScheduledExecutorService;
@@ -68,6 +70,12 @@ public final class ReplyPushLoop {
static final String QUESTIONS_NUDGE_FORMAT =
"%d workers are paused on a question — run fleet_poll(ticket=...) for each, then answer "
+ "with fleet_send(turnId=..., content=...): %s";
static final String BACKEND_INCIDENT_NUDGE_FORMAT =
"Backend credential %s is cooling for %d remaining seconds; affected profiles: %s; "
+ "affected workers: %s. Run fleet_list to see more.";
static final String BACKEND_TARGET_UNMAPPED_NUDGE_FORMAT =
"A backend error on worker %s could not be mapped to a credential (%s) — no cool-off "
+ "was applied. Run fleet_list to check that worker.";
private final PrimaryRegistry primaryRegistry;
private final AgentControl agents;
@@ -93,6 +101,13 @@ public final class ReplyPushLoop {
* how depleted an older, still-open question's count is.
*/
private final ConcurrentHashMap<String, PendingQuestion> pendingQuestions = new ConcurrentHashMap<>();
/** Backend incidents awaiting one successful delivery, keyed by incident id and owning lead. */
private final ConcurrentHashMap<IncidentLead, PendingIncident> pendingIncidents = new ConcurrentHashMap<>();
/** Incident/lead pairs already delivered. They make repeated incident reports one-shot. */
private final Set<IncidentLead> deliveredIncidents = ConcurrentHashMap.newKeySet();
private final ConcurrentHashMap<UnmappedTargetLead, PendingUnmappedTarget> pendingUnmappedTargets =
new ConcurrentHashMap<>();
private final Set<UnmappedTargetLead> deliveredUnmappedTargets = ConcurrentHashMap.newKeySet();
/** CB-590: leads with an active combined reminder schedule (replies and/or tickets and/or questions). */
private final ConcurrentHashMap<String, Boolean> activeLeads = new ConcurrentHashMap<>();
@@ -178,7 +193,20 @@ public final class ReplyPushLoop {
* tracked per question, not per lead per source).
*/
private record PendingQuestion(String turnId, String ticket, String target, String lead,
String question, int nudgeCount) {
String question, int nudgeCount) {
}
private record IncidentLead(String incidentId, String lead) {
}
private record PendingIncident(IncidentLead key, String credential, List<String> profiles,
List<String> targets, int remainingCoolOffSeconds, int nudgeCount) {
}
private record UnmappedTargetLead(String target, String reason, String lead) {
}
private record PendingUnmappedTarget(UnmappedTargetLead key, int nudgeCount) {
}
/** Questions still open for {@code lead}, snapshotted fresh for one tick. */
@@ -192,6 +220,24 @@ public final class ReplyPushLoop {
.collect(Collectors.toUnmodifiableSet());
}
private List<PendingIncident> pendingIncidentsFor(String lead) {
return pendingIncidents.values().stream().filter(i -> lead.equals(i.key().lead())).toList();
}
private Set<IncidentLead> pendingIncidentKeysFor(String lead) {
return pendingIncidentsFor(lead).stream().map(PendingIncident::key)
.collect(Collectors.toUnmodifiableSet());
}
private List<PendingUnmappedTarget> pendingUnmappedTargetsFor(String lead) {
return pendingUnmappedTargets.values().stream().filter(i -> lead.equals(i.key().lead())).toList();
}
private Set<UnmappedTargetLead> pendingUnmappedTargetKeysFor(String lead) {
return pendingUnmappedTargetsFor(lead).stream().map(PendingUnmappedTarget::key)
.collect(Collectors.toUnmodifiableSet());
}
/**
* The reply-source reminder count {@link #decide} should see for {@code lead} on this tick:
* the <em>minimum</em> nudge count among the reply targets currently pending for it (CB-598).
@@ -236,6 +282,15 @@ public final class ReplyPushLoop {
return min == Integer.MAX_VALUE ? 0 : min;
}
private int minIncidentNudgeCountFor(String lead) {
return pendingIncidentsFor(lead).stream().mapToInt(PendingIncident::nudgeCount).min().orElse(0);
}
private int minUnmappedTargetNudgeCountFor(String lead) {
return pendingUnmappedTargetsFor(lead).stream().mapToInt(PendingUnmappedTarget::nudgeCount)
.min().orElse(0);
}
/**
* Pure decision function: examine everything pending for {@code lead} — reply targets and
* tickets alike — and return what the loop should do.
@@ -272,17 +327,35 @@ public final class ReplyPushLoop {
* @param questionReminderCount the lowest nudge count among questions open for this lead
*/
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount) {
return decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
minIncidentNudgeCountFor(lead));
}
/** As above, with backend incidents as a fourth, independently bounded source. */
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount,
int incidentReminderCount) {
return decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
incidentReminderCount, minUnmappedTargetNudgeCountFor(lead));
}
/** As above, with unmapped backend targets as a fifth, independently bounded source. */
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount,
int incidentReminderCount, int unmappedTargetReminderCount) {
boolean hasReplyWork = !pendingReplyTargetsFor(lead).isEmpty();
boolean hasTicketWork = !pendingTicketIdsFor(lead).isEmpty();
boolean hasQuestionWork = !pendingQuestionTurnIdsFor(lead).isEmpty();
if (!hasReplyWork && !hasTicketWork && !hasQuestionWork) {
boolean hasIncidentWork = !pendingIncidentKeysFor(lead).isEmpty();
boolean hasUnmappedTargetWork = !pendingUnmappedTargetKeysFor(lead).isEmpty();
if (!hasReplyWork && !hasTicketWork && !hasQuestionWork && !hasIncidentWork && !hasUnmappedTargetWork) {
log.debug("push: nothing pending for lead {}, stopping reminder", lead);
return Action.STOP;
}
boolean replyEligible = hasReplyWork && replyReminderCount < maxReminders;
boolean ticketEligible = hasTicketWork && ticketReminderCount < maxReminders;
boolean questionEligible = hasQuestionWork && questionReminderCount < maxReminders;
if (!replyEligible && !ticketEligible && !questionEligible) {
boolean incidentEligible = hasIncidentWork && incidentReminderCount < maxReminders;
boolean unmappedTargetEligible = hasUnmappedTargetWork && unmappedTargetReminderCount < maxReminders;
if (!replyEligible && !ticketEligible && !questionEligible && !incidentEligible && !unmappedTargetEligible) {
log.debug("push: reminder cap ({}) reached for lead {} on every source with pending work, stopping",
maxReminders, lead);
countNudge("exhausted");
@@ -394,6 +467,48 @@ public final class ReplyPushLoop {
pendingQuestions.remove(turnId);
}
/**
* Queue a one-shot backend credential outage notice for every distinct lead that owns an
* affected worker. A successful injection records its {@code (incidentId, lead)} key, so a
* repeat report never reminds that lead again.
*/
public void onBackendIncident(String incidentId, Collection<String> targets, String credential,
Collection<String> profiles, int remainingCoolOffSeconds) {
Map<String, List<String>> targetsByLead = new ConcurrentHashMap<>();
for (String target : targets) {
var lead = primaryRegistry.nudgeTargetFor(target);
if (lead.isEmpty()) {
log.warn("push: backend incident {} has no known lead for target {}", incidentId, target);
continue;
}
targetsByLead.computeIfAbsent(lead.get(), _ -> new ArrayList<>()).add(target);
}
List<String> profileNames = profiles.stream().sorted().toList();
for (var entry : targetsByLead.entrySet()) {
IncidentLead key = new IncidentLead(incidentId, entry.getKey());
if (deliveredIncidents.contains(key)) continue;
pendingIncidents.putIfAbsent(key, new PendingIncident(key, credential, profileNames,
entry.getValue().stream().sorted().toList(), remainingCoolOffSeconds, 0));
startOrCoalesce(entry.getKey());
}
}
/**
* Tell the owning lead that a classified backend error could not be tied to a credential.
* Without an owning lead, emit a warning because no control can act on the target.
*/
public void onBackendTargetUnmapped(String target, String reason) {
var lead = primaryRegistry.nudgeTargetFor(target);
if (lead.isEmpty()) {
log.warn("push: backend target {} could not map to a credential: {}", target, reason);
return;
}
UnmappedTargetLead key = new UnmappedTargetLead(target, reason, lead.get());
if (deliveredUnmappedTargets.contains(key)) return;
pendingUnmappedTargets.putIfAbsent(key, new PendingUnmappedTarget(key, 0));
startOrCoalesce(lead.get());
}
// --- the schedule ----------------------------------------------------------------------------
/** Start a reminder schedule for {@code lead}, or join the one already running. */
@@ -425,18 +540,25 @@ public final class ReplyPushLoop {
Set<String> repliesBefore = pendingReplyTargetsFor(lead);
Set<String> ticketsBefore = pendingTicketIdsFor(lead);
Set<String> questionsBefore = pendingQuestionTurnIdsFor(lead);
Set<IncidentLead> incidentsBefore = pendingIncidentKeysFor(lead);
Set<UnmappedTargetLead> unmappedTargetsBefore = pendingUnmappedTargetKeysFor(lead);
int replyReminderCount = minReplyNudgeCountFor(lead);
int ticketReminderCount = minTicketNudgeCountFor(lead);
int questionReminderCount = minQuestionNudgeCountFor(lead);
var action = decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount);
int incidentReminderCount = minIncidentNudgeCountFor(lead);
int unmappedTargetReminderCount = minUnmappedTargetNudgeCountFor(lead);
var action = decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
incidentReminderCount, unmappedTargetReminderCount);
switch (action) {
case INJECT -> {
injectNudge(lead, replyReminderCount, ticketReminderCount, questionReminderCount);
injectNudge(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
incidentReminderCount, unmappedTargetReminderCount);
scheduleNext(lead);
}
// Re-check after the configured backoff; the lead may become injectable soon.
case WAIT_BUSY -> scheduleNext(lead);
case STOP -> stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore);
case STOP -> stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, incidentsBefore,
unmappedTargetsBefore);
}
}
@@ -480,11 +602,25 @@ public final class ReplyPushLoop {
* decision-to-release window reclaims the schedule slot exactly like a raced-in reply or ticket.
*/
void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
Set<String> questionsBefore) {
Set<String> questionsBefore) {
stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, pendingIncidentKeysFor(lead));
}
private void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
Set<String> questionsBefore, Set<IncidentLead> incidentsBefore) {
stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, incidentsBefore,
pendingUnmappedTargetKeysFor(lead));
}
private void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
Set<String> questionsBefore, Set<IncidentLead> incidentsBefore,
Set<UnmappedTargetLead> unmappedTargetsBefore) {
activeLeads.remove(lead);
boolean racedIn = pendingReplyTargetsFor(lead).stream().anyMatch(t -> !repliesBefore.contains(t))
|| pendingTicketIdsFor(lead).stream().anyMatch(t -> !ticketsBefore.contains(t))
|| pendingQuestionTurnIdsFor(lead).stream().anyMatch(t -> !questionsBefore.contains(t));
|| pendingQuestionTurnIdsFor(lead).stream().anyMatch(t -> !questionsBefore.contains(t))
|| pendingIncidentKeysFor(lead).stream().anyMatch(i -> !incidentsBefore.contains(i))
|| pendingUnmappedTargetKeysFor(lead).stream().anyMatch(i -> !unmappedTargetsBefore.contains(i));
if (racedIn && activeLeads.putIfAbsent(lead, Boolean.TRUE) == null) {
log.debug("push: new work for lead {} raced the reminder loop's stop — restarting", lead);
scheduleNext(lead);
@@ -495,18 +631,22 @@ public final class ReplyPushLoop {
/** Send one combined nudge covering everything currently pending for {@code lead}. */
private void injectNudge(String lead, int replyReminderCount, int ticketReminderCount,
int questionReminderCount) {
int questionReminderCount, int incidentReminderCount,
int unmappedTargetReminderCount) {
// Re-read rather than threading it down from decide(): a reply can drain, a ticket be
// collected, or a question be answered (or another arrive), between the decision and the
// injection.
Set<String> replyTargets = pendingReplyTargetsFor(lead);
List<PendingTicket> tickets = pendingTicketsFor(lead);
List<PendingQuestion> questions = pendingQuestionsFor(lead);
if (replyTargets.isEmpty() && tickets.isEmpty() && questions.isEmpty()) {
List<PendingIncident> incidents = pendingIncidentsFor(lead);
List<PendingUnmappedTarget> unmappedTargets = pendingUnmappedTargetsFor(lead);
if (replyTargets.isEmpty() && tickets.isEmpty() && questions.isEmpty() && incidents.isEmpty()
&& unmappedTargets.isEmpty()) {
log.debug("push: pending work for lead {} drained before the nudge could be sent", lead);
return;
}
String nudge = formatNudge(replyTargets, tickets, questions);
String nudge = formatNudge(replyTargets, tickets, questions, incidents, unmappedTargets);
try {
agents.send(lead, nudge);
log.debug("push: nudge sent to lead {} (reply {}/{}, ticket {}/{}, question {}/{}; "
@@ -515,6 +655,16 @@ public final class ReplyPushLoop {
questionReminderCount + 1, maxReminders,
replyTargets.size(), tickets.size(), questions.size());
countNudge("delivered");
for (PendingIncident incident : incidents) {
if (pendingIncidents.remove(incident.key(), incident)) {
deliveredIncidents.add(incident.key());
}
}
for (PendingUnmappedTarget unmappedTarget : unmappedTargets) {
if (pendingUnmappedTargets.remove(unmappedTarget.key(), unmappedTarget)) {
deliveredUnmappedTargets.add(unmappedTarget.key());
}
}
} catch (RuntimeException e) {
log.warn("push: failed to nudge lead {} (reply {}/{}, ticket {}/{}, question {}/{}): {}",
lead, replyReminderCount + 1, maxReminders, ticketReminderCount + 1, maxReminders,
@@ -526,12 +676,13 @@ public final class ReplyPushLoop {
// item already at or over the cap keeps riding along in the text (still pending, still
// named) but its extra bumps here are inert: decide() already treats it as ineligible once
// its count reaches maxReminders.
bumpNudgeCounts(replyTargets, tickets, questions);
bumpNudgeCounts(replyTargets, tickets, questions, incidents, unmappedTargets);
}
/** Record that every one of these items was just named in a sent (or attempted) nudge. */
private void bumpNudgeCounts(Set<String> replyTargets, List<PendingTicket> tickets,
List<PendingQuestion> questions) {
List<PendingQuestion> questions, List<PendingIncident> incidents,
List<PendingUnmappedTarget> unmappedTargets) {
for (String target : replyTargets) {
pendingReplies.computeIfPresent(target, (t, e) -> new ReplyEntry(e.lead(), e.nudgeCount() + 1));
}
@@ -544,6 +695,14 @@ public final class ReplyPushLoop {
new PendingQuestion(e.turnId(), e.ticket(), e.target(), e.lead(), e.question(),
e.nudgeCount() + 1));
}
for (PendingIncident incident : incidents) {
pendingIncidents.computeIfPresent(incident.key(), (id, e) -> new PendingIncident(e.key(),
e.credential(), e.profiles(), e.targets(), e.remainingCoolOffSeconds(), e.nudgeCount() + 1));
}
for (PendingUnmappedTarget unmappedTarget : unmappedTargets) {
pendingUnmappedTargets.computeIfPresent(unmappedTarget.key(), (id, e) ->
new PendingUnmappedTarget(e.key(), e.nudgeCount() + 1));
}
}
/** Schedule the next tick on the scheduler thread pool. */
@@ -556,7 +715,8 @@ public final class ReplyPushLoop {
/** Render everything pending for one lead as a single nudge line. */
private static String formatNudge(Set<String> replyTargets, List<PendingTicket> tickets,
List<PendingQuestion> questions) {
List<PendingQuestion> questions, List<PendingIncident> incidents,
List<PendingUnmappedTarget> unmappedTargets) {
List<String> parts = new ArrayList<>();
if (!replyTargets.isEmpty()) {
parts.add(formatRepliesNudge(replyTargets));
@@ -567,6 +727,12 @@ public final class ReplyPushLoop {
if (!questions.isEmpty()) {
parts.add(formatQuestionsNudge(questions));
}
if (!incidents.isEmpty()) {
parts.addAll(incidents.stream().map(ReplyPushLoop::formatBackendIncidentNudge).toList());
}
if (!unmappedTargets.isEmpty()) {
parts.addAll(unmappedTargets.stream().map(ReplyPushLoop::formatUnmappedTargetNudge).toList());
}
return String.join(" | ", parts);
}
@@ -606,6 +772,16 @@ public final class ReplyPushLoop {
return QUESTIONS_NUDGE_FORMAT.formatted(pending.size(), ids);
}
private static String formatBackendIncidentNudge(PendingIncident incident) {
return BACKEND_INCIDENT_NUDGE_FORMAT.formatted(incident.credential(), incident.remainingCoolOffSeconds(),
String.join(", ", incident.profiles()), String.join(", ", incident.targets()));
}
private static String formatUnmappedTargetNudge(PendingUnmappedTarget unmappedTarget) {
return BACKEND_TARGET_UNMAPPED_NUDGE_FORMAT.formatted(unmappedTarget.key().target(),
unmappedTarget.key().reason());
}
// --- lifecycle -----------------------------------------------------------------------------
/**
@@ -627,6 +803,10 @@ public final class ReplyPushLoop {
pendingReplies.clear();
pendingTickets.clear();
pendingQuestions.clear();
pendingIncidents.clear();
deliveredIncidents.clear();
pendingUnmappedTargets.clear();
deliveredUnmappedTargets.clear();
}
/** @see #stop() */
@@ -4,11 +4,8 @@ import ch.qos.logback.classic.Level;
import ch.qos.logback.classic.LoggerContext;
import ch.qos.logback.classic.spi.ILoggingEvent;
import ch.qos.logback.core.read.ListAppender;
import com.fasterxml.jackson.databind.JsonNode;
import com.fasterxml.jackson.databind.ObjectMapper;
import dev.ltms.fleet.herdr.AgentControl;
import dev.ltms.fleet.herdr.FakeHerdr;
import dev.ltms.fleet.herdr.HerdrClient;
import dev.ltms.fleet.msg.Rendezvous;
import dev.ltms.fleet.msg.TestTurnTokens;
import dev.ltms.fleet.msg.TurnToken;
@@ -762,202 +759,4 @@ class CompletionResolverTest {
assertEquals("partial (configured: [terra]; not configured: [gx10])",
CompletionResolver.coverage(Set.of("terra", "gx10"), Set.of("terra")));
}
// --- fleetd#201 Unit 1: target-keyed backend-error pattern + typed sink ----------------------
@Test
void aConfiguredBackendErrorPatternClassifiesAMatchAsAFailureAndNotifiesTheSinkOnce() {
String block = "⏺ 503 Service Unavailable: upstream credential rejected\n❯ ";
FakeHerdr herdr = new FakeHerdr().readText(block);
Rendezvous rendezvous = new Rendezvous();
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
java.util.List<String> notified = new java.util.ArrayList<>();
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
var waiter = rendezvous.open("term_a");
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
assertTrue(waiter.isDone(), "a configured-pattern match still resolves the blocked send");
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind());
assertEquals(1, notified.size(), "the sink fires exactly once for the winning classification");
assertEquals("term_a: 503 Service Unavailable: upstream credential rejected", notified.get(0));
}
@Test
void aLosingBackendErrorClassificationNeverNotifiesTheSink() {
// The waiter was already resolved (e.g. by the worker's own fleet_reply) before this scrape
// landed — resolveFailure loses the race and must return false, so the sink must not fire.
String block = "⏺ 503 Service Unavailable: upstream credential rejected\n❯ ";
FakeHerdr herdr = new FakeHerdr().readText(block);
Rendezvous rendezvous = new Rendezvous();
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
java.util.List<String> notified = new java.util.ArrayList<>();
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
var waiter = rendezvous.open("term_a");
var turn = new CompletionResolver.InFlight(waiter, null);
assertTrue(rendezvous.resolveCompletion(waiter, "already replied")); // fleet_reply won first
resolver.resolve("term_a", turn);
assertTrue(notified.isEmpty(), "a classification that loses the race must not fire the sink");
assertEquals("already replied", waiter.getNow(null).text(), "the earlier resolution stands untouched");
}
@Test
void aBackendErrorClassificationThatLosesARealRaceDuringTheScrapeNeverNotifiesTheSink() {
// The test above pre-resolves the waiter BEFORE calling resolve(), so it is caught by
// resolve()'s own top-of-method isDone() guard before ever reaching the win-gated
// resolveFailure call this unit adds — it proves the outcome, not the specific gate. This
// test forces the actual race window fleetd#201 must be safe in: a genuine fleet_reply lands
// WHILE the herdr scrape for this turn is in flight, i.e. AFTER resolve() has already passed
// the top guard (waiter not done yet) and committed to reading the pane, but BEFORE it
// reaches the "if (rendezvous.resolveFailure(...))" line below. The side effect is attached
// to the one real I/O step resolve() performs between those two points: the herdr
// "agent.read" call itself.
Rendezvous rendezvous = new Rendezvous();
var waiter = rendezvous.open("term_a");
ObjectMapper mapper = new ObjectMapper();
HerdrClient racingDuringScrape = new HerdrClient() {
@Override
public JsonNode call(String method, Object params) {
if ("agent.read".equals(method)) {
// The real fleet_reply that wins the race, landing mid-scrape.
assertTrue(rendezvous.resolve("term_a", "a genuine fleet_reply landed first"),
"the racing reply must land while the waiter is still open");
}
try {
return mapper.readTree(mapper.writeValueAsString(java.util.Map.of(
"type", "agent_read",
"read", java.util.Map.of("text",
"⏺ 503 Service Unavailable: upstream credential rejected\n❯ "))));
} catch (Exception e) {
throw new RuntimeException(e);
}
}
@Override
public void close() {
}
};
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
java.util.List<String> notified = new java.util.ArrayList<>();
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
CompletionResolver resolver = new CompletionResolver(new AgentControl(racingDuringScrape), rendezvous,
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
var turn = new CompletionResolver.InFlight(waiter, null); // not done yet — passes the top guard
resolver.resolve("term_a", turn);
assertTrue(notified.isEmpty(),
"a classification that loses a real race during the scrape must not fire the sink");
assertEquals(Rendezvous.Kind.REPLY, waiter.getNow(null).kind(), "the genuine fleet_reply stands");
assertEquals("a genuine fleet_reply landed first", waiter.getNow(null).text());
}
@Test
void exhaustionKeepsWinningOverBackendErrorEvenWhenBothPatternsMatchTheSameLine() {
// A line matching both a configured exhausted pattern AND a configured backend-error pattern
// must classify BACKEND_EXHAUSTED and call only the ExhaustionSink — the ordering fleetd#578
// already relies on must not change.
String block = "⏺ The usage limit has been reached. Try again later.\n❯ ";
FakeHerdr herdr = new FakeHerdr().readText(block);
Rendezvous rendezvous = new Rendezvous();
ExhaustedPatternLookup exhausted = target -> Pattern.compile("usage limit has been reached");
BackendErrorPatternLookup backendErrors = target -> Pattern.compile("(?i)usage limit");
java.util.List<String> exhaustedNotified = new java.util.ArrayList<>();
java.util.List<String> backendErrorNotified = new java.util.ArrayList<>();
ExhaustionSink exhaustionSink = (target, reason) -> exhaustedNotified.add(target);
BackendErrorSink backendErrorSink = (target, matchedLine, reason) -> backendErrorNotified.add(target);
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
exhausted, exhaustionSink, backendErrors, backendErrorSink);
var waiter = rendezvous.open("term_a");
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, waiter.getNow(null).kind(),
"exhaustion must keep winning over backend error");
assertEquals(1, exhaustedNotified.size(), "the exhaustion sink fires");
assertTrue(backendErrorNotified.isEmpty(), "the backend-error sink must never fire for this turn");
}
@Test
void aConfiguredPatternAlsoClassifiesTheRawScrapeFallbackAndNotifiesTheSink() {
// No ⏺ marker and leading TUI chrome ⇒ lastAssistantBlock() yields "", so classification must
// fall back to the raw scrape (fleetd#211) — and it must use the configured pattern too.
String block = """
╭──────────────────────────────────────╮
503 Service Unavailable: upstream credential rejected
""";
FakeHerdr herdr = new FakeHerdr().readText(block);
Rendezvous rendezvous = new Rendezvous();
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
java.util.List<String> notified = new java.util.ArrayList<>();
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
var waiter = rendezvous.open("term_a");
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
assertTrue(waiter.isDone(), "a raw-scrape configured-pattern match still resolves the blocked send");
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
"classified from the raw scrape even though the trimmed block was empty");
assertEquals(1, notified.size(), "the sink fires exactly once from the raw-scrape path");
assertTrue(notified.get(0).contains("503 Service Unavailable"), notified.get(0));
}
// --- fleetd#201 Unit 1: classification inside the fleetd#164 MIN_TURN_NANOS floor -------------
@Test
void aMatchingErrorInsideTheFloorIsTypedAndNotifiesTheSink() {
FakeHerdr herdr = new FakeHerdr().readText("⏺ 503 Service Unavailable: upstream credential rejected\n❯ ");
Rendezvous rendezvous = new Rendezvous();
long[] clock = {10_000_000_000L};
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
java.util.List<String> notified = new java.util.ArrayList<>();
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink, () -> clock[0]);
var waiter = rendezvous.open("term_a");
var turn = new CompletionResolver.InFlight(waiter, null, clock[0]); // delivered "now"
clock[0] += CompletionResolver.MIN_TURN_NANOS - 1; // inside the floor
resolver.resolve("term_a", turn);
assertTrue(waiter.isDone());
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind());
assertEquals(1, notified.size(), "a matching error inside the floor notifies the sink");
assertTrue(notified.get(0).contains("503 Service Unavailable"), notified.get(0));
}
@Test
void aNonMatchInsideTheFloorStaysGenericAndNeverNotifiesTheSink() {
FakeHerdr herdr = new FakeHerdr().readText("⏺ still starting up\n❯ ");
Rendezvous rendezvous = new Rendezvous();
long[] clock = {10_000_000_000L};
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
java.util.List<String> notified = new java.util.ArrayList<>();
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink, () -> clock[0]);
var waiter = rendezvous.open("term_a");
var turn = new CompletionResolver.InFlight(waiter, null, clock[0]); // delivered "now"
clock[0] += CompletionResolver.MIN_TURN_NANOS - 1; // inside the floor
resolver.resolve("term_a", turn);
assertTrue(waiter.isDone());
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
"still a failure — the floor itself, not the pattern, is why");
assertTrue(waiter.getNow(null).text().contains("too fast to be real work"),
"a non-match inside the floor stays the generic too-fast reason: " + waiter.getNow(null).text());
assertTrue(notified.isEmpty(), "a non-match must never notify the typed sink");
}
}
@@ -43,6 +43,7 @@ class ReplyPushLoopTest {
private static final String PRIMARY = "term_primary";
private static final String WORKER = "term_worker";
private static final String WORKER2 = "term_worker2";
private static final String OTHER_PRIMARY = "term_other_primary";
private static final ObjectMapper MAPPER = new ObjectMapper();
private PrimaryRegistry registry;
@@ -529,6 +530,106 @@ class ReplyPushLoopTest {
assertTrue(nudge.contains("task-1"), "the ticket must not be dropped: " + nudge);
}
// --- CB-201: backend credential outage nudges -----------------------------------------------
@Test
void oneBackendIncidentForTwoWorkersOfOneLeadSendsOnceAndIsOneShot() throws Exception {
var rec = recordingClient();
agents = new AgentControl(rec);
registry.recordDelegation(WORKER, PRIMARY);
registry.recordDelegation(WORKER2, PRIMARY);
var loop = loop(2, 100);
loop.onBackendIncident("incident-1", List.of(WORKER, WORKER2), "cred-a",
List.of("terra", "codex"), 47);
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS), "one lead gets one outage injection");
Thread.sleep(200);
assertEquals(1, rec.sendCount(), "two workers for one lead must send only one outage notice");
String nudge = rec.sentParams().getFirst().getValue().toString();
assertTrue(nudge.contains("cred-a"));
assertTrue(nudge.contains("codex, terra"));
assertTrue(nudge.contains(WORKER) && nudge.contains(WORKER2));
assertTrue(nudge.contains("47 remaining seconds"));
assertTrue(nudge.contains("fleet_list"));
loop.onBackendIncident("incident-1", List.of(WORKER, WORKER2), "cred-a",
List.of("terra", "codex"), 47);
Thread.sleep(250);
assertEquals(1, rec.sendCount(), "a delivered incident id must never be sent again");
}
@Test
void oneBackendIncidentSendsOnceToEachAffectedLead() throws Exception {
var rec = recordingClient();
rec.sendLatch = new CountDownLatch(2);
agents = new AgentControl(rec);
registry.recordDelegation(WORKER, PRIMARY);
registry.recordDelegation(WORKER2, OTHER_PRIMARY);
var loop = loop(1, 50);
loop.onBackendIncident("incident-2", List.of(WORKER, WORKER2), "cred-a", List.of("terra"), 60);
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS), "each affected lead gets its one notice");
assertEquals(2, rec.sendCount());
}
@Test
void backendIncidentWaitsForBusyLeadThenCombinesWithFailedTicket() throws Exception {
var rec = new BusyThenIdleHerdrClient(2);
agents = new AgentControl(rec);
var loop = loop(1, 50);
loop.onTicketTerminal("task-failed", WORKER, true);
loop.onBackendIncident("incident-3", List.of(WORKER), "cred-b", List.of("terra"), 31);
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS), "busy lead should receive deferred combined nudge");
assertEquals(1, rec.sendCount(), "ticket and outage must use the same injection");
String nudge = rec.sentParams().getFirst().getValue().toString();
assertTrue(nudge.contains("task-failed") && nudge.contains("cred-b"),
"the one nudge must name both the failed ticket and outage: " + nudge);
}
@Test
void failedBackendIncidentSendRetriesWithinTheExistingReminderCap() throws Exception {
var rec = new ThrowOnceHerdrClient();
agents = new AgentControl(rec);
var loop = loop(2, 50);
loop.onBackendIncident("incident-4", List.of(WORKER), "cred-c", List.of("terra"), 20);
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS), "failed first send must retry under maxReminders");
assertEquals(2, rec.promptAttempts.get(), "one failed send and one retry use the existing cap");
}
@Test
void unmappedBackendTargetUsesTheKnownLeadSchedule() throws Exception {
var rec = recordingClient();
agents = new AgentControl(rec);
var loop = loop(1, 50);
loop.onBackendTargetUnmapped(WORKER, "no credential matched");
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS));
String nudge = ((Map<?, ?>) rec.sentParams().getFirst().getValue()).get("text").toString();
assertEquals("A backend error on worker term_worker could not be mapped to a credential "
+ "(no credential matched) — no cool-off was applied. "
+ "Run fleet_list to check that worker.",
nudge);
}
@Test
void stopClearsPendingBackendIncidents() {
agents = agentWithStatus("idle");
var loop = loop(1, 100_000);
loop.onBackendIncident("incident-stop", List.of(WORKER), "cred-stop", List.of("terra"), 60);
loop.stop();
assertEquals(ReplyPushLoop.Action.STOP, loop.decide(PRIMARY, 0, 0, 0),
"stop must clear a pending backend incident as well as older sources");
}
// --- CB-590 follow-up: per-source reminder budgets — the regression this round exists for ---
@Test
@@ -928,6 +1029,31 @@ class ReplyPushLoopTest {
return new RecordingHerdrClient();
}
/** Fails its first prompt call, then records the retry. */
private static final class ThrowOnceHerdrClient implements HerdrClient {
private final AtomicInteger promptAttempts = new AtomicInteger();
private final CountDownLatch sendLatch = new CountDownLatch(1);
@Override
public JsonNode call(String method, Object params) {
if ("agent.get".equals(method)) {
return MAPPER.createObjectNode().set("agent", MAPPER.createObjectNode()
.put("terminal_id", PRIMARY).put("agent_status", "idle"));
}
if ("agent.prompt".equals(method) && promptAttempts.getAndIncrement() == 0) {
throw new IllegalStateException("first prompt fails");
}
if ("agent.prompt".equals(method)) {
sendLatch.countDown();
}
return MAPPER.createObjectNode();
}
@Override
public void close() {
}
}
/**
* Thread-safe fake that reports {@code working} (not injectable) for its first
* {@code busyChecks} status calls, then {@code idle} forever after — used to prove work queued