Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 7840e9adf6 | |||
| ee932fd85b |
@@ -1,35 +0,0 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Per-target lookup for a profile's configured backend-error pattern (fleetd#201 / #227): how
|
||||
* {@link CompletionResolver} recognizes a backend that failed a turn outright (a credential
|
||||
* outage, a provider 5xx) from whatever ends up on the member's pane, apart from a genuine
|
||||
* completion.
|
||||
*
|
||||
* <p>The pattern is always profile config, never a vendor string in Java source: every backend
|
||||
* words its failure differently, so a hardcoded sentence would only ever match one of them. A
|
||||
* target with no configured pattern is not "off" the way {@link ExhaustedPatternLookup#none()}
|
||||
* is — {@link CompletionResolver} falls back to its narrow {@code (?i)\bAPI Error\s*:}
|
||||
* compatibility pattern instead, so classification still happens, just without a profile-specific
|
||||
* match. {@link #legacy()} is the explicit stand-in existing (pre-fleetd#201) callers pass to get
|
||||
* exactly that fallback-only behavior until a caller wires a real, profile-driven lookup
|
||||
* (fleetd#201 Unit 5).
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface BackendErrorPatternLookup {
|
||||
|
||||
/** The compiled pattern configured for {@code target}'s profile, or {@code null} if none. */
|
||||
Pattern patternFor(String target);
|
||||
|
||||
/**
|
||||
* Legacy lookup — no profile has a configured pattern, so {@link CompletionResolver} classifies
|
||||
* every target using only its built-in {@code (?i)\bAPI Error\s*:} compatibility pattern. The
|
||||
* explicit stand-in existing constructors pass so their behavior is unchanged until a caller
|
||||
* wires a real, profile-driven lookup.
|
||||
*/
|
||||
static BackendErrorPatternLookup legacy() {
|
||||
return target -> null;
|
||||
}
|
||||
}
|
||||
@@ -1,35 +0,0 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
/**
|
||||
* Notified when {@link CompletionResolver} actually delivers a typed backend-error classification
|
||||
* to a waiting send (fleetd#201 / #227) — never on a race that lost. {@link CompletionResolver}
|
||||
* calls this only after {@code Rendezvous.resolveFailure} returns {@code true} for that exact
|
||||
* waiter, mirroring the win-only race rule {@link ExhaustionSink} already uses.
|
||||
*
|
||||
* <p>The public send result is unchanged by this classification — it is still a failed send
|
||||
* ({@code Rendezvous.Kind#FAILED}); this sink is the internal seam a later stage (fleetd#201 Unit
|
||||
* 5, the policy that cools a credential off after two such failures in 60 seconds and tells the
|
||||
* lead) consumes. {@link CompletionResolver} knows only {@code target} (a herdr terminal id); it
|
||||
* has no notion of profiles or credentials, so mapping {@code target} to whatever should be
|
||||
* quarantined is entirely the sink's job — exactly like {@link ExhaustionSink}.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface BackendErrorSink {
|
||||
|
||||
/**
|
||||
* @param target the herdr terminal id whose turn was classified as a backend error
|
||||
* @param matchedLine the single pane line that matched the backend-error pattern
|
||||
* @param reason the full failure reason carried by the classification (the matched line
|
||||
* plus whatever pane context the classifying path attaches)
|
||||
*/
|
||||
void onBackendError(String target, String matchedLine, String reason);
|
||||
|
||||
/**
|
||||
* Inert sink — nothing happens on a backend-error classification. The explicit stand-in a
|
||||
* caller (or a test not exercising this feature) passes instead of a defaulting overload,
|
||||
* exactly like {@link ExhaustionSink#none()}.
|
||||
*/
|
||||
static BackendErrorSink none() {
|
||||
return (target, matchedLine, reason) -> { };
|
||||
}
|
||||
}
|
||||
@@ -89,8 +89,6 @@ public final class CompletionResolver implements TurnListener {
|
||||
private final Rendezvous rendezvous;
|
||||
private final ExhaustedPatternLookup exhaustedPatterns;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
private final BackendErrorPatternLookup backendErrorPatterns;
|
||||
private final BackendErrorSink backendErrorSink;
|
||||
private final LongSupplier nowNanos;
|
||||
|
||||
/**
|
||||
@@ -131,48 +129,10 @@ public final class CompletionResolver implements TurnListener {
|
||||
* classification actually resolves a waiter. Required for the same
|
||||
* reason as {@code exhaustedPatterns} — pass {@link ExhaustionSink#none()}
|
||||
* to opt out.
|
||||
*
|
||||
* <p>Transition constructor (fleetd#201 Unit 1): delegates to the full constructor below with
|
||||
* {@link BackendErrorPatternLookup#legacy()} and {@link BackendErrorSink#none()}, so every
|
||||
* existing caller keeps today's behavior — the narrow built-in {@code API Error:} match, no
|
||||
* sink notified — until a caller wires a real, profile-driven backend-error lookup and sink
|
||||
* (fleetd#201 Unit 5).
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
|
||||
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), System::nanoTime);
|
||||
}
|
||||
|
||||
/**
|
||||
* Transition constructor (fleetd#201 Unit 1): same legacy backend-error defaults as the 4-arg
|
||||
* constructor above, but with the injectable clock. Kept so existing fleetd#164 timing tests
|
||||
* (and any other caller that wants a controllable clock but not the backend-error feature)
|
||||
* keep compiling unchanged.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, LongSupplier nowNanos) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
|
||||
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), nowNanos);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor (fleetd#201 Unit 1): adds the target-keyed backend-error pattern
|
||||
* lookup and sink alongside the existing exhaustion pair. Required, like {@code exhaustedPatterns}
|
||||
* and {@code exhaustionSink} — pass {@link BackendErrorPatternLookup#legacy()} and
|
||||
* {@link BackendErrorSink#none()} to opt out.
|
||||
*
|
||||
* @param backendErrorPatterns fleetd#201: per-target lookup for a profile's configured
|
||||
* backend-error pattern; a target with none configured is matched
|
||||
* against the built-in compatibility pattern instead (never "off").
|
||||
* @param backendErrorSink fleetd#201: notified when a backend-error classification actually
|
||||
* resolves a waiter — never on a race that lost.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
|
||||
BackendErrorSink backendErrorSink) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, backendErrorPatterns, backendErrorSink,
|
||||
System::nanoTime);
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, System::nanoTime);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -185,14 +145,11 @@ public final class CompletionResolver implements TurnListener {
|
||||
* package.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
|
||||
BackendErrorSink backendErrorSink, LongSupplier nowNanos) {
|
||||
ExhaustionSink exhaustionSink, LongSupplier nowNanos) {
|
||||
this.agents = agents;
|
||||
this.rendezvous = rendezvous;
|
||||
this.exhaustedPatterns = Objects.requireNonNull(exhaustedPatterns, "exhaustedPatterns");
|
||||
this.exhaustionSink = Objects.requireNonNull(exhaustionSink, "exhaustionSink");
|
||||
this.backendErrorPatterns = Objects.requireNonNull(backendErrorPatterns, "backendErrorPatterns");
|
||||
this.backendErrorSink = Objects.requireNonNull(backendErrorSink, "backendErrorSink");
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
}
|
||||
|
||||
@@ -275,7 +232,7 @@ public final class CompletionResolver implements TurnListener {
|
||||
// screen, since that is usually the backend's own error.
|
||||
long elapsedNanos = nowNanos.getAsLong() - turn.deliveredAtNanos();
|
||||
if (elapsedNanos < MIN_TURN_NANOS) {
|
||||
failTooFast(target, turn, waiter, elapsedNanos);
|
||||
fail(target, turn, tooFastReason(target, elapsedNanos));
|
||||
return;
|
||||
}
|
||||
String tail;
|
||||
@@ -345,27 +302,18 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
return;
|
||||
}
|
||||
// fleetd#164 (part 2) / fleetd#201: a scrape that read cleanly and produced content still
|
||||
// isn't a real reply when that content is the backend's own rejection (e.g. an HTTP 400
|
||||
// before the worker did any work). Classify it as a failure naming the member, rather than
|
||||
// handing the caller a scrape that reads like a completed answer, and — only on the
|
||||
// resolution that actually wins the race, mirroring the exhaustion sink above — notify the
|
||||
// typed backend-error sink so a later stage can act on repeated failures.
|
||||
String backendError = firstMatchingLine(assistantBlock, backendErrorPatternOrFallback(target));
|
||||
// fleetd#164 (part 2): a scrape that read cleanly and produced content still isn't a real
|
||||
// reply when that content is the backend's own rejection (e.g. an HTTP 400 before the worker
|
||||
// did any work). Classify it as a failure naming the member, rather than handing the caller a
|
||||
// scrape that reads like a completed answer.
|
||||
String backendError = firstMatchingLine(assistantBlock, BACKEND_ERROR);
|
||||
if (backendError != null) {
|
||||
// Carry the whole scrape, not just the matched line. The pattern is a heuristic: a member
|
||||
// that forgot fleet_reply while reporting *about* a backend error matches it too. Failing
|
||||
// is still right — the caller must not read a scrape as an answer — but dropping the rest
|
||||
// of the pane would destroy the report, which is the same defect fleetd#164 is about.
|
||||
String reason = "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + tail;
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
// fleetd#201 Unit 1: only on the resolution that actually won the race — a late
|
||||
// duplicate must never double-count one backend failure.
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
fail(target, turn, "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + tail);
|
||||
return;
|
||||
}
|
||||
String completion = clipped ? tail + "\n" + CLIPPED_PANE_TAIL_MARKER : tail;
|
||||
@@ -405,20 +353,14 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
return true;
|
||||
}
|
||||
String backendError = firstMatchingLine(raw, backendErrorPatternOrFallback(target));
|
||||
String backendError = firstMatchingLine(raw, BACKEND_ERROR);
|
||||
if (backendError != null) {
|
||||
// Carry the pane, not just the matched line — the same fleetd#164 rule the normal path
|
||||
// above applies. Here it matters more, not less: the trimmed block was empty, so the raw
|
||||
// scrape is the ONLY copy of whatever the member managed to say. Clipped to the same cap
|
||||
// the normal path uses, since a raw screen has no boundary trimming to bound it.
|
||||
String reason = "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + clip(raw);
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback from the raw scrape: {}", target, reason);
|
||||
// fleetd#201 Unit 1: only on the resolution that actually won the race.
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
fail(target, turn, "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + clip(raw));
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
@@ -460,53 +402,22 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#164 (floor) / fleetd#201 (classification): fail a turn that settled inside
|
||||
* {@link #MIN_TURN_NANOS} — a crash signature (e.g. a backend HTTP 400 before the worker did
|
||||
* anything) that a bare {@code BUSY -> DONE} transition cannot be told apart from a genuinely
|
||||
* fast completion. Runs the same backend-error classification the normal and raw-scrape paths
|
||||
* apply, against whatever is on screen right now: a match is a typed failure that notifies
|
||||
* {@link #backendErrorSink} (only on the resolution that wins the race); a non-match stays the
|
||||
* original generic too-fast failure, naming the member and both timings, with whatever the pane
|
||||
* shows appended so the caller sees the cause, not just "it failed".
|
||||
* fleetd#164: the failure reason for a turn that settled inside {@link #MIN_TURN_NANOS} — names
|
||||
* the member and both timings, and appends whatever the pane shows (usually the backend's own
|
||||
* error) so the caller sees the cause, not just "it failed".
|
||||
*/
|
||||
private void failTooFast(String target, InFlight turn, CompletableFuture<Rendezvous.Resolution> waiter,
|
||||
long elapsedNanos) {
|
||||
private String tooFastReason(String target, long elapsedNanos) {
|
||||
String scrape;
|
||||
try {
|
||||
scrape = agents.read(target, SCRAPE_SOURCE);
|
||||
scrape = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
} catch (RuntimeException e) {
|
||||
scrape = "";
|
||||
}
|
||||
String clippedScrape = clip(scrape);
|
||||
String baseReason = String.format(
|
||||
String reason = String.format(
|
||||
"member %s went BUSY -> DONE in %dms (floor %dms) — too fast to be real work, most "
|
||||
+ "likely a backend error before any work started",
|
||||
target, elapsedNanos / 1_000_000, MIN_TURN_NANOS / 1_000_000);
|
||||
String backendError = firstMatchingLine(scrape, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
String reason = baseReason + ": " + clippedScrape;
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
// fleetd#201 Unit 1: only on the resolution that actually won the race.
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
fail(target, turn, clippedScrape.isBlank() ? baseReason : baseReason + ": " + clippedScrape);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#201: the pattern to classify a backend error against for {@code target} — its
|
||||
* profile's configured {@link BackendErrorPatternLookup} entry when there is one, else the
|
||||
* built-in {@link #BACKEND_ERROR} compatibility pattern. A target with no configured pattern is
|
||||
* never "off": it always falls back to this narrow default, exactly like the classification
|
||||
* behaved before fleetd#201 (Unit 5 reports a target relying on this fallback separately, as
|
||||
* legacy-default coverage rather than full per-profile coverage).
|
||||
*/
|
||||
private Pattern backendErrorPatternOrFallback(String target) {
|
||||
Pattern configured = backendErrorPatterns.patternFor(target);
|
||||
return configured != null ? configured : BACKEND_ERROR;
|
||||
return scrape.isBlank() ? reason : reason + ": " + scrape;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -9,8 +9,10 @@ import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collection;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
@@ -68,6 +70,12 @@ public final class ReplyPushLoop {
|
||||
static final String QUESTIONS_NUDGE_FORMAT =
|
||||
"%d workers are paused on a question — run fleet_poll(ticket=...) for each, then answer "
|
||||
+ "with fleet_send(turnId=..., content=...): %s";
|
||||
static final String BACKEND_INCIDENT_NUDGE_FORMAT =
|
||||
"Backend credential %s is cooling for %d remaining seconds; affected profiles: %s; "
|
||||
+ "affected workers: %s. Run fleet_list to see more.";
|
||||
static final String BACKEND_TARGET_UNMAPPED_NUDGE_FORMAT =
|
||||
"A backend error on worker %s could not be mapped to a credential (%s) — no cool-off "
|
||||
+ "was applied. Run fleet_list to check that worker.";
|
||||
|
||||
private final PrimaryRegistry primaryRegistry;
|
||||
private final AgentControl agents;
|
||||
@@ -93,6 +101,13 @@ public final class ReplyPushLoop {
|
||||
* how depleted an older, still-open question's count is.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, PendingQuestion> pendingQuestions = new ConcurrentHashMap<>();
|
||||
/** Backend incidents awaiting one successful delivery, keyed by incident id and owning lead. */
|
||||
private final ConcurrentHashMap<IncidentLead, PendingIncident> pendingIncidents = new ConcurrentHashMap<>();
|
||||
/** Incident/lead pairs already delivered. They make repeated incident reports one-shot. */
|
||||
private final Set<IncidentLead> deliveredIncidents = ConcurrentHashMap.newKeySet();
|
||||
private final ConcurrentHashMap<UnmappedTargetLead, PendingUnmappedTarget> pendingUnmappedTargets =
|
||||
new ConcurrentHashMap<>();
|
||||
private final Set<UnmappedTargetLead> deliveredUnmappedTargets = ConcurrentHashMap.newKeySet();
|
||||
/** CB-590: leads with an active combined reminder schedule (replies and/or tickets and/or questions). */
|
||||
private final ConcurrentHashMap<String, Boolean> activeLeads = new ConcurrentHashMap<>();
|
||||
|
||||
@@ -178,7 +193,20 @@ public final class ReplyPushLoop {
|
||||
* tracked per question, not per lead per source).
|
||||
*/
|
||||
private record PendingQuestion(String turnId, String ticket, String target, String lead,
|
||||
String question, int nudgeCount) {
|
||||
String question, int nudgeCount) {
|
||||
}
|
||||
|
||||
private record IncidentLead(String incidentId, String lead) {
|
||||
}
|
||||
|
||||
private record PendingIncident(IncidentLead key, String credential, List<String> profiles,
|
||||
List<String> targets, int remainingCoolOffSeconds, int nudgeCount) {
|
||||
}
|
||||
|
||||
private record UnmappedTargetLead(String target, String reason, String lead) {
|
||||
}
|
||||
|
||||
private record PendingUnmappedTarget(UnmappedTargetLead key, int nudgeCount) {
|
||||
}
|
||||
|
||||
/** Questions still open for {@code lead}, snapshotted fresh for one tick. */
|
||||
@@ -192,6 +220,24 @@ public final class ReplyPushLoop {
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
private List<PendingIncident> pendingIncidentsFor(String lead) {
|
||||
return pendingIncidents.values().stream().filter(i -> lead.equals(i.key().lead())).toList();
|
||||
}
|
||||
|
||||
private Set<IncidentLead> pendingIncidentKeysFor(String lead) {
|
||||
return pendingIncidentsFor(lead).stream().map(PendingIncident::key)
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
private List<PendingUnmappedTarget> pendingUnmappedTargetsFor(String lead) {
|
||||
return pendingUnmappedTargets.values().stream().filter(i -> lead.equals(i.key().lead())).toList();
|
||||
}
|
||||
|
||||
private Set<UnmappedTargetLead> pendingUnmappedTargetKeysFor(String lead) {
|
||||
return pendingUnmappedTargetsFor(lead).stream().map(PendingUnmappedTarget::key)
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* The reply-source reminder count {@link #decide} should see for {@code lead} on this tick:
|
||||
* the <em>minimum</em> nudge count among the reply targets currently pending for it (CB-598).
|
||||
@@ -236,6 +282,15 @@ public final class ReplyPushLoop {
|
||||
return min == Integer.MAX_VALUE ? 0 : min;
|
||||
}
|
||||
|
||||
private int minIncidentNudgeCountFor(String lead) {
|
||||
return pendingIncidentsFor(lead).stream().mapToInt(PendingIncident::nudgeCount).min().orElse(0);
|
||||
}
|
||||
|
||||
private int minUnmappedTargetNudgeCountFor(String lead) {
|
||||
return pendingUnmappedTargetsFor(lead).stream().mapToInt(PendingUnmappedTarget::nudgeCount)
|
||||
.min().orElse(0);
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure decision function: examine everything pending for {@code lead} — reply targets and
|
||||
* tickets alike — and return what the loop should do.
|
||||
@@ -272,17 +327,35 @@ public final class ReplyPushLoop {
|
||||
* @param questionReminderCount the lowest nudge count among questions open for this lead
|
||||
*/
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount) {
|
||||
return decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
minIncidentNudgeCountFor(lead));
|
||||
}
|
||||
|
||||
/** As above, with backend incidents as a fourth, independently bounded source. */
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount,
|
||||
int incidentReminderCount) {
|
||||
return decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
incidentReminderCount, minUnmappedTargetNudgeCountFor(lead));
|
||||
}
|
||||
|
||||
/** As above, with unmapped backend targets as a fifth, independently bounded source. */
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount,
|
||||
int incidentReminderCount, int unmappedTargetReminderCount) {
|
||||
boolean hasReplyWork = !pendingReplyTargetsFor(lead).isEmpty();
|
||||
boolean hasTicketWork = !pendingTicketIdsFor(lead).isEmpty();
|
||||
boolean hasQuestionWork = !pendingQuestionTurnIdsFor(lead).isEmpty();
|
||||
if (!hasReplyWork && !hasTicketWork && !hasQuestionWork) {
|
||||
boolean hasIncidentWork = !pendingIncidentKeysFor(lead).isEmpty();
|
||||
boolean hasUnmappedTargetWork = !pendingUnmappedTargetKeysFor(lead).isEmpty();
|
||||
if (!hasReplyWork && !hasTicketWork && !hasQuestionWork && !hasIncidentWork && !hasUnmappedTargetWork) {
|
||||
log.debug("push: nothing pending for lead {}, stopping reminder", lead);
|
||||
return Action.STOP;
|
||||
}
|
||||
boolean replyEligible = hasReplyWork && replyReminderCount < maxReminders;
|
||||
boolean ticketEligible = hasTicketWork && ticketReminderCount < maxReminders;
|
||||
boolean questionEligible = hasQuestionWork && questionReminderCount < maxReminders;
|
||||
if (!replyEligible && !ticketEligible && !questionEligible) {
|
||||
boolean incidentEligible = hasIncidentWork && incidentReminderCount < maxReminders;
|
||||
boolean unmappedTargetEligible = hasUnmappedTargetWork && unmappedTargetReminderCount < maxReminders;
|
||||
if (!replyEligible && !ticketEligible && !questionEligible && !incidentEligible && !unmappedTargetEligible) {
|
||||
log.debug("push: reminder cap ({}) reached for lead {} on every source with pending work, stopping",
|
||||
maxReminders, lead);
|
||||
countNudge("exhausted");
|
||||
@@ -394,6 +467,48 @@ public final class ReplyPushLoop {
|
||||
pendingQuestions.remove(turnId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Queue a one-shot backend credential outage notice for every distinct lead that owns an
|
||||
* affected worker. A successful injection records its {@code (incidentId, lead)} key, so a
|
||||
* repeat report never reminds that lead again.
|
||||
*/
|
||||
public void onBackendIncident(String incidentId, Collection<String> targets, String credential,
|
||||
Collection<String> profiles, int remainingCoolOffSeconds) {
|
||||
Map<String, List<String>> targetsByLead = new ConcurrentHashMap<>();
|
||||
for (String target : targets) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.warn("push: backend incident {} has no known lead for target {}", incidentId, target);
|
||||
continue;
|
||||
}
|
||||
targetsByLead.computeIfAbsent(lead.get(), _ -> new ArrayList<>()).add(target);
|
||||
}
|
||||
List<String> profileNames = profiles.stream().sorted().toList();
|
||||
for (var entry : targetsByLead.entrySet()) {
|
||||
IncidentLead key = new IncidentLead(incidentId, entry.getKey());
|
||||
if (deliveredIncidents.contains(key)) continue;
|
||||
pendingIncidents.putIfAbsent(key, new PendingIncident(key, credential, profileNames,
|
||||
entry.getValue().stream().sorted().toList(), remainingCoolOffSeconds, 0));
|
||||
startOrCoalesce(entry.getKey());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Tell the owning lead that a classified backend error could not be tied to a credential.
|
||||
* Without an owning lead, emit a warning because no control can act on the target.
|
||||
*/
|
||||
public void onBackendTargetUnmapped(String target, String reason) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.warn("push: backend target {} could not map to a credential: {}", target, reason);
|
||||
return;
|
||||
}
|
||||
UnmappedTargetLead key = new UnmappedTargetLead(target, reason, lead.get());
|
||||
if (deliveredUnmappedTargets.contains(key)) return;
|
||||
pendingUnmappedTargets.putIfAbsent(key, new PendingUnmappedTarget(key, 0));
|
||||
startOrCoalesce(lead.get());
|
||||
}
|
||||
|
||||
// --- the schedule ----------------------------------------------------------------------------
|
||||
|
||||
/** Start a reminder schedule for {@code lead}, or join the one already running. */
|
||||
@@ -425,18 +540,25 @@ public final class ReplyPushLoop {
|
||||
Set<String> repliesBefore = pendingReplyTargetsFor(lead);
|
||||
Set<String> ticketsBefore = pendingTicketIdsFor(lead);
|
||||
Set<String> questionsBefore = pendingQuestionTurnIdsFor(lead);
|
||||
Set<IncidentLead> incidentsBefore = pendingIncidentKeysFor(lead);
|
||||
Set<UnmappedTargetLead> unmappedTargetsBefore = pendingUnmappedTargetKeysFor(lead);
|
||||
int replyReminderCount = minReplyNudgeCountFor(lead);
|
||||
int ticketReminderCount = minTicketNudgeCountFor(lead);
|
||||
int questionReminderCount = minQuestionNudgeCountFor(lead);
|
||||
var action = decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount);
|
||||
int incidentReminderCount = minIncidentNudgeCountFor(lead);
|
||||
int unmappedTargetReminderCount = minUnmappedTargetNudgeCountFor(lead);
|
||||
var action = decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
incidentReminderCount, unmappedTargetReminderCount);
|
||||
switch (action) {
|
||||
case INJECT -> {
|
||||
injectNudge(lead, replyReminderCount, ticketReminderCount, questionReminderCount);
|
||||
injectNudge(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
incidentReminderCount, unmappedTargetReminderCount);
|
||||
scheduleNext(lead);
|
||||
}
|
||||
// Re-check after the configured backoff; the lead may become injectable soon.
|
||||
case WAIT_BUSY -> scheduleNext(lead);
|
||||
case STOP -> stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore);
|
||||
case STOP -> stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, incidentsBefore,
|
||||
unmappedTargetsBefore);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -480,11 +602,25 @@ public final class ReplyPushLoop {
|
||||
* decision-to-release window reclaims the schedule slot exactly like a raced-in reply or ticket.
|
||||
*/
|
||||
void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
|
||||
Set<String> questionsBefore) {
|
||||
Set<String> questionsBefore) {
|
||||
stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, pendingIncidentKeysFor(lead));
|
||||
}
|
||||
|
||||
private void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
|
||||
Set<String> questionsBefore, Set<IncidentLead> incidentsBefore) {
|
||||
stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, incidentsBefore,
|
||||
pendingUnmappedTargetKeysFor(lead));
|
||||
}
|
||||
|
||||
private void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
|
||||
Set<String> questionsBefore, Set<IncidentLead> incidentsBefore,
|
||||
Set<UnmappedTargetLead> unmappedTargetsBefore) {
|
||||
activeLeads.remove(lead);
|
||||
boolean racedIn = pendingReplyTargetsFor(lead).stream().anyMatch(t -> !repliesBefore.contains(t))
|
||||
|| pendingTicketIdsFor(lead).stream().anyMatch(t -> !ticketsBefore.contains(t))
|
||||
|| pendingQuestionTurnIdsFor(lead).stream().anyMatch(t -> !questionsBefore.contains(t));
|
||||
|| pendingQuestionTurnIdsFor(lead).stream().anyMatch(t -> !questionsBefore.contains(t))
|
||||
|| pendingIncidentKeysFor(lead).stream().anyMatch(i -> !incidentsBefore.contains(i))
|
||||
|| pendingUnmappedTargetKeysFor(lead).stream().anyMatch(i -> !unmappedTargetsBefore.contains(i));
|
||||
if (racedIn && activeLeads.putIfAbsent(lead, Boolean.TRUE) == null) {
|
||||
log.debug("push: new work for lead {} raced the reminder loop's stop — restarting", lead);
|
||||
scheduleNext(lead);
|
||||
@@ -495,18 +631,22 @@ public final class ReplyPushLoop {
|
||||
|
||||
/** Send one combined nudge covering everything currently pending for {@code lead}. */
|
||||
private void injectNudge(String lead, int replyReminderCount, int ticketReminderCount,
|
||||
int questionReminderCount) {
|
||||
int questionReminderCount, int incidentReminderCount,
|
||||
int unmappedTargetReminderCount) {
|
||||
// Re-read rather than threading it down from decide(): a reply can drain, a ticket be
|
||||
// collected, or a question be answered (or another arrive), between the decision and the
|
||||
// injection.
|
||||
Set<String> replyTargets = pendingReplyTargetsFor(lead);
|
||||
List<PendingTicket> tickets = pendingTicketsFor(lead);
|
||||
List<PendingQuestion> questions = pendingQuestionsFor(lead);
|
||||
if (replyTargets.isEmpty() && tickets.isEmpty() && questions.isEmpty()) {
|
||||
List<PendingIncident> incidents = pendingIncidentsFor(lead);
|
||||
List<PendingUnmappedTarget> unmappedTargets = pendingUnmappedTargetsFor(lead);
|
||||
if (replyTargets.isEmpty() && tickets.isEmpty() && questions.isEmpty() && incidents.isEmpty()
|
||||
&& unmappedTargets.isEmpty()) {
|
||||
log.debug("push: pending work for lead {} drained before the nudge could be sent", lead);
|
||||
return;
|
||||
}
|
||||
String nudge = formatNudge(replyTargets, tickets, questions);
|
||||
String nudge = formatNudge(replyTargets, tickets, questions, incidents, unmappedTargets);
|
||||
try {
|
||||
agents.send(lead, nudge);
|
||||
log.debug("push: nudge sent to lead {} (reply {}/{}, ticket {}/{}, question {}/{}; "
|
||||
@@ -515,6 +655,16 @@ public final class ReplyPushLoop {
|
||||
questionReminderCount + 1, maxReminders,
|
||||
replyTargets.size(), tickets.size(), questions.size());
|
||||
countNudge("delivered");
|
||||
for (PendingIncident incident : incidents) {
|
||||
if (pendingIncidents.remove(incident.key(), incident)) {
|
||||
deliveredIncidents.add(incident.key());
|
||||
}
|
||||
}
|
||||
for (PendingUnmappedTarget unmappedTarget : unmappedTargets) {
|
||||
if (pendingUnmappedTargets.remove(unmappedTarget.key(), unmappedTarget)) {
|
||||
deliveredUnmappedTargets.add(unmappedTarget.key());
|
||||
}
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("push: failed to nudge lead {} (reply {}/{}, ticket {}/{}, question {}/{}): {}",
|
||||
lead, replyReminderCount + 1, maxReminders, ticketReminderCount + 1, maxReminders,
|
||||
@@ -526,12 +676,13 @@ public final class ReplyPushLoop {
|
||||
// item already at or over the cap keeps riding along in the text (still pending, still
|
||||
// named) but its extra bumps here are inert: decide() already treats it as ineligible once
|
||||
// its count reaches maxReminders.
|
||||
bumpNudgeCounts(replyTargets, tickets, questions);
|
||||
bumpNudgeCounts(replyTargets, tickets, questions, incidents, unmappedTargets);
|
||||
}
|
||||
|
||||
/** Record that every one of these items was just named in a sent (or attempted) nudge. */
|
||||
private void bumpNudgeCounts(Set<String> replyTargets, List<PendingTicket> tickets,
|
||||
List<PendingQuestion> questions) {
|
||||
List<PendingQuestion> questions, List<PendingIncident> incidents,
|
||||
List<PendingUnmappedTarget> unmappedTargets) {
|
||||
for (String target : replyTargets) {
|
||||
pendingReplies.computeIfPresent(target, (t, e) -> new ReplyEntry(e.lead(), e.nudgeCount() + 1));
|
||||
}
|
||||
@@ -544,6 +695,14 @@ public final class ReplyPushLoop {
|
||||
new PendingQuestion(e.turnId(), e.ticket(), e.target(), e.lead(), e.question(),
|
||||
e.nudgeCount() + 1));
|
||||
}
|
||||
for (PendingIncident incident : incidents) {
|
||||
pendingIncidents.computeIfPresent(incident.key(), (id, e) -> new PendingIncident(e.key(),
|
||||
e.credential(), e.profiles(), e.targets(), e.remainingCoolOffSeconds(), e.nudgeCount() + 1));
|
||||
}
|
||||
for (PendingUnmappedTarget unmappedTarget : unmappedTargets) {
|
||||
pendingUnmappedTargets.computeIfPresent(unmappedTarget.key(), (id, e) ->
|
||||
new PendingUnmappedTarget(e.key(), e.nudgeCount() + 1));
|
||||
}
|
||||
}
|
||||
|
||||
/** Schedule the next tick on the scheduler thread pool. */
|
||||
@@ -556,7 +715,8 @@ public final class ReplyPushLoop {
|
||||
|
||||
/** Render everything pending for one lead as a single nudge line. */
|
||||
private static String formatNudge(Set<String> replyTargets, List<PendingTicket> tickets,
|
||||
List<PendingQuestion> questions) {
|
||||
List<PendingQuestion> questions, List<PendingIncident> incidents,
|
||||
List<PendingUnmappedTarget> unmappedTargets) {
|
||||
List<String> parts = new ArrayList<>();
|
||||
if (!replyTargets.isEmpty()) {
|
||||
parts.add(formatRepliesNudge(replyTargets));
|
||||
@@ -567,6 +727,12 @@ public final class ReplyPushLoop {
|
||||
if (!questions.isEmpty()) {
|
||||
parts.add(formatQuestionsNudge(questions));
|
||||
}
|
||||
if (!incidents.isEmpty()) {
|
||||
parts.addAll(incidents.stream().map(ReplyPushLoop::formatBackendIncidentNudge).toList());
|
||||
}
|
||||
if (!unmappedTargets.isEmpty()) {
|
||||
parts.addAll(unmappedTargets.stream().map(ReplyPushLoop::formatUnmappedTargetNudge).toList());
|
||||
}
|
||||
return String.join(" | ", parts);
|
||||
}
|
||||
|
||||
@@ -606,6 +772,16 @@ public final class ReplyPushLoop {
|
||||
return QUESTIONS_NUDGE_FORMAT.formatted(pending.size(), ids);
|
||||
}
|
||||
|
||||
private static String formatBackendIncidentNudge(PendingIncident incident) {
|
||||
return BACKEND_INCIDENT_NUDGE_FORMAT.formatted(incident.credential(), incident.remainingCoolOffSeconds(),
|
||||
String.join(", ", incident.profiles()), String.join(", ", incident.targets()));
|
||||
}
|
||||
|
||||
private static String formatUnmappedTargetNudge(PendingUnmappedTarget unmappedTarget) {
|
||||
return BACKEND_TARGET_UNMAPPED_NUDGE_FORMAT.formatted(unmappedTarget.key().target(),
|
||||
unmappedTarget.key().reason());
|
||||
}
|
||||
|
||||
// --- lifecycle -----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@@ -627,6 +803,10 @@ public final class ReplyPushLoop {
|
||||
pendingReplies.clear();
|
||||
pendingTickets.clear();
|
||||
pendingQuestions.clear();
|
||||
pendingIncidents.clear();
|
||||
deliveredIncidents.clear();
|
||||
pendingUnmappedTargets.clear();
|
||||
deliveredUnmappedTargets.clear();
|
||||
}
|
||||
|
||||
/** @see #stop() */
|
||||
|
||||
@@ -4,11 +4,8 @@ import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.LoggerContext;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.TestTurnTokens;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
@@ -762,202 +759,4 @@ class CompletionResolverTest {
|
||||
assertEquals("partial (configured: [terra]; not configured: [gx10])",
|
||||
CompletionResolver.coverage(Set.of("terra", "gx10"), Set.of("terra")));
|
||||
}
|
||||
|
||||
// --- fleetd#201 Unit 1: target-keyed backend-error pattern + typed sink ----------------------
|
||||
|
||||
@Test
|
||||
void aConfiguredBackendErrorPatternClassifiesAMatchAsAFailureAndNotifiesTheSinkOnce() {
|
||||
String block = "⏺ 503 Service Unavailable: upstream credential rejected\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "a configured-pattern match still resolves the blocked send");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind());
|
||||
assertEquals(1, notified.size(), "the sink fires exactly once for the winning classification");
|
||||
assertEquals("term_a: 503 Service Unavailable: upstream credential rejected", notified.get(0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLosingBackendErrorClassificationNeverNotifiesTheSink() {
|
||||
// The waiter was already resolved (e.g. by the worker's own fleet_reply) before this scrape
|
||||
// landed — resolveFailure loses the race and must return false, so the sink must not fire.
|
||||
String block = "⏺ 503 Service Unavailable: upstream credential rejected\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null);
|
||||
assertTrue(rendezvous.resolveCompletion(waiter, "already replied")); // fleet_reply won first
|
||||
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(notified.isEmpty(), "a classification that loses the race must not fire the sink");
|
||||
assertEquals("already replied", waiter.getNow(null).text(), "the earlier resolution stands untouched");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBackendErrorClassificationThatLosesARealRaceDuringTheScrapeNeverNotifiesTheSink() {
|
||||
// The test above pre-resolves the waiter BEFORE calling resolve(), so it is caught by
|
||||
// resolve()'s own top-of-method isDone() guard before ever reaching the win-gated
|
||||
// resolveFailure call this unit adds — it proves the outcome, not the specific gate. This
|
||||
// test forces the actual race window fleetd#201 must be safe in: a genuine fleet_reply lands
|
||||
// WHILE the herdr scrape for this turn is in flight, i.e. AFTER resolve() has already passed
|
||||
// the top guard (waiter not done yet) and committed to reading the pane, but BEFORE it
|
||||
// reaches the "if (rendezvous.resolveFailure(...))" line below. The side effect is attached
|
||||
// to the one real I/O step resolve() performs between those two points: the herdr
|
||||
// "agent.read" call itself.
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
var waiter = rendezvous.open("term_a");
|
||||
ObjectMapper mapper = new ObjectMapper();
|
||||
HerdrClient racingDuringScrape = new HerdrClient() {
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
if ("agent.read".equals(method)) {
|
||||
// The real fleet_reply that wins the race, landing mid-scrape.
|
||||
assertTrue(rendezvous.resolve("term_a", "a genuine fleet_reply landed first"),
|
||||
"the racing reply must land while the waiter is still open");
|
||||
}
|
||||
try {
|
||||
return mapper.readTree(mapper.writeValueAsString(java.util.Map.of(
|
||||
"type", "agent_read",
|
||||
"read", java.util.Map.of("text",
|
||||
"⏺ 503 Service Unavailable: upstream credential rejected\n❯ "))));
|
||||
} catch (Exception e) {
|
||||
throw new RuntimeException(e);
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
};
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(racingDuringScrape), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
|
||||
|
||||
var turn = new CompletionResolver.InFlight(waiter, null); // not done yet — passes the top guard
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(notified.isEmpty(),
|
||||
"a classification that loses a real race during the scrape must not fire the sink");
|
||||
assertEquals(Rendezvous.Kind.REPLY, waiter.getNow(null).kind(), "the genuine fleet_reply stands");
|
||||
assertEquals("a genuine fleet_reply landed first", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void exhaustionKeepsWinningOverBackendErrorEvenWhenBothPatternsMatchTheSameLine() {
|
||||
// A line matching both a configured exhausted pattern AND a configured backend-error pattern
|
||||
// must classify BACKEND_EXHAUSTED and call only the ExhaustionSink — the ordering fleetd#578
|
||||
// already relies on must not change.
|
||||
String block = "⏺ The usage limit has been reached. Try again later.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup exhausted = target -> Pattern.compile("usage limit has been reached");
|
||||
BackendErrorPatternLookup backendErrors = target -> Pattern.compile("(?i)usage limit");
|
||||
java.util.List<String> exhaustedNotified = new java.util.ArrayList<>();
|
||||
java.util.List<String> backendErrorNotified = new java.util.ArrayList<>();
|
||||
ExhaustionSink exhaustionSink = (target, reason) -> exhaustedNotified.add(target);
|
||||
BackendErrorSink backendErrorSink = (target, matchedLine, reason) -> backendErrorNotified.add(target);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
exhausted, exhaustionSink, backendErrors, backendErrorSink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, waiter.getNow(null).kind(),
|
||||
"exhaustion must keep winning over backend error");
|
||||
assertEquals(1, exhaustedNotified.size(), "the exhaustion sink fires");
|
||||
assertTrue(backendErrorNotified.isEmpty(), "the backend-error sink must never fire for this turn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aConfiguredPatternAlsoClassifiesTheRawScrapeFallbackAndNotifiesTheSink() {
|
||||
// No ⏺ marker and leading TUI chrome ⇒ lastAssistantBlock() yields "", so classification must
|
||||
// fall back to the raw scrape (fleetd#211) — and it must use the configured pattern too.
|
||||
String block = """
|
||||
╭──────────────────────────────────────╮
|
||||
503 Service Unavailable: upstream credential rejected
|
||||
""";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "a raw-scrape configured-pattern match still resolves the blocked send");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"classified from the raw scrape even though the trimmed block was empty");
|
||||
assertEquals(1, notified.size(), "the sink fires exactly once from the raw-scrape path");
|
||||
assertTrue(notified.get(0).contains("503 Service Unavailable"), notified.get(0));
|
||||
}
|
||||
|
||||
// --- fleetd#201 Unit 1: classification inside the fleetd#164 MIN_TURN_NANOS floor -------------
|
||||
|
||||
@Test
|
||||
void aMatchingErrorInsideTheFloorIsTypedAndNotifiesTheSink() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ 503 Service Unavailable: upstream credential rejected\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
long[] clock = {10_000_000_000L};
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink, () -> clock[0]);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null, clock[0]); // delivered "now"
|
||||
clock[0] += CompletionResolver.MIN_TURN_NANOS - 1; // inside the floor
|
||||
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone());
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind());
|
||||
assertEquals(1, notified.size(), "a matching error inside the floor notifies the sink");
|
||||
assertTrue(notified.get(0).contains("503 Service Unavailable"), notified.get(0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonMatchInsideTheFloorStaysGenericAndNeverNotifiesTheSink() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ still starting up\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
long[] clock = {10_000_000_000L};
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink, () -> clock[0]);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null, clock[0]); // delivered "now"
|
||||
clock[0] += CompletionResolver.MIN_TURN_NANOS - 1; // inside the floor
|
||||
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone());
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"still a failure — the floor itself, not the pattern, is why");
|
||||
assertTrue(waiter.getNow(null).text().contains("too fast to be real work"),
|
||||
"a non-match inside the floor stays the generic too-fast reason: " + waiter.getNow(null).text());
|
||||
assertTrue(notified.isEmpty(), "a non-match must never notify the typed sink");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -43,6 +43,7 @@ class ReplyPushLoopTest {
|
||||
private static final String PRIMARY = "term_primary";
|
||||
private static final String WORKER = "term_worker";
|
||||
private static final String WORKER2 = "term_worker2";
|
||||
private static final String OTHER_PRIMARY = "term_other_primary";
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
|
||||
private PrimaryRegistry registry;
|
||||
@@ -529,6 +530,106 @@ class ReplyPushLoopTest {
|
||||
assertTrue(nudge.contains("task-1"), "the ticket must not be dropped: " + nudge);
|
||||
}
|
||||
|
||||
// --- CB-201: backend credential outage nudges -----------------------------------------------
|
||||
|
||||
@Test
|
||||
void oneBackendIncidentForTwoWorkersOfOneLeadSendsOnceAndIsOneShot() throws Exception {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
registry.recordDelegation(WORKER, PRIMARY);
|
||||
registry.recordDelegation(WORKER2, PRIMARY);
|
||||
var loop = loop(2, 100);
|
||||
|
||||
loop.onBackendIncident("incident-1", List.of(WORKER, WORKER2), "cred-a",
|
||||
List.of("terra", "codex"), 47);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS), "one lead gets one outage injection");
|
||||
Thread.sleep(200);
|
||||
assertEquals(1, rec.sendCount(), "two workers for one lead must send only one outage notice");
|
||||
String nudge = rec.sentParams().getFirst().getValue().toString();
|
||||
assertTrue(nudge.contains("cred-a"));
|
||||
assertTrue(nudge.contains("codex, terra"));
|
||||
assertTrue(nudge.contains(WORKER) && nudge.contains(WORKER2));
|
||||
assertTrue(nudge.contains("47 remaining seconds"));
|
||||
assertTrue(nudge.contains("fleet_list"));
|
||||
|
||||
loop.onBackendIncident("incident-1", List.of(WORKER, WORKER2), "cred-a",
|
||||
List.of("terra", "codex"), 47);
|
||||
Thread.sleep(250);
|
||||
assertEquals(1, rec.sendCount(), "a delivered incident id must never be sent again");
|
||||
}
|
||||
|
||||
@Test
|
||||
void oneBackendIncidentSendsOnceToEachAffectedLead() throws Exception {
|
||||
var rec = recordingClient();
|
||||
rec.sendLatch = new CountDownLatch(2);
|
||||
agents = new AgentControl(rec);
|
||||
registry.recordDelegation(WORKER, PRIMARY);
|
||||
registry.recordDelegation(WORKER2, OTHER_PRIMARY);
|
||||
var loop = loop(1, 50);
|
||||
|
||||
loop.onBackendIncident("incident-2", List.of(WORKER, WORKER2), "cred-a", List.of("terra"), 60);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS), "each affected lead gets its one notice");
|
||||
assertEquals(2, rec.sendCount());
|
||||
}
|
||||
|
||||
@Test
|
||||
void backendIncidentWaitsForBusyLeadThenCombinesWithFailedTicket() throws Exception {
|
||||
var rec = new BusyThenIdleHerdrClient(2);
|
||||
agents = new AgentControl(rec);
|
||||
var loop = loop(1, 50);
|
||||
|
||||
loop.onTicketTerminal("task-failed", WORKER, true);
|
||||
loop.onBackendIncident("incident-3", List.of(WORKER), "cred-b", List.of("terra"), 31);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS), "busy lead should receive deferred combined nudge");
|
||||
assertEquals(1, rec.sendCount(), "ticket and outage must use the same injection");
|
||||
String nudge = rec.sentParams().getFirst().getValue().toString();
|
||||
assertTrue(nudge.contains("task-failed") && nudge.contains("cred-b"),
|
||||
"the one nudge must name both the failed ticket and outage: " + nudge);
|
||||
}
|
||||
|
||||
@Test
|
||||
void failedBackendIncidentSendRetriesWithinTheExistingReminderCap() throws Exception {
|
||||
var rec = new ThrowOnceHerdrClient();
|
||||
agents = new AgentControl(rec);
|
||||
var loop = loop(2, 50);
|
||||
|
||||
loop.onBackendIncident("incident-4", List.of(WORKER), "cred-c", List.of("terra"), 20);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS), "failed first send must retry under maxReminders");
|
||||
assertEquals(2, rec.promptAttempts.get(), "one failed send and one retry use the existing cap");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unmappedBackendTargetUsesTheKnownLeadSchedule() throws Exception {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
var loop = loop(1, 50);
|
||||
|
||||
loop.onBackendTargetUnmapped(WORKER, "no credential matched");
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS));
|
||||
String nudge = ((Map<?, ?>) rec.sentParams().getFirst().getValue()).get("text").toString();
|
||||
assertEquals("A backend error on worker term_worker could not be mapped to a credential "
|
||||
+ "(no credential matched) — no cool-off was applied. "
|
||||
+ "Run fleet_list to check that worker.",
|
||||
nudge);
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopClearsPendingBackendIncidents() {
|
||||
agents = agentWithStatus("idle");
|
||||
var loop = loop(1, 100_000);
|
||||
loop.onBackendIncident("incident-stop", List.of(WORKER), "cred-stop", List.of("terra"), 60);
|
||||
|
||||
loop.stop();
|
||||
|
||||
assertEquals(ReplyPushLoop.Action.STOP, loop.decide(PRIMARY, 0, 0, 0),
|
||||
"stop must clear a pending backend incident as well as older sources");
|
||||
}
|
||||
|
||||
// --- CB-590 follow-up: per-source reminder budgets — the regression this round exists for ---
|
||||
|
||||
@Test
|
||||
@@ -928,6 +1029,31 @@ class ReplyPushLoopTest {
|
||||
return new RecordingHerdrClient();
|
||||
}
|
||||
|
||||
/** Fails its first prompt call, then records the retry. */
|
||||
private static final class ThrowOnceHerdrClient implements HerdrClient {
|
||||
private final AtomicInteger promptAttempts = new AtomicInteger();
|
||||
private final CountDownLatch sendLatch = new CountDownLatch(1);
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
if ("agent.get".equals(method)) {
|
||||
return MAPPER.createObjectNode().set("agent", MAPPER.createObjectNode()
|
||||
.put("terminal_id", PRIMARY).put("agent_status", "idle"));
|
||||
}
|
||||
if ("agent.prompt".equals(method) && promptAttempts.getAndIncrement() == 0) {
|
||||
throw new IllegalStateException("first prompt fails");
|
||||
}
|
||||
if ("agent.prompt".equals(method)) {
|
||||
sendLatch.countDown();
|
||||
}
|
||||
return MAPPER.createObjectNode();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Thread-safe fake that reports {@code working} (not injectable) for its first
|
||||
* {@code busyChecks} status calls, then {@code idle} forever after — used to prove work queued
|
||||
|
||||
Reference in New Issue
Block a user