Merge CB-580: fail a ticket when its member reaches a terminal health state
Verified by the lead on0af902e: mvn -f bridged/pom.xml clean install, unpiped, in a scratch worktree — MVN_EXIT=0, Tests run: 689, Failures: 0, Errors: 0, BUILD SUCCESS. Reviewed by the lead reading the diff. The member's own report was lost to the idle reaper before it was collected, so there was no author write-up to review against. All three defects that sank3b2f395are absent: * failTarget is required — one constructor, Objects.requireNonNull, no defaulting overload anywhere in the repo, and Bridged.java:365 updated to pass messages::abandon. * The failure fires inside reportTransition after its `if (previous == next) return;` guard, so an unchanged tick cannot reach it. * No AtomicReference; MessageService is passed directly, so there is no empty window. abandon(String target, String reason) confirmed as CB-568's target-wide operation that resolves waiters as a failure rather than letting them time out. Known limitation, accepted and covered by its own test: states.put records the new state before the bounded retries run, so if all three attempts throw, the tickets stay pending and no later tick retries. It is logged at WARN, and the retries carry no backoff.
This commit was merged in pull request #56.
This commit is contained in:
@@ -20,8 +20,10 @@ import org.slf4j.LoggerFactory;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.function.BiConsumer;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
class FleetHealthMonitorTest {
|
||||
@Test void oneTickUsesOneFleetListForAnyRosterSize() {
|
||||
@@ -37,7 +39,8 @@ class FleetHealthMonitorTest {
|
||||
herdr.calls.clear();
|
||||
MessageService messages = new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox());
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents, sessions::roster, messages, scheduler, () -> 1, 60);
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents, sessions::roster, messages, scheduler, () -> 1, 60,
|
||||
(_, _) -> { });
|
||||
monitor.tick();
|
||||
monitor.stop();
|
||||
assertEquals(1, herdr.calls.stream().filter(call -> call.method().equals("agent.list")).count());
|
||||
@@ -49,7 +52,7 @@ class FleetHealthMonitorTest {
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents, java.util.List::of,
|
||||
new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox()),
|
||||
scheduler, () -> 1, 60);
|
||||
scheduler, () -> 1, 60, (_, _) -> { });
|
||||
monitor.tick();
|
||||
herdr.healthy(true);
|
||||
monitor.tick();
|
||||
@@ -68,7 +71,7 @@ class FleetHealthMonitorTest {
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents, java.util.List::of,
|
||||
new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox()),
|
||||
scheduler, () -> 1, 60);
|
||||
scheduler, () -> 1, 60, (_, _) -> { });
|
||||
monitor.reportTransition("term_a", HealthState.TURN_BOUNDARY_LOST);
|
||||
monitor.reportTransition("term_a", HealthState.TURN_BOUNDARY_LOST);
|
||||
monitor.stop();
|
||||
@@ -78,4 +81,90 @@ class FleetHealthMonitorTest {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
// --- CB-580: a member that reaches GONE/NEVER_READY must fail its waiting tickets
|
||||
|
||||
private static FleetHealthMonitor monitorWith(BiConsumer<String, String> failTarget) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
return new FleetHealthMonitor(agents, java.util.List::of,
|
||||
new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox()),
|
||||
scheduler, () -> 1, 60, failTarget);
|
||||
}
|
||||
|
||||
@Test void terminalTransitionFailsTheTargetOnce() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.stop();
|
||||
assertEquals(1, failTarget.calls.size());
|
||||
assertEquals("term_a", failTarget.calls.get(0).target());
|
||||
assertTrue(failTarget.calls.get(0).reason().contains("GONE"));
|
||||
}
|
||||
|
||||
@Test void neverReadyNamesItselfAsTheReason() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.NEVER_READY);
|
||||
monitor.stop();
|
||||
assertEquals(1, failTarget.calls.size());
|
||||
assertTrue(failTarget.calls.get(0).reason().contains("NEVER_READY"));
|
||||
}
|
||||
|
||||
@Test void stayingInATerminalStateProducesOneFailureNotN() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.stop();
|
||||
assertEquals(1, failTarget.calls.size());
|
||||
}
|
||||
|
||||
@Test void aNonTerminalFaultStateDoesNotFailTheTarget() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.TURN_BOUNDARY_LOST);
|
||||
monitor.stop();
|
||||
assertEquals(0, failTarget.calls.size());
|
||||
}
|
||||
|
||||
@Test void failTargetRetryIsBounded() {
|
||||
AlwaysThrowingFailTarget failTarget = new AlwaysThrowingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.stop();
|
||||
assertEquals(FleetHealthMonitor.MAX_FAIL_TARGET_ATTEMPTS, failTarget.calls);
|
||||
}
|
||||
|
||||
@Test void exhaustedRetryStillDoesNotRefireOnAnUnchangedTick() {
|
||||
AlwaysThrowingFailTarget failTarget = new AlwaysThrowingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
int afterFirstTransition = failTarget.calls;
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.stop();
|
||||
assertEquals(afterFirstTransition, failTarget.calls);
|
||||
}
|
||||
|
||||
private record RecordedCall(String target, String reason) { }
|
||||
|
||||
private static final class RecordingFailTarget implements BiConsumer<String, String> {
|
||||
final java.util.List<RecordedCall> calls = new java.util.ArrayList<>();
|
||||
|
||||
@Override public void accept(String target, String reason) {
|
||||
calls.add(new RecordedCall(target, reason));
|
||||
}
|
||||
}
|
||||
|
||||
private static final class AlwaysThrowingFailTarget implements BiConsumer<String, String> {
|
||||
int calls = 0;
|
||||
|
||||
@Override public void accept(String target, String reason) {
|
||||
calls++;
|
||||
throw new RuntimeException("boom");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user