t386: pin the per-member drift baseline the fix's own tests left open
The two tests merged with #386 both start with the member already BUSY, so a single global drift baseline passes them. This one sleeps the host while nothing is busy and only then starts a turn, which fails without the per-member map.
This commit is contained in:
@@ -30,6 +30,7 @@ import java.util.concurrent.ScheduledFuture;
|
||||
import java.util.concurrent.ScheduledThreadPoolExecutor;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
@@ -277,6 +278,58 @@ class FleetHealthMonitorTest {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #386 follow-up, added on merge. The fix carries a PER-MEMBER drift baseline, so drift
|
||||
* from a sleep that happened BEFORE a member went busy is never charged to that member. The
|
||||
* two tests shipped with the fix both start with the member already BUSY, so a single global
|
||||
* baseline passes them — this one fails without the per-member map.
|
||||
*
|
||||
* <p>Order matters: the host sleeps while nothing is busy, and only then does a member take a
|
||||
* turn. Its stall clock must start at zero.
|
||||
*/
|
||||
@Test void driftFromASleepBeforeAMemberWentBusyIsNotChargedToThatMember() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
AtomicLong mono = new AtomicLong(0);
|
||||
AtomicLong real = new AtomicLong(0);
|
||||
AtomicReference<List<MemberSession>> roster = new AtomicReference<>(List.of());
|
||||
FakeHerdr herdr = new FakeHerdr().withAgent("busy", "term_busy", "pane_busy", "tab_busy");
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents, roster::get,
|
||||
new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox()),
|
||||
scheduler, mono::get, real::get, 60, 600, (_, _) -> { });
|
||||
|
||||
monitor.tick(); // baseline, no members yet
|
||||
|
||||
// The host sleeps for 700s with nobody busy: the monotonic clock stands still.
|
||||
real.set(TimeUnit.SECONDS.toNanos(700));
|
||||
monitor.tick();
|
||||
|
||||
// Awake again. Only NOW does a member start a turn, with a fresh activity stamp taken
|
||||
// from the monotonic clock. Both clocks advance together from here.
|
||||
mono.set(TimeUnit.SECONDS.toNanos(10));
|
||||
real.set(TimeUnit.SECONDS.toNanos(710));
|
||||
roster.set(List.of(member("term_busy", MemberSession.State.BUSY,
|
||||
0, TimeUnit.SECONDS.toNanos(10))));
|
||||
monitor.tick();
|
||||
|
||||
mono.set(TimeUnit.SECONDS.toNanos(20));
|
||||
real.set(TimeUnit.SECONDS.toNanos(720));
|
||||
monitor.tick();
|
||||
monitor.stop();
|
||||
|
||||
assertEquals(0, appender.list.stream().filter(event -> event.getFormattedMessage()
|
||||
.contains("member=term_busy state=STALL_SUSPECTED")).count(),
|
||||
"the member has been busy for 10s, not 710s — the earlier sleep is not its stall");
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetHealthMonitor monitorWithClocks(FakeHerdr herdr, List<MemberSession> roster,
|
||||
java.util.concurrent.ScheduledExecutorService scheduler, LongSupplier clock,
|
||||
LongSupplier realtimeClock, long intervalSeconds, long workingSuspectAfterSeconds,
|
||||
|
||||
Reference in New Issue
Block a user