diff --git a/fleetd/src/test/java/dev/ltms/fleet/health/FleetHealthMonitorTest.java b/fleetd/src/test/java/dev/ltms/fleet/health/FleetHealthMonitorTest.java index 149422c..e9a0346 100644 --- a/fleetd/src/test/java/dev/ltms/fleet/health/FleetHealthMonitorTest.java +++ b/fleetd/src/test/java/dev/ltms/fleet/health/FleetHealthMonitorTest.java @@ -30,6 +30,7 @@ import java.util.concurrent.ScheduledFuture; import java.util.concurrent.ScheduledThreadPoolExecutor; import java.util.concurrent.TimeUnit; import java.util.concurrent.atomic.AtomicLong; +import java.util.concurrent.atomic.AtomicReference; import java.util.function.BiConsumer; import java.util.function.LongSupplier; @@ -277,6 +278,58 @@ class FleetHealthMonitorTest { } } + /** + * fleetd #386 follow-up, added on merge. The fix carries a PER-MEMBER drift baseline, so drift + * from a sleep that happened BEFORE a member went busy is never charged to that member. The + * two tests shipped with the fix both start with the member already BUSY, so a single global + * baseline passes them — this one fails without the per-member map. + * + *
Order matters: the host sleeps while nothing is busy, and only then does a member take a
+ * turn. Its stall clock must start at zero.
+ */
+ @Test void driftFromASleepBeforeAMemberWentBusyIsNotChargedToThatMember() {
+ Logger logger = (Logger) LoggerFactory.getLogger(FleetHealthMonitor.class);
+ ListAppender> roster = new AtomicReference<>(List.of());
+ FakeHerdr herdr = new FakeHerdr().withAgent("busy", "term_busy", "pane_busy", "tab_busy");
+ var scheduler = Executors.newSingleThreadScheduledExecutor();
+ AgentControl agents = new AgentControl(herdr);
+ FleetHealthMonitor monitor = new FleetHealthMonitor(agents, roster::get,
+ new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox()),
+ scheduler, mono::get, real::get, 60, 600, (_, _) -> { });
+
+ monitor.tick(); // baseline, no members yet
+
+ // The host sleeps for 700s with nobody busy: the monotonic clock stands still.
+ real.set(TimeUnit.SECONDS.toNanos(700));
+ monitor.tick();
+
+ // Awake again. Only NOW does a member start a turn, with a fresh activity stamp taken
+ // from the monotonic clock. Both clocks advance together from here.
+ mono.set(TimeUnit.SECONDS.toNanos(10));
+ real.set(TimeUnit.SECONDS.toNanos(710));
+ roster.set(List.of(member("term_busy", MemberSession.State.BUSY,
+ 0, TimeUnit.SECONDS.toNanos(10))));
+ monitor.tick();
+
+ mono.set(TimeUnit.SECONDS.toNanos(20));
+ real.set(TimeUnit.SECONDS.toNanos(720));
+ monitor.tick();
+ monitor.stop();
+
+ assertEquals(0, appender.list.stream().filter(event -> event.getFormattedMessage()
+ .contains("member=term_busy state=STALL_SUSPECTED")).count(),
+ "the member has been busy for 10s, not 710s — the earlier sleep is not its stall");
+ } finally {
+ logger.detachAppender(appender);
+ }
+ }
+
private static FleetHealthMonitor monitorWithClocks(FakeHerdr herdr, List