From 4af919af0ced48cafddb88c6270458139ee57b9d Mon Sep 17 00:00:00 2001 From: Dai Ha Date: Thu, 1 Oct 2026 16:41:40 +0200 Subject: [PATCH] fleetd #629 follow-up: bound FleetdAssemblyFleetAppTest with a SEPARATE_THREAD @Timeout MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The mutation cycle for #629 proved deleting the fix (reverting FleetdAssembly's awaitHerdr call back to a hardcoded Fleetd::sleepHerdrPoll) doesn't just make a test fail — it hangs forever, because the test's fake nanoClock() only advances when ports.herdrPollWait() is actually called. SAME_THREAD @Timeout (JUnit's default) can't catch that: it only measures elapsed time after the test method returns on its own, which never happens here. A class-level @Timeout(10s, SEPARATE_THREAD) does, since it runs the test on its own thread and interrupts it on timeout. Verified by re-running the same mutation: the suite now fails fast with a named TimeoutException instead of hanging indefinitely. --- .../ltms/fleet/FleetdAssemblyFleetAppTest.java | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/fleetd/src/test/java/dev/ltms/fleet/FleetdAssemblyFleetAppTest.java b/fleetd/src/test/java/dev/ltms/fleet/FleetdAssemblyFleetAppTest.java index 36ca40d..52e49ba 100644 --- a/fleetd/src/test/java/dev/ltms/fleet/FleetdAssemblyFleetAppTest.java +++ b/fleetd/src/test/java/dev/ltms/fleet/FleetdAssemblyFleetAppTest.java @@ -8,6 +8,7 @@ import dev.ltms.fleet.herdr.HerdrClient; import io.javalin.Javalin; import org.junit.jupiter.api.AfterEach; import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.Timeout; import org.junit.jupiter.api.io.TempDir; import java.net.URI; @@ -71,7 +72,23 @@ import static org.junit.jupiter.api.Assertions.assertEquals; * below is what this class relies on for CB-185's {@code FleetApp} half; {@code * FleetAppTwoDaemonTest} remains the full behavioural proof that {@code FleetApp} itself merges * {@code /sessions} correctly once handed two clients. + * + *

fleetd #629 follow-up. The fix below (see {@link TwoHerdrResourcePorts}) + * makes {@link #healthzGoesRedWhenTheLeadDaemonIsDownEvenThoughTheMemberIsUp}'s fake {@code + * nanoClock()} frozen unless {@code herdrPollWait()} itself advances it. That is a sharper pin + * than an assertion — if a future edit to {@code FleetdAssembly} ever bypasses {@code + * ports.herdrPollWait()} again (e.g. reverting to a hardcoded {@code Thread.sleep}), the clock + * never advances, {@code Fleetd#awaitHerdr}'s deadline is never reached, and this test hangs + * forever instead of failing — proven by deliberately reintroducing that exact regression while + * fixing this ticket. {@code @Timeout} turns that silent hang into a bounded, named test failure: + * {@code SEPARATE_THREAD} so JUnit's timeout governor can actually interrupt a thread stuck in a + * real {@code Thread.sleep} loop (the default {@code SAME_THREAD} mode cannot — it only measures + * elapsed time after the test method returns on its own, which never happens here). 10 seconds is + * roughly 150x the real passing times measured here (~0.06s), so a slow CI machine has no reason + * to flake, and it is still 3x faster than discovering the regression by burning a CI job's whole + * wall-clock budget. */ +@Timeout(value = 10, unit = TimeUnit.SECONDS, threadMode = Timeout.ThreadMode.SEPARATE_THREAD) class FleetdAssemblyFleetAppTest { private static final class TwoHerdrResourcePorts implements ResourcePorts {