Compare commits
16 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 141ae3b04d | |||
| faefea14c4 | |||
| ea6896f2ef | |||
| d7f94cafa2 | |||
| 096f08c866 | |||
| 4eb720029c | |||
| 0db6d31dc2 | |||
| 158a2a84b5 | |||
| 41ebc9cf69 | |||
| 619769bb52 | |||
| 180c953c42 | |||
| 09c37061c1 | |||
| 2be287ea03 | |||
| c3b0406826 | |||
| 9dca376604 | |||
| 91be5079d6 |
@@ -20,6 +20,13 @@ fleetd.out
|
||||
fleetd/fleetd.out
|
||||
logs/
|
||||
|
||||
# fleetd #635 follow-up — scripts/config-edit.sh's backup directory. No leading slash, so this is
|
||||
# ignored at every depth: the real one lives under fleetd/ (also named in fleetd/.gitignore, next
|
||||
# to the config it backs up), and scripts/test-config-edit.sh's own throwaway fixtures build one
|
||||
# under the repo root while the suite runs. --config can point anywhere, so the directory name is
|
||||
# ignored everywhere rather than only where the live daemon happens to use it.
|
||||
.config-backups/
|
||||
|
||||
# fleetd #480: the lead rollover handover file. `leadRollover.handoverPath` points here, and the
|
||||
# outgoing lead rewrites it on every rollover. It is a snapshot of one moment's live state —
|
||||
# unpushed branches, running builds, open questions — so it is stale the moment it is written and
|
||||
|
||||
@@ -7,6 +7,13 @@ dependency-reduced-pom.xml
|
||||
fleetd.yaml
|
||||
bridged.yaml
|
||||
|
||||
# fleetd #635 follow-up — scripts/config-edit.sh's backups of fleetd.yaml. A backup of a file
|
||||
# that must never be committed inherits that requirement. The directory is the real protection
|
||||
# (it keeps working even if the backup naming changes); the glob is a backstop for a stray
|
||||
# backup written the old way, directly beside fleetd.yaml, or by an older copy of the script.
|
||||
.config-backups/
|
||||
fleetd.yaml.bak.*
|
||||
|
||||
# CB-505 audit trail + daemon stdout/stderr — runtime records, never source
|
||||
logs/
|
||||
|
||||
|
||||
@@ -0,0 +1,183 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.fleet.msg.LeadChannel;
|
||||
import dev.ltms.fleet.msg.LeadChannelHandle;
|
||||
import dev.ltms.fleet.msg.LeadMessage;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertSame;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 step 4 ranks 3 and 8: the assembled daemon must use the AMQP openers from {@link
|
||||
* ResourcePorts}, and its startup report must describe the object the runtime actually owns. These
|
||||
* fakes never open a socket.
|
||||
*/
|
||||
class FleetdAssemblyAmqpOpenersTest {
|
||||
|
||||
private static final String COORD_ID = "assembly-test";
|
||||
|
||||
private static final class DurableReplyInbox implements ReplyInbox {
|
||||
@Override public void own(String target) { }
|
||||
@Override public void release(String target) { }
|
||||
@Override public void publish(String target, String msgId, String content) { }
|
||||
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||
@Override public boolean ack(String target, String msgId) { return false; }
|
||||
}
|
||||
|
||||
private static final class DurableLeadMailbox implements LeadChannelHandle {
|
||||
@Override public void publish(String toCoordId, LeadMessage message) { }
|
||||
@Override public List<LeadMessage> peek() { return List.of(); }
|
||||
@Override public void ack(String msgId) { }
|
||||
@Override public String selfCoordId() { return COORD_ID; }
|
||||
@Override public boolean heldDurable() { return true; }
|
||||
@Override public MailboxState inspect(String coordId) { return MailboxState.unknown(coordId); }
|
||||
@Override public void close() { }
|
||||
}
|
||||
|
||||
private static final class RecordingPorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
final DurableReplyInbox replyInbox = new DurableReplyInbox();
|
||||
final DurableLeadMailbox leadMailbox = new DurableLeadMailbox();
|
||||
final AtomicInteger replyOpenCalls = new AtomicInteger();
|
||||
final AtomicInteger mailboxOpenCalls = new AtomicInteger();
|
||||
final boolean openSucceeds;
|
||||
|
||||
RecordingPorts(boolean openSucceeds) {
|
||||
this.openSucceeds = openSucceeds;
|
||||
}
|
||||
|
||||
@Override public Map<String, String> environment() { return Map.of(); }
|
||||
@Override public HerdrClient connectHerdr(Path socketPath) { return herdr; }
|
||||
@Override public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> {
|
||||
replyOpenCalls.incrementAndGet();
|
||||
if (!openSucceeds) throw new IllegalStateException("fake reply broker is down");
|
||||
return replyInbox;
|
||||
};
|
||||
}
|
||||
@Override public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfId, prefetch) -> {
|
||||
mailboxOpenCalls.incrementAndGet();
|
||||
if (!openSucceeds) throw new IllegalStateException("fake coordination broker is down");
|
||||
return leadMailbox;
|
||||
};
|
||||
}
|
||||
@Override public LongSupplier nanoClock() { return System::nanoTime; }
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
@Override public LongSupplier wallClockNanos() { return System::nanoTime; }
|
||||
@Override public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
@Override public void addShutdownHook(Runnable hook) { }
|
||||
@Override public void startHttp(Javalin app, String host, int port) { }
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Files.createDirectories(dir);
|
||||
Path config = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(config, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
broker:
|
||||
uri: "amqp://fake-reply-broker/vh"
|
||||
coordinator:
|
||||
uri: "amqp://fake-coordination-broker/vh"
|
||||
selfId: "assembly-test"
|
||||
""");
|
||||
return FleetConfig.load(config);
|
||||
}
|
||||
|
||||
private static FleetdRuntime assemble(Path dir, RecordingPorts ports) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
return FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, new ConfigRef(dir.resolve("fleetd.yaml"), cfg),
|
||||
new SubscriptionGuard(cfg.guard().hostSet())), ports);
|
||||
}
|
||||
|
||||
private static boolean reportContains(ListAppender<ILoggingEvent> appender, String text) {
|
||||
return appender.list.stream().map(ILoggingEvent::getFormattedMessage).anyMatch(message -> message.contains(text));
|
||||
}
|
||||
|
||||
@Test
|
||||
void assembledAmqpOpenersAndTheirReportsAgreeOnDurableAndFallbackStates(@TempDir Path dir) throws Exception {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
Level oldLevel = logger.getLevel();
|
||||
ListAppender<ILoggingEvent> reports = new ListAppender<>();
|
||||
reports.start();
|
||||
logger.setLevel(Level.INFO);
|
||||
logger.addAppender(reports);
|
||||
try {
|
||||
RecordingPorts durablePorts = new RecordingPorts(true);
|
||||
FleetdRuntime durable = assemble(dir.resolve("durable"), durablePorts);
|
||||
try {
|
||||
// Control: this fails loudly if the assembly did not run or used an inert opener.
|
||||
assertEquals(1, durablePorts.replyOpenCalls.get(), "assembly must call replyInboxOpener once");
|
||||
assertEquals(1, durablePorts.mailboxOpenCalls.get(), "assembly must call leadMailboxOpener once");
|
||||
assertSame(durablePorts.replyInbox, durable.replyInbox(),
|
||||
"the durable reply report must describe the exact inbox the runtime owns");
|
||||
assertSame(durablePorts.leadMailbox, durable.leadMailbox(),
|
||||
"the coordination-on report must describe the exact mailbox the runtime owns");
|
||||
assertNotNull(durable.leadCoordLoop(), "a durable mailbox must start lead coordination");
|
||||
assertTrue(reportContains(reports, "reply inbox: AMQP broker (durable)"));
|
||||
assertTrue(reportContains(reports, "lead coordination: ON as coord-id " + COORD_ID));
|
||||
} finally {
|
||||
durable.close();
|
||||
}
|
||||
|
||||
reports.list.clear();
|
||||
RecordingPorts fallbackPorts = new RecordingPorts(false);
|
||||
FleetdRuntime fallback = assemble(dir.resolve("fallback"), fallbackPorts);
|
||||
try {
|
||||
assertEquals(1, fallbackPorts.replyOpenCalls.get(), "assembly must call the failing reply opener once");
|
||||
assertEquals(1, fallbackPorts.mailboxOpenCalls.get(), "assembly must call the failing mailbox opener once");
|
||||
assertTrue(fallback.replyInbox() instanceof InMemoryReplyInbox,
|
||||
"a failed reply opener must make the runtime own the in-memory fallback");
|
||||
assertNull(fallback.leadMailbox(), "a failed mailbox opener must leave coordination off");
|
||||
assertNull(fallback.leadCoordLoop(), "coordination must not start without a mailbox");
|
||||
assertTrue(reportContains(reports, "reply inbox: in-memory (soft-state)"));
|
||||
assertTrue(reportContains(reports, "lead-to-lead messaging is OFF"));
|
||||
} finally {
|
||||
fallback.close();
|
||||
}
|
||||
} finally {
|
||||
logger.detachAppender(reports);
|
||||
logger.setLevel(oldLevel);
|
||||
}
|
||||
}
|
||||
}
|
||||
+198
@@ -0,0 +1,198 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.health.FleetHealthMonitor;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 rank 7 — {@code FleetdAssembly.java:429} wires {@link FleetHealthMonitor}'s {@code
|
||||
* failTarget} callback with {@code Fleetd.healthFailTarget(messages)}. {@link
|
||||
* FleetdHealthFailTargetWiringTest} already pins that the FACTORY itself delegates to {@code
|
||||
* messages::abandon}, but it calls {@code Fleetd.healthFailTarget} directly — it never drives {@code
|
||||
* FleetdAssembly.assembleAndStart} and so cannot see whether the real call site at {@code :429}
|
||||
* still passes it the real, assembled {@link MessageService}. Swapping that argument for a no-op
|
||||
* {@code (a, b) -> {}} compiles clean and leaves the whole suite — including the factory-level test
|
||||
* — green: a dead member's waiting ticket then sits {@code PENDING} for the full 30-minute async
|
||||
* timeout instead of failing immediately.
|
||||
*
|
||||
* <p>This test assembles the real daemon with {@code health.enabled: true}, pulls the REAL {@code
|
||||
* failTarget} {@link BiConsumer} out of the REAL, assembled {@link FleetHealthMonitor} (via
|
||||
* reflection — the field is package-private to {@code dev.ltms.fleet.health}, and nothing public
|
||||
* exposes it; {@code StatusPollerResilienceTest} already uses the same technique in this suite), and
|
||||
* invokes it directly against the REAL {@link MessageService} {@link FleetdRuntime#messages()}
|
||||
* returns. A no-op lambda swapped in at the call site leaves the ticket {@code PENDING} forever,
|
||||
* which this test catches; the real one fails it.
|
||||
*/
|
||||
class FleetdAssemblyHealthFailTargetBehaviouralTest {
|
||||
|
||||
private static final String TARGET = "term_a";
|
||||
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no broker: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
health:
|
||||
enabled: true
|
||||
intervalSeconds: 30
|
||||
""");
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the real assembled FleetHealthMonitor's failTarget reaches the real "
|
||||
+ "MessageService.abandon, not a no-op")
|
||||
void assembledHealthFailTargetReachesRealMessages(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
assertNotNull(ports.shutdownHook, "FleetdAssembly must have registered a shutdown hook");
|
||||
|
||||
// Surefire runs the whole suite in one JVM fork, so the scheduler/loops this assembly starts
|
||||
// (SessionReaper, StatusPoller, the health monitor) must be torn down here, on the failure
|
||||
// path too — hence the try/finally, not just a statement at the end of the happy path.
|
||||
try {
|
||||
FleetHealthMonitor healthMonitor = runtime.healthMonitor();
|
||||
assertNotNull(healthMonitor, "health.enabled: true in this test's config, so "
|
||||
+ "FleetdAssembly.assembleAndStart must have built a real FleetHealthMonitor");
|
||||
|
||||
Field field = FleetHealthMonitor.class.getDeclaredField("failTarget");
|
||||
field.setAccessible(true);
|
||||
BiConsumer<String, String> failTarget = (BiConsumer<String, String>) field.get(healthMonitor);
|
||||
assertNotNull(failTarget, "FleetHealthMonitor's failTarget must never be null — the "
|
||||
+ "constructor itself requires it");
|
||||
|
||||
MessageService messages = runtime.messages();
|
||||
|
||||
// --- loud control: prove the assembled MessageService is actually wired up and a ticket is
|
||||
// genuinely PENDING before failTarget ever runs. If this fails, the test below would pass
|
||||
// vacuously on a MessageService that never got a ticket in the first place. TARGET has no
|
||||
// live agent behind it (no session was ever acquired), so nothing resolves this ticket on
|
||||
// its own — it stays PENDING until failTarget (or a timeout) ends it.
|
||||
String ticket = messages.sendAsync(TARGET, "long task");
|
||||
MessageService.TaskView before = messages.poll(ticket);
|
||||
assertEquals(MessageService.Phase.PENDING, before.phase(),
|
||||
"control: the async ticket must be PENDING before failTarget runs");
|
||||
|
||||
failTarget.accept(TARGET, "member unreachable (health monitor)");
|
||||
|
||||
MessageService.TaskView after = awaitTerminal(messages, ticket);
|
||||
assertEquals(MessageService.Phase.FAILED, after.phase(),
|
||||
"FleetdAssembly.java:429 must pass Fleetd.healthFailTarget(messages) built from the "
|
||||
+ "SAME assembled MessageService — a no-op BiConsumer at that call site leaves "
|
||||
+ "this ticket PENDING for the full 30-minute async timeout instead of failing it");
|
||||
assertTrue(after.detail() != null && after.detail().contains("member unreachable"),
|
||||
"the failure reason passed to failTarget.accept must reach MessageService.abandon and "
|
||||
+ "end up in the ticket's detail");
|
||||
} finally {
|
||||
// Proof the teardown actually ran, not just an assurance that a finally was added: the
|
||||
// captured shutdown hook's close order (FleetdAssemblyLifecycleTest) closes the herdr
|
||||
// client last, so ports.herdr.closed flips to true only if this hook really executed.
|
||||
ports.shutdownHook.run();
|
||||
assertTrue(ports.herdr.closed, "the captured shutdown hook must have run and closed herdr — "
|
||||
+ "proof this test's assembled background loops/scheduler were torn down");
|
||||
}
|
||||
}
|
||||
|
||||
private static MessageService.TaskView awaitTerminal(MessageService messages, String ticket)
|
||||
throws InterruptedException {
|
||||
long deadline = System.currentTimeMillis() + 5000;
|
||||
MessageService.TaskView view = messages.poll(ticket);
|
||||
while (view.phase() == MessageService.Phase.PENDING && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(10);
|
||||
view = messages.poll(ticket);
|
||||
}
|
||||
return view;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,232 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 rank 6 — {@code FleetdAssembly.java:447} wires {@code
|
||||
* sessions.onRelease(Fleetd.releaseCleanup(messages, replyInbox, primaryRegistry))}. {@link
|
||||
* FleetdReleaseCleanupWiringTest} already pins that the FACTORY {@code Fleetd.releaseCleanup}
|
||||
* itself reaches all three collaborators — but it calls the factory directly, never {@code
|
||||
* FleetdAssembly.assembleAndStart}, so it cannot see whether the real call site at {@code :447}
|
||||
* still registers it (as opposed to a no-op {@code detail -> { }}) or still passes it the REAL,
|
||||
* assembled {@code messages}/{@code replyInbox}/{@code primaryRegistry}. Swapping the registered
|
||||
* listener for a no-op at that call site compiles clean and leaves the whole suite — including the
|
||||
* factory-level test — green: EVERY teardown then leaks a stuck rendezvous waiter, an unreleased
|
||||
* reply-inbox consumer, and a stale lead binding, all three at once.
|
||||
*
|
||||
* <p>This test assembles the real daemon with {@code idleSleepGuard.enabled: false} — the ONLY
|
||||
* other {@code onRelease} registration in {@code FleetdAssembly} (see {@code
|
||||
* dev.ltms.fleet.power.IdleSleepGuard}'s own wiring at {@code FleetdAssembly.java:228}) — so the
|
||||
* real {@link SessionManager}'s release-listener list holds exactly the one listener this call site
|
||||
* registers. It pulls that REAL listener out via reflection (the list itself is private, like
|
||||
* {@code StatusPollerResilienceTest}'s use of the same technique elsewhere in this suite), invokes
|
||||
* it directly, and asserts all three collaborator effects against the REAL, assembled {@link
|
||||
* MessageService} ({@link FleetdRuntime#messages()}), the REAL {@link ReplyInbox} ({@link
|
||||
* FleetdRuntime#replyInbox()}), and the REAL {@link PrimaryRegistry} — reached through {@link
|
||||
* FleetdRuntime#pushLoop()}, the only other accessor that was handed the same {@code
|
||||
* primaryRegistry} instance ({@code FleetdAssembly.java:382}), since {@code FleetMcp} never exposes
|
||||
* it.
|
||||
*/
|
||||
class FleetdAssemblyReleaseCleanupBehaviouralTest {
|
||||
|
||||
private static final String TARGET = "term_a";
|
||||
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no broker: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
""");
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the real assembled release listener reaches messages.abandon, "
|
||||
+ "replyInbox.release, AND primaryRegistry.forgetDelegation — all three leaks at once")
|
||||
void assembledReleaseListenerReachesAllThreeCollaborators(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
assertNotNull(ports.shutdownHook, "FleetdAssembly must have registered a shutdown hook");
|
||||
|
||||
// Surefire runs the whole suite in one JVM fork, so the scheduler/loops this assembly starts
|
||||
// must be torn down here, on the failure path too — hence the try/finally, not just a
|
||||
// statement at the end of the happy path.
|
||||
try {
|
||||
// --- reach into SessionManager's private release-listener list. idleSleepGuard.enabled:
|
||||
// false above means FleetdAssembly.java:228 never registers, so this list must hold EXACTLY
|
||||
// the one listener :447 registers.
|
||||
Field listenersField = SessionManager.class.getDeclaredField("releaseListeners");
|
||||
listenersField.setAccessible(true);
|
||||
List<Consumer<SessionManager.ReleaseDetail>> releaseListeners =
|
||||
(List<Consumer<SessionManager.ReleaseDetail>>) listenersField.get(runtime.sessions());
|
||||
assertEquals(1, releaseListeners.size(), "control: with idleSleepGuard.enabled: false, "
|
||||
+ "FleetdAssembly.java:447 must be the ONLY onRelease registration — a different "
|
||||
+ "count means this test is no longer isolating the call site it claims to pin");
|
||||
Consumer<SessionManager.ReleaseDetail> releaseListener = releaseListeners.get(0);
|
||||
|
||||
MessageService messages = runtime.messages();
|
||||
ReplyInbox replyInbox = runtime.replyInbox();
|
||||
|
||||
// primaryRegistry is never exposed by FleetdRuntime directly — ReplyPushLoop is the other
|
||||
// collaborator FleetdAssembly.java:382 hands the SAME instance to, so reach it from there.
|
||||
Field primaryRegistryField = ReplyPushLoop.class.getDeclaredField("primaryRegistry");
|
||||
primaryRegistryField.setAccessible(true);
|
||||
PrimaryRegistry primaryRegistry = (PrimaryRegistry) primaryRegistryField.get(runtime.pushLoop());
|
||||
assertNotNull(primaryRegistry, "control: the assembled ReplyPushLoop must hold a real "
|
||||
+ "PrimaryRegistry instance");
|
||||
|
||||
// --- loud controls: set up the "before" state each collaborator's effect is measured
|
||||
// against, against the REAL assembled objects. If any of these three fails, the test below
|
||||
// would pass vacuously because the subject it claims to observe never existed in the first
|
||||
// place.
|
||||
String ticket = messages.sendAsync(TARGET, "long task");
|
||||
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket).phase(),
|
||||
"control: the async ticket must be PENDING before the release listener runs");
|
||||
|
||||
replyInbox.own(TARGET);
|
||||
replyInbox.publish(TARGET, "msg-1", "hello");
|
||||
assertEquals(1, replyInbox.peek(TARGET).size(),
|
||||
"control: the reply inbox must own TARGET and hold one message before the release "
|
||||
+ "listener runs");
|
||||
|
||||
primaryRegistry.recordDelegation(TARGET, "lead-1");
|
||||
assertEquals("lead-1", primaryRegistry.nudgeTargetFor(TARGET).orElse(null),
|
||||
"control: the delegation must be recorded before the release listener runs");
|
||||
|
||||
// --- the one call under test: invoke the REAL, assembled release listener directly, the
|
||||
// same way SessionManager.release(...) would on a real teardown.
|
||||
releaseListener.accept(new SessionManager.ReleaseDetail(TARGET, null, null, null, null));
|
||||
|
||||
MessageService.TaskView after = awaitTerminal(messages, ticket);
|
||||
assertEquals(MessageService.Phase.FAILED, after.phase(),
|
||||
"FleetdAssembly.java:447 must register a listener that calls messages.abandon(...) "
|
||||
+ "on the SAME assembled MessageService — an inert listener leaves this "
|
||||
+ "ticket PENDING for the full 30-minute async timeout");
|
||||
assertTrue(after.detail() != null && after.detail().contains("released"),
|
||||
"the abandon reason must say the worker session was released");
|
||||
|
||||
assertTrue(replyInbox.peek(TARGET).isEmpty(),
|
||||
"FleetdAssembly.java:447 must register a listener that calls replyInbox.release(...) "
|
||||
+ "— an inert listener leaves the inbox still owning TARGET with its message");
|
||||
|
||||
assertTrue(primaryRegistry.nudgeTargetFor(TARGET).isEmpty(),
|
||||
"FleetdAssembly.java:447 must register a listener that calls "
|
||||
+ "primaryRegistry.forgetDelegation(...) — an inert listener leaves the stale "
|
||||
+ "delegation in place");
|
||||
} finally {
|
||||
// Proof the teardown actually ran, not just an assurance that a finally was added: the
|
||||
// captured shutdown hook's close order (FleetdAssemblyLifecycleTest) closes the herdr
|
||||
// client last, so ports.herdr.closed flips to true only if this hook really executed.
|
||||
ports.shutdownHook.run();
|
||||
assertTrue(ports.herdr.closed, "the captured shutdown hook must have run and closed herdr — "
|
||||
+ "proof this test's assembled background loops/scheduler were torn down");
|
||||
}
|
||||
}
|
||||
|
||||
private static MessageService.TaskView awaitTerminal(MessageService messages, String ticket)
|
||||
throws InterruptedException {
|
||||
long deadline = System.currentTimeMillis() + 5000;
|
||||
MessageService.TaskView view = messages.poll(ticket);
|
||||
while (view.phase() == MessageService.Phase.PENDING && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(10);
|
||||
view = messages.poll(ticket);
|
||||
}
|
||||
return view;
|
||||
}
|
||||
}
|
||||
+241
@@ -0,0 +1,241 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.lead.LeadContextGauge;
|
||||
import dev.ltms.fleet.msg.LeadHeartbeatLoop;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.lang.reflect.Method;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #630, and fleetd #612's ranks for this call site. {@code FleetdAssembly.java:402}
|
||||
* computes {@code requireOperatorConfirm} from the effective {@code leadRollover.requireOperatorConfirm}
|
||||
* config, and {@code :409} threads it as the 14th argument into the full {@link LeadHeartbeatLoop}
|
||||
* constructor. Measured on 26f1986: dropping that one argument so the 13-argument overload is
|
||||
* selected instead (it delegates with {@code true} hardcoded — see that overload's own javadoc,
|
||||
* fleetd #621) compiles with 0 errors and leaves all 1883 tests green, both with and without the
|
||||
* argument. In production this means the daemon keeps starting and keeps nudging, but the
|
||||
* context-high notice silently goes back to telling EVERY lead to ask the operator before a
|
||||
* context roll — on a host that set {@code requireOperatorConfirm: false} specifically so it would
|
||||
* not have to. That is the operator's own fix silently reverting, with a fully green suite.
|
||||
*
|
||||
* <p>{@code LeadHeartbeatLoopTest} already proves {@link LeadHeartbeatLoop}'s package-private
|
||||
* {@code contextNotice(boolean, LeadContextGauge.Reading, boolean, boolean)} branches correctly on
|
||||
* its own {@code requireOperatorConfirm} argument — that the METHOD works. It says nothing about
|
||||
* which value {@code FleetdAssembly} actually passes into the constructed loop, so it is not
|
||||
* reused here as coverage for the call site.
|
||||
*
|
||||
* <p>This test assembles the real daemon TWICE — once with {@code leadRollover.requireOperatorConfirm:
|
||||
* false}, once with {@code true} — pulls the REAL {@code requireOperatorConfirm} field out of the
|
||||
* REAL, assembled {@link LeadHeartbeatLoop} each time (reflection: the field, and {@code
|
||||
* contextNotice} itself, are package-private to {@code dev.ltms.fleet.msg}, and nothing public
|
||||
* exposes either — the same technique {@code StatusPollerResilienceTest} already uses in this
|
||||
* suite), and calls the REAL {@code contextNotice} method with that field's value to produce the
|
||||
* actual notice text the assembled loop would append to a nudge. Both directions are asserted: a
|
||||
* one-directional test here would pass on a constant.
|
||||
*/
|
||||
class FleetdAssemblyRequireOperatorConfirmBehaviouralTest {
|
||||
|
||||
private static final class RecordingResourcePorts implements ResourcePorts {
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no broker: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException("no coordinator: block is configured");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return System::nanoTime;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir, boolean requireOperatorConfirm) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
leadHeartbeat:
|
||||
idleAfterSeconds: 600
|
||||
backoffMs: 15000
|
||||
quietNudgeCap: 5
|
||||
leadRollover:
|
||||
handoverPath: handover.md
|
||||
requireOperatorConfirm: %s
|
||||
""".formatted(requireOperatorConfirm));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
/** Carries both the assembled loop under test AND its {@link RecordingResourcePorts}, so the
|
||||
* caller can tear the assembly down (this test assembles the real daemon TWICE — see the class
|
||||
* javadoc — and each assembly needs its own teardown, not just the last one). */
|
||||
private record Assembled(LeadHeartbeatLoop heartbeat, RecordingResourcePorts ports) {
|
||||
}
|
||||
|
||||
private static Assembled assembleHeartbeat(Path dir, boolean requireOperatorConfirm) throws Exception {
|
||||
FleetConfig cfg = writeConfig(dir, requireOperatorConfirm);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
RecordingResourcePorts ports = new RecordingResourcePorts();
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
assertNotNull(ports.shutdownHook, "FleetdAssembly must have registered a shutdown hook");
|
||||
|
||||
LeadHeartbeatLoop heartbeat = runtime.heartbeat();
|
||||
assertNotNull(heartbeat, "control: leadHeartbeat: is configured, so FleetdAssembly.assembleAndStart "
|
||||
+ "must have built a real LeadHeartbeatLoop");
|
||||
return new Assembled(heartbeat, ports);
|
||||
}
|
||||
|
||||
/** Pulls the REAL {@code requireOperatorConfirm} field off the REAL, assembled loop. */
|
||||
private static boolean assembledRequireOperatorConfirm(LeadHeartbeatLoop heartbeat) throws Exception {
|
||||
Field field = LeadHeartbeatLoop.class.getDeclaredField("requireOperatorConfirm");
|
||||
field.setAccessible(true);
|
||||
return field.getBoolean(heartbeat);
|
||||
}
|
||||
|
||||
/** Calls the REAL, package-private {@code contextNotice(boolean, Reading, boolean, boolean)} via reflection. */
|
||||
private static String contextNotice(boolean enabled, LeadContextGauge.Reading reading, boolean alreadyNotified,
|
||||
boolean requireOperatorConfirm) throws Exception {
|
||||
Method method = LeadHeartbeatLoop.class.getDeclaredMethod("contextNotice", boolean.class,
|
||||
LeadContextGauge.Reading.class, boolean.class, boolean.class);
|
||||
method.setAccessible(true);
|
||||
return (String) method.invoke(null, enabled, reading, alreadyNotified, requireOperatorConfirm);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the real assembled LeadHeartbeatLoop's context-high notice tracks "
|
||||
+ "leadRollover.requireOperatorConfirm — BOTH directions")
|
||||
void assembledRequireOperatorConfirmControlsNoticeWording(@TempDir Path dir) throws Exception {
|
||||
LeadContextGauge.Reading highReading = new LeadContextGauge.Reading(LeadContextGauge.State.HIGH,
|
||||
250_000L, 2);
|
||||
|
||||
String noticeFalse;
|
||||
String noticeTrue;
|
||||
|
||||
// --- direction 1: requireOperatorConfirm: false -----------------------------------------
|
||||
Path falseDir = dir.resolve("false");
|
||||
Files.createDirectories(falseDir);
|
||||
Assembled assembledFalse = assembleHeartbeat(falseDir, false);
|
||||
// Surefire runs the whole suite in one JVM fork, so each assembly's scheduler/loops must be
|
||||
// torn down here, on the failure path too — hence try/finally per assembly (this test
|
||||
// assembles TWICE, so both need their own teardown, not just the last one).
|
||||
try {
|
||||
boolean fieldFalse = assembledRequireOperatorConfirm(assembledFalse.heartbeat());
|
||||
assertFalse(fieldFalse, "FleetdAssembly.java:402/:409 must thread leadRollover."
|
||||
+ "requireOperatorConfirm: false into the assembled LeadHeartbeatLoop's own field — "
|
||||
+ "dropping the 14th constructor argument selects the 13-argument overload, which "
|
||||
+ "hardcodes true regardless of config (fleetd #621), and this would read true instead");
|
||||
|
||||
noticeFalse = contextNotice(true, highReading, false, fieldFalse);
|
||||
assertTrue(noticeFalse.contains("Decide for yourself when to confirm"),
|
||||
"with requireOperatorConfirm: false, the assembled loop's own notice must tell the "
|
||||
+ "lead it can decide for itself — got: " + noticeFalse);
|
||||
assertFalse(noticeFalse.contains("ask the operator") || noticeFalse.contains("Only the operator"),
|
||||
"with requireOperatorConfirm: false, the assembled loop's own notice must NOT ask the "
|
||||
+ "operator — got: " + noticeFalse);
|
||||
} finally {
|
||||
// Proof the teardown actually ran, not just an assurance that a finally was added: the
|
||||
// captured shutdown hook's close order (FleetdAssemblyLifecycleTest) closes the herdr
|
||||
// client last, so ports.herdr.closed flips to true only if this hook really executed.
|
||||
assembledFalse.ports().shutdownHook.run();
|
||||
assertTrue(assembledFalse.ports().herdr.closed, "the captured shutdown hook must have run "
|
||||
+ "and closed herdr — proof this assembly's background loops/scheduler were torn down");
|
||||
}
|
||||
|
||||
// --- direction 2: requireOperatorConfirm: true -------------------------------------------
|
||||
Path trueDir = dir.resolve("true");
|
||||
Files.createDirectories(trueDir);
|
||||
Assembled assembledTrue = assembleHeartbeat(trueDir, true);
|
||||
try {
|
||||
boolean fieldTrue = assembledRequireOperatorConfirm(assembledTrue.heartbeat());
|
||||
assertTrue(fieldTrue, "FleetdAssembly.java:402/:409 must thread leadRollover."
|
||||
+ "requireOperatorConfirm: true into the assembled LeadHeartbeatLoop's own field");
|
||||
|
||||
noticeTrue = contextNotice(true, highReading, false, fieldTrue);
|
||||
assertTrue(noticeTrue.contains("ask the operator") && noticeTrue.contains("Only the operator can approve the roll"),
|
||||
"with requireOperatorConfirm: true, the assembled loop's own notice must ask the "
|
||||
+ "operator — got: " + noticeTrue);
|
||||
assertFalse(noticeTrue.contains("Decide for yourself when to confirm"),
|
||||
"with requireOperatorConfirm: true, the assembled loop's own notice must NOT tell "
|
||||
+ "the lead it can decide for itself — got: " + noticeTrue);
|
||||
} finally {
|
||||
assembledTrue.ports().shutdownHook.run();
|
||||
assertTrue(assembledTrue.ports().herdr.closed, "the captured shutdown hook must have run "
|
||||
+ "and closed herdr — proof this assembly's background loops/scheduler were torn down");
|
||||
}
|
||||
|
||||
// --- the two directions must actually differ: a constant return would pass both assertion
|
||||
// blocks above vacuously if they happened to share wording, so compare them directly too.
|
||||
assertTrue(!noticeFalse.equals(noticeTrue),
|
||||
"the two directions must produce genuinely different notice text — got the same "
|
||||
+ "text for both: " + noticeFalse);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,218 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.inject.CompletionResolver;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 step 4, ranks 1 and 2 (publish side) — {@link FleetdAssembly} lines
|
||||
* {@code liveExhaustedPatterns}/{@code exhaustedPatterns} (CB-578 stage A, the ticket's own "worst
|
||||
* consequence in the whole sweep": a genuine usage-limit refusal handed back to a waiting caller
|
||||
* AS REAL COMPLETED WORK) and {@code Fleetd.publishExhaustionSink(...)} (CB-578 stage B: the
|
||||
* credential that hit the limit is never quarantined). None of these three lines is driven by an
|
||||
* existing test through the real assembly: {@code FleetdExhaustedPatternLookupWiringTest} and
|
||||
* {@code FleetdLiveExhaustedPatternsWiringTest} (fleetd #589) call {@code Fleetd.liveExhaustedPatterns}
|
||||
* / {@code Fleetd.exhaustedPatternLookup} directly as factories, never through {@link
|
||||
* FleetdAssembly#assembleAndStart} — they prove the FACTORY classifies correctly, never that THIS
|
||||
* call site is the one that actually got wired into the running {@link CompletionResolver}. {@link
|
||||
* FleetdBackendQuarantineAssemblyTest} drives {@code BackendQuarantine.withEscalation(...)}
|
||||
* directly, a different call site from {@code publishExhaustionSink} here.
|
||||
*
|
||||
* <p>This test drives the REAL assembled {@link CompletionResolver} ({@link
|
||||
* FleetdRuntime#completion()}) with a profile carrying a configured {@code exhaustedPattern},
|
||||
* through a pane scrape that matches it, and asserts both halves of the production consequence:
|
||||
* (1) the resolution is {@link Rendezvous.Kind#BACKEND_EXHAUSTED}, never a plain completion handed
|
||||
* back as real work, and (2) the profile's credential is actually quarantined afterward, through
|
||||
* the REAL {@link BackendQuarantine} the same assembly built ({@link
|
||||
* FleetdRuntime#mcp()}{@code .quarantineSource().quarantine()}) — never a copy.
|
||||
*
|
||||
* <p>Same {@code ControllableResourcePorts} shape as {@code FleetdCompletionResolverAssemblyTest}:
|
||||
* a fake, advanceable {@code nanoClock} so {@code CompletionResolver.MIN_TURN_NANOS} clears without
|
||||
* a real sleep, and {@link FakeHerdr#readText} to drive the pane scrape.
|
||||
*/
|
||||
class FleetdExhaustedPatternAssemblyTest {
|
||||
|
||||
private static final class ControllableResourcePorts implements ResourcePorts {
|
||||
|
||||
final FakeHerdr herdr;
|
||||
final AtomicLong nowNanos = new AtomicLong(1_000_000_000L); // arbitrary non-zero start
|
||||
Runnable shutdownHook;
|
||||
|
||||
ControllableResourcePorts(FakeHerdr herdr) {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
void advanceSeconds(long seconds) {
|
||||
nowNanos.addAndGet(TimeUnit.SECONDS.toNanos(seconds));
|
||||
}
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> {
|
||||
throw new UnsupportedOperationException("replyInboxOpener must not be called — no broker: block");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException("leadMailboxOpener must not be called — no coordinator: block");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Deliberately never bind a real port.
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir, int cooldownSeconds) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
quarantineCooldownSeconds: %d
|
||||
profiles:
|
||||
exhaustprofile:
|
||||
baseUrl: http://exhausthost.local:8000
|
||||
model: sonnet
|
||||
exhaustedPattern: "usage limit reached"
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- exhausthost.local
|
||||
""".formatted(cooldownSeconds));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #589's own description of this gap ({@code Fleetd#exhaustedPatternLookup}'s javadoc):
|
||||
* "the worst consequence in the whole #589 sweep" — a genuine usage-limit refusal stops being
|
||||
* classified as {@code BACKEND_EXHAUSTED} and is handed back to a waiting {@code fleet_send} as
|
||||
* if it were real completed work. Pins {@code FleetdAssembly}'s {@code liveExhaustedPatterns}
|
||||
* AND {@code exhaustedPatterns} lines (rank 1) together with {@code publishExhaustionSink}
|
||||
* (rank 2, the non-OpenCode half) in one flow: classify, then quarantine.
|
||||
*/
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] a scrape matching the profile's exhaustedPattern resolves "
|
||||
+ "BACKEND_EXHAUSTED (never a plain completion) and quarantines the credential")
|
||||
void assembledResolverClassifiesExhaustionAndQuarantinesTheCredential(@TempDir Path dir) throws Exception {
|
||||
int cooldownSeconds = 120;
|
||||
FleetConfig cfg = writeConfig(dir, cooldownSeconds);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
ControllableResourcePorts ports = new ControllableResourcePorts(new FakeHerdr());
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
try {
|
||||
MemberSession session = runtime.sessions().acquire("exhaustprofile", null, dir.toString(), null);
|
||||
String target = session.terminalId();
|
||||
|
||||
CompletionResolver completion = runtime.completion();
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = new CompletableFuture<>();
|
||||
|
||||
ports.herdr.readText("idle, nothing yet");
|
||||
completion.onDelivered(target, new TurnToken(target, waiter, null));
|
||||
// The matched text must START the pane line (CompletionResolver.startsWithExhaustion) —
|
||||
// no preceding sentence — for the quarantine side-effect to fire, same as production.
|
||||
ports.herdr.readText("usage limit reached: try again in a few hours");
|
||||
ports.advanceSeconds(3); // clear CompletionResolver.MIN_TURN_NANOS (2s), no real sleep
|
||||
completion.resolveBeforePostAction(target);
|
||||
|
||||
// CONTROL: the waiter must have resolved synchronously at all — if the assembled
|
||||
// CompletionResolver were never actually driven (e.g. a wiring break upstream silently
|
||||
// left the resolver unreachable), this fails loudly before the real assertions below
|
||||
// ever run, rather than passing on an untouched waiter.
|
||||
Rendezvous.Resolution resolution = waiter.getNow(null);
|
||||
assertTrue(resolution != null, "CONTROL: the waiter must have resolved synchronously — "
|
||||
+ "if this is null, the assembled resolver was never actually exercised");
|
||||
|
||||
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, resolution.kind(),
|
||||
"a scrape matching the profile's configured exhaustedPattern must classify as "
|
||||
+ "BACKEND_EXHAUSTED, not a plain completion handed back as real work — "
|
||||
+ "replacing FleetdAssembly's liveExhaustedPatterns/exhaustedPatterns "
|
||||
+ "lines with their inert forms (Map.of() / target -> null) must fail "
|
||||
+ "this assertion; got: " + resolution);
|
||||
assertTrue(resolution.text().contains("usage limit reached"), resolution.text());
|
||||
|
||||
BackendQuarantine quarantine = runtime.mcp().quarantineSource().quarantine();
|
||||
assertTrue(quarantine.isQuarantined("exhaustprofile"),
|
||||
"the real publishExhaustionSink-built sink must have quarantined the profile's "
|
||||
+ "credential (effectiveCredentialId() == the profile name here, no "
|
||||
+ "credentialId configured) — replacing FleetdAssembly's "
|
||||
+ "publishExhaustionSink call site with a hardcoded ExhaustionSink.none() "
|
||||
+ "must fail this assertion, since nothing would ever call "
|
||||
+ "quarantine.quarantine(...)");
|
||||
OptionalLong remaining = quarantine.remainingSeconds("exhaustprofile");
|
||||
assertTrue(remaining.isPresent() && remaining.getAsLong() > 0
|
||||
&& remaining.getAsLong() <= cooldownSeconds,
|
||||
"a fresh quarantine must block for at most the configured base cooldown: " + remaining);
|
||||
} finally {
|
||||
if (ports.shutdownHook != null) ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
}
|
||||
+197
@@ -0,0 +1,197 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.member.HerdrPeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.lang.reflect.Field;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #612 step 4, rank 2 (OpenCode half) — {@link FleetdAssembly}'s {@code
|
||||
* forwardingExhaustionSink} line ({@code Fleetd.forwardingExhaustionSink(exhaustionSinkRef)}),
|
||||
* handed to {@link dev.ltms.fleet.member.OpenCodeLauncher} so its fleetd #175 model-mismatch check
|
||||
* can quarantine a credential before {@code sessions} exists to build the real sink (the
|
||||
* construction-order cycle documented at that call site). The ticket calls this independent from
|
||||
* {@code publishExhaustionSink} (pinned by {@link FleetdExhaustedPatternAssemblyTest}): a credential
|
||||
* that hits a usage limit through THIS path is never quarantined if {@code forwardingExhaustionSink}
|
||||
* is swapped for a hardcoded {@link ExhaustionSink#none()} at that call site — the OpenCode
|
||||
* launcher's own quarantine check keeps compiling and keeps "running", but it permanently talks to
|
||||
* a sink that does nothing, independent of whatever {@code publishExhaustionSink} does later.
|
||||
*
|
||||
* <p>{@code FleetdExhaustionSinkForwardingWiringTest} (fleetd #589) already proves {@code
|
||||
* Fleetd.forwardingExhaustionSink(ref)} forwards to whatever {@code ref} holds — as a bare factory
|
||||
* call, never through {@link FleetdAssembly#assembleAndStart}. It proves nothing about whether
|
||||
* THIS call site is the one FleetdAssembly actually wires into the real {@code OpenCodeLauncher}
|
||||
* it builds, which is exactly the #602/#606-shaped gap this ticket exists to close.
|
||||
*
|
||||
* <p>No accessor on {@link FleetdRuntime} reaches the adapter instances (by design — see that
|
||||
* class's own javadoc: only the final collaborators it owns directly are exposed), so this test
|
||||
* reaches the REAL, assembled {@code OpenCodeLauncher}'s {@code exhaustionSink} field the same way
|
||||
* {@code SessionManager}/{@code CompositePeerLauncher} wire it internally: a short, targeted
|
||||
* reflective walk ({@code SessionManager.launcher} → {@code CompositePeerLauncher.byProfile} →
|
||||
* {@code OpenCodeLauncher.exhaustionSink}) onto the exact object the assembly built — never a copy,
|
||||
* and never a read of the source text. Reflection is used the same way elsewhere in this suite
|
||||
* (e.g. {@code StatusPollerWatchdogTest}) to reach a private collaborator a production constructor
|
||||
* intentionally does not expose a public accessor for.
|
||||
*/
|
||||
class FleetdOpenCodeExhaustionForwardingAssemblyTest {
|
||||
|
||||
private static final class ControllableResourcePorts implements ResourcePorts {
|
||||
|
||||
final FakeHerdr herdr = new FakeHerdr();
|
||||
final AtomicLong nowNanos = new AtomicLong(1_000_000_000L);
|
||||
Runnable shutdownHook;
|
||||
|
||||
@Override
|
||||
public Map<String, String> environment() {
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public HerdrClient connectHerdr(Path socketPath) {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.AmqpOpener replyInboxOpener() {
|
||||
return (uri, prefetch) -> {
|
||||
throw new UnsupportedOperationException("replyInboxOpener must not be called — no broker: block");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Fleetd.LeadMailboxOpener leadMailboxOpener() {
|
||||
return (uri, selfCoordId, prefetch) -> {
|
||||
throw new UnsupportedOperationException("leadMailboxOpener must not be called — no coordinator: block");
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public Runnable herdrPollWait() {
|
||||
// Never invoked: this test's FakeHerdr answers immediately, so awaitHerdr never polls.
|
||||
return () -> {
|
||||
throw new UnsupportedOperationException("herdrPollWait must not be called — herdr is healthy");
|
||||
};
|
||||
}
|
||||
@Override
|
||||
public LongSupplier nanoClock() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public LongSupplier wallClockNanos() {
|
||||
return nowNanos::get;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ScheduledExecutorService newScheduler(String purpose) {
|
||||
return Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void addShutdownHook(Runnable hook) {
|
||||
this.shutdownHook = hook;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void startHttp(Javalin app, String host, int port) {
|
||||
// Deliberately never bind a real port.
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig writeConfig(Path dir, int cooldownSeconds) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
idleSleepGuard:
|
||||
enabled: false
|
||||
quarantineCooldownSeconds: %d
|
||||
profiles:
|
||||
gemini:
|
||||
kind: opencode
|
||||
model: google/gemini-2.5-pro
|
||||
""".formatted(cooldownSeconds));
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
/** Reach a declared field by name on {@code target}'s runtime class, bypassing the access check. */
|
||||
private static Object readField(Object target, Class<?> declaringClass, String fieldName) throws Exception {
|
||||
Field field = declaringClass.getDeclaredField(fieldName);
|
||||
field.setAccessible(true);
|
||||
return field.get(target);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[BEHAVIOURAL] the real assembled OpenCodeLauncher's exhaustionSink field forwards "
|
||||
+ "an onExhausted call into the real daemon's BackendQuarantine")
|
||||
void assembledOpenCodeLauncherExhaustionSinkQuarantinesTheCredential(@TempDir Path dir) throws Exception {
|
||||
int cooldownSeconds = 90;
|
||||
FleetConfig cfg = writeConfig(dir, cooldownSeconds);
|
||||
ConfigRef config = new ConfigRef(dir.resolve("fleetd.yaml"), cfg);
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
ControllableResourcePorts ports = new ControllableResourcePorts();
|
||||
|
||||
FleetdRuntime runtime = FleetdAssembly.assembleAndStart(new AssemblyInputs(cfg, config, guard), ports);
|
||||
try {
|
||||
PeerLauncher launcherField = (PeerLauncher) readField(runtime.sessions(),
|
||||
runtime.sessions().getClass(), "launcher");
|
||||
// CONTROL: the composite launcher must actually be the real production type with a
|
||||
// "gemini" -> OpenCodeLauncher entry — if this fails, nothing below exercised the real
|
||||
// assembly at all, rather than silently passing on an empty/wrong object.
|
||||
assertTrue(launcherField instanceof CompositePeerLauncher,
|
||||
"CONTROL: SessionManager.launcher must be the real CompositePeerLauncher the "
|
||||
+ "assembly built, got: " + launcherField);
|
||||
@SuppressWarnings("unchecked")
|
||||
Map<String, HerdrPeerLauncher> byProfile = (Map<String, HerdrPeerLauncher>)
|
||||
readField(launcherField, CompositePeerLauncher.class, "byProfile");
|
||||
HerdrPeerLauncher adapter = byProfile.get("gemini");
|
||||
assertTrue(adapter != null && adapter.getClass().getSimpleName().equals("OpenCodeLauncher"),
|
||||
"CONTROL: the 'gemini' profile must resolve to a real OpenCodeLauncher adapter, "
|
||||
+ "got: " + adapter);
|
||||
|
||||
ExhaustionSink sink = (ExhaustionSink) readField(adapter, adapter.getClass(), "exhaustionSink");
|
||||
assertTrue(sink != null, "CONTROL: OpenCodeLauncher.exhaustionSink must never be null");
|
||||
|
||||
// The exact call OpenCodeLauncher.SessionAwareHandle#checkModelMatch makes on a real
|
||||
// model mismatch (fleetd #175): target, reason, and its own already-known profile name.
|
||||
sink.onExhausted("term_gemini_1", "opencode model mismatch (test)", "gemini");
|
||||
|
||||
BackendQuarantine quarantine = runtime.mcp().quarantineSource().quarantine();
|
||||
assertTrue(quarantine.isQuarantined("gemini"),
|
||||
"the real forwardingExhaustionSink-wired field must have delegated into the "
|
||||
+ "published production sink, which quarantines the profile's credential "
|
||||
+ "('gemini' here — no credentialId configured) — replacing "
|
||||
+ "FleetdAssembly's forwardingExhaustionSink call site with a hardcoded "
|
||||
+ "ExhaustionSink.none() must fail this assertion, since the field read "
|
||||
+ "above would then BE the inert no-op and nothing would ever reach "
|
||||
+ "quarantine.quarantine(...)");
|
||||
assertEquals(cooldownSeconds, quarantine.remainingSeconds("gemini").orElseThrow(
|
||||
() -> new AssertionError("credential must report a remaining cooldown")));
|
||||
} finally {
|
||||
if (ports.shutdownHook != null) ports.shutdownHook.run();
|
||||
}
|
||||
}
|
||||
}
|
||||
Executable
+732
@@ -0,0 +1,732 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# The one auditable way to edit the live fleetd.yaml.
|
||||
#
|
||||
# fleetd ticket #635 — why this exists at all: fleetd.yaml is gitignored and holds the live
|
||||
# fleet's settings. A bad raw edit reaches a daemon that is already serving, so a direct `Edit`
|
||||
# on it is refused by policy. This script is the allow-listed alternative, and it is not just
|
||||
# convenience — it is the thing a raw file write can never give you: a backup, a parse check
|
||||
# BEFORE the file is installed, and the daemon's own reload verdict read back afterwards. An
|
||||
# edit to a live config is not finished when the bytes are written. It is finished when the
|
||||
# daemon has said what it did with them.
|
||||
#
|
||||
# What the daemon says, and how this script finds it — measured against `ConfigRef.java` on
|
||||
# fleetd commit 158a2a8, 2026-10-01:
|
||||
#
|
||||
# 1. `ConfigRef` re-reads fleetd.yaml only when the WATCHER sees the mtime move (every 10s by
|
||||
# default — read the real interval out of the daemon's own startup line, "config watch: ...
|
||||
# re-read when it changes (every Ns)"). So a verdict never appears before the next tick.
|
||||
# 2. `ConfigRef.Outcome.summary()` logs exactly one of five strings (ConfigRef.java:371-391):
|
||||
# config reload refused — <error message>
|
||||
# config reload refused — these keys cannot change under a running daemon: <keys>. ...
|
||||
# config reloaded
|
||||
# config reloaded; these changes need a restart to take effect: <keys>
|
||||
# config reloaded; partially live — <key: detail | ...>
|
||||
# A parse/validation failure logs a DIFFERENT line instead, before any summary ever runs
|
||||
# (ConfigRef.java:425): "config reload from <path> refused, keeping the running config:
|
||||
# <message>". This script recognises both shapes of refusal.
|
||||
# 3. The em dash in those strings is a real multi-byte character — match the stable prefix
|
||||
# "config reload refused" (or "...refused, keeping the running config" for the parse-failure
|
||||
# shape), never the dash itself.
|
||||
# 4. A cold-key change (bind/herdrSocket/memberHerdrSocket/broker/auth) throws away the WHOLE
|
||||
# reload — the running config keeps every old value, not only the cold one.
|
||||
# 5. A deferred/split change IS applied (current.set(fresh) runs) — "needs a restart" is a
|
||||
# SUCCESS with a follow-up, never a failure.
|
||||
#
|
||||
# Four outcomes, and they stay four (see the exit code table below). The one most likely to be
|
||||
# gotten wrong is "cannot tell" (exit 5): the daemon may be down, or the watcher may be stalled,
|
||||
# and folding that into either "refused" or "applied" is worse than never checking at all,
|
||||
# because a caller then acts on a verdict nobody actually read. So exit 5 never restores — a
|
||||
# visible, recoverable edit beats an invisible revert of a GOOD edit.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/config-edit.sh --check
|
||||
# scripts/config-edit.sh --set <yq-path>=<value> [--set ...]
|
||||
# scripts/config-edit.sh --from <candidate.yaml>
|
||||
# scripts/config-edit.sh --dry-run --set <yq-path>=<value>
|
||||
# scripts/config-edit.sh --restore
|
||||
#
|
||||
# `--set .a.b=` (an empty value — a forgotten typo) is REFUSED, not accepted as "clear the
|
||||
# field": a null value falls back to its default rather than erroring, which is silent, not
|
||||
# safe. To clear a key on purpose, write a literal null: `--set .a.b=null`. Every other value
|
||||
# is always written as a YAML string (via yq's strenv(), never spliced into the expression), so
|
||||
# there is currently no --set spelling for the literal three-character STRING "null" itself — use
|
||||
# --from for that rare case.
|
||||
#
|
||||
# Overrides (so this is drivable with no daemon — see scripts/test-config-edit.sh):
|
||||
# --config <path> default: fleetd/fleetd.yaml
|
||||
# --log <path> default: fleetd/fleetd.out
|
||||
# --wait-seconds <n> default: 4x the watch interval this script reads out of --log (10 -> 40)
|
||||
#
|
||||
# Exit codes (the --check/--restore/usage-error paths are reported separately, see below):
|
||||
# 0 applied; verdict read; clean
|
||||
# 3 applied; verdict read; needs a restart (deferred or split keys named)
|
||||
# 4 REFUSED by the daemon; backup restored (and the restore's own verdict reported if seen)
|
||||
# 5 CANNOT TELL — no verdict line inside the wait window. Nothing is restored.
|
||||
#
|
||||
# Never prints a secret. fleetd.yaml keeps credentials out by indirection (broker.uriEnv,
|
||||
# gitTokenEnv) but this script does not rely on that staying true: every diff it prints is piped
|
||||
# through `redact`, which (a) blanks the userinfo of any `scheme://user:pass@host` and (b) masks
|
||||
# the whole value on any line whose key looks like a credential. See `redact` below.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
SELF="$REPO/scripts/config-edit.sh"
|
||||
|
||||
CONFIG="$REPO/fleetd/fleetd.yaml"
|
||||
LOG="$REPO/fleetd/fleetd.out"
|
||||
WAIT_SECONDS_OVERRIDE=""
|
||||
FALLBACK_PORT=8765
|
||||
|
||||
MODE=""
|
||||
DRY_RUN=0
|
||||
SETS=()
|
||||
FROM_FILE=""
|
||||
|
||||
# fleetd #635 follow-up — a signal (or any early exit while a candidate is still uninstalled) must
|
||||
# not leave a `.config-edit.XXXXXX` file sitting beside the live config forever. CAND is global
|
||||
# (never a function-local) on purpose: this ONE trap, set once, covers every path that ever
|
||||
# creates a candidate — run_edit and dry_run_diff both assign it, and clear it back to "" once the
|
||||
# file is consumed (installed, or explicitly removed), so a later, unrelated exit never retries a
|
||||
# path that already served its purpose.
|
||||
CAND=""
|
||||
cleanup_candidate() { [ -n "$CAND" ] && rm -f "$CAND" 2>/dev/null; return 0; }
|
||||
trap cleanup_candidate EXIT INT TERM
|
||||
|
||||
say() { printf '\n\033[1m== %s\033[0m\n' "$*"; }
|
||||
ok() { printf ' ok %s\n' "$*"; }
|
||||
warn() { printf ' WARN %s\n' "$*"; }
|
||||
die() { printf '\n FAIL %s\n\n' "$*" >&2; exit 1; }
|
||||
|
||||
set_mode() {
|
||||
local new="$1"
|
||||
if [ -n "$MODE" ] && [ "$MODE" != "$new" ]; then
|
||||
die "cannot combine --$MODE and --$new in one invocation"
|
||||
fi
|
||||
MODE="$new"
|
||||
}
|
||||
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--check) set_mode check; shift ;;
|
||||
--restore) set_mode restore; shift ;;
|
||||
--set)
|
||||
[ $# -ge 2 ] || die "--set requires <yq-path>=<value>"
|
||||
set_mode set
|
||||
SETS+=("$2")
|
||||
shift 2 ;;
|
||||
--from)
|
||||
[ $# -ge 2 ] || die "--from requires a candidate file path"
|
||||
set_mode from
|
||||
FROM_FILE="$2"
|
||||
shift 2 ;;
|
||||
--dry-run) DRY_RUN=1; shift ;;
|
||||
--config)
|
||||
[ $# -ge 2 ] || die "--config requires a path"
|
||||
CONFIG="$2"; shift 2 ;;
|
||||
--log)
|
||||
[ $# -ge 2 ] || die "--log requires a path"
|
||||
LOG="$2"; shift 2 ;;
|
||||
--wait-seconds)
|
||||
[ $# -ge 2 ] || die "--wait-seconds requires a number of seconds"
|
||||
WAIT_SECONDS_OVERRIDE="$2"; shift 2 ;;
|
||||
-h|--help) sed -n '3,70p' "$SELF"; exit 0 ;;
|
||||
*) echo "unknown option: $1 (try --help)" >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
[ -n "$MODE" ] || die "no action given — use --check, --set, --from, or --restore (see --help)"
|
||||
|
||||
# ------------------------------------------------------------------------------------- redaction
|
||||
#
|
||||
# Two independent passes, applied to every diff this script ever prints:
|
||||
# 1. `scheme://user:pass@host` -> `scheme://<redacted>@host`, globally (the `g` flag matters —
|
||||
# a line can carry more than one URI).
|
||||
# 2. Any line whose key looks like TOKEN|SECRET|PASSWORD|PASSWD|PASSPHRASE|CREDENTIAL|URI|_KEY,
|
||||
# matched case-insensitively against the key text (uriEnv, gitTokenEnv, ... are camelCase,
|
||||
# not SCREAMING_CASE) has its whole value blanked, diff marker and indentation kept so the
|
||||
# shape of the change is still visible. Deliberately conservative: a false-positive
|
||||
# redaction on an unrelated line costs nothing, an unredacted secret is a security defect
|
||||
# (acceptance criterion 7).
|
||||
#
|
||||
# fleetd #635 follow-up (ticket comment 17670, defect 7) — a masked key line is not the whole
|
||||
# story: a YAML block scalar (`|`, `|-`, `>`, `>-`, ...) puts the VALUE on the lines that follow
|
||||
# the key, each indented deeper than it. The key-name match above only ever sees the key line
|
||||
# itself, so those continuation lines used to flow straight through unredacted while the key line
|
||||
# right above them printed a reassuring "<redacted>" — an incomplete redactor that looks complete
|
||||
# is worse than one that visibly does nothing, because it stops a reviewer from looking further.
|
||||
# The fix is structural, not another name to match: once a key line is masked, every following
|
||||
# line indented STRICTLY DEEPER than that key is masked too, by indentation alone, until the
|
||||
# indentation returns to the key's own level or shallower. This needs no knowledge of the key's
|
||||
# name, so it covers a block scalar under any masked key — but ONLY while that key's own line is
|
||||
# itself inside the hunk being printed. `diff -u` prints just three lines of context, so a block
|
||||
# scalar's body often reaches this function with its key line left out; there is then nothing to
|
||||
# anchor to, `masked` is never set, and the body prints in full. A blank line inside a block
|
||||
# scalar loses the anchor the same way, because a blank diff line measures as indent 0. Both are
|
||||
# measured and filed as fleetd #639 — do not read this paragraph as a guarantee that a masked
|
||||
# key's value can never be printed.
|
||||
#
|
||||
# `redact` is always fed `diff -u` output, and every line of a unified diff starts with exactly
|
||||
# one of ' ', '+', '-' (the three body markers; '@'/'-'/'+' for the three header-line kinds too).
|
||||
# That one leading character is NOT part of the YAML indentation, and must be stripped before
|
||||
# indentation is measured or a key is matched — otherwise a changed ('+' or '-') line reads one
|
||||
# column shallower than it really is, and either wrongly escapes a continuation mask or wrongly
|
||||
# ends one early. Tabs are out of scope: YAML forbids them for indentation, and this is a bounded
|
||||
# fix, not a YAML parser.
|
||||
redact() {
|
||||
local line prefix content indent lead key
|
||||
local masked=0 masked_indent=0 saved_nocasematch=0
|
||||
shopt -q nocasematch && saved_nocasematch=1
|
||||
shopt -s nocasematch
|
||||
sed -E 's#://[^@]*@#://<redacted>@#g' | while IFS= read -r line || [ -n "$line" ]; do
|
||||
case "$line" in
|
||||
[\ +-]*) prefix="${line:0:1}"; content="${line:1}" ;;
|
||||
*) prefix=""; content="$line" ;;
|
||||
esac
|
||||
|
||||
indent=0
|
||||
while [ "${content:$indent:1}" = " " ]; do indent=$((indent + 1)); done
|
||||
|
||||
if [ "$masked" = 1 ] && [ "$indent" -gt "$masked_indent" ]; then
|
||||
printf '%s%*s<redacted>\n' "$prefix" "$indent" ""
|
||||
continue
|
||||
fi
|
||||
masked=0
|
||||
|
||||
if [[ "$content" =~ ^([[:space:]]*)([A-Za-z0-9_.-]+:) ]]; then
|
||||
lead="${BASH_REMATCH[1]}"
|
||||
key="${BASH_REMATCH[2]}"
|
||||
if [[ "$key" =~ (TOKEN|SECRET|PASSWORD|PASSWD|PASSPHRASE|CREDENTIAL|URI|_KEY) ]]; then
|
||||
printf '%s%s%s <redacted>\n' "$prefix" "$lead" "$key"
|
||||
masked=1
|
||||
masked_indent="$indent"
|
||||
continue
|
||||
fi
|
||||
fi
|
||||
printf '%s\n' "$line"
|
||||
done
|
||||
[ "$saved_nocasematch" = 1 ] || shopt -u nocasematch
|
||||
}
|
||||
|
||||
# ------------------------------------------------------------------------------------- the probe
|
||||
#
|
||||
# Probe the SOCKET, never `pgrep`/`ps -f` — both print argv, and argv holds `NAME=value`, making
|
||||
# either a credential channel. The port comes from the config's own `bind.port`; 8765 is only a
|
||||
# fallback when that key is absent or the file does not parse yet.
|
||||
resolve_port() {
|
||||
local file="$1" port
|
||||
if [ -f "$file" ] && command -v yq >/dev/null 2>&1; then
|
||||
port="$(yq eval '.bind.port' "$file" 2>/dev/null || true)"
|
||||
else
|
||||
port=""
|
||||
fi
|
||||
case "$port" in
|
||||
''|null) echo "$FALLBACK_PORT" ;;
|
||||
*) echo "$port" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
daemon_listening() {
|
||||
local port="$1"
|
||||
if command -v nc >/dev/null 2>&1; then
|
||||
nc -z -w1 127.0.0.1 "$port" 2>/dev/null
|
||||
else
|
||||
( exec 3<>"/dev/tcp/127.0.0.1/$port" ) 2>/dev/null
|
||||
fi
|
||||
}
|
||||
|
||||
# --------------------------------------------------------------------------------- the log marker
|
||||
#
|
||||
# Take the log's line count BEFORE touching anything. Every later read of "what did the daemon
|
||||
# say" starts strictly after this mark, so a refusal from hours ago can never be mistaken for
|
||||
# this edit's verdict. Same approach as scripts/redeploy-fleetd.sh's RESTART_MARK.
|
||||
log_mark() {
|
||||
local file="$1"
|
||||
if [ -f "$file" ]; then
|
||||
wc -l < "$file" 2>/dev/null || echo 0
|
||||
else
|
||||
echo 0
|
||||
fi
|
||||
}
|
||||
|
||||
read_verdict_after_marker() {
|
||||
local file="$1" mark="$2"
|
||||
[ -f "$file" ] || return 0
|
||||
tail -n "+$((mark + 1))" "$file" 2>/dev/null || true
|
||||
}
|
||||
|
||||
# Classifies one log LINE. Echoes one of: refused | clean | needs-restart | none. Always
|
||||
# succeeds (every branch ends in `echo`), so it is safe to call from inside `$( )`.
|
||||
classify_verdict_line() {
|
||||
local line="$1"
|
||||
case "$line" in
|
||||
*'config reload refused'*) echo refused ;;
|
||||
*'config reload from '*'refused, keeping the running config'*) echo refused ;;
|
||||
*'config reloaded'*)
|
||||
case "$line" in
|
||||
*'need a restart'*|*'partially live'*) echo needs-restart ;;
|
||||
*) echo clean ;;
|
||||
esac ;;
|
||||
*) echo none ;;
|
||||
esac
|
||||
return 0
|
||||
}
|
||||
|
||||
scan_region_for_verdict() {
|
||||
local region="$1" line kind
|
||||
[ -n "$region" ] || return 1
|
||||
while IFS= read -r line || [ -n "$line" ]; do
|
||||
kind="$(classify_verdict_line "$line")"
|
||||
if [ "$kind" != "none" ]; then
|
||||
VERDICT_KIND="$kind"
|
||||
VERDICT_LINE="$line"
|
||||
return 0
|
||||
fi
|
||||
done <<< "$region"
|
||||
return 1
|
||||
}
|
||||
|
||||
# Sets VERDICT_KIND/VERDICT_LINE and returns 0 on the first verdict line found after $mark;
|
||||
# returns 1 (VERDICT_KIND=none) if none appeared inside $wait_s seconds. Checks once before each
|
||||
# sleep AND once more after the last sleep, the same boundary idiom
|
||||
# scripts/redeploy-fleetd.sh's wait_for_daemon_exit/wait_for_new_pid already use.
|
||||
wait_for_verdict() {
|
||||
local log="$1" mark="$2" wait_s="$3" _i region
|
||||
VERDICT_KIND="none"
|
||||
VERDICT_LINE=""
|
||||
for _i in $(seq "$wait_s"); do
|
||||
region="$(read_verdict_after_marker "$log" "$mark")"
|
||||
scan_region_for_verdict "$region" && return 0
|
||||
sleep 1
|
||||
done
|
||||
region="$(read_verdict_after_marker "$log" "$mark")"
|
||||
scan_region_for_verdict "$region" && return 0
|
||||
return 1
|
||||
}
|
||||
|
||||
last_verdict_line() {
|
||||
local file="$1" line out=""
|
||||
[ -f "$file" ] || return 0
|
||||
while IFS= read -r line || [ -n "$line" ]; do
|
||||
if [ "$(classify_verdict_line "$line")" != "none" ]; then
|
||||
out="$line"
|
||||
fi
|
||||
done < "$file"
|
||||
printf '%s' "$out"
|
||||
}
|
||||
|
||||
default_wait_seconds() {
|
||||
local log="$1" interval=""
|
||||
if [ -f "$log" ]; then
|
||||
interval="$(grep -F 'config watch:' "$log" 2>/dev/null | tail -1 \
|
||||
| sed -E 's/.*\(every ([0-9]+)s\).*/\1/' || true)"
|
||||
fi
|
||||
case "$interval" in
|
||||
''|*[!0-9]*) interval=10 ;;
|
||||
esac
|
||||
echo $((interval * 4))
|
||||
}
|
||||
|
||||
# ----------------------------------------------------------------------------------- the backup
|
||||
#
|
||||
# Timestamped, never pruned — "keep backups" per the ticket. A pid suffix avoids a same-second
|
||||
# collision between two invocations.
|
||||
#
|
||||
# fleetd #635 follow-up — lands under a DEDICATED, gitignored directory beside the config
|
||||
# (<dir>/.config-backups/), never beside the config file itself. The whole reason fleetd.yaml is
|
||||
# gitignored is that it must never be committed, and a backup of it inherits that requirement — a
|
||||
# bare `fleetd.yaml.bak.*` next to a tracked directory is one `git add -A`/`git add .` away from
|
||||
# committing the live config. A directory beats a glob on its own: the glob only protects today's
|
||||
# naming, a location keeps working even if the naming changes later. (See .gitignore for the glob
|
||||
# kept anyway, as a backstop for a stray backup written the old way.)
|
||||
BACKUP_DIRNAME=".config-backups"
|
||||
|
||||
backup_dir_for() {
|
||||
local src="$1"
|
||||
printf '%s/%s' "$(dirname "$src")" "$BACKUP_DIRNAME"
|
||||
}
|
||||
|
||||
backup_config() {
|
||||
local src="$1" ts backup dir base
|
||||
dir="$(backup_dir_for "$src")"
|
||||
mkdir -p "$dir" \
|
||||
|| die "could not create the backup directory $dir — refusing to edit without a backup. The live config at $src was NOT touched."
|
||||
ts="$(date -u +%Y%m%dT%H%M%S)Z"
|
||||
base="$(basename "$src")"
|
||||
backup="${dir}/${base}.bak.${ts}.$$"
|
||||
cp "$src" "$backup" \
|
||||
|| die "could not create a backup at $backup — refusing to edit without one. The live config at $src was NOT touched."
|
||||
printf '%s' "$backup"
|
||||
}
|
||||
|
||||
newest_backup() {
|
||||
local cfg="$1" dir base
|
||||
dir="$(backup_dir_for "$cfg")"
|
||||
base="$(basename "$cfg")"
|
||||
ls -t "${dir}/${base}".bak.* 2>/dev/null | head -1 || true
|
||||
}
|
||||
|
||||
# ------------------------------------------------------------------------------------ the file mode
|
||||
#
|
||||
# fleetd #635 follow-up — `mv` from a mktemp candidate carries mktemp's 0600 onto the live path
|
||||
# forever (measured: 644 -> 600 after one --set), and a restore does not undo it either, because
|
||||
# `cp` onto an EXISTING file keeps the DESTINATION's mode, not the source's. Capture the live
|
||||
# file's mode before anything touches it, and reapply it to whatever lands on that path
|
||||
# afterwards — the candidate before install, and the config again after a restore — so an edit
|
||||
# changes the file's CONTENT only, never its permissions. BSD `stat -f '%Lp'` first (matches this
|
||||
# project's dev machine), GNU `stat -c '%a'` as the fallback. Prints nothing when the file does
|
||||
# not exist yet, so apply_mode then does nothing and a first-ever edit falls back to the normal
|
||||
# umask default rather than inventing a number.
|
||||
file_mode() {
|
||||
local file="$1"
|
||||
[ -f "$file" ] || return 0
|
||||
stat -f '%Lp' "$file" 2>/dev/null || stat -c '%a' "$file" 2>/dev/null || true
|
||||
}
|
||||
|
||||
apply_mode() {
|
||||
local file="$1" mode="$2"
|
||||
[ -n "$mode" ] || return 0
|
||||
chmod "$mode" "$file" 2>/dev/null || true
|
||||
}
|
||||
|
||||
# --------------------------------------------------------------------------- candidate builders
|
||||
#
|
||||
# Never edit the live file in place. Each builder fills $1 (a temp file already sitting in the
|
||||
# SAME directory as the live config, so the later `mv` install is a rename, not a cross-device
|
||||
# copy — see run_edit).
|
||||
# fleetd #635 follow-up — a forgotten value (`--set .a.b=`, a plausible typo) must never be
|
||||
# accepted as "clear the field". `*=*` alone cannot tell "--set .a.b=" from "--set .a.b=7" apart
|
||||
# — both contain an `=` — so the guard has to look at the VALUE, not the shape of the argument.
|
||||
# An empty value refuses outright: nothing is installed, and the message names the likely cause
|
||||
# AND the two ways to actually mean it (clear on purpose, or an intentional empty string via
|
||||
# --from). Measured against the real daemon loader: a quoted empty string reads back as a null
|
||||
# field (`quoted empty -> OK int=null`), and a null numeric field FALLS BACK TO ITS DEFAULT rather
|
||||
# than erroring — so this is not a cosmetic nit, it is the one shape of edit that widens capacity
|
||||
# silently instead of failing loudly, which is exactly what this script exists to catch.
|
||||
#
|
||||
# A deliberate clear needs its own spelling, because `""` and YAML `null` are NOT the same value
|
||||
# to the loader (`""` is a valid empty String; `null` means absent, and an Integer field reads
|
||||
# either the same way — null — but a String field would keep `""` as a real value). `--set
|
||||
# .a.b=null` is that spelling: it writes a literal, unquoted `null` via yq, never the string
|
||||
# "null" through strenv(). One consequence worth knowing: there is currently no --set spelling
|
||||
# for the three-character STRING "null" itself (it collides with the clear spelling) — use
|
||||
# --from for that rare case.
|
||||
apply_set_pairs() {
|
||||
local cand="$1" kv path value
|
||||
shift
|
||||
for kv in "$@"; do
|
||||
case "$kv" in
|
||||
*=*) : ;;
|
||||
*) die "--set expects <yq-path>=<value>, got: '$kv'" ;;
|
||||
esac
|
||||
path="${kv%%=*}"
|
||||
path="${path#.}"
|
||||
value="${kv#*=}"
|
||||
if [ -z "$value" ]; then
|
||||
die "--set '$kv' has an EMPTY value — refusing. Nothing was installed. A forgotten value
|
||||
would NULL the field, and a null value falls back to its default rather than erroring —
|
||||
silent, not safe. Did you mean --set .${path}=null to clear it on purpose, or --from a
|
||||
file if you need a genuinely empty string?"
|
||||
fi
|
||||
# fleetd #635 follow-up (ticket comment 17673, defect 8) — these two failure messages used to
|
||||
# echo the full "$kv" (path=value, exactly as the operator typed it), unredacted. The operator
|
||||
# already has the value, so a terminal is not where this leaks — the risk is where the output
|
||||
# goes NEXT: this fleet pastes command output into tickets, PRs and fleet_reply bodies, and a
|
||||
# failure is exactly when someone copies it to ask for help. Print the PATH, which is what's
|
||||
# needed to fix the command, and never the value. $kv is not key:value-shaped YAML, so piping
|
||||
# it through redact would just pass it straight through — a false sense of coverage, the same
|
||||
# mistake as defect 7.
|
||||
if [ "$value" = "null" ]; then
|
||||
yq eval -i ".${path} = null" "$cand" \
|
||||
|| die "yq could not clear --set '.${path}=null' — nothing was installed. The live config is unchanged."
|
||||
continue
|
||||
fi
|
||||
CONFIG_EDIT_SET_VALUE="$value" yq eval -i ".${path} = strenv(CONFIG_EDIT_SET_VALUE)" "$cand" \
|
||||
|| die "yq could not apply --set '.${path}=<value>' — nothing was installed. The live config is unchanged."
|
||||
done
|
||||
}
|
||||
|
||||
build_from_set() {
|
||||
local cand="$1"
|
||||
cp "$CONFIG" "$cand"
|
||||
apply_set_pairs "$cand" "${SETS[@]}"
|
||||
}
|
||||
|
||||
build_from_file() {
|
||||
local cand="$1"
|
||||
[ -f "$FROM_FILE" ] || die "--from file not found: $FROM_FILE"
|
||||
cp "$FROM_FILE" "$cand"
|
||||
}
|
||||
|
||||
parse_check() {
|
||||
yq eval '.' "$1" >/dev/null 2>&1
|
||||
}
|
||||
|
||||
install_candidate() {
|
||||
local cand="$1" live="$2"
|
||||
mv -f "$cand" "$live"
|
||||
}
|
||||
|
||||
# -------------------------------------------------------------------------------- the report path
|
||||
#
|
||||
# Prints the literal command the operator (or a test) can run to restore the backup by hand — the
|
||||
# absolute path to THIS script plus the overrides actually in force, so it works from any cwd.
|
||||
restore_command_line() {
|
||||
printf '%q --restore --config %q --log %q --wait-seconds %q' "$SELF" "$CONFIG" "$LOG" "$WAIT_SECONDS"
|
||||
}
|
||||
|
||||
# State 4 only: restore the pre-edit backup, then wait for a SECOND verdict confirming the
|
||||
# restore itself reloaded cleanly. Never claims a restore it did not observe — if the second wait
|
||||
# also times out, it says so plainly rather than reporting "restored" as though confirmed.
|
||||
restore_and_confirm() {
|
||||
local backup="$1" mark2 orig_mode
|
||||
orig_mode="$(file_mode "$CONFIG")"
|
||||
mark2="$(log_mark "$LOG")"
|
||||
cp "$backup" "$CONFIG" \
|
||||
|| die "could not restore $backup onto $CONFIG — the live config is left as the REFUSED edit. Fix this by hand immediately: cp \"$backup\" \"$CONFIG\""
|
||||
apply_mode "$CONFIG" "$orig_mode"
|
||||
ok "restored from $backup"
|
||||
if wait_for_verdict "$LOG" "$mark2" "$WAIT_SECONDS"; then
|
||||
case "$VERDICT_KIND" in
|
||||
refused) warn "the RESTORE was also refused by the daemon: $VERDICT_LINE" ;;
|
||||
*) ok "restore confirmed: $VERDICT_LINE" ;;
|
||||
esac
|
||||
else
|
||||
warn "the restore is on disk, but no confirming verdict line appeared within ${WAIT_SECONDS}s"
|
||||
warn "cannot confirm the restore reloaded cleanly — check $LOG by hand"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# The four-outcome decision. Echoed as a function so run_edit/restore_mode share one place that
|
||||
# can return 0/3/4/5 — never duplicated, never re-worded between the two callers.
|
||||
report_outcome() {
|
||||
local mark="$1" backup="$2" kind line
|
||||
|
||||
say "waiting for the daemon's verdict (up to ${WAIT_SECONDS}s)"
|
||||
if wait_for_verdict "$LOG" "$mark" "$WAIT_SECONDS"; then
|
||||
kind="$VERDICT_KIND"; line="$VERDICT_LINE"
|
||||
else
|
||||
kind="none"
|
||||
fi
|
||||
|
||||
case "$kind" in
|
||||
clean)
|
||||
ok "daemon verdict: $line"
|
||||
say "result: applied cleanly"
|
||||
return 0 ;;
|
||||
needs-restart)
|
||||
ok "daemon verdict: $line"
|
||||
say "result: applied — a restart is needed for the change(s) named above"
|
||||
return 3 ;;
|
||||
refused)
|
||||
warn "daemon verdict: $line"
|
||||
say "result: REFUSED — restoring the backup"
|
||||
restore_and_confirm "$backup"
|
||||
return 4 ;;
|
||||
none)
|
||||
warn "no verdict line appeared within ${WAIT_SECONDS}s after $LOG line $mark"
|
||||
warn "CANNOT TELL whether the daemon applied this edit, refused it, or is simply down."
|
||||
warn "Nothing was restored — the edit is still on disk at $CONFIG."
|
||||
echo
|
||||
echo " backup: $backup"
|
||||
echo " to restore it by hand:"
|
||||
echo " $(restore_command_line)"
|
||||
return 5 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# ------------------------------------------------------------------------------------- the modes
|
||||
check_mode() {
|
||||
say "config-edit --check"
|
||||
if [ -f "$CONFIG" ]; then
|
||||
if parse_check "$CONFIG"; then
|
||||
ok "config parses: $CONFIG"
|
||||
else
|
||||
warn "config does NOT parse as valid YAML: $CONFIG"
|
||||
fi
|
||||
else
|
||||
warn "no config file at $CONFIG"
|
||||
fi
|
||||
|
||||
local port
|
||||
port="$(resolve_port "$CONFIG")"
|
||||
if daemon_listening "$port"; then
|
||||
ok "daemon is listening on 127.0.0.1:$port"
|
||||
else
|
||||
warn "no daemon detected listening on 127.0.0.1:$port"
|
||||
fi
|
||||
|
||||
ok "watch interval assumed: $(( $(default_wait_seconds "$LOG") / 4 ))s (derives --wait-seconds default of $(default_wait_seconds "$LOG")s)"
|
||||
|
||||
local verdict
|
||||
verdict="$(last_verdict_line "$LOG")"
|
||||
if [ -n "$verdict" ]; then
|
||||
ok "last verdict in log: $verdict"
|
||||
else
|
||||
warn "no reload verdict line found in $LOG"
|
||||
fi
|
||||
|
||||
local backup
|
||||
backup="$(newest_backup "$CONFIG")"
|
||||
if [ -n "$backup" ]; then
|
||||
ok "newest backup: $backup"
|
||||
else
|
||||
warn "no backups found for $CONFIG"
|
||||
fi
|
||||
|
||||
if command -v yq >/dev/null 2>&1; then
|
||||
ok "yq: $(yq --version 2>&1)"
|
||||
else
|
||||
warn "yq not found on PATH"
|
||||
fi
|
||||
|
||||
return 0
|
||||
}
|
||||
|
||||
# Shared by --set and --from: backup, build, parse-check, redacted diff, install, await verdict.
|
||||
run_edit() {
|
||||
local builder="$1"
|
||||
[ -f "$CONFIG" ] || die "no config at $CONFIG — nothing to edit"
|
||||
|
||||
local mark orig_mode
|
||||
mark="$(log_mark "$LOG")"
|
||||
orig_mode="$(file_mode "$CONFIG")"
|
||||
|
||||
say "probe"
|
||||
local port
|
||||
port="$(resolve_port "$CONFIG")"
|
||||
if daemon_listening "$port"; then
|
||||
ok "daemon appears to be listening on 127.0.0.1:$port"
|
||||
else
|
||||
warn "no daemon detected listening on 127.0.0.1:$port — a verdict may never appear"
|
||||
fi
|
||||
|
||||
say "backup"
|
||||
local backup
|
||||
backup="$(backup_config "$CONFIG")"
|
||||
ok "backup: $backup"
|
||||
|
||||
say "candidate"
|
||||
local cand
|
||||
CAND="$(mktemp "$(dirname "$CONFIG")/.config-edit.XXXXXX")" \
|
||||
|| die "could not create a candidate temp file next to $CONFIG"
|
||||
cand="$CAND"
|
||||
if ! "$builder" "$cand"; then
|
||||
rm -f "$cand"; CAND=""
|
||||
die "could not build the candidate — nothing was installed. The live config at $CONFIG is unchanged."
|
||||
fi
|
||||
|
||||
if ! parse_check "$cand"; then
|
||||
rm -f "$cand"; CAND=""
|
||||
die "candidate does not parse as valid YAML — nothing was installed. The live config at $CONFIG is unchanged."
|
||||
fi
|
||||
ok "candidate parses"
|
||||
|
||||
apply_mode "$cand" "$orig_mode"
|
||||
|
||||
say "change (redacted)"
|
||||
diff -u "$backup" "$cand" | redact || true
|
||||
|
||||
say "install"
|
||||
install_candidate "$cand" "$CONFIG" \
|
||||
|| die "could not install the candidate onto $CONFIG — the live config was NOT changed. The validated candidate is sitting at $cand; investigate before retrying."
|
||||
CAND=""
|
||||
ok "installed: $CONFIG"
|
||||
|
||||
local rc=0
|
||||
report_outcome "$mark" "$backup" || rc=$?
|
||||
return "$rc"
|
||||
}
|
||||
|
||||
dry_run_diff() {
|
||||
local builder="$1"
|
||||
[ -f "$CONFIG" ] || die "no config at $CONFIG — nothing to diff against"
|
||||
local cand
|
||||
CAND="$(mktemp "$(dirname "$CONFIG")/.config-edit.XXXXXX")" \
|
||||
|| die "could not create a candidate temp file next to $CONFIG"
|
||||
cand="$CAND"
|
||||
if ! "$builder" "$cand"; then
|
||||
rm -f "$cand"; CAND=""
|
||||
die "could not build the candidate — this was a --dry-run, nothing would have been installed either"
|
||||
fi
|
||||
if ! parse_check "$cand"; then
|
||||
rm -f "$cand"; CAND=""
|
||||
die "candidate does not parse as valid YAML — this was a --dry-run, nothing would have been installed either"
|
||||
fi
|
||||
say "dry run — diff (redacted), nothing installed"
|
||||
diff -u "$CONFIG" "$cand" | redact || true
|
||||
rm -f "$cand"; CAND=""
|
||||
return 0
|
||||
}
|
||||
|
||||
restore_mode() {
|
||||
[ -f "$CONFIG" ] || die "no config at $CONFIG to restore onto"
|
||||
local backup dir base
|
||||
backup="$(newest_backup "$CONFIG")"
|
||||
if [ -z "$backup" ]; then
|
||||
# fleetd #635 follow-up (ticket comment 17664) — this message must name the directory the
|
||||
# code actually searches (backup_dir_for, same as newest_backup), not the old beside-the-
|
||||
# config glob. A backup written the OLD way is real and NOT searched any more — say so and
|
||||
# give the one-line recovery command — but do NOT make the search itself look there; that
|
||||
# would be a behaviour change nobody asked for. The message is the only thing being fixed.
|
||||
dir="$(backup_dir_for "$CONFIG")"
|
||||
base="$(basename "$CONFIG")"
|
||||
die "no backup found matching ${dir}/${base}.bak.* — nothing to restore.
|
||||
A backup written the OLD way, directly beside the config (${CONFIG}.bak.*), is NOT
|
||||
searched — that location was retired so a backup of a file that must never be committed
|
||||
cannot sit next to a tracked directory. If one exists there, recover it by hand:
|
||||
cp ${CONFIG}.bak.<timestamp>.<pid> $CONFIG"
|
||||
fi
|
||||
[ -f "$backup" ] || die "backup candidate $backup vanished"
|
||||
|
||||
say "restore"
|
||||
ok "restoring $backup onto $CONFIG"
|
||||
local mark orig_mode
|
||||
mark="$(log_mark "$LOG")"
|
||||
orig_mode="$(file_mode "$CONFIG")"
|
||||
cp "$backup" "$CONFIG" || die "could not copy $backup onto $CONFIG"
|
||||
apply_mode "$CONFIG" "$orig_mode"
|
||||
ok "installed: $CONFIG"
|
||||
|
||||
local rc=0
|
||||
report_outcome "$mark" "$backup" || rc=$?
|
||||
return "$rc"
|
||||
}
|
||||
|
||||
# -------------------------------------------------------------------------------------- dispatch
|
||||
|
||||
if [ -n "$WAIT_SECONDS_OVERRIDE" ]; then
|
||||
WAIT_SECONDS="$WAIT_SECONDS_OVERRIDE"
|
||||
else
|
||||
WAIT_SECONDS="$(default_wait_seconds "$LOG")"
|
||||
fi
|
||||
|
||||
RC=0
|
||||
case "$MODE" in
|
||||
check)
|
||||
check_mode || RC=$?
|
||||
;;
|
||||
set)
|
||||
[ "${#SETS[@]}" -gt 0 ] || die "--set requires at least one <yq-path>=<value>"
|
||||
if [ "$DRY_RUN" = 1 ]; then
|
||||
dry_run_diff build_from_set || RC=$?
|
||||
else
|
||||
run_edit build_from_set || RC=$?
|
||||
fi
|
||||
;;
|
||||
from)
|
||||
[ -n "$FROM_FILE" ] || die "--from requires a candidate file path"
|
||||
if [ "$DRY_RUN" = 1 ]; then
|
||||
dry_run_diff build_from_file || RC=$?
|
||||
else
|
||||
run_edit build_from_file || RC=$?
|
||||
fi
|
||||
;;
|
||||
restore)
|
||||
restore_mode || RC=$?
|
||||
;;
|
||||
esac
|
||||
|
||||
exit "$RC"
|
||||
Executable
+585
@@ -0,0 +1,585 @@
|
||||
#!/usr/bin/env bash
|
||||
# Self-contained checks for scripts/config-edit.sh — fleetd ticket #635.
|
||||
#
|
||||
# Drives the REAL config-edit.sh as a subprocess against a FIXTURE config and a FIXTURE log in a
|
||||
# throwaway temp directory this file creates and removes. Never touches fleetd/fleetd.yaml or
|
||||
# fleetd/fleetd.out, and never starts, stops, or contacts a daemon — there is no daemon here, so
|
||||
# each test PLAYS the daemon: it starts config-edit.sh in the background (it is waiting on the
|
||||
# log), appends the verdict line it wants, then collects the real exit code.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
EDIT="$ROOT/scripts/config-edit.sh"
|
||||
TMP="$(mktemp -d "$ROOT/.config-edit-test.XXXXXX")"
|
||||
trap 'rm -rf "$TMP"' EXIT
|
||||
|
||||
fail() {
|
||||
printf 'FAIL: %s\n' "$*" >&2
|
||||
return 1
|
||||
}
|
||||
|
||||
assert_equals() {
|
||||
local expected="$1" actual="$2" description="$3"
|
||||
[ "$expected" = "$actual" ] || fail "$description: expected $expected, got $actual"
|
||||
}
|
||||
|
||||
assert_contains() {
|
||||
local needle="$1" text="$2" description="$3"
|
||||
printf '%s' "$text" | grep -qF -- "$needle" || fail "$description: missing [$needle]"
|
||||
}
|
||||
|
||||
assert_not_contains() {
|
||||
local needle="$1" text="$2" description="$3"
|
||||
if printf '%s' "$text" | grep -qF -- "$needle"; then
|
||||
fail "$description: must NOT contain [$needle], but it does"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# A fresh fixture pair per test: $1/fleetd.yaml (the config) and $1/fleetd.out (the log), plus a
|
||||
# small wait-seconds budget so no test takes long. Returns the fixture dir via stdout.
|
||||
new_fixture() {
|
||||
local dir
|
||||
dir="$(mktemp -d "$TMP/fixture.XXXXXX")"
|
||||
cat > "$dir/fleetd.yaml" <<'YAML'
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 19999
|
||||
broker:
|
||||
uri: amqp://user:hunter2@host/vhost
|
||||
profiles:
|
||||
sonnet:
|
||||
weight: 3
|
||||
maxLoad: 5
|
||||
YAML
|
||||
: > "$dir/fleetd.out"
|
||||
printf '%s' "$dir"
|
||||
}
|
||||
|
||||
# Runs config-edit.sh in the background against $dir's fixtures, with the given extra args, and
|
||||
# a short --wait-seconds. Sets RUN_PID. Caller appends to $dir/fleetd.out (or not, for the
|
||||
# silence test) and then calls collect_run to block for the exit code.
|
||||
start_run() {
|
||||
local dir="$1" wait_s="$2"; shift 2
|
||||
(
|
||||
# config-edit.sh deliberately exits 3/4/5 on several of these tests. This subshell inherits
|
||||
# the parent's `set -e`, and without disabling it here the FIRST nonzero exit would kill the
|
||||
# subshell before the `echo $? > rc` line ever ran — the real code would never reach the file.
|
||||
set +e
|
||||
"$EDIT" --config "$dir/fleetd.yaml" --log "$dir/fleetd.out" --wait-seconds "$wait_s" "$@" \
|
||||
> "$dir/stdout.log" 2>&1
|
||||
echo $? > "$dir/rc"
|
||||
) &
|
||||
RUN_PID=$!
|
||||
}
|
||||
|
||||
collect_run() {
|
||||
local dir="$1"
|
||||
# wait echoes back the backgrounded subshell's own exit status (here, deliberately 3/4/5 on
|
||||
# several tests) — under `set -e` a bare nonzero `wait` would abort this whole test script, so
|
||||
# it is neutralized with `|| true`; the real code is read from $dir/rc right after.
|
||||
wait "$RUN_PID" || true
|
||||
RUN_OUTPUT="$(cat "$dir/stdout.log")"
|
||||
RUN_RC="$(cat "$dir/rc")"
|
||||
}
|
||||
|
||||
# -------------------------------------------------------------- acceptance criterion 1: refusal
|
||||
test_refusal_restores_byte_for_byte() {
|
||||
local dir
|
||||
dir="$(new_fixture)"
|
||||
cp "$dir/fleetd.yaml" "$dir/pre-edit.yaml"
|
||||
|
||||
start_run "$dir" 5 --set '.broker.uri=amqp://changed@host/x'
|
||||
sleep 1
|
||||
printf 'config reload refused — these keys cannot change under a running daemon: broker. Restart fleetd to apply them.\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
|
||||
assert_equals 4 "$RUN_RC" "refusal exit code"
|
||||
cmp -s "$dir/fleetd.yaml" "$dir/pre-edit.yaml" \
|
||||
|| fail "refusal must restore the config byte for byte onto the pre-edit backup"
|
||||
}
|
||||
|
||||
# -------------------------------------------------------------- acceptance criterion 2: clean
|
||||
test_clean_reload_keeps_the_edit() {
|
||||
local dir
|
||||
dir="$(new_fixture)"
|
||||
|
||||
start_run "$dir" 5 --set '.profiles.sonnet.weight=7'
|
||||
sleep 1
|
||||
printf 'config reloaded\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
|
||||
assert_equals 0 "$RUN_RC" "clean reload exit code"
|
||||
assert_equals "7" "$(yq eval '.profiles.sonnet.weight' "$dir/fleetd.yaml")" "clean reload live value"
|
||||
}
|
||||
|
||||
# ----------------------------------------------------- acceptance criterion 3: deferred != clean
|
||||
test_deferred_reload_is_told_apart_from_clean() {
|
||||
local dir
|
||||
dir="$(new_fixture)"
|
||||
|
||||
start_run "$dir" 5 --set '.profiles.sonnet.weight=9'
|
||||
sleep 1
|
||||
printf 'config reloaded; these changes need a restart to take effect: profiles\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
|
||||
assert_equals 3 "$RUN_RC" "deferred reload exit code"
|
||||
[ "$RUN_RC" != 0 ] || fail "deferred reload must not report exit 0"
|
||||
assert_contains "profiles" "$RUN_OUTPUT" "deferred reload names the key"
|
||||
assert_contains "restart" "$RUN_OUTPUT" "deferred reload says a restart is needed"
|
||||
}
|
||||
|
||||
# -------------------------------------------------------------- acceptance criterion 4: silence
|
||||
test_silence_is_its_own_answer() {
|
||||
local dir
|
||||
dir="$(new_fixture)"
|
||||
|
||||
start_run "$dir" 2 --set '.profiles.sonnet.weight=11'
|
||||
# Feed the log nothing.
|
||||
collect_run "$dir"
|
||||
|
||||
assert_equals 5 "$RUN_RC" "silence exit code"
|
||||
assert_equals "11" "$(yq eval '.profiles.sonnet.weight' "$dir/fleetd.yaml")" "the edited value must still be on disk"
|
||||
assert_contains '--restore' "$RUN_OUTPUT" "silence prints the --restore command"
|
||||
|
||||
local backup restore_cmd
|
||||
backup="$(ls -t "$dir"/.config-backups/fleetd.yaml.bak.* | head -1)"
|
||||
[ -n "$backup" ] || fail "silence must still have taken a backup"
|
||||
|
||||
restore_cmd="$(printf '%s\n' "$RUN_OUTPUT" | grep -F -- '--restore --config' | sed -E 's/^[[:space:]]*//')"
|
||||
[ -n "$restore_cmd" ] || fail "could not find the printed --restore invocation in the output"
|
||||
# Running this --restore invocation installs the backup, then itself waits for a confirming
|
||||
# verdict that this fixture never feeds — so it legitimately exits 5 ("cannot tell") here, same
|
||||
# as any edit with no daemon on the other end. Only a usage/internal error (1 or 2) is a real
|
||||
# failure of the command itself; the actual assertion is the byte-for-byte cmp below.
|
||||
local restore_rc=0
|
||||
eval "$restore_cmd" > "$dir/restore.log" 2>&1 || restore_rc=$?
|
||||
case "$restore_rc" in
|
||||
0|3|4|5) : ;;
|
||||
*) fail "the printed --restore command errored out (exit $restore_rc): $(cat "$dir/restore.log")" ;;
|
||||
esac
|
||||
|
||||
cmp -s "$dir/fleetd.yaml" "$backup" \
|
||||
|| fail "running the printed --restore command must put the file back to the original backup"
|
||||
}
|
||||
|
||||
# --------------------------------------------------------- acceptance criterion 5: bad candidate
|
||||
test_broken_candidate_never_reaches_live_path() {
|
||||
local dir rc=0
|
||||
dir="$(new_fixture)"
|
||||
printf 'foo: [unclosed\n' > "$dir/broken.yaml"
|
||||
|
||||
"$EDIT" --from "$dir/broken.yaml" --config "$dir/fleetd.yaml" --log "$dir/fleetd.out" --wait-seconds 2 \
|
||||
> "$dir/stdout.log" 2>&1 || rc=$?
|
||||
|
||||
[ "$rc" -ne 0 ] || fail "a broken --from candidate must exit non-zero"
|
||||
cmp -s "$dir/fleetd.yaml" <(new_fixture_yaml) \
|
||||
|| fail "the broken candidate must never reach the live fixture config"
|
||||
}
|
||||
new_fixture_yaml() {
|
||||
cat <<'YAML'
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 19999
|
||||
broker:
|
||||
uri: amqp://user:hunter2@host/vhost
|
||||
profiles:
|
||||
sonnet:
|
||||
weight: 3
|
||||
maxLoad: 5
|
||||
YAML
|
||||
}
|
||||
|
||||
# -------------------------------------------------------------------- acceptance criterion 6
|
||||
test_marker_skips_lines_before_it() {
|
||||
local dir
|
||||
dir="$(new_fixture)"
|
||||
printf 'config reload refused — something ancient\n' > "$dir/fleetd.out"
|
||||
|
||||
start_run "$dir" 5 --set '.profiles.sonnet.weight=5'
|
||||
sleep 1
|
||||
printf 'config reloaded\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
|
||||
assert_equals 0 "$RUN_RC" "a stale refusal before the marker must not be read as this edit's verdict"
|
||||
}
|
||||
|
||||
# ------------------------------------------------------------------- acceptance criterion 7 (+13)
|
||||
# fleetd #635 follow-up (ticket comment 17659) — the two assertions below this comment were the
|
||||
# WHOLE test before the follow-up, and both are negative-only: they pass just as happily when the
|
||||
# diff is never printed at all as when it is printed and correctly redacted. A mutant that deletes
|
||||
# `diff -u "$backup" "$cand" | redact` from the edit path survives them, because an absent output
|
||||
# contains neither "hunter2" nor "user:" either — see the mutation-and-revert proof in the reply.
|
||||
# Criterion 13 is the fix: a LOUD positive control that only passes when a diff was demonstrably
|
||||
# printed AND the redaction demonstrably ran on real content, not merely that nothing leaked.
|
||||
test_redaction_holds() {
|
||||
local dir
|
||||
dir="$(new_fixture)"
|
||||
|
||||
start_run "$dir" 5 --set '.profiles.sonnet.weight=4'
|
||||
sleep 1
|
||||
printf 'config reloaded\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
|
||||
assert_equals 0 "$RUN_RC" "redaction-case reload exit code"
|
||||
assert_not_contains "hunter2" "$RUN_OUTPUT" "full output must never contain the password"
|
||||
assert_not_contains "user:" "$RUN_OUTPUT" "full output must never contain the userinfo"
|
||||
# acceptance criterion 13 — positive control: the diff's default 3-line context around the
|
||||
# changed "weight" key also covers the fixture's "uri:" line, so a genuinely-printed, genuinely-
|
||||
# redacted diff must contain BOTH the redaction marker and the changed key's name. A test that
|
||||
# only ever asserts absence cannot tell "redacted" from "never printed" apart; this can.
|
||||
assert_contains "<redacted>" "$RUN_OUTPUT" "the redaction must be PROVEN to have run on real content, not merely absent"
|
||||
assert_contains "weight" "$RUN_OUTPUT" "a diff must have been demonstrably printed at all"
|
||||
}
|
||||
|
||||
# ------------------------------------------------------- acceptance criterion 9: forgotten value
|
||||
# `--set .a.b=` is a plausible typo (the value simply forgotten), and it must be refused outright
|
||||
# rather than silently nulling the field — a null numeric field falls back to its default, which
|
||||
# widens capacity instead of failing loudly. No background verdict feeder here: a refused --set
|
||||
# must never even reach the daemon, so this never starts a background run at all.
|
||||
test_forgotten_value_refuses_and_installs_nothing() {
|
||||
local dir rc=0
|
||||
dir="$(new_fixture)"
|
||||
cp "$dir/fleetd.yaml" "$dir/pre-edit.yaml"
|
||||
|
||||
"$EDIT" --set '.profiles.sonnet.maxLoad=' \
|
||||
--config "$dir/fleetd.yaml" --log "$dir/fleetd.out" --wait-seconds 2 \
|
||||
> "$dir/stdout.log" 2>&1 || rc=$?
|
||||
RUN_OUTPUT="$(cat "$dir/stdout.log")"
|
||||
|
||||
[ "$rc" -ne 0 ] || fail "an empty --set value must exit non-zero, got 0"
|
||||
cmp -s "$dir/fleetd.yaml" "$dir/pre-edit.yaml" \
|
||||
|| fail "an empty --set value must install nothing — the live fixture changed"
|
||||
assert_contains "EMPTY value" "$RUN_OUTPUT" "the refusal must name the empty value"
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------- acceptance criterion 10: explicit null
|
||||
# `--set .a.b=null` is the deliberate-clear spelling, and it must write a REAL yaml null, never
|
||||
# the string "''" — those are different values to the daemon's loader (fleetd ticket #635's
|
||||
# follow-up comment measured `""` reading back as a null field anyway, which is exactly why the
|
||||
# two forms must not collapse onto each other: `--set path=` refuses instead of silently reaching
|
||||
# this same null outcome through the back door). Read the RAW line with grep, never only through
|
||||
# `yq` — `yq eval` reports `null` for both an actual null and a missing/absent key, so it cannot
|
||||
# tell "wrote null" apart from "wrote nothing"; only the literal line on disk can.
|
||||
test_explicit_null_writes_bare_null_not_empty_string() {
|
||||
local dir
|
||||
dir="$(new_fixture)"
|
||||
|
||||
start_run "$dir" 5 --set '.profiles.sonnet.maxLoad=null'
|
||||
sleep 1
|
||||
printf 'config reloaded\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
|
||||
assert_equals 0 "$RUN_RC" "explicit null clear exit code"
|
||||
local raw_line
|
||||
raw_line="$(grep -E 'maxLoad' "$dir/fleetd.yaml")"
|
||||
assert_contains "null" "$raw_line" "the installed line must spell a bare null"
|
||||
assert_not_contains '""' "$raw_line" "the installed line must NOT be a quoted empty string"
|
||||
}
|
||||
|
||||
# ------------------------------------------------------- acceptance criterion 11: backup never committable
|
||||
# A backup of fleetd.yaml inherits fleetd.yaml's own "never commit this" requirement (fleetd #635
|
||||
# follow-up, ticket comment 17655). Proves two things: the backup lands somewhere `git
|
||||
# check-ignore` reports as ignored (equivalently, a path `git status --porcelain` never lists as
|
||||
# untracked), AND that --restore still finds and uses it from that location.
|
||||
test_backup_is_never_committable() {
|
||||
local dir backup
|
||||
dir="$(new_fixture)"
|
||||
cp "$dir/fleetd.yaml" "$dir/pre-edit.yaml"
|
||||
|
||||
start_run "$dir" 5 --set '.profiles.sonnet.weight=55'
|
||||
sleep 1
|
||||
printf 'config reloaded\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
assert_equals 0 "$RUN_RC" "setup edit exit code"
|
||||
|
||||
backup="$(ls -t "$dir"/.config-backups/fleetd.yaml.bak.* 2>/dev/null | head -1)"
|
||||
[ -n "$backup" ] || fail "no backup found under .config-backups/ — did the location change?"
|
||||
|
||||
git -C "$ROOT" check-ignore -q -- "$backup" \
|
||||
|| fail "the backup at $backup is NOT gitignored — it would survive a git add -A"
|
||||
if git -C "$ROOT" status --porcelain -- "$backup" 2>/dev/null | grep -q '^??'; then
|
||||
fail "git status still lists the backup as untracked: $backup"
|
||||
fi
|
||||
|
||||
start_run "$dir" 5 --restore
|
||||
sleep 1
|
||||
printf 'config reloaded\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
assert_equals 0 "$RUN_RC" "--restore after the backup-location change exit code"
|
||||
cmp -s "$dir/fleetd.yaml" "$dir/pre-edit.yaml" \
|
||||
|| fail "--restore from the new backup location must still put the file back byte for byte"
|
||||
}
|
||||
|
||||
# --------------------------------------------------------- acceptance criterion 12: file mode
|
||||
# `mv` from a mktemp candidate carries mktemp's 0600 forever, and a plain `cp` onto an existing
|
||||
# file keeps the DESTINATION's mode rather than the source's, so a restore does not undo the
|
||||
# narrowing either (fleetd #635 follow-up, ticket comment 17657). Proves the mode survives an edit
|
||||
# AND a subsequent restore, from two different starting points — 644 is the common case, 600
|
||||
# proves the fix PRESERVES whatever mode was there rather than hardcoding 644.
|
||||
test_file_mode_survives_edit_and_restore() {
|
||||
local dir want got
|
||||
for want in 644 600; do
|
||||
dir="$(new_fixture)"
|
||||
chmod "$want" "$dir/fleetd.yaml"
|
||||
|
||||
start_run "$dir" 5 --set ".profiles.sonnet.weight=${want}"
|
||||
sleep 1
|
||||
printf 'config reloaded\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
assert_equals 0 "$RUN_RC" "mode-preservation setup edit exit code ($want)"
|
||||
got="$(stat -f '%Lp' "$dir/fleetd.yaml" 2>/dev/null || stat -c '%a' "$dir/fleetd.yaml")"
|
||||
assert_equals "$want" "$got" "mode must survive a --set ($want)"
|
||||
|
||||
start_run "$dir" 5 --restore
|
||||
sleep 1
|
||||
printf 'config reloaded\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
assert_equals 0 "$RUN_RC" "mode-preservation restore exit code ($want)"
|
||||
got="$(stat -f '%Lp' "$dir/fleetd.yaml" 2>/dev/null || stat -c '%a' "$dir/fleetd.yaml")"
|
||||
assert_equals "$want" "$got" "mode must survive a --restore ($want)"
|
||||
done
|
||||
}
|
||||
|
||||
# ----------------------------------------- acceptance criterion 14: restore message names the real directory
|
||||
# fleetd #635 follow-up (ticket comment 17664, defect 6) — the --restore "no backup found"
|
||||
# message used to print the OLD beside-the-config glob even though newest_backup had already
|
||||
# moved to searching the managed directory. Proves BOTH directions: the not-found message names
|
||||
# the directory actually searched (not merely that it says SOMETHING), and that a real backup
|
||||
# sitting in that directory still lets --restore succeed — otherwise the fix could regress into
|
||||
# a message that is always printed regardless of whether a backup exists.
|
||||
test_restore_message_names_the_searched_directory() {
|
||||
local dir rc=0
|
||||
|
||||
# Direction 1: no backup anywhere — the message must name .config-backups/, not the bare
|
||||
# beside-the-config glob the OLD code printed.
|
||||
dir="$(new_fixture)"
|
||||
"$EDIT" --restore --config "$dir/fleetd.yaml" --log "$dir/fleetd.out" --wait-seconds 2 \
|
||||
> "$dir/stdout.log" 2>&1 || rc=$?
|
||||
RUN_OUTPUT="$(cat "$dir/stdout.log")"
|
||||
|
||||
assert_equals 1 "$rc" "--restore with no backup anywhere exit code"
|
||||
assert_contains ".config-backups/fleetd.yaml.bak.*" "$RUN_OUTPUT" \
|
||||
"the not-found message must name the directory actually searched, not the old beside-the-config glob"
|
||||
|
||||
# Direction 2: a real backup IS present in .config-backups/ — --restore must still succeed, so
|
||||
# the message fix cannot have turned into one that prints regardless of whether a backup exists.
|
||||
dir="$(new_fixture)"
|
||||
cp "$dir/fleetd.yaml" "$dir/pre-edit.yaml"
|
||||
start_run "$dir" 5 --set '.profiles.sonnet.weight=77'
|
||||
sleep 1
|
||||
printf 'config reloaded\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
assert_equals 0 "$RUN_RC" "setup edit exit code for criterion 14's second half"
|
||||
|
||||
start_run "$dir" 5 --restore
|
||||
sleep 1
|
||||
printf 'config reloaded\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
assert_equals 0 "$RUN_RC" "--restore with a real backup present must still succeed"
|
||||
cmp -s "$dir/fleetd.yaml" "$dir/pre-edit.yaml" \
|
||||
|| fail "--restore with a real backup present must put the file back byte for byte"
|
||||
}
|
||||
|
||||
# ------------------------- acceptance criterion 15a: block-scalar continuation lines are redacted
|
||||
# fleetd #635 follow-up (ticket comment 17670, defect 7) — redact() used to look only AT the key
|
||||
# line. A YAML block scalar (`|`) puts its value on the lines that FOLLOW the key, each indented
|
||||
# deeper than it, so the real secret flowed through untouched while the key line right above it
|
||||
# printed a reassuring "<redacted>" — worse than no redaction, because the marker stops a reader
|
||||
# from looking further. The edited key here ("retries") sits directly next to the block scalar,
|
||||
# well inside diff -u's default 3-line context window, so the printed hunk is GUARANTEED to
|
||||
# include the secret's lines — placing the edit further away would let this pass today even
|
||||
# without the fix, proving nothing (the ticket comment's own warning, from the lead's first
|
||||
# reproduction attempt). The positive control runs FIRST: without it, "the secret never entered
|
||||
# the diff at all" would pass identically to "it entered and was correctly redacted".
|
||||
new_fixture_block_scalar() {
|
||||
local dir
|
||||
dir="$(mktemp -d "$TMP/fixture.XXXXXX")"
|
||||
cat > "$dir/fleetd.yaml" <<'YAML'
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 19999
|
||||
broker:
|
||||
uri: amqp://user:hunter2@host/vhost
|
||||
auth:
|
||||
token: |
|
||||
FAKELEAK-BLOCK-SCALAR
|
||||
retries: 1
|
||||
profiles:
|
||||
sonnet:
|
||||
weight: 3
|
||||
maxLoad: 5
|
||||
YAML
|
||||
: > "$dir/fleetd.out"
|
||||
printf '%s' "$dir"
|
||||
}
|
||||
|
||||
test_block_scalar_continuation_is_redacted() {
|
||||
local dir
|
||||
dir="$(new_fixture_block_scalar)"
|
||||
|
||||
start_run "$dir" 5 --set '.auth.retries=2'
|
||||
sleep 1
|
||||
printf 'config reloaded\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
|
||||
assert_equals 0 "$RUN_RC" "block-scalar case reload exit code"
|
||||
# Positive control FIRST: the key's own (masked) line must really be in the printed diff, or the
|
||||
# negative assertion right after proves nothing — see the comment above this test.
|
||||
assert_contains "token:" "$RUN_OUTPUT" "block-scalar case: the key's line must be in the printed diff"
|
||||
assert_contains "<redacted>" "$RUN_OUTPUT" "block-scalar case: redaction must be proven to have run on real content"
|
||||
assert_not_contains "FAKELEAK-BLOCK-SCALAR" "$RUN_OUTPUT" "block-scalar case: the block scalar's VALUE must never leak"
|
||||
}
|
||||
|
||||
# ------------------------------------- acceptance criterion 15b: "passphrase" is also recognised
|
||||
# "passphrase" was in none of TOKEN|SECRET|PASSWORD|PASSWD|CREDENTIAL|URI|_KEY (ticket comment
|
||||
# 17670). This is a plain key:value line, not a block scalar — kept in its OWN fixture and OWN
|
||||
# function, separate from criterion 15a, so that a failure in one case can never mask a failure in
|
||||
# the other (a single combined test would abort under `set -e` at its first failing assertion,
|
||||
# and the second case would then never even run).
|
||||
new_fixture_passphrase() {
|
||||
local dir
|
||||
dir="$(mktemp -d "$TMP/fixture.XXXXXX")"
|
||||
cat > "$dir/fleetd.yaml" <<'YAML'
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 19999
|
||||
broker:
|
||||
uri: amqp://user:hunter2@host/vhost
|
||||
auth:
|
||||
passphrase: FAKELEAK-PASSPHRASE
|
||||
retries: 1
|
||||
profiles:
|
||||
sonnet:
|
||||
weight: 3
|
||||
maxLoad: 5
|
||||
YAML
|
||||
: > "$dir/fleetd.out"
|
||||
printf '%s' "$dir"
|
||||
}
|
||||
|
||||
test_passphrase_key_is_redacted() {
|
||||
local dir
|
||||
dir="$(new_fixture_passphrase)"
|
||||
|
||||
start_run "$dir" 5 --set '.auth.retries=2'
|
||||
sleep 1
|
||||
printf 'config reloaded\n' >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
|
||||
assert_equals 0 "$RUN_RC" "passphrase case reload exit code"
|
||||
assert_contains "passphrase:" "$RUN_OUTPUT" "passphrase case: the key's line must be in the printed diff"
|
||||
assert_contains "<redacted>" "$RUN_OUTPUT" "passphrase case: redaction must be proven to have run on real content"
|
||||
assert_not_contains "FAKELEAK-PASSPHRASE" "$RUN_OUTPUT" "passphrase case: the passphrase VALUE must never leak"
|
||||
}
|
||||
|
||||
# ----------------------------------- acceptance criterion 16: a failing --set must not echo value
|
||||
# fleetd #635 follow-up (ticket comment 17673, defect 8) — apply_set_pairs used to echo the FULL
|
||||
# "$kv" (path=value, exactly as typed) in its yq-failure messages, so a broken --set with a
|
||||
# secret-looking value printed that value right back out. The path alone is what the positive
|
||||
# control proves is still there — it is what the operator needs to fix their command — and the
|
||||
# negative assertion proves the value itself never appears. Kept to exactly this one failure
|
||||
# shape (an invalid yq path/expression), matching the ticket's own reproduction.
|
||||
test_failing_set_does_not_echo_its_value() {
|
||||
local dir rc=0
|
||||
dir="$(new_fixture)"
|
||||
cp "$dir/fleetd.yaml" "$dir/pre-edit.yaml"
|
||||
|
||||
"$EDIT" --dry-run --set '.broker.["bad=FAKELEAK-SETVALUE' \
|
||||
--config "$dir/fleetd.yaml" --log "$dir/fleetd.out" --wait-seconds 2 \
|
||||
> "$dir/stdout.log" 2>&1 || rc=$?
|
||||
RUN_OUTPUT="$(cat "$dir/stdout.log")"
|
||||
|
||||
[ "$rc" -ne 0 ] || fail "a --set with an invalid yq expression must exit non-zero, got 0"
|
||||
cmp -s "$dir/fleetd.yaml" "$dir/pre-edit.yaml" \
|
||||
|| fail "a failing --set must install nothing — the live fixture changed"
|
||||
# Positive control FIRST: the path must still be in the message, or the negative assertion right
|
||||
# after proves nothing (the message could simply have disappeared entirely).
|
||||
assert_contains '.broker.["bad' "$RUN_OUTPUT" "the failure message must still name the PATH"
|
||||
assert_not_contains "FAKELEAK-SETVALUE" "$RUN_OUTPUT" "the failure message must NEVER echo the VALUE"
|
||||
}
|
||||
|
||||
# dry-run must never touch the live file and must still redact.
|
||||
test_dry_run_never_installs_and_redacts() {
|
||||
local dir before
|
||||
dir="$(new_fixture)"
|
||||
before="$(cat "$dir/fleetd.yaml")"
|
||||
|
||||
"$EDIT" --dry-run --set '.profiles.sonnet.weight=99' \
|
||||
--config "$dir/fleetd.yaml" --log "$dir/fleetd.out" --wait-seconds 2 \
|
||||
> "$dir/stdout.log" 2>&1
|
||||
local rc=$?
|
||||
RUN_OUTPUT="$(cat "$dir/stdout.log")"
|
||||
|
||||
assert_equals 0 "$rc" "dry-run exit code"
|
||||
assert_equals "$before" "$(cat "$dir/fleetd.yaml")" "dry-run must never write the live config"
|
||||
assert_not_contains "hunter2" "$RUN_OUTPUT" "dry-run diff must also be redacted"
|
||||
assert_contains "99" "$RUN_OUTPUT" "dry-run diff must show the candidate value"
|
||||
# Same positive-control reasoning as acceptance criterion 13, applied to the dry-run diff path.
|
||||
assert_contains "<redacted>" "$RUN_OUTPUT" "the dry-run diff's redaction must be PROVEN to have run, not merely absent"
|
||||
}
|
||||
|
||||
# --check is read-only and always exits 0, even against a dead "daemon".
|
||||
test_check_is_read_only_and_exits_zero() {
|
||||
local dir before rc=0
|
||||
dir="$(new_fixture)"
|
||||
before="$(cat "$dir/fleetd.yaml")"
|
||||
|
||||
"$EDIT" --check --config "$dir/fleetd.yaml" --log "$dir/fleetd.out" \
|
||||
> "$dir/stdout.log" 2>&1 || rc=$?
|
||||
|
||||
assert_equals 0 "$rc" "--check exit code"
|
||||
assert_equals "$before" "$(cat "$dir/fleetd.yaml")" "--check must never modify the config"
|
||||
}
|
||||
|
||||
test_refusal_shape_from_parse_failure_wording_is_recognised() {
|
||||
local dir
|
||||
dir="$(new_fixture)"
|
||||
|
||||
start_run "$dir" 5 --set '.profiles.sonnet.weight=6'
|
||||
sleep 1
|
||||
printf 'config reload from %s refused, keeping the running config: boom\n' "$dir/fleetd.yaml" >> "$dir/fleetd.out"
|
||||
collect_run "$dir"
|
||||
|
||||
assert_equals 4 "$RUN_RC" "the parse-failure refusal shape must also exit 4, not be read as silence"
|
||||
}
|
||||
|
||||
echo "== acceptance criterion 1: refusal restores byte for byte =="
|
||||
test_refusal_restores_byte_for_byte
|
||||
echo "== acceptance criterion 2: clean reload keeps the edit =="
|
||||
test_clean_reload_keeps_the_edit
|
||||
echo "== acceptance criterion 3: deferred reload told apart from clean =="
|
||||
test_deferred_reload_is_told_apart_from_clean
|
||||
echo "== acceptance criterion 4: silence is its own answer =="
|
||||
test_silence_is_its_own_answer
|
||||
echo "== acceptance criterion 5: broken candidate never reaches the live path =="
|
||||
test_broken_candidate_never_reaches_live_path
|
||||
echo "== acceptance criterion 6: the marker works =="
|
||||
test_marker_skips_lines_before_it
|
||||
echo "== acceptance criterion 7 (+13: redaction is proven to have run) =="
|
||||
test_redaction_holds
|
||||
echo "== acceptance criterion 9: a forgotten value refuses and installs nothing =="
|
||||
test_forgotten_value_refuses_and_installs_nothing
|
||||
echo "== acceptance criterion 10: an explicit clear writes a bare null =="
|
||||
test_explicit_null_writes_bare_null_not_empty_string
|
||||
echo "== acceptance criterion 11: a backup is never committable =="
|
||||
test_backup_is_never_committable
|
||||
echo "== acceptance criterion 12: the file mode survives an edit and a restore =="
|
||||
test_file_mode_survives_edit_and_restore
|
||||
echo "== acceptance criterion 14: the restore message names the directory actually searched =="
|
||||
test_restore_message_names_the_searched_directory
|
||||
echo "== acceptance criterion 15a: a block scalar's continuation lines are redacted =="
|
||||
test_block_scalar_continuation_is_redacted
|
||||
echo "== acceptance criterion 15b: a passphrase key is also recognised =="
|
||||
test_passphrase_key_is_redacted
|
||||
echo "== acceptance criterion 16: a failing --set must not echo its value =="
|
||||
test_failing_set_does_not_echo_its_value
|
||||
echo "== extra: dry-run never installs, and redacts =="
|
||||
test_dry_run_never_installs_and_redacts
|
||||
echo "== extra: --check is read-only and always exits 0 =="
|
||||
test_check_is_read_only_and_exits_zero
|
||||
echo "== extra: the parse-failure refusal shape is also recognised =="
|
||||
test_refusal_shape_from_parse_failure_wording_is_recognised
|
||||
|
||||
printf 'PASS: config-edit acceptance criteria\n'
|
||||
Reference in New Issue
Block a user