a2108a8a14
Tuning a fleet meant restarting bridged, and a restart tears down every lead
and worker it owns. Changing one pool's weight cost the whole fleet's state,
so in practice nobody changed it.
ConfigRef holds the live BridgedConfig in an AtomicReference. Consumers read
it at the point of use, so a change reaches the next spawn with nothing
rebuilt. The launchers that used to capture config into fields now take
suppliers: the fleet tabLabel template, the profile map, the placement policy
and the fleet block.
Keys fall into three classes, and the difference is what already exists when
the reload happens:
hot fleet: (pools + tabLabel), placement:, and an existing profile's
weight / maxLoad / model / tabLabel — live on the next spawn.
deferred lifecycle:, leadHeartbeat:, guard:, worktreeRoot:, spawnReady*,
and adding/removing a profile — accepted, but the startup wiring
keeps the old value. The reload logs these by name.
cold bind:, herdrSocket:, broker:, auth: — refuses the WHOLE reload.
A cold change refuses everything rather than applying the hot half. A
half-applied reload leaves the daemon matching no file on disk, which is the
worst thing a reload can do to an operator reading that file to work out what
the daemon is doing. Refusing keeps the invariant that the live config is
always some version of the file.
A parse failure or a failed startup validator is refused the same way, and
the running config stays live: a file being saved is sometimes read
mid-write, and degrading a working daemon over a half-written file is a bad
trade. The same four validators startup runs are re-run, so a config that
could not have booted cannot slip in through a reload.
ConfigWatcher polls the modified time on a daemon thread, opt-in through
configReload.enabled (default off, so an upgraded daemon is unchanged). It
stamps the timestamp BEFORE reloading, so a refused file is not retried every
tick — the next save earns a fresh attempt. A missing file is skipped
silently, because editors unlink briefly mid-save.
MicroProfile Config was the first idea and does not fit: @ConfigMapping needs
interfaces, resolves once at bootstrap, and reload would still mean rebuild
and swap. The port would also lose the raw-YAML duplicate-key detection,
since duplicates have already collapsed once the tree is flattened to
properties.
634 tests.
97 lines
3.9 KiB
Java
97 lines
3.9 KiB
Java
package dev.ltms.bridged.config;
|
|
|
|
import org.slf4j.Logger;
|
|
import org.slf4j.LoggerFactory;
|
|
|
|
import java.io.IOException;
|
|
import java.nio.file.Files;
|
|
import java.nio.file.Path;
|
|
import java.util.concurrent.Executors;
|
|
import java.util.concurrent.ScheduledExecutorService;
|
|
import java.util.concurrent.TimeUnit;
|
|
|
|
/**
|
|
* Polls {@code bridged.yaml}'s modified time and asks {@link ConfigRef} to reload when it moves
|
|
* (CB-559). Opt-in through {@code configReload.enabled}.
|
|
*
|
|
* <p><strong>Why polling and not a filesystem watch.</strong> {@code WatchService} on macOS has no
|
|
* native backend — it falls back to polling internally anyway, at an interval this code does not
|
|
* control — and editors save config files in ways that produce a different event mix per editor
|
|
* (write-in-place, write-and-rename, write-temp-and-swap). A modified-time check treats all of them
|
|
* the same and is a single {@code stat} per tick, which at a ten-second cadence costs nothing worth
|
|
* measuring.
|
|
*
|
|
* <p><strong>A missing or unreadable file is not a reason to act.</strong> Many editors briefly
|
|
* unlink the file during a save. Reloading on "it vanished" would mean reloading from a file that no
|
|
* longer exists; reporting an error every tick would bury the log. So an unreadable file is skipped
|
|
* silently and the next tick tries again — the running config stays live, which is the correct
|
|
* outcome either way.
|
|
*/
|
|
public final class ConfigWatcher {
|
|
|
|
private static final Logger log = LoggerFactory.getLogger(ConfigWatcher.class);
|
|
|
|
private final ConfigRef ref;
|
|
private final long intervalSeconds;
|
|
private final ScheduledExecutorService scheduler;
|
|
|
|
private volatile long lastSeenMillis;
|
|
|
|
public ConfigWatcher(ConfigRef ref, long intervalSeconds) {
|
|
this.ref = ref;
|
|
this.intervalSeconds = intervalSeconds;
|
|
this.lastSeenMillis = modifiedMillis(ref.path());
|
|
this.scheduler = Executors.newSingleThreadScheduledExecutor(r -> {
|
|
Thread t = new Thread(r, "config-watcher");
|
|
// A daemon thread: an operator's config watch must never be the reason the JVM refuses
|
|
// to exit after everything else has shut down.
|
|
t.setDaemon(true);
|
|
return t;
|
|
});
|
|
}
|
|
|
|
/** Begin watching. A ref with no file (a fixed one) is a no-op rather than an error. */
|
|
public void start() {
|
|
if (ref.path() == null) {
|
|
log.debug("config watch not started — this config has no file behind it");
|
|
return;
|
|
}
|
|
scheduler.scheduleWithFixedDelay(this::tick, intervalSeconds, intervalSeconds,
|
|
TimeUnit.SECONDS);
|
|
log.info("config watch: {} re-read when it changes (every {}s)", ref.path(), intervalSeconds);
|
|
}
|
|
|
|
/** One poll. Never throws — an exception here would silently cancel the schedule. */
|
|
void tick() {
|
|
try {
|
|
long now = modifiedMillis(ref.path());
|
|
if (now == 0 || now == lastSeenMillis) {
|
|
return;
|
|
}
|
|
// Stamp BEFORE reloading. A file whose reload is refused (a bad edit, or a cold key)
|
|
// must not be retried every tick — that would log the same refusal forever. The next
|
|
// save moves the timestamp again and earns a fresh attempt.
|
|
lastSeenMillis = now;
|
|
ref.reload();
|
|
} catch (RuntimeException e) {
|
|
log.warn("config watch tick failed, still watching: {}", e.getMessage());
|
|
}
|
|
}
|
|
|
|
private static long modifiedMillis(Path path) {
|
|
if (path == null) {
|
|
return 0;
|
|
}
|
|
try {
|
|
return Files.getLastModifiedTime(path).toMillis();
|
|
} catch (IOException e) {
|
|
return 0; // mid-save, or gone: say nothing and try again next tick
|
|
}
|
|
}
|
|
|
|
/** Stop polling. Called from the daemon's ordered shutdown hook, alongside the other loops. */
|
|
public void stop() {
|
|
scheduler.shutdownNow();
|
|
}
|
|
}
|