Compare commits

..

1 Commits

Author SHA1 Message Date
Dai Ha 0efe1567c0 CB-586: prune refs/wip/* older than 24h whose tree is reachable from main
CI / build (pull_request) Successful in 1m13s
CI / contract (pull_request) Successful in 1m21s
Add the CB-586 retention rule to GitWorktrees and drive it from the reaper:
a snapshot is deleted only when its tree content is already reachable from
main AND the ref is older than 24h. Reachability keeps the last copy of a
worker's work; the age floor stops a fresh snapshot being swept while a
lead is still looking at it. Every deletion logs the ref name and commit
sha so it is recoverable from the reflog. The /members response gains a
wipRefs{count,costBytes} census the operator can read without shelling
into the repo.
2026-08-16 19:02:33 +02:00
14 changed files with 487 additions and 323 deletions
@@ -83,10 +83,6 @@ public final class Bridged {
static void main(String[] args) {
Path configPath = Path.of(args.length > 0 ? args[0] : "bridged.yaml");
BridgedConfig cfg = BridgedConfig.load(configPath);
// CB-594: report which secret env vars the config actually needs, by name, before anything
// else can fail on a silently-empty one. A daemon started without a login shell (launchd)
// boots fine either way — this is the only thing that says so out loud.
reportRequiredSecrets(cfg);
// CB-559: `cfg` stays the startup snapshot — every validation and every piece of one-time
// wiring below reads it, and must, because those decisions cannot be unmade. `config` is the
// live reference the hot paths read per use. Which keys can actually move is ConfigRef's
@@ -547,63 +543,6 @@ public final class Bridged {
return target -> presence.isPresent(target) || leads.get().containsKey(target);
}
/**
* CB-594: which env vars the loaded config actually needs, and why — every non-{@code
* subscription} profile's {@code tokenEnv} (a subscription profile never reads one, see
* {@link BridgedConfig.Profile#isSubscription()}), plus every profile's {@code gitTokenEnv}
* where set (opt-in). Derived from the config, not hard-coded, so a new profile is covered for
* free. A var required by more than one profile is one entry naming every profile that needs
* it. Deliberately excludes {@code auth.tokenEnv}: that one is already enforced loudly, by a
* startup throw, a few lines above this method's call site.
*
* <p>Package-private and pure (no I/O, no logging) so the derivation is unit-testable without
* capturing log output; {@link #reportRequiredSecrets(BridgedConfig)} is the logging caller.
*/
static Map<String, List<String>> requiredSecretEnvVars(BridgedConfig cfg) {
Map<String, List<String>> requiredBy = new LinkedHashMap<>();
cfg.profiles().forEach((name, profile) -> {
if (!profile.isSubscription()) {
requiredBy.computeIfAbsent(profile.tokenEnv(), _ -> new ArrayList<>())
.add("profile '" + name + "' tokenEnv");
}
if (profile.hasGitToken()) {
requiredBy.computeIfAbsent(profile.gitTokenEnv(), _ -> new ArrayList<>())
.add("profile '" + name + "' gitTokenEnv");
}
});
return requiredBy;
}
/**
* CB-594: log, by name only, which required env vars (see {@link #requiredSecretEnvVars}) are
* set in the daemon's own process environment — the environment every profile's {@code
* tokenEnv}/{@code gitTokenEnv} is read from at spawn time (see
* {@code HerdrPeerLauncher.resolveEnv}). Never logs a value, a prefix, or a length.
*
* <p>A missing entry only warns — it must never refuse to start. A daemon that boots and says
* what is wrong is strictly more useful than one that will not boot at all.
*/
private static void reportRequiredSecrets(BridgedConfig cfg) {
Map<String, List<String>> requiredBy = requiredSecretEnvVars(cfg);
if (requiredBy.isEmpty()) {
log.info("startup secrets: no profile references a token env var — nothing to check");
return;
}
Map<String, String> env = System.getenv();
requiredBy.forEach((varName, sources) -> {
String value = env.get(varName);
if (value != null && !value.isBlank()) {
log.info("startup secret {}: set ({})", varName, String.join(", ", sources));
} else {
log.warn("startup secret {}: MISSING ({}) — the daemon will start anyway, and this "
+ "failure stays invisible until a worker actually needs it. Fix "
+ "${SHARED_ENV}/tools/secrets.sh and restart bridged from a LOGIN "
+ "shell (see scripts/redeploy-bridged.sh).",
varName, String.join(", ", sources));
}
});
}
/**
* Poll herdr's {@code ping} until it answers or {@link #HERDR_WAIT_SECONDS} elapses (CB-504).
*
@@ -234,7 +234,14 @@ public final class BridgedApp {
List<Map<String, Object>> out = sessions.roster().stream()
.map(s -> SessionManager.rosterView(s, live.get(s.terminalId())))
.toList();
ctx.status(200).json(Map.of("workers", out));
Map<String, Object> body = new LinkedHashMap<>();
body.put("workers", out);
// CB-586: operator visibility for the refs/wip snapshot store without shelling into the
// repo — how many snapshot refs exist and roughly what they cost. Present only once a
// worktree session has established the repo, so a never-snapshotted fleet reports nothing.
sessions.wipRefs().ifPresent(st -> body.put("wipRefs",
Map.of("count", st.count(), "costBytes", st.costBytes())));
ctx.status(200).json(body);
}
/** The configured worker profiles and which one a no-argument spawn uses. */
@@ -12,9 +12,12 @@ import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.StandardCopyOption;
import java.security.SecureRandom;
import java.util.ArrayList;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Optional;
import java.util.Set;
import java.util.concurrent.TimeUnit;
import java.util.concurrent.atomic.AtomicLong;
import java.util.stream.Collectors;
@@ -299,6 +302,129 @@ public final class GitWorktrees implements Worktrees {
return index;
}
/**
* One {@code refs/wip/<branch>} snapshot ref as read by {@link #listWipRefs}: its full ref name,
* the snapshot commit's sha, and that commit's committer time in unix millis (the age of the
* snapshot — a snapshot is written once and never rewritten, so the commit date is the ref's).
*/
private record WipRef(String refName, String sha, long committerMillis) {
String branch() {
return refName.substring("refs/wip/".length());
}
}
@Override
public WipRefStats wipRefs(String repoRoot) {
List<WipRef> refs = listWipRefs(repoRoot);
long costBytes = 0;
for (WipRef ref : refs) {
costBytes += treeSize(repoRoot, ref.sha());
}
return new WipRefStats(refs.size(), costBytes);
}
@Override
public int pruneWipRefs(String repoRoot, long minAgeMillis) {
// The rule is documented on Worktrees#pruneWipRefs: delete only a snapshot whose tree
// content is already reachable from main AND that is older than minAgeMillis. Reachability
// is the floor that keeps a worker's last copy; the age floor keeps a just-written snapshot
// from being swept while a lead may still be looking at it.
List<WipRef> refs = listWipRefs(repoRoot);
if (refs.isEmpty()) {
return 0;
}
long nowMillis = System.currentTimeMillis();
// Resolve what main carries once per sweep, not once per ref.
Set<String> mainObjects = reachableObjectsFromMain(repoRoot);
int deleted = 0;
for (WipRef ref : refs) {
long ageMillis = nowMillis - ref.committerMillis();
if (ageMillis <= minAgeMillis) {
continue; // too recent — never swept, even if it looks recoverable (CB-586)
}
String tree = exec("git", "-C", repoRoot, "rev-parse", ref.sha() + "^{tree}").trim();
if (!mainObjects.contains(tree)) {
// Last copy of the snapshot's content — the worker's work exists nowhere else.
// Never delete automatically (CB-586 criterion 2).
continue;
}
exec("git", "-C", repoRoot, "update-ref", "-d", ref.refName());
deleted++;
log.info("pruned snapshot ref refs/wip/{} commit={} (age {}h): its tree is already "
+ "reachable from main, so the work is preserved; recover from reflog via "
+ "git update-ref refs/wip/{} {}",
ref.branch(), ref.sha(), TimeUnit.MILLISECONDS.toHours(ageMillis),
ref.branch(), ref.sha());
}
return deleted;
}
/**
* Every {@code refs/wip/*} ref (see {@link WipRef}). The committer date is read as a unix
* count of seconds and converted to millis. {@code %00} (NUL) separates the fields because a
* branch name may contain spaces.
*/
private List<WipRef> listWipRefs(String repoRoot) {
String out = exec("git", "-C", repoRoot, "for-each-ref",
"--format=%(refname)%00%(objectname)%00%(committerdate:unix)", "refs/wip/");
List<WipRef> refs = new ArrayList<>();
for (String line : out.split("\\R")) {
if (line.isBlank()) {
continue;
}
String[] parts = line.split("\u0000", -1);
if (parts.length == 3 && !parts[1].isBlank()) {
refs.add(new WipRef(parts[0], parts[1], Long.parseLong(parts[2]) * 1000L));
}
}
return refs;
}
/**
* The set of object shas reachable from {@code main}, or an empty set when {@code main} cannot
* be resolved. An empty set is the safe direction: the retention sweep then concludes nothing
* is recoverable, so it deletes nothing — a repo with no {@code main} must never cause a
* worker's last copy of a snapshot to be dropped on a reachability misreading.
*/
private Set<String> reachableObjectsFromMain(String repoRoot) {
if (exitCode("git", "-C", repoRoot, "rev-parse", "--verify", "main") != 0) {
log.debug("refs/wip retention: no 'main' ref in {} — treating nothing as reachable", repoRoot);
return Set.of();
}
String out = exec("git", "-C", repoRoot, "rev-list", "--objects", "main");
Set<String> objects = new HashSet<>();
for (String line : out.split("\\R")) {
if (line.isBlank()) {
continue;
}
int sp = line.indexOf(' ');
objects.add(sp < 0 ? line : line.substring(0, sp));
}
return objects;
}
/** Approximate cost of a snapshot: the sum of every blob's size in its committed tree. */
private long treeSize(String repoRoot, String sha) {
String out = exec("git", "-C", repoRoot, "ls-tree", "-r", "-l", sha);
long total = 0;
for (String line : out.split("\\R")) {
if (line.isBlank()) {
continue;
}
// ls-tree -l row: "<mode> <type> <object> <size>\t<path>"; the size is only numeric for
// blobs (trees read "-"), so gate on the type token and take the 4th whitespace field.
String[] parts = line.split("\\s+");
if (parts.length >= 4 && "blob".equals(parts[1])) {
try {
total += Long.parseLong(parts[3]);
} catch (NumberFormatException ignored) {
// a '-' size (or any anomaly) contributes nothing to the rough figure
}
}
}
return total;
}
/** Resolve the directory that will hold per-session worktree checkouts. */
private Path resolveRoot(String repoRoot) {
if (configuredRoot != null && !configuredRoot.isBlank()) {
@@ -53,6 +53,14 @@ public final class SessionManager implements TurnListener {
private final int contextCap;
private final boolean clearAfterTurn;
private volatile MemberLifecycle memberLifecycle = MemberLifecycle.NONE;
/**
* CB-586: the repo root the fleet actually works in, remembered the first time a worktree
* session is spawned (worktrees are checkouts of it). {@code refs/wip/*} live there, and this
* single cached value is what the snapshot retention sweep and the operator-visible census run
* against. The daemon is bridged into one project at a time, so "the first worktree's repo" is
* the repo; {@code null} until any worktree is spawned, meaning nothing to sweep or measure.
*/
private volatile String fleetRepoRoot;
/** CB-520: notified with a terminalId on every acquire; no-op until wired. */
private final List<Consumer<String>> acquireListeners = new java.util.concurrent.CopyOnWriteArrayList<>();
@@ -450,6 +458,11 @@ public final class SessionManager implements TurnListener {
// The non-worktree path always used this chain; only this branch was missed.
String repoRoot = worktrees.repoRoot(
launcher.effectiveCwd(new SpawnRequest(preResolvedProfile, requestedCwd, callerCwd)));
if (fleetRepoRoot == null) {
// CB-586: remember the repo whose worktrees the fleet spawns — its refs/wip/* are the
// snapshot store the retention sweep and the operator census operate on.
fleetRepoRoot = repoRoot;
}
String branch = "worker/" + slug(wt.ticketSlug()) + "-" + nonce();
String path = null;
PeerHandle handle;
@@ -740,6 +753,26 @@ public final class SessionManager implements TurnListener {
return registry.size();
}
/**
* CB-586: the operator-visible census of {@code refs/wip/*} in the repo the fleet works in —
* how many snapshot refs exist and roughly what they cost. Empty (no repo known) until at
* least one worktree session has been spawned, exactly so a fleet that has never snapshotted
* anything surfaces nothing new, as it did before CB-586.
*/
public Optional<Worktrees.WipRefStats> wipRefs() {
String repo = fleetRepoRoot;
return repo == null ? Optional.empty() : Optional.of(worktrees.wipRefs(repo));
}
/**
* CB-586: run the snapshot retention sweep in the fleet's repo (a no-op until a worktree has
* been spawned, which establishes the repo). Returns how many {@code refs/wip/*} it deleted.
*/
public int sweepWipRefs(long minAgeMillis) {
String repo = fleetRepoRoot;
return repo == null ? 0 : worktrees.pruneWipRefs(repo, minAgeMillis);
}
/**
* The registered session owning {@code terminalId}, or {@code null} if none does.
*
@@ -14,12 +14,20 @@ public final class SessionReaper {
private static final Logger log = LoggerFactory.getLogger(SessionReaper.class);
private static final long DEFAULT_INTERVAL_MILLIS = 5000;
/** CB-586: the refs/wip age floor — never sweep a snapshot younger than 24h (the CB-586 rule). */
private static final long WIP_MIN_AGE_MILLIS = TimeUnit.HOURS.toMillis(24);
/**
* CB-586: how often the retention sweep runs. Given the 24h age floor, running it every few
* hours means a ref is dropped within hours of becoming eligible, never within minutes.
*/
private static final long WIP_SWEEP_INTERVAL_NANOS = TimeUnit.HOURS.toNanos(6);
private final SessionManager sessions;
private final long idleTtlNanos;
private final long intervalMillis;
private volatile boolean running;
private Thread thread;
private volatile long lastWipSweepNanos = Long.MIN_VALUE;
/** Construct a reaper with the default 5-second polling interval. */
public SessionReaper(SessionManager sessions, long idleTtlSeconds) {
@@ -49,10 +57,32 @@ public final class SessionReaper {
} catch (RuntimeException e) {
log.warn("session reaper iteration failed; continuing", e);
}
maybeSweepWipRefs();
sleep();
}
}
/**
* CB-586: run the refs/wip retention sweep on a slow cadence (hours, not the per-iteration
* millisecond loop). Best-effort — a failure must never take the idle-reap loop down with it.
*/
private void maybeSweepWipRefs() {
long now = System.nanoTime();
if (now - lastWipSweepNanos < WIP_SWEEP_INTERVAL_NANOS) {
return;
}
try {
int deleted = sessions.sweepWipRefs(WIP_MIN_AGE_MILLIS);
if (deleted > 0) {
log.info("refs/wip retention sweep deleted {} snapshot ref(s) older than 24h whose "
+ "content was already reachable from main", deleted);
}
} catch (RuntimeException e) {
log.warn("refs/wip retention sweep failed; continuing", e);
}
lastWipSweepNanos = now;
}
private void sleep() {
try {
Thread.sleep(intervalMillis);
@@ -54,4 +54,48 @@ public interface Worktrees {
* tolerance — a worktree that is gone holds nothing to snapshot)
*/
Optional<String> snapshot(String worktreePath, String branch, String message);
/**
* CB-586: how many {@code refs/wip/*} snapshot refs exist in {@code repoRoot} and roughly what
* they cost. This is the operator-visible surface for the snapshot growth CB-578 stage C left
* behind — counts of refs alone hide that each one pins a whole tree for {@code git gc}.
*
* @param repoRoot the repository to scan
* @return count of snapshot refs, and {@code costBytes} = the approximate total working-tree
* size of every snapshot's committed content (summed per ref, so shared objects are
* counted once per ref that carries them)
*/
WipRefStats wipRefs(String repoRoot);
/**
* CB-586: run the {@code refs/wip/*} retention sweep and return how many refs it deleted.
*
* <p>The retention rule is <em>reachability plus an age floor</em>. A snapshot ref is deleted
* only when <strong>both</strong> hold:
* <ol>
* <li>its commit's <em>tree content</em> is already reachable from {@code main} — the work
* the snapshot preserved has been recovered, so dropping the ref loses nothing; and</li>
* <li>the ref is older than {@code minAgeMillis} — a very recent snapshot is never swept
* while a lead may still be looking at it.</li>
* </ol>
*
* <p>Reachability is the safety property. A snapshot exists precisely because the work was not
* committed anywhere else, so a snapshot whose content is <em>not</em> reachable from
* {@code main} is the <strong>last copy</strong> of a worker's work and must never be deleted
* automatically — that is the failure CB-576 and CB-578 stage C were built to stop. Age alone
* must never drive a deletion, because age-based sweeping is exactly how the last copy gets
* destroyed. (Both numbers and the rule are CB-586's decision; this method only implements it.)
*
* <p>Every deletion logs the ref name and the commit sha, so an operator who finds they lost
* the wrong thing can still recover it from git's reflog.
*
* @param repoRoot the repository whose {@code refs/wip/*} to sweep
* @param minAgeMillis the age floor; a ref younger than this is never touched
* @return the number of snapshot refs deleted
*/
int pruneWipRefs(String repoRoot, long minAgeMillis);
/** CB-586: the operator-visible census of {@code refs/wip/*} in one repository. */
record WipRefStats(int count, long costBytes) {
}
}
@@ -1,114 +0,0 @@
package dev.ltms.bridged;
import dev.ltms.bridged.config.BridgedConfig;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.List;
import java.util.Map;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertTrue;
/**
* CB-594: {@link Bridged#requiredSecretEnvVars(BridgedConfig)} is what decides what the startup
* secret report checks — it must derive that set from the config, not a hand-written list, or a
* new profile's token silently stops being reported.
*/
class RequiredSecretEnvVarsTest {
private static BridgedConfig load(Path dir, String yaml) throws Exception {
Path f = dir.resolve("bridged.yaml");
Files.writeString(f, yaml);
return BridgedConfig.load(f);
}
@Test
void collectsATokenEnvPerNonSubscriptionProfile(@TempDir Path dir) throws Exception {
BridgedConfig cfg = load(dir, """
profiles:
local:
baseUrl: http://gx00.gw:8000
tokenEnv: AI_GATEWAY_TOKEN
""");
Map<String, List<String>> required = Bridged.requiredSecretEnvVars(cfg);
assertTrue(required.containsKey("AI_GATEWAY_TOKEN"));
assertEquals(List.of("profile 'local' tokenEnv"), required.get("AI_GATEWAY_TOKEN"));
}
@Test
void aSubscriptionProfileNeedsNoTokenEnv(@TempDir Path dir) throws Exception {
BridgedConfig cfg = load(dir, """
profiles:
opus:
subscription: true
model: claude-opus-5
""");
assertTrue(Bridged.requiredSecretEnvVars(cfg).isEmpty(),
"subscription: true never reads ANTHROPIC_AUTH_TOKEN — see Profile#isSubscription");
}
@Test
void gitTokenEnvIsOptInAndCollectedWhenSet(@TempDir Path dir) throws Exception {
BridgedConfig cfg = load(dir, """
profiles:
local:
baseUrl: http://gx00.gw:8000
tokenEnv: AI_GATEWAY_TOKEN
gitTokenEnv: WORKER_GITEA_TOKEN
""");
Map<String, List<String>> required = Bridged.requiredSecretEnvVars(cfg);
assertTrue(required.containsKey("WORKER_GITEA_TOKEN"));
assertEquals(List.of("profile 'local' gitTokenEnv"), required.get("WORKER_GITEA_TOKEN"));
}
@Test
void noGitTokenEnvMeansNothingIsRequiredForIt(@TempDir Path dir) throws Exception {
BridgedConfig cfg = load(dir, """
profiles:
local:
baseUrl: http://gx00.gw:8000
tokenEnv: AI_GATEWAY_TOKEN
""");
assertFalse(Bridged.requiredSecretEnvVars(cfg).containsKey("WORKER_GITEA_TOKEN"));
}
@Test
void aVarSharedByTwoProfilesIsReportedOnceNamingBoth(@TempDir Path dir) throws Exception {
BridgedConfig cfg = load(dir, """
profiles:
local:
baseUrl: http://gx00.gw:8000
tokenEnv: AI_GATEWAY_TOKEN
gitTokenEnv: WORKER_GITEA_TOKEN
gx:
kind: opencode
baseUrl: https://llm.ltms.dev/v1
tokenEnv: AI_GATEWAY_TOKEN
gitTokenEnv: WORKER_GITEA_TOKEN
""");
Map<String, List<String>> required = Bridged.requiredSecretEnvVars(cfg);
assertEquals(List.of("profile 'local' tokenEnv", "profile 'gx' tokenEnv"),
required.get("AI_GATEWAY_TOKEN"));
assertEquals(List.of("profile 'local' gitTokenEnv", "profile 'gx' gitTokenEnv"),
required.get("WORKER_GITEA_TOKEN"));
}
@Test
void noProfilesMeansNothingIsRequired(@TempDir Path dir) throws Exception {
BridgedConfig cfg = load(dir, "bind:\n host: 127.0.0.1\n port: 8765\n");
assertTrue(Bridged.requiredSecretEnvVars(cfg).isEmpty());
}
}
@@ -27,11 +27,15 @@ public final class FakeWorktrees implements Worktrees {
public record SnapshotCall(String worktreePath, String branch, String message) {
}
public record PruneCall(String repoRoot, long minAgeMillis) {
}
private final List<AddCall> addCalls = new CopyOnWriteArrayList<>();
private final List<RemoveCall> removeCalls = new CopyOnWriteArrayList<>();
private final List<OverlayCall> overlayCalls = new CopyOnWriteArrayList<>();
private final List<RepoRootCall> repoRootCalls = new CopyOnWriteArrayList<>();
private final List<SnapshotCall> snapshotCalls = new CopyOnWriteArrayList<>();
private final List<PruneCall> pruneCalls = new CopyOnWriteArrayList<>();
private final Set<String> existingPaths = ConcurrentHashMap.newKeySet();
private final Set<String> trackedPaths = ConcurrentHashMap.newKeySet();
private final AtomicLong snapshotSeq = new AtomicLong();
@@ -40,6 +44,8 @@ public final class FakeWorktrees implements Worktrees {
private volatile boolean dirty = false;
private volatile String repoRoot = "/repo";
private volatile String prefix = "/worktrees";
private volatile WipRefStats wipRefs = new WipRefStats(0, 0L);
private volatile int pruneResult = 0;
public FakeWorktrees withRepoRoot(String root) {
this.repoRoot = root;
@@ -82,6 +88,18 @@ public final class FakeWorktrees implements Worktrees {
return this;
}
/** Configure the value returned by {@link #wipRefs}. */
public FakeWorktrees withWipRefs(WipRefStats stats) {
this.wipRefs = stats;
return this;
}
/** Configure the value returned by {@link #pruneWipRefs}. */
public FakeWorktrees withPruneResult(int deleted) {
this.pruneResult = deleted;
return this;
}
@Override
public String add(String repoRoot, String branch, String baseRef) {
addCalls.add(new AddCall(repoRoot, branch, baseRef));
@@ -135,6 +153,17 @@ public final class FakeWorktrees implements Worktrees {
return Optional.of("wip" + snapshotSeq.incrementAndGet());
}
@Override
public WipRefStats wipRefs(String repoRoot) {
return wipRefs;
}
@Override
public int pruneWipRefs(String repoRoot, long minAgeMillis) {
pruneCalls.add(new PruneCall(repoRoot, minAgeMillis));
return pruneResult;
}
public List<AddCall> addCalls() {
return List.copyOf(addCalls);
}
@@ -167,6 +196,10 @@ public final class FakeWorktrees implements Worktrees {
return List.copyOf(snapshotCalls);
}
public List<PruneCall> pruneCalls() {
return List.copyOf(pruneCalls);
}
public SnapshotCall lastSnapshot() {
return snapshotCalls.isEmpty() ? null : snapshotCalls.getLast();
}
@@ -3,6 +3,7 @@ package dev.ltms.bridged.session;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.HashSet;
@@ -138,6 +139,53 @@ class GitWorktreesTest {
return out;
}
/** Write {@code content} as a blob into the object database; returns its sha. */
private static String blobOf(Path cwd, String content) throws Exception {
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "hash-object", "-w", "--stdin")
.redirectErrorStream(true).start();
p.getOutputStream().write(content.getBytes(StandardCharsets.UTF_8));
p.getOutputStream().close();
String out = new String(p.getInputStream().readAllBytes()).trim();
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git hash-object timed out");
assertEquals(0, p.exitValue(), "git hash-object failed:\n" + out);
return out;
}
/** Build a single-file tree object from {@code blob}; returns the tree's sha. */
private static String treeOf(Path cwd, String path, String blob) throws Exception {
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "mktree")
.redirectErrorStream(true).start();
p.getOutputStream().write(("100644 blob " + blob + "\t" + path + "\n").getBytes(StandardCharsets.UTF_8));
p.getOutputStream().close();
String out = new String(p.getInputStream().readAllBytes()).trim();
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git mktree timed out");
assertEquals(0, p.exitValue(), "git mktree failed:\n" + out);
return out;
}
/** {@code git commit-tree} rooted at {@code tree} with a chosen committer date; returns the sha. */
private static String commitTree(Path cwd, String tree, String parent, String committerDate,
String message) throws Exception {
ProcessBuilder pb = new ProcessBuilder("git", "-C", cwd.toString(), "commit-tree",
tree, "-p", parent, "-m", message);
pb.environment().put("GIT_COMMITTER_DATE", committerDate);
Process p = pb.redirectErrorStream(true).start();
String out = new String(p.getInputStream().readAllBytes()).trim();
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git commit-tree timed out");
assertEquals(0, p.exitValue(), "git commit-tree failed:\n" + out);
return out;
}
/** {@code git update-ref <ref> <sha>} — create the snapshot ref directly. */
private static void updateRef(Path cwd, String ref, String sha) throws Exception {
git(cwd, "update-ref", ref, sha);
}
/** True when {@code ref} exists in the repo (for-each-ref on a missing ref is empty, not an error). */
private static boolean refExists(Path cwd, String ref) throws Exception {
return !forEachRef(cwd, ref).trim().isEmpty();
}
/**
* The heart of CB-525: a provisioned worktree must not inherit the primary's MCP servers. Without
* the isolation step the checked-out {@code .mcp.json} carries them in, and a worker navigating
@@ -474,4 +522,104 @@ class GitWorktreesTest {
assertThrows(WorktreeException.class, () -> gitWorktrees.snapshot(wt, branch, "test snapshot"),
"an unresolvable real index must fail loudly, not silently snapshot from an empty index");
}
/**
* CB-586, criterion 2. A snapshot whose content is NOT reachable from {@code main} is the last
* copy of a worker's work, and must never be deleted automatically — even when it is old and
* even when the caller passes a zero age floor. Uses the real snapshot path on a dirty worktree,
* so the unreachable tree is exactly the shape CB-576/CB-578 stage C exist to protect.
*/
@Test
void anUnreachableSnapshotIsNeverPruned(@TempDir Path tmp) throws Exception {
Path repo = initRepo(tmp.resolve("repo"));
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
String branch = "cb-586-unreachable";
String wt = gitWorktrees.add(repo.toString(), branch, "HEAD");
Files.writeString(Path.of(wt).resolve("worker-draft.txt"), "work that exists nowhere else\n");
Optional<String> ref = gitWorktrees.snapshot(wt, branch, "snapshot with unreachable content");
assertTrue(ref.isPresent());
// Age floor 0 makes age a non-issue: only reachability can save it — and it must.
assertEquals(0, gitWorktrees.pruneWipRefs(repo.toString(), 0),
"the unreachable snapshot is the last copy and must not be pruned");
assertTrue(refExists(repo, "refs/wip/" + branch),
"an unreachable snapshot must survive the sweep");
}
/**
* CB-586, criterion 1 (the reachable half). A snapshot whose tree content IS already reachable
* from {@code main} and which is older than the age floor is pure duplication — the work is
* recovered — so it must be pruned.
*/
@Test
void aReachableSnapshotOlderThanTheFloorIsPruned(@TempDir Path tmp) throws Exception {
Path repo = initRepo(tmp.resolve("repo"));
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
// A snapshot whose tree is exactly main's current tree: fully reachable from main.
String mainTree = revParse(repo, "main^{tree}");
String old = commitTree(repo, mainTree, revParse(repo, "HEAD"), "2020-01-01T00:00:00", "snapshot");
updateRef(repo, "refs/wip/recovered", old);
assertEquals(1, gitWorktrees.pruneWipRefs(repo.toString(), TimeUnit.HOURS.toMillis(24)),
"an old, main-reachable snapshot must be pruned");
assertFalse(refExists(repo, "refs/wip/recovered"),
"the reachable snapshot's ref must be gone after the sweep");
}
/**
* CB-586, the age floor. A snapshot whose content IS reachable from {@code main} but which is
* younger than the age floor must not be swept — a lead may still be looking at it.
*/
@Test
void aReachableButRecentSnapshotIsNotPruned(@TempDir Path tmp) throws Exception {
Path repo = initRepo(tmp.resolve("repo"));
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
// Reachable from main, but committed "now" — a fresh snapshot. The 24h floor must protect it.
String mainTree = revParse(repo, "main^{tree}");
String fresh = commitTree(repo, mainTree, revParse(repo, "HEAD"),
"2038-01-01T00:00:00", "snapshot just taken");
updateRef(repo, "refs/wip/fresh", fresh);
assertEquals(0, gitWorktrees.pruneWipRefs(repo.toString(), TimeUnit.HOURS.toMillis(24)),
"a recent snapshot must be kept even when reachable");
assertTrue(refExists(repo, "refs/wip/fresh"),
"the recent reachable snapshot must survive the sweep");
}
/**
* CB-586, criterion 5. A fleet that has never snapshotted anything has no {@code refs/wip/*},
* so a sweep is a no-op and the census reports none — identical to before CB-586 existed.
*/
@Test
void aFleetWithNoSnapshotsPrunesNothingAndReportsNothing(@TempDir Path tmp) throws Exception {
Path repo = initRepo(tmp.resolve("repo"));
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
assertEquals(0, gitWorktrees.pruneWipRefs(repo.toString(), 0),
"no snapshot refs means nothing to prune");
Worktrees.WipRefStats stats = gitWorktrees.wipRefs(repo.toString());
assertEquals(0, stats.count(), "a never-snapshotted fleet has zero refs/wip refs");
assertEquals(0L, stats.costBytes(), "a never-snapshotted fleet costs zero bytes");
}
/** CB-586, criterion 4: the census reports how many refs exist and roughly what they cost. */
@Test
void wipRefsReportsCountAndCost(@TempDir Path tmp) throws Exception {
Path repo = initRepo(tmp.resolve("repo"));
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
String blob = blobOf(repo, "a recoverable snapshot's worth of content");
String tree = treeOf(repo, "snapshot.txt", blob);
updateRef(repo, "refs/wip/one", commitTree(repo, tree, revParse(repo, "HEAD"),
"2020-01-01T00:00:00", "snapshot"));
updateRef(repo, "refs/wip/two", commitTree(repo, tree, revParse(repo, "HEAD"),
"2020-01-02T00:00:00", "snapshot"));
Worktrees.WipRefStats stats = gitWorktrees.wipRefs(repo.toString());
assertEquals(2, stats.count(), "two snapshot refs are reported");
assertTrue(stats.costBytes() > 0, "the cost of the snapshots is a positive byte count");
}
}
@@ -138,6 +138,16 @@ class SessionManagerTest {
return java.util.Optional.of("wip" + snapshotSeq.incrementAndGet());
}
@Override
public WipRefStats wipRefs(String repoRoot) {
return new WipRefStats(0, 0L);
}
@Override
public int pruneWipRefs(String repoRoot, long minAgeMillis) {
return 0;
}
List<String> removeCalls() {
return List.copyOf(removeCalls);
}
@@ -444,4 +444,37 @@ class WorktreeSessionManagerTest {
"a failed snapshot leaves no ref to report");
}
/**
* CB-586, criteria 4 and 1. Once a worktree session is spawned the repo is known, so the
* operator census and the retention sweep delegate to that repo's {@code refs/wip/*}.
*/
@Test
void wipRefsAndSweepDelegateToTheFleetRepoOnceKnown() {
FakeHerdr herdr = new FakeHerdr();
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt")
.withWipRefs(new Worktrees.WipRefStats(3, 42L)).withPruneResult(2);
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
assertTrue(sessions.wipRefs().isEmpty(),
"no worktree spawned yet means no repo is known and nothing to report");
assertEquals(0, sessions.sweepWipRefs(TimeUnit.HOURS.toMillis(24)),
"no worktree spawned yet means the sweep is a no-op");
sessions.acquire("ltms-local", null, "/caller/proj", null,
new WorktreeRequest("cb-586", null));
Worktrees.WipRefStats stats = sessions.wipRefs().orElseThrow();
assertEquals(3, stats.count(), "the census comes from the fleet repo");
assertEquals(42L, stats.costBytes(), "the cost comes from the fleet repo");
assertEquals(2, sessions.sweepWipRefs(TimeUnit.HOURS.toMillis(24)),
"the sweep runs against the fleet repo");
List<FakeWorktrees.PruneCall> prunes = worktrees.pruneCalls();
assertEquals(1, prunes.size(), "the no-op short-circuits before reaching the seam, so only "
+ "the repo-known sweep issues a call");
assertEquals("/repo", prunes.getFirst().repoRoot(), "the sweep targets the fleet repo");
assertEquals(TimeUnit.HOURS.toMillis(24), prunes.getFirst().minAgeMillis(),
"the caller's age floor is passed through");
}
}
+13 -43
View File
@@ -1,7 +1,7 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
<!--
CB-504 / CB-594 — launchd agent for bridged (macOS).
CB-504 — launchd agent for bridged (macOS).
This is the real supervision target today: the dogfooded daemon runs on macOS, where there is
no systemd. A systemd unit ships alongside (deploy/bridged.service) for the Linux gateways
@@ -9,33 +9,14 @@
Install:
cp deploy/dev.ltms.bridged.plist ~/Library/LaunchAgents/
# edit the paths + JAVA_HOME below to match this host, then:
launchctl load -w ~/Library/LaunchAgents/dev.ltms.bridged.plist
launchctl list | grep bridged
The paths below are already filled in for this host (resolved 2026-08-16 from
`/usr/libexec/java_home`... except that reported the system Applet-plugin JVM, not the jenv-
managed JDK 25 actually used to build/run bridged, so JAVA_HOME here is the real one:
`JENV_VERSION=25.0.3 java -XshowSettings:properties -version 2>&1 | grep java.home`; `which mvn`;
`echo $HOME`). If this file is copied to a different host, re-resolve all three paths and check
no placeholder path is left behind; scripts/redeploy-bridged.sh's check mode does not (and
cannot) check this file for you.
CB-594 — launchd cannot run a login shell (see the PATH comment on EnvironmentVariables below,
and scripts/bridged-launchd-wrapper.sh for the fix): ProgramArguments below execs THAT wrapper,
not java directly, so WORKER_GITEA_TOKEN and AI_GATEWAY_TOKEN still get sourced from
${SHARED_ENV}/tools/secrets.sh even though launchd itself never sources anything.
Note on ordering: launchd has no "start after herdr" primitive for user agents, and neither
does systemd in a way that survives a socket appearing late. bridged retries the herdr socket
on startup instead, so an agent that comes up before herdr converges rather than dying — that
retry is the actual fix; KeepAlive below is the backstop.
CB-594 — KeepAlive vs. scripts/redeploy-bridged.sh: a bare SIGTERM makes this JVM exit 143 even
with its shutdown hook running to completion (measured, see the CB-594 report), which
SuccessfulExit:false below reads as a crash and races to restart the OLD jar. The redeploy
script now detects a loaded agent and uses `launchctl unload`/`load` instead of a raw kill, so
only one supervisor ever touches the process at a time — read that script's own output on a
redeploy for the confirmation.
-->
<plist version="1.0">
<dict>
@@ -44,23 +25,22 @@
<key>ProgramArguments</key>
<array>
<string>/Users/dai.ha/LTMS/claude-bridge/scripts/bridged-launchd-wrapper.sh</string>
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin/java</string>
<string>/Users/CHANGEME/Tool/jdk-25.0.2.jdk/Contents/Home/bin/java</string>
<string>-jar</string>
<string>/Users/dai.ha/LTMS/claude-bridge/bridged/target/bridged.jar</string>
<string>/Users/CHANGEME/src/claude-bridge/bridged/target/bridged.jar</string>
<string>bridged.yaml</string>
</array>
<!-- Config path in ProgramArguments is relative, so the working directory must be the module. -->
<key>WorkingDirectory</key>
<string>/Users/dai.ha/LTMS/claude-bridge/bridged</string>
<string>/Users/CHANGEME/src/claude-bridge/bridged</string>
<key>EnvironmentVariables</key>
<dict>
<key>JAVA_HOME</key>
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home</string>
<string>/Users/CHANGEME/Tool/jdk-25.0.2.jdk/Contents/Home</string>
<key>HERDR_SOCKET_PATH</key>
<string>/Users/dai.ha/.config/herdr/herdr.sock</string>
<string>/Users/CHANGEME/.config/herdr/herdr.sock</string>
<!--
PATH matters more than it looks (CB-511): bridged propagates its own PATH to every worker
it spawns, so this line decides whether the fleet can run a build at all. launchd does NOT
@@ -68,14 +48,11 @@
bare /usr/bin:/bin and no JDK or Maven. Keep the toolchain entries first.
-->
<key>PATH</key>
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin:/Users/dai.ha/Softwares/apache-maven/bin:/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin</string>
<string>/Users/CHANGEME/Tool/jdk-25.0.2.jdk/Contents/Home/bin:/Users/CHANGEME/Tool/apache-maven-3.9.16/bin:/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin</string>
<!--
Worker/API tokens are NOT set here: this file is committed. CB-594 —
scripts/bridged-launchd-wrapper.sh (named in ProgramArguments above) is what supplies
them, by execing a login shell that sources ${SHARED_ENV}/tools/secrets.sh before the
daemon itself starts. bridged also reads the API token from the env var named by
auth.tokenEnv (default BRIDGED_API_TOKEN) and only in auth.mode: token — the wrapper
covers that one too, since it is the same login shell.
Worker/API tokens are NOT set here: this file is committed. Export them from a private
launchd override or a wrapper script. bridged reads the API token from the env var named
by auth.tokenEnv (default BRIDGED_API_TOKEN) and only in auth.mode: token.
-->
</dict>
@@ -92,17 +69,10 @@
<key>ThrottleInterval</key>
<integer>10</integer>
<!--
CB-594 — same file scripts/redeploy-bridged.sh already tails ($BRIDGED/bridged.out), and both
streams point at it, not two separate log files: the script's fresh-line / ERROR-count checks
after a restart read this one path regardless of whether launchd or the script started the
process, and a stdout/stderr split would make half of what happens during a launchd-driven
restart invisible to it.
-->
<key>StandardOutPath</key>
<string>/Users/dai.ha/LTMS/claude-bridge/bridged/bridged.out</string>
<string>/Users/CHANGEME/src/claude-bridge/bridged/logs/bridged.out.log</string>
<key>StandardErrorPath</key>
<string>/Users/dai.ha/LTMS/claude-bridge/bridged/bridged.out</string>
<string>/Users/CHANGEME/src/claude-bridge/bridged/logs/bridged.err.log</string>
<key>ProcessType</key>
<string>Background</string>
-32
View File
@@ -1,32 +0,0 @@
#!/usr/bin/env bash
#
# CB-594 — the only reason this file exists: launchd does not run a login shell.
#
# WORKER_GITEA_TOKEN and AI_GATEWAY_TOKEN live in ${SHARED_ENV}/tools/secrets.sh, sourced only by a
# LOGIN shell (.zprofile/.zshrc etc). launchd execs a job's ProgramArguments directly — no shell, no
# profile, nothing sourced (the plist's own PATH comment documents the same gap one variable over).
# A daemon started that way boots fine and looks healthy; the failure is invisible until a worker
# tries to open a PR (WORKER_GITEA_TOKEN empty) or a gateway profile gets a 401 (AI_GATEWAY_TOKEN
# empty) — hours later, with nothing tying the two together (CB-591, CLAUDE.md "Redeploying the
# daemon"). Bridged now also logs which required secret names resolved at startup (see
# Bridged.reportRequiredSecrets), but that log line can only tell the truth if the tokens had a
# chance to be sourced in the first place — which is this script's entire job.
#
# So: launchd execs THIS script instead of java directly. This script execs a login shell
# ('zsh -l'), which sources secrets.sh, and that shell execs the real command in its place — one
# process throughout (exec, not a subshell fork), so launchd's PID tracking, KeepAlive, and
# StandardOut/ErrorPath all still see the one process they expect.
#
# The plist passes the full command as THIS script's own arguments, e.g.:
# ProgramArguments = [ .../bridged-launchd-wrapper.sh, /path/to/java, -jar, /path/to/bridged.jar,
# bridged.yaml ]
# so the wrapper stays generic and the actual command lives in exactly one place (the plist), not
# duplicated here.
set -euo pipefail
if [ "$#" -eq 0 ]; then
echo "bridged-launchd-wrapper.sh: no command given — check the plist's ProgramArguments" >&2
exit 2
fi
exec /bin/zsh -lc 'exec "$@"' -- "$@"
+9 -72
View File
@@ -5,14 +5,13 @@
# A merge is not a deployment: the running daemon holds the jar it was started with, so code merged
# to main does nothing until this runs. See CLAUDE.md -> "Redeploying the daemon".
#
# This script exists to turn six remembered traps into one auditable command:
# This script exists to turn five remembered traps into one auditable command:
#
# 1. A piped `mvn` hides BUILD FAILURE behind a zero exit, so the build here is never piped.
# 2. The daemon must start from a LOGIN shell, or the tokens it hands to members are empty:
# WORKER_GITEA_TOKEN (workers cannot open a PR) and AI_GATEWAY_TOKEN (401 at llm.ltms.dev).
# Both are read from the DAEMON's own environment at spawn time, so a value added to
# secrets.sh after startup is absent. Nothing logs this here, so the script checks and says
# so — and since CB-594, bridged's own startup log says so too, by env var name.
# secrets.sh after startup is absent. Nothing logs this, so the script checks and says so.
# 3. An old daemon that never actually died looks identical from the outside, so the script waits
# for the process to exit and for the port to free before it starts a new one.
# 4. "It started" is not "it works": the script polls /healthz until it answers, and reports the
@@ -20,13 +19,6 @@
# mismatch.
# 5. Restarting under live members drops their tickets, so the script refuses unless you confirm
# the fleet is drained.
# 6. CB-594 — the launchd agent (deploy/dev.ltms.bridged.plist), if installed and loaded, is a
# SECOND supervisor: its KeepAlive.SuccessfulExit=false restarts the daemon on any nonzero
# exit, and a bare SIGTERM makes this JVM exit 143 even with its shutdown hook running to
# completion (measured — see the CB-594 report). A plain `kill` here would race launchd's own
# restart of the OLD jar. So this script detects whether the agent is loaded and, only then,
# swaps `kill` + manual `nohup` for `launchctl unload`/`load` — the one supervisor in control
# at any moment is whichever one you asked to act, never both.
#
# Usage:
# scripts/redeploy-bridged.sh # build, confirm, restart, verify
@@ -51,17 +43,13 @@ HEALTH='http://127.0.0.1:8765/healthz'
STOP_WAIT=30 # seconds to wait for a clean exit before reporting failure
HEALTH_WAIT=60 # seconds to wait for /healthz to answer after start
# CB-594: the launchd agent this script must not fight with (see trap 6 above).
LAUNCHD_LABEL='dev.ltms.bridged'
LAUNCHD_PLIST="$HOME/Library/LaunchAgents/$LAUNCHD_LABEL.plist"
DO_BUILD=1; ASSUME_YES=0; CHECK_ONLY=0
for arg in "$@"; do
case "$arg" in
--yes|-y) ASSUME_YES=1 ;;
--no-build) DO_BUILD=0 ;;
--check) CHECK_ONLY=1 ;;
-h|--help) sed -n '3,37p' "${BASH_SOURCE[0]}"; exit 0 ;;
-h|--help) sed -n '3,30p' "${BASH_SOURCE[0]}"; exit 0 ;;
*) echo "unknown option: $arg (try --help)" >&2; exit 2 ;;
esac
done
@@ -73,11 +61,6 @@ die() { printf '\n FAIL %s\n\n' "$*" >&2; exit 1; }
jar_id() { [ -f "$JAR" ] && shasum -a 256 "$JAR" | cut -c1-12 || echo "absent"; }
running_pid() { pgrep -f "$PATTERN" || true; }
# `launchctl list <label>` exits 0 iff the label is loaded (registered with launchd) — true whether
# or not it is currently running, which is exactly "supervision is active" for our purposes. Read-
# only: neither helper below changes anything, so both are also safe under --check.
launchd_installed() { [ -f "$LAUNCHD_PLIST" ]; }
launchd_loaded() { launchctl list "$LAUNCHD_LABEL" >/dev/null 2>&1; }
# ---------------------------------------------------------------- report state
@@ -91,22 +74,6 @@ fi
ok "jar on disk: $(jar_id) ($([ -f "$JAR" ] && date -r "$JAR" '+%Y-%m-%d %H:%M:%S' || echo 'none'))"
ok "HEAD: $(git -C "$REPO" log --oneline -1)"
# CB-594: supervision state. Installed and loaded are different facts — a copied-but-never-loaded
# plist supervises nothing, and a loaded label with no file backing it (rare, but possible after an
# edited/moved plist) is still what launchd will act on.
if launchd_installed; then
ok "launchd agent installed: $LAUNCHD_PLIST"
else
warn "launchd agent NOT installed (no supervision — a crash will not restart the daemon)."
fi
SUPERVISED=0
if launchd_loaded; then
SUPERVISED=1
ok "launchd agent loaded ($LAUNCHD_LABEL) — launchd supervises this daemon"
else
warn "launchd agent not loaded — this script is the only thing that will restart the daemon."
fi
# The trap with no log line. Checked in a LOGIN shell, because that is how the daemon is started
# below. Never prints the value — only whether it resolved.
if zsh -lc '[ -n "${WORKER_GITEA_TOKEN:-}" ]' 2>/dev/null; then
@@ -170,26 +137,11 @@ if [ -n "$OLD_PID" ] && [ "$ASSUME_YES" = 0 ]; then
fi
# ------------------------------------------------------------------ stop
#
# CB-594: when SUPERVISED, launchd owns the stop — never a raw `kill` here. A bare SIGTERM makes
# this JVM exit 143 even with its shutdown hook running to completion (verified separately: a
# throwaway Java process with an equivalent shutdown hook, sent SIGTERM from a login shell that
# could `wait` on it directly, reported exit code 143 every time — never 0). launchd's
# KeepAlive.SuccessfulExit=false treats any nonzero exit as a crash and restarts the OLD jar,
# which would race this script's own restart of the NEW one. `launchctl unload` avoids that race
# by deregistering the job first, so no KeepAlive is left armed when the process actually stops.
if [ -n "$OLD_PID" ]; then
say "stop"
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)" # verify a FRESH line appears later
if [ "$SUPERVISED" = 1 ]; then
echo " supervision is ON: using 'launchctl unload' (not kill) so launchd's own KeepAlive"
echo " cannot restart the OLD jar out from under this script — see the CB-594 comment above."
launchctl unload -w "$LAUNCHD_PLIST" \
|| die "launchctl unload failed — the daemon may still be under supervision; investigate before retrying"
else
kill "$OLD_PID"
fi
kill "$OLD_PID"
for _ in $(seq "$STOP_WAIT"); do
[ -z "$(running_pid)" ] && break
sleep 1
@@ -200,33 +152,18 @@ if [ -n "$OLD_PID" ]; then
leave worktrees and panes behind. Investigate, then kill -9 by hand if you accept that."
fi
ok "pid $OLD_PID exited"
elif [ "$SUPERVISED" = 1 ]; then
# Loaded but not currently running (e.g. throttled after a crash loop). Unload it anyway so the
# start step below does a clean load, never a load stacked on an already-loaded label.
say "stop"
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
launchctl unload -w "$LAUNCHD_PLIST" 2>/dev/null || true
ok "launchd agent unloaded (was already not running)"
else
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
fi
# ------------------------------------------------------------------ start
# Unsupervised: login shell (zsh -l) is what puts the secrets on the daemon's environment, and cwd
# must be bridged/ because the daemon resolves bridged.yaml, logs/ and target/ relative to it.
# Supervised: launchd does both — deploy/dev.ltms.bridged.plist points ProgramArguments at
# scripts/bridged-launchd-wrapper.sh (CB-594), which is what execs the login shell in launchd's
# place, and WorkingDirectory in the plist already pins bridged/.
# Login shell (zsh -l) is what puts the secrets on the daemon's environment. cwd must be bridged/
# because the daemon resolves bridged.yaml, logs/ and target/ relative to it.
say "start"
if [ "$SUPERVISED" = 1 ]; then
echo " supervision is ON: using 'launchctl load' so launchd starts and keeps supervising this"
echo " process, instead of a manual nohup that launchd would know nothing about."
launchctl load -w "$LAUNCHD_PLIST" || die "launchctl load failed"
else
# Absolute jar path so `ps` names which checkout is running.
( cd "$BRIDGED" && zsh -lc "nohup java -jar '$JAR' >> bridged.out 2>&1 &" )
fi
# Absolute jar path so `ps` names which checkout is running; cwd still bridged/ because the daemon
# resolves bridged.yaml, logs/ and target/ relative to it.
( cd "$BRIDGED" && zsh -lc "nohup java -jar '$JAR' >> bridged.out 2>&1 &" )
for _ in $(seq 10); do
NEW_PID="$(running_pid)"