diff --git a/fleetd/src/main/java/dev/ltms/fleet/lead/LeadContextGauge.java b/fleetd/src/main/java/dev/ltms/fleet/lead/LeadContextGauge.java
new file mode 100644
index 0000000..8b11260
--- /dev/null
+++ b/fleetd/src/main/java/dev/ltms/fleet/lead/LeadContextGauge.java
@@ -0,0 +1,295 @@
+package dev.ltms.fleet.lead;
+
+import com.fasterxml.jackson.databind.JsonNode;
+import com.fasterxml.jackson.databind.ObjectMapper;
+
+import java.io.IOException;
+import java.io.RandomAccessFile;
+import java.nio.charset.StandardCharsets;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.util.ArrayList;
+import java.util.List;
+import java.util.Map;
+import java.util.concurrent.ConcurrentHashMap;
+import java.util.concurrent.atomic.AtomicInteger;
+import java.util.function.LongSupplier;
+
+/**
+ * Reads how full a lead's own Claude Code context window is, from the transcript Claude Code
+ * itself writes — never from the lead's pane (fleetd has {@code AgentControl.read} for that, and
+ * this must not use it: a pane holds terminal text, not the structured usage numbers a transcript
+ * carries, and scraping it would also race the lead's own rendering).
+ *
+ *
Why this exists. A lead auto-compacts when its context fills — on the host
+ * this was built for, that happened 30 times in one session, discarding roughly 250,000 tokens
+ * and costing 46 seconds to 3 minutes each time, and fleetd had no way to see it coming. This
+ * class is the first thing that looks.
+ *
+ *
The route. Claude Code appends one JSON object per line to
+ * {@code /projects//.jsonl}. {@code } is an undocumented,
+ * internal encoding of the working directory — this class never derives it. Instead it lists the
+ * one-level-deep subdirectories of {@code /projects/} and looks for
+ * {@code .jsonl} by name, so the slug rule can change without breaking this reader.
+ *
+ *
+ * - Live context is read off the last record in the read window that carries
+ * a {@code message.usage} object: {@code input_tokens + cache_read_input_tokens +
+ * cache_creation_input_tokens}. This is what actually fills the window — a plain
+ * {@code input_tokens} count alone understates it once the conversation has any cached
+ * prefix, which on a long-lived lead is always.
+ * - Compaction history is a count of {@code subtype: "compact_boundary"}
+ * records seen in the same read window — see {@link Reading#compactions()}. It is a count
+ * within the window this reader actually looked at, not a lifetime total: a session with
+ * more compactions than fit in {@link #TAIL_BYTES} of transcript will undercount. That
+ * trade-off is deliberate — see {@link #TAIL_BYTES}.
+ *
+ *
+ * Three states, not two (OK / HIGH / UNKNOWN). Every path that cannot
+ * positively establish the live token count — a missing file, an unreadable one, a peer that
+ * is not a Claude backend, or a last line that fails to parse as JSON — returns
+ * {@link State#UNKNOWN} with no token number, never a default "0" or "ok" that would read as
+ * "this lead is fine" when the honest answer is "I could not look". A last-line parse failure in
+ * particular is treated as UNKNOWN rather than falling back to the previous good line: Claude Code
+ * appends and flushes one line at a time, so a malformed last line means either a write caught
+ * mid-flush or a format this reader no longer understands — reporting the stale previous number as
+ * if it were current would be exactly the false confidence this whole feature exists to avoid.
+ *
+ *
Bounded cost. {@code fleet_list} is polled constantly, so every read is
+ * capped two ways: {@link #TAIL_BYTES} bounds how much of the transcript is ever read from disk
+ * (never the whole 52 MB a long-lived transcript reaches on the host this was measured on), and
+ * {@link #DEFAULT_CACHE_TTL_MILLIS} bounds how often that bounded read actually happens — a burst
+ * of {@code fleet_list} calls inside one TTL window reads the file once. One instance's cache is
+ * keyed by {@code (configDir, sessionId)}, so it is safe to share across every lead a single
+ * {@code fleet_list} call reports on.
+ */
+public final class LeadContextGauge {
+
+ private static final String PROJECTS_DIR = "projects";
+ private static final ObjectMapper MAPPER = new ObjectMapper();
+
+ /**
+ * How many trailing bytes of a transcript a single read ever pulls off disk. Chosen so one
+ * read comfortably spans many recent turns — each usage or compact_boundary record is at most
+ * a few KB — while staying nowhere near the 52 MB a long session's real transcript reaches on
+ * the host this was built for; reading that whole file on every {@code fleet_list} call is
+ * exactly the cost this bound exists to avoid. 2 MiB holds on the order of hundreds of recent
+ * lines even when a turn's tool output is unusually large, which is far more than needed to
+ * find the most recent usage record and any recent compaction.
+ */
+ static final int TAIL_BYTES = 2 * 1024 * 1024;
+
+ /**
+ * How long a {@link Reading} is served from cache before the file is read again.
+ * {@code fleet_list} is called constantly (by design — it is the fleet's own status probe), so
+ * without a TTL a burst of calls would re-read the transcript tail once per call. 5 seconds is
+ * short enough that a caller watching for a state change never waits long, and long enough that
+ * a poll loop calling every second or two only touches disk once per window.
+ */
+ static final long DEFAULT_CACHE_TTL_MILLIS = 5_000;
+
+ /**
+ * Live tokens at or above this count report {@link State#HIGH}. On the host this was measured
+ * on, auto-compaction actually fires around 267,000–270,000 tokens, but the point of a HIGH
+ * state is to warn before that happens, not at it — 200,000 is the standard Claude context
+ * window size and a sensible built-in default: no config key is required to pick it, and a
+ * lead crossing it is already deep enough into its window that a compaction is foreseeable.
+ */
+ static final long HIGH_THRESHOLD_TOKENS = 200_000;
+
+ /** The only peer kind this reader understands ({@code Agent.agentType()}'s wire value). */
+ private static final String CLAUDE_AGENT_TYPE = "claude";
+
+ public enum State { OK, HIGH, UNKNOWN }
+
+ /**
+ * @param state {@link State#UNKNOWN} whenever {@code tokens} could not be established
+ * @param tokens live context tokens, or {@code null} exactly when {@code state} is
+ * {@link State#UNKNOWN}
+ * @param compactions {@code compact_boundary} records seen in the read window (see class
+ * javadoc) — {@code 0} both for "genuinely none seen" and for "unknown",
+ * since a caller that already sees {@code state: UNKNOWN} has no reason to
+ * trust this number either way
+ */
+ public record Reading(State state, Long tokens, int compactions) {
+ static Reading unknown() {
+ return new Reading(State.UNKNOWN, null, 0);
+ }
+ }
+
+ private record CacheEntry(Reading reading, long readAtMillis) {
+ }
+
+ private final LongSupplier clock;
+ private final long ttlMillis;
+ private final Map cache = new ConcurrentHashMap<>();
+ /** Test seam only (package-private) — counts real disk reads, i.e. cache misses. */
+ private final AtomicInteger diskReads = new AtomicInteger();
+
+ public LeadContextGauge() {
+ this(System::currentTimeMillis, DEFAULT_CACHE_TTL_MILLIS);
+ }
+
+ /** Test seam: an injectable clock and TTL so cache expiry is provable without sleeping. */
+ LeadContextGauge(LongSupplier clock, long ttlMillis) {
+ this.clock = clock;
+ this.ttlMillis = ttlMillis;
+ }
+
+ /** How many times this instance has actually read a transcript off disk — test seam only. */
+ int diskReadCount() {
+ return diskReads.get();
+ }
+
+ /**
+ * @param configDir the lead's {@code CLAUDE_CONFIG_DIR}, or {@code null}/blank to use the
+ * default {@code /.claude} — the right answer for the common case
+ * where the lead's profile sets no {@code configDir} override
+ * @param sessionId the lead's own Claude session id ({@code Agent.sessionId()}), or
+ * {@code null} when herdr has not resolved one yet
+ * @param agentType the detected peer kind ({@code Agent.agentType()}); anything other than
+ * {@code "claude"} (including {@code null}, meaning undetected) reports
+ * {@link State#UNKNOWN} — this reader only understands Claude Code's own
+ * transcript format
+ */
+ public Reading read(String configDir, String sessionId, String agentType) {
+ if (sessionId == null || sessionId.isBlank()) {
+ return Reading.unknown();
+ }
+ if (!CLAUDE_AGENT_TYPE.equalsIgnoreCase(agentType)) {
+ return Reading.unknown();
+ }
+ String base = (configDir == null || configDir.isBlank())
+ ? System.getProperty("user.home") + "/.claude"
+ : configDir;
+ String cacheKey = base + '\u0000' + sessionId;
+ long now = clock.getAsLong();
+ CacheEntry cached = cache.get(cacheKey);
+ if (cached != null && now - cached.readAtMillis() < ttlMillis) {
+ return cached.reading();
+ }
+ Reading fresh = readUncached(base, sessionId);
+ cache.put(cacheKey, new CacheEntry(fresh, now));
+ return fresh;
+ }
+
+ private Reading readUncached(String base, String sessionId) {
+ diskReads.incrementAndGet();
+ Path file = findTranscript(base, sessionId);
+ if (file == null) {
+ return Reading.unknown();
+ }
+ TailRead tail;
+ try {
+ tail = tailBytes(file, TAIL_BYTES);
+ } catch (IOException e) {
+ return Reading.unknown();
+ }
+ return parse(tail);
+ }
+
+ /**
+ * Finds {@code .jsonl} under {@code /projects/}, one level deep — never by
+ * deriving the slug directory from a working directory (see class javadoc). Bounded to a
+ * single {@code list()} of {@code projects/} itself: it never recurses further, so the cost is
+ * the number of project directories, not the size of any transcript inside them.
+ */
+ private Path findTranscript(String base, String sessionId) {
+ Path projectsDir = Path.of(base, PROJECTS_DIR);
+ if (!Files.isDirectory(projectsDir)) {
+ return null;
+ }
+ String filename = sessionId + ".jsonl";
+ Path direct = projectsDir.resolve(filename);
+ if (Files.isRegularFile(direct)) {
+ return direct;
+ }
+ try (var children = Files.list(projectsDir)) {
+ return children.filter(Files::isDirectory)
+ .map(dir -> dir.resolve(filename))
+ .filter(Files::isRegularFile)
+ .findFirst()
+ .orElse(null);
+ } catch (IOException e) {
+ return null;
+ }
+ }
+
+ /** Package-private (not {@code private}): {@link #tailBytes} is a test seam, see its javadoc. */
+ record TailRead(byte[] bytes, boolean fromStart) {
+ }
+
+ /**
+ * Reads at most {@code maxBytes} trailing bytes of {@code file}. Package-private (not
+ * {@code private}) so a test can assert directly on the returned array's length — "bytes
+ * actually read", not on any parsed answer — without needing a file anywhere near
+ * {@link #TAIL_BYTES} in size to prove the cap holds.
+ */
+ static TailRead tailBytes(Path file, int maxBytes) throws IOException {
+ try (RandomAccessFile raf = new RandomAccessFile(file.toFile(), "r")) {
+ long length = raf.length();
+ long start = Math.max(0, length - maxBytes);
+ raf.seek(start);
+ byte[] buf = new byte[(int) (length - start)];
+ raf.readFully(buf);
+ return new TailRead(buf, start == 0);
+ }
+ }
+
+ /**
+ * Parses the tail into a {@link Reading}. The first line is dropped unconditionally whenever
+ * the tail is not the whole file (it starts mid-line, cut by {@link #TAIL_BYTES} — an expected
+ * artefact of the bound, not a data problem). The window's actual last line, by contrast, must
+ * parse cleanly: see the "three states" section of the class javadoc for why a bad last line
+ * means {@link State#UNKNOWN} rather than a fall-back to the previous good line.
+ */
+ private Reading parse(TailRead tail) {
+ String text = new String(tail.bytes(), StandardCharsets.UTF_8);
+ List lines = new ArrayList<>(List.of(text.split("\n", -1)));
+ if (!lines.isEmpty() && lines.get(lines.size() - 1).isEmpty()) {
+ lines.remove(lines.size() - 1); // trailing newline leaves a phantom empty last element
+ }
+ if (!tail.fromStart() && !lines.isEmpty()) {
+ lines.remove(0); // first line is a fragment cut by our own tail bound, not real data
+ }
+ if (lines.isEmpty()) {
+ return Reading.unknown();
+ }
+ JsonNode lastNode;
+ try {
+ lastNode = MAPPER.readTree(lines.get(lines.size() - 1));
+ } catch (IOException e) {
+ return Reading.unknown();
+ }
+ Long tokens = null;
+ int compactions = 0;
+ for (int i = 0; i < lines.size(); i++) {
+ JsonNode node = i == lines.size() - 1 ? lastNode : tryParse(lines.get(i));
+ if (node == null) {
+ continue;
+ }
+ JsonNode usage = node.path("message").path("usage");
+ if (usage.isObject()) {
+ tokens = usage.path("input_tokens").asLong(0)
+ + usage.path("cache_read_input_tokens").asLong(0)
+ + usage.path("cache_creation_input_tokens").asLong(0);
+ }
+ if ("compact_boundary".equals(node.path("subtype").asText(null))) {
+ compactions++;
+ }
+ }
+ if (tokens == null) {
+ return new Reading(State.UNKNOWN, null, compactions);
+ }
+ State state = tokens >= HIGH_THRESHOLD_TOKENS ? State.HIGH : State.OK;
+ return new Reading(state, tokens, compactions);
+ }
+
+ private JsonNode tryParse(String line) {
+ try {
+ return MAPPER.readTree(line);
+ } catch (IOException e) {
+ return null;
+ }
+ }
+}
diff --git a/fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java b/fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java
index 6fa0559..b475773 100644
--- a/fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java
+++ b/fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java
@@ -12,6 +12,7 @@ import dev.ltms.fleet.metrics.Metrics;
import dev.ltms.fleet.inject.MemberPresence;
import dev.ltms.fleet.inject.CompletionResolver;
import dev.ltms.fleet.herdr.HerdrException;
+import dev.ltms.fleet.lead.LeadContextGauge;
import dev.ltms.fleet.lead.LeadRollover;
import dev.ltms.fleet.msg.LeadChannel;
import dev.ltms.fleet.msg.LeadMessage;
@@ -126,6 +127,17 @@ public final class FleetMcp {
* clean {@code NOT_CONFIGURED} refusal rather than throwing. See {@link #handover}.
*/
private final LeadRollover leadRollover;
+ /**
+ * "Lead context gauge": how full each lead's own Claude Code context window is, reported on
+ * {@code fleet_list}'s {@code leads} rows (see {@link #contextView}). Built unconditionally, in
+ * the field initializer rather than a constructor parameter — this is not a togglable feature
+ * with an on/off config knob the way {@link OutageSource}/{@link LeadSeatSource} are: it needs
+ * no config at all (see {@link LeadContextGauge}'s own javadoc for the built-in defaults), so
+ * there is no "off" value to thread through every existing constructor call site. One instance
+ * per daemon so its read cache (keyed by session, TTL'd) is actually shared across
+ * {@code fleet_list} calls rather than rebuilt — and therefore useless — on every call.
+ */
+ private final LeadContextGauge leadContextGauge = new LeadContextGauge();
/** Capacity facts used by {@code fleet_list}; production must supply the placement live count. */
public record CapacitySource(Function liveCount, Function maxLoad,
@@ -498,7 +510,7 @@ public final class FleetMcp {
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_list", Map.of()), null);
if (denied != null) return denied;
return listFleet(workers, sessions, messages, capacity, healthCoverage, loopHealth, quarantine, outage,
- leadSeats, callers.leads(),
+ leadSeats, leadContextGauge, callers.leads(),
callerTerminal(exchange),
new CoordinationSource(leadChannel, peers),
coordinatorVisibleTo(principal(exchange)));
@@ -1601,7 +1613,7 @@ public final class FleetMcp {
Map leads, String selfTerm,
CoordinationSource coordination) {
return listFleet(workers, sessions, messages, capacity, healthCoverage, loopHealth, quarantine,
- OutageSource.none(), LeadSeatSource.none(), leads, selfTerm, coordination, false);
+ OutageSource.none(), LeadSeatSource.none(), new LeadContextGauge(), leads, selfTerm, coordination, false);
}
/** As above, plus fleetd #201 Unit 5 cool-off facts (see {@link OutageSource}). */
@@ -1610,7 +1622,7 @@ public final class FleetMcp {
QuarantineSource quarantine, OutageSource outage,
Map leads, String selfTerm) {
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, outage,
- LeadSeatSource.none(), leads, selfTerm, CoordinationSource.none(), false);
+ LeadSeatSource.none(), new LeadContextGauge(), leads, selfTerm, CoordinationSource.none(), false);
}
/**
@@ -1627,7 +1639,7 @@ public final class FleetMcp {
QuarantineSource quarantine, Map leads, String selfTerm,
CoordinationSource coordination) {
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, OutageSource.none(),
- LeadSeatSource.none(), leads, selfTerm, coordination, false);
+ LeadSeatSource.none(), new LeadContextGauge(), leads, selfTerm, coordination, false);
}
/** As above, plus fleetd #201 Unit 5 cool-off facts (see {@link OutageSource}). */
@@ -1636,7 +1648,7 @@ public final class FleetMcp {
QuarantineSource quarantine, OutageSource outage,
Map leads, String selfTerm, CoordinationSource coordination) {
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, outage,
- LeadSeatSource.none(), leads, selfTerm, coordination, false);
+ LeadSeatSource.none(), new LeadContextGauge(), leads, selfTerm, coordination, false);
}
/**
@@ -1658,7 +1670,7 @@ public final class FleetMcp {
LeadSeatSource leadSeats, Map leads, String selfTerm,
CoordinationSource coordination) {
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, outage,
- leadSeats, leads, selfTerm, coordination, false);
+ leadSeats, new LeadContextGauge(), leads, selfTerm, coordination, false);
}
/**
@@ -1682,14 +1694,23 @@ public final class FleetMcp {
LeadSeatSource leadSeats, Map leads, String selfTerm,
CoordinationSource coordination, boolean callerIsPrimary) {
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine,
- outage, leadSeats, leads, selfTerm, coordination, callerIsPrimary);
+ outage, leadSeats, new LeadContextGauge(), leads, selfTerm, coordination, callerIsPrimary);
}
+ /**
+ * The canonical implementation. {@code contextGauge} is the "lead context gauge" (see
+ * {@link LeadContextGauge}) — every wrapper overload above passes a freshly constructed one,
+ * which is correct for them (none of them exercise repeated calls where a shared cache would
+ * matter); the one caller that matters for caching, {@code fleet_list}'s MCP handler, passes
+ * its own single long-lived instance instead (see {@code FleetMcp}'s {@code leadContextGauge}
+ * field).
+ */
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
CapacitySource capacity, HealthCoverageSource healthCoverage,
LoopHealthSource loopHealth,
QuarantineSource quarantine, OutageSource outage,
- LeadSeatSource leadSeats, Map leads, String selfTerm,
+ LeadSeatSource leadSeats, LeadContextGauge contextGauge,
+ Map leads, String selfTerm,
CoordinationSource coordination, boolean callerIsPrimary) {
try {
Map live = workers.list().stream()
@@ -1698,7 +1719,7 @@ public final class FleetMcp {
.collect(Collectors.toMap(Agent::terminalId, Function.identity(), (_, b) -> b));
List