Compare commits
4 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| d59ece6dec | |||
| aa4c0b84c3 | |||
| 40c593cd09 | |||
| 979adf82eb |
@@ -65,6 +65,27 @@
|
||||
# credential the member holds in full.
|
||||
#
|
||||
set -uo pipefail
|
||||
# `pipefail` is not what catches the parser failure handled below (fleetd #500): in
|
||||
# `printf '%s' "$POLICY_JSON" | jq -r '...'`, jq is the LAST element of the pipe, so the pipeline's
|
||||
# own exit status is already jq's status, with or without pipefail. It is kept as insurance for if
|
||||
# a post-processing stage is ever appended after the parser (e.g. `| tail -n +2`) — at that point
|
||||
# the parser would sit upstream and pipefail becomes the only thing that still reports its status.
|
||||
|
||||
# --- refuse on an interpreter that cannot run this script (fleetd #500) -------------------------
|
||||
#
|
||||
# mapfile, used below to parse the policy response, was added in bash 4.0. macOS ships bash 3.2.57
|
||||
# at /bin/bash, which predates it. This script's own `set -uo pipefail` does not catch a missing
|
||||
# mapfile: the builtin just fails with "command not found" on stderr, and every line below that
|
||||
# reads the array it would have filled uses a `:-` default or a slice, neither of which `set -u`
|
||||
# catches on an unset array. Left unguarded, that chain ends in the "0 known names" refusal further
|
||||
# down — a claim about the POLICY, for a failure that is actually about the INTERPRETER. So the
|
||||
# interpreter is checked once, explicitly, before it is asked to do anything mapfile depends on.
|
||||
if (( ${BASH_VERSINFO[0]} < 4 )); then
|
||||
echo "refusing to run: this script uses mapfile, which needs bash 4 or newer. This shell is bash" \
|
||||
"${BASH_VERSION:-<unknown, no \$BASH_VERSION>}. Re-run it under a newer bash, for example:" \
|
||||
"\"\$(command -v bash)\" \"$0\"" "$@" >&2
|
||||
exit 3
|
||||
fi
|
||||
|
||||
FLEETD_HOST="${FLEETD_HOST:-http://127.0.0.1:8765}"
|
||||
POLICY_URL="${FLEETD_HOST%/}/member-credentials"
|
||||
@@ -124,16 +145,32 @@ fi
|
||||
# One parse pass: line 1 = present (true/false/null), line 2 = policy mode (possibly blank),
|
||||
# lines 3-5 = knownCount/allowedCount/blockedCount, remaining lines = the known[] names. A single
|
||||
# pass avoids re-parsing (and re-risking a truthiness bug) five separate times.
|
||||
#
|
||||
# This used to feed the parser straight into `mapfile -t _FIELDS < <(producer)`. That form cannot
|
||||
# see the producer fail: `<` `<(...)` is a process substitution, not a pipeline, so `set -o
|
||||
# pipefail` does not reach inside it, and mapfile's own exit status reports whether the BUILTIN
|
||||
# ran, not whether the command substituted into it succeeded — a failing jq or python3 there still
|
||||
# leaves mapfile at rc=0 with an empty array, read as a parse that genuinely found nothing (fleetd
|
||||
# #500). Capturing the parser's output with command substitution first, and checking ITS exit
|
||||
# status, reports the producer's real failure while the fact still exists — before it is handed to
|
||||
# mapfile at all.
|
||||
#
|
||||
# mapfile then reads from that captured string with `<<<` (a herestring), not `< <(...)`: `<<<`
|
||||
# materialises the whole string in memory first, where `< <(...)` would stream it. That only
|
||||
# matters for a large producer; this one is a short credential-name policy response, so the
|
||||
# tradeoff is irrelevant here — noted because it would not be for every producer.
|
||||
if command -v jq >/dev/null 2>&1; then
|
||||
mapfile -t _FIELDS < <(printf '%s' "$POLICY_JSON" | jq -r '
|
||||
_FIELDS_RAW="$(printf '%s' "$POLICY_JSON" | jq -r '
|
||||
(.present | tostring),
|
||||
(.policy // ""),
|
||||
(.knownCount // 0 | tostring),
|
||||
(.allowedCount // 0 | tostring),
|
||||
(.blockedCount // 0 | tostring),
|
||||
(.known[]? // empty)')
|
||||
(.known[]? // empty)')"
|
||||
_PARSE_STATUS=$?
|
||||
_PARSER_NAME="jq"
|
||||
else
|
||||
mapfile -t _FIELDS < <(printf '%s' "$POLICY_JSON" | python3 - <<'PY'
|
||||
_FIELDS_RAW="$(printf '%s' "$POLICY_JSON" | python3 - <<'PY'
|
||||
import json, sys
|
||||
data = json.load(sys.stdin)
|
||||
print(str(data.get("present")))
|
||||
@@ -144,7 +181,34 @@ print(data.get("blockedCount") if data.get("blockedCount") is not None else 0)
|
||||
for n in (data.get("known") or []):
|
||||
print(n)
|
||||
PY
|
||||
)
|
||||
)"
|
||||
_PARSE_STATUS=$?
|
||||
_PARSER_NAME="python3"
|
||||
fi
|
||||
|
||||
if [ "$_PARSE_STATUS" -ne 0 ]; then
|
||||
echo "refusing to run: could not parse the policy fetched from $POLICY_URL — $_PARSER_NAME exited" \
|
||||
"non-zero (status $_PARSE_STATUS). That is a parser failure, not a claim about the policy" \
|
||||
"itself; the policy response has not been read." >&2
|
||||
exit 4
|
||||
fi
|
||||
|
||||
mapfile -t _FIELDS <<< "$_FIELDS_RAW"
|
||||
|
||||
# Arity check — the CORRECTNESS fix (fleetd #500). A parser that exits 0 can still return fewer
|
||||
# than the 5 fixed fields (present, policy mode, 3 counts) that every line below this expects,
|
||||
# whatever the reason: a producer that printed nothing, malformed JSON that jq/python3 still
|
||||
# accepted, or a schema change upstream. The slice just below this (`_FIELDS[@]:5`) does not fire
|
||||
# `set -u` on an unset OR a short array, and every fixed-field read above used a `:-` default, so
|
||||
# without this check a short `_FIELDS` reaches the "0 known names" guard further down with the
|
||||
# same look as a policy that genuinely has 0 names. Check the count here, at the one point the
|
||||
# fact is still present, before the slice consumes it.
|
||||
if (( ${#_FIELDS[@]} < 5 )); then
|
||||
echo "refusing to run: the policy parser ($_PARSER_NAME) returned ${#_FIELDS[@]} field(s); at" \
|
||||
"least 5 are required (present, policy mode, knownCount, allowedCount, blockedCount). The" \
|
||||
"parse ran but its shape is wrong — this is not a claim about how many names the policy" \
|
||||
"knows." >&2
|
||||
exit 5
|
||||
fi
|
||||
|
||||
PRESENT="${_FIELDS[0]:-null}"
|
||||
@@ -164,19 +228,21 @@ case "$KNOWN_COUNT_REPORTED" in
|
||||
;;
|
||||
esac
|
||||
|
||||
# --- guard the denominator explicitly — never proceed on a zero/short count ---------------------
|
||||
# --- guard the denominator explicitly — never proceed on a zero count ---------------------------
|
||||
#
|
||||
# This is the exact trap named in the ticket: an empty (or truncated) NAMES array passes every
|
||||
# subsequent "is it set" check vacuously and prints a table that LOOKS complete. So this is checked
|
||||
# before anything else runs, with a message that says why, not just that it failed.
|
||||
# This is the exact trap named in the ticket: an empty NAMES array passes every subsequent "is it
|
||||
# set" check vacuously and prints a table that LOOKS complete. By this point the interpreter gate,
|
||||
# the parser-exit-status check, and the arity check above have already ruled out "the interpreter
|
||||
# couldn't run mapfile", "the parser failed", and "the parser returned the wrong shape" — so a zero
|
||||
# count reaching here really does mean the policy itself reports 0 known names, not a swallowed
|
||||
# failure upstream. That is still checked before anything else runs, with a message that says so.
|
||||
if [ "${#NAMES[@]}" -eq 0 ] || [ "$KNOWN_COUNT_REPORTED" -eq 0 ]; then
|
||||
cat >&2 <<EOF
|
||||
refusing to run: the policy fetched from $POLICY_URL contains 0 known names (present=${PRESENT:-unknown}).
|
||||
|
||||
Either memberCredentials: is absent/empty on the running daemon (nothing is protected — see fleetd's
|
||||
own startup warning), or the response could not be parsed. Either way, checking zero names would
|
||||
print a clean-looking table for a policy that protects nothing, or for a probe that read nothing.
|
||||
This is refused rather than reported as a pass.
|
||||
memberCredentials: is absent or empty on the running daemon — nothing is protected (see fleetd's own
|
||||
startup warning). Checking zero names would print a clean-looking table for a policy that protects
|
||||
nothing. This is refused rather than reported as a pass.
|
||||
EOF
|
||||
exit 1
|
||||
fi
|
||||
|
||||
+99
-10
@@ -56,6 +56,13 @@ set -euo pipefail
|
||||
REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
MODULE="$REPO/fleetd"
|
||||
JAR="$MODULE/target/fleetd.jar"
|
||||
# fleetd #493: never build into the path a running process holds. The build writes here first
|
||||
# (Maven's shade plugin has finalName=fleetd, so `clean install` still lands its output at
|
||||
# target/fleetd.jar — that part is unchanged and out of this script's control), but this script
|
||||
# now moves it out to JAR_STAGED immediately, and only swaps it back to JAR (a plain `mv`, so a
|
||||
# rename, never a byte-by-byte overwrite) after the OLD daemon has been confirmed exited. See
|
||||
# stage_built_jar/swap_staged_jar below.
|
||||
JAR_STAGED="$MODULE/target/fleetd-new.jar"
|
||||
OUT="$MODULE/fleetd.out"
|
||||
# Matches BOTH the absolute form and the relative `java -jar target/fleetd.jar` a hand-start
|
||||
# produces from inside fleetd/. Anchoring on the absolute path alone was a real bug: the daemon
|
||||
@@ -115,8 +122,62 @@ ok() { printf ' ok %s\n' "$*"; }
|
||||
warn() { printf ' WARN %s\n' "$*"; }
|
||||
die() { printf '\n FAIL %s\n\n' "$*" >&2; exit 1; }
|
||||
|
||||
jar_id() { [ -f "$JAR" ] && shasum -a 256 "$JAR" | cut -c1-12 || echo "absent"; }
|
||||
# Reports the hash of $JAR by default, or of whatever path is passed — used to report the STAGED
|
||||
# jar right after a build (before it has been swapped in) without ever changing what a bare
|
||||
# `jar_id` (no args) means: the live path, $JAR. --check and the final "pid ..., jar ..." line
|
||||
# both call it with no args on purpose, so neither can ever be fooled by a leftover staged file.
|
||||
jar_id() { local f="${1:-$JAR}"; [ -f "$f" ] && shasum -a 256 "$f" | cut -c1-12 || echo "absent"; }
|
||||
running_pid() { pgrep -f "$PATTERN" || true; }
|
||||
|
||||
# fleetd #493 — three small, independently testable pieces of "never build into the path a
|
||||
# running process holds":
|
||||
#
|
||||
# stage_built_jar moves the jar Maven just produced OUT of the live path and onto the staging
|
||||
# path, immediately after a successful build. Dies (leaving the OLD daemon
|
||||
# untouched — this runs before the stop step) if Maven reported success but
|
||||
# left no jar behind, or if the move itself fails.
|
||||
# require_no_build_jar the --no-build path never builds or stages anything: it must find a
|
||||
# jar already sitting at the live path from an earlier successful run, and
|
||||
# die with the same truthful message this script has always used if not.
|
||||
# wait_for_daemon_exit polls running_pid() for up to $1 seconds and reports whether the OLD
|
||||
# daemon actually exited — extracted to its own function so the main flow
|
||||
# can be relied on to call swap_staged_jar only AFTER this returns success,
|
||||
# and so a test can prove that ordering by reading the script's own source.
|
||||
# swap_staged_jar the actual swap: a plain `mv` of the staged jar onto the live path. Called
|
||||
# only once the OLD daemon is confirmed gone (see wait_for_daemon_exit above),
|
||||
# so this is never a write into a path a running process holds — by the time
|
||||
# it runs, nothing holds that path anymore. If it fails, the caller must not
|
||||
# start a new daemon: die() below already refuses that by exiting the script.
|
||||
stage_built_jar() {
|
||||
[ -f "$JAR" ] || die "build succeeded but produced no jar at $JAR — cannot stage it for restart.
|
||||
The running daemon was NOT touched."
|
||||
mv -f "$JAR" "$JAR_STAGED" \
|
||||
|| die "could not move the freshly built jar from $JAR to the staging path $JAR_STAGED.
|
||||
The running daemon was NOT touched."
|
||||
}
|
||||
|
||||
require_no_build_jar() {
|
||||
[ -f "$JAR" ] || die "no jar at $JAR — run without --no-build"
|
||||
}
|
||||
|
||||
wait_for_daemon_exit() {
|
||||
local timeout="$1" _i
|
||||
for _i in $(seq "$timeout"); do
|
||||
[ -z "$(running_pid)" ] && return 0
|
||||
sleep 1
|
||||
done
|
||||
[ -z "$(running_pid)" ]
|
||||
}
|
||||
|
||||
swap_staged_jar() {
|
||||
local staged="$1" live="$2"
|
||||
[ -f "$staged" ] || die "no staged jar at $staged to swap in — the daemon was NOT started."
|
||||
mv -f "$staged" "$live" \
|
||||
|| die "could not move the staged jar from $staged into place at $live — the daemon was NOT
|
||||
started. The built jar is still sitting at $staged; a manual 'mv \"$staged\" \"$live\"'
|
||||
may recover this once you find out why the move failed."
|
||||
}
|
||||
|
||||
# `launchctl list <label>` exits 0 iff the label is loaded (registered with launchd) — true whether
|
||||
# or not it is currently running, which is exactly "supervision is active" for our purposes. Read-
|
||||
# only: neither helper below changes anything, so both are also safe under --check.
|
||||
@@ -512,6 +573,9 @@ fi
|
||||
|
||||
if [ "$DO_BUILD" = 1 ]; then
|
||||
say "build"
|
||||
# fleetd #493: wipe a leftover staged jar from a previous failed/interrupted run BEFORE doing
|
||||
# anything else, so that run's leftovers can never be mistaken for this run's output.
|
||||
rm -f "$JAR_STAGED"
|
||||
BUILD_LOG="$(mktemp -t fleetd-build)"
|
||||
echo " log: $BUILD_LOG"
|
||||
if ! mvn -f "$MODULE/pom.xml" clean install > "$BUILD_LOG" 2>&1; then
|
||||
@@ -521,13 +585,19 @@ if [ "$DO_BUILD" = 1 ]; then
|
||||
fi
|
||||
grep -E '^\[INFO\] Tests run:.*Failures' "$BUILD_LOG" | tail -1 | sed 's/^\[INFO\] / /' || true
|
||||
ok "BUILD SUCCESS"
|
||||
ok "jar now: $(jar_id)"
|
||||
# fleetd #493: move the freshly built jar off the live path immediately — the running (OLD)
|
||||
# daemon, if any, is still up at this point (build always runs before stop). From here until the
|
||||
# swap step below (after the OLD daemon is confirmed gone), $JAR_STAGED is the only artefact this
|
||||
# script treats as "the new jar" — $JAR itself is not touched again until the swap.
|
||||
stage_built_jar
|
||||
ok "jar now: $(jar_id "$JAR_STAGED")"
|
||||
else
|
||||
say "build skipped (--no-build)"
|
||||
# fleetd #493: --no-build never builds or stages anything — it restarts whatever jar is already
|
||||
# sitting at the live path from an earlier successful run. Same check, same message as before.
|
||||
require_no_build_jar
|
||||
fi
|
||||
|
||||
[ -f "$JAR" ] || die "no jar at $JAR — run without --no-build"
|
||||
|
||||
# ----------------------------------------------------------------- drain gate
|
||||
|
||||
if [ -n "$OLD_PID" ] && [ "$ASSUME_YES" = 0 ]; then
|
||||
@@ -539,7 +609,17 @@ if [ -n "$OLD_PID" ] && [ "$ASSUME_YES" = 0 ]; then
|
||||
echo " you still want, BEFORE continuing."
|
||||
echo
|
||||
read -r -p " Fleet drained? type yes to restart: " reply
|
||||
[ "$reply" = "yes" ] || die "aborted — nothing changed"
|
||||
if [ "$reply" != "yes" ]; then
|
||||
# fleetd #493: "nothing changed" would be a lie once a build has run — the freshly built jar
|
||||
# already moved to $JAR_STAGED (stage_built_jar, above), so the live path has one fewer file
|
||||
# than before this run started, even though the running daemon itself was never touched.
|
||||
if [ "$DO_BUILD" = 1 ] && [ -f "$JAR_STAGED" ]; then
|
||||
die "aborted — the running daemon was NOT touched, but the freshly built jar is sitting at
|
||||
$JAR_STAGED, not yet swapped into $JAR. Rerun (with or without --no-build) to finish the
|
||||
restart, or remove $JAR_STAGED by hand if you want to discard this build."
|
||||
fi
|
||||
die "aborted — nothing changed"
|
||||
fi
|
||||
fi
|
||||
|
||||
# ------------------------------------------------------------------ stop
|
||||
@@ -585,11 +665,7 @@ if [ -n "$OLD_PID" ]; then
|
||||
refusing to guess how to stop a daemon under an unknown supervisor. The daemon was NOT
|
||||
stopped." ;;
|
||||
esac
|
||||
for _ in $(seq "$STOP_WAIT"); do
|
||||
[ -z "$(running_pid)" ] && break
|
||||
sleep 1
|
||||
done
|
||||
if [ -n "$(running_pid)" ]; then
|
||||
if ! wait_for_daemon_exit "$STOP_WAIT"; then
|
||||
die "pid $OLD_PID still alive after ${STOP_WAIT}s. Not escalating to kill -9 automatically:
|
||||
the shutdown hook releases sessions and worktrees in order, and killing it hard can
|
||||
leave worktrees and panes behind. Investigate, then kill -9 by hand if you accept that."
|
||||
@@ -613,6 +689,19 @@ else
|
||||
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
|
||||
fi
|
||||
|
||||
# ------------------------------------------------------------------ swap
|
||||
#
|
||||
# fleetd #493: every branch above has now either confirmed the OLD daemon actually exited
|
||||
# (wait_for_daemon_exit, above) or established there was never one running to begin with. Only
|
||||
# NOW is it safe to put the freshly built jar at the path the NEXT `java -jar` (direct, or via
|
||||
# launchd/systemd's ExecStart) will read from — this mv is the one and only write to $JAR anywhere
|
||||
# in this script's mutating flow. If it fails, do not start: die() below exits before "start" runs.
|
||||
if [ "$DO_BUILD" = 1 ]; then
|
||||
say "swap"
|
||||
swap_staged_jar "$JAR_STAGED" "$JAR"
|
||||
ok "jar in place: $(jar_id)"
|
||||
fi
|
||||
|
||||
# ------------------------------------------------------------------ start
|
||||
# Unsupervised: login shell (zsh -l) is what puts the secrets on the daemon's environment, and cwd
|
||||
# must be fleetd/ because the daemon resolves fleetd.yaml, logs/ and target/ relative to it.
|
||||
|
||||
@@ -206,6 +206,144 @@ test_assert_single_daemon_rejects_two_pids() {
|
||||
printf '%s' "$output" | grep -qF '4343' || fail "refusal message does not list the pids it found"
|
||||
}
|
||||
|
||||
# fleetd #493 — never build into the path a running process holds. stage_built_jar/swap_staged_jar
|
||||
# are exercised directly against real files on disk (not stubs), because the whole point is file
|
||||
# behavior (does the content move, does the source disappear, does a failure leave both sides
|
||||
# intact) that a stubbed function cannot prove.
|
||||
test_stage_built_jar_moves_off_live_path() {
|
||||
local dir jar staged saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
||||
dir="$TMP/stage-ok"; mkdir -p "$dir"
|
||||
jar="$dir/fleetd.jar"; staged="$dir/fleetd-new.jar"
|
||||
printf 'built jar bytes' > "$jar"
|
||||
JAR="$jar"; JAR_STAGED="$staged"
|
||||
stage_built_jar || fail "stage_built_jar rejected a real build output"
|
||||
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
||||
[ ! -f "$jar" ] || fail "stage_built_jar left the jar behind at the live path $jar"
|
||||
[ -f "$staged" ] || fail "stage_built_jar did not create the staged jar at $staged"
|
||||
grep -qF 'built jar bytes' "$staged" || fail "staged jar does not carry the built content"
|
||||
}
|
||||
|
||||
test_stage_built_jar_dies_when_build_produced_nothing() {
|
||||
local dir output rc=0 saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
||||
dir="$TMP/stage-missing"; mkdir -p "$dir"
|
||||
JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar"
|
||||
output="$(stage_built_jar 2>&1)" || rc=$?
|
||||
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
||||
[ "$rc" -ne 0 ] || fail "stage_built_jar accepted a missing build output"
|
||||
printf '%s' "$output" | grep -qF "$dir/fleetd.jar" \
|
||||
|| fail "refusal message does not name the missing jar path"
|
||||
}
|
||||
|
||||
test_swap_staged_jar_moves_staged_onto_live() {
|
||||
local dir staged live
|
||||
dir="$TMP/swap-ok"; mkdir -p "$dir"
|
||||
staged="$dir/fleetd-new.jar"; live="$dir/fleetd.jar"
|
||||
printf 'swapped jar bytes' > "$staged"
|
||||
swap_staged_jar "$staged" "$live" || fail "swap_staged_jar rejected a real staged jar"
|
||||
[ ! -f "$staged" ] || fail "swap_staged_jar left the staged file behind at $staged"
|
||||
[ -f "$live" ] || fail "swap_staged_jar did not create the live jar at $live"
|
||||
grep -qF 'swapped jar bytes' "$live" || fail "live jar does not carry the staged content"
|
||||
}
|
||||
|
||||
# The heart of the ticket's item 3: a failed swap must refuse to start. This function dies on
|
||||
# failure, and die() exits — so like the require_drivable_supervisor tests above, the call goes
|
||||
# inside a command substitution to contain that exit to a subshell.
|
||||
test_swap_staged_jar_dies_without_staged_file() {
|
||||
local dir output rc=0
|
||||
dir="$TMP/swap-missing"; mkdir -p "$dir"
|
||||
output="$(swap_staged_jar "$dir/fleetd-new.jar" "$dir/fleetd.jar" 2>&1)" || rc=$?
|
||||
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a missing staged jar"
|
||||
[ ! -f "$dir/fleetd.jar" ] || fail "swap_staged_jar must not create the live jar when nothing was staged"
|
||||
printf '%s' "$output" | grep -qF "$dir/fleetd-new.jar" \
|
||||
|| fail "refusal message does not name the missing staged path"
|
||||
}
|
||||
|
||||
test_swap_staged_jar_dies_when_mv_fails() {
|
||||
local dir staged live output rc=0
|
||||
dir="$TMP/swap-fail"; mkdir -p "$dir/src"
|
||||
staged="$dir/src/fleetd-new.jar"
|
||||
printf 'fake jar bytes' > "$staged"
|
||||
live="$dir/no-such-dir/fleetd.jar" # parent directory does not exist -> mv fails
|
||||
output="$(swap_staged_jar "$staged" "$live" 2>&1)" || rc=$?
|
||||
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a failing mv"
|
||||
[ -f "$staged" ] || fail "swap_staged_jar must leave the staged jar in place when the move fails"
|
||||
[ ! -f "$live" ] || fail "swap_staged_jar must not report success when the move failed"
|
||||
printf '%s' "$output" | grep -qF "$staged" \
|
||||
|| fail "refusal message does not name the staged path that could not be moved"
|
||||
}
|
||||
|
||||
# --no-build must still resolve $JAR (never the staged path — there is nothing to stage on this
|
||||
# path) and must still die with the exact wording documented in the script's own header comment.
|
||||
test_require_no_build_jar_dies_when_absent() {
|
||||
local saved_jar="$JAR" output rc=0 missing="$TMP/no-build-absent/fleetd.jar"
|
||||
JAR="$missing"
|
||||
output="$(require_no_build_jar 2>&1)" || rc=$?
|
||||
JAR="$saved_jar"
|
||||
[ "$rc" -ne 0 ] || fail "require_no_build_jar accepted a missing jar"
|
||||
printf '%s' "$output" | grep -qF "no jar at $missing — run without --no-build" \
|
||||
|| fail "refusal message does not match the documented --no-build wording"
|
||||
}
|
||||
|
||||
test_require_no_build_jar_accepts_present_jar() {
|
||||
local saved_jar="$JAR" dir
|
||||
dir="$TMP/no-build-present"; mkdir -p "$dir"
|
||||
JAR="$dir/fleetd.jar"
|
||||
printf 'existing jar' > "$JAR"
|
||||
require_no_build_jar || fail "require_no_build_jar rejected an existing jar"
|
||||
JAR="$saved_jar"
|
||||
}
|
||||
|
||||
# wait_for_daemon_exit is the seam the swap ordering depends on: it must not report success while
|
||||
# running_pid() still answers, and must report success the moment it clears. `sleep` is shadowed so
|
||||
# the timeout-loop test does not actually wait out its budget.
|
||||
test_wait_for_daemon_exit_returns_true_once_pid_clears() {
|
||||
# running_pid() runs inside a $(...) — a subshell — every time wait_for_daemon_exit calls it, so
|
||||
# a plain shell variable it increments would reset on each call instead of accumulating. Count in
|
||||
# a file instead, which is the one thing that actually survives across those subshells.
|
||||
local counter_file="$TMP/wait-exit-calls" final_calls
|
||||
printf '0' > "$counter_file"
|
||||
running_pid() {
|
||||
local n
|
||||
n="$(cat "$counter_file")"
|
||||
n=$((n + 1))
|
||||
printf '%s' "$n" > "$counter_file"
|
||||
if [ "$n" -lt 3 ]; then printf '4242'; else printf ''; fi
|
||||
}
|
||||
sleep() { :; }
|
||||
wait_for_daemon_exit 10 || fail "wait_for_daemon_exit did not report success once the pid cleared"
|
||||
final_calls="$(cat "$counter_file")"
|
||||
[ "$final_calls" -ge 3 ] || fail "wait_for_daemon_exit returned before actually re-checking running_pid"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
|
||||
}
|
||||
|
||||
test_wait_for_daemon_exit_times_out_if_pid_never_clears() {
|
||||
local rc=0
|
||||
running_pid() { printf '4242'; }
|
||||
sleep() { :; }
|
||||
wait_for_daemon_exit 3 || rc=$?
|
||||
[ "$rc" -ne 0 ] || fail "wait_for_daemon_exit reported success while the pid never cleared"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
|
||||
}
|
||||
|
||||
# fleetd #493 item 2: "put the swap after that wait, before the start." Sourcing stops before the
|
||||
# main flow ever runs (see the SOURCED guard in redeploy-fleetd.sh), so the ordering guarantee
|
||||
# itself — as opposed to the pure functions it's built from — can only be checked by reading the
|
||||
# script's own call sites, the same way test_recovery_patterns_match_source below checks Java
|
||||
# source shape instead of behavior it cannot invoke directly.
|
||||
test_swap_ordered_after_wait_and_before_start() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" wait_line swap_line start_line
|
||||
wait_line="$(grep -Fn 'wait_for_daemon_exit "$STOP_WAIT"' "$src" | head -1 | cut -d: -f1)"
|
||||
swap_line="$(grep -Fn 'swap_staged_jar "$JAR_STAGED" "$JAR"' "$src" | head -1 | cut -d: -f1)"
|
||||
start_line="$(grep -Fn 'say "start"' "$src" | head -1 | cut -d: -f1)"
|
||||
[ -n "$wait_line" ] || fail "could not find the wait-for-exit call site in redeploy-fleetd.sh"
|
||||
[ -n "$swap_line" ] || fail "could not find the swap call site in redeploy-fleetd.sh"
|
||||
[ -n "$start_line" ] || fail "could not find the start section in redeploy-fleetd.sh"
|
||||
[ "$swap_line" -gt "$wait_line" ] \
|
||||
|| fail "swap_staged_jar (line $swap_line) is not after wait_for_daemon_exit (line $wait_line)"
|
||||
[ "$swap_line" -lt "$start_line" ] \
|
||||
|| fail "swap_staged_jar (line $swap_line) is not before the start section (line $start_line)"
|
||||
}
|
||||
|
||||
test_no_errors() {
|
||||
cat > "$TMP/no-errors.log" <<'LOG'
|
||||
2026-09-05 12:00:00 INFO fleetd listening
|
||||
@@ -417,6 +555,16 @@ test_require_drivable_supervisor_accepts_known_kinds
|
||||
test_count_daemon_pids
|
||||
test_assert_single_daemon_accepts_one_pid
|
||||
test_assert_single_daemon_rejects_two_pids
|
||||
test_stage_built_jar_moves_off_live_path
|
||||
test_stage_built_jar_dies_when_build_produced_nothing
|
||||
test_swap_staged_jar_moves_staged_onto_live
|
||||
test_swap_staged_jar_dies_without_staged_file
|
||||
test_swap_staged_jar_dies_when_mv_fails
|
||||
test_require_no_build_jar_dies_when_absent
|
||||
test_require_no_build_jar_accepts_present_jar
|
||||
test_wait_for_daemon_exit_returns_true_once_pid_clears
|
||||
test_wait_for_daemon_exit_times_out_if_pid_never_clears
|
||||
test_swap_ordered_after_wait_and_before_start
|
||||
test_no_errors
|
||||
test_recovery_patterns_match_source
|
||||
test_attributed_recovered_connection_error
|
||||
|
||||
Reference in New Issue
Block a user