Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 599419f9e6 | |||
| b17f37a683 |
+173
-15
@@ -34,7 +34,11 @@
|
||||
# three different answers, drives whichever one it finds through its own control plane
|
||||
# (`launchctl` / `systemctl --user`), and REFUSES outright — never falls back to `kill` — when
|
||||
# it finds a supervision signal it cannot map to exactly one of the two it knows how to drive.
|
||||
# A wrong guess here is how two daemons end up running against one herdr session.
|
||||
# A wrong guess here is how two daemons end up running against one herdr session. Follow-up:
|
||||
# "not currently loaded" is not the same fact as "unsupervised" — a unit that is installed but
|
||||
# activating/failed/pending-restart, or a `systemctl` call that could not answer at all (e.g.
|
||||
# no user-bus access), both now read as a fifth answer, "unclear", and REFUSE the same way
|
||||
# "ambiguous" does, rather than silently falling through to "none".
|
||||
# 8. fleetd #492 — a post-restart check counts running fleetd processes and fails the whole run if
|
||||
# more than one is alive. That is the one thing none of the checks above (healthz 200, jar id,
|
||||
# the fresh "listening" line) can see: every one of them is satisfied by EITHER daemon.
|
||||
@@ -71,6 +75,30 @@ LAUNCHD_PLIST="$HOME/Library/LaunchAgents/$LAUNCHD_LABEL.plist"
|
||||
# "dev.ltms.fleetd" — systemd user units here are not namespaced the way the launchd label is).
|
||||
SYSTEMD_UNIT='fleetd'
|
||||
|
||||
# fleetd #492 follow-up: detect_supervisor packs TWO values (kind, detail) onto the one stdout
|
||||
# line that survives its $(...) call — see the constraints comment above that function. This is
|
||||
# the separator between them: the ASCII "unit separator" byte, chosen because it never occurs in
|
||||
# any of the prose detail strings and needs no escaping in a `case`/glob pattern.
|
||||
SUPERVISOR_DETAIL_SEP=$'\x1f'
|
||||
|
||||
# fleetd #492 follow-up: set by systemd_loaded/systemd_installed when the underlying `systemctl`
|
||||
# call could not answer cleanly — it exited non-zero AND wrote something to stderr, which is a real
|
||||
# tool failure (e.g. it cannot reach the user bus over a non-lingering ssh session), never the same
|
||||
# fact as a clean negative answer ("not active", no stderr). Initialized here, not just inside the
|
||||
# probes, so detect_supervisor can read them under `set -u` even before either probe has ever run,
|
||||
# and so a test that stubs a probe with a plain `return 0`/`return 1` body (leaving these untouched)
|
||||
# reads a deterministic 0 rather than whatever a previous probe call left behind.
|
||||
SYSTEMD_LOADED_ERRORED=0
|
||||
SYSTEMD_INSTALLED_ERRORED=0
|
||||
# fleetd #492 follow-up: SUPERVISOR_UNCLEAR_DETAIL is the specific supervisor/reason that
|
||||
# require_drivable_supervisor's die() names on an "unclear" answer. Deliberately NOT pre-declared
|
||||
# here (unlike the two flags above): it is set only by the real call site, right after it unpacks
|
||||
# detect_supervisor's stdout (see the constraints comment above detect_supervisor). If that call
|
||||
# site is ever skipped or broken, a bare `set -u` reference to this variable in
|
||||
# require_drivable_supervisor must fail loudly with "unbound variable" — a pre-declared empty
|
||||
# default would instead silently print an empty reason, hiding exactly the value this ticket
|
||||
# exists to surface.
|
||||
|
||||
DO_BUILD=1; ASSUME_YES=0; CHECK_ONLY=0
|
||||
for arg in "$@"; do
|
||||
case "$arg" in
|
||||
@@ -100,39 +128,128 @@ launchd_loaded() { launchctl list "$LAUNCHD_LABEL" >/dev/null 2>&1; }
|
||||
# (this repo is developed on macOS) can substitute each one independently — the same seam
|
||||
# launchd_installed/launchd_loaded above already use.
|
||||
#
|
||||
# fleetd #492 follow-up: both functions used to throw `systemctl`'s stderr straight into
|
||||
# /dev/null, which meant "systemctl answered no" and "systemctl could not answer at all" (e.g. it
|
||||
# cannot reach the user bus over a non-lingering ssh session) looked identical — both a plain
|
||||
# nonzero exit. They now capture stderr separately and set their own *_ERRORED flag ONLY when the
|
||||
# call exited non-zero AND wrote something to stderr — a real tool failure, never a clean "not
|
||||
# installed"/"not active" answer (which exits non-zero with empty stderr). detect_supervisor reads
|
||||
# the flag right after calling the probe, so a probe that could not answer routes to "unclear",
|
||||
# never silently becomes "none".
|
||||
#
|
||||
# "installed": a unit FILE by this name exists, regardless of its current state — the systemd
|
||||
# analogue of the plist file existing on disk. `list-unit-files` reads unit definitions without
|
||||
# depending on runtime state, so this stays read-only and safe under --check.
|
||||
systemd_installed() {
|
||||
command -v systemctl >/dev/null 2>&1 \
|
||||
&& systemctl --user list-unit-files "$SYSTEMD_UNIT.service" --no-legend 2>/dev/null | grep -q .
|
||||
SYSTEMD_INSTALLED_ERRORED=0
|
||||
command -v systemctl >/dev/null 2>&1 || return 1
|
||||
local err_file out rc=0
|
||||
if ! err_file="$(mktemp -t systemd-installed-err)"; then
|
||||
SYSTEMD_INSTALLED_ERRORED=1
|
||||
return 1
|
||||
fi
|
||||
out="$(systemctl --user list-unit-files "$SYSTEMD_UNIT.service" --no-legend 2>"$err_file")" || rc=$?
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
if [ -s "$err_file" ]; then
|
||||
SYSTEMD_INSTALLED_ERRORED=1
|
||||
fi
|
||||
rm -f "$err_file"
|
||||
return "$rc"
|
||||
fi
|
||||
rm -f "$err_file"
|
||||
printf '%s' "$out" | grep -q .
|
||||
}
|
||||
# "loaded": systemd currently supervises this unit as an active job — the systemd analogue of
|
||||
# `launchctl list <label>` succeeding. Measured on the second host: `systemctl --user is-active
|
||||
# fleetd` -> "active".
|
||||
# fleetd` -> "active". A clean "no" (inactive/failed/activating/deactivating) exits non-zero with
|
||||
# nothing on stderr; a probe that could not reach systemd at all exits non-zero WITH a stderr
|
||||
# message — see the fleetd #492 follow-up note above.
|
||||
systemd_loaded() {
|
||||
command -v systemctl >/dev/null 2>&1 && systemctl --user is-active "$SYSTEMD_UNIT" >/dev/null 2>&1
|
||||
SYSTEMD_LOADED_ERRORED=0
|
||||
command -v systemctl >/dev/null 2>&1 || return 1
|
||||
local err_file rc=0
|
||||
if ! err_file="$(mktemp -t systemd-loaded-err)"; then
|
||||
SYSTEMD_LOADED_ERRORED=1
|
||||
return 1
|
||||
fi
|
||||
systemctl --user is-active "$SYSTEMD_UNIT" >/dev/null 2>"$err_file" || rc=$?
|
||||
if [ "$rc" -ne 0 ] && [ -s "$err_file" ]; then
|
||||
SYSTEMD_LOADED_ERRORED=1
|
||||
fi
|
||||
rm -f "$err_file"
|
||||
return "$rc"
|
||||
}
|
||||
|
||||
# fleetd #492: three real answers, not two — launchd, systemd, or genuinely unsupervised — plus a
|
||||
# fourth, "ambiguous", for the one case this script cannot tell apart: both signals firing at once.
|
||||
# That is exactly "I cannot tell who supervises this process", and guessing wrong here is how two
|
||||
# daemons end up running against one herdr session (see trap 7 in the header). Pure and
|
||||
# side-effect-free: reads the two probes above and decides — never mutates anything, so it is safe
|
||||
# under --check and testable by overriding launchd_loaded/systemd_loaded after sourcing.
|
||||
# daemons end up running against one herdr session (see trap 7 in the header).
|
||||
#
|
||||
# fleetd #492 follow-up: a fifth answer, "unclear", for two more situations that must NEVER be read
|
||||
# as "none" (measured — see the report this ticket is a follow-up to):
|
||||
# - installed-but-not-loaded, on EITHER supervisor. `systemctl --user is-active` answers "no" for
|
||||
# `activating`, `deactivating`, `failed`, and while an auto-restart is pending — every one of
|
||||
# those is a host that IS under systemd (or launchd) and whose supervisor is about to act again.
|
||||
# `*_installed` already knows the unit/agent exists; this is the first place that fact is
|
||||
# actually consulted in the decision, not just printed as a warning.
|
||||
# - a probe that could not answer at all. systemd_loaded/systemd_installed set their own
|
||||
# *_ERRORED flag (see the comment above them) when `systemctl` exits non-zero WITH a stderr
|
||||
# message — a real tool failure, e.g. it cannot reach the user bus over a non-lingering ssh
|
||||
# session — never conflated with a clean negative answer.
|
||||
# "none" now means only: neither supervisor is installed, neither is loaded, and neither probe
|
||||
# errored.
|
||||
#
|
||||
# fleetd #492 follow-up — constraints every caller of this function depends on (learned the hard
|
||||
# way: an earlier version of this fix set a SUPERVISOR_UNCLEAR_DETAIL global from inside here and
|
||||
# it was silently lost, because every real call site invokes this as `$(detect_supervisor)`):
|
||||
# 1. It is called as `$(detect_supervisor)`, so ONLY STDOUT crosses back to the caller. Anything
|
||||
# this function needs to tell its caller — the "unclear" detail included — must be printed,
|
||||
# never assigned to a global: a global set inside a `$( )` subshell dies with that subshell.
|
||||
# This function packs BOTH values (kind and detail) onto that one stdout line, joined by
|
||||
# $SUPERVISOR_DETAIL_SEP, and the caller unpacks them on its own side of the subshell boundary.
|
||||
# 2. This script runs under `set -euo pipefail` (line 50), so an unset variable is a loud
|
||||
# failure. Do not add a `${VAR:-default}` anywhere downstream to paper over a value that
|
||||
# should always be there — that hides a lost value instead of surfacing it (fleetd #497's
|
||||
# defect class).
|
||||
# 3. Every `case` on this function's return value needs an explicit final `*)` arm, chosen by
|
||||
# whether that caller ACTS on the value (`die` — an unrecognised value must never be silently
|
||||
# driven) or only DISPLAYS it (`echo`/`warn` and continue — a diagnostic must not go silent on
|
||||
# exactly the value it most needs to report).
|
||||
#
|
||||
# Pure and side-effect-free besides the two *_ERRORED flags (read back within this same call, never
|
||||
# by the caller — see the constraints above): reads the four probes and decides — never mutates
|
||||
# anything, so it is safe under --check and testable by overriding
|
||||
# launchd_installed/launchd_loaded/systemd_installed/systemd_loaded after sourcing.
|
||||
detect_supervisor() {
|
||||
local ld=0 sd=0
|
||||
local ld=0 sd=0 li=0 si=0 kind detail=""
|
||||
SYSTEMD_LOADED_ERRORED=0
|
||||
SYSTEMD_INSTALLED_ERRORED=0
|
||||
|
||||
launchd_loaded && ld=1
|
||||
systemd_loaded && sd=1
|
||||
if [ "$ld" = 1 ] && [ "$sd" = 1 ]; then
|
||||
echo "ambiguous"
|
||||
launchd_installed && li=1
|
||||
systemd_installed && si=1
|
||||
|
||||
if [ "$SYSTEMD_LOADED_ERRORED" = 1 ] || [ "$SYSTEMD_INSTALLED_ERRORED" = 1 ]; then
|
||||
detail="the systemd --user probe for '$SYSTEMD_UNIT' could not answer cleanly (systemctl exited non-zero and reported an error on stderr, not a clean negative — e.g. it cannot reach the user bus)"
|
||||
kind="unclear"
|
||||
elif [ "$ld" = 1 ] && [ "$sd" = 1 ]; then
|
||||
kind="ambiguous"
|
||||
elif [ "$li" = 1 ] && [ "$ld" = 0 ]; then
|
||||
detail="the launchd agent ($LAUNCHD_LABEL) is installed ($LAUNCHD_PLIST exists) but is not currently loaded"
|
||||
kind="unclear"
|
||||
elif [ "$si" = 1 ] && [ "$sd" = 0 ]; then
|
||||
detail="the systemd --user unit ($SYSTEMD_UNIT) is installed but not currently active — it may be activating, deactivating, failed, or waiting on an auto-restart"
|
||||
kind="unclear"
|
||||
elif [ "$ld" = 1 ]; then
|
||||
echo "launchd"
|
||||
kind="launchd"
|
||||
elif [ "$sd" = 1 ]; then
|
||||
echo "systemd"
|
||||
kind="systemd"
|
||||
else
|
||||
echo "none"
|
||||
kind="none"
|
||||
fi
|
||||
|
||||
printf '%s%s%s' "$kind" "$SUPERVISOR_DETAIL_SEP" "$detail"
|
||||
}
|
||||
|
||||
# fleetd #492: turns anything detect_supervisor returns that is NOT exactly one of the two
|
||||
@@ -150,6 +267,19 @@ require_drivable_supervisor() {
|
||||
OLD jar out from under it — the exact failure this ticket (fleetd #492) exists to
|
||||
prevent. Stop one of the two supervisors by hand, confirm only one remains loaded, then
|
||||
rerun." ;;
|
||||
unclear)
|
||||
# fleetd #492 follow-up: SUPERVISOR_UNCLEAR_DETAIL crosses back from detect_supervisor's
|
||||
# subshell via its stdout, unpacked by the caller BEFORE it calls this function (see the
|
||||
# constraints comment above detect_supervisor). No ${VAR:-default} here on purpose: if the
|
||||
# detail is somehow missing, `set -u` makes this reference fail loudly instead of silently
|
||||
# naming nothing — a default that hides a lost value is the same defect class as fleetd
|
||||
# #497.
|
||||
die "a supervisor looks present but this script cannot tell whether it actually drives this
|
||||
daemon: $SUPERVISOR_UNCLEAR_DETAIL. Guessing wrong here is the same failure 'ambiguous'
|
||||
above exists to prevent: driving the daemon while an unseen supervisor revives the OLD
|
||||
jar out from under it (fleetd #492). Check 'launchctl list $LAUNCHD_LABEL' and
|
||||
'systemctl --user status $SYSTEMD_UNIT' by hand, resolve whichever looks unclear, then
|
||||
rerun." ;;
|
||||
*)
|
||||
die "detect_supervisor returned an unrecognized value '$kind' — refusing to guess which
|
||||
supervisor, if any, controls this daemon." ;;
|
||||
@@ -301,7 +431,12 @@ fi
|
||||
# outright — before touching anything — if that cannot be told apart (see require_drivable_
|
||||
# supervisor above). --check reaches this same line, so a host with an undrivable supervisor is
|
||||
# reported as a failure even in --check, without ever reaching the build/stop/start steps.
|
||||
SUPERVISOR_KIND="$(detect_supervisor)"
|
||||
# fleetd #492 follow-up: detect_supervisor runs as $(...), so only the printed line survives —
|
||||
# unpack kind and detail from it HERE, in this shell, before calling anything downstream. See the
|
||||
# constraints comment above detect_supervisor for why this cannot be done any other way.
|
||||
SUPERVISOR_RAW="$(detect_supervisor)"
|
||||
SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}"
|
||||
SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}"
|
||||
require_drivable_supervisor "$SUPERVISOR_KIND"
|
||||
ok "supervisor detected: $SUPERVISOR_KIND"
|
||||
SUPERVISED=0
|
||||
@@ -320,6 +455,13 @@ case "$SUPERVISOR_KIND" in
|
||||
none)
|
||||
warn "no supervisor loaded — this script is the only thing that will restart the daemon."
|
||||
;;
|
||||
*)
|
||||
# fleetd #492 follow-up: this block only DISPLAYS state, it changes nothing yet — so a value
|
||||
# it doesn't recognise gets reported, not an abort that goes silent on exactly the state most
|
||||
# worth seeing. (Unreachable today: require_drivable_supervisor above already died on
|
||||
# "ambiguous"/"unclear" before this case runs. Guards the value nobody has invented yet.)
|
||||
warn "unrecognised supervisor kind: '$SUPERVISOR_KIND' — detect_supervisor returned a value this block does not know; continuing to report the rest of the state."
|
||||
;;
|
||||
esac
|
||||
|
||||
# The trap with no log line. Checked in a LOGIN shell, because that is how the daemon is started
|
||||
@@ -434,6 +576,14 @@ if [ -n "$OLD_PID" ]; then
|
||||
none)
|
||||
kill "$OLD_PID"
|
||||
;;
|
||||
*)
|
||||
# fleetd #492 follow-up: this block ACTS (stops the daemon one specific way per kind) — an
|
||||
# unrecognised value must never fall through to a default action, silently picking the wrong
|
||||
# one (or none at all) while reporting success. (Unreachable today: require_drivable_
|
||||
# supervisor already died before this runs. Guards the value nobody has invented yet.)
|
||||
die "detect_supervisor returned an unrecognized value '$SUPERVISOR_KIND' at the stop step —
|
||||
refusing to guess how to stop a daemon under an unknown supervisor. The daemon was NOT
|
||||
stopped." ;;
|
||||
esac
|
||||
for _ in $(seq "$STOP_WAIT"); do
|
||||
[ -z "$(running_pid)" ] && break
|
||||
@@ -507,6 +657,14 @@ case "$SUPERVISOR_KIND" in
|
||||
# Absolute jar path so `ps` names which checkout is running.
|
||||
( cd "$MODULE" && zsh -lc "nohup java -jar '$JAR' >> fleetd.out 2>&1 &" )
|
||||
;;
|
||||
*)
|
||||
# fleetd #492 follow-up: this block ACTS (starts the daemon one specific way per kind) — an
|
||||
# unrecognised value must never fall through to a default action, silently picking the wrong
|
||||
# one (or none at all) while reporting success. (Unreachable today: require_drivable_
|
||||
# supervisor already died before this runs. Guards the value nobody has invented yet.)
|
||||
die "detect_supervisor returned an unrecognized value '$SUPERVISOR_KIND' at the start step —
|
||||
refusing to guess how to start a daemon under an unknown supervisor. The daemon was NOT
|
||||
started." ;;
|
||||
esac
|
||||
|
||||
for _ in $(seq 10); do
|
||||
|
||||
@@ -25,6 +25,17 @@ classify_fixture() {
|
||||
classify_amqp_connection_errors "$TMP/$name"
|
||||
}
|
||||
|
||||
# fleetd #492 follow-up: detect_supervisor's stdout is now "kind<SEP>detail" (see the constraints
|
||||
# comment above detect_supervisor in redeploy-fleetd.sh) — every test below that only cares about
|
||||
# the kind must split it out with the SAME in-shell parameter expansion the real call site (:438)
|
||||
# uses, never a bare string comparison against the raw output.
|
||||
supervisor_kind_of() {
|
||||
printf '%s' "${1%%"$SUPERVISOR_DETAIL_SEP"*}"
|
||||
}
|
||||
supervisor_detail_of() {
|
||||
printf '%s' "${1#*"$SUPERVISOR_DETAIL_SEP"}"
|
||||
}
|
||||
|
||||
|
||||
# fleetd #492 — supervisor detection. Detect_supervisor() reads launchd_loaded/systemd_loaded, so
|
||||
# each test overrides BOTH pairs (installed + loaded) explicitly, rather than relying on either
|
||||
@@ -37,7 +48,7 @@ test_detect_supervisor_launchd_only() {
|
||||
launchd_loaded() { return 0; }
|
||||
systemd_installed() { return 1; }
|
||||
systemd_loaded() { return 1; }
|
||||
assert_equals "launchd" "$(detect_supervisor)" "launchd-only detection"
|
||||
assert_equals "launchd" "$(supervisor_kind_of "$(detect_supervisor)")" "launchd-only detection"
|
||||
}
|
||||
|
||||
test_detect_supervisor_systemd_only() {
|
||||
@@ -45,7 +56,7 @@ test_detect_supervisor_systemd_only() {
|
||||
launchd_loaded() { return 1; }
|
||||
systemd_installed() { return 0; }
|
||||
systemd_loaded() { return 0; }
|
||||
assert_equals "systemd" "$(detect_supervisor)" "systemd-only detection"
|
||||
assert_equals "systemd" "$(supervisor_kind_of "$(detect_supervisor)")" "systemd-only detection"
|
||||
}
|
||||
|
||||
test_detect_supervisor_none() {
|
||||
@@ -53,7 +64,105 @@ test_detect_supervisor_none() {
|
||||
launchd_loaded() { return 1; }
|
||||
systemd_installed() { return 1; }
|
||||
systemd_loaded() { return 1; }
|
||||
assert_equals "none" "$(detect_supervisor)" "unsupervised detection"
|
||||
assert_equals "none" "$(supervisor_kind_of "$(detect_supervisor)")" "unsupervised detection"
|
||||
}
|
||||
|
||||
# fleetd #492 follow-up — detect_supervisor must never answer "none" when the truth is "could not
|
||||
# tell". `systemd_installed`/`systemd_loaded` already know a unit file exists; this proves that
|
||||
# fact is now actually consulted, not just printed as a warning: an installed-but-not-loaded unit
|
||||
# reads as unclear, because is-active answers "no" for activating/deactivating/failed/pending
|
||||
# auto-restart too, and every one of those is a host that IS under systemd.
|
||||
test_detect_supervisor_systemd_installed_not_loaded_is_unclear() {
|
||||
launchd_installed() { return 1; }
|
||||
launchd_loaded() { return 1; }
|
||||
systemd_installed() { return 0; } # the unit file IS there
|
||||
systemd_loaded() { return 1; } # is-active says no — could be activating/failed/pending restart
|
||||
local raw
|
||||
raw="$(detect_supervisor)"
|
||||
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "systemd installed-but-not-loaded must read as unclear, not none"
|
||||
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$SYSTEMD_UNIT" \
|
||||
|| fail "detail does not name the systemd unit it found installed-but-not-loaded"
|
||||
}
|
||||
|
||||
# Same fact, the launchd side: a plist on disk that is not currently loaded (unloaded without being
|
||||
# removed, or about to be reloaded) must not read as "no supervisor" either.
|
||||
test_detect_supervisor_launchd_installed_not_loaded_is_unclear() {
|
||||
launchd_installed() { return 0; } # the plist IS there
|
||||
launchd_loaded() { return 1; } # launchctl list says not loaded
|
||||
systemd_installed() { return 1; }
|
||||
systemd_loaded() { return 1; }
|
||||
local raw
|
||||
raw="$(detect_supervisor)"
|
||||
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "launchd installed-but-not-loaded must read as unclear, not none"
|
||||
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$LAUNCHD_LABEL" \
|
||||
|| fail "detail does not name the launchd label it found installed-but-not-loaded"
|
||||
}
|
||||
|
||||
# Drives the REAL systemd_loaded/systemd_installed bodies (never stubbed) through a `systemctl`
|
||||
# stub placed first on PATH that exits non-zero AND writes to stderr — the shape of a systemctl
|
||||
# that runs but cannot reach the user bus (measured elsewhere as a headless ssh session with no
|
||||
# lingering). This must read as unclear, never none: a probe that could not answer at all is not
|
||||
# the same fact as "no supervisor is loaded".
|
||||
test_detect_supervisor_systemd_probe_error_is_unclear() {
|
||||
# Re-source first to restore the REAL launchd_*/systemd_* probe bodies. Earlier tests in this
|
||||
# file permanently override them with stub `return 0`/`return 1` bodies (that is the whole point
|
||||
# of those tests), and a bash function definition is global for the rest of the process — without
|
||||
# this, systemd_loaded here would still be whatever the previous test left it as, never touching
|
||||
# a real `systemctl` call at all.
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
local bin_dir result rc=0
|
||||
bin_dir="$TMP/stub-bin-systemctl-errors"
|
||||
mkdir -p "$bin_dir"
|
||||
cat > "$bin_dir/systemctl" <<'STUB'
|
||||
#!/usr/bin/env bash
|
||||
echo "Failed to connect to bus: No such file or directory" >&2
|
||||
exit 1
|
||||
STUB
|
||||
chmod +x "$bin_dir/systemctl"
|
||||
|
||||
PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$?
|
||||
[ "$rc" -ne 0 ] || fail "systemd_loaded must not report loaded=true when systemctl only errored"
|
||||
assert_equals "1" "$SYSTEMD_LOADED_ERRORED" "systemd_loaded must flag a probe error, not a clean negative"
|
||||
|
||||
launchd_installed() { return 1; }
|
||||
launchd_loaded() { return 1; }
|
||||
result="$(PATH="$bin_dir:$PATH" detect_supervisor)"
|
||||
assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe error must read as unclear, not none"
|
||||
printf '%s' "$(supervisor_detail_of "$result")" | grep -qF "$SYSTEMD_UNIT" \
|
||||
|| fail "detail does not name the systemd unit whose probe errored"
|
||||
}
|
||||
|
||||
# fleetd #492 follow-up (Item 1): this must go through the REAL call-site shape at :437-440, not a
|
||||
# hand-constructed "unclear" value — a test that builds "unclear" directly proves the switch, not
|
||||
# the handoff, and that is exactly the gap that let SUPERVISOR_UNCLEAR_DETAIL never reach the real
|
||||
# caller in b17f37a. detect_supervisor runs as $(detect_supervisor): a subshell. Only stdout
|
||||
# survives that boundary, so kind AND detail must both cross on it — this test proves they do.
|
||||
test_require_drivable_supervisor_refuses_unclear() {
|
||||
launchd_installed() { return 1; }
|
||||
launchd_loaded() { return 1; }
|
||||
systemd_installed() { return 0; }
|
||||
systemd_loaded() { return 1; }
|
||||
|
||||
local SUPERVISOR_RAW SUPERVISOR_KIND SUPERVISOR_UNCLEAR_DETAIL output rc=0
|
||||
# Exactly what :437-439 does — do not shortcut this by constructing "unclear" by hand.
|
||||
SUPERVISOR_RAW="$(detect_supervisor)"
|
||||
SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}"
|
||||
SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}"
|
||||
|
||||
assert_equals "unclear" "$SUPERVISOR_KIND" "setup: expected unclear before testing the refusal"
|
||||
[ -n "$SUPERVISOR_UNCLEAR_DETAIL" ] \
|
||||
|| fail "detail did not survive the \$(...) call-site boundary — SUPERVISOR_UNCLEAR_DETAIL is empty in the parent shell"
|
||||
printf '%s' "$SUPERVISOR_UNCLEAR_DETAIL" | grep -qF "$SYSTEMD_UNIT" \
|
||||
|| fail "detail that crossed the subshell boundary does not name the systemd unit it found installed-but-not-loaded"
|
||||
|
||||
output="$(require_drivable_supervisor "$SUPERVISOR_KIND" 2>&1)" || rc=$?
|
||||
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an unclear (undrivable) supervisor"
|
||||
# Check for the ACTUAL DETAIL TEXT, not just "$SYSTEMD_UNIT" — the die() message's boilerplate
|
||||
# recovery instructions name the unit unconditionally either way ("systemctl --user status
|
||||
# $SYSTEMD_UNIT"), so a bare unit-name grep here would pass even on a lost/fallback detail. Only
|
||||
# the specific detail string proves the crossed value, not the boilerplate, reached the message.
|
||||
printf '%s' "$output" | grep -qF "$SUPERVISOR_UNCLEAR_DETAIL" \
|
||||
|| fail "refusal message does not contain the specific detail that crossed the subshell boundary"
|
||||
}
|
||||
|
||||
# The heart of the ticket: a supervisor this script cannot drive must refuse, never fall through to
|
||||
@@ -299,7 +408,11 @@ test_unattributable_quiet_mutation_is_caught() {
|
||||
test_detect_supervisor_launchd_only
|
||||
test_detect_supervisor_systemd_only
|
||||
test_detect_supervisor_none
|
||||
test_detect_supervisor_systemd_installed_not_loaded_is_unclear
|
||||
test_detect_supervisor_launchd_installed_not_loaded_is_unclear
|
||||
test_detect_supervisor_systemd_probe_error_is_unclear
|
||||
test_require_drivable_supervisor_refuses_ambiguous
|
||||
test_require_drivable_supervisor_refuses_unclear
|
||||
test_require_drivable_supervisor_accepts_known_kinds
|
||||
test_count_daemon_pids
|
||||
test_assert_single_daemon_accepts_one_pid
|
||||
|
||||
Reference in New Issue
Block a user