|
|
|
@@ -5,7 +5,7 @@
|
|
|
|
|
# A merge is not a deployment: the running daemon holds the jar it was started with, so code merged
|
|
|
|
|
# to main does nothing until this runs. See CLAUDE.md -> "Redeploying the daemon".
|
|
|
|
|
#
|
|
|
|
|
# This script exists to turn six remembered traps into one auditable command:
|
|
|
|
|
# This script exists to turn eight remembered traps into one auditable command:
|
|
|
|
|
#
|
|
|
|
|
# 1. A piped `mvn` hides BUILD FAILURE behind a zero exit, so the build here is never piped.
|
|
|
|
|
# 2. The daemon must start from a LOGIN shell, or the tokens it hands to members are empty:
|
|
|
|
@@ -27,6 +27,21 @@
|
|
|
|
|
# restart of the OLD jar. So this script detects whether the agent is loaded and, only then,
|
|
|
|
|
# swaps `kill` + manual `nohup` for `launchctl unload`/`load` — the one supervisor in control
|
|
|
|
|
# at any moment is whichever one you asked to act, never both.
|
|
|
|
|
# 7. fleetd #492 — a systemd --user unit is a THIRD possible supervisor (seen on a second host):
|
|
|
|
|
# Restart=on-failure treats this JVM's SIGTERM exit code (143, per CB-594 above) as a failure
|
|
|
|
|
# too, so a bare `kill` there would race systemd's own restart of the OLD jar exactly like
|
|
|
|
|
# launchd would. This script now tells launchd, systemd, and "genuinely unsupervised" apart as
|
|
|
|
|
# three different answers, drives whichever one it finds through its own control plane
|
|
|
|
|
# (`launchctl` / `systemctl --user`), and REFUSES outright — never falls back to `kill` — when
|
|
|
|
|
# it finds a supervision signal it cannot map to exactly one of the two it knows how to drive.
|
|
|
|
|
# A wrong guess here is how two daemons end up running against one herdr session. Follow-up:
|
|
|
|
|
# "not currently loaded" is not the same fact as "unsupervised" — a unit that is installed but
|
|
|
|
|
# activating/failed/pending-restart, or a `systemctl` call that could not answer at all (e.g.
|
|
|
|
|
# no user-bus access), both now read as a fifth answer, "unclear", and REFUSE the same way
|
|
|
|
|
# "ambiguous" does, rather than silently falling through to "none".
|
|
|
|
|
# 8. fleetd #492 — a post-restart check counts running fleetd processes and fails the whole run if
|
|
|
|
|
# more than one is alive. That is the one thing none of the checks above (healthz 200, jar id,
|
|
|
|
|
# the fresh "listening" line) can see: every one of them is satisfied by EITHER daemon.
|
|
|
|
|
#
|
|
|
|
|
# Usage:
|
|
|
|
|
# scripts/redeploy-fleetd.sh # build, confirm, restart, verify
|
|
|
|
@@ -55,13 +70,42 @@ HEALTH_WAIT=60 # seconds to wait for /healthz to answer after start
|
|
|
|
|
LAUNCHD_LABEL='dev.ltms.fleetd'
|
|
|
|
|
LAUNCHD_PLIST="$HOME/Library/LaunchAgents/$LAUNCHD_LABEL.plist"
|
|
|
|
|
|
|
|
|
|
# fleetd #492: the systemd --user unit this script must not fight with either (see trap 7 above).
|
|
|
|
|
# Measured on the second host: `systemctl --user cat fleetd` names the unit "fleetd" (not
|
|
|
|
|
# "dev.ltms.fleetd" — systemd user units here are not namespaced the way the launchd label is).
|
|
|
|
|
SYSTEMD_UNIT='fleetd'
|
|
|
|
|
|
|
|
|
|
# fleetd #492 follow-up: detect_supervisor packs TWO values (kind, detail) onto the one stdout
|
|
|
|
|
# line that survives its $(...) call — see the constraints comment above that function. This is
|
|
|
|
|
# the separator between them: the ASCII "unit separator" byte, chosen because it never occurs in
|
|
|
|
|
# any of the prose detail strings and needs no escaping in a `case`/glob pattern.
|
|
|
|
|
SUPERVISOR_DETAIL_SEP=$'\x1f'
|
|
|
|
|
|
|
|
|
|
# fleetd #492 follow-up: set by systemd_loaded/systemd_installed when the underlying `systemctl`
|
|
|
|
|
# call could not answer cleanly — it exited non-zero AND wrote something to stderr, which is a real
|
|
|
|
|
# tool failure (e.g. it cannot reach the user bus over a non-lingering ssh session), never the same
|
|
|
|
|
# fact as a clean negative answer ("not active", no stderr). Initialized here, not just inside the
|
|
|
|
|
# probes, so detect_supervisor can read them under `set -u` even before either probe has ever run,
|
|
|
|
|
# and so a test that stubs a probe with a plain `return 0`/`return 1` body (leaving these untouched)
|
|
|
|
|
# reads a deterministic 0 rather than whatever a previous probe call left behind.
|
|
|
|
|
SYSTEMD_LOADED_ERRORED=0
|
|
|
|
|
SYSTEMD_INSTALLED_ERRORED=0
|
|
|
|
|
# fleetd #492 follow-up: SUPERVISOR_UNCLEAR_DETAIL is the specific supervisor/reason that
|
|
|
|
|
# require_drivable_supervisor's die() names on an "unclear" answer. Deliberately NOT pre-declared
|
|
|
|
|
# here (unlike the two flags above): it is set only by the real call site, right after it unpacks
|
|
|
|
|
# detect_supervisor's stdout (see the constraints comment above detect_supervisor). If that call
|
|
|
|
|
# site is ever skipped or broken, a bare `set -u` reference to this variable in
|
|
|
|
|
# require_drivable_supervisor must fail loudly with "unbound variable" — a pre-declared empty
|
|
|
|
|
# default would instead silently print an empty reason, hiding exactly the value this ticket
|
|
|
|
|
# exists to surface.
|
|
|
|
|
|
|
|
|
|
DO_BUILD=1; ASSUME_YES=0; CHECK_ONLY=0
|
|
|
|
|
for arg in "$@"; do
|
|
|
|
|
case "$arg" in
|
|
|
|
|
--yes|-y) ASSUME_YES=1 ;;
|
|
|
|
|
--no-build) DO_BUILD=0 ;;
|
|
|
|
|
--check) CHECK_ONLY=1 ;;
|
|
|
|
|
-h|--help) sed -n '3,37p' "${BASH_SOURCE[0]}"; exit 0 ;;
|
|
|
|
|
-h|--help) sed -n '3,48p' "${BASH_SOURCE[0]}"; exit 0 ;;
|
|
|
|
|
*) echo "unknown option: $arg (try --help)" >&2; exit 2 ;;
|
|
|
|
|
esac
|
|
|
|
|
done
|
|
|
|
@@ -79,6 +123,193 @@ running_pid() { pgrep -f "$PATTERN" || true; }
|
|
|
|
|
launchd_installed() { [ -f "$LAUNCHD_PLIST" ]; }
|
|
|
|
|
launchd_loaded() { launchctl list "$LAUNCHD_LABEL" >/dev/null 2>&1; }
|
|
|
|
|
|
|
|
|
|
# fleetd #492: same two questions for systemd --user. Kept as separate, overridable functions
|
|
|
|
|
# (never an inline `systemctl` call at each use site) so a test on a box with no systemd at all
|
|
|
|
|
# (this repo is developed on macOS) can substitute each one independently — the same seam
|
|
|
|
|
# launchd_installed/launchd_loaded above already use.
|
|
|
|
|
#
|
|
|
|
|
# fleetd #492 follow-up: both functions used to throw `systemctl`'s stderr straight into
|
|
|
|
|
# /dev/null, which meant "systemctl answered no" and "systemctl could not answer at all" (e.g. it
|
|
|
|
|
# cannot reach the user bus over a non-lingering ssh session) looked identical — both a plain
|
|
|
|
|
# nonzero exit. They now capture stderr separately and set their own *_ERRORED flag ONLY when the
|
|
|
|
|
# call exited non-zero AND wrote something to stderr — a real tool failure, never a clean "not
|
|
|
|
|
# installed"/"not active" answer (which exits non-zero with empty stderr). detect_supervisor reads
|
|
|
|
|
# the flag right after calling the probe, so a probe that could not answer routes to "unclear",
|
|
|
|
|
# never silently becomes "none".
|
|
|
|
|
#
|
|
|
|
|
# "installed": a unit FILE by this name exists, regardless of its current state — the systemd
|
|
|
|
|
# analogue of the plist file existing on disk. `list-unit-files` reads unit definitions without
|
|
|
|
|
# depending on runtime state, so this stays read-only and safe under --check.
|
|
|
|
|
systemd_installed() {
|
|
|
|
|
SYSTEMD_INSTALLED_ERRORED=0
|
|
|
|
|
command -v systemctl >/dev/null 2>&1 || return 1
|
|
|
|
|
local err_file out rc=0
|
|
|
|
|
if ! err_file="$(mktemp -t systemd-installed-err)"; then
|
|
|
|
|
SYSTEMD_INSTALLED_ERRORED=1
|
|
|
|
|
return 1
|
|
|
|
|
fi
|
|
|
|
|
out="$(systemctl --user list-unit-files "$SYSTEMD_UNIT.service" --no-legend 2>"$err_file")" || rc=$?
|
|
|
|
|
if [ "$rc" -ne 0 ]; then
|
|
|
|
|
if [ -s "$err_file" ]; then
|
|
|
|
|
SYSTEMD_INSTALLED_ERRORED=1
|
|
|
|
|
fi
|
|
|
|
|
rm -f "$err_file"
|
|
|
|
|
return "$rc"
|
|
|
|
|
fi
|
|
|
|
|
rm -f "$err_file"
|
|
|
|
|
printf '%s' "$out" | grep -q .
|
|
|
|
|
}
|
|
|
|
|
# "loaded": systemd currently supervises this unit as an active job — the systemd analogue of
|
|
|
|
|
# `launchctl list <label>` succeeding. Measured on the second host: `systemctl --user is-active
|
|
|
|
|
# fleetd` -> "active". A clean "no" (inactive/failed/activating/deactivating) exits non-zero with
|
|
|
|
|
# nothing on stderr; a probe that could not reach systemd at all exits non-zero WITH a stderr
|
|
|
|
|
# message — see the fleetd #492 follow-up note above.
|
|
|
|
|
systemd_loaded() {
|
|
|
|
|
SYSTEMD_LOADED_ERRORED=0
|
|
|
|
|
command -v systemctl >/dev/null 2>&1 || return 1
|
|
|
|
|
local err_file rc=0
|
|
|
|
|
if ! err_file="$(mktemp -t systemd-loaded-err)"; then
|
|
|
|
|
SYSTEMD_LOADED_ERRORED=1
|
|
|
|
|
return 1
|
|
|
|
|
fi
|
|
|
|
|
systemctl --user is-active "$SYSTEMD_UNIT" >/dev/null 2>"$err_file" || rc=$?
|
|
|
|
|
if [ "$rc" -ne 0 ] && [ -s "$err_file" ]; then
|
|
|
|
|
SYSTEMD_LOADED_ERRORED=1
|
|
|
|
|
fi
|
|
|
|
|
rm -f "$err_file"
|
|
|
|
|
return "$rc"
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
# fleetd #492: three real answers, not two — launchd, systemd, or genuinely unsupervised — plus a
|
|
|
|
|
# fourth, "ambiguous", for the one case this script cannot tell apart: both signals firing at once.
|
|
|
|
|
# That is exactly "I cannot tell who supervises this process", and guessing wrong here is how two
|
|
|
|
|
# daemons end up running against one herdr session (see trap 7 in the header).
|
|
|
|
|
#
|
|
|
|
|
# fleetd #492 follow-up: a fifth answer, "unclear", for two more situations that must NEVER be read
|
|
|
|
|
# as "none" (measured — see the report this ticket is a follow-up to):
|
|
|
|
|
# - installed-but-not-loaded, on EITHER supervisor. `systemctl --user is-active` answers "no" for
|
|
|
|
|
# `activating`, `deactivating`, `failed`, and while an auto-restart is pending — every one of
|
|
|
|
|
# those is a host that IS under systemd (or launchd) and whose supervisor is about to act again.
|
|
|
|
|
# `*_installed` already knows the unit/agent exists; this is the first place that fact is
|
|
|
|
|
# actually consulted in the decision, not just printed as a warning.
|
|
|
|
|
# - a probe that could not answer at all. systemd_loaded/systemd_installed set their own
|
|
|
|
|
# *_ERRORED flag (see the comment above them) when `systemctl` exits non-zero WITH a stderr
|
|
|
|
|
# message — a real tool failure, e.g. it cannot reach the user bus over a non-lingering ssh
|
|
|
|
|
# session — never conflated with a clean negative answer.
|
|
|
|
|
# "none" now means only: neither supervisor is installed, neither is loaded, and neither probe
|
|
|
|
|
# errored.
|
|
|
|
|
#
|
|
|
|
|
# fleetd #492 follow-up — constraints every caller of this function depends on (learned the hard
|
|
|
|
|
# way: an earlier version of this fix set a SUPERVISOR_UNCLEAR_DETAIL global from inside here and
|
|
|
|
|
# it was silently lost, because every real call site invokes this as `$(detect_supervisor)`):
|
|
|
|
|
# 1. It is called as `$(detect_supervisor)`, so ONLY STDOUT crosses back to the caller. Anything
|
|
|
|
|
# this function needs to tell its caller — the "unclear" detail included — must be printed,
|
|
|
|
|
# never assigned to a global: a global set inside a `$( )` subshell dies with that subshell.
|
|
|
|
|
# This function packs BOTH values (kind and detail) onto that one stdout line, joined by
|
|
|
|
|
# $SUPERVISOR_DETAIL_SEP, and the caller unpacks them on its own side of the subshell boundary.
|
|
|
|
|
# 2. This script runs under `set -euo pipefail` (line 50), so an unset variable is a loud
|
|
|
|
|
# failure. Do not add a `${VAR:-default}` anywhere downstream to paper over a value that
|
|
|
|
|
# should always be there — that hides a lost value instead of surfacing it (fleetd #497's
|
|
|
|
|
# defect class).
|
|
|
|
|
# 3. Every `case` on this function's return value needs an explicit final `*)` arm, chosen by
|
|
|
|
|
# whether that caller ACTS on the value (`die` — an unrecognised value must never be silently
|
|
|
|
|
# driven) or only DISPLAYS it (`echo`/`warn` and continue — a diagnostic must not go silent on
|
|
|
|
|
# exactly the value it most needs to report).
|
|
|
|
|
#
|
|
|
|
|
# Pure and side-effect-free besides the two *_ERRORED flags (read back within this same call, never
|
|
|
|
|
# by the caller — see the constraints above): reads the four probes and decides — never mutates
|
|
|
|
|
# anything, so it is safe under --check and testable by overriding
|
|
|
|
|
# launchd_installed/launchd_loaded/systemd_installed/systemd_loaded after sourcing.
|
|
|
|
|
detect_supervisor() {
|
|
|
|
|
local ld=0 sd=0 li=0 si=0 kind detail=""
|
|
|
|
|
SYSTEMD_LOADED_ERRORED=0
|
|
|
|
|
SYSTEMD_INSTALLED_ERRORED=0
|
|
|
|
|
|
|
|
|
|
launchd_loaded && ld=1
|
|
|
|
|
systemd_loaded && sd=1
|
|
|
|
|
launchd_installed && li=1
|
|
|
|
|
systemd_installed && si=1
|
|
|
|
|
|
|
|
|
|
if [ "$SYSTEMD_LOADED_ERRORED" = 1 ] || [ "$SYSTEMD_INSTALLED_ERRORED" = 1 ]; then
|
|
|
|
|
detail="the systemd --user probe for '$SYSTEMD_UNIT' could not answer cleanly (systemctl exited non-zero and reported an error on stderr, not a clean negative — e.g. it cannot reach the user bus)"
|
|
|
|
|
kind="unclear"
|
|
|
|
|
elif [ "$ld" = 1 ] && [ "$sd" = 1 ]; then
|
|
|
|
|
kind="ambiguous"
|
|
|
|
|
elif [ "$li" = 1 ] && [ "$ld" = 0 ]; then
|
|
|
|
|
detail="the launchd agent ($LAUNCHD_LABEL) is installed ($LAUNCHD_PLIST exists) but is not currently loaded"
|
|
|
|
|
kind="unclear"
|
|
|
|
|
elif [ "$si" = 1 ] && [ "$sd" = 0 ]; then
|
|
|
|
|
detail="the systemd --user unit ($SYSTEMD_UNIT) is installed but not currently active — it may be activating, deactivating, failed, or waiting on an auto-restart"
|
|
|
|
|
kind="unclear"
|
|
|
|
|
elif [ "$ld" = 1 ]; then
|
|
|
|
|
kind="launchd"
|
|
|
|
|
elif [ "$sd" = 1 ]; then
|
|
|
|
|
kind="systemd"
|
|
|
|
|
else
|
|
|
|
|
kind="none"
|
|
|
|
|
fi
|
|
|
|
|
|
|
|
|
|
printf '%s%s%s' "$kind" "$SUPERVISOR_DETAIL_SEP" "$detail"
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
# fleetd #492: turns anything detect_supervisor returns that is NOT exactly one of the two
|
|
|
|
|
# supervisors this script knows how to drive into a die() — never a fall-through to the `kill`
|
|
|
|
|
# path. Kept as its own function so a test can call it directly (in a subshell, since it die()s)
|
|
|
|
|
# without running the whole report-state flow or needing a real launchd/systemd.
|
|
|
|
|
require_drivable_supervisor() {
|
|
|
|
|
local kind="$1"
|
|
|
|
|
case "$kind" in
|
|
|
|
|
launchd|systemd|none) ;;
|
|
|
|
|
ambiguous)
|
|
|
|
|
die "both launchd ($LAUNCHD_LABEL) and systemd --user ($SYSTEMD_UNIT) report themselves as
|
|
|
|
|
loaded for this daemon at the same time. This script cannot tell which one actually
|
|
|
|
|
supervises the running process, and driving either alone risks the OTHER reviving the
|
|
|
|
|
OLD jar out from under it — the exact failure this ticket (fleetd #492) exists to
|
|
|
|
|
prevent. Stop one of the two supervisors by hand, confirm only one remains loaded, then
|
|
|
|
|
rerun." ;;
|
|
|
|
|
unclear)
|
|
|
|
|
# fleetd #492 follow-up: SUPERVISOR_UNCLEAR_DETAIL crosses back from detect_supervisor's
|
|
|
|
|
# subshell via its stdout, unpacked by the caller BEFORE it calls this function (see the
|
|
|
|
|
# constraints comment above detect_supervisor). No ${VAR:-default} here on purpose: if the
|
|
|
|
|
# detail is somehow missing, `set -u` makes this reference fail loudly instead of silently
|
|
|
|
|
# naming nothing — a default that hides a lost value is the same defect class as fleetd
|
|
|
|
|
# #497.
|
|
|
|
|
die "a supervisor looks present but this script cannot tell whether it actually drives this
|
|
|
|
|
daemon: $SUPERVISOR_UNCLEAR_DETAIL. Guessing wrong here is the same failure 'ambiguous'
|
|
|
|
|
above exists to prevent: driving the daemon while an unseen supervisor revives the OLD
|
|
|
|
|
jar out from under it (fleetd #492). Check 'launchctl list $LAUNCHD_LABEL' and
|
|
|
|
|
'systemctl --user status $SYSTEMD_UNIT' by hand, resolve whichever looks unclear, then
|
|
|
|
|
rerun." ;;
|
|
|
|
|
*)
|
|
|
|
|
die "detect_supervisor returned an unrecognized value '$kind' — refusing to guess which
|
|
|
|
|
supervisor, if any, controls this daemon." ;;
|
|
|
|
|
esac
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
# fleetd #492: the exact symptom a racing supervisor produces — count how many fleetd processes are
|
|
|
|
|
# alive right now. Takes the pid list as a parameter (rather than calling running_pid() itself) so a
|
|
|
|
|
# test can pass a canned two-line string without a real second process running. Pure except for the
|
|
|
|
|
# die() in assert_single_daemon below.
|
|
|
|
|
count_daemon_pids() {
|
|
|
|
|
local pids="$1"
|
|
|
|
|
if [ -z "$pids" ]; then
|
|
|
|
|
echo 0
|
|
|
|
|
else
|
|
|
|
|
printf '%s\n' "$pids" | grep -c .
|
|
|
|
|
fi
|
|
|
|
|
}
|
|
|
|
|
assert_single_daemon() {
|
|
|
|
|
local pids="$1" count
|
|
|
|
|
count="$(count_daemon_pids "$pids")"
|
|
|
|
|
if [ "$count" -gt 1 ]; then
|
|
|
|
|
die "more than one fleetd process is running after this restart (pids: $(printf '%s' "$pids" | tr '\n' ' ')).
|
|
|
|
|
This is the exact failure a racing supervisor produces: the OLD jar was revived by its
|
|
|
|
|
supervisor while this script started a NEW copy. Two daemons on one herdr session kill
|
|
|
|
|
each other's members. Investigate with 'pgrep -f \"$PATTERN\"' and stop the wrong one by
|
|
|
|
|
hand — do not assume either pid is the one you want."
|
|
|
|
|
fi
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
# CB-600: the script computes its own log path from where it sits on disk (REPO, above); the
|
|
|
|
|
# plist hard-codes an absolute StandardOutPath. Nothing forced the two to agree — if this script
|
|
|
|
|
# were ever run from a checkout other than the one the loaded plist names, launchd would start and
|
|
|
|
@@ -182,25 +413,57 @@ fi
|
|
|
|
|
ok "jar on disk: $(jar_id) ($([ -f "$JAR" ] && date -r "$JAR" '+%Y-%m-%d %H:%M:%S' || echo 'none'))"
|
|
|
|
|
ok "HEAD: $(git -C "$REPO" log --oneline -1)"
|
|
|
|
|
|
|
|
|
|
# CB-594: supervision state. Installed and loaded are different facts — a copied-but-never-loaded
|
|
|
|
|
# plist supervises nothing, and a loaded label with no file backing it (rare, but possible after an
|
|
|
|
|
# edited/moved plist) is still what launchd will act on.
|
|
|
|
|
# CB-594 / fleetd #492: supervision state. Installed and loaded are different facts — a
|
|
|
|
|
# copied-but-never-loaded plist (or an unloaded systemd unit) supervises nothing, and a loaded
|
|
|
|
|
# label/unit with no file backing it is still what its supervisor will act on.
|
|
|
|
|
if launchd_installed; then
|
|
|
|
|
ok "launchd agent installed: $LAUNCHD_PLIST"
|
|
|
|
|
else
|
|
|
|
|
warn "launchd agent NOT installed (no supervision — a crash will not restart the daemon)."
|
|
|
|
|
warn "launchd agent NOT installed."
|
|
|
|
|
fi
|
|
|
|
|
SUPERVISED=0
|
|
|
|
|
if launchd_loaded; then
|
|
|
|
|
SUPERVISED=1
|
|
|
|
|
ok "launchd agent loaded ($LAUNCHD_LABEL) — launchd supervises this daemon"
|
|
|
|
|
# CB-600: fail loudly here, before ANY other check runs, if this script and the loaded plist
|
|
|
|
|
# would read different log files — every check after this point is worthless otherwise.
|
|
|
|
|
check_log_path_matches_plist "$OUT" "$LAUNCHD_PLIST"
|
|
|
|
|
if systemd_installed; then
|
|
|
|
|
ok "systemd --user unit installed: $SYSTEMD_UNIT"
|
|
|
|
|
else
|
|
|
|
|
warn "launchd agent not loaded — this script is the only thing that will restart the daemon."
|
|
|
|
|
warn "systemd --user unit NOT installed ($SYSTEMD_UNIT)."
|
|
|
|
|
fi
|
|
|
|
|
|
|
|
|
|
# fleetd #492: decide which of the two (if either) actually supervises this daemon, and refuse
|
|
|
|
|
# outright — before touching anything — if that cannot be told apart (see require_drivable_
|
|
|
|
|
# supervisor above). --check reaches this same line, so a host with an undrivable supervisor is
|
|
|
|
|
# reported as a failure even in --check, without ever reaching the build/stop/start steps.
|
|
|
|
|
# fleetd #492 follow-up: detect_supervisor runs as $(...), so only the printed line survives —
|
|
|
|
|
# unpack kind and detail from it HERE, in this shell, before calling anything downstream. See the
|
|
|
|
|
# constraints comment above detect_supervisor for why this cannot be done any other way.
|
|
|
|
|
SUPERVISOR_RAW="$(detect_supervisor)"
|
|
|
|
|
SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}"
|
|
|
|
|
SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}"
|
|
|
|
|
require_drivable_supervisor "$SUPERVISOR_KIND"
|
|
|
|
|
ok "supervisor detected: $SUPERVISOR_KIND"
|
|
|
|
|
SUPERVISED=0
|
|
|
|
|
case "$SUPERVISOR_KIND" in
|
|
|
|
|
launchd)
|
|
|
|
|
SUPERVISED=1
|
|
|
|
|
ok "launchd agent loaded ($LAUNCHD_LABEL) — launchd supervises this daemon"
|
|
|
|
|
# CB-600: fail loudly here, before ANY other check runs, if this script and the loaded plist
|
|
|
|
|
# would read different log files — every check after this point is worthless otherwise.
|
|
|
|
|
check_log_path_matches_plist "$OUT" "$LAUNCHD_PLIST"
|
|
|
|
|
;;
|
|
|
|
|
systemd)
|
|
|
|
|
SUPERVISED=1
|
|
|
|
|
ok "systemd --user unit active ($SYSTEMD_UNIT) — systemd supervises this daemon"
|
|
|
|
|
;;
|
|
|
|
|
none)
|
|
|
|
|
warn "no supervisor loaded — this script is the only thing that will restart the daemon."
|
|
|
|
|
;;
|
|
|
|
|
*)
|
|
|
|
|
# fleetd #492 follow-up: this block only DISPLAYS state, it changes nothing yet — so a value
|
|
|
|
|
# it doesn't recognise gets reported, not an abort that goes silent on exactly the state most
|
|
|
|
|
# worth seeing. (Unreachable today: require_drivable_supervisor above already died on
|
|
|
|
|
# "ambiguous"/"unclear" before this case runs. Guards the value nobody has invented yet.)
|
|
|
|
|
warn "unrecognised supervisor kind: '$SUPERVISOR_KIND' — detect_supervisor returned a value this block does not know; continuing to report the rest of the state."
|
|
|
|
|
;;
|
|
|
|
|
esac
|
|
|
|
|
|
|
|
|
|
# The trap with no log line. Checked in a LOGIN shell, because that is how the daemon is started
|
|
|
|
|
# below. Never prints the value — only whether it resolved.
|
|
|
|
|
if zsh -lc '[ -n "${WORKER_GITEA_TOKEN:-}" ]' 2>/dev/null; then
|
|
|
|
@@ -281,25 +544,47 @@ fi
|
|
|
|
|
|
|
|
|
|
# ------------------------------------------------------------------ stop
|
|
|
|
|
#
|
|
|
|
|
# CB-594: when SUPERVISED, launchd owns the stop — never a raw `kill` here. A bare SIGTERM makes
|
|
|
|
|
# this JVM exit 143 even with its shutdown hook running to completion (verified separately: a
|
|
|
|
|
# throwaway Java process with an equivalent shutdown hook, sent SIGTERM from a login shell that
|
|
|
|
|
# could `wait` on it directly, reported exit code 143 every time — never 0). launchd's
|
|
|
|
|
# KeepAlive.SuccessfulExit=false treats any nonzero exit as a crash and restarts the OLD jar,
|
|
|
|
|
# which would race this script's own restart of the NEW one. `launchctl unload` avoids that race
|
|
|
|
|
# by deregistering the job first, so no KeepAlive is left armed when the process actually stops.
|
|
|
|
|
# CB-594 / fleetd #492: when SUPERVISED, the supervisor owns the stop — never a raw `kill` here. A
|
|
|
|
|
# bare SIGTERM makes this JVM exit 143 even with its shutdown hook running to completion (verified
|
|
|
|
|
# separately: a throwaway Java process with an equivalent shutdown hook, sent SIGTERM from a login
|
|
|
|
|
# shell that could `wait` on it directly, reported exit code 143 every time — never 0). launchd's
|
|
|
|
|
# KeepAlive.SuccessfulExit=false and systemd's Restart=on-failure both treat any nonzero exit as a
|
|
|
|
|
# crash and restart the OLD jar, which would race this script's own restart of the NEW one.
|
|
|
|
|
# `launchctl unload` avoids that race by deregistering the job first, so no KeepAlive is left
|
|
|
|
|
# armed when the process actually stops. `systemctl --user stop` needs no such dance: unlike
|
|
|
|
|
# KeepAlive, systemd's Restart= does not fire on a deliberate stop, only on an unexpected exit of
|
|
|
|
|
# an active unit.
|
|
|
|
|
|
|
|
|
|
if [ -n "$OLD_PID" ]; then
|
|
|
|
|
say "stop"
|
|
|
|
|
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)" # verify a FRESH line appears later
|
|
|
|
|
if [ "$SUPERVISED" = 1 ]; then
|
|
|
|
|
echo " supervision is ON: using 'launchctl unload' (not kill) so launchd's own KeepAlive"
|
|
|
|
|
echo " cannot restart the OLD jar out from under this script — see the CB-594 comment above."
|
|
|
|
|
launchctl unload -w "$LAUNCHD_PLIST" \
|
|
|
|
|
|| die "launchctl unload failed — the daemon may still be under supervision; investigate before retrying"
|
|
|
|
|
else
|
|
|
|
|
kill "$OLD_PID"
|
|
|
|
|
fi
|
|
|
|
|
case "$SUPERVISOR_KIND" in
|
|
|
|
|
launchd)
|
|
|
|
|
echo " supervision is ON (launchd): using 'launchctl unload' (not kill) so launchd's own"
|
|
|
|
|
echo " KeepAlive cannot restart the OLD jar out from under this script — see the CB-594"
|
|
|
|
|
echo " comment above."
|
|
|
|
|
launchctl unload -w "$LAUNCHD_PLIST" \
|
|
|
|
|
|| die "launchctl unload failed — the daemon may still be under supervision; investigate before retrying"
|
|
|
|
|
;;
|
|
|
|
|
systemd)
|
|
|
|
|
echo " supervision is ON (systemd --user): using 'systemctl --user stop' (not kill) so"
|
|
|
|
|
echo " systemd's own Restart=on-failure cannot restart the OLD jar out from under this"
|
|
|
|
|
echo " script — see the fleetd #492 comment above."
|
|
|
|
|
systemctl --user stop "$SYSTEMD_UNIT" \
|
|
|
|
|
|| die "'systemctl --user stop $SYSTEMD_UNIT' failed — the daemon may still be under supervision; investigate before retrying"
|
|
|
|
|
;;
|
|
|
|
|
none)
|
|
|
|
|
kill "$OLD_PID"
|
|
|
|
|
;;
|
|
|
|
|
*)
|
|
|
|
|
# fleetd #492 follow-up: this block ACTS (stops the daemon one specific way per kind) — an
|
|
|
|
|
# unrecognised value must never fall through to a default action, silently picking the wrong
|
|
|
|
|
# one (or none at all) while reporting success. (Unreachable today: require_drivable_
|
|
|
|
|
# supervisor already died before this runs. Guards the value nobody has invented yet.)
|
|
|
|
|
die "detect_supervisor returned an unrecognized value '$SUPERVISOR_KIND' at the stop step —
|
|
|
|
|
refusing to guess how to stop a daemon under an unknown supervisor. The daemon was NOT
|
|
|
|
|
stopped." ;;
|
|
|
|
|
esac
|
|
|
|
|
for _ in $(seq "$STOP_WAIT"); do
|
|
|
|
|
[ -z "$(running_pid)" ] && break
|
|
|
|
|
sleep 1
|
|
|
|
@@ -310,13 +595,20 @@ if [ -n "$OLD_PID" ]; then
|
|
|
|
|
leave worktrees and panes behind. Investigate, then kill -9 by hand if you accept that."
|
|
|
|
|
fi
|
|
|
|
|
ok "pid $OLD_PID exited"
|
|
|
|
|
elif [ "$SUPERVISED" = 1 ]; then
|
|
|
|
|
elif [ "$SUPERVISOR_KIND" = "launchd" ]; then
|
|
|
|
|
# Loaded but not currently running (e.g. throttled after a crash loop). Unload it anyway so the
|
|
|
|
|
# start step below does a clean load, never a load stacked on an already-loaded label.
|
|
|
|
|
say "stop"
|
|
|
|
|
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
|
|
|
|
|
launchctl unload -w "$LAUNCHD_PLIST" 2>/dev/null || true
|
|
|
|
|
ok "launchd agent unloaded (was already not running)"
|
|
|
|
|
elif [ "$SUPERVISOR_KIND" = "systemd" ]; then
|
|
|
|
|
# Same case for systemd: the unit is known/active-capable but not currently running. `stop` on an
|
|
|
|
|
# already-stopped unit is a harmless no-op — kept for symmetry with the launchd branch above.
|
|
|
|
|
say "stop"
|
|
|
|
|
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
|
|
|
|
|
systemctl --user stop "$SYSTEMD_UNIT" 2>/dev/null || true
|
|
|
|
|
ok "systemd --user unit stopped (was already not running)"
|
|
|
|
|
else
|
|
|
|
|
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
|
|
|
|
|
fi
|
|
|
|
@@ -324,34 +616,56 @@ fi
|
|
|
|
|
# ------------------------------------------------------------------ start
|
|
|
|
|
# Unsupervised: login shell (zsh -l) is what puts the secrets on the daemon's environment, and cwd
|
|
|
|
|
# must be fleetd/ because the daemon resolves fleetd.yaml, logs/ and target/ relative to it.
|
|
|
|
|
# Supervised: launchd does both — deploy/dev.ltms.fleetd.plist points ProgramArguments at
|
|
|
|
|
# Supervised (launchd): launchd does both — deploy/dev.ltms.fleetd.plist points ProgramArguments at
|
|
|
|
|
# scripts/fleetd-launchd-wrapper.sh (CB-594), which is what execs the login shell in launchd's
|
|
|
|
|
# place, and WorkingDirectory in the plist already pins fleetd/.
|
|
|
|
|
# Supervised (systemd --user): the unit does both too — measured on the second host, ExecStart is
|
|
|
|
|
# `/bin/zsh -lc "exec java -jar target/fleetd.jar fleetd.yaml"` (a login shell, same reason as
|
|
|
|
|
# above) and WorkingDirectory is already pinned to fleetd/.
|
|
|
|
|
|
|
|
|
|
say "start"
|
|
|
|
|
if [ "$SUPERVISED" = 1 ]; then
|
|
|
|
|
echo " supervision is ON: using 'launchctl load' so launchd starts and keeps supervising this"
|
|
|
|
|
echo " process, instead of a manual nohup that launchd would know nothing about."
|
|
|
|
|
# CB-600: 'launchctl unload -w' above already persisted Disabled=true for this label. A load -w
|
|
|
|
|
# that succeeds clears it; a load -w that FAILS leaves the agent both stopped and disabled — worse
|
|
|
|
|
# than before this script ran, because a later reboot or login will not bring it back either. One
|
|
|
|
|
# retry covers a transient race (e.g. launchd not yet fully done deregistering); if it still fails,
|
|
|
|
|
# die with the exact recovery command rather than a bare "failed".
|
|
|
|
|
if ! launchctl load -w "$LAUNCHD_PLIST" 2>/dev/null; then
|
|
|
|
|
warn "launchctl load failed on the first attempt — retrying once after a short pause"
|
|
|
|
|
sleep 2
|
|
|
|
|
launchctl load -w "$LAUNCHD_PLIST" || die "launchctl load failed twice.
|
|
|
|
|
The agent is now STOPPED and DISABLED — it will NOT come back on its own, not even after a
|
|
|
|
|
reboot or login, because 'launchctl unload -w' above persisted Disabled=true and load -w
|
|
|
|
|
never got the chance to clear it. Recover with:
|
|
|
|
|
launchctl load -w \"$LAUNCHD_PLIST\"
|
|
|
|
|
If that still fails, check 'launchctl list $LAUNCHD_LABEL', validate the plist with
|
|
|
|
|
'plutil -lint \"$LAUNCHD_PLIST\"', and check $OUT before assuming a retry will succeed."
|
|
|
|
|
fi
|
|
|
|
|
else
|
|
|
|
|
# Absolute jar path so `ps` names which checkout is running.
|
|
|
|
|
( cd "$MODULE" && zsh -lc "nohup java -jar '$JAR' >> fleetd.out 2>&1 &" )
|
|
|
|
|
fi
|
|
|
|
|
case "$SUPERVISOR_KIND" in
|
|
|
|
|
launchd)
|
|
|
|
|
echo " supervision is ON (launchd): using 'launchctl load' so launchd starts and keeps"
|
|
|
|
|
echo " supervising this process, instead of a manual nohup that launchd would know nothing"
|
|
|
|
|
echo " about."
|
|
|
|
|
# CB-600: 'launchctl unload -w' above already persisted Disabled=true for this label. A load -w
|
|
|
|
|
# that succeeds clears it; a load -w that FAILS leaves the agent both stopped and disabled — worse
|
|
|
|
|
# than before this script ran, because a later reboot or login will not bring it back either. One
|
|
|
|
|
# retry covers a transient race (e.g. launchd not yet fully done deregistering); if it still fails,
|
|
|
|
|
# die with the exact recovery command rather than a bare "failed".
|
|
|
|
|
if ! launchctl load -w "$LAUNCHD_PLIST" 2>/dev/null; then
|
|
|
|
|
warn "launchctl load failed on the first attempt — retrying once after a short pause"
|
|
|
|
|
sleep 2
|
|
|
|
|
launchctl load -w "$LAUNCHD_PLIST" || die "launchctl load failed twice.
|
|
|
|
|
The agent is now STOPPED and DISABLED — it will NOT come back on its own, not even after a
|
|
|
|
|
reboot or login, because 'launchctl unload -w' above persisted Disabled=true and load -w
|
|
|
|
|
never got the chance to clear it. Recover with:
|
|
|
|
|
launchctl load -w \"$LAUNCHD_PLIST\"
|
|
|
|
|
If that still fails, check 'launchctl list $LAUNCHD_LABEL', validate the plist with
|
|
|
|
|
'plutil -lint \"$LAUNCHD_PLIST\"', and check $OUT before assuming a retry will succeed."
|
|
|
|
|
fi
|
|
|
|
|
;;
|
|
|
|
|
systemd)
|
|
|
|
|
echo " supervision is ON (systemd --user): using 'systemctl --user start' so systemd starts"
|
|
|
|
|
echo " and keeps supervising this process, instead of a manual nohup it would know nothing"
|
|
|
|
|
echo " about."
|
|
|
|
|
systemctl --user start "$SYSTEMD_UNIT" || die "'systemctl --user start $SYSTEMD_UNIT' failed.
|
|
|
|
|
Check 'systemctl --user status $SYSTEMD_UNIT' and $OUT before assuming a retry will succeed."
|
|
|
|
|
;;
|
|
|
|
|
none)
|
|
|
|
|
# Absolute jar path so `ps` names which checkout is running.
|
|
|
|
|
( cd "$MODULE" && zsh -lc "nohup java -jar '$JAR' >> fleetd.out 2>&1 &" )
|
|
|
|
|
;;
|
|
|
|
|
*)
|
|
|
|
|
# fleetd #492 follow-up: this block ACTS (starts the daemon one specific way per kind) — an
|
|
|
|
|
# unrecognised value must never fall through to a default action, silently picking the wrong
|
|
|
|
|
# one (or none at all) while reporting success. (Unreachable today: require_drivable_
|
|
|
|
|
# supervisor already died before this runs. Guards the value nobody has invented yet.)
|
|
|
|
|
die "detect_supervisor returned an unrecognized value '$SUPERVISOR_KIND' at the start step —
|
|
|
|
|
refusing to guess how to start a daemon under an unknown supervisor. The daemon was NOT
|
|
|
|
|
started." ;;
|
|
|
|
|
esac
|
|
|
|
|
|
|
|
|
|
for _ in $(seq 10); do
|
|
|
|
|
NEW_PID="$(running_pid)"
|
|
|
|
@@ -411,6 +725,13 @@ FRESH_LOG="$(mktemp -t fleetd-fresh-log)"
|
|
|
|
|
trap 'rm -f "$FRESH_LOG"' EXIT
|
|
|
|
|
tail -n "+$((RESTART_MARK + 1))" "$OUT" > "$FRESH_LOG" 2>/dev/null || true
|
|
|
|
|
classify_amqp_connection_errors "$FRESH_LOG"
|
|
|
|
|
|
|
|
|
|
# fleetd #492: checked here, after healthz and the fresh-log check have both had time to run, so a
|
|
|
|
|
# supervisor that revives the OLD jar a few seconds late is caught too. Every check above (healthz
|
|
|
|
|
# 200, jar id, the fresh 'listening' line) is satisfied by EITHER daemon if two are alive — this is
|
|
|
|
|
# the only one that can tell.
|
|
|
|
|
assert_single_daemon "$(running_pid)"
|
|
|
|
|
|
|
|
|
|
say "result"
|
|
|
|
|
ok "pid $NEW_PID, jar $(jar_id)"
|
|
|
|
|
if [ "$REDEPLOY_ERROR_COUNT" -eq 0 ]; then
|
|
|
|
|