From 3da44eed636f2ae40576abfb80230af73ba9312f Mon Sep 17 00:00:00 2001 From: Dai Ha Date: Sat, 12 Sep 2026 15:06:15 +0700 Subject: [PATCH 1/2] fleetd #550: replace shasum with a portable hash256 helper, add a Linux CI job for the shell suite MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit jar_id() in redeploy-fleetd.sh called shasum directly, which does not exist on GNU coreutils Linux (Debian/Ubuntu/etc.) — there it silently reported an existing jar as "absent" with exit 0, because the missing command made `cut` succeed on empty input and pipefail's failure was then swallowed by the `|| echo "absent"` fallback. The shell test suite hit the same tool at test-redeploy-fleetd.sh:298-299 and died at exit 127 with zero FAIL lines printed — the same shape as a clean pass on the one channel anyone would check. Adds one hash256() helper (prefer sha256sum, fall back to shasum -a 256, same idiom already used in probe-member-credentials.sh) and points jar_id and the test suite's own reference hash at it. jar_id now has three distinct answers instead of two: absent, a hash, or "unhashable" when neither hasher is on PATH — "absent" is never used for a file that exists. Adds a CI job (shell-tests) that runs scripts/test-redeploy-fleetd.sh on ubuntu-latest, gated on the step's own exit code rather than a FAIL-line count, since a suite that dies before running is exactly what a green run also looks like by that count. New tests: test_jar_id_reports_unhashable_when_no_hasher_on_path (stubbed PATH with neither hasher) and test_no_unguarded_macos_only_hasher_calls (a shape check across every script under scripts/, not named lines — #545 already showed this idiom spreading from two sites to six). --- .gitea/workflows/ci.yml | 17 +++++++++ scripts/redeploy-fleetd.sh | 26 ++++++++++++- scripts/test-redeploy-fleetd.sh | 67 ++++++++++++++++++++++++++++++++- 3 files changed, 107 insertions(+), 3 deletions(-) diff --git a/.gitea/workflows/ci.yml b/.gitea/workflows/ci.yml index 6cd955a..4fe1422 100644 --- a/.gitea/workflows/ci.yml +++ b/.gitea/workflows/ci.yml @@ -58,6 +58,23 @@ jobs: done exit 0 + # fleetd #550 — nothing ran scripts/test-redeploy-fleetd.sh in CI before this, on any platform, + # so it had run only on macOS by hand and two Linux-only bugs (this issue's items 1 and 2) + # survived undetected: shasum is a macOS-only tool (it ships with Perl; GNU coreutils, i.e. every + # mainstream Linux distro including this runner's ubuntu-latest, does not have it and ships + # sha256sum instead). The gate here is the step's own exit code, nothing else: a `run:` step in + # Gitea/GitHub Actions already fails the job on a non-zero exit with no extra scripting needed, + # so this deliberately does NOT grep the output for a `FAIL:` count. That is the #550 item-2 + # lesson one level up — a suite that dies before it runs a single test prints zero FAIL lines, + # which is exactly what a clean pass also prints, so counting FAIL lines can never be the gate. + shell-tests: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - name: redeploy-fleetd.sh shell suite + run: bash scripts/test-redeploy-fleetd.sh + # CB-521 — actually run the AMQP contract test in CI, against a REAL broker. The broker is a # RabbitMQ SERVICE CONTAINER, not Testcontainers-with-Docker: the runner image has no Docker, so # AmqpReplyInboxContractTest reads AMQP_URI (set below to the service's network alias) and binds diff --git a/scripts/redeploy-fleetd.sh b/scripts/redeploy-fleetd.sh index 533c0ea..a5df9fb 100755 --- a/scripts/redeploy-fleetd.sh +++ b/scripts/redeploy-fleetd.sh @@ -140,11 +140,35 @@ ok() { printf ' ok %s\n' "$*"; } warn() { printf ' WARN %s\n' "$*"; } die() { printf '\n FAIL %s\n\n' "$*" >&2; exit 1; } +# fleetd #550 — shasum is macOS-only (it ships with Perl, which Debian/Ubuntu/etc. do not install +# by default); GNU coreutils (every mainstream Linux distro) ships sha256sum instead and has no +# shasum at all. Prefer sha256sum, fall back to shasum -a 256 — same idiom as +# probe-member-credentials.sh's `hasher` selection — and when NEITHER is on PATH, say so plainly. +# That third answer matters: without it, a missing hasher makes `cut` succeed on empty input, and +# under `set -o pipefail` the pipeline as a whole still fails, so a caller's own `|| echo "absent"` +# then reports a file that is right there as though it were gone. `hash256` never does that — it +# only ever hashes or says it could not. +hash256() { + local f="$1" + if command -v sha256sum >/dev/null 2>&1; then + sha256sum "$f" | cut -c1-12 + elif command -v shasum >/dev/null 2>&1; then + shasum -a 256 "$f" | cut -c1-12 + else + echo "unhashable" + fi +} + # Reports the hash of $JAR by default, or of whatever path is passed — used to report the STAGED # jar right after a build (before it has been swapped in) without ever changing what a bare # `jar_id` (no args) means: the live path, $JAR. --check and the final "pid ..., jar ..." line # both call it with no args on purpose, so neither can ever be fooled by a leftover staged file. -jar_id() { local f="${1:-$JAR}"; [ -f "$f" ] && shasum -a 256 "$f" | cut -c1-12 || echo "absent"; } +# fleetd #550 — THREE distinct answers now, not two: `[ -f "$f" ]` already separates "the jar is +# not there" (-> "absent") from "the jar is there"; for the second case, hash256 itself separates +# "hashed it" (a 12-char hex string) from "could not hash it" (-> "unhashable", when no hasher is +# on PATH). "absent" must never be the answer for a file that exists — that conflation, on Linux, +# was the whole defect this ticket fixes. +jar_id() { local f="${1:-$JAR}"; [ -f "$f" ] && hash256 "$f" || echo "absent"; } running_pid() { pgrep -f "$PATTERN" || true; } # fleetd #493 — three small, independently testable pieces of "never build into the path a diff --git a/scripts/test-redeploy-fleetd.sh b/scripts/test-redeploy-fleetd.sh index 18effa0..6421700 100755 --- a/scripts/test-redeploy-fleetd.sh +++ b/scripts/test-redeploy-fleetd.sh @@ -209,6 +209,33 @@ test_mktemp_dash_t_templates_have_x_placeholders() { || fail "mktemp -t template(s) with no X placeholder (fails under GNU coreutils): $bad" } +# fleetd #550 — the shape, not the named lines: #545 showed the exact same failure mode (a +# macOS-only idiom used with no portable fallback) spread from two sites to six across 91 commits +# before anyone tested the SHAPE rather than specific lines. This is the shasum sibling: any script +# under scripts/ that actually INVOKES the macOS-only hasher to compute a hash (as opposed to +# merely probing whether it exists with `command -v`, or mentioning it in prose) must also check +# for the portable one first in that same file — the prefer-portable-fall-back-to-macOS-only idiom +# probe-member-credentials.sh:273-279 and this ticket's own hash256 helper both follow. +# +# The needle is built from two concatenated pieces, deliberately never written as one literal +# string in this file: written whole, it would match THIS CHECK'S OWN source line once the loop +# below reaches this very file, and the check would then "pass" by matching itself rather than any +# real invocation elsewhere — a zero-findings result indistinguishable from a clean file. +test_no_unguarded_macos_only_hasher_calls() { + local needle f bad="" usage + needle='shasum'; needle="$needle -a" + for f in "$ROOT"/scripts/*.sh; do + [ -f "$f" ] || continue + usage="$(grep -Fn "$needle" "$f" || true)" + if [ -n "$usage" ]; then + grep -q 'command -v sha256sum' "$f" \ + || bad="$bad$(basename "$f") " + fi + done + [ -z "$bad" ] \ + || fail "script(s) invoke the macOS-only hasher with no portable-hasher-first fallback guard in the same file: $bad" +} + # fleetd #492 follow-up (Item 1): this must go through the REAL call-site shape at :437-440, not a # hand-constructed "unclear" value — a test that builds "unclear" directly proves the switch, not # the handoff, and that is exactly the gap that let SUPERVISOR_UNCLEAR_DETAIL never reach the real @@ -295,8 +322,11 @@ test_jar_id_defaults_to_live_and_reports_explicit_path() { JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar" printf 'live jar bytes' > "$JAR" printf 'staged jar bytes, not the same content' > "$JAR_STAGED" - live_hash="$(shasum -a 256 "$JAR" | cut -c1-12)" - staged_hash="$(shasum -a 256 "$JAR_STAGED" | cut -c1-12)" + # fleetd #550: this reference hash must be computed the same portable way jar_id() itself now + # computes one — a bare, unguarded call to the macOS-only hasher here was exactly the item-2 + # defect, dying with "command not found" on any Linux runner that has no such hasher at all. + live_hash="$(hash256 "$JAR")" + staged_hash="$(hash256 "$JAR_STAGED")" default_result="$(jar_id)" explicit_result="$(jar_id "$JAR_STAGED")" JAR="$saved_jar"; JAR_STAGED="$saved_staged" @@ -323,6 +353,37 @@ test_jar_id_reports_absent_for_missing_file() { assert_equals "absent" "$explicit_result" "jar_id with an explicit missing path must report absent" } +# fleetd #550 — the whole point of this ticket: a jar that IS there but could not be hashed must +# never read the same as a jar that is not there at all. Drives the REAL hash256/jar_id bodies +# (never stubbed) through a stub PATH that contains neither of the two hashers this script knows — +# same technique test_detect_supervisor_systemd_probe_setup_failure_is_unclear uses for `mktemp`, +# except here the stub directory is used to REPLACE PATH rather than prepend to it, because the +# point is to make BOTH hashers unreachable, not to intercept one specific command while leaving +# everything else on the real PATH reachable. `[ -f ... ]` and the shell's own `command`/`echo` +# builtins need no PATH at all, so this is safe even with PATH reduced to an empty directory. +test_jar_id_reports_unhashable_when_no_hasher_on_path() { + local bin_dir dir saved_jar="$JAR" default_result explicit_result + bin_dir="$TMP/stub-bin-no-hasher"; mkdir -p "$bin_dir" + dir="$TMP/jar-id-no-hasher"; mkdir -p "$dir" + JAR="$dir/fleetd.jar" + printf 'a real jar that exists but nothing here can hash' > "$JAR" + [ -f "$JAR" ] || fail "test fixture error: \$JAR does not exist at $JAR" + default_result="$(PATH="$bin_dir" jar_id)" + explicit_result="$(PATH="$bin_dir" jar_id "$JAR")" + JAR="$saved_jar" + [ "$default_result" != "absent" ] \ + || fail "jar_id reported absent for a file that exists, only because no hasher was on PATH" + [ "$explicit_result" != "absent" ] \ + || fail "jar_id (explicit path) reported absent for a file that exists, only because no hasher was on PATH" + # A 12-char hex hash is the OTHER wrong answer here: with no hasher at all, nothing could have + # produced one, so a value that merely happens to look like one would mean the stub failed to + # hide the real hashers rather than that jar_id degraded correctly. + printf '%s' "$default_result" | grep -Eq '^[0-9a-f]{12}$' \ + && fail "test fixture error: PATH stub did not actually hide the real hasher(s) — got what looks like a real hash" + assert_equals "unhashable" "$default_result" "jar_id with no hasher on PATH must report a third, distinct state — never absent, never a hash" + assert_equals "unhashable" "$explicit_result" "jar_id (explicit path) with no hasher on PATH must report the same third state" +} + # fleetd #493 — never build into the path a running process holds. stage_built_jar/swap_staged_jar # are exercised directly against real files on disk (not stubs), because the whole point is file # behavior (does the content move, does the source disappear, does a failure leave both sides @@ -1167,6 +1228,7 @@ test_detect_supervisor_launchd_installed_not_loaded_is_unclear test_detect_supervisor_systemd_probe_error_is_unclear test_detect_supervisor_systemd_probe_setup_failure_is_unclear test_mktemp_dash_t_templates_have_x_placeholders +test_no_unguarded_macos_only_hasher_calls test_require_drivable_supervisor_refuses_ambiguous test_require_drivable_supervisor_refuses_unclear test_require_drivable_supervisor_accepts_known_kinds @@ -1175,6 +1237,7 @@ test_assert_single_daemon_accepts_one_pid test_assert_single_daemon_rejects_two_pids test_jar_id_defaults_to_live_and_reports_explicit_path test_jar_id_reports_absent_for_missing_file +test_jar_id_reports_unhashable_when_no_hasher_on_path test_stage_built_jar_moves_off_live_path test_stage_built_jar_dies_when_build_produced_nothing test_swap_staged_jar_moves_staged_onto_live -- 2.52.0 From b8182c96c285b2f516583677676f3a613e948f1e Mon Sep 17 00:00:00 2001 From: Dai Ha Date: Sat, 12 Sep 2026 15:16:43 +0700 Subject: [PATCH 2/2] fleetd #550: pin hash256's algorithm against a literal SHA-256 test vector test_jar_id_defaults_to_live_and_reports_explicit_path's reference hash is computed by calling hash256 itself (needed so it doesn't call the Linux-crashing bare shasum directly). That made subject and reference the same instrument: they agree no matter which algorithm hash256 actually runs, so a mutation swapping both of hash256's arms for the wrong algorithm was invisible to the suite. Adds test_hash256_computes_a_real_sha256, pinned against the published SHA-256 test vector for the 3-byte input "abc" (ba7816bf8f01...), written as a literal constant rather than computed by any hasher at test time. Verified the constant myself both ways (sha256sum and shasum -a 256) before writing it in. --- scripts/test-redeploy-fleetd.sh | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/scripts/test-redeploy-fleetd.sh b/scripts/test-redeploy-fleetd.sh index 6421700..384e9a2 100755 --- a/scripts/test-redeploy-fleetd.sh +++ b/scripts/test-redeploy-fleetd.sh @@ -335,6 +335,26 @@ test_jar_id_defaults_to_live_and_reports_explicit_path() { assert_equals "$staged_hash" "$explicit_result" "jar_id \"\$JAR_STAGED\" must report the hash of the staged jar, not fall back to \$JAR" } +# fleetd #550 — closes a gap the test above leaves open. That test's own reference hash is now ALSO +# computed by calling hash256 (needed for item 2: the old bare macOS-only-hasher call there was the +# Linux crash), so its subject (jar_id, via hash256) and its reference (also hash256) share one +# instrument — they agree no matter which algorithm hash256 actually runs, so a mutation that swaps +# BOTH of hash256's arms for the wrong algorithm is invisible to it. This test's expected value +# comes from neither hasher: it is the published SHA-256 test vector for the 3-byte input "abc" +# (no trailing newline), written here as a literal constant, so it can still tell "hashed +# correctly" from "hashed, just with the wrong algorithm" — which is what this whole ticket is +# about. +test_hash256_computes_a_real_sha256() { + local dir f result + dir="$TMP/hash256-known-vector"; mkdir -p "$dir" + f="$dir/abc.txt" + printf 'abc' > "$f" + result="$(hash256 "$f")" + # SHA-256("abc") = ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad, the standard + # FIPS 180 test vector — first 12 hex chars, matching hash256's own `cut -c1-12`. + assert_equals "ba7816bf8f01" "$result" "hash256 of the literal 3-byte input 'abc' must be the known SHA-256 prefix, not some other algorithm's" +} + # fleetd #517 — jar_id()'s "absent" branch was unpinned by any test: the existing test above (#511) # proves both halves of the present-file contract but never exercises the missing-file path. This # word matters more than a string usually would: "absent" is the #413 signal that a `mvn clean` @@ -1236,6 +1256,7 @@ test_count_daemon_pids test_assert_single_daemon_accepts_one_pid test_assert_single_daemon_rejects_two_pids test_jar_id_defaults_to_live_and_reports_explicit_path +test_hash256_computes_a_real_sha256 test_jar_id_reports_absent_for_missing_file test_jar_id_reports_unhashable_when_no_hasher_on_path test_stage_built_jar_moves_off_live_path -- 2.52.0