#!/usr/bin/env bash # Regression test for the single assumption this whole caching scheme rests # on: that a build running inside a hardlink clone cannot mutate the directory # it was cloned from. # # That assumption is FALSE for a plain `cp -al`. Measured, and asserted below # as an explicit control: build in a raw `cp -al` clone and the source's # dep-info file (`.fingerprint//dep-*` under Cargo's build-dir layout # v1, `build///fingerprint/dep-*` under v2 — and under content # freshness only, see the probe below), `build//output`, # `build//out/**` and `deps/*.d` all change, # because Cargo and build scripts write those with a plain truncating write # rather than the write-then-rename Cargo uses for real artifacts. # # The consequence is not a slow build, it is a wrong one: a PR clone rewrites # the base's dep-info to describe the PR's sources while the base's cache # still holds the artifact built from the base's sources; once the PR merges, # the base's next run finds the checksums match its (now merged) sources, # reports `Fresh`, and links a binary built from the pre-merge code. # # cache-lib.sh's unshare_mutable_paths() is what closes that, and this test is # what proves it stays closed. The control matters as much as the fix: a # scenario that passes for both would prove nothing. # # Needs a working cargo on PATH. Everything happens under a mktemp -d scratch # tree. Run by hand: bash scripts/hardlink-clone-selftest.sh set -euo pipefail script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) . "$script_dir/cache-lib.sh" command -v cargo >/dev/null || { echo "SKIP: no cargo on PATH"; exit 0; } scratch=$(mktemp -d) trap 'rm -rf "$scratch"' EXIT pass_count=0 fail() { echo "ASSERTION FAILED: $*" >&2; exit 1; } ok() { pass_count=$((pass_count + 1)); echo "PASS: $*"; } # Content hash of every file in a tree, keyed by relative path. snapshot_tree() { (cd "$1" && find . -type f -print0 | sort -z | xargs -0 -r sha1sum) 2>/dev/null; } # `diff` exits 1 when the trees differ, which is the expected case here and # must not trip `pipefail` — the difference IS the result. mutated_paths() { { diff <(printf '%s' "$1") <(printf '%s' "$2") || true; } 2>/dev/null \ | awk '/^[<>]/ { print $3 }' | sort -u } # A crate with a build script, because build-script OUT_DIR writes are one of # the two mutation families and are invisible without one. mkcrate() { local dir="$1" mkdir -p "$dir/src" cat > "$dir/Cargo.toml" <<'TOML' [package] name = "probe" version = "0.1.0" edition = "2021" [workspace] TOML cat > "$dir/build.rs" <<'RS' use std::{env, fs, path::PathBuf}; fn main() { println!("cargo::rerun-if-changed=src/lib.rs"); let out = PathBuf::from(env::var("OUT_DIR").unwrap()); let src = fs::read_to_string("src/lib.rs").unwrap(); fs::write(out.join("gen.txt"), format!("generated from {} bytes", src.len())).unwrap(); } RS } crate_dir="$scratch/probe" mkcrate "$crate_dir" cd "$crate_dir" export CARGO_INCREMENTAL=0 # Every assertion below reads cargo's own words out of a build log # (`Compiling libdep`, `Fresh probe`). A CI image that forces colour splices an # ANSI reset between the status word and the crate name, at which point every # one of those greps silently stops matching and the suite reports the # opposite of what happened — observed on gitdan-ci's runner image, where # scenario 2 failed while the log it printed plainly showed `Compiling libdep`. # Pin the format the assertions are written against. export CARGO_TERM_COLOR=never # Checksum freshness is where the worst failure lives (the dep-* file carries # per-source checksums and is rewritten in place). Only available on nightly; # without it the test still covers the build/ and *.d families. CONTENT_A='pub fn f() -> u32 { 1 }' CONTENT_B='pub fn f() -> u32 { 22222 } pub fn g() -> u32 { 7 }' # Probe the BEHAVIOUR, not the channel and not the flag. Two weaker probes # were tried against gitdan-ci's runner and each let the suite assert a # property the toolchain did not have: # # `cargo +nightly -V` — answers "did a proxy called with # +nightly exit 0". `-V` # short-circuits before `-Z` is # even parsed. # `cargo +nightly -Z checksum-freshness — answers "is this flag still # locate-project` accepted", which since cargo PR # #17382 (2026-08-22) is a # different question from "is # content freshness on". That PR # demoted the flag to a gate and # gave `build.fingerprint` the # choice, defaulting to `mtime` — # so 1.100.0-nightly accepts the # flag and resolves freshness by # mtime unless # CARGO_BUILD_FINGERPRINT=content # is set too. Measured 2026-08-26; # see daniel/gitdan#62. # # The scenario at the end of this file depends on one thing and it is neither # of those: that changed content with an OLDER mtime rebuilds. Under mtime # freshness the correct answer is Fresh, so under mtime freshness that # scenario asserts a bug. So the probe simply performs that experiment, on its # own crate and its own target dir, with no clone anywhere near it — which is # also what makes it a control rather than a restatement of the scenario: the # probe establishes that the toolchain rebuilds on content, the scenario # establishes that a hardlink clone did not take that away. # THREE OUTCOMES, NOT TWO. An experiment that cannot tell a negative result # from a failed measurement is not settling the question, and the two are not # interchangeable here: "this toolchain resolves freshness by mtime" is a # statement about Cargo, while "a probe build failed" is a statement about this # machine. Collapsing them — which an earlier cut of this did, by returning # non-zero for both — makes a half-installed toolchain print a confident and # wrong explanation and quietly drop a scenario. The scenario still has to be # skipped in either case; what must not happen is the log claiming to know why. # # 0 content freshness measured ACTIVE — both builds ran, the backdated # rebuild recompiled # 3 measured INACTIVE — both builds ran, the backdated # rebuild reported Fresh # anything else NOT MEASURED — nothing was learned about the # toolchain # # THE ANSWER CODES ARE 0 AND 3, AND THE GAP IS THE MECHANISM. Bash produces 1 # for an ordinary command failure, 2 for a usage error, 126/127 for a command # it could not run, 128+n for a signal, and — this is the one that matters — # 1 for an unbound-variable or other EXPANSION failure, which happens before # the command runs and is therefore invisible to a `||` guard and to an ERR # trap alike. It never produces 3. So "not an answer code" is decided by a # property of the shell rather than by an enumeration of the ways a step can # go wrong, and a step added later without a guard, or with a guard that # cannot fire, lands on NOT MEASURED by construction. # # That is the whole reason INACTIVE is not 1. It was, and three review rounds # on this function each found a narrower way for a shell-generated 1 to be read # as a measurement — an unguarded command, then a typo'd variable name on a # line that HAS its guard. Each was closed by narrowing the failure surface, # which is a game with no last move. Moving the answer off the codes bash can # generate ends it instead: there is no longer a mutation that turns an error # into an answer, only mutations that turn an error into a different error. # # The guards below stay, and so does the trap, but their job is now reporting # rather than correctness: they make a failed step land on 2 with its logs # printed instead of on some incidental status, which is nicer to debug and # lands in the same place either way. # # One piece of that reporting layer is load-bearing and not obvious. A command # on the left of `||` — or in an `if` condition — runs with errexit suppressed, # and that suppression propagates into a subshell and is NOT undone by a # `set -e` inside it (measured on bash 5.3: an unguarded `false` there falls # through to `exit 0`). Calling with errexit disarmed at the site is the only # form that lets the subshell re-arm it; hence the `set +e` bracket. The ERR # trap is then required on top, because a bare `set -e` abort exits with the # FAILING COMMAND's status, and `false` gives 1. # # WHY NOT-MEASURED SKIPS RATHER THAN FAILS. The scenario it gates is the only thing in # this suite that depends on freshness mode; everything else still runs and # still catches real regressions. Failing instead would turn a statement about # one machine's toolchain into a red gate reading "the hardlink scheme is # broken" across the three repos consuming this action — the same category # error the three-state split exists to prevent, one level up. What would # change the answer is not-measured becoming the everyday CI outcome; it is not # — gitdan-ci's outcome is a measurement either way. It reported a measured # INACTIVE until 2026-08-26, for the reason recorded at the `export` below, and # an ACTIVE once both switches were set. CHECKSUM_MODE="off" CHECKSUM_REASON="no nightly on PATH accepting -Z checksum-freshness" CARGO_BIN=(cargo) checksum_freshness_probe() { local d="$scratch/freshness-probe" t="$scratch/freshness-probe-target" mkcrate "$d" || return 2 # 2 is simply "not 0 and not 3"; see the header ( set -e trap 'exit 2' ERR cd "$d" || exit 2 printf '%s\n' "$CONTENT_A" > src/lib.rs || exit 2 CARGO_TARGET_DIR="$t" cargo +nightly build -q > "$scratch/freshness-probe-warm.log" 2>&1 || exit 2 printf '%s\n' "$CONTENT_B" > src/lib.rs || exit 2 touch -d '@1000000000' src/lib.rs || exit 2 CARGO_TARGET_DIR="$t" cargo +nightly build -v > "$scratch/freshness-probe.log" 2>&1 || exit 2 # 3, not 1: see the header. This is the only statement in the subshell that # may report a measurement, and it is the only one that may exit 3. if grep -qE '^\s+Fresh probe' "$scratch/freshness-probe.log"; then exit 3; fi exit 0 ) } if cargo +nightly -Z checksum-freshness locate-project > /dev/null 2>&1; then export CARGO_UNSTABLE_CHECKSUM_FRESHNESS=true # BOTH, since cargo PR #17382 (2026-08-22): the -Z flag only unlocks the # feature and `build.fingerprint` selects it, defaulting to `mtime`. Setting # the gate alone is what made this suite report a measured INACTIVE on # 1.100.0-nightly and skip its strongest scenario (daniel/gitdan#62). Safe to # export unconditionally — a Cargo that does not know the key ignores it # silently, verified 2026-08-26 on 1.93.1 stable and 1.96.0-nightly, both of # which still measure ACTIVE from the gate alone. export CARGO_BUILD_FINGERPRINT=content # Errexit off across the call, so the subshell can arm its own — see the # header. `probe_rc` is read before it is restored. probe_rc=0 set +e checksum_freshness_probe probe_rc=$? set -e case "$probe_rc" in 0) CARGO_BIN=(cargo +nightly) CHECKSUM_MODE="on" CHECKSUM_REASON="" ;; 3) unset CARGO_UNSTABLE_CHECKSUM_FRESHNESS CARGO_BUILD_FINGERPRINT CHECKSUM_REASON="this nightly accepts -Z checksum-freshness and build.fingerprint=content but still resolves freshness by mtime" ;; *) unset CARGO_UNSTABLE_CHECKSUM_FRESHNESS CARGO_BUILD_FINGERPRINT CHECKSUM_MODE="unmeasured" CHECKSUM_REASON="the probe exited ${probe_rc}, which is not one of its answer codes, so this was NOT MEASURED — this toolchain may or may not resolve freshness by content" # Loud, because the cost is silently lost coverage on a machine that # might have had it. The suite continues: everything else it asserts is # independent of freshness mode. echo "::warning::hardlink-clone-selftest: could not measure whether this toolchain resolves freshness by content — the probe exited ${probe_rc}. This is a failure to measure, not a finding about Cargo." tail -n 15 "$scratch/freshness-probe-warm.log" "$scratch/freshness-probe.log" 2>/dev/null | sed 's/^/ /' >&2 || true ;; esac fi cd "$crate_dir" echo "=== checksum-freshness mode: ${CHECKSUM_MODE}${CHECKSUM_REASON:+ — ${CHECKSUM_REASON}} ===" build_base() { local dir="$1" printf '%s\n' "$CONTENT_A" > src/lib.rs CARGO_TARGET_DIR="$dir" "${CARGO_BIN[@]}" build -q } echo echo "=== control: a raw \`cp -al\` clone DOES mutate its source ===" base_ctl="$scratch/base-ctl"; clone_ctl="$scratch/clone-ctl" build_base "$base_ctl" before=$(snapshot_tree "$base_ctl") cp -al "$base_ctl" "$clone_ctl" strip_cargo_locks "$clone_ctl" # locks alone are not the hazard under test printf '%s\n' "$CONTENT_B" > src/lib.rs CARGO_TARGET_DIR="$clone_ctl" "${CARGO_BIN[@]}" build -q after=$(snapshot_tree "$base_ctl") ctl_mutated=$(mutated_paths "$before" "$after") if [ -z "$ctl_mutated" ]; then fail "control produced no mutation — the test can no longer distinguish fixed from broken" fi ok "raw cp -al clone mutates the source ($(printf '%s\n' "$ctl_mutated" | wc -l) paths)" printf '%s\n' "$ctl_mutated" | sed 's/^/ /' # Reported, not asserted, and the distinction is the point. The control's job # is to prove the hazard exists at all, which the non-empty set above already # does; this line records WHICH families a given Cargo exhibits. # # The dep-info file is the worst of them — it carries the per-source # checksums, so mutating it through a shared inode turns a hardlink clone into # silent stale-artifact reuse rather than a slow build. Failing on its absence # would mean this suite goes red whenever upstream stops doing something we # never wanted it to do — and it would go red in the CONTROL, where a failure # reads as "the hazard is gone" rather than "upstream changed". Nothing is lost # by reporting it: the fix scenario below asserts the source is byte-identical # after a full rebuild in the clone, which covers every family this Cargo has, # named or not. # # THE PATTERN MUST MATCH BOTH LAYOUTS, and that is not a detail. Cargo's # build-dir layout v2 moved the file from `/.fingerprint//dep-*` # to `/build///fingerprint/dep-*` (stabilised by cargo PR # #17354, cargo 1.100.0, stable 2026-11-12; nightly default since 1.99). An # earlier cut of this line looked for the v1 path only, so on 2026-08-26, # against 1.100.0-nightly with content freshness genuinely on, it printed # "does NOT rewrite ... in place" directly beneath a control listing that # showed the rewrite. A reporting line that can contradict the data three # lines above it is worse than no line at all. `fingerprint/.*dep-` matches # either layout and neither `.d` family. if [ "$CHECKSUM_MODE" = "on" ]; then if printf '%s' "$ctl_mutated" | grep -q 'fingerprint/.*dep-'; then echo " note: this cargo DOES rewrite its dep-info fingerprint file in place under content freshness" else echo " note: this cargo does NOT rewrite its dep-info fingerprint file in place; only the build/ and *.d families appear above" fi fi echo echo "=== fix: hardlink_clone_into() leaves the source byte-identical ===" base_fix="$scratch/base-fix"; clone_fix="$scratch/clone-fix" build_base "$base_fix" before=$(snapshot_tree "$base_fix") hardlink_clone_into "$base_fix" "$clone_fix" "selftest" || fail "hardlink_clone_into reported the destination already existed" # The clone's contract, asserted before anything builds in it: artifacts # share inodes (that is what makes the clone near-free), and every file Cargo # rewrites in place does not (that is what makes it sound). Checking after a # rebuild would prove nothing — the rebuild replaces those files anyway. shared=0; unshared=0 while IFS= read -r f; do rel="${f#"$base_fix"/}" [ -e "$clone_fix/$rel" ] || continue if [ "$(stat -c '%i' "$f")" = "$(stat -c '%i' "$clone_fix/$rel")" ]; then case "$rel" in */.fingerprint/*|*/build/*|*.d|.rustc_info.json) fail "mutable path still shares an inode with the source: $rel" ;; esac shared=$((shared + 1)) else unshared=$((unshared + 1)) fi done < <(find "$base_fix" -type f) [ "$shared" -gt 0 ] || fail "nothing is shared — the clone degenerated into a full copy" [ "$unshared" -gt 0 ] || fail "nothing was unshared — unshare_mutable_paths did not run" ok "fresh clone shares ${shared} artifact files and privately owns ${unshared} mutable ones" printf '%s\n' "$CONTENT_B" > src/lib.rs CARGO_TARGET_DIR="$clone_fix" "${CARGO_BIN[@]}" build -q after=$(snapshot_tree "$base_fix") fix_mutated=$(mutated_paths "$before" "$after") if [ -n "$fix_mutated" ]; then echo " still mutated:" >&2 printf '%s\n' "$fix_mutated" | sed 's/^/ /' >&2 fail "a build in the clone mutated the source through a shared inode" fi ok "no file in the source changed after a full rebuild in the clone" echo if [ "$CHECKSUM_MODE" = "on" ]; then echo "=== the whole point: the source's next build is still correct ===" # The source's cache holds artifacts built from CONTENT_A. Advance the # source to CONTENT_B (as a merge would) and rebuild in it. If the clone had # corrupted its dep-info, Cargo would report Fresh and keep the stale rlib. # # CHECKSUM-FRESHNESS ONLY, and the backdated mtime is why. Under checksum # freshness the dep-info file's per-source checksums decide, so a 2001 # timestamp on changed content must still rebuild — the assertion below. # Under Cargo's ordinary MTIME freshness the same timestamp means the source # is older than the artifact, and reporting Fresh is the correct answer; # asserting otherwise asserts a bug. This scenario was written against a # machine with a nightly installed and, run without one, failed on that # correct answer. printf '%s\n' "$CONTENT_B" > src/lib.rs touch -d '@1000000000' src/lib.rs log="$scratch/rebuild.log" CARGO_TARGET_DIR="$base_fix" "${CARGO_BIN[@]}" build -v > "$log" 2>&1 || { cat "$log"; fail "rebuild in the source failed"; } if grep -qE '^\s+Fresh probe' "$log"; then fail "source declared its own crate Fresh against sources it has never built — stale-artifact reuse" fi ok "source correctly rebuilt its crate after advancing to the clone's content" else echo "=== skipped: the source's-next-build scenario needs content-based freshness ===" echo " reason: ${CHECKSUM_REASON}" fi echo echo "hardlink-clone-selftest: ${pass_count} assertions passed"