Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions os/installer/pithead-install
Original file line number Diff line number Diff line change
Expand Up @@ -114,6 +114,10 @@ wipe_data_partition() { # $1 disk, $2 mode (data|all)
echo "==> wiping user data, preserving the synced chains"
mnt=$(mktemp -d)
mount "$part" "$mnt"
# A failed step below must not leave the partition mounted-busy for the operator's retry —
# the script's own EXIT trap is installed later in main(), so this window needs its own.
# shellcheck disable=SC2064 # expand NOW: $mnt is a local, out of scope when the trap fires
trap "umount '$mnt' 2>/dev/null || true; rmdir '$mnt' 2>/dev/null || true" EXIT
keep="$mnt/.wipe-keep.$$"
mkdir -p "$keep"
# Same-filesystem renames: preserving a 250 GB chain must not copy it.
Expand All @@ -132,6 +136,7 @@ wipe_data_partition() { # $1 disk, $2 mode (data|all)
rmdir "$keep" 2>/dev/null || true
umount "$mnt"
rmdir "$mnt"
trap - EXIT
}

main() {
Expand Down
23 changes: 20 additions & 3 deletions tests/os/run.sh
Original file line number Diff line number Diff line change
Expand Up @@ -322,6 +322,17 @@ phase_boot() {
}
ok "wizard serves the token gate ($ip)"

# Hugepages are load-bearing (the RandomX dataset must land in hugetlbfs, not the cgroup —
# the Dockerfile's own words): the baked sysctl reserves 3072 2M pages, and a boot that
# silently lost them starves the miner while everything else looks healthy.
local hp
hp=$(_ssh "awk '/^HugePages_Total/{print \$2}' /proc/meminfo" 2>/dev/null) || hp=""
if [ -n "$hp" ] && [ "$hp" -ge 3072 ]; then
ok "hugepage pool reserved at boot ($hp pages)"
else
bad "hugepage pool missing or short at boot (HugePages_Total: ${hp:-unreadable}, want >= 3072)"
fi

# #895: machine-id must be assigned once and then STAY. An empty-baked image with no
# restore-from-/data unit regenerates a new transient id on EVERY boot (systemd's own
# first-boot semantics on a permanently read-only /etc) — worse than the bug being fixed.
Expand Down Expand Up @@ -835,7 +846,8 @@ phase_install() {
test -e \"\$T/pithead/reinstall-sentinel\" && s=\$s,USER-DATA-SURVIVED
umount \"\$T\" 2>&1 || s=\$s,UMOUNT-FAILED
echo \"verdict=\$s\"" 2>&1)
if printf '%s' "$wout" | grep -q "verdict=OK"; then
# Anchored: "verdict=OK,USER-DATA-SURVIVED" must NOT pass — the suffix flags are the failure.
if printf '%s' "$wout" | grep -qx "verdict=OK"; then
ok "wipe=data KEPT both chains and dropped the user data"
else
bad "wipe=data got the split wrong: $(printf '%s' "$wout" | tr '\n' ' ' | cut -c1-300)"
Expand Down Expand Up @@ -1830,12 +1842,17 @@ phase_fault() {
# refusal — refusing to install is correct, crashing is not, and bricking is disqualifying.
info "fault C — install a deliberately corrupted bundle"
_ssh "dd if=/dev/urandom of=/data/update.bundle bs=1M seek=8 count=2 conv=notrunc" >/dev/null 2>&1 || true
out=$(_ssh "$(_install_cmd /data/update.bundle) 2>&1" || true)
local corrupt_rc=0
out=$(_ssh "$(_install_cmd /data/update.bundle) 2>&1") || corrupt_rc=$?
if printf '%s' "$out" | grep -qi "panic"; then
bad "C: the updater PANICKED on a corrupt bundle instead of refusing it"
info " $(printf '%s' "$out" | grep -i panic | head -1 | cut -c1-150)"
elif [ "$corrupt_rc" -eq 0 ]; then
# The old check only asserted the absence of "panic" — an install that quietly ACCEPTED
# the tampered bundle read as a pass. The refusal itself is the assertion.
bad "C: a corrupted/tampered bundle was ACCEPTED (installer exited 0)"
else
ok "C: a corrupt bundle is refused without crashing"
ok "C: a corrupt bundle is refused without crashing (exit $corrupt_rc)"
fi
if _wait_ssh 300; then
ok "C: still boots after being handed a corrupt bundle (marker '$(_marker)')"
Expand Down
11 changes: 11 additions & 0 deletions tests/os/verify-image.sh
Original file line number Diff line number Diff line change
Expand Up @@ -109,6 +109,17 @@ chk "RebootWatchdogSec set (reboot itself is watched)" 'grep -q "^RebootWatchdog
chk "cpu-governor unit present" '[ -s "$ROOT/etc/systemd/system/pithead-cpu-governor.service" ]'
chk "cpu-governor unit enabled" 'test -L "$ROOT/etc/systemd/system/multi-user.target.wants/pithead-cpu-governor.service"'
chk "cpu-governor asks for performance" 'grep -q "frequency-set -g performance" "$ROOT/etc/systemd/system/pithead-cpu-governor.service"'
# The factory-reset / wedged-/data recovery path: every other boot-path unit is pinned here, and
# this one is the last resort an operator has on a shell-less box — a build that drops it ships a
# machine that cannot recover itself.
chk "data-reset unit shipped" '[ -s "$ROOT/etc/systemd/system/pithead-data-reset.service" ]'
chk "data-reset unit enabled (local-fs transaction)" 'test -L "$ROOT/etc/systemd/system/local-fs.target.wants/pithead-data-reset.service"'
chk "data-reset script present and executable" 'test -x "$ROOT/usr/local/sbin/pithead-data-reset"'
chk "data-reset ordered before /data mounts (a mounted partition cannot be reformatted)" \
'grep -q "^Before=data.mount local-fs.target" "$ROOT/etc/systemd/system/pithead-data-reset.service"'
# Hugepages: the sysctl the Dockerfile calls load-bearing for the memory caps.
chk "hugepage reservation baked (RandomX dataset must land in hugetlbfs)" \
'grep -q "vm.nr_hugepages=3072" "$ROOT/etc/sysctl.d/99-pithead-hugepages.conf"'

echo "==> docker-export artefacts (all six members)"
chk "no /.dockerenv" '[ ! -e "$ROOT/.dockerenv" ]'
Expand Down