diff --git a/bench/run_clickbench.sh b/bench/run_clickbench.sh index 71d5aecb..fd6b656d 100755 --- a/bench/run_clickbench.sh +++ b/bench/run_clickbench.sh @@ -161,7 +161,7 @@ PSQL="$BINDIR/psql -h /tmp -p $CB_PORT -U postgres -d clickbench -X -q" # 0. Preconditions, once, loudly # --------------------------------------------------------------------------- note "== preconditions" -for t in curl awk zcat "$BINDIR/psql" "$BINDIR/initdb" "$BINDIR/pg_ctl"; do +for t in curl awk zcat sha256sum cmp "$BINDIR/psql" "$BINDIR/initdb" "$BINDIR/pg_ctl"; do command -v "$t" >/dev/null 2>&1 || [ -x "$t" ] || die "missing tool: $t" done mkdir -p "$CB_DATA" || die "cannot write $CB_DATA" @@ -173,15 +173,63 @@ note " rows: $CB_ROWS tries: $CB_TRIES arms: $CB_ARMS" # 1. The definition, fetched rather than vendored # --------------------------------------------------------------------------- note "== fetching the ClickBench definition (not stored in this repository)" +# +# Re-fetched EVERY run, on purpose: the point of not vendoring the definition is +# that the benchmark tracks upstream, and a cached copy silently pins it to +# whatever was current the first time this ever ran on the machine. An earlier +# version kept the cache when the file was merely present, so "fetched at run +# time" was true once and false afterwards. +# +# PGC_CB_OFFLINE=1 keeps an existing copy without reaching the network, for a +# machine that has none. It says so in the output, because a run against a stale +# definition is not comparable to one against the current definition and the +# difference must not be invisible in the log. +# +# curl gets --fail, and the fetched bytes are checked for the shape they are +# supposed to have before they replace a good copy. Both matter, and the second +# is not redundant: +# +# -sSL without --fail treats HTTP 404 as SUCCESS. GitHub answers a bad path +# with a 21-byte "404: Not Found" body and a 404 status, so curl exited 0 and +# wrote that page over the real definition. `[ -s ]` was satisfied -- the page +# is not empty -- the digest was computed and printed, and the run continued +# against an error page. Measured while writing this, with a deliberately bad +# URL. The give-away was both files having the same digest. +# +# So: --fail rejects the status, the grep rejects a body that is not the file we +# asked for, and neither replaces the existing copy until both pass. for f in create.sql queries.sql; do - if [ ! -s "$CB_DATA/$f" ]; then - curl -sSL --retry 3 --max-time 120 -o "$CB_DATA/$f" "$CB_URL_BASE/$f" \ - || die "could not fetch $f" - note " fetched $f" + case "$f" in + create.sql) shape='CREATE TABLE' ;; + queries.sql) shape='SELECT' ;; + esac + if [ "${PGC_CB_OFFLINE:-0}" = 1 ]; then + [ -s "$CB_DATA/$f" ] || die "PGC_CB_OFFLINE=1 but $CB_DATA/$f is not there" + note " OFFLINE: using the existing $f, which may not be current" + elif curl -fsSL --retry 3 --max-time 120 -o "$CB_DATA/$f.new" "$CB_URL_BASE/$f" \ + && [ -s "$CB_DATA/$f.new" ] \ + && grep -qi "$shape" "$CB_DATA/$f.new"; then + if [ -s "$CB_DATA/$f" ] && ! cmp -s "$CB_DATA/$f" "$CB_DATA/$f.new"; then + note " fetched $f -- CHANGED since the last run on this machine" + else + note " fetched $f" + fi + mv "$CB_DATA/$f.new" "$CB_DATA/$f" else - note " have $f already" + rm -f "$CB_DATA/$f.new" + die "could not fetch $f (set PGC_CB_OFFLINE=1 to run against the copy already here)" fi done + +# The digest of what actually ran. A published number is only reproducible if the +# definition it came from can be identified, and "fetched from main" does not +# identify anything -- main moves. Cite these beside any result. +CB_SHA_CREATE=$(sha256sum "$CB_DATA/create.sql" | cut -c1-16) +CB_SHA_QUERIES=$(sha256sum "$CB_DATA/queries.sql" | cut -c1-16) +note " definition: create.sql $CB_SHA_CREATE queries.sql $CB_SHA_QUERIES" +require "the create.sql digest was computed" "${#CB_SHA_CREATE}" "16" || exit 1 +require "the queries.sql digest was computed" "${#CB_SHA_QUERIES}" "16" || exit 1 + NQUERIES=$(grep -c 'SELECT' "$CB_DATA/queries.sql") require "the query file holds 43 queries" "$NQUERIES" "43" || exit 1 # The column count is asserted against the CREATED TABLE further down, not diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 8b5c7eaa..fe8b3620 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -739,6 +739,29 @@ project's MIT license does not carry. The harness fetches the definition from upstream at run time and copies nothing. Our comparison oracle is the heap arm of the same run. +The definition is re-fetched on every run, so the benchmark tracks the current +upstream rather than a copy taken once. That means upstream can change what is +measured between two runs, and a result is only comparable to another result +taken against the same definition. Each run therefore prints the SHA-256 of both +fetched files: + +``` + definition: create.sql 42d28575fd59fb4a queries.sql a7d6673357348ee9 +``` + +Those are the real digests of upstream `main` as of 2026-08-06, 43 queries. + +Cite those beside any number taken from a run. + +`PGC_CB_OFFLINE=1` runs against the copy already on the machine, for a host with +no network. It says so in the output. A run against a stale definition is not +comparable to one against the current definition. + +The table below predates the digest being recorded, so it cannot cite one. The +run was on 2026-08-05, and upstream carried the digests above a day later. That +is an inference and not a measurement. The next run is the first that will state +it. + The numbers below are one run on 2026-08-05. The conditions were PostgreSQL 18.4 non-assert, 16 cores, 62 GB of memory, and 11,110,833 rows. That row count is every ninth row of the real 100 million row table. The reported time is the best