From 3569b2091b8477034913e5c5166936a3ff40d36e Mon Sep 17 00:00:00 2001 From: "Joshua (D) Drake" <136637981+ChronicallyJD@users.noreply.github.com> Date: Wed, 5 Aug 2026 11:33:59 -0600 Subject: [PATCH 1/3] bench: a standing join benchmark, and the finding it makes checkable (#401) Every other query on the benchmark page reads one table. Star schemas and dimension joins are a large part of what columnar storage is bought for, and we had no measurement of them. Not "we know and it is bad". We did not know. Five shapes: a no-join control, a selective dimension join, an unselective join, a multi-dimension star, and a wide projection under a join. Arms interleaved per shape rather than swept, because a sweep gives its first arm the cold cache (#271). TimescaleDB is supported and announces itself as skipped when absent. The result, at 20,000,000 rows on realistic data: no join, the control 0.53x we are 1.87 times FASTER selective dimension join 1.36x wide projection under a join 2.98x storage 0.141x 7.1 times smaller Read the first two together. Same rows, same bytes, same encoding. Our vectorized aggregate only sits directly above our scan, so a join between the scan and the aggregate disables it, and we then compete row at a time. A join does not cost us through the join. It costs us by disabling the thing we are fast at. The harness ASSERTS that mechanism instead of describing it. It fails the run if the control is not vectorized and fails it if the join arm is. Either half alone is consistent with the feature simply being switched off. Two mistakes of mine are built into its shape. The data shape is an arm, not an assumption, because my first result on this issue used random() float8 and reported a gap the realistic shape more than halves. And the no-join control is mandatory, because I once refuted my own correct hypothesis by running the decomposition with the vectorized aggregate turned off, which made both arms equally slow. docs/benchmarks.md carried "we have not measured them. See #401" from #402. That is now false, so it is replaced by the measurement rather than left to rot. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01UqprqkCXuH8SegiZejE1Tw --- bench/run_bench_join.sh | 313 ++++++++++++++++++++++++++++++++++++++++ docs/benchmarks.md | 73 +++++++++- 2 files changed, 382 insertions(+), 4 deletions(-) create mode 100644 bench/run_bench_join.sh diff --git a/bench/run_bench_join.sh b/bench/run_bench_join.sh new file mode 100644 index 0000000..6cf6870 --- /dev/null +++ b/bench/run_bench_join.sh @@ -0,0 +1,313 @@ +#!/usr/bin/env bash +# +# Join-heavy benchmark (issue #401). +# +# Every query in the main benchmark reads one table. Star schemas and dimension +# joins are the shape columnar storage is usually bought for, and until this +# existed we had no measurement of them at all. Not "we know and it is bad". We +# did not know. +# +# --------------------------------------------------------------------------- +# The two mistakes this harness is built to prevent, because I made both +# --------------------------------------------------------------------------- +# +# 1. THE DATA SHAPE IS AN ARM, NOT AN ASSUMPTION. My first result on #401 used +# random() float8 metrics: 20 million distinct values per column, maximum +# entropy, nothing to compress. That is the one shape where columnar storage +# cannot win, and I published "we lose by up to 6.2x" from it. Quantising the +# metrics to two decimals, which is what an instrument actually reports, more +# than halved the worst gap. So both shapes are measured here and both are +# reported. Neither is allowed to be implicit. +# +# 2. THE NO-JOIN CONTROL IS MANDATORY. The finding this benchmark exists to +# surface is invisible without it, and I once "refuted" my own correct +# hypothesis by running the decomposition with the vectorized aggregate +# turned off, so the no-join arm was unvectorized too and both looked equally +# slow. S0 runs the same aggregate over the same rows with no join, and the +# premise checks below assert that the aggregate really is vectorized there +# and really is not under the join. +# +# --------------------------------------------------------------------------- +# What it measures +# --------------------------------------------------------------------------- +# +# S0 no join at all, the control that makes the rest legible +# S1 selective dimension join, small filtered dimension against a large fact +# S2 unselective join, the dimension barely filters, so the join does real work +# S3 multi-dimension star, where join order starts to matter +# S4 wide projection under a join, does the join force decoding of columns +# nothing needs +# +# Arms: heap, pgColumnar, and TimescaleDB when it is installed. A missing +# TimescaleDB is announced, never silently skipped. +# +# --------------------------------------------------------------------------- +# Usage +# --------------------------------------------------------------------------- +# +# bench/run_bench_join.sh [PG_CONFIG] +# +# BENCH_SCALE fact rows (default 20000000) +# BENCH_REPS timed repetitions, median reported (default 3) +# BENCH_PORT throwaway cluster port (default 55471) +# BENCH_SHAPES quantised,random (default both) +# BENCH_ARMS heap,columnar,timescale (default heap,columnar) +# BENCH_KEEP 1 to leave the cluster up afterwards (default 0) +# +# Run on an idle machine against a non-assert build. Written fresh for +# pgColumnar; it reuses no upstream benchmark script. +set -uo pipefail + +PG_CONFIG="${1:-/usr/local/pg18n/bin/pg_config}" +SCALE="${BENCH_SCALE:-20000000}" +REPS="${BENCH_REPS:-3}" +PORT="${BENCH_PORT:-55471}" +SHAPES="${BENCH_SHAPES:-quantised,random}" +ARMS="${BENCH_ARMS:-heap,columnar}" +KEEP="${BENCH_KEEP:-0}" + +BINDIR="$("$PG_CONFIG" --bindir)" || { echo "FATAL no pg_config"; exit 1; } +WORKDIR="$(mktemp -d /tmp/pgc-joinbench.XXXXXX)" +PGDATA="$WORKDIR/data" +fail=0 + +note() { printf '%s\n' "$*"; } +require() { # require + if [ "$2" != "$3" ]; then + printf 'FAIL premise: %s (got [%s] want [%s])\n' "$1" "$2" "$3" + fail=1 + return 1 + fi + printf 'ok premise: %s\n' "$1" + return 0 +} + +note "== join benchmark, $SCALE fact rows, $REPS reps, shapes [$SHAPES], arms [$ARMS]" +note " $("$PG_CONFIG" --version)" + +"$BINDIR/initdb" -D "$PGDATA" --locale=C -U postgres > "$WORKDIR/initdb.log" 2>&1 \ + || { tail -10 "$WORKDIR/initdb.log"; echo FATAL initdb; exit 1; } +NCPU=$(nproc) +# TimescaleDB has to be preloaded before the cluster starts, so decide here. +# A requested arm whose library is absent is announced, never silently dropped. +PRELOAD=pgcolumnar +case ",$ARMS," in + *,timescale,*) + if [ -f "$("$PG_CONFIG" --pkglibdir)/timescaledb.so" ]; then + PRELOAD="timescaledb,pgcolumnar" + else + note "SKIP the timescale arm was asked for and timescaledb.so is not installed" + ARMS="${ARMS//timescale/}"; ARMS="${ARMS//,,/,}"; ARMS="${ARMS%,}"; ARMS="${ARMS#,}" + fi ;; +esac +MEMGB=$(awk '/MemTotal/ {printf "%d", $2/1024/1024}' /proc/meminfo) +cat >> "$PGDATA/postgresql.conf" </dev/null 2>&1 \ + || { tail -20 "$WORKDIR/server.log"; echo FATAL start; exit 1; } +cleanup() { + [ "$KEEP" = 1 ] || "$BINDIR/pg_ctl" -D "$PGDATA" -w stop -m immediate >/dev/null 2>&1 + [ "$KEEP" = 1 ] || rm -rf "$WORKDIR" +} +trap cleanup EXIT +"$BINDIR/createdb" -h /tmp -p "$PORT" -U postgres joinbench >/dev/null 2>&1 +P="$BINDIR/psql -h /tmp -p $PORT -U postgres -d joinbench -X -At" +$P -c "CREATE EXTENSION pgcolumnar;" >/dev/null 2>&1 || { echo FATAL extension; exit 1; } + +# Serial, so the comparison is of the storage and not of the worker count. The +# main harness does the same and states why: parallelism is a separate axis, and +# mixing it in makes a storage difference unreadable. +SERIAL="SET max_parallel_workers_per_gather=0;" +# The accelerations that matter for an aggregate over our scan. Off by default +# (#401, #421), and a benchmark that leaves them off measures the wrong thing. +VEC="SET pgcolumnar.enable_vectorization=on; SET pgcolumnar.enable_ungrouped_vector_agg=on;" + +# --------------------------------------------------------------------------- +# Fixture +# --------------------------------------------------------------------------- +# One fact table per (arm, shape). Dimensions are heap in every arm, which is the +# realistic deployment and is also what keeps the join method comparable. +# +# quantised: metrics rounded to two decimals, what an instrument reports. +# random: raw random(), maximum entropy, the worst case for any column store. +metric_expr() { # metric_expr + case "$1" in + quantised) echo "round((random()*100)::numeric, 2)::float8" ;; + random) echo "random()" ;; + esac +} + +note "== dimensions" +$P -c "CREATE TABLE hosts (host_id int PRIMARY KEY, hostname text, region text, tier int); + INSERT INTO hosts SELECT g, 'host'||g, 'r'||(g%20), g%5 FROM generate_series(1,2000) g; + CREATE TABLE regions (region text PRIMARY KEY, continent text); + INSERT INTO regions SELECT 'r'||g, 'c'||(g%4) FROM generate_series(0,19) g; + ANALYZE hosts; ANALYZE regions;" >/dev/null 2>&1 +require "the dimension loaded" "$($P -c 'SELECT count(*) FROM hosts')" "2000" || exit 1 + +IFS=',' read -r -a SHAPE_A <<< "$SHAPES" +IFS=',' read -r -a ARM_A <<< "$ARMS" + +fact_name() { echo "fact_$1_$2"; } # arm, shape + +for shape in "${SHAPE_A[@]}"; do + m=$(metric_expr "$shape") + for arm in "${ARM_A[@]}"; do + t=$(fact_name "$arm" "$shape") + case "$arm" in + columnar) using="USING pgcolumnar" ;; + *) using="" ;; + esac + note "== loading $t" + $P -c "CREATE TABLE $t (ts timestamptz, host_id int, m1 float8, m2 float8, + m3 float8, m4 float8, m5 float8, m6 float8, + m7 float8, m8 float8) $using;" >/dev/null 2>&1 + $P -c "INSERT INTO $t SELECT + '2024-01-01'::timestamptz + (g % 8640000) * interval '10 second', + 1 + (g % 2000), $m, $m, $m, $m, $m, $m, $m, $m + FROM generate_series(1,$SCALE) g;" >/dev/null 2>&1 + $P -c "VACUUM ANALYZE $t;" >/dev/null 2>&1 + n=$($P -c "SELECT count(*) FROM $t") + sz=$($P -c "SELECT pg_total_relation_size('$t')") + note " $t: $n rows, $sz bytes" + require "$t loaded every row" "$n" "$SCALE" || fail=1 + # No index on either fact table, so neither arm gets an index advantage. + ni=$($P -c "SELECT count(*) FROM pg_index WHERE indrelid='$t'::regclass") + require "$t carries no index" "$ni" "0" || fail=1 + done +done +[ "$fail" = 0 ] || { echo "FATAL a fixture premise failed"; exit 1; } + +# --------------------------------------------------------------------------- +# The shapes +# --------------------------------------------------------------------------- +# %T is the fact table. S0 is the control and has no join at all. +q_for() { # q_for + case "$1" in + S0) echo "SELECT avg(m1) FROM %T" ;; + S1) echo "SELECT avg(f.m1) FROM %T f JOIN hosts h USING (host_id) WHERE h.tier = 3" ;; + S2) echo "SELECT avg(f.m1) FROM %T f JOIN hosts h USING (host_id) WHERE h.tier < 5" ;; + S3) echo "SELECT r.continent, avg(f.m1) FROM %T f JOIN hosts h USING (host_id) + JOIN regions r ON r.region = h.region WHERE h.tier < 4 + GROUP BY r.continent ORDER BY 1" ;; + S4) echo "SELECT avg(f.m1+f.m2+f.m3+f.m4+f.m5+f.m6+f.m7+f.m8) + FROM %T f JOIN hosts h USING (host_id) WHERE h.tier = 3" ;; + esac +} +q_desc() { + case "$1" in + S0) echo "no join (control)" ;; + S1) echo "selective dimension join" ;; + S2) echo "unselective join" ;; + S3) echo "multi-dimension star" ;; + S4) echo "wide projection under a join" ;; + esac +} +SHAPE_IDS="S0 S1 S2 S3 S4" + +# One timed run, in milliseconds. +timed() { # timed + local sql="${2//%T/$1}" out + out=$("$BINDIR/psql" -h /tmp -p "$PORT" -U postgres -d joinbench -X -t \ + -c "$SERIAL $VEC" -c '\timing on' -c "$sql" 2>&1) + grep -qiE '^ERROR' <<<"$out" && { echo ERR; return; } + grep -oE 'Time: [0-9.]+ ms' <<<"$out" | tail -1 | grep -oE '[0-9.]+' +} +median() { printf '%s\n' "$@" | sort -n | awk '{a[NR]=$1} END {print a[int((NR+1)/2)]}'; } +plan_of() { # plan_of
+ "$BINDIR/psql" -h /tmp -p "$PORT" -U postgres -d joinbench -X -At \ + -c "$SERIAL $VEC" -c "EXPLAIN (COSTS OFF) ${2//%T/$1}" 2>&1 +} + +# --------------------------------------------------------------------------- +# The premises that make the headline claim checkable rather than narrated +# --------------------------------------------------------------------------- +note "== premises on the plans" +for shape in "${SHAPE_A[@]}"; do + ct=$(fact_name columnar "$shape") + printf '%s\n' "${ARM_A[@]}" | grep -q columnar || continue + + # The claim this whole issue turns on: our vectorized aggregate sits directly + # above our scan, so a join between them disables it. Asserted on both sides, + # because either half alone is consistent with the feature simply being off. + p0=$(plan_of "$ct" "$(q_for S0)") + p1=$(plan_of "$ct" "$(q_for S1)") + require "[$shape] the no-join control really is vectorized" \ + "$(grep -qi 'Vectorized' <<<"$p0" && echo yes || echo no)" "yes" || fail=1 + require "[$shape] a join between scan and aggregate disables it" \ + "$(grep -qi 'Vectorized' <<<"$p1" && echo yes || echo no)" "no" || fail=1 + + # A different join method either side makes the comparison meaningless. + for shp in S1 S2 S4; do + hm=$(plan_of "$(fact_name heap "$shape")" "$(q_for $shp)" | grep -oE 'Hash Join|Merge Join|Nested Loop' | head -1) + cm=$(plan_of "$ct" "$(q_for $shp)" | grep -oE 'Hash Join|Merge Join|Nested Loop' | head -1) + require "[$shape] $shp uses the same join method in both arms ($hm)" "$cm" "$hm" || fail=1 + done +done + +# --------------------------------------------------------------------------- +# Measure, interleaved +# --------------------------------------------------------------------------- +# One shape at a time across every arm before moving on. A sweep gives its first +# arm the cold cache and every later arm a warm one (#271). +note "== timings (median of $REPS, interleaved across arms)" +declare -A MS +for shape in "${SHAPE_A[@]}"; do + for shp in $SHAPE_IDS; do + for arm in "${ARM_A[@]}"; do + t=$(fact_name "$arm" "$shape") + runs=() + for r in $(seq 1 "$REPS"); do runs+=("$(timed "$t" "$(q_for $shp)")"); done + MS["$shape:$shp:$arm"]=$(median "${runs[@]}") + done + printf ' %-10s %-3s' "$shape" "$shp" + for arm in "${ARM_A[@]}"; do printf ' %-9s=%-10s' "$arm" "${MS["$shape:$shp:$arm"]}"; done + printf '\n' + done +done + +# --------------------------------------------------------------------------- +# Report +# --------------------------------------------------------------------------- +echo +echo "=============== JOIN BENCHMARK (#401) ===============" +echo "rows: $SCALE reps: $REPS serial $(nproc) cores" +echo +for shape in "${SHAPE_A[@]}"; do + echo "-- data shape: $shape" + printf '%-4s %-32s %12s %12s %10s\n' id shape heap columnar 'col/heap' + for shp in $SHAPE_IDS; do + h="${MS["$shape:$shp:heap"]:-}" + c="${MS["$shape:$shp:columnar"]:-}" + r="-" + case "$h$c" in + *ERR*|"") ;; + *) r=$(awk -v a="$c" -v b="$h" 'BEGIN { printf "%.2f", a / b }') ;; + esac + printf '%-4s %-32s %12s %12s %10s\n' "$shp" "$(q_desc $shp)" "$h" "$c" "$r" + done + hs=$($P -c "SELECT pg_total_relation_size('$(fact_name heap $shape)')") + cs=$($P -c "SELECT pg_total_relation_size('$(fact_name columnar $shape)')") + printf '%-4s %-32s %12s %12s %10s\n' "" "total relation size (bytes)" "$hs" "$cs" \ + "$(awk -v a="$cs" -v b="$hs" 'BEGIN { printf "%.3f", a / b }')" + echo +done +echo "S0 against S1 is the finding: the same aggregate over the same rows, with" +echo "and without a join. Read those two lines together before anything else." +if [ "$fail" != 0 ]; then + echo + echo "A PREMISE FAILED. Do not quote these numbers." + exit 1 +fi +exit 0 diff --git a/docs/benchmarks.md b/docs/benchmarks.md index d15f409..9e8e91d 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -653,16 +653,81 @@ both are larger than anything in it: The point lookup is the one number that moved the wrong way, and it moved a long way. See the note above it. +## Joins + +Every other query on this page reads one table. Star schemas and dimension joins +are a large part of what columnar storage is bought for, so `bench/run_bench_join.sh` +measures them. + +```sh +BENCH_SCALE=20000000 bench/run_bench_join.sh /path/to/pg18n/bin/pg_config +``` + +The numbers below are one run on 2026-08-05. The conditions were PostgreSQL 18.4 +non-assert, 16 cores, 20,000,000 fact rows, serial execution and the median of +three repetitions. The fact table has eight metric columns. The dimensions are +heap tables in both arms, which is the realistic deployment. + +The data shape is an arm rather than an assumption. `quantised` rounds each metric +to two decimals, which is what an instrument reports. `random` stores raw +`random()`, which has maximum entropy and cannot be compressed. An earlier +measurement of these shapes used `random()` without noticing, and reported a gap +that was much larger than the realistic one. + +### Realistic data + +| shape | heap | columnar | columnar over heap | +| --- | ---: | ---: | ---: | +| no join, the control | 1,536.8 ms | 820.2 ms | **0.53x** | +| selective dimension join | 1,952.3 ms | 2,655.6 ms | 1.36x | +| unselective join | 3,152.8 ms | 3,694.1 ms | 1.17x | +| multi-dimension star | 6,560.8 ms | 6,936.6 ms | 1.06x | +| wide projection under a join | 2,293.0 ms | 6,821.8 ms | 2.98x | +| total relation size | 2,185,166,848 B | 307,109,888 B | 0.141x | + +### Incompressible data + +| shape | heap | columnar | columnar over heap | +| --- | ---: | ---: | ---: | +| no join, the control | 1,533.6 ms | 1,912.0 ms | 1.25x | +| selective dimension join | 1,968.6 ms | 3,957.4 ms | 2.01x | +| unselective join | 3,161.9 ms | 4,990.6 ms | 1.58x | +| multi-dimension star | 6,581.0 ms | 8,176.0 ms | 1.24x | +| wide projection under a join | 2,311.6 ms | 16,269.1 ms | 7.04x | +| total relation size | 2,185,166,848 B | 1,306,492,928 B | 0.598x | + +### What the control says + +Read the first two rows of the first table together. **Without a join we are 1.87 +times faster. Add a join and we are 1.36 times slower.** The rows are the same, +the bytes are the same, and the encoding is the same. + +The reason is structural. Our vectorized aggregate only sits directly above our +scan. Put a join between the scan and the aggregate and it cannot apply. We +then compete row at a time, against a format built for exactly that. The harness asserts +this rather than describing it. It fails the run if the control is not vectorized, +and it fails the run if the join arm is. + +So a join does not cost us through the join itself. It costs us by disabling the +thing we are fast at. + +### What the two data shapes say + +Compressibility is a second and separate effect. On incompressible data we lose +even with no join at all, and the wide projection shape goes from 2.98x to 7.04x. +Storage goes from 7.1 times smaller to 1.7 times smaller. + +Neither effect explains the other. Both are real. + ## What this page does not measure **Every query on this page reads one table.** The TSBS workload is time-series shaped. All eight of its queries are a scan, a filter and an aggregate over a single relation. None of them joins. -So the join-heavy analytical shapes are absent: star schemas, fact tables joined -to dimensions, and the selective-dimension-join pattern. Those are a large part of -what columnar storage is usually bought for, and this page says nothing about -them. Not that we do badly on them. We have not measured them. See #401. +The join-heavy analytical shapes are measured separately, in the Joins section +above, on a fixture built for them. That section is the evidence about star +schemas and dimension joins. This section is about the TSBS numbers only. Joins themselves work. A columnar table joins a heap table in either direction, and joins another columnar table. Hash, merge and nested-loop strategies all return From 8e56e216f72f6f3ba5c28c0b309c6007bb610813 Mon Sep 17 00:00:00 2001 From: "Joshua (D) Drake" <136637981+ChronicallyJD@users.noreply.github.com> Date: Wed, 5 Aug 2026 12:51:57 -0600 Subject: [PATCH 2/3] bench: withdraw the TimescaleDB arm rather than ship one that builds a heap table (#401) The harness advertised BENCH_ARMS=timescale and, for that arm, ran case "$arm" in columnar) using="USING pgcolumnar" ;; *) using="" ;; esac which is a plain heap table. No hypertable, no columnstore. It would have run, produced plausible numbers, and labelled heap as TimescaleDB. That is the vacuous-arm shape I have spent the week filing against other people, behind a flag nobody had exercised. Building it properly turned up #428 instead: pgColumnar and TimescaleDB both register a custom scan node named "ColumnarScan", so a parallel worker resolves the node to the wrong extension's callbacks and any parallel query over a compressed hypertable fails. Until that is fixed the arm could only run serially, which is not a fair comparison against two arms that may parallelise. So the flag is removed rather than left accepting a value that produces a mislabelled number, and the header says why. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01UqprqkCXuH8SegiZejE1Tw --- bench/run_bench_join.sh | 22 ++++++++++------------ 1 file changed, 10 insertions(+), 12 deletions(-) diff --git a/bench/run_bench_join.sh b/bench/run_bench_join.sh index 6cf6870..4b6506b 100644 --- a/bench/run_bench_join.sh +++ b/bench/run_bench_join.sh @@ -38,8 +38,8 @@ # S4 wide projection under a join, does the join force decoding of columns # nothing needs # -# Arms: heap, pgColumnar, and TimescaleDB when it is installed. A missing -# TimescaleDB is announced, never silently skipped. +# Arms: heap and pgColumnar. See the BENCH_ARMS note below for why TimescaleDB is +# not one of them. # # --------------------------------------------------------------------------- # Usage @@ -51,7 +51,14 @@ # BENCH_REPS timed repetitions, median reported (default 3) # BENCH_PORT throwaway cluster port (default 55471) # BENCH_SHAPES quantised,random (default both) -# BENCH_ARMS heap,columnar,timescale (default heap,columnar) +# BENCH_ARMS heap,columnar (default heap,columnar) +# +# A TimescaleDB arm is NOT offered. It needs create_hypertable plus the +# columnstore conversion, and until #428 is fixed a parallel query over a +# compressed hypertable fails outright when pgcolumnar is loaded, so the arm +# could only ever run serially and would not be a fair comparison. The flag is +# absent rather than accepting a value that would have built a plain heap table +# and labelled it TimescaleDB. # BENCH_KEEP 1 to leave the cluster up afterwards (default 0) # # Run on an idle machine against a non-assert build. Written fresh for @@ -91,15 +98,6 @@ NCPU=$(nproc) # TimescaleDB has to be preloaded before the cluster starts, so decide here. # A requested arm whose library is absent is announced, never silently dropped. PRELOAD=pgcolumnar -case ",$ARMS," in - *,timescale,*) - if [ -f "$("$PG_CONFIG" --pkglibdir)/timescaledb.so" ]; then - PRELOAD="timescaledb,pgcolumnar" - else - note "SKIP the timescale arm was asked for and timescaledb.so is not installed" - ARMS="${ARMS//timescale/}"; ARMS="${ARMS//,,/,}"; ARMS="${ARMS%,}"; ARMS="${ARMS#,}" - fi ;; -esac MEMGB=$(awk '/MemTotal/ {printf "%d", $2/1024/1024}' /proc/meminfo) cat >> "$PGDATA/postgresql.conf" < Date: Wed, 5 Aug 2026 14:55:36 -0600 Subject: [PATCH 3/3] bench: the ratio guard only caught BOTH sides missing, not one (#401) case "$h$c" in *ERR*|"") ;; *) r=$(awk ... a / b) ;; esac "$h$c" is empty only when both are. One empty side concatenates to a non-empty string, falls through, and divides: heap=[1500] col=[ ] -> 0.00 reads as a 100 percent win heap=[ ] col=[ 800] -> inf 0.00 is the dangerous one, in a table whose whole purpose is to be quoted somewhere without the run log beside it. timed() returns empty on every path the ERROR branch does not catch: a cancelled query, a lost connection, or psql changing its timing format. Replaced with a ratio() that tests each side separately, refuses ERR, and refuses a zero denominator. It is self-tested before any number is printed, because a ratio helper that silently passes a half-empty pair is worse than none. bench/ does not source test/lib.sh, so this is the local counterpart of check_ratio from #418 and #422. Same class as the bc failure that started #418, and the third time this shape has turned up in my own work today. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01UqprqkCXuH8SegiZejE1Tw --- bench/run_bench_join.sh | 45 ++++++++++++++++++++++++++++++++++++----- 1 file changed, 40 insertions(+), 5 deletions(-) diff --git a/bench/run_bench_join.sh b/bench/run_bench_join.sh index 4b6506b..f9af9d4 100644 --- a/bench/run_bench_join.sh +++ b/bench/run_bench_join.sh @@ -89,6 +89,45 @@ require() { # require return 0 } +# Ratio of two timings, or "-" when either side is missing. +# +# Testing the CONCATENATION "$a$b" only catches BOTH sides missing. One empty +# side concatenates to a non-empty string, falls through, and divides: +# +# heap=[1500] col=[ ] -> 0.00 reads as a 100 percent win +# heap=[ ] col=[ 800] -> inf +# +# 0.00 is the dangerous one, in a table whose whole purpose is to be quoted +# somewhere without the run log beside it. Same class as #418, and timed() +# returns empty on any path the ERROR branch does not catch: a cancelled query, +# a lost connection, or psql changing its timing format. +# +# test/lib.sh has check_ratio for suites. This is bench/, which does not source +# the test harness, so the guard lives here and is self-tested below. +ratio() { # ratio -> "n.nn" or "-" + case "$1" in '' | *ERR*) echo '-'; return ;; esac + case "$2" in '' | *ERR*) echo '-'; return ;; esac + if [ "$(awk -v x="$2" 'BEGIN { print (x + 0 == 0) ? 1 : 0 }')" = 1 ]; then + echo '-'; return + fi + awk -v a="$1" -v b="$2" 'BEGIN { printf "%.2f", a / b }' +} + +# Prove the guard rejects what it claims to, before any number is printed. A +# ratio helper that silently passes a half-empty pair is worse than none, and +# this is the third time this shape has appeared in my own work today. +ratio_selftest() { + local bad=0 got + got=$(ratio "" "800"); [ "$got" = "-" ] || { echo "FATAL ratio() accepted an empty numerator"; bad=1; } + got=$(ratio "1500" ""); [ "$got" = "-" ] || { echo "FATAL ratio() accepted an empty denominator"; bad=1; } + got=$(ratio "1500" "0"); [ "$got" = "-" ] || { echo "FATAL ratio() divided by zero"; bad=1; } + got=$(ratio "ERR" "800"); [ "$got" = "-" ] || { echo "FATAL ratio() accepted ERR"; bad=1; } + got=$(ratio "800" "1600");[ "$got" = "0.50" ] || { echo "FATAL ratio() got $got for 800/1600"; bad=1; } + [ "$bad" = 0 ] || exit 1 + note " ratio guard self-test: ok" +} +ratio_selftest + note "== join benchmark, $SCALE fact rows, $REPS reps, shapes [$SHAPES], arms [$ARMS]" note " $("$PG_CONFIG" --version)" @@ -288,11 +327,7 @@ for shape in "${SHAPE_A[@]}"; do for shp in $SHAPE_IDS; do h="${MS["$shape:$shp:heap"]:-}" c="${MS["$shape:$shp:columnar"]:-}" - r="-" - case "$h$c" in - *ERR*|"") ;; - *) r=$(awk -v a="$c" -v b="$h" 'BEGIN { printf "%.2f", a / b }') ;; - esac + r=$(ratio "$c" "$h") printf '%-4s %-32s %12s %12s %10s\n' "$shp" "$(q_desc $shp)" "$h" "$c" "$r" done hs=$($P -c "SELECT pg_total_relation_size('$(fact_name heap $shape)')")