diff --git a/Dockerfile b/Dockerfile index 4c9a399..063a81c 100644 --- a/Dockerfile +++ b/Dockerfile @@ -3,6 +3,13 @@ FROM ubuntu:24.04 # Prevent interactive prompts during package installation ENV DEBIAN_FRONTEND=noninteractive +# Populated by BuildKit with "amd64" or "arm64". Must be declared WITHOUT a +# default: a default shadows the value the builder injects, which would silently +# fetch the wrong architecture's binaries. Steps below fall back to +# `dpkg --print-architecture` (same amd64/arm64 vocabulary) when it is unset, +# so non-BuildKit builds still resolve the host architecture correctly. +ARG TARGETARCH + # ============================================================================ # Base system dependencies (GitHub Actions Runner) # ============================================================================ @@ -55,19 +62,78 @@ RUN curl --proto '=https' --tlsv1.2 -sSf https://just.systems/install.sh | bash # ============================================================================ # Install Pkl (Apple's configuration language - used by canvas) # ============================================================================ -RUN curl -L -o /usr/local/bin/pkl https://github.com/apple/pkl/releases/download/0.30.1/pkl-linux-amd64 && \ +# Version matches what bender-driver/dti-fsic-driver/vehicle-message-definitions +# actually pin in their "Install PKL CLI" workflow step (verified against those +# repos' ci.yml, not assumed) -- see the useradd block below for why this alone +# does not fix those workflows' install step. +RUN ARCH="${TARGETARCH:-$(dpkg --print-architecture)}" && \ + case "$ARCH" in \ + amd64) PKL_ARCH=amd64 ;; \ + arm64) PKL_ARCH=aarch64 ;; \ + *) echo "Unsupported architecture: $ARCH" >&2; exit 1 ;; \ + esac && \ + curl -fL -o /usr/local/bin/pkl "https://github.com/apple/pkl/releases/download/0.31.1/pkl-linux-${PKL_ARCH}" && \ chmod +x /usr/local/bin/pkl # ============================================================================ # Install uv (fast Python package manager) and maturin (Rust-Python build tool) # ============================================================================ -RUN curl -LsSf https://astral.sh/uv/install.sh | sh && \ - # Add uv to PATH - . $HOME/.local/bin/env && \ - # Install maturin globally via uv - uv tool install maturin +# Installed into shared, world-readable locations rather than under /root, which +# is mode 0700: a tool symlinked out of /root is unusable by the unprivileged +# runner user that actually executes jobs. +ENV UV_TOOL_DIR=/opt/uv/tools + +RUN curl -LsSf https://astral.sh/uv/install.sh | env UV_INSTALL_DIR=/usr/local/bin sh && \ + # Install maturin globally, with its launcher on the shared PATH + UV_TOOL_BIN_DIR=/usr/local/bin uv tool install maturin && \ + chmod -R a+rX /opt/uv + +# ============================================================================ +# Web UI and Tauri desktop dependencies (hbf) +# ============================================================================ +# Deliberately placed AFTER the espup layer. Docker invalidates every layer +# below an edited one, and rebuilding the Xtensa toolchain costs many minutes, +# so anything added later must stay later. +# +# `cargo build -p hbf-gui` links against webkit2gtk-4.1 and fails at +# pkg-config time without the -dev package; librsvg2 and appindicator3 are +# Tauri's SVG and tray-icon dependencies. This mirrors the apt list hbf CI +# installs per job, minus what the firmware layers above already provide +# (libudev-dev, pkg-config, libssl-dev). +RUN apt-get update && \ + apt-get install -y --no-install-recommends \ + libwebkit2gtk-4.1-dev libayatana-appindicator3-dev \ + librsvg2-dev && \ + apt-get clean && rm -rf /var/lib/apt/lists/* + +# Node is needed even though bun is the package manager, because bun does not +# replace it as a script *interpreter*. hbf's `ts_export` test execs +# `ui/node_modules/.bin/prettier` directly from Rust; that file is a .cjs script +# whose shebang is `#!/usr/bin/env node`, so without node the exec fails with +# status 127 and the drift check reports "bindings would drift". `bun run lint` +# and `bun run check` are unaffected because `bun run` interprets the JS itself +# and never consults the shebang -- which is exactly why this gap is invisible +# until something shells out to a .bin entry. +# +# npm comes along for `npx`, which the same test falls back to when the +# project-local binary is absent. GitHub-hosted runners preinstall both, which is +# why this only surfaced on the fleet. +RUN apt-get update && \ + apt-get install -y --no-install-recommends \ + nodejs npm && \ + apt-get clean && rm -rf /var/lib/apt/lists/* && \ + node --version && npx --version -ENV PATH="/root/.local/bin:${PATH}" +# bun builds the SvelteKit bundle that `tauri::generate_context!()` embeds at +# COMPILE time, so it is a build dependency of hbf-gui rather than a test-only +# tool. Pinned to the version hbf CI's `oven-sh/setup-bun` requests so lockfile +# resolution is identical on both. BUN_INSTALL places the binary on the shared +# PATH instead of under /root, which is mode 0700 and therefore invisible to the +# unprivileged runner user -- the same trap the uv block above documents. +ENV BUN_INSTALL=/usr/local +RUN curl -fsSL https://bun.sh/install | bash -s "bun-v1.3.14" && \ + chmod a+rx /usr/local/bin/bun && \ + bun --version # ============================================================================ # Create runner directory and download GitHub Actions Runner @@ -75,13 +141,19 @@ ENV PATH="/root/.local/bin:${PATH}" RUN mkdir -p /actions-runner WORKDIR /actions-runner -RUN LATEST_TAG=$(curl -s https://api.github.com/repos/actions/runner/releases/latest | jq -r .tag_name) && \ +RUN ARCH="${TARGETARCH:-$(dpkg --print-architecture)}" && \ + case "$ARCH" in \ + amd64) RUNNER_ARCH=x64 ;; \ + arm64) RUNNER_ARCH=arm64 ;; \ + *) echo "Unsupported architecture: $ARCH" >&2; exit 1 ;; \ + esac && \ + LATEST_TAG=$(curl -s https://api.github.com/repos/actions/runner/releases/latest | jq -r .tag_name) && \ RUNNER_VERSION=${LATEST_TAG#v} && \ - echo "Downloading Runner Version: ${RUNNER_VERSION}" && \ - curl -L -o actions-runner-linux-x64-${RUNNER_VERSION}.tar.gz \ - "https://github.com/actions/runner/releases/download/v${RUNNER_VERSION}/actions-runner-linux-x64-${RUNNER_VERSION}.tar.gz" && \ - tar xzf actions-runner-linux-x64-${RUNNER_VERSION}.tar.gz && \ - rm actions-runner-linux-x64-${RUNNER_VERSION}.tar.gz + echo "Downloading Runner Version: ${RUNNER_VERSION} (${RUNNER_ARCH})" && \ + curl -fL -o runner.tar.gz \ + "https://github.com/actions/runner/releases/download/v${RUNNER_VERSION}/actions-runner-linux-${RUNNER_ARCH}-${RUNNER_VERSION}.tar.gz" && \ + tar xzf runner.tar.gz && \ + rm runner.tar.gz # ============================================================================ # Setup SSH for private repository access (submodules) @@ -91,9 +163,50 @@ RUN mkdir -p /root/.ssh && \ ssh-keyscan github.com >> /root/.ssh/known_hosts && \ chmod 644 /root/.ssh/known_hosts -# Copy entrypoint script +# ============================================================================ +# sccache -- shared compilation cache +# ============================================================================ +# Placement is deliberate on both sides. It sits AFTER the espup and bun layers, +# so adding it never invalidates the multi-GB Xtensa toolchain, and BEFORE +# `COPY entrypoint.sh`, because that COPY invalidates every layer after it +# whenever the entrypoint changes -- re-downloading sccache on each entrypoint +# tweak would be pure waste. +# +# Why it exists: every replica keeps its own `target/`, and hbf's reaches +# 11-13 GB, so twelve of them took a 926 GB volume down to 294 MB free on +# 2026-08-09, at which point CI began failing with +# `collect2: ld terminated with signal 7 [Bus error]` -- disk exhaustion wearing +# a linker bug's clothing. +# +# sccache does NOT shrink `target/`. It caches rustc invocations in a store +# outside it, so the rlibs and test executables still land there at full size. +# What it buys is that DELETING a target dir becomes cheap, which is what makes +# those dirs disposable rather than something to hoard. Bounding disk therefore +# needs sccache AND a recurring sweep; sccache on its own does not do it. +# +# The musl build is static, so it is indifferent to the glibc version of whatever +# base image this is rebuilt on. +ARG SCCACHE_VERSION=v0.17.0 +RUN ARCH="${TARGETARCH:-$(dpkg --print-architecture)}" && \ + case "$ARCH" in \ + amd64) SCCACHE_ARCH=x86_64 ;; \ + arm64) SCCACHE_ARCH=aarch64 ;; \ + *) echo "ERROR: unsupported architecture for sccache: $ARCH" >&2; exit 1 ;; \ + esac && \ + SCCACHE_PKG="sccache-${SCCACHE_VERSION}-${SCCACHE_ARCH}-unknown-linux-musl" && \ + curl -fsSL -o /tmp/sccache.tar.gz \ + "https://github.com/mozilla/sccache/releases/download/${SCCACHE_VERSION}/${SCCACHE_PKG}.tar.gz" && \ + tar -xzf /tmp/sccache.tar.gz -C /tmp && \ + install -m 0755 "/tmp/${SCCACHE_PKG}/sccache" /usr/local/bin/sccache && \ + rm -rf /tmp/sccache.tar.gz "/tmp/${SCCACHE_PKG}" && \ + sccache --version + +# Copy entrypoint script, the post-job sweep hook, and the pre-job gitconfig +# reset hook COPY entrypoint.sh /entrypoint.sh -RUN chmod +x /entrypoint.sh +COPY job-completed-hook.sh /usr/local/bin/job-completed-hook.sh +COPY job-started-hook.sh /usr/local/bin/job-started-hook.sh +RUN chmod +x /entrypoint.sh /usr/local/bin/job-completed-hook.sh /usr/local/bin/job-started-hook.sh # Create a non-root user and copy tools RUN useradd -m runner && \ @@ -105,9 +218,12 @@ RUN useradd -m runner && \ cp -r /root/.rustup/* /home/runner/.rustup/ 2>/dev/null || true && \ # Copy export-esp.sh to runner home cp /root/export-esp.sh /home/runner/export-esp.sh 2>/dev/null || true && \ - # Copy uv and tools to runner user - mkdir -p /home/runner/.local && \ - cp -r /root/.local/* /home/runner/.local/ 2>/dev/null || true && \ + # uv and its tools (maturin) live in /usr/local/bin and /opt/uv, which are + # already on the shared PATH and readable by this user — nothing to copy. + # Pre-create the sccache directory so its named volume is seeded with runner + # ownership. A volume mounted over a path that does not exist in the image is + # created root-owned, which the unprivileged runner cannot write to. + mkdir -p /home/runner/.cache/sccache && \ # Copy SSH config to runner user mkdir -p /home/runner/.ssh && \ cp /root/.ssh/known_hosts /home/runner/.ssh/ && \ @@ -116,7 +232,30 @@ RUN useradd -m runner && \ # Add source export-esp.sh to runner's bashrc echo 'source $HOME/export-esp.sh 2>/dev/null || true' >> /home/runner/.bashrc && \ # Fix ownership - chown -R runner:runner /home/runner + chown -R runner:runner /home/runner && \ + # Several consuming repos' workflows self-install a version-pinned tool by + # curling a binary straight into /usr/local/bin and chmod +x-ing it -- e.g. + # bender-driver/dti-fsic-driver/vehicle-message-definitions all run: + # curl -L -o /usr/local/bin/pkl https://.../pkl- && chmod +x ... + # On a GitHub-hosted runner this succeeds because the job owns the whole VM. + # Here it hits EACCES: /usr/local/bin is root:root 0755 from the apt/curl + # installs above, and `curl -o` truncates the EXISTING pkl binary in place + # (an open() with O_TRUNC), which needs write on that file's inode, not just + # search/exec on the directory. A PATH-based redirect (e.g. exporting a + # writable $RUNNER_TEMP/bin) cannot fix this: the destination is a literal + # absolute path in those workflows, not something resolved via PATH, and + # editing every consuming repo's workflow is exactly the per-repo workaround + # this fleet's image is meant to avoid. So the directory itself has to + # become writable by the user that actually runs jobs. + # + # chown rather than chmod a+w to match this file's own idiom (chown -R + # runner:runner appears twice above) instead of leaving a world-writable + # system directory. /usr/local/bin holds nothing but the tools this image + # installs (just, pkl, uv/maturin, sccache, bun -- no apt package puts + # anything here), so handing it to runner does not touch anything owned by + # another principal, and the runner user already executes arbitrary job + # code with far broader access than this. + chown -R runner:runner /usr/local/bin # Environment variables for runner user ENV RUSTUP_HOME=/home/runner/.rustup \ diff --git a/README.md b/README.md index ad8a82d..d5ad604 100644 --- a/README.md +++ b/README.md @@ -15,7 +15,9 @@ This runner includes all tools required for the firmware CI pipeline: ### Build Tools - **just** - Command runner used by the firmware project -- **Pkl** (v0.29.1) - Apple's configuration language (used by canvas) +- **Pkl** (v0.31.1) - Apple's configuration language (used by canvas); `/usr/local/bin` + is writable by the `runner` user so consuming workflows can self-install a + different pinned version without hitting `EACCES` - **maturin** - Build Python wheels from Rust code ### Python @@ -29,6 +31,23 @@ This runner includes all tools required for the firmware CI pipeline: - **SSH** - Pre-configured with GitHub's host keys for private submodule access - Standard build essentials (`build-essential`, `pkg-config`, `libssl-dev`) +## Architectures + +The image builds for both `linux/amd64` and `linux/arm64` (e.g. Apple Silicon via +OrbStack/Docker Desktop). Architecture-specific downloads (GitHub Actions runner, +Pkl) are selected from BuildKit's `TARGETARCH`; the Rust, ESP (`espup`) and Python +toolchains resolve their own host architecture. + +Docker Compose and `docker build` produce a native image by default. To build +explicitly for one architecture: + +```bash +docker buildx build --platform linux/arm64 -t github-runner . +``` + +> Note: if `TARGETARCH` is unset (a build without BuildKit), the Dockerfile falls +> back to `dpkg --print-architecture`, i.e. the base image's own architecture. + ## Usage ### Environment Variables @@ -40,6 +59,62 @@ This runner includes all tools required for the firmware CI pipeline: | `RUNNER_TOKEN` | One of `GITHUB_PAT` / `RUNNER_TOKEN` | Static runner registration token from GitHub. Expires ~1 hour after creation, so restarts after that will fail unless refreshed. Ignored if `GITHUB_PAT` is set. | | `RUNNER_NAME` | No | Base name for the runner (default: `runner`) | | `RUNNER_LABELS` | No | Comma-separated labels for the runner | +| `RUNNER_CPUS` | No | CPUs per replica; also caps `CARGO_BUILD_JOBS` (default: `2`) | +| `RUNNER_MEMORY` | No | Memory per replica (default: `6g`) | + +### Parallel Jobs + +A GitHub Actions runner executes **one job at a time** — there is no concurrency +setting inside the runner. Total parallelism is therefore just `RUNNER_COUNT`. + +Eight replicas (`runner-1` .. `runner-8`) are declared explicitly in +`docker-compose.yml`, at 2 CPUs and 6 GB each, sized for a 16-core / 64 GB host. +`CARGO_BUILD_JOBS` is pinned to `RUNNER_CPUS` — without that, cargo sizes its +thread pool from the *host* core count and every replica would spawn ~16 +threads, oversubscribing the machine. + +**Memory, not CPU, is what limits the replica count.** 8 x 6 GB = 48 GB of the +~58 GB the OrbStack VM exposes. Adding replicas without lowering `RUNNER_MEMORY` +will overcommit and get builds OOM-killed. + +A single CI run only reaches 5 concurrent jobs (four checks in parallel, then +three builds behind `needs`). The reason more replicas still help is that +`concurrency` in `firmware_ci.yml` is keyed per *branch*, so several runs +execute at once and jobs queue globally. + +To run fewer runners, name the services; to run bigger ones, raise the limits: + +```bash +docker compose up -d --build runner-1 runner-2 runner-3 +RUNNER_CPUS=4 RUNNER_MEMORY=10g docker compose up -d --build +``` + +> Replicas are separate services rather than `deploy.replicas` because a scaled +> service shares one set of volumes, and sccache cannot safely share a cache +> directory between concurrent server processes (see below). + +### Caching + +Two caches survive container recreation, both **per replica**: + +- **`cargo-registry-N`** — the crate download cache. +- **`sccache-N`** — the compiler cache. `setup-rust-dual` in the firmware repo + points sccache at `$HOME/.cache/sccache`. + +Neither may be shared between replicas. sccache keeps its LRU index in memory +per server process, so containers sharing one directory evict against each +other. The registry was shared in an earlier revision and broke CI: unpacked +sources under `registry/src` disappear mid-compile when another container's +cargo garbage-collects the global cache, producing +`could not execute process ... No such file or directory`. The cost of not +sharing is N copies of the same crate downloads, which is the right trade. + +The runner's `_work` directory is deliberately **not** persisted. The firmware +workflow checks out with `clean: false` to reuse `target/`, but a stale +submodule `target/` surviving `git submodule deinit` is what produced +`could not parse/generate dep info ... No such file or directory` build +failures. sccache is content-hashed and immune to that staleness, so it is the +right layer to persist; `_work` is not. ### Running with Docker Compose @@ -48,6 +123,10 @@ This runner includes all tools required for the firmware CI pipeline: export URL=https://github.com/jkuracing export GITHUB_PAT= -# Start the runner -docker compose up -d +# Start the runners +docker compose up -d --build ``` + +> Always pass `--build`. Plain `docker compose up -d` only builds when the image +> is missing, so it will happily keep running a stale image after the Dockerfile +> or `entrypoint.sh` changes. diff --git a/docker-compose.yml b/docker-compose.yml index 235563e..8d26029 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -1,16 +1,260 @@ +# Replicas are declared explicitly rather than via `deploy.replicas` because +# every replica of a scaled service shares one set of volumes, and NOTHING here +# is safe to share between concurrently building runners: +# +# - sccache keeps its LRU index in memory per server process, so several +# containers on one cache DIRECTORY evict against each other. This is why +# the shared sccache below is reached over S3 instead: a server backend has +# no such per-process index, so sharing it is safe where a directory is not. +# - The cargo registry was shared here initially and caused real CI failures: +# error: could not compile `crc32fast` (lib) +# Caused by: could not execute process `.../bin/rustc --crate-name +# crc32fast .../registry/src/index.crates.io-*/crc32fast-1.5.0/src/lib.rs` +# Caused by: No such file or directory (os error 2) +# Unpacked sources under registry/src vanish mid-compile when another +# container's cargo garbage-collects the global cache, so the spawn fails on +# a working directory that no longer exists. Cargo's package-cache lock does +# not protect a build for its whole duration across separate containers. +# +# Each runner therefore gets its own cargo registry volume, which is only +# expressible as its own service. The cost is N copies of the crate downloads. +# +# sccache is the exception and is now SHARED, via the `sccache-s3` service below. +# Twelve private caches would each have to warm from scratch, so the first build +# on every replica stayed cold -- most of what a cache exists to prevent. The +# per-replica sccache volumes that used to be mounted here are gone: with an S3 +# backend sccache keeps no local cache directory. +# +# `docker compose up -d --build` starts all 12. To run fewer, name them: +# docker compose up -d --build runner-1 runner-2 runner-3 +# +# `docker compose build` MUST be run with no service argument. Each service +# declares its own `build: .`, so compose tags a separate image per service, and +# `docker compose build runner-1` silently leaves the others on the old image -- +# they still start and register, so nothing looks wrong until a job needs a tool +# only the rebuilt image has. +x-runner: &runner + build: . + # `on-failure:5` used to be the policy here and it cost the fleet ten days of + # downtime: a bug in entrypoint.sh's config cleanup made every replica exit 1 + # on restart, the five retries were spent in seconds, and Docker then left all + # containers dead with no surviving process to notice. A runner fleet + # should heal rather than latch off, and the entrypoint mints a fresh + # registration token per start, so a genuinely broken image loops visibly in + # the logs instead of failing silently. + restart: unless-stopped + stop_grace_period: 5m + # Ordering only, deliberately NOT `condition: service_healthy`. Gating twelve + # runners on the cache's health would turn an optimisation into a single point + # of failure for the whole fleet -- a sick MinIO would mean no CI at all rather + # than slow CI. entrypoint.sh probes the bucket itself and degrades to + # wrapper-less builds if it cannot be reached. + depends_on: [sccache-s3] + environment: + URL: ${URL} + GITHUB_PAT: ${GITHUB_PAT} + # Defaulted to empty: entrypoint.sh prefers GITHUB_PAT and only falls back + # to a static token, but an unset variable makes `docker compose` print a + # warning per service per invocation -- noise that buries the real errors + # underneath. + RUNNER_TOKEN: ${RUNNER_TOKEN:-} + RUNNER_NAME: ${RUNNER_NAME:-} + # A runner is offered a job only when its label set is a SUPERSET of the + # job's `runs-on`. Both labels therefore go on every replica: splitting them + # across replicas would leave some containers ineligible for hbf jobs and + # idle whenever hbf work is all that is queued. firmware_ci.yml asks for + # [fw-builder]; hbf asks for [hbf-builder]; every replica can serve either. + RUNNER_LABELS: ${RUNNER_LABELS:-fw-builder,hbf-builder} + # Match cargo's internal parallelism to this replica's CPU allotment. + # cargo defaults to one codegen unit per *host* core, so without this each + # replica would spawn ~16 threads and every replica would oversubscribe the + # machine. The cpus limit below only throttles the result; capping the thread + # count is what actually avoids the thrashing. + CARGO_BUILD_JOBS: ${RUNNER_CPUS:-2} + # Disk, not memory or CPU, is what this fleet runs out of. Measured + # 2026-08-09 with the host volume down to 294 MB free: the 12 container + # writable layers held 170 GB, and 85% of that was ONE directory per + # replica -- hbf's target/ at 11-13 GB (firmware's is 0.9 GB). The + # per-replica cargo registry and sccache volumes that this file goes to + # such lengths to keep separate are only ~29 GB combined, so they are not + # the problem and sharing them would not fix it. target/ cannot be shared + # at all: cargo takes an exclusive lock per target directory, so one + # shared dir would serialise all 12 replicas and destroy the parallelism + # that is the entire point of the fleet. The lever is to make each target/ + # smaller, not fewer of them. + # + # The cargo knobs that act on this -- CARGO_INCREMENTAL and + # CARGO_PROFILE_DEV_DEBUG -- are deliberately NOT set here, and + # entrypoint.sh owns their defaults instead. A value set in this file only + # reaches a job once `up -d` RECREATES the container, which destroys _work + # and with it the warm target/ dir on the writable layer -- the very thing + # being economised. Exported from the entrypoint, a plain `docker restart` + # applies them and keeps the caches. Add either name here to override, and + # see entrypoint.sh for why /actions-runner/.env cannot serve this purpose. + # + # Setting them on the fleet rather than in either repo's workflow keeps + # hosted runners and developer laptops unaffected. Neither repo sets + # CARGO_INCREMENTAL, so the fleet's value is what a job sees. + # + # sccache was DEAD here until 2026-08-09 -- absent from the image, no + # RUSTC_WRAPPER ever set, and 1.1 GB per replica of cache last written + # 2026-07-28. That is why wiping a target/ dir used to mean a genuinely cold + # rebuild. It is now installed in the image and wired to the shared bucket by + # entrypoint.sh, which enables the wrapper only if that bucket answers, so a + # cache outage degrades build speed instead of failing CI. + # + # CARGO_INCREMENTAL=0 is a PREREQUISITE for it, not a rival: sccache cannot + # cache incrementally compiled units and silently bypasses them. + # + # What sccache does NOT do is shrink target/ -- the rlibs and test + # executables still land there at full size, so a warm fleet returns to + # roughly 140 GB. It makes deleting those dirs cheap, which is what makes a + # recurring sweep affordable. Disk stays bounded by sccache AND the sweep + # together; neither suffices alone. + deploy: + resources: + limits: + # 12 replicas at 2 CPU / 4 GB on a 16-core / 64 GB host. + # + # This block previously read "Memory is the binding constraint, not CPU" + # at 8 x 6 GB. Measured under a full load of firmware and hbf jobs, that + # was wrong on both counts: peak usage across all replicas was 756 MiB + # against the 6 GiB limit -- an 8x overshoot -- and only 4 of 8 + # containers were computing at all (~195% CPU each), the rest sitting + # near idle on network and setup. Roughly half the host's cores were + # unused while jobs queued. + # + # The real constraint was SLOTS. An hbf run measured 16.8 minutes of job + # time inside an 11.8 minute span -- an average concurrency of 1.4 -- + # because firmware held 7 of the 8 slots. So: more, smaller replicas. + # + # Total memory is unchanged at 48 GB of the ~58 GB the VM exposes. CPU is + # deliberately oversubscribed 1.5:1 (12 x 2 = 24 on 16 cores), which the + # observed idle time justifies. + # + # DISK is now the limiting factor, not memory: each replica keeps its own + # target/ for both repositories. At 12 replicas this reached 170 GB of + # container layers and took the host volume down to 294 MB free, which + # does not fail as "out of disk" -- it surfaces as + # `collect2: ld terminated with signal 7 [Bus error]`, the linker dying + # mid-write. hbf's own ci.yml documents the same symptom on hosted + # runners. Treat any inexplicable linker or codegen failure across + # several replicas as a disk check first. + # + # See CARGO_INCREMENTAL / CARGO_PROFILE_DEV_DEBUG above for the fix, and + # note it is PROSPECTIVE: existing target/ dirs keep their incremental + # state and fat debuginfo until rebuilt, so landing those variables + # reclaims nothing on its own. Run `docker system df` before going wider. + cpus: ${RUNNER_CPUS:-2} + memory: ${RUNNER_MEMORY:-4g} + services: - github-runner: - build: . - restart: on-failure:5 - stop_grace_period: 5m + # The one cache all 12 replicas share. See the header comment for why this is + # S3 rather than a shared directory, and why the cargo registry deliberately is + # NOT shared the same way. + # + # MinIO is used in preference to Redis because a compilation cache of this size + # belongs on disk: Redis would hold the whole thing in RAM, and 48 of the VM's + # ~58 GB is already committed to the replicas, so it would compete with the + # builds it is meant to accelerate. + # + # Not published to the host. Only the runners need it, and the fleet's compose + # network is enough -- exposing an unauthenticated-by-default object store on a + # laptop's interfaces would be a poor trade for a build cache. + sccache-s3: + image: minio/minio:latest + restart: unless-stopped + command: server /data environment: - URL: ${URL} - GITHUB_PAT: ${GITHUB_PAT} - RUNNER_TOKEN: ${RUNNER_TOKEN} - RUNNER_NAME: ${RUNNER_NAME} - RUNNER_LABELS: ${RUNNER_LABELS} - deploy: - replicas: ${RUNNER_COUNT:-1} + MINIO_ROOT_USER: ${SCCACHE_ACCESS_KEY:-sccache} + MINIO_ROOT_PASSWORD: ${SCCACHE_SECRET_KEY:-sccache-secret} + volumes: [sccache-s3-data:/data] + healthcheck: + test: ["CMD", "mc", "ready", "local"] + interval: 10s + timeout: 5s + retries: 5 + start_period: 20s + + # Creates the bucket and, more importantly, BOUNDS it. sccache has a built-in + # LRU for a local cache directory (SCCACHE_CACHE_SIZE) but none whatsoever for + # an S3 backend -- it never deletes anything it uploads. Left alone this bucket + # would grow without limit and become the disk problem it was added to solve. + # + # Expiry by age rather than a hard quota is the primary bound: a hard quota + # makes writes start failing once reached, which sccache reports as errors on + # every miss, whereas expiry keeps the working set and simply forgets what has + # not been useful lately. 14 days comfortably covers a fortnight of branches + # while capping the bucket at roughly one fortnight of unique compilations. + # The quota is a generous backstop against a pathological week. + # + # Runs to completion and exits; `restart: "no"` keeps it from looping. + sccache-s3-init: + image: minio/mc:latest + depends_on: + sccache-s3: + condition: service_healthy + restart: "no" + entrypoint: > + /bin/sh -c " + mc alias set fleet http://sccache-s3:9000 + '${SCCACHE_ACCESS_KEY:-sccache}' '${SCCACHE_SECRET_KEY:-sccache-secret}' && + mc mb --ignore-existing fleet/sccache && + mc ilm rule add --expire-days 14 fleet/sccache 2>/dev/null || true && + mc quota set fleet/sccache --size ${SCCACHE_MAX_SIZE:-20GiB} 2>/dev/null || true && + echo 'sccache bucket ready:' && mc du fleet/sccache + " + + runner-1: + <<: *runner + volumes: [cargo-registry-1:/home/runner/.cargo/registry] + runner-2: + <<: *runner + volumes: [cargo-registry-2:/home/runner/.cargo/registry] + runner-3: + <<: *runner + volumes: [cargo-registry-3:/home/runner/.cargo/registry] + runner-4: + <<: *runner + volumes: [cargo-registry-4:/home/runner/.cargo/registry] + runner-5: + <<: *runner + volumes: [cargo-registry-5:/home/runner/.cargo/registry] + runner-6: + <<: *runner + volumes: [cargo-registry-6:/home/runner/.cargo/registry] + runner-7: + <<: *runner + volumes: [cargo-registry-7:/home/runner/.cargo/registry] + runner-8: + <<: *runner + volumes: [cargo-registry-8:/home/runner/.cargo/registry] + runner-9: + <<: *runner + volumes: [cargo-registry-9:/home/runner/.cargo/registry] + runner-10: + <<: *runner + volumes: [cargo-registry-10:/home/runner/.cargo/registry] + runner-11: + <<: *runner + volumes: [cargo-registry-11:/home/runner/.cargo/registry] + runner-12: + <<: *runner + volumes: [cargo-registry-12:/home/runner/.cargo/registry] volumes: - runner-data: + # The single shared compilation cache, bounded by the ilm rule and quota that + # sccache-s3-init applies. Replaces the twelve per-replica sccache volumes. + sccache-s3-data: + cargo-registry-1: + cargo-registry-2: + cargo-registry-3: + cargo-registry-4: + cargo-registry-5: + cargo-registry-6: + cargo-registry-7: + cargo-registry-8: + cargo-registry-9: + cargo-registry-10: + cargo-registry-11: + cargo-registry-12: diff --git a/entrypoint.sh b/entrypoint.sh index b74e239..9aef3cb 100644 --- a/entrypoint.sh +++ b/entrypoint.sh @@ -18,6 +18,16 @@ FULL_RUNNER_NAME="${RUNNER_NAME}-${HOSTNAME}" echo "Fixing permissions for /actions-runner..." chown -R runner:runner /actions-runner +# These are named volumes. Docker seeds them from the image with the right +# ownership, but a volume created before the directory existed in the image (or +# by another image) comes back root-owned and silently breaks every build. +for vol_dir in /home/runner/.cargo/registry /home/runner/.cache/sccache; do + if [[ -d "$vol_dir" ]] && [[ "$(stat -c %U "$vol_dir")" != "runner" ]]; then + echo "Fixing permissions for ${vol_dir}..." + chown -R runner:runner "$vol_dir" + fi +done + # Fetches a short-lived token ($1: "registration-token" or "remove-token") from the # GitHub API, using GITHUB_PAT. Prints the token on stdout, returns non-zero on failure. fetch_runner_token() { @@ -71,8 +81,24 @@ if [[ -f /home/runner/export-esp.sh ]]; then fi echo "Removing any existing runner configuration..." -# Clean up previous runs (crucial for ephemeral runners) -rm -f .runner .credentials .credentials_rsaparams +# Clean up previous runs (crucial for ephemeral runners). +# +# `.runner_migrated` MUST be in this list. The runner self-updates in place, and +# a post-update runner drops that marker beside its config. `config.sh` treats +# the marker ALONE as proof the runner is already configured -- verified by +# creating only `.runner_migrated` and passing a deliberately bogus token: it +# fails with "Cannot configure the runner because it is already configured" +# without even attempting to authenticate. +# +# Because the old list stopped at `.credentials_rsaparams`, every replica that +# had auto-updated crash-looped on its next restart until `restart: +# on-failure:5` exhausted its retries, which silently took the entire fleet +# offline about ten days after it was last rebuilt. Deleting the marker is +# correct rather than merely expedient: this entrypoint always reconfigures from +# a freshly minted registration token, so there is no migrated state worth +# preserving across a restart. +rm -f .runner .credentials .credentials_rsaparams \ + .runner_migrated .credentials_migrated echo "Configuring GitHub Actions Runner as ${FULL_RUNNER_NAME}..." echo "URL: $URL" @@ -125,6 +151,197 @@ handle_shutdown() { } trap handle_shutdown SIGTERM SIGINT +# Cargo knobs applied to every job this replica runs. +# +# Exporting here is what makes these changeable by a plain `docker restart`. +# `docker compose` bakes a container's environment at CREATION time, so setting +# them only in docker-compose.yml means they reach a job solely after +# `up -d` -- which RECREATES the container, destroying `_work` along with the +# 11-13 GB warm `target/` dir that lives on the writable layer. Restart keeps +# it. The runner inherits this process's environment and hands it to each job +# step, so an export reaches the compiler. +# +# Do NOT move these into /actions-runner/.env. That file is read only by the +# systemd unit `svc.sh` generates; this entrypoint execs ./run.sh directly and +# run.sh contains no reference to it -- verified, not assumed. The stock .env is +# empty here and `env.sh` merely writes it for that service path. +# +# The defaults live here rather than in docker-compose.yml so that one file owns +# them; compose or `docker run -e` can still override either value. + +# Worth 2.8 GB per replica (14 GB in a long-lived developer checkout, which is +# what this keeps the fleet from becoming). +# +# This is a trade, NOT free: an earlier version of this comment called +# incremental state "pure waste in CI, since every job is a different commit", +# which is wrong here. hbf's ci.yml deliberately uses `clean: false` to keep +# `target/` warm across jobs, so successive jobs on one replica genuinely can +# hit an incremental cache. +# +# The exposure is bounded and judged worth the disk: +# - It only ever covers hbf's own dozen workspace crates. Registry +# dependencies -- which are the whole 8.4 GB bulk of debug/deps -- are +# compiled non-incrementally regardless of this setting. +# - A hit needs the SAME replica to rebuild a NEARLY IDENTICAL commit. Jobs +# go to whichever of the 12 replicas is free, with no branch affinity, so +# that is luck rather than design. +# - Where nothing changed at all, cargo's ordinary fingerprinting skips the +# crate outright and incremental adds nothing. +# - It is not free even when it hits: incremental raises the codegen-unit +# count, which costs some link time back. +# +# It is also a prerequisite for sccache rather than a rival to it -- sccache +# cannot cache incrementally compiled units and silently bypasses them. sccache +# would hit across every crate, replica and commit rather than only the +# same-replica-similar-commit case, so trading incremental for it is a clear win +# whenever someone wires it up. +export CARGO_INCREMENTAL="${CARGO_INCREMENTAL:-0}" + +# A floor, NOT a saving -- do not expect this to reclaim anything. hbf's ci.yml +# already sets it workflow-wide as of 3a6e615, and the target dirs measured on +# this fleet are already the reduced size: objdump on the largest test +# executable shows .debug_loc at 0 bytes with .debug_line the dominant section, +# the line-tables-only signature. Set here so the property holds for every job +# whatever an individual workflow remembers to configure; firmware_ci.yml sets +# only CARGO_PROFILE_RELEASE_DEBUG and leaves its dev profile uncovered. +export CARGO_PROFILE_DEV_DEBUG="${CARGO_PROFILE_DEV_DEBUG:-line-tables-only}" + +# sccache, backed by the fleet's shared S3 (MinIO) bucket. +# +# Shared rather than per-replica on purpose. Twelve private caches would each +# have to warm from scratch, so the first build on every replica stays cold -- +# which is most of what a cache is supposed to prevent. Pointed at one bucket, +# whichever replica compiles a crate first serves the other eleven. Note the +# per-replica *local* sccache volumes this fleet used to mount are unnecessary +# in this mode and have been removed from docker-compose.yml: with an S3 backend +# sccache does not use a local cache directory. +# +# A shared local DIRECTORY would not be safe here -- each sccache server process +# keeps its own in-memory LRU index, so several containers writing one directory +# corrupt each other's accounting. A server backend has no such problem, which +# is the distinction the header comment in docker-compose.yml draws for the cargo +# registry as well. +SCCACHE_BUCKET="${SCCACHE_BUCKET:-sccache}" +SCCACHE_ENDPOINT="${SCCACHE_ENDPOINT:-http://sccache-s3:9000}" +SCCACHE_REGION="${SCCACHE_REGION:-us-east-1}" +AWS_ACCESS_KEY_ID="${AWS_ACCESS_KEY_ID:-sccache}" +AWS_SECRET_ACCESS_KEY="${AWS_SECRET_ACCESS_KEY:-sccache-secret}" + +# Enable the wrapper ONLY if the bucket is actually reachable. A cache is an +# optimisation and must never be able to break CI: if MinIO is down, mis-DNSed or +# still starting, the correct outcome is slower builds, not failed ones. sccache +# degrades gracefully once running, but a server that cannot start at all would +# take every `cargo` invocation down with it, so this is checked up front rather +# than hoped for. Retried because the runner and MinIO come up concurrently. +# The window is generous (~60s) because compose only orders startup here and does +# not wait for health -- see docker-compose.yml for why gating the fleet on the +# cache would be worse. On a cold `up -d` MinIO may still be initialising its +# volume while the runners boot, and a replica that gives up early would run +# every job uncached until something restarted it. +sccache_reachable=0 +for attempt in $(seq 1 20); do + if curl -fsS --max-time 3 "${SCCACHE_ENDPOINT}/minio/health/live" >/dev/null 2>&1; then + sccache_reachable=1 + break + fi + echo "sccache: ${SCCACHE_ENDPOINT} not ready (attempt ${attempt}/20), retrying..." + sleep 3 +done + +if [[ "$sccache_reachable" == "1" ]]; then + export SCCACHE_BUCKET SCCACHE_ENDPOINT SCCACHE_REGION + export AWS_ACCESS_KEY_ID AWS_SECRET_ACCESS_KEY + export SCCACHE_S3_USE_SSL="${SCCACHE_S3_USE_SSL:-false}" + # Surfaces storage errors in the job log instead of silently degrading to a 0% + # hit rate, which is exactly how the previous sccache attempt here died + # unnoticed -- it left 1.1 GB per replica of cache last written 2026-07-28 and + # no wrapper ever configured. + export SCCACHE_ERROR_LOG=/tmp/sccache.log + export SCCACHE_LOG="${SCCACHE_LOG:-warn}" + export RUSTC_WRAPPER=sccache + + # Start the server HERE, explicitly, and never let it idle out. Both halves are + # load-bearing, and this was found the hard way. + # + # The sccache server takes its cache configuration from whichever process first + # starts it. Started explicitly with the environment above it comes up on s3 + # ("Cache location s3, name: sccache"); left to be spawned implicitly by + # cargo's first `sccache rustc ...` wrapper call it came up on LOCAL DISK + # instead, reporting `Cache location Local disk` with every compile request + # invisible to the shared bucket. That failure is silent -- builds succeed at + # full speed-looking cost, the bucket stays empty, and the only symptom is a + # cache that never hits. It is exactly the shape of the previous dead sccache + # attempt here, so it gets a real fix rather than a hope. + # + # SCCACHE_IDLE_TIMEOUT=0 keeps the server alive for the container's lifetime. + # The default is 600s, after which the server exits and the NEXT wrapper call + # respawns it -- landing back on local disk and silently unsharing the cache + # between jobs. A runner is idle far longer than ten minutes between jobs, so + # the default would have made this bug the normal case. + export SCCACHE_IDLE_TIMEOUT=0 + gosu runner env \ + SCCACHE_BUCKET="$SCCACHE_BUCKET" SCCACHE_ENDPOINT="$SCCACHE_ENDPOINT" \ + SCCACHE_REGION="$SCCACHE_REGION" SCCACHE_S3_USE_SSL="$SCCACHE_S3_USE_SSL" \ + AWS_ACCESS_KEY_ID="$AWS_ACCESS_KEY_ID" AWS_SECRET_ACCESS_KEY="$AWS_SECRET_ACCESS_KEY" \ + SCCACHE_IDLE_TIMEOUT=0 SCCACHE_ERROR_LOG="$SCCACHE_ERROR_LOG" \ + SCCACHE_LOG="$SCCACHE_LOG" \ + sccache --start-server 2>&1 | tail -2 || true + + # Assert the server really is on s3, via a throwaway compile first. + # + # `sccache --show-stats` on its own is NOT a trustworthy probe: with no server + # running it reports `Cache location Local disk` from client-side defaults + # without starting one (no SCCACHE_ERROR_LOG is even created), which reads as a + # broken S3 config when nothing is wrong. Diagnosing that cost real time here. + # Routing one trivial compilation through the wrapper guarantees a server + # exists, so the backend line that follows describes reality. + # + # This matters because a cache silently on local disk is worse than no cache: + # it consumes the very volume this exists to relieve and returns a 0% hit rate, + # which is precisely how the previous sccache attempt on this fleet died + # unnoticed. + sccache_canary="$(mktemp -d)" + echo 'fn main() {}' > "${sccache_canary}/canary.rs" + chown -R runner:runner "$sccache_canary" + gosu runner env RUSTC_WRAPPER=sccache SCCACHE_IDLE_TIMEOUT=0 \ + SCCACHE_BUCKET="$SCCACHE_BUCKET" SCCACHE_ENDPOINT="$SCCACHE_ENDPOINT" \ + SCCACHE_REGION="$SCCACHE_REGION" SCCACHE_S3_USE_SSL="$SCCACHE_S3_USE_SSL" \ + AWS_ACCESS_KEY_ID="$AWS_ACCESS_KEY_ID" AWS_SECRET_ACCESS_KEY="$AWS_SECRET_ACCESS_KEY" \ + sccache rustc --crate-name canary --crate-type lib --emit=metadata \ + -o "${sccache_canary}/canary.rmeta" "${sccache_canary}/canary.rs" >/dev/null 2>&1 || true + rm -rf "$sccache_canary" + + sccache_backend="$(gosu runner sccache --show-stats 2>/dev/null | grep -i 'Cache location' || true)" + case "$sccache_backend" in + *s3*) echo "sccache: ENABLED -> ${SCCACHE_ENDPOINT}/${SCCACHE_BUCKET}" ;; + *) echo "sccache: WARNING -- not on s3 after canary compile, got: ${sccache_backend:-}" ;; + esac +else + echo "sccache: DISABLED (${SCCACHE_ENDPOINT} unreachable) -- builds will be slower but will still succeed" +fi + +# Bound `target/` between jobs. sccache makes rebuilding cheap but does NOT make +# target/ small -- the rlibs and test executables still land there at full size, +# so without this the twelve replicas drift back to ~140 GB and refill the disk. +# The runner runs this hook between jobs, so unlike a host cron racing twelve +# replicas it can never delete a target dir out from under a live compile. +export ACTIONS_RUNNER_HOOK_JOB_COMPLETED=/usr/local/bin/job-completed-hook.sh +export SWEEP_MAX_GB="${SWEEP_MAX_GB:-4}" + +# Reset $HOME/.gitconfig before every job. Canvas-consuming repos' shared setup +# snippet writes a git `insteadOf` rewrite with `git config --global set` (then +# `--add`), which is safe on an ephemeral GitHub-hosted runner but accumulates +# in this container's persistent $HOME/.gitconfig job after job until a later +# `set` call hits an already multi-valued key and fails outright. See +# job-started-hook.sh for why this is a job-STARTED hook rather than only +# living in job-completed-hook.sh: it must run regardless of whether the +# previous job finished, was cancelled, or was killed. +export ACTIONS_RUNNER_HOOK_JOB_STARTED=/usr/local/bin/job-started-hook.sh + +echo "Cargo: CARGO_INCREMENTAL=${CARGO_INCREMENTAL} CARGO_PROFILE_DEV_DEBUG=${CARGO_PROFILE_DEV_DEBUG} RUSTC_WRAPPER=${RUSTC_WRAPPER:-}" +echo "Sweep: target/ budget ${SWEEP_MAX_GB} GB per replica, enforced after each job" +echo "Gitconfig: reset to a clean baseline before each job (ACTIONS_RUNNER_HOOK_JOB_STARTED)" + echo "Starting runner..." gosu runner ./run.sh & RUNNER_PID=$! diff --git a/job-completed-hook.sh b/job-completed-hook.sh new file mode 100644 index 0000000..ada2cb2 --- /dev/null +++ b/job-completed-hook.sh @@ -0,0 +1,76 @@ +#!/bin/bash +# Runs after EVERY job on this replica, via ACTIONS_RUNNER_HOOK_JOB_COMPLETED. +# +# Why this exists +# --------------- +# sccache made rebuilding a `target/` dir cheap, but it does not make `target/` +# small: rlibs, shared objects and test executables still land there at full +# size, so hbf's reaches 11-13 GB per replica. Twelve of those is ~140 GB, which +# is how a 926 GB volume ended up at 294 MB free on 2026-08-09 -- surfacing not +# as "out of disk" but as `collect2: ld terminated with signal 7 [Bus error]`. +# +# So the disk bound is this hook, and sccache is what makes it affordable: a +# swept replica refills from the shared bucket instead of recompiling. A cold +# rebuild measured 11 misses; after `cargo clean` the same build took 13 hits and +# added no misses, i.e. it came entirely from the cache. +# +# Why a job hook rather than a cron +# --------------------------------- +# The runner invokes this between jobs, so it can never delete a `target/` out +# from under a running compile -- which a host-side cron racing 12 replicas could +# easily do. It also needs no state on the host and no scheduler to keep alive. +# +# Budget +# ------ +# Total is meant to stay under 100 GB: +# images ~10 + cargo registries ~16 + sccache bucket <=20 = ~46 GB fixed +# 12 replicas x SWEEP_MAX_GB = the rest +# At the default of 4 GB that lands near 94 GB. Raising it trades disk for fewer +# sweeps (and so faster jobs); lowering it does the reverse. Note this bounds +# STEADY STATE, not the peak: a build in flight can exceed the threshold, and is +# only swept once it finishes. +set -uo pipefail + +SWEEP_MAX_GB="${SWEEP_MAX_GB:-4}" +# The runner root's _work, NOT $RUNNER_WORKSPACE. +# +# This originally read `${RUNNER_WORKSPACE:-/actions-runner/_work}`, which was +# wrong: RUNNER_WORKSPACE is per-REPOSITORY (`_work/`), so the budget was +# enforced once per repo rather than once per replica. With firmware and hbf both +# checked out, each replica could hold 2 x SWEEP_MAX_GB. Measured the morning +# after rollout: five replicas sat at 6.8-7.2 GB against a nominal 4 GB budget, +# and the fleet total reached 52 GB against a 48 GB ceiling. SWEEP_WORK_DIR +# stays overridable for testing. +WORK_DIR="${SWEEP_WORK_DIR:-/actions-runner/_work}" +# _work holds more than checkouts (_tool, _temp, _actions), so measure the whole +# thing -- that is what actually occupies the writable layer. +[[ -d "$WORK_DIR" ]] || exit 0 + +used_mb=$(du -sm "$WORK_DIR" 2>/dev/null | cut -f1) +[[ -n "$used_mb" ]] || exit 0 +limit_mb=$((SWEEP_MAX_GB * 1024)) + +if (( used_mb <= limit_mb )); then + echo "sweep: _work at ${used_mb} MB, under the ${limit_mb} MB budget -- keeping it warm" + exit 0 +fi + +echo "sweep: _work at ${used_mb} MB exceeds ${limit_mb} MB -- removing target dirs" + +# Largest first, stopping as soon as the budget is met, so a replica keeps as +# much warmth as the budget allows instead of being emptied wholesale. Only +# genuine cargo target dirs are touched: the CACHEDIR.TAG / debug / release test +# avoids deleting a source directory that merely happens to be called "target". +while IFS= read -r dir; do + (( used_mb <= limit_mb )) && break + [[ -f "${dir}/CACHEDIR.TAG" || -d "${dir}/debug" || -d "${dir}/release" ]] || continue + freed=$(du -sm "$dir" 2>/dev/null | cut -f1) + rm -rf "$dir" && used_mb=$((used_mb - ${freed:-0})) + echo "sweep: removed ${dir} (${freed:-?} MB), now ~${used_mb} MB" +done < <(find "$WORK_DIR" -type d -name target -prune 2>/dev/null \ + | while IFS= read -r d; do echo "$(du -sm "$d" 2>/dev/null | cut -f1) $d"; done \ + | sort -rn | cut -d' ' -f2-) + +# Never fail the job. This hook runs after the work that matters is already +# done and reported; a sweep problem must not turn a green job red. +exit 0 diff --git a/job-started-hook.sh b/job-started-hook.sh new file mode 100644 index 0000000..41bacef --- /dev/null +++ b/job-started-hook.sh @@ -0,0 +1,48 @@ +#!/bin/bash +# Runs before EVERY job on this replica, via ACTIONS_RUNNER_HOOK_JOB_STARTED. +# +# Why this exists +# ---------------- +# A shared setup snippet used across canvas-consuming repos (bender-driver, +# dti-fsic-driver, vehicle-message-definitions, and others being migrated onto +# this fleet) configures a git `insteadOf` rewrite with +# `git config --global set` (then `--add` for a second value under the same +# key). On an ephemeral GitHub-hosted runner this is harmless: the VM, and +# $HOME/.gitconfig with it, is destroyed the moment the job ends. +# +# On this fleet the container -- and therefore the runner user's +# $HOME/.gitconfig -- outlives any one job, so values written by `--add` +# accumulate under the same key across every job a replica has ever run. A +# later job's plain `set` call then collides with an already multi-valued key +# and the whole step fails with: +# error: cannot overwrite multiple values with a single value +# This is invisible in any single job and only shows up after a replica has +# served enough canvas-consuming jobs to pile up a second value -- which is +# exactly what surfaced once bender-driver/dti-fsic-driver/ +# vehicle-message-definitions started sharing this fleet with firmware/hbf. +# +# Why a job-STARTED hook, and not (only) job-completed-hook.sh +# -------------------------------------------------------------- +# job-completed-hook.sh (see its own header) only runs after a job finishes +# normally. A cancelled, timed-out, or forcibly-killed job skips it entirely, +# and that job's accumulated $HOME/.gitconfig survives into the next one -- +# exactly the collision this hook exists to prevent. A job-STARTED hook runs +# before every job regardless of how the PREVIOUS job ended, so it is the only +# placement that actually closes the gap rather than narrowing it. +# +# What "clean baseline" means here +# --------------------------------- +# Nothing in this image's build or entrypoint.sh ever writes to +# $HOME/.gitconfig for the runner user -- verified by grep, not assumed -- so +# the baseline a freshly created container starts with is simply the file's +# absence. This hook reproduces exactly that, every time, rather than trying +# to selectively undo just the insteadOf rewrite (which would have to know +# every key any consuming repo's setup snippet might someday add). +set -uo pipefail + +rm -f "${HOME:-/home/runner}/.gitconfig" + +# Never fail the job. Like job-completed-hook.sh, this runs adjacent to work +# that must not be put at risk by a cleanup step -- a non-zero exit here would +# fail the job it is meant to protect. +exit 0