From 75ae855006fc367a037d16276975a48fa205557f Mon Sep 17 00:00:00 2001 From: Riccardo Persello Date: Mon, 27 Jul 2026 10:23:28 +0200 Subject: [PATCH 01/11] Support arm64 builds alongside amd64 Select architecture-specific downloads from BuildKit's TARGETARCH so the image builds natively on arm64 (e.g. Apple Silicon via OrbStack) as well as amd64: - GitHub Actions runner: linux-x64 / linux-arm64 - Pkl: pkl-linux-amd64 / pkl-linux-aarch64 TARGETARCH is declared without a default, since a default shadows the value the builder injects and would silently fetch amd64 binaries into an arm64 image. Steps fall back to `dpkg --print-architecture` when it is unset so non-BuildKit builds still resolve the host architecture. Downloads now use `curl -f` so a 404 fails the build instead of writing an HTML error page in place of the binary. The Rust, espup and uv toolchains already resolve their own host architecture, and entrypoint.sh has no architecture assumptions. Verified by building --platform linux/arm64 and running the image: pkl 0.30.1 (native), runner 2.336.0, rustc/cargo-nextest on aarch64-unknown-linux-gnu, and the esp toolchain with xtensa-esp-elf-gcc 15.2.0. Co-Authored-By: Claude Claude-Session: https://claude.ai/code/session_01CM6943zQZosnY2QugpiQxf --- Dockerfile | 33 ++++++++++++++++++++++++++------- README.md | 17 +++++++++++++++++ 2 files changed, 43 insertions(+), 7 deletions(-) diff --git a/Dockerfile b/Dockerfile index 4c9a399..2a05743 100644 --- a/Dockerfile +++ b/Dockerfile @@ -3,6 +3,13 @@ FROM ubuntu:24.04 # Prevent interactive prompts during package installation ENV DEBIAN_FRONTEND=noninteractive +# Populated by BuildKit with "amd64" or "arm64". Must be declared WITHOUT a +# default: a default shadows the value the builder injects, which would silently +# fetch the wrong architecture's binaries. Steps below fall back to +# `dpkg --print-architecture` (same amd64/arm64 vocabulary) when it is unset, +# so non-BuildKit builds still resolve the host architecture correctly. +ARG TARGETARCH + # ============================================================================ # Base system dependencies (GitHub Actions Runner) # ============================================================================ @@ -55,7 +62,13 @@ RUN curl --proto '=https' --tlsv1.2 -sSf https://just.systems/install.sh | bash # ============================================================================ # Install Pkl (Apple's configuration language - used by canvas) # ============================================================================ -RUN curl -L -o /usr/local/bin/pkl https://github.com/apple/pkl/releases/download/0.30.1/pkl-linux-amd64 && \ +RUN ARCH="${TARGETARCH:-$(dpkg --print-architecture)}" && \ + case "$ARCH" in \ + amd64) PKL_ARCH=amd64 ;; \ + arm64) PKL_ARCH=aarch64 ;; \ + *) echo "Unsupported architecture: $ARCH" >&2; exit 1 ;; \ + esac && \ + curl -fL -o /usr/local/bin/pkl "https://github.com/apple/pkl/releases/download/0.30.1/pkl-linux-${PKL_ARCH}" && \ chmod +x /usr/local/bin/pkl # ============================================================================ @@ -75,13 +88,19 @@ ENV PATH="/root/.local/bin:${PATH}" RUN mkdir -p /actions-runner WORKDIR /actions-runner -RUN LATEST_TAG=$(curl -s https://api.github.com/repos/actions/runner/releases/latest | jq -r .tag_name) && \ +RUN ARCH="${TARGETARCH:-$(dpkg --print-architecture)}" && \ + case "$ARCH" in \ + amd64) RUNNER_ARCH=x64 ;; \ + arm64) RUNNER_ARCH=arm64 ;; \ + *) echo "Unsupported architecture: $ARCH" >&2; exit 1 ;; \ + esac && \ + LATEST_TAG=$(curl -s https://api.github.com/repos/actions/runner/releases/latest | jq -r .tag_name) && \ RUNNER_VERSION=${LATEST_TAG#v} && \ - echo "Downloading Runner Version: ${RUNNER_VERSION}" && \ - curl -L -o actions-runner-linux-x64-${RUNNER_VERSION}.tar.gz \ - "https://github.com/actions/runner/releases/download/v${RUNNER_VERSION}/actions-runner-linux-x64-${RUNNER_VERSION}.tar.gz" && \ - tar xzf actions-runner-linux-x64-${RUNNER_VERSION}.tar.gz && \ - rm actions-runner-linux-x64-${RUNNER_VERSION}.tar.gz + echo "Downloading Runner Version: ${RUNNER_VERSION} (${RUNNER_ARCH})" && \ + curl -fL -o runner.tar.gz \ + "https://github.com/actions/runner/releases/download/v${RUNNER_VERSION}/actions-runner-linux-${RUNNER_ARCH}-${RUNNER_VERSION}.tar.gz" && \ + tar xzf runner.tar.gz && \ + rm runner.tar.gz # ============================================================================ # Setup SSH for private repository access (submodules) diff --git a/README.md b/README.md index ad8a82d..46bc2c0 100644 --- a/README.md +++ b/README.md @@ -29,6 +29,23 @@ This runner includes all tools required for the firmware CI pipeline: - **SSH** - Pre-configured with GitHub's host keys for private submodule access - Standard build essentials (`build-essential`, `pkg-config`, `libssl-dev`) +## Architectures + +The image builds for both `linux/amd64` and `linux/arm64` (e.g. Apple Silicon via +OrbStack/Docker Desktop). Architecture-specific downloads (GitHub Actions runner, +Pkl) are selected from BuildKit's `TARGETARCH`; the Rust, ESP (`espup`) and Python +toolchains resolve their own host architecture. + +Docker Compose and `docker build` produce a native image by default. To build +explicitly for one architecture: + +```bash +docker buildx build --platform linux/arm64 -t github-runner . +``` + +> Note: if `TARGETARCH` is unset (a build without BuildKit), the Dockerfile falls +> back to `dpkg --print-architecture`, i.e. the base image's own architecture. + ## Usage ### Environment Variables From adb9482da66bb5f398450a171e68e50bb682e16d Mon Sep 17 00:00:00 2001 From: Riccardo Persello Date: Mon, 27 Jul 2026 10:26:21 +0200 Subject: [PATCH 02/11] Fix maturin being unusable by the runner user uv installed itself and maturin under /root/.local, and the image copied that tree into /home/runner/.local. The copy brought along maturin's launcher symlink, which points at an absolute path inside /root/.local/share/uv/tools. /root is mode 0700, so the unprivileged runner user that actually executes jobs could not traverse it: $ gosu runner maturin --version error: exec: "maturin": executable file not found in $PATH It worked as root, which is why this went unnoticed. Both architectures were affected. Install uv into /usr/local/bin and its tools into /opt/uv (via UV_INSTALL_DIR / UV_TOOL_BIN_DIR / UV_TOOL_DIR) so they sit on the shared PATH and are readable by every user. This also removes the need to copy the tree into the runner's home at all. Verified in the arm64 image: `gosu runner maturin --version` reports 1.14.1 and `maturin list-python` resolves CPython 3.12. Co-Authored-By: Claude Claude-Session: https://claude.ai/code/session_01CM6943zQZosnY2QugpiQxf --- Dockerfile | 19 ++++++++++--------- 1 file changed, 10 insertions(+), 9 deletions(-) diff --git a/Dockerfile b/Dockerfile index 2a05743..b1234e8 100644 --- a/Dockerfile +++ b/Dockerfile @@ -74,13 +74,15 @@ RUN ARCH="${TARGETARCH:-$(dpkg --print-architecture)}" && \ # ============================================================================ # Install uv (fast Python package manager) and maturin (Rust-Python build tool) # ============================================================================ -RUN curl -LsSf https://astral.sh/uv/install.sh | sh && \ - # Add uv to PATH - . $HOME/.local/bin/env && \ - # Install maturin globally via uv - uv tool install maturin +# Installed into shared, world-readable locations rather than under /root, which +# is mode 0700: a tool symlinked out of /root is unusable by the unprivileged +# runner user that actually executes jobs. +ENV UV_TOOL_DIR=/opt/uv/tools -ENV PATH="/root/.local/bin:${PATH}" +RUN curl -LsSf https://astral.sh/uv/install.sh | env UV_INSTALL_DIR=/usr/local/bin sh && \ + # Install maturin globally, with its launcher on the shared PATH + UV_TOOL_BIN_DIR=/usr/local/bin uv tool install maturin && \ + chmod -R a+rX /opt/uv # ============================================================================ # Create runner directory and download GitHub Actions Runner @@ -124,9 +126,8 @@ RUN useradd -m runner && \ cp -r /root/.rustup/* /home/runner/.rustup/ 2>/dev/null || true && \ # Copy export-esp.sh to runner home cp /root/export-esp.sh /home/runner/export-esp.sh 2>/dev/null || true && \ - # Copy uv and tools to runner user - mkdir -p /home/runner/.local && \ - cp -r /root/.local/* /home/runner/.local/ 2>/dev/null || true && \ + # uv and its tools (maturin) live in /usr/local/bin and /opt/uv, which are + # already on the shared PATH and readable by this user — nothing to copy. # Copy SSH config to runner user mkdir -p /home/runner/.ssh && \ cp /root/.ssh/known_hosts /home/runner/.ssh/ && \ From 75c4e46280689b5a9ada210031326546201de438 Mon Sep 17 00:00:00 2001 From: Riccardo Persello Date: Mon, 27 Jul 2026 11:10:09 +0200 Subject: [PATCH 03/11] Add shared cargo cache and per-replica resource limits A runner executes one job at a time, so parallelism is purely the replica count. Default it to 4 and bound what each replica may consume. Sized for a 16-core / 64 GB host: 4 replicas x 4 CPUs x 10 GB, leaving headroom for the host OS. CARGO_BUILD_JOBS is pinned to RUNNER_CPUS because cargo otherwise sizes its thread pool from the host core count, so each replica would spawn ~16 threads and N replicas would oversubscribe the machine N-fold; the cpus limit alone only throttles the result rather than preventing the thrashing. All three knobs are overridable via RUNNER_COUNT / RUNNER_CPUS / RUNNER_MEMORY. Replicas now share a cargo-registry volume instead of each re-downloading the full dependency set. Only the registry is shared, not the whole CARGO_HOME: cargo locks that directory so concurrent access is safe, whereas a shared target/ dir would race. entrypoint.sh repairs ownership of the registry volume when it comes back root-owned, which happens for a volume not seeded from the image and would otherwise silently break every build. Also drops the runner-data volume, which was declared but never mounted. Verified on the arm64 image: compose applies the limits outside swarm (NanoCpus=4000000000, Memory=10737418240) across 4 replicas; the runner user can write to the registry volume both when seeded from the image and after the root-owned repair path; and a crate fetched in one container is served to a second via `cargo fetch --offline`, confirming real sharing. Co-Authored-By: Claude Claude-Session: https://claude.ai/code/session_01CM6943zQZosnY2QugpiQxf --- README.md | 37 +++++++++++++++++++++++++++++++++++-- docker-compose.yml | 22 ++++++++++++++++++++-- entrypoint.sh | 8 ++++++++ 3 files changed, 63 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 46bc2c0..e93f9b7 100644 --- a/README.md +++ b/README.md @@ -57,6 +57,35 @@ docker buildx build --platform linux/arm64 -t github-runner . | `RUNNER_TOKEN` | One of `GITHUB_PAT` / `RUNNER_TOKEN` | Static runner registration token from GitHub. Expires ~1 hour after creation, so restarts after that will fail unless refreshed. Ignored if `GITHUB_PAT` is set. | | `RUNNER_NAME` | No | Base name for the runner (default: `runner`) | | `RUNNER_LABELS` | No | Comma-separated labels for the runner | +| `RUNNER_COUNT` | No | Number of runner replicas (default: `4`) | +| `RUNNER_CPUS` | No | CPUs per replica; also caps `CARGO_BUILD_JOBS` (default: `4`) | +| `RUNNER_MEMORY` | No | Memory per replica (default: `10g`) | + +### Parallel Jobs + +A GitHub Actions runner executes **one job at a time** — there is no concurrency +setting inside the runner. Total parallelism is therefore just `RUNNER_COUNT`. + +The defaults (4 replicas x 4 CPUs x 10 GB) target a 16-core / 64 GB host. Each +replica gets a hard CPU and memory limit, and `CARGO_BUILD_JOBS` is pinned to +`RUNNER_CPUS` — without that, cargo sizes its thread pool from the *host* core +count and every replica would spawn ~16 threads, oversubscribing the machine. + +Raising `RUNNER_COUNT` past the core count trades per-job latency for throughput: +8 replicas x 2 CPUs runs twice as many jobs, but each Rust build is much slower. +Prefer more replicas only if your jobs are mostly light (fmt, clippy, tests) +rather than full firmware builds. + +```bash +RUNNER_COUNT=8 RUNNER_CPUS=2 RUNNER_MEMORY=6g docker compose up -d --build +``` + +### Caching + +Replicas share a `cargo-registry` volume, so crates are downloaded once rather +than once per replica. Only the registry is shared — cargo locks it, making +concurrent access safe, whereas a shared `target/` directory would race. +Build artifacts are **not** shared or persisted across `docker compose down`. ### Running with Docker Compose @@ -65,6 +94,10 @@ docker buildx build --platform linux/arm64 -t github-runner . export URL=https://github.com/jkuracing export GITHUB_PAT= -# Start the runner -docker compose up -d +# Start the runners +docker compose up -d --build ``` + +> Always pass `--build`. Plain `docker compose up -d` only builds when the image +> is missing, so it will happily keep running a stale image after the Dockerfile +> or `entrypoint.sh` changes. diff --git a/docker-compose.yml b/docker-compose.yml index 235563e..8201dd6 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -9,8 +9,26 @@ services: RUNNER_TOKEN: ${RUNNER_TOKEN} RUNNER_NAME: ${RUNNER_NAME} RUNNER_LABELS: ${RUNNER_LABELS} + # Match cargo's internal parallelism to this replica's CPU allotment. + # cargo defaults to one codegen unit per *host* core, so without this each + # replica would spawn ~16 threads and N replicas would oversubscribe the + # machine N-fold. The cpus limit below only throttles the result; capping + # the thread count is what actually avoids the thrashing. + CARGO_BUILD_JOBS: ${RUNNER_CPUS:-4} + volumes: + # Shared crate download cache. Only the registry is shared, not the whole + # CARGO_HOME: cargo locks this directory, so concurrent replicas are safe, + # whereas a shared target/ dir would race. Without this every replica + # re-downloads the full dependency set on a cold start. + - cargo-registry:/home/runner/.cargo/registry deploy: - replicas: ${RUNNER_COUNT:-1} + replicas: ${RUNNER_COUNT:-4} + resources: + limits: + # Sized for a 16-core / 64 GB host, leaving headroom for macOS itself. + # 4 x 4 CPUs saturates the machine without oversubscribing it. + cpus: ${RUNNER_CPUS:-4} + memory: ${RUNNER_MEMORY:-10g} volumes: - runner-data: + cargo-registry: diff --git a/entrypoint.sh b/entrypoint.sh index b74e239..96d66b7 100644 --- a/entrypoint.sh +++ b/entrypoint.sh @@ -18,6 +18,14 @@ FULL_RUNNER_NAME="${RUNNER_NAME}-${HOSTNAME}" echo "Fixing permissions for /actions-runner..." chown -R runner:runner /actions-runner +# The shared cargo registry is a named volume. Docker seeds it from the image +# with the right ownership, but a volume created before that directory existed +# (or by another image) comes back root-owned and silently breaks every build. +if [[ -d /home/runner/.cargo/registry ]] && [[ "$(stat -c %U /home/runner/.cargo/registry)" != "runner" ]]; then + echo "Fixing permissions for the shared cargo registry..." + chown -R runner:runner /home/runner/.cargo/registry +fi + # Fetches a short-lived token ($1: "registration-token" or "remove-token") from the # GitHub API, using GITHUB_PAT. Prints the token on stdout, returns non-zero on failure. fetch_runner_token() { From 6f0cafd253e11105238bf77c92412fcd020e7c02 Mon Sep 17 00:00:00 2001 From: Riccardo Persello Date: Mon, 27 Jul 2026 11:12:50 +0200 Subject: [PATCH 04/11] Default RUNNER_LABELS to fw-builder Every job in the firmware repo's firmware_ci.yml targets `runs-on: labels: [fw-builder]`. With no labels set, a runner registers successfully and then sits idle forever, since no job ever matches it. Co-Authored-By: Claude Claude-Session: https://claude.ai/code/session_01CM6943zQZosnY2QugpiQxf --- docker-compose.yml | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/docker-compose.yml b/docker-compose.yml index 8201dd6..72bcb53 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -8,7 +8,9 @@ services: GITHUB_PAT: ${GITHUB_PAT} RUNNER_TOKEN: ${RUNNER_TOKEN} RUNNER_NAME: ${RUNNER_NAME} - RUNNER_LABELS: ${RUNNER_LABELS} + # firmware_ci.yml targets `runs-on: labels: [fw-builder]`, so a runner + # without this label is registered but never assigned any job. + RUNNER_LABELS: ${RUNNER_LABELS:-fw-builder} # Match cargo's internal parallelism to this replica's CPU allotment. # cargo defaults to one codegen unit per *host* core, so without this each # replica would spawn ~16 threads and N replicas would oversubscribe the From ab95f76a2c394775216e6c5ec577beb192a319bf Mon Sep 17 00:00:00 2001 From: Riccardo Persello Date: Mon, 27 Jul 2026 11:36:42 +0200 Subject: [PATCH 05/11] Scale to 8 replicas at 2 CPUs / 6 GB each MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Observed CI state shows runs queuing while host CPU sits idle, so job throughput is runner-starved rather than compute-bound. A single run only reaches 5 concurrent jobs, but firmware_ci.yml keys its concurrency group per branch, so several runs execute simultaneously and jobs queue globally — more replicas do get used. Memory is the binding constraint: 8 x 6 GB = 48 GB of the ~58 GB the OrbStack VM exposes, leaving host headroom. CPUs are oversubscribed 1:1 (8 x 2 = 16) since jobs spend much of their wall time on network and link steps rather than pegged compute. Co-Authored-By: Claude Claude-Session: https://claude.ai/code/session_01CM6943zQZosnY2QugpiQxf --- README.md | 22 +++++++++++++--------- docker-compose.yml | 15 +++++++++------ 2 files changed, 22 insertions(+), 15 deletions(-) diff --git a/README.md b/README.md index e93f9b7..2b78cd8 100644 --- a/README.md +++ b/README.md @@ -57,27 +57,31 @@ docker buildx build --platform linux/arm64 -t github-runner . | `RUNNER_TOKEN` | One of `GITHUB_PAT` / `RUNNER_TOKEN` | Static runner registration token from GitHub. Expires ~1 hour after creation, so restarts after that will fail unless refreshed. Ignored if `GITHUB_PAT` is set. | | `RUNNER_NAME` | No | Base name for the runner (default: `runner`) | | `RUNNER_LABELS` | No | Comma-separated labels for the runner | -| `RUNNER_COUNT` | No | Number of runner replicas (default: `4`) | -| `RUNNER_CPUS` | No | CPUs per replica; also caps `CARGO_BUILD_JOBS` (default: `4`) | -| `RUNNER_MEMORY` | No | Memory per replica (default: `10g`) | +| `RUNNER_COUNT` | No | Number of runner replicas (default: `8`) | +| `RUNNER_CPUS` | No | CPUs per replica; also caps `CARGO_BUILD_JOBS` (default: `2`) | +| `RUNNER_MEMORY` | No | Memory per replica (default: `6g`) | ### Parallel Jobs A GitHub Actions runner executes **one job at a time** — there is no concurrency setting inside the runner. Total parallelism is therefore just `RUNNER_COUNT`. -The defaults (4 replicas x 4 CPUs x 10 GB) target a 16-core / 64 GB host. Each +The defaults (8 replicas x 2 CPUs x 6 GB) target a 16-core / 64 GB host. Each replica gets a hard CPU and memory limit, and `CARGO_BUILD_JOBS` is pinned to `RUNNER_CPUS` — without that, cargo sizes its thread pool from the *host* core count and every replica would spawn ~16 threads, oversubscribing the machine. -Raising `RUNNER_COUNT` past the core count trades per-job latency for throughput: -8 replicas x 2 CPUs runs twice as many jobs, but each Rust build is much slower. -Prefer more replicas only if your jobs are mostly light (fmt, clippy, tests) -rather than full firmware builds. +**Memory, not CPU, is what limits the replica count.** 8 x 6 GB = 48 GB of the +~58 GB the OrbStack VM exposes. Raising `RUNNER_COUNT` without lowering +`RUNNER_MEMORY` will overcommit and get builds OOM-killed. + +A single CI run only reaches 5 concurrent jobs (four checks in parallel, then +three builds behind `needs`). The reason more replicas still help is that +`concurrency` in `firmware_ci.yml` is keyed per *branch*, so several runs +execute at once and jobs queue globally. ```bash -RUNNER_COUNT=8 RUNNER_CPUS=2 RUNNER_MEMORY=6g docker compose up -d --build +RUNNER_COUNT=4 RUNNER_CPUS=4 RUNNER_MEMORY=10g docker compose up -d --build ``` ### Caching diff --git a/docker-compose.yml b/docker-compose.yml index 72bcb53..a0dbad9 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -16,7 +16,7 @@ services: # replica would spawn ~16 threads and N replicas would oversubscribe the # machine N-fold. The cpus limit below only throttles the result; capping # the thread count is what actually avoids the thrashing. - CARGO_BUILD_JOBS: ${RUNNER_CPUS:-4} + CARGO_BUILD_JOBS: ${RUNNER_CPUS:-2} volumes: # Shared crate download cache. Only the registry is shared, not the whole # CARGO_HOME: cargo locks this directory, so concurrent replicas are safe, @@ -24,13 +24,16 @@ services: # re-downloads the full dependency set on a cold start. - cargo-registry:/home/runner/.cargo/registry deploy: - replicas: ${RUNNER_COUNT:-4} + replicas: ${RUNNER_COUNT:-8} resources: limits: - # Sized for a 16-core / 64 GB host, leaving headroom for macOS itself. - # 4 x 4 CPUs saturates the machine without oversubscribing it. - cpus: ${RUNNER_CPUS:-4} - memory: ${RUNNER_MEMORY:-10g} + # Sized for a 16-core / 64 GB host. Memory is the binding constraint, + # not CPU: 8 x 6g = 48 GB of the ~58 GB the VM exposes, leaving + # headroom for the host. CPUs are deliberately oversubscribed 1:1 + # (8 x 2 = 16) because jobs spend much of their wall time on network + # and link steps rather than pegged compute. + cpus: ${RUNNER_CPUS:-2} + memory: ${RUNNER_MEMORY:-6g} volumes: cargo-registry: From bce3379764f437a6eaf9470363efc1afafce902b Mon Sep 17 00:00:00 2001 From: Riccardo Persello Date: Mon, 27 Jul 2026 11:49:06 +0200 Subject: [PATCH 06/11] Persist sccache per replica across container recreation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Containers are not long-lived, so $HOME/.cache/sccache in the writable layer is discarded on every recreate and each new container recompiles from cold. setup-rust-dual in the firmware repo configures a 25 GB sccache there and describes it as living on persistent runner storage, so that cache is worth keeping across recreates. It cannot be one shared volume. sccache maintains its LRU index in memory per server process, so several containers pointed at one cache directory evict against each other and corrupt it. Since every replica of a scaled service shares one set of volumes, `deploy.replicas` cannot express per-replica storage — the replicas are now eight explicit services built from a YAML anchor, each with its own sccache volume. The crate registry stays shared, which is safe because cargo locks it. _work is still not persisted: the stale submodule target/ it would preserve is exactly what was breaking builds. The Dockerfile pre-creates the sccache directory so its volume is seeded with runner ownership, and the entrypoint's ownership repair now covers both volume paths — a volume that is non-empty and root-owned is not re-seeded by Docker and would otherwise be unwritable by the runner user. Verified: compose resolves 8 services each with a distinct sccache volume and a shared cargo-registry; the runner user can write the sccache volume when seeded from the image; and with a deliberately root-owned non-empty volume, an unrepaired write fails with EACCES while the entrypoint loop restores ownership and the write succeeds. Co-Authored-By: Claude Claude-Session: https://claude.ai/code/session_01CM6943zQZosnY2QugpiQxf --- Dockerfile | 4 ++ README.md | 42 +++++++++++++----- docker-compose.yml | 107 ++++++++++++++++++++++++++++++--------------- entrypoint.sh | 16 ++++--- 4 files changed, 115 insertions(+), 54 deletions(-) diff --git a/Dockerfile b/Dockerfile index b1234e8..624058d 100644 --- a/Dockerfile +++ b/Dockerfile @@ -128,6 +128,10 @@ RUN useradd -m runner && \ cp /root/export-esp.sh /home/runner/export-esp.sh 2>/dev/null || true && \ # uv and its tools (maturin) live in /usr/local/bin and /opt/uv, which are # already on the shared PATH and readable by this user — nothing to copy. + # Pre-create the sccache directory so its named volume is seeded with runner + # ownership. A volume mounted over a path that does not exist in the image is + # created root-owned, which the unprivileged runner cannot write to. + mkdir -p /home/runner/.cache/sccache && \ # Copy SSH config to runner user mkdir -p /home/runner/.ssh && \ cp /root/.ssh/known_hosts /home/runner/.ssh/ && \ diff --git a/README.md b/README.md index 2b78cd8..0469300 100644 --- a/README.md +++ b/README.md @@ -57,7 +57,6 @@ docker buildx build --platform linux/arm64 -t github-runner . | `RUNNER_TOKEN` | One of `GITHUB_PAT` / `RUNNER_TOKEN` | Static runner registration token from GitHub. Expires ~1 hour after creation, so restarts after that will fail unless refreshed. Ignored if `GITHUB_PAT` is set. | | `RUNNER_NAME` | No | Base name for the runner (default: `runner`) | | `RUNNER_LABELS` | No | Comma-separated labels for the runner | -| `RUNNER_COUNT` | No | Number of runner replicas (default: `8`) | | `RUNNER_CPUS` | No | CPUs per replica; also caps `CARGO_BUILD_JOBS` (default: `2`) | | `RUNNER_MEMORY` | No | Memory per replica (default: `6g`) | @@ -66,30 +65,49 @@ docker buildx build --platform linux/arm64 -t github-runner . A GitHub Actions runner executes **one job at a time** — there is no concurrency setting inside the runner. Total parallelism is therefore just `RUNNER_COUNT`. -The defaults (8 replicas x 2 CPUs x 6 GB) target a 16-core / 64 GB host. Each -replica gets a hard CPU and memory limit, and `CARGO_BUILD_JOBS` is pinned to -`RUNNER_CPUS` — without that, cargo sizes its thread pool from the *host* core -count and every replica would spawn ~16 threads, oversubscribing the machine. +Eight replicas (`runner-1` .. `runner-8`) are declared explicitly in +`docker-compose.yml`, at 2 CPUs and 6 GB each, sized for a 16-core / 64 GB host. +`CARGO_BUILD_JOBS` is pinned to `RUNNER_CPUS` — without that, cargo sizes its +thread pool from the *host* core count and every replica would spawn ~16 +threads, oversubscribing the machine. **Memory, not CPU, is what limits the replica count.** 8 x 6 GB = 48 GB of the -~58 GB the OrbStack VM exposes. Raising `RUNNER_COUNT` without lowering -`RUNNER_MEMORY` will overcommit and get builds OOM-killed. +~58 GB the OrbStack VM exposes. Adding replicas without lowering `RUNNER_MEMORY` +will overcommit and get builds OOM-killed. A single CI run only reaches 5 concurrent jobs (four checks in parallel, then three builds behind `needs`). The reason more replicas still help is that `concurrency` in `firmware_ci.yml` is keyed per *branch*, so several runs execute at once and jobs queue globally. +To run fewer runners, name the services; to run bigger ones, raise the limits: + ```bash -RUNNER_COUNT=4 RUNNER_CPUS=4 RUNNER_MEMORY=10g docker compose up -d --build +docker compose up -d --build runner-1 runner-2 runner-3 +RUNNER_CPUS=4 RUNNER_MEMORY=10g docker compose up -d --build ``` +> Replicas are separate services rather than `deploy.replicas` because a scaled +> service shares one set of volumes, and sccache cannot safely share a cache +> directory between concurrent server processes (see below). + ### Caching -Replicas share a `cargo-registry` volume, so crates are downloaded once rather -than once per replica. Only the registry is shared — cargo locks it, making -concurrent access safe, whereas a shared `target/` directory would race. -Build artifacts are **not** shared or persisted across `docker compose down`. +Two caches survive container recreation: + +- **`cargo-registry`** — shared by all replicas. Crates are downloaded once + rather than once per runner. Sharing is safe because cargo locks the registry. +- **`sccache-N`** — one volume *per replica*. `setup-rust-dual` in the firmware + repo points sccache at `$HOME/.cache/sccache`, and sccache keeps an in-memory + LRU index per server process, so several containers sharing one cache + directory would evict against each other and corrupt it. + +The runner's `_work` directory is deliberately **not** persisted. The firmware +workflow checks out with `clean: false` to reuse `target/`, but a stale +submodule `target/` surviving `git submodule deinit` is what produced +`could not parse/generate dep info ... No such file or directory` build +failures. sccache is content-hashed and immune to that staleness, so it is the +right layer to persist; `_work` is not. ### Running with Docker Compose diff --git a/docker-compose.yml b/docker-compose.yml index a0dbad9..b824987 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -1,39 +1,76 @@ +# Replicas are declared explicitly rather than via `deploy.replicas` because +# every replica of a scaled service shares one set of volumes. sccache keeps an +# in-memory LRU index per server process, so pointing several containers at one +# cache directory lets them evict against each other and corrupt it. Each runner +# therefore needs its OWN sccache volume, which requires its own service. +# +# The crate registry is different: cargo locks it, so a single shared volume is +# safe and avoids N copies of the same downloads. +# +# `docker compose up -d --build` starts all 8. To run fewer, name them: +# docker compose up -d --build runner-1 runner-2 runner-3 +x-runner: &runner + build: . + restart: on-failure:5 + stop_grace_period: 5m + environment: + URL: ${URL} + GITHUB_PAT: ${GITHUB_PAT} + RUNNER_TOKEN: ${RUNNER_TOKEN} + RUNNER_NAME: ${RUNNER_NAME} + # firmware_ci.yml targets `runs-on: labels: [fw-builder]`, so a runner + # without this label is registered but never assigned any job. + RUNNER_LABELS: ${RUNNER_LABELS:-fw-builder} + # Match cargo's internal parallelism to this replica's CPU allotment. + # cargo defaults to one codegen unit per *host* core, so without this each + # replica would spawn ~16 threads and 8 replicas would oversubscribe the + # machine 8-fold. The cpus limit below only throttles the result; capping + # the thread count is what actually avoids the thrashing. + CARGO_BUILD_JOBS: ${RUNNER_CPUS:-2} + deploy: + resources: + limits: + # Sized for a 16-core / 64 GB host. Memory is the binding constraint, + # not CPU: 8 x 6g = 48 GB of the ~58 GB the VM exposes, leaving + # headroom for the host. CPUs are deliberately oversubscribed 1:1 + # (8 x 2 = 16) because jobs spend much of their wall time on network + # and link steps rather than pegged compute. + cpus: ${RUNNER_CPUS:-2} + memory: ${RUNNER_MEMORY:-6g} + services: - github-runner: - build: . - restart: on-failure:5 - stop_grace_period: 5m - environment: - URL: ${URL} - GITHUB_PAT: ${GITHUB_PAT} - RUNNER_TOKEN: ${RUNNER_TOKEN} - RUNNER_NAME: ${RUNNER_NAME} - # firmware_ci.yml targets `runs-on: labels: [fw-builder]`, so a runner - # without this label is registered but never assigned any job. - RUNNER_LABELS: ${RUNNER_LABELS:-fw-builder} - # Match cargo's internal parallelism to this replica's CPU allotment. - # cargo defaults to one codegen unit per *host* core, so without this each - # replica would spawn ~16 threads and N replicas would oversubscribe the - # machine N-fold. The cpus limit below only throttles the result; capping - # the thread count is what actually avoids the thrashing. - CARGO_BUILD_JOBS: ${RUNNER_CPUS:-2} - volumes: - # Shared crate download cache. Only the registry is shared, not the whole - # CARGO_HOME: cargo locks this directory, so concurrent replicas are safe, - # whereas a shared target/ dir would race. Without this every replica - # re-downloads the full dependency set on a cold start. - - cargo-registry:/home/runner/.cargo/registry - deploy: - replicas: ${RUNNER_COUNT:-8} - resources: - limits: - # Sized for a 16-core / 64 GB host. Memory is the binding constraint, - # not CPU: 8 x 6g = 48 GB of the ~58 GB the VM exposes, leaving - # headroom for the host. CPUs are deliberately oversubscribed 1:1 - # (8 x 2 = 16) because jobs spend much of their wall time on network - # and link steps rather than pegged compute. - cpus: ${RUNNER_CPUS:-2} - memory: ${RUNNER_MEMORY:-6g} + runner-1: + <<: *runner + volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-1:/home/runner/.cache/sccache] + runner-2: + <<: *runner + volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-2:/home/runner/.cache/sccache] + runner-3: + <<: *runner + volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-3:/home/runner/.cache/sccache] + runner-4: + <<: *runner + volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-4:/home/runner/.cache/sccache] + runner-5: + <<: *runner + volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-5:/home/runner/.cache/sccache] + runner-6: + <<: *runner + volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-6:/home/runner/.cache/sccache] + runner-7: + <<: *runner + volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-7:/home/runner/.cache/sccache] + runner-8: + <<: *runner + volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-8:/home/runner/.cache/sccache] volumes: cargo-registry: + sccache-1: + sccache-2: + sccache-3: + sccache-4: + sccache-5: + sccache-6: + sccache-7: + sccache-8: diff --git a/entrypoint.sh b/entrypoint.sh index 96d66b7..653b5b0 100644 --- a/entrypoint.sh +++ b/entrypoint.sh @@ -18,13 +18,15 @@ FULL_RUNNER_NAME="${RUNNER_NAME}-${HOSTNAME}" echo "Fixing permissions for /actions-runner..." chown -R runner:runner /actions-runner -# The shared cargo registry is a named volume. Docker seeds it from the image -# with the right ownership, but a volume created before that directory existed -# (or by another image) comes back root-owned and silently breaks every build. -if [[ -d /home/runner/.cargo/registry ]] && [[ "$(stat -c %U /home/runner/.cargo/registry)" != "runner" ]]; then - echo "Fixing permissions for the shared cargo registry..." - chown -R runner:runner /home/runner/.cargo/registry -fi +# These are named volumes. Docker seeds them from the image with the right +# ownership, but a volume created before the directory existed in the image (or +# by another image) comes back root-owned and silently breaks every build. +for vol_dir in /home/runner/.cargo/registry /home/runner/.cache/sccache; do + if [[ -d "$vol_dir" ]] && [[ "$(stat -c %U "$vol_dir")" != "runner" ]]; then + echo "Fixing permissions for ${vol_dir}..." + chown -R runner:runner "$vol_dir" + fi +done # Fetches a short-lived token ($1: "registration-token" or "remove-token") from the # GitHub API, using GITHUB_PAT. Prints the token on stdout, returns non-zero on failure. From 862ab62c1163a8856cf3d1013b620c1205e13aab Mon Sep 17 00:00:00 2001 From: Riccardo Persello Date: Mon, 27 Jul 2026 12:12:18 +0200 Subject: [PATCH 07/11] Give each runner its own cargo registry volume Sharing one cargo-registry volume across the 8 runners introduced a new CI failure absent from every run before it: error: could not compile `crc32fast` (lib) Caused by: could not execute process `.../bin/rustc --crate-name crc32fast .../registry/src/index.crates.io-*/crc32fast-1.5.0/src/lib.rs` Caused by: No such file or directory (os error 2) `could not execute process` appears 0 times across runs predating the shared volume and immediately after it, with the vanished path inside the shared registry. Unpacked sources under registry/src are removed mid-compile when another container's cargo garbage-collects the global cache, so the rustc spawn fails on a working directory that no longer exists. Cargo's package-cache lock does not cover a build for its whole duration, and it cannot arbitrate between separate containers. Give each runner its own registry volume, matching sccache. This costs N copies of the crate downloads and removes the only remaining shared mutable state between concurrently building runners. Note this is distinct from the pre-existing `could not parse/generate dep info` failures, which point at a submodule's target/ rather than the registry and are addressed separately in the firmware repo. Co-Authored-By: Claude Claude-Session: https://claude.ai/code/session_01CM6943zQZosnY2QugpiQxf --- README.md | 21 +++++++++++++-------- docker-compose.yml | 47 +++++++++++++++++++++++++++++++--------------- 2 files changed, 45 insertions(+), 23 deletions(-) diff --git a/README.md b/README.md index 0469300..318003d 100644 --- a/README.md +++ b/README.md @@ -93,14 +93,19 @@ RUNNER_CPUS=4 RUNNER_MEMORY=10g docker compose up -d --build ### Caching -Two caches survive container recreation: - -- **`cargo-registry`** — shared by all replicas. Crates are downloaded once - rather than once per runner. Sharing is safe because cargo locks the registry. -- **`sccache-N`** — one volume *per replica*. `setup-rust-dual` in the firmware - repo points sccache at `$HOME/.cache/sccache`, and sccache keeps an in-memory - LRU index per server process, so several containers sharing one cache - directory would evict against each other and corrupt it. +Two caches survive container recreation, both **per replica**: + +- **`cargo-registry-N`** — the crate download cache. +- **`sccache-N`** — the compiler cache. `setup-rust-dual` in the firmware repo + points sccache at `$HOME/.cache/sccache`. + +Neither may be shared between replicas. sccache keeps its LRU index in memory +per server process, so containers sharing one directory evict against each +other. The registry was shared in an earlier revision and broke CI: unpacked +sources under `registry/src` disappear mid-compile when another container's +cargo garbage-collects the global cache, producing +`could not execute process ... No such file or directory`. The cost of not +sharing is N copies of the same crate downloads, which is the right trade. The runner's `_work` directory is deliberately **not** persisted. The firmware workflow checks out with `clean: false` to reuse `target/`, but a stale diff --git a/docker-compose.yml b/docker-compose.yml index b824987..83d7a05 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -1,11 +1,21 @@ # Replicas are declared explicitly rather than via `deploy.replicas` because -# every replica of a scaled service shares one set of volumes. sccache keeps an -# in-memory LRU index per server process, so pointing several containers at one -# cache directory lets them evict against each other and corrupt it. Each runner -# therefore needs its OWN sccache volume, which requires its own service. +# every replica of a scaled service shares one set of volumes, and NOTHING here +# is safe to share between concurrently building runners: # -# The crate registry is different: cargo locks it, so a single shared volume is -# safe and avoids N copies of the same downloads. +# - sccache keeps its LRU index in memory per server process, so several +# containers on one cache directory evict against each other. +# - The cargo registry was shared here initially and caused real CI failures: +# error: could not compile `crc32fast` (lib) +# Caused by: could not execute process `.../bin/rustc --crate-name +# crc32fast .../registry/src/index.crates.io-*/crc32fast-1.5.0/src/lib.rs` +# Caused by: No such file or directory (os error 2) +# Unpacked sources under registry/src vanish mid-compile when another +# container's cargo garbage-collects the global cache, so the spawn fails on +# a working directory that no longer exists. Cargo's package-cache lock does +# not protect a build for its whole duration across separate containers. +# +# Each runner therefore gets its own registry AND sccache volume, which is only +# expressible as its own service. The cost is N copies of the crate downloads. # # `docker compose up -d --build` starts all 8. To run fewer, name them: # docker compose up -d --build runner-1 runner-2 runner-3 @@ -41,31 +51,38 @@ x-runner: &runner services: runner-1: <<: *runner - volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-1:/home/runner/.cache/sccache] + volumes: [cargo-registry-1:/home/runner/.cargo/registry, sccache-1:/home/runner/.cache/sccache] runner-2: <<: *runner - volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-2:/home/runner/.cache/sccache] + volumes: [cargo-registry-2:/home/runner/.cargo/registry, sccache-2:/home/runner/.cache/sccache] runner-3: <<: *runner - volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-3:/home/runner/.cache/sccache] + volumes: [cargo-registry-3:/home/runner/.cargo/registry, sccache-3:/home/runner/.cache/sccache] runner-4: <<: *runner - volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-4:/home/runner/.cache/sccache] + volumes: [cargo-registry-4:/home/runner/.cargo/registry, sccache-4:/home/runner/.cache/sccache] runner-5: <<: *runner - volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-5:/home/runner/.cache/sccache] + volumes: [cargo-registry-5:/home/runner/.cargo/registry, sccache-5:/home/runner/.cache/sccache] runner-6: <<: *runner - volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-6:/home/runner/.cache/sccache] + volumes: [cargo-registry-6:/home/runner/.cargo/registry, sccache-6:/home/runner/.cache/sccache] runner-7: <<: *runner - volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-7:/home/runner/.cache/sccache] + volumes: [cargo-registry-7:/home/runner/.cargo/registry, sccache-7:/home/runner/.cache/sccache] runner-8: <<: *runner - volumes: [cargo-registry:/home/runner/.cargo/registry, sccache-8:/home/runner/.cache/sccache] + volumes: [cargo-registry-8:/home/runner/.cargo/registry, sccache-8:/home/runner/.cache/sccache] volumes: - cargo-registry: + cargo-registry-1: + cargo-registry-2: + cargo-registry-3: + cargo-registry-4: + cargo-registry-5: + cargo-registry-6: + cargo-registry-7: + cargo-registry-8: sccache-1: sccache-2: sccache-3: From 3f0f1e680c9b93c7754f754077b5a5a9d4413abd Mon Sep 17 00:00:00 2001 From: Riccardo Persello Date: Sat, 8 Aug 2026 12:58:53 +0200 Subject: [PATCH 08/11] Stop the runner self-update from killing the whole fleet All eight replicas had been offline for about ten days, which also meant firmware CI could not run at all. The failure was silent: nothing was left running to report it. The runner self-updates in place, and a post-update runner drops a `.runner_migrated` marker beside its config. `config.sh` treats that marker ALONE as proof the runner is already configured -- confirmed by creating only `.runner_migrated` and passing a deliberately bogus token, which fails with "Cannot configure the runner because it is already configured" without even attempting to authenticate. The cleanup here stopped at `.credentials_rsaparams`, so every replica that had auto-updated exited 1 on its next restart. Deleting the marker is correct rather than expedient: this entrypoint always reconfigures from a freshly minted registration token, so no migrated state is worth preserving across a restart. `restart: on-failure:5` turned that per-restart failure into a permanent one -- five retries were spent in seconds, after which Docker left the containers dead. A runner fleet should heal rather than latch off, so it becomes `unless-stopped`. The entrypoint mints one token per start, so even a genuinely broken image loops visibly in the logs instead of failing silently. Default RUNNER_TOKEN and RUNNER_NAME to empty as well. Both are optional when GITHUB_PAT is set, but leaving them unset made `docker compose` print two warnings per service -- sixteen lines that buried the real error underneath. Regression test: plant `.runner_migrated` in a live container, restart it, and confirm the count of "Listening for Jobs" lines increases. Do not test this by grepping `docker logs | tail -N` for that string without counting: a container that booted fine and then broke still has the line in its history, which reports a crash-looping runner as healthy. Co-Authored-By: Claude Claude-Session: https://claude.ai/code/session_018cbZiFrvjiXMRHLq1L9PDf --- docker-compose.yml | 17 ++++++++++++++--- entrypoint.sh | 20 ++++++++++++++++++-- 2 files changed, 32 insertions(+), 5 deletions(-) diff --git a/docker-compose.yml b/docker-compose.yml index 83d7a05..2152f78 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -21,13 +21,24 @@ # docker compose up -d --build runner-1 runner-2 runner-3 x-runner: &runner build: . - restart: on-failure:5 + # `on-failure:5` used to be the policy here and it cost the fleet ten days of + # downtime: a bug in entrypoint.sh's config cleanup made every replica exit 1 + # on restart, the five retries were spent in seconds, and Docker then left all + # eight containers dead with no surviving process to notice. A runner fleet + # should heal rather than latch off, and the entrypoint mints a fresh + # registration token per start, so a genuinely broken image loops visibly in + # the logs instead of failing silently. + restart: unless-stopped stop_grace_period: 5m environment: URL: ${URL} GITHUB_PAT: ${GITHUB_PAT} - RUNNER_TOKEN: ${RUNNER_TOKEN} - RUNNER_NAME: ${RUNNER_NAME} + # Defaulted to empty: entrypoint.sh prefers GITHUB_PAT and only falls back + # to a static token, but an unset variable makes `docker compose` print a + # warning per service per invocation -- 16 lines of noise that bury the + # real errors underneath. + RUNNER_TOKEN: ${RUNNER_TOKEN:-} + RUNNER_NAME: ${RUNNER_NAME:-} # firmware_ci.yml targets `runs-on: labels: [fw-builder]`, so a runner # without this label is registered but never assigned any job. RUNNER_LABELS: ${RUNNER_LABELS:-fw-builder} diff --git a/entrypoint.sh b/entrypoint.sh index 653b5b0..88c35b2 100644 --- a/entrypoint.sh +++ b/entrypoint.sh @@ -81,8 +81,24 @@ if [[ -f /home/runner/export-esp.sh ]]; then fi echo "Removing any existing runner configuration..." -# Clean up previous runs (crucial for ephemeral runners) -rm -f .runner .credentials .credentials_rsaparams +# Clean up previous runs (crucial for ephemeral runners). +# +# `.runner_migrated` MUST be in this list. The runner self-updates in place, and +# a post-update runner drops that marker beside its config. `config.sh` treats +# the marker ALONE as proof the runner is already configured -- verified by +# creating only `.runner_migrated` and passing a deliberately bogus token: it +# fails with "Cannot configure the runner because it is already configured" +# without even attempting to authenticate. +# +# Because the old list stopped at `.credentials_rsaparams`, every replica that +# had auto-updated crash-looped on its next restart until `restart: +# on-failure:5` exhausted its retries, which silently took the entire fleet +# offline about ten days after it was last rebuilt. Deleting the marker is +# correct rather than merely expedient: this entrypoint always reconfigures from +# a freshly minted registration token, so there is no migrated state worth +# preserving across a restart. +rm -f .runner .credentials .credentials_rsaparams \ + .runner_migrated .credentials_migrated echo "Configuring GitHub Actions Runner as ${FULL_RUNNER_NAME}..." echo "URL: $URL" From 7adc233efa2f7be3b760a262b9090e16c5190aef Mon Sep 17 00:00:00 2001 From: Riccardo Persello Date: Sat, 8 Aug 2026 12:59:21 +0200 Subject: [PATCH 09/11] Build hbf as well as firmware: bun, webkit and a second label hbf CI runs on GitHub-hosted runners today and reinstalls its toolchain on every job. Moving it here needs two things the image lacked. Add bun and the Tauri desktop dependencies. `cargo build -p hbf-gui` links against webkit2gtk-4.1 and fails at pkg-config time without the -dev package; librsvg2 and appindicator3 are Tauri's SVG and tray-icon dependencies. bun builds the SvelteKit bundle that `tauri::generate_context!()` embeds at COMPILE time, which makes it a build dependency rather than a test-only tool, and it is pinned to the version hbf CI's `oven-sh/setup-bun` requests so lockfile resolution matches. BUN_INSTALL puts the binary on the shared PATH instead of under /root, which is mode 0700 and so invisible to the unprivileged runner user -- the same trap the uv block already documents. Both layers go AFTER espup deliberately. Docker invalidates every layer below an edited one, and rebuilding the Xtensa toolchain costs many minutes. Verified the `esp` toolchain survived the rebuild untouched. Image grows 8.86 -> 9.6 GB. Add `hbf-builder` to every replica rather than reserving a subset for it. A runner is offered a job only when its labels are a SUPERSET of the job's `runs-on`, so splitting them (1-6 fw-builder, 7-8 hbf-builder) would leave six containers ineligible for hbf work and idle whenever hbf work is all that is queued. Both labels everywhere means any replica serves either repo, and capacity is added by adding replicas. One trap worth recording: `docker compose build` must be run with NO service argument. Each service declares its own `build: .`, so compose tags a separate image per service, and `docker compose build runner-1` silently leaves the other seven on the old image -- they still start and register, so nothing looks wrong until a job needs a tool only the rebuilt image has. hbf CI cannot move here yet: `.github/actions/setup-canvas` downloads pkl-linux-amd64 while these runners are arm64, and it writes to /usr/local/bin, which the runner user cannot do. pkl is also skewed three ways (0.30.1 here, 0.31.1 in hbf CI, 0.32.1 on the dev machine). Co-Authored-By: Claude Claude-Session: https://claude.ai/code/session_018cbZiFrvjiXMRHLq1L9PDf --- Dockerfile | 29 +++++++++++++++++++++++++++++ docker-compose.yml | 10 +++++++--- 2 files changed, 36 insertions(+), 3 deletions(-) diff --git a/Dockerfile b/Dockerfile index 624058d..dd7fb67 100644 --- a/Dockerfile +++ b/Dockerfile @@ -84,6 +84,35 @@ RUN curl -LsSf https://astral.sh/uv/install.sh | env UV_INSTALL_DIR=/usr/local/b UV_TOOL_BIN_DIR=/usr/local/bin uv tool install maturin && \ chmod -R a+rX /opt/uv +# ============================================================================ +# Web UI and Tauri desktop dependencies (hbf) +# ============================================================================ +# Deliberately placed AFTER the espup layer. Docker invalidates every layer +# below an edited one, and rebuilding the Xtensa toolchain costs many minutes, +# so anything added later must stay later. +# +# `cargo build -p hbf-gui` links against webkit2gtk-4.1 and fails at +# pkg-config time without the -dev package; librsvg2 and appindicator3 are +# Tauri's SVG and tray-icon dependencies. This mirrors the apt list hbf CI +# installs per job, minus what the firmware layers above already provide +# (libudev-dev, pkg-config, libssl-dev). +RUN apt-get update && \ + apt-get install -y --no-install-recommends \ + libwebkit2gtk-4.1-dev libayatana-appindicator3-dev \ + librsvg2-dev && \ + apt-get clean && rm -rf /var/lib/apt/lists/* + +# bun builds the SvelteKit bundle that `tauri::generate_context!()` embeds at +# COMPILE time, so it is a build dependency of hbf-gui rather than a test-only +# tool. Pinned to the version hbf CI's `oven-sh/setup-bun` requests so lockfile +# resolution is identical on both. BUN_INSTALL places the binary on the shared +# PATH instead of under /root, which is mode 0700 and therefore invisible to the +# unprivileged runner user -- the same trap the uv block above documents. +ENV BUN_INSTALL=/usr/local +RUN curl -fsSL https://bun.sh/install | bash -s "bun-v1.3.14" && \ + chmod a+rx /usr/local/bin/bun && \ + bun --version + # ============================================================================ # Create runner directory and download GitHub Actions Runner # ============================================================================ diff --git a/docker-compose.yml b/docker-compose.yml index 2152f78..a12ebbb 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -39,9 +39,13 @@ x-runner: &runner # real errors underneath. RUNNER_TOKEN: ${RUNNER_TOKEN:-} RUNNER_NAME: ${RUNNER_NAME:-} - # firmware_ci.yml targets `runs-on: labels: [fw-builder]`, so a runner - # without this label is registered but never assigned any job. - RUNNER_LABELS: ${RUNNER_LABELS:-fw-builder} + # A runner is offered a job only when its label set is a SUPERSET of the + # job's `runs-on`. Both labels therefore go on every replica: splitting them + # across replicas (say 1-6 fw-builder, 7-8 hbf-builder) would leave six + # containers ineligible for hbf jobs and idle whenever hbf work is all that + # is queued. firmware_ci.yml asks for [fw-builder]; hbf asks for + # [hbf-builder]; every replica can serve either. + RUNNER_LABELS: ${RUNNER_LABELS:-fw-builder,hbf-builder} # Match cargo's internal parallelism to this replica's CPU allotment. # cargo defaults to one codegen unit per *host* core, so without this each # replica would spawn ~16 threads and 8 replicas would oversubscribe the From 3942a4aa31fac8d280f28f672a3b2112d33b22ff Mon Sep 17 00:00:00 2001 From: Riccardo Persello Date: Sat, 8 Aug 2026 16:11:14 +0200 Subject: [PATCH 10/11] Install Node alongside bun hbf's `ts_export` test execs `ui/node_modules/.bin/prettier` directly from Rust. That file is a .cjs script whose shebang is `#!/usr/bin/env node`, so on this image the exec failed with status 127 and the drift check reported "bindings would drift from CI's regen" -- a misleading message for a missing interpreter. bun does not substitute for node here. `bun run lint` and `bun run check` work because `bun run` interprets the JS itself and never consults the shebang, which is why the gap stays invisible until something shells out to a .bin entry. npm comes along for `npx`, which the same test falls back to when the project-local binary is absent. GitHub-hosted runners preinstall both, so this could only surface here. Co-Authored-By: Claude Claude-Session: https://claude.ai/code/session_018cbZiFrvjiXMRHLq1L9PDf --- Dockerfile | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/Dockerfile b/Dockerfile index dd7fb67..877b77f 100644 --- a/Dockerfile +++ b/Dockerfile @@ -102,6 +102,24 @@ RUN apt-get update && \ librsvg2-dev && \ apt-get clean && rm -rf /var/lib/apt/lists/* +# Node is needed even though bun is the package manager, because bun does not +# replace it as a script *interpreter*. hbf's `ts_export` test execs +# `ui/node_modules/.bin/prettier` directly from Rust; that file is a .cjs script +# whose shebang is `#!/usr/bin/env node`, so without node the exec fails with +# status 127 and the drift check reports "bindings would drift". `bun run lint` +# and `bun run check` are unaffected because `bun run` interprets the JS itself +# and never consults the shebang -- which is exactly why this gap is invisible +# until something shells out to a .bin entry. +# +# npm comes along for `npx`, which the same test falls back to when the +# project-local binary is absent. GitHub-hosted runners preinstall both, which is +# why this only surfaced on the fleet. +RUN apt-get update && \ + apt-get install -y --no-install-recommends \ + nodejs npm && \ + apt-get clean && rm -rf /var/lib/apt/lists/* && \ + node --version && npx --version + # bun builds the SvelteKit bundle that `tauri::generate_context!()` embeds at # COMPILE time, so it is a build dependency of hbf-gui rather than a test-only # tool. Pinned to the version hbf CI's `oven-sh/setup-bun` requests so lockfile From 92a81192b2390401b49ed25fb2d5fc7a37f02582 Mon Sep 17 00:00:00 2001 From: Riccardo Persello Date: Sat, 8 Aug 2026 16:43:25 +0200 Subject: [PATCH 11/11] Scale to 12 replicas at 2 CPU / 4 GB This block claimed "Memory is the binding constraint, not CPU" at 8 x 6 GB. Measured under a full load of firmware and hbf jobs, that was wrong on both counts: peak memory across all replicas was 756 MiB against the 6 GiB limit -- an 8x overshoot -- and only 4 of 8 containers were computing at all (~195% CPU each), the rest sitting near idle on network and setup. Roughly half the host's cores went unused while jobs queued. The real constraint was SLOTS. An hbf run measured 16.8 minutes of job time inside an 11.8 minute span -- an average concurrency of 1.4 -- because firmware held 7 of the 8 slots, so a pipeline that takes 2m41s on a hosted runner took 11m48s here. Hence more, smaller replicas. Total memory is unchanged at 48 GB of the ~58 GB the VM exposes. CPU is now oversubscribed 1.5:1 (12 x 2 = 24 on 16 cores), which the measured idle time justifies. Disk is the limiting factor now, not memory: each replica keeps its own target/ for both repositories on one 200 GB volume. Reclaiming 27 GB of stale build cache brought it to 43% before this change; check `docker system df` before going wider. Co-Authored-By: Claude Claude-Session: https://claude.ai/code/session_018cbZiFrvjiXMRHLq1L9PDf --- docker-compose.yml | 75 +++++++++++++++++++++++++++++++++++----------- 1 file changed, 58 insertions(+), 17 deletions(-) diff --git a/docker-compose.yml b/docker-compose.yml index a12ebbb..0bcc36a 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -17,14 +17,20 @@ # Each runner therefore gets its own registry AND sccache volume, which is only # expressible as its own service. The cost is N copies of the crate downloads. # -# `docker compose up -d --build` starts all 8. To run fewer, name them: +# `docker compose up -d --build` starts all 12. To run fewer, name them: # docker compose up -d --build runner-1 runner-2 runner-3 +# +# `docker compose build` MUST be run with no service argument. Each service +# declares its own `build: .`, so compose tags a separate image per service, and +# `docker compose build runner-1` silently leaves the others on the old image -- +# they still start and register, so nothing looks wrong until a job needs a tool +# only the rebuilt image has. x-runner: &runner build: . # `on-failure:5` used to be the policy here and it cost the fleet ten days of # downtime: a bug in entrypoint.sh's config cleanup made every replica exit 1 # on restart, the five retries were spent in seconds, and Docker then left all - # eight containers dead with no surviving process to notice. A runner fleet + # containers dead with no surviving process to notice. A runner fleet # should heal rather than latch off, and the entrypoint mints a fresh # registration token per start, so a genuinely broken image loops visibly in # the logs instead of failing silently. @@ -35,33 +41,48 @@ x-runner: &runner GITHUB_PAT: ${GITHUB_PAT} # Defaulted to empty: entrypoint.sh prefers GITHUB_PAT and only falls back # to a static token, but an unset variable makes `docker compose` print a - # warning per service per invocation -- 16 lines of noise that bury the - # real errors underneath. + # warning per service per invocation -- noise that buries the real errors + # underneath. RUNNER_TOKEN: ${RUNNER_TOKEN:-} RUNNER_NAME: ${RUNNER_NAME:-} # A runner is offered a job only when its label set is a SUPERSET of the # job's `runs-on`. Both labels therefore go on every replica: splitting them - # across replicas (say 1-6 fw-builder, 7-8 hbf-builder) would leave six - # containers ineligible for hbf jobs and idle whenever hbf work is all that - # is queued. firmware_ci.yml asks for [fw-builder]; hbf asks for - # [hbf-builder]; every replica can serve either. + # across replicas would leave some containers ineligible for hbf jobs and + # idle whenever hbf work is all that is queued. firmware_ci.yml asks for + # [fw-builder]; hbf asks for [hbf-builder]; every replica can serve either. RUNNER_LABELS: ${RUNNER_LABELS:-fw-builder,hbf-builder} # Match cargo's internal parallelism to this replica's CPU allotment. # cargo defaults to one codegen unit per *host* core, so without this each - # replica would spawn ~16 threads and 8 replicas would oversubscribe the - # machine 8-fold. The cpus limit below only throttles the result; capping - # the thread count is what actually avoids the thrashing. + # replica would spawn ~16 threads and every replica would oversubscribe the + # machine. The cpus limit below only throttles the result; capping the thread + # count is what actually avoids the thrashing. CARGO_BUILD_JOBS: ${RUNNER_CPUS:-2} deploy: resources: limits: - # Sized for a 16-core / 64 GB host. Memory is the binding constraint, - # not CPU: 8 x 6g = 48 GB of the ~58 GB the VM exposes, leaving - # headroom for the host. CPUs are deliberately oversubscribed 1:1 - # (8 x 2 = 16) because jobs spend much of their wall time on network - # and link steps rather than pegged compute. + # 12 replicas at 2 CPU / 4 GB on a 16-core / 64 GB host. + # + # This block previously read "Memory is the binding constraint, not CPU" + # at 8 x 6 GB. Measured under a full load of firmware and hbf jobs, that + # was wrong on both counts: peak usage across all replicas was 756 MiB + # against the 6 GiB limit -- an 8x overshoot -- and only 4 of 8 + # containers were computing at all (~195% CPU each), the rest sitting + # near idle on network and setup. Roughly half the host's cores were + # unused while jobs queued. + # + # The real constraint was SLOTS. An hbf run measured 16.8 minutes of job + # time inside an 11.8 minute span -- an average concurrency of 1.4 -- + # because firmware held 7 of the 8 slots. So: more, smaller replicas. + # + # Total memory is unchanged at 48 GB of the ~58 GB the VM exposes. CPU is + # deliberately oversubscribed 1.5:1 (12 x 2 = 24 on 16 cores), which the + # observed idle time justifies. + # + # DISK is now the limiting factor, not memory: each replica keeps its own + # target/ for both repositories on one 200 GB volume, which was 47% full + # at 8 replicas. Watch `docker system df` before going wider. cpus: ${RUNNER_CPUS:-2} - memory: ${RUNNER_MEMORY:-6g} + memory: ${RUNNER_MEMORY:-4g} services: runner-1: @@ -88,6 +109,18 @@ services: runner-8: <<: *runner volumes: [cargo-registry-8:/home/runner/.cargo/registry, sccache-8:/home/runner/.cache/sccache] + runner-9: + <<: *runner + volumes: [cargo-registry-9:/home/runner/.cargo/registry, sccache-9:/home/runner/.cache/sccache] + runner-10: + <<: *runner + volumes: [cargo-registry-10:/home/runner/.cargo/registry, sccache-10:/home/runner/.cache/sccache] + runner-11: + <<: *runner + volumes: [cargo-registry-11:/home/runner/.cargo/registry, sccache-11:/home/runner/.cache/sccache] + runner-12: + <<: *runner + volumes: [cargo-registry-12:/home/runner/.cargo/registry, sccache-12:/home/runner/.cache/sccache] volumes: cargo-registry-1: @@ -98,6 +131,10 @@ volumes: cargo-registry-6: cargo-registry-7: cargo-registry-8: + cargo-registry-9: + cargo-registry-10: + cargo-registry-11: + cargo-registry-12: sccache-1: sccache-2: sccache-3: @@ -106,3 +143,7 @@ volumes: sccache-6: sccache-7: sccache-8: + sccache-9: + sccache-10: + sccache-11: + sccache-12: