From 1f5307fdecbd393ccb206093f20e17c43a822d57 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 15 Jun 2026 01:38:23 +0200 Subject: [PATCH] feat(infra): migrated runner caches to PVC-backed Bun/Cargo and RustFS sccache - Updated bun-install action to set mounted cache mode and use PVC cache paths. - Removed RustFS Bun restore/save and maintenance scripts, replacing them with mounted cache setup. - Removed zstd from runner image installation and baked-tool verification checks. - Updated infra docs to describe split caching with RustFS for sccache and PVC for Bun/Cargo. --- .github/actions/build-native-kata/action.yml | 134 --------- .github/actions/build-native/action.yml | 192 ++++++++----- .github/actions/bun-install/action.yml | 27 +- .github/actions/bun-install/rustfs-cache.sh | 126 --------- .github/workflows/ci.yml | 31 ++- infra/docs/02-kata-runtime.md | 6 +- infra/docs/03-runner-image.md | 22 +- infra/docs/04-arc-and-caching.md | 239 ++++++++++------ infra/docs/README.md | 18 +- infra/reload-runner.sh | 4 +- infra/runner.Dockerfile | 2 +- infra/rustfs-cache-maintenance.sh | 276 ------------------- 12 files changed, 345 insertions(+), 732 deletions(-) delete mode 100644 .github/actions/build-native-kata/action.yml delete mode 100755 .github/actions/bun-install/rustfs-cache.sh delete mode 100755 infra/rustfs-cache-maintenance.sh diff --git a/.github/actions/build-native-kata/action.yml b/.github/actions/build-native-kata/action.yml deleted file mode 100644 index 6f2223ade..000000000 --- a/.github/actions/build-native-kata/action.yml +++ /dev/null @@ -1,134 +0,0 @@ -name: Build native addon (omp-kata) -description: > - Build the pi_natives cdylib on the preloaded omp-kata runner image, using - baked toolchains and the shared RustFS-backed sccache instead of per-job tool - setup downloads. - -inputs: - hash: - description: Rust source hash used in the artifact name - required: true - platform: - description: Target platform (linux, darwin, win32) - required: true - arch: - description: Target arch (x64, arm64) - required: true - variant: - description: Optional build variant (baseline, modern); required for native x64 builds. - required: false - default: "" - target: - description: Optional rustc target triple for cross-compilation - required: false - default: "" - rust_checks: - description: Run clippy/rustfmt checks (only one matrix entry should set this) - required: false - default: "false" - save_cache: - description: Kept for interface parity with the GitHub-hosted action; unused here. - required: false - default: "false" - -runs: - using: composite - steps: - - uses: ./.github/actions/ensure-rust-toolchain - with: - toolchain: nightly-2026-04-29 - components: ${{ inputs.rust_checks == 'true' && 'clippy,rustfmt' || '' }} - target: ${{ inputs.target }} - - name: Configure native Rust flags - if: inputs.target == '' - shell: bash - env: - TARGET_ARCH: ${{ inputs.arch }} - TARGET_VARIANT: ${{ inputs.variant }} - run: | - case "$TARGET_ARCH:$TARGET_VARIANT" in - x64:modern) - rustflags="-C target-cpu=x86-64-v3" - ;; - x64:baseline) - rustflags="-C target-cpu=x86-64-v2" - ;; - x64:*) - echo "::error::x64 native builds require variant=modern or variant=baseline" - exit 1 - ;; - *) - if [ -n "${RUSTFLAGS:-}" ]; then - echo "Using caller-provided RUSTFLAGS=$RUSTFLAGS" - exit 0 - fi - rustflags="-C target-cpu=native" - ;; - esac - - echo "RUSTFLAGS=$rustflags" >> "$GITHUB_ENV" - echo "Configured RUSTFLAGS=$rustflags" - - uses: ./.github/actions/ensure-sccache - with: - version: "0.15.0" - - name: Enable sccache for cargo - shell: bash - run: | - { - echo "RUSTC_WRAPPER=sccache" - echo "CARGO_INCREMENTAL=0" - } >> "$GITHUB_ENV" - echo "sccache backend: shared S3 ($SCCACHE_BUCKET @ $SCCACHE_ENDPOINT)" - - uses: ./.github/actions/ensure-cargo-tool - if: inputs.target == '' - with: - binary: cargo-nextest - crate: cargo-nextest - - uses: ./.github/actions/bun-install - - uses: ./.github/actions/ensure-zig - if: inputs.target != '' && !endsWith(inputs.target, '-msvc') - with: - version: "0.16.0" - - uses: ./.github/actions/ensure-cargo-tool - if: inputs.target != '' && !endsWith(inputs.target, '-msvc') - with: - binary: cargo-zigbuild - crate: cargo-zigbuild - - uses: ./.github/actions/ensure-cargo-tool - if: endsWith(inputs.target, '-msvc') - with: - binary: cargo-xwin - crate: cargo-xwin - - name: Cache cargo-xwin Windows SDK - if: endsWith(inputs.target, '-msvc') - uses: actions/cache@v4 - with: - path: ~/.cache/cargo-xwin - key: cargo-xwin-${{ runner.os }}-v1 - - name: Accept xwin license - if: endsWith(inputs.target, '-msvc') - shell: bash - run: echo "XWIN_ACCEPT_LICENSE=1" >> "$GITHUB_ENV" - - name: Rust checks - if: inputs.rust_checks == 'true' - shell: bash - run: bun run check:rs - - name: Test workspace (Rust) - if: inputs.target == '' && inputs.platform != 'darwin' - shell: bash - run: bun run test:rs - - name: Build native addon(s) - shell: bash - env: - CROSS_TARGET: ${{ inputs.target }} - TARGET_PLATFORM: ${{ inputs.platform }} - TARGET_ARCH: ${{ inputs.arch }} - TARGET_VARIANTS: ${{ inputs.variant }} - run: bun run ci:build:native - - name: Upload native addon(s) - uses: actions/upload-artifact@v4 - with: - name: pi-natives-${{ inputs.platform }}-${{ inputs.arch }}${{ inputs.variant && format('-{0}', inputs.variant) || '' }}-h${{ inputs.hash }} - path: packages/natives/native/pi_natives.${{ inputs.platform }}-${{ inputs.arch }}*.node - if-no-files-found: error - retention-days: 90 diff --git a/.github/actions/build-native/action.yml b/.github/actions/build-native/action.yml index c1da1cd60..7a6a89911 100644 --- a/.github/actions/build-native/action.yml +++ b/.github/actions/build-native/action.yml @@ -1,5 +1,12 @@ name: Build native addon -description: Build the pi_natives cdylib for one platform/arch/variant and upload it as a hash-tagged artifact. +description: > + Build the pi_natives cdylib for one platform/arch/variant and upload it as a + hash-tagged artifact. Self-detects the runner via $SCCACHE_BUCKET (injected + only on the self-hosted omp-kata pods): on-infra it uses the image's baked + toolchains + the RustFS-backed sccache; on GitHub-hosted runners it installs + the toolchains and uses Swatinem target/ cache + the GitHub Actions sccache + backend. PRs run on GitHub-hosted runners, so the on-infra path only ever + serves trusted push/main + release builds. inputs: hash: @@ -24,48 +31,64 @@ inputs: required: false default: "false" save_cache: - description: Whether Swatinem/rust-cache should write a cache entry + description: Whether Swatinem/rust-cache should write a cache entry (GitHub-hosted only) required: false default: "false" runs: using: composite steps: - - uses: dtolnay/rust-toolchain@nightly + - name: Detect runner environment + id: detect + shell: bash + run: | + # $SCCACHE_BUCKET is injected only on the self-hosted omp-kata runner + # pods (envFrom sccache-s3); its presence is the repo's single + # "on can.internal infra?" signal. On-infra: baked toolchains, RustFS + # sccache, no GitHub target/ cache. Off-infra (GitHub-hosted): install + # toolchains, Swatinem target/ cache, GitHub Actions sccache backend. + if [ -n "${SCCACHE_BUCKET:-}" ]; then + echo "on_infra=true" >> "$GITHUB_OUTPUT" + echo "runner: self-hosted omp-kata (baked tools + RustFS sccache)" + else + echo "on_infra=false" >> "$GITHUB_OUTPUT" + echo "runner: GitHub-hosted (install tools + Swatinem cache + GHA sccache)" + fi + + # --- Rust toolchain ----------------------------------------------------- + - name: Ensure baked Rust toolchain (omp-kata) + if: steps.detect.outputs.on_infra == 'true' + uses: ./.github/actions/ensure-rust-toolchain + with: + toolchain: nightly-2026-04-29 + components: ${{ inputs.rust_checks == 'true' && 'clippy,rustfmt' || '' }} + target: ${{ inputs.target }} + - name: Install Rust toolchain (GitHub-hosted) + if: steps.detect.outputs.on_infra == 'false' + uses: dtolnay/rust-toolchain@nightly with: toolchain: nightly-2026-04-29 components: ${{ inputs.rust_checks == 'true' && 'clippy, rustfmt' || '' }} targets: ${{ inputs.target }} - - name: Install Linux build prerequisites - if: runner.os == 'Linux' + - name: Install Linux build prerequisites (GitHub-hosted) + if: steps.detect.outputs.on_infra == 'false' && runner.os == 'Linux' shell: bash run: | sudo apt-get update sudo apt-get install -y build-essential - - name: Prepend rustup toolchain bin to PATH + - name: Prepend rustup toolchain bin to PATH (GitHub-hosted) + if: steps.detect.outputs.on_infra == 'false' shell: bash run: | - # Homebrew on macOS runners ships rustup-init with shadow proxies - # for `cargo`/`rustc`/etc. that error out as the installer - # ("unexpected argument 'metadata' found"). Force the real - # toolchain binaries to win on PATH. + # Homebrew on macOS runners ships rustup-init with shadow proxies for + # `cargo`/`rustc`/etc. that error out as the installer ("unexpected + # argument 'metadata' found"). Force the real toolchain binaries to win + # on PATH. (ensure-rust-toolchain already does this on omp-kata.) toolchain_bin="$(dirname "$(rustup which cargo)")" echo "$toolchain_bin" >> "$GITHUB_PATH" echo "Prepended $toolchain_bin to PATH" - # `Swatinem/rust-cache` keys target/ off its restore-time environment, so - # set RUSTFLAGS before deciding whether to restore it. If x64 target-cpu is - # only selected inside ci-build-native.ts/build-native.ts, cargo invalidates - # the restored target/ but rust-cache sees an exact key and refuses to save - # the rebuilt artifacts, causing macOS x64 baseline to rebuild forever. - # - # Include the native source hash in the shared key as well: rust-cache's - # lockfile scan misses the workspace root Cargo.toml version that Cargo - # fingerprints for workspace crates. Without it, release version bumps can - # get an exact hit for artifacts Cargo must rebuild. - # - # On GitHub-hosted runners, rust-cache can still warm target/. On omp-kata, - # the shared RustFS-backed sccache is the primary reuse layer and avoids a - # second GitHub-cache restore for Cargo dirs/target. + + # --- Rust flags (shared) ------------------------------------------------ - name: Configure native Rust flags if: inputs.target == '' shell: bash @@ -95,34 +118,40 @@ runs: echo "RUSTFLAGS=$rustflags" >> "$GITHUB_ENV" echo "Configured RUSTFLAGS=$rustflags" - - name: Decide Rust artifact cache - id: rust_artifact_cache - shell: bash - run: | - if [ -n "${SCCACHE_BUCKET:-}" ]; then - echo "use_rust_cache=false" >> "$GITHUB_OUTPUT" - echo "Rust target cache: skipped on shared-sccache runners" - else - echo "use_rust_cache=true" >> "$GITHUB_OUTPUT" - echo "Rust target cache: GitHub Actions" - fi - - uses: Swatinem/rust-cache@v2 - if: steps.rust_artifact_cache.outputs.use_rust_cache == 'true' + + # --- target/ cache (GitHub-hosted only) --------------------------------- + # Swatinem keys target/ off its restore-time environment, so it must run + # after RUSTFLAGS is set: if x64 target-cpu were only selected later, cargo + # would invalidate the restored target/ while rust-cache saw an exact key + # and refused to save the rebuilt artifacts (macOS x64 baseline rebuilds + # forever). The native source hash is in the shared key too: rust-cache's + # lockfile scan misses the workspace-root Cargo.toml version Cargo + # fingerprints, so a version bump could otherwise get an exact hit for + # artifacts Cargo must rebuild. On omp-kata the mounted Cargo registry + + # RustFS sccache are the reuse layers, so there is no second cache restore. + - name: Cache Rust target/ (GitHub-hosted) + if: steps.detect.outputs.on_infra == 'false' + uses: Swatinem/rust-cache@v2 with: shared-key: native-${{ inputs.platform }}-${{ inputs.arch }}-${{ inputs.variant || 'default' }}-h${{ inputs.hash }} cache-on-failure: true save-if: ${{ inputs.save_cache == 'true' }} cache-workspace-crates: true - - name: Setup sccache + + # --- sccache ------------------------------------------------------------ + - name: Ensure baked sccache (omp-kata) + if: steps.detect.outputs.on_infra == 'true' + uses: ./.github/actions/ensure-sccache + with: + version: "0.15.0" + - name: Setup sccache (GitHub-hosted) + if: steps.detect.outputs.on_infra == 'false' uses: mozilla-actions/sccache-action@v0.0.10 - name: Enable sccache for cargo - # `CARGO_INCREMENTAL=0` is required: sccache silently skips caching when - # incremental is enabled, turning the wrapper into a no-op. The backend - # is conditional: self-hosted omp-kata runners inject a shared S3 (RustFS) - # cache via pod env (SCCACHE_BUCKET/ENDPOINT/REGION + AWS creds) that - # sccache reads from the inherited environment; GitHub-hosted runners - # (macOS, ubuntu-arm) can't reach the private RustFS and keep the GHA - # cache backend. + # CARGO_INCREMENTAL=0 is required: sccache silently skips caching when + # incremental is enabled, turning the wrapper into a no-op. The backend is + # conditional: omp-kata reads the shared S3 (RustFS) config from the + # inherited pod env; GitHub-hosted runners use the GHA cache backend. shell: bash run: | { @@ -135,33 +164,64 @@ runs: echo "SCCACHE_GHA_ENABLED=true" >> "$GITHUB_ENV" echo "sccache backend: GitHub Actions cache" fi - - uses: taiki-e/install-action@v2 - if: inputs.target == '' + + # --- cargo-nextest (native test runner; non-cross builds only) ---------- + - name: Ensure baked cargo-nextest (omp-kata) + if: steps.detect.outputs.on_infra == 'true' && inputs.target == '' + uses: ./.github/actions/ensure-cargo-tool + with: + binary: cargo-nextest + crate: cargo-nextest + - name: Install cargo-nextest (GitHub-hosted) + if: steps.detect.outputs.on_infra == 'false' && inputs.target == '' + uses: taiki-e/install-action@v2 with: tool: nextest + - uses: ./.github/actions/bun-install - # Cross-compile toolchain selection: non-MSVC targets (e.g. - # `aarch64-unknown-linux-gnu`) build with `cargo-zigbuild`; MSVC targets - # (e.g. `x86_64-pc-windows-msvc`) build with `cargo-xwin`. The napi CLI's - # `--cross-compile` flag picks the backend; we just install what it needs. - - name: Setup zig (non-MSVC cross-compile) - if: inputs.target != '' && !endsWith(inputs.target, '-msvc') + + # --- Cross-compile toolchains ------------------------------------------- + # Non-MSVC targets (e.g. aarch64-unknown-linux-gnu) build with + # cargo-zigbuild (needs zig); MSVC targets (e.g. x86_64-pc-windows-msvc) + # build with cargo-xwin (needs clang/lld/llvm). The napi CLI's + # --cross-compile flag picks the backend; we just install what it needs. + # Cross builds only run on push/main + release (omp-kata), so the + # GitHub-hosted cross branches exist for portability and never fire here. + - name: Ensure baked zig (omp-kata, non-MSVC cross) + if: steps.detect.outputs.on_infra == 'true' && inputs.target != '' && !endsWith(inputs.target, '-msvc') + uses: ./.github/actions/ensure-zig + with: + version: "0.16.0" + - name: Setup zig (GitHub-hosted, non-MSVC cross) + if: steps.detect.outputs.on_infra == 'false' && inputs.target != '' && !endsWith(inputs.target, '-msvc') uses: mlugg/setup-zig@v2 with: version: 0.16.0 - - name: Install cargo-zigbuild (non-MSVC cross-compile) - if: inputs.target != '' && !endsWith(inputs.target, '-msvc') + - name: Ensure baked cargo-zigbuild (omp-kata, non-MSVC cross) + if: steps.detect.outputs.on_infra == 'true' && inputs.target != '' && !endsWith(inputs.target, '-msvc') + uses: ./.github/actions/ensure-cargo-tool + with: + binary: cargo-zigbuild + crate: cargo-zigbuild + - name: Install cargo-zigbuild (GitHub-hosted, non-MSVC cross) + if: steps.detect.outputs.on_infra == 'false' && inputs.target != '' && !endsWith(inputs.target, '-msvc') uses: taiki-e/install-action@v2 with: tool: cargo-zigbuild - - name: Install LLVM tooling (MSVC cross-compile) - if: endsWith(inputs.target, '-msvc') + - name: Install LLVM tooling (GitHub-hosted, MSVC cross) + if: steps.detect.outputs.on_infra == 'false' && endsWith(inputs.target, '-msvc') shell: bash run: | sudo apt-get update sudo apt-get install -y clang lld llvm - - name: Install cargo-xwin (MSVC cross-compile) - if: endsWith(inputs.target, '-msvc') + - name: Ensure baked cargo-xwin (omp-kata, MSVC cross) + if: steps.detect.outputs.on_infra == 'true' && endsWith(inputs.target, '-msvc') + uses: ./.github/actions/ensure-cargo-tool + with: + binary: cargo-xwin + crate: cargo-xwin + - name: Install cargo-xwin (GitHub-hosted, MSVC cross) + if: steps.detect.outputs.on_infra == 'false' && endsWith(inputs.target, '-msvc') uses: taiki-e/install-action@v2 with: tool: cargo-xwin @@ -175,15 +235,17 @@ runs: if: endsWith(inputs.target, '-msvc') shell: bash run: echo "XWIN_ACCEPT_LICENSE=1" >> "$GITHUB_ENV" + + # --- Checks, build, upload (shared) ------------------------------------- - name: Rust checks if: inputs.rust_checks == 'true' shell: bash run: bun run check:rs - name: Test workspace (Rust) - # macOS has no `#[cfg(target_os = "macos")]` tests in the workspace, - # and Windows-only tests are no longer exercised in CI (`win32-x64` - # cross-builds on Linux). Skipping the duplicate Linux runs on macOS - # saves ~10 min of parallel runner time. + # macOS has no `#[cfg(target_os = "macos")]` tests in the workspace, and + # Windows-only tests are no longer exercised in CI (win32-x64 cross-builds + # on Linux). Skipping the duplicate Linux runs on macOS saves ~10 min of + # parallel runner time. if: inputs.target == '' && inputs.platform != 'darwin' shell: bash run: bun run test:rs @@ -201,7 +263,7 @@ runs: name: pi-natives-${{ inputs.platform }}-${{ inputs.arch }}${{ inputs.variant && format('-{0}', inputs.variant) || '' }}-h${{ inputs.hash }} path: packages/natives/native/pi_natives.${{ inputs.platform }}-${{ inputs.arch }}*.node if-no-files-found: error - # Explicit so the native_artifact_lookup canary keeps working even if org - # defaults shift; bump if Rust source ever stays stable for >90 days + # Explicit so the native_artifact_lookup canary keeps working even if + # org defaults shift; bump if Rust source ever stays stable for >90 days # of main pushes and you want to avoid rebuilds. retention-days: 90 diff --git a/.github/actions/bun-install/action.yml b/.github/actions/bun-install/action.yml index 4ccfd6a9f..5ae10831d 100644 --- a/.github/actions/bun-install/action.yml +++ b/.github/actions/bun-install/action.yml @@ -1,10 +1,10 @@ name: "bun install (shared cache)" description: > Ensure bun is on PATH, then run `bun install --frozen-lockfile` with a shared - dependency cache. bun setup is skipped when the runner image already ships it + package store. bun setup is skipped when the runner image already ships it (the preloaded omp-kata image), and only fetched on runners that lack it - (e.g. GitHub-hosted). The cache uses the in-cluster RustFS S3 when the sccache - credentials are present, and the stock actions/cache backend otherwise. + (e.g. GitHub-hosted). Self-hosted omp-kata runners use the mounted Bun store + PVC; GitHub-hosted runners use the stock actions/cache backend. runs: using: composite @@ -24,10 +24,11 @@ runs: fi # The repo keys "are we on can.internal infra?" off $SCCACHE_BUCKET (see # actions/build-native). RUNNER_ENVIRONMENT is empty on ARC pods, so it - # is not a usable signal here. - if [ -n "${SCCACHE_BUCKET:-}" ] && [ -n "${AWS_ACCESS_KEY_ID:-}" ]; then - echo "cache=rustfs" >> "$GITHUB_OUTPUT" - echo "bun cache backend: RustFS S3 ($SCCACHE_BUCKET @ $SCCACHE_ENDPOINT)" + # is not a usable signal here. On infra, the ARC pod mounts the shared + # Bun store at the default cache path; off infra, actions/cache restores it. + if [ -n "${SCCACHE_BUCKET:-}" ]; then + echo "cache=mounted" >> "$GITHUB_OUTPUT" + echo "bun cache backend: mounted PVC (${BUN_INSTALL_CACHE_DIR:-${HOME}/.bun/install/cache})" else echo "cache=gha" >> "$GITHUB_OUTPUT" echo "bun cache backend: GitHub Actions cache" @@ -48,17 +49,13 @@ runs: path: ~/.bun/install/cache key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }} - # On-infra (omp-kata): RustFS store + node_modules over the LAN. - - name: Restore bun caches (RustFS) - if: steps.env.outputs.cache == 'rustfs' + # On-infra (omp-kata): the pod mounts a shared PVC at bun's store path. + - name: Prepare mounted bun store + if: steps.env.outputs.cache == 'mounted' shell: bash - run: bash "$GITHUB_ACTION_PATH/rustfs-cache.sh" restore + run: mkdir -p "${BUN_INSTALL_CACHE_DIR:-${HOME}/.bun/install/cache}" - name: Install dependencies shell: bash run: bun install --frozen-lockfile - - name: Save bun caches (RustFS) - if: steps.env.outputs.cache == 'rustfs' - shell: bash - run: bash "$GITHUB_ACTION_PATH/rustfs-cache.sh" save diff --git a/.github/actions/bun-install/rustfs-cache.sh b/.github/actions/bun-install/rustfs-cache.sh deleted file mode 100755 index 2cfbc3d16..000000000 --- a/.github/actions/bun-install/rustfs-cache.sh +++ /dev/null @@ -1,126 +0,0 @@ -#!/usr/bin/env bash -# Shared bun dependency cache backed by the in-cluster RustFS (S3) object store. -# -# Used by .github/actions/bun-install on the self-hosted omp-kata runners. There -# the stock `actions/cache` restore of ~/.bun/install/cache costs 130-186s per -# job because GitHub's cache backend is only reachable over the node's NAT -# egress, and ~9 jobs contend on it at once. RustFS lives in the same k3s node -# (svc :9000, already allowed by the runner egress NetworkPolicy), so the same -# payload moves at LAN speed. -# -# Credentials are the ones sccache already gets via the `sccache-s3` secret -# (envFrom on every runner pod): AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY / -# SCCACHE_ENDPOINT / SCCACHE_BUCKET / SCCACHE_REGION / SCCACHE_S3_USE_SSL. -# -# Two objects per lockfile, under the bun-cache/ key prefix of the sccache -# bucket: -# store-- the bun global package store (~/.bun/install/cache) -# nm-- the installed node_modules trees (root + workspaces) -# The store additionally publishes a rolling store--latest alias, so a -# changed lockfile still warm-starts from the previous store and `bun install` -# only fetches the delta. A node_modules hit short-circuits everything: the -# subsequent `bun install --frozen-lockfile` is a no-op, so the store is neither -# fetched nor saved. -set -euo pipefail - -mode="${1:?usage: rustfs-cache.sh restore|save}" - -: "${SCCACHE_BUCKET:?SCCACHE_BUCKET required}" -: "${SCCACHE_ENDPOINT:?SCCACHE_ENDPOINT required}" -: "${AWS_ACCESS_KEY_ID:?AWS_ACCESS_KEY_ID required}" -: "${AWS_SECRET_ACCESS_KEY:?AWS_SECRET_ACCESS_KEY required}" - -region="${SCCACHE_REGION:-us-east-1}" -if [ "${SCCACHE_S3_USE_SSL:-false}" = "true" ]; then scheme=https; else scheme=http; fi -base="${scheme}://${SCCACHE_ENDPOINT}/${SCCACHE_BUCKET}/bun-cache" -os="${RUNNER_OS:-$(uname -s)}" -store_dir="${BUN_INSTALL_CACHE_DIR:-${HOME}/.bun/install/cache}" -work="${RUNNER_TEMP:-/tmp}/bun-rustfs-cache" -mkdir -p "$work" - -# Prefer multi-threaded zstd (baked into the omp-kata runner image); fall back to -# gzip so the action still works on an image that predates the zstd addition. The -# object suffix records the codec, and restore only inflates archives this host -# can actually decompress. -if command -v zstd >/dev/null 2>&1; then - tar_c=(-I "zstd -3 -T0"); ext="tzst"; alt_ext="tgz" -else - tar_c=(-I "gzip -6"); ext="tgz"; alt_ext="tzst" -fi - -lock_hash="$(sha256sum bun.lock | cut -c1-32)" -store_key="store-${os}-${lock_hash}" -store_latest="store-${os}-latest" -nm_key="nm-${os}-${lock_hash}" - -auth=(--aws-sigv4 "aws:amz:${region}:s3" --user "${AWS_ACCESS_KEY_ID}:${AWS_SECRET_ACCESS_KEY}") -# 404 (-f) and connection errors are non-zero; transient errors retry, 4xx do not. -s3_get() { curl -fsS --retry 3 --retry-connrefused "${auth[@]}" "${base}/$1" -o "$2"; } -s3_exists() { curl -fsS -I --retry 3 --retry-connrefused "${auth[@]}" "${base}/$1" -o /dev/null >/dev/null 2>&1; } -s3_put() { curl -fsS --retry 3 --retry-connrefused "${auth[@]}" -T "$2" "${base}/$1" -o /dev/null; } - -# Download . (then the alternate codec) and extract into dir $2. -# tar auto-detects the codec from the archive; a present-but-uninflatable archive -# (codec mismatch with this host) is treated as a miss. -fetch_extract() { # name dest - local name="$1" dest="$2" e f - for e in "$ext" "$alt_ext"; do - f="${work}/${name}.${e}" - if s3_get "${name}.${e}" "$f" 2>/dev/null; then - mkdir -p "$dest" - if tar -xf "$f" -C "$dest" 2>/dev/null; then rm -f "$f"; return 0; fi - rm -f "$f" - fi - done - return 1 -} - -case "$mode" in -restore) - if fetch_extract "$nm_key" "$PWD"; then - echo "bun cache: node_modules HIT ($nm_key) — install becomes a no-op" - : > "${work}/nm_hit" - exit 0 - fi - echo "bun cache: node_modules miss ($nm_key)" - if fetch_extract "$store_key" "$store_dir"; then - echo "bun cache: store HIT ($store_key)" - elif fetch_extract "$store_latest" "$store_dir"; then - echo "bun cache: store warm-start ($store_latest)" - else - echo "bun cache: store miss — cold install" - fi - ;; -save) - if [ -f "${work}/nm_hit" ]; then - echo "bun cache: node_modules was a hit — nothing to save" - exit 0 - fi - # Store (+ rolling latest): save when this exact lockfile has none yet. - if [ -d "$store_dir" ] && ! s3_exists "${store_key}.${ext}"; then - tar "${tar_c[@]}" -cf "${work}/store.${ext}" -C "$store_dir" . - s3_put "${store_key}.${ext}" "${work}/store.${ext}" - s3_put "${store_latest}.${ext}" "${work}/store.${ext}" - rm -f "${work}/store.${ext}" - echo "bun cache: saved store ($store_key + $store_latest)" - fi - # node_modules: save the installed trees for this exact lockfile. - if ! s3_exists "${nm_key}.${ext}"; then - shopt -s nullglob - nm_paths=() - for p in node_modules packages/*/node_modules python/robomp/web/node_modules; do - [ -d "$p" ] && nm_paths+=("$p") - done - if [ ${#nm_paths[@]} -gt 0 ]; then - tar "${tar_c[@]}" -cf "${work}/nm.${ext}" "${nm_paths[@]}" - s3_put "${nm_key}.${ext}" "${work}/nm.${ext}" - rm -f "${work}/nm.${ext}" - echo "bun cache: saved node_modules ($nm_key, ${#nm_paths[@]} trees)" - fi - fi - ;; -*) - echo "rustfs-cache.sh: unknown mode '$mode' (want restore|save)" >&2 - exit 2 - ;; -esac diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 1a7f56b1a..a99cd7cb6 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -44,7 +44,7 @@ jobs: # ref (or from a tagged main HEAD) is also treated as a release. release_metadata: name: Resolve release metadata - runs-on: omp-kata + runs-on: ${{ github.event_name == 'pull_request' && 'ubuntu-22.04' || 'omp-kata' }} outputs: is-release: ${{ steps.detect.outputs.is-release }} release-tag: ${{ steps.detect.outputs.release-tag }} @@ -96,7 +96,7 @@ jobs: # retention window (see build-native action) is the effective TTL. native_artifact_lookup: name: Look up cached native artifacts - runs-on: omp-kata + runs-on: ${{ github.event_name == 'pull_request' && 'ubuntu-22.04' || 'omp-kata' }} outputs: source-hash: ${{ steps.compute.outputs.source-hash }} linux-x64-run-id: ${{ steps.find.outputs.linux-x64-run-id }} @@ -187,7 +187,7 @@ jobs: # Fast lint, type check, and browser bundle build (no Rust, no native build needed) check: name: Lint, type check & web build - runs-on: omp-kata + runs-on: ${{ github.event_name == 'pull_request' && 'ubuntu-22.04' || 'omp-kata' }} steps: - uses: actions/checkout@v4 - uses: ./.github/actions/bun-install @@ -203,7 +203,7 @@ jobs: name: "Native: Linux x64 (${{ matrix.variant }})" needs: [release_metadata, native_artifact_lookup] if: ${{ needs.release_metadata.outputs.is-release == 'true' || needs.native_artifact_lookup.outputs.linux-x64-run-id == '' }} - runs-on: omp-kata + runs-on: ${{ github.event_name == 'pull_request' && 'ubuntu-22.04' || 'omp-kata' }} strategy: fail-fast: false matrix: @@ -212,7 +212,7 @@ jobs: - { variant: modern } steps: - uses: actions/checkout@v4 - - uses: ./.github/actions/build-native-kata + - uses: ./.github/actions/build-native with: hash: ${{ needs.native_artifact_lookup.outputs.source-hash }} platform: linux @@ -237,7 +237,7 @@ jobs: runs-on: ${{ matrix.os }} steps: - uses: actions/checkout@v4 - - uses: ./.github/actions/build-native-kata + - uses: ./.github/actions/build-native with: hash: ${{ needs.native_artifact_lookup.outputs.source-hash }} platform: ${{ matrix.platform }} @@ -269,12 +269,13 @@ jobs: save_cache: ${{ github.event_name == 'push' && github.ref == 'refs/heads/main' }} test_workspace: name: Test TS workspace fast - runs-on: omp-kata + runs-on: ${{ github.event_name == 'pull_request' && 'ubuntu-22.04' || 'omp-kata' }} needs: [native_linux_x64, native_artifact_lookup] if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }} timeout-minutes: 20 steps: - uses: actions/checkout@v4 + - uses: ./.github/actions/setup-system-deps - uses: ./.github/actions/bun-install - name: Resolve Linux x64 native artifact run id: source @@ -298,12 +299,13 @@ jobs: test_coding_agent_singleton: name: Test coding-agent singleton/global-state (TS) - runs-on: omp-kata + runs-on: ${{ github.event_name == 'pull_request' && 'ubuntu-22.04' || 'omp-kata' }} needs: [native_linux_x64, native_artifact_lookup] if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }} timeout-minutes: 20 steps: - uses: actions/checkout@v4 + - uses: ./.github/actions/setup-system-deps - uses: ./.github/actions/bun-install - name: Resolve Linux x64 native artifact run id: source @@ -329,7 +331,7 @@ jobs: test_ts_native: name: Test TS native/integration packages - runs-on: omp-kata + runs-on: ${{ github.event_name == 'pull_request' && 'ubuntu-22.04' || 'omp-kata' }} needs: [native_linux_x64, native_artifact_lookup] if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }} timeout-minutes: 25 @@ -359,7 +361,7 @@ jobs: test_coding_agent_ui: name: Test coding-agent UI/TUI (TS) - runs-on: omp-kata + runs-on: ${{ github.event_name == 'pull_request' && 'ubuntu-22.04' || 'omp-kata' }} needs: [native_linux_x64, native_artifact_lookup] if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }} timeout-minutes: 25 @@ -389,12 +391,13 @@ jobs: test_coding_agent_runtime: name: Test coding-agent runtime/session (TS) - runs-on: omp-kata + runs-on: ${{ github.event_name == 'pull_request' && 'ubuntu-22.04' || 'omp-kata' }} needs: [native_linux_x64, native_artifact_lookup] if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }} timeout-minutes: 25 steps: - uses: actions/checkout@v4 + - uses: ./.github/actions/setup-system-deps - uses: ./.github/actions/bun-install - name: Resolve Linux x64 native artifact run id: source @@ -420,7 +423,7 @@ jobs: test_coding_agent_native: name: Test coding-agent native/unit (TS) - runs-on: omp-kata + runs-on: ${{ github.event_name == 'pull_request' && 'ubuntu-22.04' || 'omp-kata' }} needs: [native_linux_x64, native_artifact_lookup] if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }} timeout-minutes: 25 @@ -450,7 +453,7 @@ jobs: test_smoke: name: Test CLI smoke (TS) - runs-on: omp-kata + runs-on: ${{ github.event_name == 'pull_request' && 'ubuntu-22.04' || 'omp-kata' }} needs: [native_linux_x64, native_artifact_lookup] if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }} timeout-minutes: 15 @@ -480,7 +483,7 @@ jobs: install_methods: name: Install method smoke tests - runs-on: omp-kata + runs-on: ${{ github.event_name == 'pull_request' && 'ubuntu-22.04' || 'omp-kata' }} steps: - uses: actions/checkout@v4 - uses: ./.github/actions/ensure-rust-toolchain diff --git a/infra/docs/02-kata-runtime.md b/infra/docs/02-kata-runtime.md index bbc716eb5..069d600f4 100644 --- a/infra/docs/02-kata-runtime.md +++ b/infra/docs/02-kata-runtime.md @@ -337,9 +337,9 @@ This is why `disable_block_device_use = true`: the rootfs travels in over the shared filesystem, not as a block device. Benefits: no image-to-block conversion, near-instant rootfs availability, and host/guest can both see the files. `virtio_fs_cache = "auto"` keeps the conservative page-cache behavior, but -the active worker pool is now `--thread-pool-size=4` rather than `1` so the -metadata-heavy Bun cache restore / extract path has a few host workers to fan out -across. `--announce-submounts` keeps nested mounts visible to the guest. +the active worker pool is now `--thread-pool-size=4` rather than `1` so +metadata-heavy mounted-cache and dependency-install paths have a few host workers +to fan out across. `--announce-submounts` keeps nested mounts visible to the guest. `emptydir_mode = "shared-fs"` extends the same mechanism to Kubernetes `emptyDir` volumes — they are shared into the guest over virtio-fs instead of diff --git a/infra/docs/03-runner-image.md b/infra/docs/03-runner-image.md index 280c22994..d2b6d3531 100644 --- a/infra/docs/03-runner-image.md +++ b/infra/docs/03-runner-image.md @@ -9,7 +9,7 @@ importing it into k3s containerd, pointing ARC at it, and verifying it. Navigation: previous - [02-kata-runtime.md](./02-kata-runtime.md) (Kata + containerd + RuntimeClass) | next - [04-arc-and-caching.md](./04-arc-and-caching.md) -(ARC runners, RustFS cache, egress policy). +(ARC runners, shared caches, egress policy). > The image built here is referenced by the ARC runner pod template > (`template.spec.containers[0].image` + `imagePullPolicy: IfNotPresent`), @@ -105,7 +105,7 @@ RUN curl -fsSL https://cli.github.com/packages/githubcli-archive-keyring.gpg -o && echo "deb [arch=$(dpkg --print-architecture) signed-by=/usr/share/keyrings/githubcli-archive-keyring.gpg] https://cli.github.com/packages stable main" > /etc/apt/sources.list.d/github-cli.list \ && apt-get update \ && apt-get install -y \ - build-essential pkg-config curl ca-certificates git unzip xz-utils zstd gh \ + build-essential pkg-config curl ca-certificates git unzip xz-utils gh \ clang lld llvm \ libcairo2-dev libpango1.0-dev libjpeg-dev libgif-dev librsvg2-dev \ fd-find ripgrep imagemagick \ @@ -176,11 +176,11 @@ In order: tool. - `apt-get install` pulls three groups: - **build toolchain / utilities:** `build-essential pkg-config curl - ca-certificates git unzip xz-utils zstd gh clang lld llvm`. + ca-certificates git unzip xz-utils gh clang lld llvm`. `build-essential` + `pkg-config` are needed by the native and canvas builds; - `zstd` is the codec the bun and sccache cache tarballs use (see - [04-arc-and-caching.md](./04-arc-and-caching.md)); `clang lld llvm` are the - MSVC-cross prerequisites that used to be apt-installed per job. + `gh` is used by release workflows and the coding-agent GitHub tool; `clang + lld llvm` are the MSVC-cross prerequisites that used to be apt-installed per + job. - **canvas / cairo native stack:** `libcairo2-dev libpango1.0-dev libjpeg-dev libgif-dev librsvg2-dev` - the `-dev` headers the canvas/rsvg native modules compile against. @@ -252,7 +252,7 @@ DOCKER_BUILDKIT=1 docker build -t "$IMAGE" -t omp-kata-runner:preloaded . echo "==> [2/5] verifying baked tools" docker run --rm --entrypoint bash "$IMAGE" -lc ' set -e - for b in gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do + for b in gh fd rg magick bun cargo rustc pkg-config clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do command -v "$b" >/dev/null || { echo "MISSING: $b"; exit 1; } done echo "tools OK | bun $(bun --version) | rust $(rustc --version) | sccache $(sccache --version | awk '\''{print $2}'\'') | zig $(zig version) | gh $(gh --version | head -1 | cut -d\" \" -f3)" @@ -294,7 +294,7 @@ BuildKit + the docker layer cache make an unchanged rebuild near-instant. **[2/5] verify baked tools.** Runs the freshly built image with a bash entrypoint and asserts every expected binary is on `PATH` -(`gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin`), +(`gh fd rg magick bun cargo rustc pkg-config clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin`), failing the whole script if any is missing, then prints the key version tuple (bun / rust / sccache / zig / gh). This catches a broken apt set, missing shim, or bad toolchain pin **before** anything touches the cluster. @@ -416,7 +416,7 @@ standalone: ```bash docker run --rm --entrypoint bash omp-kata-runner:preloaded -lc ' set -e - for b in gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do + for b in gh fd rg magick bun cargo rustc pkg-config clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do command -v "$b" >/dev/null || { echo "MISSING: $b"; exit 1; } done echo "tools OK | bun $(bun --version) | rust $(rustc --version) | sccache $(set -- $(sccache --version); echo "$2") | zig $(zig version) | gh $(set -- $(gh --version | head -1); echo "$3")" @@ -462,5 +462,5 @@ afterward as shown. --- Continue to [04-arc-and-caching.md](./04-arc-and-caching.md) for how ARC -references this image in the runner pod template, wires in the RustFS shared -cache, and locks down runner egress. +wires in the shared `sccache`/Bun/Cargo cache storage, and locks down runner +egress. diff --git a/infra/docs/04-arc-and-caching.md b/infra/docs/04-arc-and-caching.md index d1ca5dfa8..955ce674f 100644 --- a/infra/docs/04-arc-and-caching.md +++ b/infra/docs/04-arc-and-caching.md @@ -1,12 +1,13 @@ -# 04 - ARC runners, shared cache, and egress policy +# 04 - ARC runners, shared caches, and egress policy This is the last setup step. By now the node runs k3s with the `kata-qemu` RuntimeClass ([02-kata-runtime.md](02-kata-runtime.md)) and the preloaded runner image has been imported into the cluster containerd ([03-runner-image.md](03-runner-image.md)). Here we install **actions-runner-controller (ARC)**, register an ephemeral **scale set** whose pods each boot inside their own Kata microVM, stand up the -in-cluster **RustFS (S3)** shared cache, and lock down runner egress with a -NetworkPolicy. See [README.md](README.md) for the architecture overview. +in-cluster **RustFS (S3)** `sccache` backend and the runner cache PVC, and lock +down runner egress with a NetworkPolicy. See [README.md](README.md) for the +architecture overview. Everything below is read against the live cluster; set the kubeconfig once: @@ -95,8 +96,7 @@ helm install arc \ oci://ghcr.io/actions/actions-runner-controller-charts/gha-runner-scale-set-controller ``` -**Scale set** (`omp-kata`), using the values file from step 3: - +**Scale set** (`omp-kata`), using the runner cache PVC and values file from step 3: ```bash helm install omp-kata \ --namespace arc-runners --create-namespace \ @@ -129,6 +129,30 @@ kubectl -n arc-systems get pods ## 3. Scale-set values (`arc-omp-values.yaml`) +Create the namespace-local PVC before installing or upgrading the scale set. This +is the shared mutable filesystem cache for data whose tools already validate +against the lockfile: Bun's global package store and Cargo's registry cache. + +```yaml +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: + name: runner-cache + namespace: arc-runners +spec: + accessModes: ["ReadWriteOnce"] + storageClassName: local-path + resources: + requests: + storage: 100Gi +``` + +Apply it once: + +```bash +kubectl apply -f runner-cache-pvc.yaml +``` + This is the live `arc-omp-values.yaml` verbatim, with only the repo owner/name in `githubConfigUrl` redacted: @@ -144,6 +168,25 @@ containerMode: template: spec: runtimeClassName: kata-qemu # <-- every runner pod boots its own KVM microVM + securityContext: + # ghcr.io/actions/actions-runner runs jobs as uid/gid 1001 ("runner"). + # Let kubelet make the PVC writable by that user without changing image-owned + # ~/.cargo/bin or ~/.rustup. + fsGroup: 1001 + fsGroupChangePolicy: OnRootMismatch + initContainers: + - name: prepare-runner-cache + image: omp-kata-runner:2026-06-15-002621 + imagePullPolicy: IfNotPresent + command: + - bash + - -lc + - install -d -o 1001 -g 1001 -m 2775 /cache/bun-store /cache/cargo-registry + securityContext: + runAsUser: 0 + volumeMounts: + - name: runner-cache + mountPath: /cache containers: - name: runner # Preloaded image: stock ghcr.io/actions/actions-runner + CI deps baked in @@ -160,6 +203,15 @@ template: envFrom: - secretRef: name: sccache-s3 + volumeMounts: + # Shared stores only. Keep node_modules, Cargo target/, and Cargo git + # checkouts per-job to avoid mutable build-output or checkout poisoning. + - name: runner-cache + mountPath: /home/runner/.bun/install/cache + subPath: bun-store + - name: runner-cache + mountPath: /home/runner/.cargo/registry + subPath: cargo-registry resources: requests: cpu: "2" @@ -167,6 +219,10 @@ template: limits: cpu: "8" memory: "12Gi" + volumes: + - name: runner-cache + persistentVolumeClaim: + claimName: runner-cache ``` Field by field: @@ -194,8 +250,22 @@ Field by field: here when you rebuild the image (see [Operate](#7-operate)). - **`command: ["/home/runner/run.sh"]`** - the stock actions-runner entrypoint; overridden explicitly because the custom image keeps the upstream layout. -- **`envFrom.secretRef.name: sccache-s3`** - injects the shared-cache S3 - configuration into every job's environment ([step 5](#5-shared-cache-rustfs-s3)). +- **`envFrom.secretRef.name: sccache-s3`** - injects only the S3 configuration that + `sccache` needs ([step 5](#5-shared-caches-rustfs-s3--runner-pvc)). Bun and + Cargo no longer use RustFS. +- **`securityContext.fsGroup: 1001`** - makes the mounted PVC writable by the + image's `runner` user without replacing image-owned `~/.cargo/bin` or `~/.rustup`. +- **`initContainers.prepare-runner-cache`** - uses the same locally imported image + to create the PVC subdirectories as root before the runner starts. This avoids + relying on kubelet's subPath auto-create permissions and does not pull another + image. +- **`volumeMounts`** - mounts the shared PVC only at `~/.bun/install/cache` and + `~/.cargo/registry`. `node_modules`, Cargo `target/`, and Cargo git checkouts + stay inside the throwaway VM filesystem. +- **`volumes[].persistentVolumeClaim.claimName: runner-cache`** - binds those + mounts to the `arc-runners/runner-cache` PVC. `ReadWriteOnce` is enough on this + single-node k3s host; use a RWX-capable storage class before spreading runners + across nodes. - **`resources`** - requests `2` CPU / `4Gi`, limits `8` CPU / `12Gi`. Kata reads these and sizes the guest accordingly: the VM now boots at the same guaranteed floor (`default_vcpus: 2`, `default_memory: 4096`) and only @@ -248,12 +318,15 @@ a compromised job is boxed into a throwaway VM with no cluster reach. --- -## 5. Shared cache (RustFS S3) +## 5. Shared caches (RustFS S3 + runner PVC) GitHub's hosted cache backend is only reachable over the node's NAT egress, so on -a busy matrix (many concurrent jobs) it becomes the bottleneck. Instead an -**S3-compatible object store, RustFS, runs inside the cluster** and serves the -cache at LAN speed over `rustfs.sccache.svc.cluster.local:9000`. +a busy matrix (many concurrent jobs) it becomes the bottleneck. This setup keeps +the hot paths inside the cluster: + +- **RustFS S3** backs `sccache` for Rust compiler outputs. +- **`runner-cache` PVC** is mounted into every runner for Bun's global package + store and Cargo's crates.io registry cache. ### 5a. Deploy RustFS @@ -379,7 +452,7 @@ pointed at the endpoint with the root creds): `mb s3://sccache`. ### 5b. The `sccache-s3` secret (injected into every runner) -Every runner pod gets the cache configuration via `envFrom` ([step 3](#3-scale-set-values-arc-omp-valuesyaml)). +Every runner pod gets the `sccache` S3 configuration via `envFrom` ([step 3](#3-scale-set-values-arc-omp-valuesyaml)). The secret lives in `arc-runners` (the runners' namespace) and has six keys: ```bash @@ -413,85 +486,95 @@ kubectl -n arc-runners create secret generic sccache-s3 \ - `SCCACHE_REGION: us-east-1` - arbitrary region label SigV4 requires. - `SCCACHE_S3_USE_SSL: false` - the endpoint is plain HTTP on the cluster network. -### 5c. The two consumers +### 5c. The cache consumers The presence of `$SCCACHE_BUCKET` in the environment is the repo's single signal -for "am I on the self-hosted infra?". Both consumers branch on it and fall back -to GitHub-hosted cache backends off-infra (GitHub-hosted macOS/arm runners never -get the secret and so cannot reach the private RustFS). +for "am I on the self-hosted infra?". Cache behavior branches on it and falls +back to GitHub-hosted cache backends off-infra. Off-infra covers GitHub-hosted +macOS/arm runners (which never get the secret and cannot reach the private +RustFS) and **every pull request**: `ci.yml` pins PR jobs to GitHub-hosted +`ubuntu-22.04`, so omp-kata only runs trusted `push`/main + release builds (see +[5d](#5d-poisoning-boundary-and-pressure)). -**(a) sccache for Rust** - [`.github/actions/build-native`](../../.github/actions/build-native/action.yml). -It installs `sccache`, then sets `RUSTC_WRAPPER=sccache` and `CARGO_INCREMENTAL=0` -(sccache silently no-ops with incremental enabled). The backend is conditional: +**(a) sccache for Rust compiler outputs** - +[`.github/actions/build-native`](../../.github/actions/build-native/action.yml). +One action serves both environments: a "Detect runner environment" step reads +`$SCCACHE_BUCKET` and branches each toolchain/cache step on it. It sets +`RUSTC_WRAPPER=sccache` and `CARGO_INCREMENTAL=0` (sccache silently no-ops with +incremental enabled). The sccache backend is conditional: -- `$SCCACHE_BUCKET` set - sccache reads `SCCACHE_BUCKET/ENDPOINT/REGION` and the - AWS creds straight from the inherited pod env and uses the **shared S3 (RustFS)**. -- otherwise - it exports `SCCACHE_GHA_ENABLED=true` and uses the **GitHub Actions - cache**. +- `$SCCACHE_BUCKET` set (omp-kata) - sccache reads `SCCACHE_BUCKET/ENDPOINT/REGION` + and the AWS creds straight from the inherited pod env and uses the **shared S3 + (RustFS)**; toolchains come from the baked image via the `ensure-*` actions. +- otherwise (GitHub-hosted) - it installs the toolchains, exports + `SCCACHE_GHA_ENABLED=true`, and uses the **GitHub Actions cache**. -`Swatinem/rust-cache` still caches `target/` on top; sccache fills the gaps when -`target/` is cold. +`Swatinem/rust-cache` runs only on GitHub-hosted runners (it caches Cargo +`target/`). On omp-kata the mounted Cargo registry handles crate downloads and +sccache fills the compile-output gap when `target/` is cold. -**(b) bun dependency cache** - [`.github/actions/bun-install`](../../.github/actions/bun-install/action.yml), -a composite action wrapping `bun install --frozen-lockfile`. A "Detect cache -backend" step checks `$SCCACHE_BUCKET` + `$AWS_ACCESS_KEY_ID`: +**(b) Cargo registry cache** - the scale-set pod template mounts +`runner-cache:/cargo-registry` at `/home/runner/.cargo/registry`. Cargo uses it +automatically because the image keeps `CARGO_HOME=/home/runner/.cargo`. -- on-infra - it runs `rustfs-cache.sh restore` before install and `... save` after; -- off-infra - it uses stock `actions/cache@v4` for the bun store. +Only the registry cache is shared. Cargo `target/` stays per-job, and +`/home/runner/.cargo/git` stays per-job too; this repo has no git dependencies, +and git checkouts are a worse shared mutable-cache boundary than crates.io +archives with lockfile checksums. -`rustfs-cache.sh` talks to RustFS directly with `curl --aws-sigv4` (S3 SigV4), no -SDK. It keys two objects per lockfile under the `bun-cache/` prefix of the -`sccache` bucket, derived from `sha256(bun.lock)`: +**(c) Bun package store** - +[`.github/actions/bun-install`](../../.github/actions/bun-install/action.yml) +wraps `bun install --frozen-lockfile`. On omp-kata, the pod template mounts +`runner-cache:/bun-store` at Bun's default store path +(`/home/runner/.bun/install/cache`), so the action only ensures the directory +exists before running Bun. Off-infra it still uses stock `actions/cache@v4` for +the same store path. -- `store--` - the bun global package store (`~/.bun/install/cache`), - plus a rolling `store--latest` alias so a changed lockfile still warm-starts - from the previous store and `bun install` fetches only the delta. -- `nm--` - the installed `node_modules` trees (repo root, every - `packages/*`, and `python/robomp/web`). +`node_modules` is deliberately not shared. It is lockfile-, platform-, script-, +and workspace-state-sensitive, and concurrent jobs would write through the same +tree. The clean VM still runs `bun install --frozen-lockfile`; it just reuses the +package tarball/extract store. -On **restore**, a `node_modules` hit short-circuits everything - the subsequent -`bun install --frozen-lockfile` is a no-op, so the store is neither fetched nor -saved. Archives are multi-threaded **zstd** (`.tzst`, baked into the runner image) -with a **gzip** (`.tgz`) fallback so the action still works on an older image; the -suffix records the codec and restore only inflates what the host can decompress. -On **save**, it writes the store/`node_modules` objects only when this exact -lockfile has none yet (`s3_exists` check), avoiding redundant uploads. +### 5d. Poisoning boundary and pressure -### 5d. Retention and PVC pressure +The shared writable PVC and the sccache S3 bucket are both poisonable by any job +that runs on `omp-kata`, and a poisoned entry could be consumed by a later +trusted build (a supply-chain risk). The primary defense is to **keep untrusted +code off the self-hosted runner entirely**: -The Bun cache intentionally keeps exact lockfile objects once written: +- `ci.yml` routes every pull-request job to GitHub-hosted `ubuntu-22.04` + (`runs-on` resolves to `omp-kata` only for `push`/main, manual dispatch, and + release). That expression lives in the base workflow, which GitHub uses + verbatim for `pull_request` events, so a fork cannot override it. Fork/PR code + therefore never sees the PVC, the `sccache-s3` creds, or RustFS - it runs + sandboxed on GitHub-hosted runners with the off-infra cache backends. +- As defense in depth, set the repo's **Settings -> Actions -> Fork pull request + workflows** policy to *Require approval for all outside collaborators* (or all + forks). GitHub's public-repo default only gates first-time contributors, which + would otherwise let a returning contributor's workflow start without review. -- `store--` and `nm--` are immutable warm caches; -- only `store--latest` is overwritten. +omp-kata thus only ever serves trusted `push`/main + release builds. The +mounted-cache design also narrows the blast radius of those trusted runs: -That means `bun.lock` churn will accumulate old exact objects on the `rustfs-data` -PVC. The repo ships [`infra/rustfs-cache-maintenance.sh`](../rustfs-cache-maintenance.sh) -to keep that under control from an ops checkout: +- no shared `node_modules`; +- no shared Cargo `target/`; +- no shared Cargo git checkouts; +- Bun still installs from `bun.lock`; +- Cargo registry entries are checked against Cargo's lockfile/source checksums; +- Rust compiler outputs stay in sccache's content-addressed backend. -```bash -CI_HOST= ./infra/rustfs-cache-maintenance.sh report -CI_HOST= ./infra/rustfs-cache-maintenance.sh prune -``` +Pressure now has two places to watch: -Its policy is deliberately conservative: +- `sccache/rustfs-data` for Rust compiler objects; +- `arc-runners/runner-cache` for Bun store + Cargo registry files. -- always keep every `store--latest` alias; -- always keep the newest exact `store-*` object that matches each `latest` alias - by ETag; -- always keep the newest `KEEP_EXACT_PER_OS` exact objects per OS (`3` by default) - for both `store-*` and `nm-*`; -- never delete exact objects newer than `MAX_AGE_DAYS` (`30` by default); -- once the PVC reaches `PRUNE_TRIGGER_PERCENT` (`80` by default), prune oldest - remaining exact objects toward `TARGET_PERCENT` (`70` by default), even if they - are newer than the age threshold. +For the runner PVC, the safe cleanup is simple and coarse: scale `omp-kata` to +zero, delete either `bun-store/` or `cargo-registry/` from the bound local-path +volume, then let the next jobs repopulate it. There are no RustFS Bun objects and +no `node_modules` archives to prune anymore. -The same script is the PVC-pressure alert: after `report` or `prune` it reads -`df -P /data` from the live RustFS pod and exits `1` at `WARN_PERCENT` (`80`) -and `2` at `CRITICAL_PERCENT` (`90`). Wire that into cron / systemd / your -monitoring runner; a failing exit is the signal that the PVC is too full. - -This caching is why RustFS sits inside the egress allow-list on `tcp/9000` -([step 6](#6-runner-egress-lockdown)). +RustFS remains inside the egress allow-list on `tcp/9000` because sccache still +uses it ([step 6](#6-runner-egress-lockdown)). --- @@ -570,7 +653,7 @@ The allow-list, rule by rule: service CIDR (`10.43.0.0/16`), so this rule alone gives a job **zero** in-cluster reach - rules 1 and 3 punch the only two holes the job legitimately needs. - **Rule 3 - RustFS cache.** TCP 9000 to the service CIDR (`10.43.0.0/16`) and the - `sccache` namespace - the shared cache from [step 5](#5-shared-cache-rustfs-s3). + `sccache` namespace - the sccache backend from [step 5](#5-shared-caches-rustfs-s3--runner-pvc). - **Ingress.** `policyTypes` lists `Ingress` but no ingress rule is defined, which is a **default-deny**: nothing can open a connection *into* a runner pod. @@ -621,10 +704,12 @@ kubectl -n arc-systems logs deploy/arc-gha-rs-controller -f kubectl -n arc-runners logs ``` -**Verify the cache is being used.** A warm job logs `bun cache: ... HIT` and +**Verify the caches are being used.** A warm job logs +`bun cache backend: mounted PVC (...)` and `sccache backend: shared S3 (sccache @ rustfs.sccache.svc.cluster.local:9000)` in -its step output. To inspect objects directly, point any S3 client at the endpoint -(with the RustFS root creds) and list `s3://sccache/bun-cache/`. +its step output. To inspect the mounted cache, scale to zero and check the +`runner-cache` local-path volume on the host; to inspect sccache objects, point +an S3 client at RustFS and list `s3://sccache/`. **Resize a job's VM** - edit the `resources` block in `arc-omp-values.yaml` ([step 3](#3-scale-set-values-arc-omp-valuesyaml); requests = guaranteed VM size, diff --git a/infra/docs/README.md b/infra/docs/README.md index d0ed0738d..b7994f729 100644 --- a/infra/docs/README.md +++ b/infra/docs/README.md @@ -1,6 +1,6 @@ # Self-hosted Kata CI -This is a self-hosted GitHub Actions setup where **every CI job runs inside its own throwaway Kata Containers QEMU/KVM microVM**. A single bare-metal Linux host runs a one-node [k3s](https://k3s.io) cluster; [actions-runner-controller (ARC)](https://github.com/actions/actions-runner-controller) watches GitHub for queued jobs and, for each one, creates a just-in-time ephemeral runner pod that boots a fresh microVM (its own guest kernel, isolated from the host), runs exactly one job, and is then destroyed. Runners share an in-cluster **RustFS** (S3-compatible) object store for `sccache` and Bun dependency caching, and egress the public internet through host NAT under a restrictive NetworkPolicy. The result is hardware-isolated, scale-to-zero CI on hardware you control. +This is a self-hosted GitHub Actions setup where **every CI job on the self-hosted `omp-kata` label runs inside its own throwaway Kata Containers QEMU/KVM microVM**. A single bare-metal Linux host runs a one-node [k3s](https://k3s.io) cluster; [actions-runner-controller (ARC)](https://github.com/actions/actions-runner-controller) watches GitHub for queued jobs and, for each one, creates a just-in-time ephemeral runner pod that boots a fresh microVM (its own guest kernel, isolated from the host), runs exactly one job, and is then destroyed. Pull requests deliberately run on GitHub-hosted runners instead, so the self-hosted fleet only serves trusted `push`/main + release builds and untrusted PR code never reaches the shared caches (see [04-arc-and-caching.md](04-arc-and-caching.md)). Runners share an in-cluster **RustFS** (S3-compatible) object store for `sccache`, plus a namespace-local PVC mounted as the Bun package store and Cargo registry cache. Public internet egress goes through host NAT under a restrictive NetworkPolicy. The result is hardware-isolated, scale-to-zero CI on hardware you control. These docs are written as a **from-scratch setup guide**: read this overview first, then follow the numbered guides in order to reproduce the system on your own host. @@ -24,8 +24,9 @@ flowchart LR JOB["actions/runner + one job's steps"] end end - subgraph CACHE["ns: sccache"] - RUSTFS["RustFS (S3)
svc rustfs:9000 · PVC rustfs-data 100Gi"] + subgraph CACHE["shared caches"] + RUSTFS["ns: sccache
RustFS (S3) svc rustfs:9000 · PVC rustfs-data 100Gi"] + PVC["ns: arc-runners
PVC runner-cache 100Gi
Bun store + Cargo registry"] end SEC["Secret sccache-s3
S3 creds + endpoint"] end @@ -37,7 +38,8 @@ flowchart LR POD --> VM SEC -.->|"envFrom"| POD NP -.->|"filters egress"| POD - JOB -->|"sccache + Bun cache (S3 SigV4)"| RUSTFS + JOB -->|"sccache (S3 SigV4)"| RUSTFS + JOB -->|"Bun store + Cargo registry mounts"| PVC POD -->|"allowed egress"| NAT NAT -->|"checkout / API / internet"| GH ``` @@ -48,7 +50,7 @@ Key properties baked into this design: - **Scale-to-zero.** `minRunners: 0` / `maxRunners: 10` — when no jobs are queued, zero runner pods (and zero microVMs) exist. - **Host-kernel isolation.** Jobs see the microVM's guest kernel, not the host kernel, so a kernel exploit in a job does not reach the host. - **No external registry.** The runner image is built on the host and imported straight into k3s' containerd. -- **Shared, in-cluster cache.** `sccache` and the Bun dependency cache both target RustFS over the cluster network; nothing cache-related leaves the host. +- **Shared, in-cluster cache.** `sccache` targets RustFS over the cluster network; Bun's package store and Cargo's registry cache are mounted from the runner cache PVC. Cache traffic stays on the host. ## End-to-end job lifecycle @@ -57,7 +59,7 @@ Key properties baked into this design: 3. The listener signals demand to the **ARC controller**, which scales the **EphemeralRunnerSet** up by one. 4. The controller creates a single **JIT-registered ephemeral runner pod** in `arc-runners`, with `runtimeClassName: kata-qemu` and the `sccache-s3` secret injected via `envFrom`. 5. containerd hands the pod to the Kata shim, which **boots a fresh QEMU/KVM microVM** (own guest kernel; the container rootfs is shared in over virtio-fs). No templating — every job gets a clean VM. -6. The runner agent inside the microVM **registers just-in-time and picks up exactly one job**. Steps run isolated from the host, using RustFS over S3 for `sccache`/Bun caching and NAT egress for the public internet, all constrained by the `runner-egress-lockdown` NetworkPolicy. +6. The runner agent inside the microVM **registers just-in-time and picks up exactly one job**. Steps run isolated from the host, using RustFS over S3 for `sccache`, mounted PVC paths for Bun/Cargo package caches, and NAT egress for the public internet, all constrained by the `runner-egress-lockdown` NetworkPolicy. 7. The job finishes; the ephemeral runner **deregisters and the pod (and its microVM) is destroyed** — never reused. 8. When no jobs remain queued, the EphemeralRunnerSet **scales back to zero**, leaving no idle runners or VMs. @@ -69,7 +71,7 @@ Key properties baked into this design: | Kata Containers runtime | QEMU/KVM microVM runtime: containerd drop-in registering `kata-qemu` + the `kata-qemu` RuntimeClass | Kata `3.31.0` | [02-kata-runtime.md](02-kata-runtime.md) | | Preloaded runner image | Custom `actions/runner` image (build toolchain, Bun, Rust nightly + cross targets, native-build deps) built on the host and imported into k3s containerd — no registry | local dated tag | [03-runner-image.md](03-runner-image.md) | | ARC (runner scale set) | actions-runner-controller, `gha-runner-scale-set` flavor: controller in `arc-systems`, one scale set + listener, GitHub App auth | ARC `0.14.2` | [04-arc-and-caching.md](04-arc-and-caching.md) | -| RustFS shared cache | In-cluster S3-compatible store (`svc rustfs:9000`, 100Gi PVC) backing `sccache` and the Bun cache, plus the `sccache-s3` secret and the egress NetworkPolicy | in-cluster service | [04-arc-and-caching.md](04-arc-and-caching.md) | +| Shared caches | RustFS S3 (`svc rustfs:9000`, 100Gi PVC) backs `sccache`; `arc-runners/runner-cache` (100Gi PVC) mounts Bun's package store and Cargo's registry cache into runner pods; the `sccache-s3` secret and egress NetworkPolicy wire access | in-cluster services/storage | [04-arc-and-caching.md](04-arc-and-caching.md) | ## Prerequisites @@ -107,4 +109,4 @@ Work through the numbered guides in order — each builds on the previous: 1. **[01-host-and-cluster.md](01-host-and-cluster.md)** — Host prep (KVM, firewall/NAT) and the single-node k3s install, networking, and CNI. 2. **[02-kata-runtime.md](02-kata-runtime.md)** — Install Kata Containers, wire it into k3s' containerd, and register the `kata-qemu` RuntimeClass. 3. **[03-runner-image.md](03-runner-image.md)** — Build the preloaded runner image and import it into k3s containerd. -4. **[04-arc-and-caching.md](04-arc-and-caching.md)** — Install ARC and the runner scale set, deploy the RustFS shared cache, wire up the `sccache-s3` secret, and apply the egress NetworkPolicy. +4. **[04-arc-and-caching.md](04-arc-and-caching.md)** — Install ARC and the runner scale set, deploy RustFS for `sccache`, add the runner cache PVC for Bun/Cargo, wire up the `sccache-s3` secret, and apply the egress NetworkPolicy. diff --git a/infra/reload-runner.sh b/infra/reload-runner.sh index 3497ff254..88e743e94 100755 --- a/infra/reload-runner.sh +++ b/infra/reload-runner.sh @@ -138,7 +138,7 @@ verify_baked_tools() { local runner="$1" "$runner" --namespace k8s.io run --rm --entrypoint bash "$IMAGE" -lc ' set -e - for b in gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do + for b in gh fd rg magick bun cargo rustc pkg-config clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do command -v "$b" >/dev/null || { echo "MISSING: $b"; exit 1; } done echo "tools OK | bun $(bun --version) | rust $(rustc --version) | sccache $(set -- $(sccache --version); echo "$2") | zig $(zig version) | gh $(set -- $(gh --version | head -1); echo "$3")" @@ -170,7 +170,7 @@ build_with_docker() { echo "==> [2/5] verifying baked tools" docker run --rm --entrypoint bash "$IMAGE" -lc ' set -e - for b in gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do + for b in gh fd rg magick bun cargo rustc pkg-config clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do command -v "$b" >/dev/null || { echo "MISSING: $b"; exit 1; } done echo "tools OK | bun $(bun --version) | rust $(rustc --version) | sccache $(set -- $(sccache --version); echo "$2") | zig $(zig version) | gh $(set -- $(gh --version | head -1); echo "$3")" diff --git a/infra/runner.Dockerfile b/infra/runner.Dockerfile index 5a760ffc7..80a688915 100644 --- a/infra/runner.Dockerfile +++ b/infra/runner.Dockerfile @@ -33,7 +33,7 @@ RUN curl -fsSL https://cli.github.com/packages/githubcli-archive-keyring.gpg -o && echo "deb [arch=$(dpkg --print-architecture) signed-by=/usr/share/keyrings/githubcli-archive-keyring.gpg] https://cli.github.com/packages stable main" > /etc/apt/sources.list.d/github-cli.list \ && apt-get update \ && apt-get install -y \ - build-essential pkg-config curl ca-certificates git unzip xz-utils zstd gh \ + build-essential pkg-config curl ca-certificates git unzip xz-utils gh \ clang lld llvm \ libcairo2-dev libpango1.0-dev libjpeg-dev libgif-dev librsvg2-dev \ fd-find ripgrep imagemagick \ diff --git a/infra/rustfs-cache-maintenance.sh b/infra/rustfs-cache-maintenance.sh deleted file mode 100755 index ad51a83d8..000000000 --- a/infra/rustfs-cache-maintenance.sh +++ /dev/null @@ -1,276 +0,0 @@ -#!/usr/bin/env bash -# Report/prune Bun cache objects in the shared RustFS bucket and alert on PVC -# pressure. The script runs on the CI host over SSH because the RustFS endpoint -# and the S3 credentials live there (inside k8s secrets). -# -# Modes: -# report list usage + object summary, exit 1/2 on warn/critical thresholds -# prune delete stale exact-lockfile Bun cache objects, then report/alert -# -# Usage: -# CI_HOST=my-ci-host ./infra/rustfs-cache-maintenance.sh report -# CI_HOST=my-ci-host ./infra/rustfs-cache-maintenance.sh prune -# -# Env knobs: -# CI_HOST ssh target of the CI host (required) -# KUBECONFIG_REMOTE kubeconfig path on the host [/etc/rancher/k3s/k3s.yaml] -# SECRET_NAMESPACE namespace of sccache-s3 secret [arc-runners] -# SECRET_NAME secret with RustFS client creds [sccache-s3] -# RUSTFS_NAMESPACE namespace of the RustFS deployment [sccache] -# RUSTFS_DEPLOYMENT deployment name of RustFS [rustfs] -# CACHE_PREFIX Bun cache prefix inside the bucket [bun-cache/] -# MAX_AGE_DAYS keep exact-lock objects newer than this [30] -# KEEP_EXACT_PER_OS always keep this many exact objects per OS [3] -# PRUNE_TRIGGER_PERCENT if PVC >= this %, prune oldest extra caches [80] -# TARGET_PERCENT when above trigger, prune toward this % [70] -# WARN_PERCENT report warning exit code at/above this % [80] -# CRITICAL_PERCENT report critical exit code at/above this % [90] -# DRY_RUN true = print deletes but do not delete [false] -set -euo pipefail - -: "${CI_HOST:?set CI_HOST to the ssh target of your CI host, e.g. CI_HOST=my-ci-host}" -mode="${1:-report}" -case "$mode" in report|prune) ;; *) echo "usage: $0 report|prune" >&2; exit 2 ;; esac - -KUBECONFIG_REMOTE="${KUBECONFIG_REMOTE:-/etc/rancher/k3s/k3s.yaml}" -SECRET_NAMESPACE="${SECRET_NAMESPACE:-arc-runners}" -SECRET_NAME="${SECRET_NAME:-sccache-s3}" -RUSTFS_NAMESPACE="${RUSTFS_NAMESPACE:-sccache}" -RUSTFS_DEPLOYMENT="${RUSTFS_DEPLOYMENT:-rustfs}" -CACHE_PREFIX="${CACHE_PREFIX:-bun-cache/}" -MAX_AGE_DAYS="${MAX_AGE_DAYS:-30}" -KEEP_EXACT_PER_OS="${KEEP_EXACT_PER_OS:-3}" -PRUNE_TRIGGER_PERCENT="${PRUNE_TRIGGER_PERCENT:-80}" -TARGET_PERCENT="${TARGET_PERCENT:-70}" -WARN_PERCENT="${WARN_PERCENT:-80}" -CRITICAL_PERCENT="${CRITICAL_PERCENT:-90}" -DRY_RUN="${DRY_RUN:-false}" - -ssh "$CI_HOST" bash -s -- \ - "$mode" "$KUBECONFIG_REMOTE" "$SECRET_NAMESPACE" "$SECRET_NAME" "$RUSTFS_NAMESPACE" "$RUSTFS_DEPLOYMENT" \ - "$CACHE_PREFIX" "$MAX_AGE_DAYS" "$KEEP_EXACT_PER_OS" "$PRUNE_TRIGGER_PERCENT" "$TARGET_PERCENT" \ - "$WARN_PERCENT" "$CRITICAL_PERCENT" "$DRY_RUN" <<'REMOTE' -set -euo pipefail -MODE="$1" -export KUBECONFIG="$2" -SECRET_NAMESPACE="$3" -SECRET_NAME="$4" -RUSTFS_NAMESPACE="$5" -RUSTFS_DEPLOYMENT="$6" -CACHE_PREFIX="$7" -MAX_AGE_DAYS="$8" -KEEP_EXACT_PER_OS="$9" -PRUNE_TRIGGER_PERCENT="${10}" -TARGET_PERCENT="${11}" -WARN_PERCENT="${12}" -CRITICAL_PERCENT="${13}" -DRY_RUN="${14}" - -ACCESS_KEY="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.AWS_ACCESS_KEY_ID}' | base64 -d)" -SECRET_KEY="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.AWS_SECRET_ACCESS_KEY}' | base64 -d)" -BUCKET="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.SCCACHE_BUCKET}' | base64 -d)" -ENDPOINT="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.SCCACHE_ENDPOINT}' | base64 -d)" -REGION="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.SCCACHE_REGION}' | base64 -d)" -USE_SSL="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.SCCACHE_S3_USE_SSL}' | base64 -d)" -if [ "$USE_SSL" = "true" ]; then SCHEME=https; else SCHEME=http; fi - -# The secret intentionally stores the in-cluster Service DNS because runner pods -# consume it. This host-side maintenance script runs outside cluster DNS, so talk -# to the same Service by ClusterIP instead when the endpoint points at `.svc`. -if [[ "$ENDPOINT" == *.svc.*:* || "$ENDPOINT" == *.svc:* ]]; then - svc_ip="$(kubectl get svc "$RUSTFS_DEPLOYMENT" -n "$RUSTFS_NAMESPACE" -o jsonpath='{.spec.clusterIP}')" - svc_port="${ENDPOINT##*:}" - ENDPOINT="${svc_ip}:${svc_port}" -fi - -usage_line() { - kubectl exec -n "$RUSTFS_NAMESPACE" deploy/"$RUSTFS_DEPLOYMENT" -- df -P /data | tail -1 -} - -before="$(usage_line)" -TOTAL_KIB="$(awk '{print $2}' <<<"$before")" -USED_KIB="$(awk '{print $3}' <<<"$before")" -AVAIL_KIB="$(awk '{print $4}' <<<"$before")" -USED_PCT_RAW="$(awk '{print $5}' <<<"$before")" -USED_PCT="${USED_PCT_RAW%%%}" - -echo "==> RustFS PVC before" -echo "$before" - -summary="$(python3 - "$MODE" "$SCHEME" "$ENDPOINT" "$BUCKET" "$REGION" "$ACCESS_KEY" "$SECRET_KEY" "$CACHE_PREFIX" "$MAX_AGE_DAYS" "$KEEP_EXACT_PER_OS" "$PRUNE_TRIGGER_PERCENT" "$TARGET_PERCENT" "$DRY_RUN" "$TOTAL_KIB" "$USED_KIB" "$USED_PCT" <<'PY' -from __future__ import annotations -from collections import defaultdict -from datetime import datetime, timedelta, timezone -import json -import re -import subprocess -import sys -import urllib.parse -import xml.etree.ElementTree as ET - -mode, scheme, endpoint, bucket, region, access, secret, prefix, max_age_days, keep_per_os, prune_trigger, target_pct, dry_run, total_kib, used_kib, used_pct = sys.argv[1:] -max_age_days = int(max_age_days) -keep_per_os = int(keep_per_os) -prune_trigger = int(prune_trigger) -target_pct = int(target_pct) -dry_run = dry_run.lower() == "true" -total_kib = int(total_kib) -used_kib = int(used_kib) -used_pct = int(used_pct) - -prefix = prefix.rstrip("/") + "/" -base = f"{scheme}://{endpoint}/{bucket}" -now = datetime.now(timezone.utc) -cutoff = now - timedelta(days=max_age_days) - - -def curl(method: str, url: str, extra: list[str] | None = None) -> str: - cmd = [ - "curl", "-fsS", - "--aws-sigv4", f"aws:amz:{region}:s3", - "--user", f"{access}:{secret}", - "-X", method, - ] - if extra: - cmd.extend(extra) - cmd.append(url) - return subprocess.check_output(cmd, text=True) - - -def list_objects() -> list[dict]: - objs: list[dict] = [] - token = None - while True: - q = {"list-type": "2", "prefix": prefix} - if token: - q["continuation-token"] = token - url = f"{base}/?{urllib.parse.urlencode(q)}" - root = ET.fromstring(curl("GET", url)) - for node in root.findall(".//{*}Contents"): - objs.append({ - "key": node.findtext("{*}Key", default=""), - "size": int(node.findtext("{*}Size", default="0")), - "etag": node.findtext("{*}ETag", default="").strip('"'), - "last_modified": datetime.fromisoformat(node.findtext("{*}LastModified", default="1970-01-01T00:00:00+00:00")), - }) - token = root.findtext(".//{*}NextContinuationToken") - if not token: - return objs - - -def human(n: int) -> str: - units = ["B", "KiB", "MiB", "GiB", "TiB"] - value = float(n) - for unit in units: - if value < 1024 or unit == units[-1]: - return f"{value:.1f}{unit}" - value /= 1024 - return f"{n}B" - -objs = list_objects() -exact_re = re.compile(rf"^{re.escape(prefix)}(store|nm)-([^-]+)-([0-9a-f]{{32}})\.(tzst|tgz)$") -latest_re = re.compile(rf"^{re.escape(prefix)}store-([^-]+)-latest\.(tzst|tgz)$") - -exacts: list[dict] = [] -latest_aliases: list[dict] = [] -other_prefix: list[dict] = [] -for obj in objs: - if m := exact_re.match(obj["key"]): - obj = obj | {"kind": m.group(1), "os": m.group(2), "lock": m.group(3), "codec": m.group(4)} - exacts.append(obj) - elif m := latest_re.match(obj["key"]): - obj = obj | {"kind": "store", "os": m.group(1), "lock": "latest", "codec": m.group(2)} - latest_aliases.append(obj) - else: - other_prefix.append(obj) - -protected = {obj["key"] for obj in latest_aliases} -by_group: dict[tuple[str, str], list[dict]] = defaultdict(list) -by_store_etag: dict[tuple[str, str, str], list[dict]] = defaultdict(list) -for obj in exacts: - by_group[(obj["kind"], obj["os"])] .append(obj) - if obj["kind"] == "store": - by_store_etag[(obj["os"], obj["codec"], obj["etag"])] .append(obj) - -for alias in latest_aliases: - matches = by_store_etag.get((alias["os"], alias["codec"], alias["etag"]), []) - if matches: - newest = max(matches, key=lambda o: o["last_modified"]) - protected.add(newest["key"]) - -for group, items in by_group.items(): - items.sort(key=lambda o: o["last_modified"], reverse=True) - for obj in items[:keep_per_os]: - protected.add(obj["key"]) - for obj in items: - if obj["last_modified"] >= cutoff: - protected.add(obj["key"]) - -eligible = [obj for obj in exacts if obj["key"] not in protected] -eligible.sort(key=lambda o: o["last_modified"]) -aged = [obj for obj in eligible if obj["last_modified"] < cutoff] - -selected: list[dict] = list(aged) -selected_keys = {obj["key"] for obj in selected} -need_reclaim_kib = max(0, used_kib - (total_kib * target_pct // 100)) if used_pct >= prune_trigger else 0 -reclaimed_kib = sum(obj["size"] // 1024 for obj in selected) -if need_reclaim_kib > reclaimed_kib: - for obj in eligible: - if obj["key"] in selected_keys: - continue - selected.append(obj) - selected_keys.add(obj["key"]) - reclaimed_kib += obj["size"] // 1024 - if reclaimed_kib >= need_reclaim_kib: - break - -selected.sort(key=lambda o: (o["kind"], o["os"], o["last_modified"], o["key"])) -delete_total = sum(obj["size"] for obj in selected) - -print(f"bun-cache objects: exact={len(exacts)} latest={len(latest_aliases)} other-prefix={len(other_prefix)} total={human(sum(o['size'] for o in objs))}") -print(f"policy: max_age_days={max_age_days} keep_exact_per_os={keep_per_os} prune_trigger={prune_trigger}% target={target_pct}%") -print(f"eligible exact objects: {len(eligible)} | selected for deletion: {len(selected)} | reclaimable: {human(delete_total)}") -for obj in selected[:20]: - age_days = int((now - obj['last_modified']).total_seconds() // 86400) - print(f" delete {obj['key']} age={age_days}d size={human(obj['size'])}") -if len(selected) > 20: - print(f" ... {len(selected) - 20} more") - -if mode == "prune": - for obj in selected: - if dry_run: - continue - curl("DELETE", f"{base}/{obj['key']}") - -print("JSON_SUMMARY=" + json.dumps({ - "exact": len(exacts), - "latest": len(latest_aliases), - "other": len(other_prefix), - "eligible": len(eligible), - "selected": len(selected), - "delete_bytes": delete_total, - "dry_run": dry_run, -})) -PY -)" - -echo "$summary" - -after="$(usage_line)" -AFTER_USED_PCT_RAW="$(awk '{print $5}' <<<"$after")" -AFTER_USED_PCT="${AFTER_USED_PCT_RAW%%%}" - -echo "==> RustFS PVC after" -echo "$after" - -if [ "$AFTER_USED_PCT" -ge "$CRITICAL_PERCENT" ]; then - echo "CRITICAL: rustfs-data usage ${AFTER_USED_PCT}% >= ${CRITICAL_PERCENT}%" >&2 - exit 2 -fi -if [ "$AFTER_USED_PCT" -ge "$WARN_PERCENT" ]; then - echo "WARNING: rustfs-data usage ${AFTER_USED_PCT}% >= ${WARN_PERCENT}%" >&2 - exit 1 -fi - -echo "OK: rustfs-data usage ${AFTER_USED_PCT}%" -REMOTE