From 551c5996722141a36b1136fb67a1e97487e83d0f Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 15 Jun 2026 01:04:37 +0200 Subject: [PATCH] feat(infra): added infra automation for runner, cache, and kata runtime operations - Enabled containerd-first runner builds with auto backend defaults, tool bootstrap, and docker fallback. - Added RustFS cache maintenance report/prune modes with S3-driven listing, cache key grouping, and threshold exits. - Added kata runtime tuning automation via SSH that patches config, runs smoke tests, and cleans up resources. - Expanded runner-image smoke checks to verify clang, lld, sccache, zig, and cargo-* requirements. --- infra/docs/02-kata-runtime.md | 39 +++-- infra/docs/03-runner-image.md | 17 +- infra/docs/04-arc-and-caching.md | 48 +++++- infra/reload-runner.sh | 193 ++++++++++++++++++--- infra/runner.Dockerfile | 6 +- infra/rustfs-cache-maintenance.sh | 276 ++++++++++++++++++++++++++++++ infra/tune-kata-runtime.sh | 81 +++++++++ 7 files changed, 600 insertions(+), 60 deletions(-) create mode 100755 infra/rustfs-cache-maintenance.sh create mode 100755 infra/tune-kata-runtime.sh diff --git a/infra/docs/02-kata-runtime.md b/infra/docs/02-kata-runtime.md index 680961175..bbc716eb5 100644 --- a/infra/docs/02-kata-runtime.md +++ b/infra/docs/02-kata-runtime.md @@ -266,16 +266,16 @@ rootfs_type = "ext4" cpu_features = "pmu=off" kernel_params = "cgroup_no_v1=all systemd.unified_cgroup_hierarchy=1" -default_vcpus = 1 +default_vcpus = 2 default_maxvcpus = 0 -default_memory = 2048 +default_memory = 4096 default_maxmemory = 0 memory_slots = 10 shared_fs = "virtio-fs" virtio_fs_daemon = "/opt/kata/libexec/virtiofsd" virtio_fs_cache = "auto" -virtio_fs_extra_args = ["--thread-pool-size=1", "--announce-submounts"] +virtio_fs_extra_args = ["--thread-pool-size=4", "--announce-submounts"] disable_block_device_use = true block_device_driver = "virtio-scsi" @@ -308,24 +308,24 @@ guest, matching a modern systemd userspace. This is the most important block to understand for a CI runner. -- `default_vcpus = 1` and `default_memory = 2048` (MiB) are the **boot-time** - size. Every microVM starts tiny: 1 vCPU, 2 GiB. +- `default_vcpus = 2` and `default_memory = 4096` (MiB) are the **boot-time** + size. Runner microVMs now start at the runner pod's guaranteed request: 2 vCPU, + 4 GiB, rather than booting tiny and immediately hotplugging to that floor. - `default_maxvcpus = 0` means "no fixed ceiling — use the host's physical CPU count" (32 on this box). `default_maxmemory = 0` likewise means "host total RAM". `memory_slots = 10` is the number of ACPI DIMM hotplug slots, i.e. how many memory-grow operations the guest can accept. -- `static_sandbox_resource_mgmt = false` enables **dynamic** sizing: Kata reads - the pod's container CPU/memory **limits** that the kubelet/CRI hands the shim - and **hotplugs** vCPUs and memory after boot to match. (Set it to `true` and - the VM stays fixed at the defaults regardless of pod limits.) So a runner pod - requesting `2 CPU / 4Gi` with limits `6 CPU / 12Gi` boots at 1 vCPU/2 GiB and - grows toward 6 vCPU / 12 GiB. If a pod sets no limits, the VM stays at the - defaults. +- `static_sandbox_resource_mgmt = false` still enables **dynamic** sizing: Kata + reads the pod's CPU/memory **limits** that the kubelet/CRI hands the shim and + hotplugs beyond the boot floor as needed. So a runner pod requesting `2 CPU / + 4Gi` with limits `8 CPU / 12Gi` now boots at 2 vCPU/4 GiB and grows toward + 8 vCPU / 12 GiB. If a pod sets no limits, the VM stays at the defaults. -The practical knobs: raise `default_vcpus`/`default_memory` only if jobs need a -big VM *immediately* at boot (hotplug has latency); otherwise leave them small -and let the pod's `resources.limits` drive the size. Sizing the runner pod is -covered in [`04-arc-and-caching.md`](./04-arc-and-caching.md). +The practical rule here is simple: align the defaults to the runner pod's +requests when every job creates a fresh VM and immediately needs that baseline +anyway; let `resources.limits` remain the hotplug ceiling. The repo ships +[`infra/tune-kata-runtime.sh`](../tune-kata-runtime.sh) to apply exactly this +change (plus the virtiofsd worker-pool tuning below) over SSH to the host. ### `shared_fs = "virtio-fs"` — sharing the container rootfs into the VM @@ -336,9 +336,10 @@ directory over **virtio-fs**, and the guest mounts it as the container root. This is why `disable_block_device_use = true`: the rootfs travels in over the shared filesystem, not as a block device. Benefits: no image-to-block conversion, near-instant rootfs availability, and host/guest can both see the -files. `virtio_fs_cache = "auto"` and `--thread-pool-size=1` are conservative -caching/concurrency defaults; `--announce-submounts` makes nested mounts visible -to the guest. +files. `virtio_fs_cache = "auto"` keeps the conservative page-cache behavior, but +the active worker pool is now `--thread-pool-size=4` rather than `1` so the +metadata-heavy Bun cache restore / extract path has a few host workers to fan out +across. `--announce-submounts` keeps nested mounts visible to the guest. `emptydir_mode = "shared-fs"` extends the same mechanism to Kubernetes `emptyDir` volumes — they are shared into the guest over virtio-fs instead of diff --git a/infra/docs/03-runner-image.md b/infra/docs/03-runner-image.md index 374b273a6..28f506f43 100644 --- a/infra/docs/03-runner-image.md +++ b/infra/docs/03-runner-image.md @@ -143,9 +143,9 @@ RUN curl --proto '=https' --tlsv1.2 -fsSL https://sh.rustup.rs \ && rustc --version \ && sccache --version \ && zig version \ - && cargo nextest --version \ - && cargo zigbuild --version \ - && cargo xwin --version + && cargo-nextest --version \ + && cargo-zigbuild --help >/dev/null \ + && cargo-xwin --help >/dev/null ``` ### Stage-by-stage annotation @@ -323,12 +323,11 @@ old one (the scale set is scale-to-zero, so this drains quickly). ### Running it from the repo (over SSH) You do not have to keep `reload.sh` on the host. The repo ships the version- -controlled Dockerfile plus an SSH-driven wrapper that performs the same build, -import, and rollout remotely, so the whole image lifecycle is managed from a -checkout: +controlled Dockerfile plus an SSH-driven wrapper that performs the rollout +remotely from a checkout: - [`infra/runner.Dockerfile`](../runner.Dockerfile) - the image definition (source of truth). -- [`infra/reload-runner.sh`](../reload-runner.sh) - copies that Dockerfile to the host, then runs the build / verify / import / `helm upgrade` / verify steps over SSH. +- [`infra/reload-runner.sh`](../reload-runner.sh) - copies that Dockerfile to the host, then prefers a **direct containerd build path**: bootstrap pinned `buildkitd` + `buildctl` + `nerdctl` under the remote build dir if needed, build straight into the k3s `k8s.io` namespace, smoke-test from that image store, then `helm upgrade` ARC. Set `BUILD_BACKEND=docker` to force the legacy `docker build` + `docker save | ctr images import` path. The host is never hardcoded; point it at your node with `CI_HOST`: @@ -417,10 +416,10 @@ standalone: ```bash docker run --rm --entrypoint bash omp-kata-runner:preloaded -lc ' set -e - for b in gh fd rg magick bun cargo rustc pkg-config zstd; do + for b in gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do command -v "$b" >/dev/null || { echo "MISSING: $b"; exit 1; } done - echo "tools OK | $(bun --version) | $(rustc --version)" + echo "tools OK | bun $(bun --version) | rust $(rustc --version) | sccache $(sccache --version | cut -d\" \" -f2) | zig $(zig version) | gh $(gh --version | head -1 | cut -d\" \" -f3)" ' ``` diff --git a/infra/docs/04-arc-and-caching.md b/infra/docs/04-arc-and-caching.md index 512a4eee2..d1ca5dfa8 100644 --- a/infra/docs/04-arc-and-caching.md +++ b/infra/docs/04-arc-and-caching.md @@ -196,13 +196,14 @@ Field by field: overridden explicitly because the custom image keeps the upstream layout. - **`envFrom.secretRef.name: sccache-s3`** - injects the shared-cache S3 configuration into every job's environment ([step 5](#5-shared-cache-rustfs-s3)). -- **`resources`** - requests `2` CPU / `4Gi`, limits `6` CPU / `12Gi`. Kata reads - these and sizes the guest accordingly: the VM boots at the runtime defaults - (`default_vcpus: 1`, `default_memory: 2048`) and **hotplugs** vCPUs and RAM to - cover the pod's containers, with `default_maxvcpus: 0` allowing up to all host - CPUs. Effectively the **requests are the guaranteed VM size** and the **limits - are the hotplug ceiling**. See [02-kata-runtime.md](02-kata-runtime.md) for the - hotplug mechanics. +- **`resources`** - requests `2` CPU / `4Gi`, limits `8` CPU / `12Gi`. Kata reads + these and sizes the guest accordingly: the VM now boots at the same + guaranteed floor (`default_vcpus: 2`, `default_memory: 4096`) and only + hotplugs beyond that toward the limits, with `default_maxvcpus: 0` allowing up + to all host CPUs. Effectively the **requests are the boot-time VM size** and + the **limits are the hotplug ceiling**. See [02-kata-runtime.md](02-kata-runtime.md) + for the runtime knobs and [`infra/tune-kata-runtime.sh`](../tune-kata-runtime.sh) + for the SSH-driven patch helper. --- @@ -456,6 +457,39 @@ suffix records the codec and restore only inflates what the host can decompress. On **save**, it writes the store/`node_modules` objects only when this exact lockfile has none yet (`s3_exists` check), avoiding redundant uploads. +### 5d. Retention and PVC pressure + +The Bun cache intentionally keeps exact lockfile objects once written: + +- `store--` and `nm--` are immutable warm caches; +- only `store--latest` is overwritten. + +That means `bun.lock` churn will accumulate old exact objects on the `rustfs-data` +PVC. The repo ships [`infra/rustfs-cache-maintenance.sh`](../rustfs-cache-maintenance.sh) +to keep that under control from an ops checkout: + +```bash +CI_HOST= ./infra/rustfs-cache-maintenance.sh report +CI_HOST= ./infra/rustfs-cache-maintenance.sh prune +``` + +Its policy is deliberately conservative: + +- always keep every `store--latest` alias; +- always keep the newest exact `store-*` object that matches each `latest` alias + by ETag; +- always keep the newest `KEEP_EXACT_PER_OS` exact objects per OS (`3` by default) + for both `store-*` and `nm-*`; +- never delete exact objects newer than `MAX_AGE_DAYS` (`30` by default); +- once the PVC reaches `PRUNE_TRIGGER_PERCENT` (`80` by default), prune oldest + remaining exact objects toward `TARGET_PERCENT` (`70` by default), even if they + are newer than the age threshold. + +The same script is the PVC-pressure alert: after `report` or `prune` it reads +`df -P /data` from the live RustFS pod and exits `1` at `WARN_PERCENT` (`80`) +and `2` at `CRITICAL_PERCENT` (`90`). Wire that into cron / systemd / your +monitoring runner; a failing exit is the signal that the PVC is too full. + This caching is why RustFS sits inside the egress allow-list on `tcp/9000` ([step 6](#6-runner-egress-lockdown)). diff --git a/infra/reload-runner.sh b/infra/reload-runner.sh index e51f41d62..5e6c7550c 100755 --- a/infra/reload-runner.sh +++ b/infra/reload-runner.sh @@ -1,8 +1,16 @@ #!/usr/bin/env bash # Build + roll the preloaded omp-kata runner image onto the self-hosted CI host, # driven over SSH from this repo. The Dockerfile next to this script is the -# source of truth: it is copied to the host, built there, imported into k3s -# containerd, and the ARC runner scale set is pointed at the new tag and rolled. +# source of truth: it is copied to the host, built there, and the ARC runner +# scale set is pointed at the new tag and rolled. +# +# By default the remote side prefers a direct containerd build path: bootstrap a +# pinned BuildKit + nerdctl toolchain under the remote build dir, build straight +# into the k3s containerd `k8s.io` namespace, and smoke-test from that image +# store. That avoids the old `docker save | ctr images import` tarball hop and +# cuts duplicate layer I/O on large reloads. If the direct path is unavailable +# or you set BUILD_BACKEND=docker, it falls back to the legacy Docker-daemon +# build + import path. # # The host is intentionally NOT hardcoded (this repo is public). Set CI_HOST to # your ssh target; the remaining knobs default to the reference deployment. @@ -13,13 +21,17 @@ # CI_HOST=my-ci-host ./infra/reload-runner.sh my/repo:tag # explicit repo:tag # # Env knobs (defaults match the reference deployment): -# CI_HOST ssh target of the CI host (required) -# REMOTE_CTX remote build dir for the Dockerfile [/root/omp-kata-runner-image] -# ARC_VALUES remote ARC scale-set helm values file [/root/arc-omp-values.yaml] -# ARC_RELEASE helm release name of the runner scale set [omp-kata] -# ARC_NAMESPACE namespace the runner scale set lives in [arc-runners] -# ARC_CHART_VERSION gha-runner-scale-set chart version [0.14.2] -# KUBECONFIG_REMOTE kubeconfig path on the host [/etc/rancher/k3s/k3s.yaml] +# CI_HOST ssh target of the CI host (required) +# REMOTE_CTX remote build dir for the Dockerfile [/root/omp-kata-runner-image] +# ARC_VALUES remote ARC scale-set helm values file [/root/arc-omp-values.yaml] +# ARC_RELEASE helm release name of the runner scale set [omp-kata] +# ARC_NAMESPACE namespace the runner scale set lives in [arc-runners] +# ARC_CHART_VERSION gha-runner-scale-set chart version [0.14.2] +# KUBECONFIG_REMOTE kubeconfig path on the host [/etc/rancher/k3s/k3s.yaml] +# BUILD_BACKEND auto | containerd | docker [auto] +# CONTAINERD_SOCKET_REMOTE remote containerd socket [/run/k3s/containerd/containerd.sock] +# NERDCTL_VERSION nerdctl release to bootstrap on demand [2.1.6] +# BUILDKIT_VERSION BuildKit release to bootstrap on demand [0.25.1] set -euo pipefail : "${CI_HOST:?set CI_HOST to the ssh target of your CI host, e.g. CI_HOST=my-ci-host}" @@ -29,6 +41,10 @@ ARC_RELEASE="${ARC_RELEASE:-omp-kata}" ARC_NAMESPACE="${ARC_NAMESPACE:-arc-runners}" ARC_CHART_VERSION="${ARC_CHART_VERSION:-0.14.2}" KUBECONFIG_REMOTE="${KUBECONFIG_REMOTE:-/etc/rancher/k3s/k3s.yaml}" +BUILD_BACKEND="${BUILD_BACKEND:-auto}" +CONTAINERD_SOCKET_REMOTE="${CONTAINERD_SOCKET_REMOTE:-/run/k3s/containerd/containerd.sock}" +NERDCTL_VERSION="${NERDCTL_VERSION:-2.1.6}" +BUILDKIT_VERSION="${BUILDKIT_VERSION:-0.25.1}" arg="${1:-$(date +%Y-%m-%d-%H%M%S)}" case "$arg" in *:*) IMAGE="$arg";; *) IMAGE="omp-kata-runner:$arg";; esac @@ -44,26 +60,159 @@ scp -q "$here/runner.Dockerfile" "${CI_HOST}:${REMOTE_CTX}/Dockerfile" # args (no secrets, no spaces) so it survives the ssh command-string re-parse # regardless of the host's login shell. ssh "$CI_HOST" bash -s -- \ - "$IMAGE" "$REMOTE_CTX" "$ARC_VALUES" "$ARC_RELEASE" "$ARC_NAMESPACE" "$ARC_CHART_VERSION" "$KUBECONFIG_REMOTE" <<'REMOTE' + "$IMAGE" "$REMOTE_CTX" "$ARC_VALUES" "$ARC_RELEASE" "$ARC_NAMESPACE" "$ARC_CHART_VERSION" \ + "$KUBECONFIG_REMOTE" "$BUILD_BACKEND" "$CONTAINERD_SOCKET_REMOTE" "$NERDCTL_VERSION" "$BUILDKIT_VERSION" <<'REMOTE' set -euo pipefail IMAGE="$1"; REMOTE_CTX="$2"; ARC_VALUES="$3"; ARC_RELEASE="$4"; ARC_NAMESPACE="$5"; ARC_CHART_VERSION="$6" export KUBECONFIG="$7" +BUILD_BACKEND="$8"; CONTAINERD_SOCKET="$9"; NERDCTL_VERSION="${10}"; BUILDKIT_VERSION="${11}" cd "$REMOTE_CTX" -echo "==> [1/5] building $IMAGE" -DOCKER_BUILDKIT=1 docker build -t "$IMAGE" -t omp-kata-runner:preloaded . +TOOLS_DIR="$REMOTE_CTX/.containerd-build-tools" +BIN_DIR="$TOOLS_DIR/bin" +RUN_DIR="$TOOLS_DIR/run" +ROOT_DIR="$TOOLS_DIR/root" +LOG_DIR="$TOOLS_DIR/log" +NERDCTL_BIN="$BIN_DIR/nerdctl" +BUILDKITD_BIN="$BIN_DIR/buildkitd" +BUILDKITCTL_BIN="$BIN_DIR/buildctl" +BUILDKIT_ADDR="unix://$RUN_DIR/buildkitd.sock" -echo "==> [2/5] verifying baked tools" -docker run --rm --entrypoint bash "$IMAGE" -lc ' - set -e - for b in gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do - command -v "$b" >/dev/null || { echo "MISSING: $b"; exit 1; } +BUILDKITD_PID="" +cleanup_buildkitd() { + if [ -n "${BUILDKITD_PID:-}" ]; then + kill "$BUILDKITD_PID" >/dev/null 2>&1 || true + fi +} +trap cleanup_buildkitd EXIT +extract_bin() { + local archive="$1" needle="$2" out="$3" tmp + tmp="$(mktemp -d)" + tar -C "$tmp" -xf "$archive" + cp "$(find "$tmp" -type f -name "$needle" | head -1)" "$out" + chmod +x "$out" + rm -rf "$tmp" +} + +bootstrap_containerd_tools() { + mkdir -p "$BIN_DIR" "$RUN_DIR" "$ROOT_DIR" "$LOG_DIR" + if [ ! -S "$CONTAINERD_SOCKET" ]; then + echo "containerd socket missing: $CONTAINERD_SOCKET" >&2 + return 1 + fi + if [ ! -x "$NERDCTL_BIN" ]; then + local archive="$TOOLS_DIR/nerdctl-${NERDCTL_VERSION}.tar.gz" + echo "==> [1/5] bootstrapping nerdctl ${NERDCTL_VERSION}" + curl -fsSL "https://github.com/containerd/nerdctl/releases/download/v${NERDCTL_VERSION}/nerdctl-${NERDCTL_VERSION}-linux-amd64.tar.gz" -o "$archive" + extract_bin "$archive" nerdctl "$NERDCTL_BIN" + fi + if [ ! -x "$BUILDKITD_BIN" ] || [ ! -x "$BUILDKITCTL_BIN" ]; then + local archive="$TOOLS_DIR/buildkit-v${BUILDKIT_VERSION}.tar.gz" + echo "==> [1/5] bootstrapping BuildKit ${BUILDKIT_VERSION}" + curl -fsSL "https://github.com/moby/buildkit/releases/download/v${BUILDKIT_VERSION}/buildkit-v${BUILDKIT_VERSION}.linux-amd64.tar.gz" -o "$archive" + extract_bin "$archive" buildkitd "$BUILDKITD_BIN" + extract_bin "$archive" buildctl "$BUILDKITCTL_BIN" + fi +} + +start_buildkitd() { + rm -f "$RUN_DIR/buildkitd.sock" + "$BUILDKITD_BIN" \ + --addr "$BUILDKIT_ADDR" \ + --root "$ROOT_DIR" \ + --containerd-worker=true \ + --containerd-worker-namespace k8s.io \ + --containerd-worker-addr "$CONTAINERD_SOCKET" \ + --oci-worker=false >"$LOG_DIR/buildkitd.log" 2>&1 & + BUILDKITD_PID="$!" + for _ in $(seq 1 120); do + [ -S "$RUN_DIR/buildkitd.sock" ] && return 0 + sleep 0.25 done - echo "tools OK | bun $(bun --version) | rust $(rustc --version) | sccache $(sccache --version | awk '\''{print $2}'\'') | zig $(zig version) | gh $(gh --version | head -1 | cut -d\" \" -f3)" -' + echo "buildkitd did not create $RUN_DIR/buildkitd.sock" >&2 + sed -n '1,120p' "$LOG_DIR/buildkitd.log" >&2 || true + return 1 +} -echo "==> [3/5] importing into k3s containerd (k8s.io namespace)" -docker save "$IMAGE" | k3s ctr -n k8s.io images import --platform linux/amd64 - +verify_baked_tools() { + local runner="$1" + "$runner" --namespace k8s.io run --rm --entrypoint bash "$IMAGE" -lc ' + set -e + for b in gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do + command -v "$b" >/dev/null || { echo "MISSING: $b"; exit 1; } + done + echo "tools OK | bun $(bun --version) | rust $(rustc --version) | sccache $(sccache --version | cut -d\" \" -f2) | zig $(zig version) | gh $(gh --version | head -1 | cut -d\" \" -f3)" + ' +} + +build_with_containerd() { + bootstrap_containerd_tools + start_buildkitd + + echo "==> [2/5] building $IMAGE directly into k3s containerd (k8s.io namespace)" + "$BUILDKITCTL_BIN" --addr "$BUILDKIT_ADDR" build \ + --progress=plain \ + --frontend dockerfile.v0 \ + --local context=. \ + --local dockerfile=. \ + --opt filename=Dockerfile \ + --output "type=image,name=$IMAGE,store=true" + k3s ctr -n k8s.io images tag "$IMAGE" omp-kata-runner:preloaded >/dev/null 2>&1 || true + + echo "==> [3/5] verifying baked tools from k3s containerd" + verify_baked_tools "$NERDCTL_BIN" +} + +build_with_docker() { + echo "==> [1/5] building $IMAGE with docker" + DOCKER_BUILDKIT=1 docker build -t "$IMAGE" -t omp-kata-runner:preloaded . + + echo "==> [2/5] verifying baked tools" + docker run --rm --entrypoint bash "$IMAGE" -lc ' + set -e + for b in gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do + command -v "$b" >/dev/null || { echo "MISSING: $b"; exit 1; } + done + echo "tools OK | bun $(bun --version) | rust $(rustc --version) | sccache $(sccache --version | cut -d\" \" -f2) | zig $(zig version) | gh $(gh --version | head -1 | cut -d\" \" -f3)" + ' + + echo "==> [3/5] importing into k3s containerd (k8s.io namespace)" + docker save "$IMAGE" | k3s ctr -n k8s.io images import --platform linux/amd64 - +} + +selected_backend="$BUILD_BACKEND" +case "$BUILD_BACKEND" in + auto) + if bootstrap_containerd_tools >/dev/null 2>&1; then + selected_backend=containerd + else + selected_backend=docker + fi + ;; + containerd|docker) + ;; + *) + echo "BUILD_BACKEND must be auto, containerd, or docker (got: $BUILD_BACKEND)" >&2 + exit 2 + ;; +esac + +echo "==> selected build backend: $selected_backend" +case "$selected_backend" in + containerd) + if ! build_with_containerd; then + if [ "$BUILD_BACKEND" = auto ]; then + echo "==> containerd build path failed; falling back to docker" >&2 + build_with_docker + else + exit 1 + fi + fi + ;; + docker) + build_with_docker + ;; +esac echo "==> [4/5] pointing ARC runner scale set at $IMAGE" sed -i "s#image: omp-kata-runner:.*#image: $IMAGE#" "$ARC_VALUES" @@ -78,4 +227,4 @@ echo "ARC runner image is now: $live" [ "$live" = "$IMAGE" ] && echo "OK: reloaded $IMAGE" || { echo "MISMATCH: expected $IMAGE"; exit 1; } REMOTE -echo "OK: $IMAGE built on ${CI_HOST}, imported into k3s, and rolled out to ARC." +echo "OK: $IMAGE built on ${CI_HOST}, stored in k3s containerd, and rolled out to ARC." diff --git a/infra/runner.Dockerfile b/infra/runner.Dockerfile index 7067fb61e..5a760ffc7 100644 --- a/infra/runner.Dockerfile +++ b/infra/runner.Dockerfile @@ -71,6 +71,6 @@ RUN curl --proto '=https' --tlsv1.2 -fsSL https://sh.rustup.rs \ && rustc --version \ && sccache --version \ && zig version \ - && cargo nextest --version \ - && cargo zigbuild --version \ - && cargo xwin --version + && cargo-nextest --version \ + && cargo-zigbuild --help >/dev/null \ + && cargo-xwin --help >/dev/null diff --git a/infra/rustfs-cache-maintenance.sh b/infra/rustfs-cache-maintenance.sh new file mode 100755 index 000000000..ad51a83d8 --- /dev/null +++ b/infra/rustfs-cache-maintenance.sh @@ -0,0 +1,276 @@ +#!/usr/bin/env bash +# Report/prune Bun cache objects in the shared RustFS bucket and alert on PVC +# pressure. The script runs on the CI host over SSH because the RustFS endpoint +# and the S3 credentials live there (inside k8s secrets). +# +# Modes: +# report list usage + object summary, exit 1/2 on warn/critical thresholds +# prune delete stale exact-lockfile Bun cache objects, then report/alert +# +# Usage: +# CI_HOST=my-ci-host ./infra/rustfs-cache-maintenance.sh report +# CI_HOST=my-ci-host ./infra/rustfs-cache-maintenance.sh prune +# +# Env knobs: +# CI_HOST ssh target of the CI host (required) +# KUBECONFIG_REMOTE kubeconfig path on the host [/etc/rancher/k3s/k3s.yaml] +# SECRET_NAMESPACE namespace of sccache-s3 secret [arc-runners] +# SECRET_NAME secret with RustFS client creds [sccache-s3] +# RUSTFS_NAMESPACE namespace of the RustFS deployment [sccache] +# RUSTFS_DEPLOYMENT deployment name of RustFS [rustfs] +# CACHE_PREFIX Bun cache prefix inside the bucket [bun-cache/] +# MAX_AGE_DAYS keep exact-lock objects newer than this [30] +# KEEP_EXACT_PER_OS always keep this many exact objects per OS [3] +# PRUNE_TRIGGER_PERCENT if PVC >= this %, prune oldest extra caches [80] +# TARGET_PERCENT when above trigger, prune toward this % [70] +# WARN_PERCENT report warning exit code at/above this % [80] +# CRITICAL_PERCENT report critical exit code at/above this % [90] +# DRY_RUN true = print deletes but do not delete [false] +set -euo pipefail + +: "${CI_HOST:?set CI_HOST to the ssh target of your CI host, e.g. CI_HOST=my-ci-host}" +mode="${1:-report}" +case "$mode" in report|prune) ;; *) echo "usage: $0 report|prune" >&2; exit 2 ;; esac + +KUBECONFIG_REMOTE="${KUBECONFIG_REMOTE:-/etc/rancher/k3s/k3s.yaml}" +SECRET_NAMESPACE="${SECRET_NAMESPACE:-arc-runners}" +SECRET_NAME="${SECRET_NAME:-sccache-s3}" +RUSTFS_NAMESPACE="${RUSTFS_NAMESPACE:-sccache}" +RUSTFS_DEPLOYMENT="${RUSTFS_DEPLOYMENT:-rustfs}" +CACHE_PREFIX="${CACHE_PREFIX:-bun-cache/}" +MAX_AGE_DAYS="${MAX_AGE_DAYS:-30}" +KEEP_EXACT_PER_OS="${KEEP_EXACT_PER_OS:-3}" +PRUNE_TRIGGER_PERCENT="${PRUNE_TRIGGER_PERCENT:-80}" +TARGET_PERCENT="${TARGET_PERCENT:-70}" +WARN_PERCENT="${WARN_PERCENT:-80}" +CRITICAL_PERCENT="${CRITICAL_PERCENT:-90}" +DRY_RUN="${DRY_RUN:-false}" + +ssh "$CI_HOST" bash -s -- \ + "$mode" "$KUBECONFIG_REMOTE" "$SECRET_NAMESPACE" "$SECRET_NAME" "$RUSTFS_NAMESPACE" "$RUSTFS_DEPLOYMENT" \ + "$CACHE_PREFIX" "$MAX_AGE_DAYS" "$KEEP_EXACT_PER_OS" "$PRUNE_TRIGGER_PERCENT" "$TARGET_PERCENT" \ + "$WARN_PERCENT" "$CRITICAL_PERCENT" "$DRY_RUN" <<'REMOTE' +set -euo pipefail +MODE="$1" +export KUBECONFIG="$2" +SECRET_NAMESPACE="$3" +SECRET_NAME="$4" +RUSTFS_NAMESPACE="$5" +RUSTFS_DEPLOYMENT="$6" +CACHE_PREFIX="$7" +MAX_AGE_DAYS="$8" +KEEP_EXACT_PER_OS="$9" +PRUNE_TRIGGER_PERCENT="${10}" +TARGET_PERCENT="${11}" +WARN_PERCENT="${12}" +CRITICAL_PERCENT="${13}" +DRY_RUN="${14}" + +ACCESS_KEY="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.AWS_ACCESS_KEY_ID}' | base64 -d)" +SECRET_KEY="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.AWS_SECRET_ACCESS_KEY}' | base64 -d)" +BUCKET="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.SCCACHE_BUCKET}' | base64 -d)" +ENDPOINT="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.SCCACHE_ENDPOINT}' | base64 -d)" +REGION="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.SCCACHE_REGION}' | base64 -d)" +USE_SSL="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.SCCACHE_S3_USE_SSL}' | base64 -d)" +if [ "$USE_SSL" = "true" ]; then SCHEME=https; else SCHEME=http; fi + +# The secret intentionally stores the in-cluster Service DNS because runner pods +# consume it. This host-side maintenance script runs outside cluster DNS, so talk +# to the same Service by ClusterIP instead when the endpoint points at `.svc`. +if [[ "$ENDPOINT" == *.svc.*:* || "$ENDPOINT" == *.svc:* ]]; then + svc_ip="$(kubectl get svc "$RUSTFS_DEPLOYMENT" -n "$RUSTFS_NAMESPACE" -o jsonpath='{.spec.clusterIP}')" + svc_port="${ENDPOINT##*:}" + ENDPOINT="${svc_ip}:${svc_port}" +fi + +usage_line() { + kubectl exec -n "$RUSTFS_NAMESPACE" deploy/"$RUSTFS_DEPLOYMENT" -- df -P /data | tail -1 +} + +before="$(usage_line)" +TOTAL_KIB="$(awk '{print $2}' <<<"$before")" +USED_KIB="$(awk '{print $3}' <<<"$before")" +AVAIL_KIB="$(awk '{print $4}' <<<"$before")" +USED_PCT_RAW="$(awk '{print $5}' <<<"$before")" +USED_PCT="${USED_PCT_RAW%%%}" + +echo "==> RustFS PVC before" +echo "$before" + +summary="$(python3 - "$MODE" "$SCHEME" "$ENDPOINT" "$BUCKET" "$REGION" "$ACCESS_KEY" "$SECRET_KEY" "$CACHE_PREFIX" "$MAX_AGE_DAYS" "$KEEP_EXACT_PER_OS" "$PRUNE_TRIGGER_PERCENT" "$TARGET_PERCENT" "$DRY_RUN" "$TOTAL_KIB" "$USED_KIB" "$USED_PCT" <<'PY' +from __future__ import annotations +from collections import defaultdict +from datetime import datetime, timedelta, timezone +import json +import re +import subprocess +import sys +import urllib.parse +import xml.etree.ElementTree as ET + +mode, scheme, endpoint, bucket, region, access, secret, prefix, max_age_days, keep_per_os, prune_trigger, target_pct, dry_run, total_kib, used_kib, used_pct = sys.argv[1:] +max_age_days = int(max_age_days) +keep_per_os = int(keep_per_os) +prune_trigger = int(prune_trigger) +target_pct = int(target_pct) +dry_run = dry_run.lower() == "true" +total_kib = int(total_kib) +used_kib = int(used_kib) +used_pct = int(used_pct) + +prefix = prefix.rstrip("/") + "/" +base = f"{scheme}://{endpoint}/{bucket}" +now = datetime.now(timezone.utc) +cutoff = now - timedelta(days=max_age_days) + + +def curl(method: str, url: str, extra: list[str] | None = None) -> str: + cmd = [ + "curl", "-fsS", + "--aws-sigv4", f"aws:amz:{region}:s3", + "--user", f"{access}:{secret}", + "-X", method, + ] + if extra: + cmd.extend(extra) + cmd.append(url) + return subprocess.check_output(cmd, text=True) + + +def list_objects() -> list[dict]: + objs: list[dict] = [] + token = None + while True: + q = {"list-type": "2", "prefix": prefix} + if token: + q["continuation-token"] = token + url = f"{base}/?{urllib.parse.urlencode(q)}" + root = ET.fromstring(curl("GET", url)) + for node in root.findall(".//{*}Contents"): + objs.append({ + "key": node.findtext("{*}Key", default=""), + "size": int(node.findtext("{*}Size", default="0")), + "etag": node.findtext("{*}ETag", default="").strip('"'), + "last_modified": datetime.fromisoformat(node.findtext("{*}LastModified", default="1970-01-01T00:00:00+00:00")), + }) + token = root.findtext(".//{*}NextContinuationToken") + if not token: + return objs + + +def human(n: int) -> str: + units = ["B", "KiB", "MiB", "GiB", "TiB"] + value = float(n) + for unit in units: + if value < 1024 or unit == units[-1]: + return f"{value:.1f}{unit}" + value /= 1024 + return f"{n}B" + +objs = list_objects() +exact_re = re.compile(rf"^{re.escape(prefix)}(store|nm)-([^-]+)-([0-9a-f]{{32}})\.(tzst|tgz)$") +latest_re = re.compile(rf"^{re.escape(prefix)}store-([^-]+)-latest\.(tzst|tgz)$") + +exacts: list[dict] = [] +latest_aliases: list[dict] = [] +other_prefix: list[dict] = [] +for obj in objs: + if m := exact_re.match(obj["key"]): + obj = obj | {"kind": m.group(1), "os": m.group(2), "lock": m.group(3), "codec": m.group(4)} + exacts.append(obj) + elif m := latest_re.match(obj["key"]): + obj = obj | {"kind": "store", "os": m.group(1), "lock": "latest", "codec": m.group(2)} + latest_aliases.append(obj) + else: + other_prefix.append(obj) + +protected = {obj["key"] for obj in latest_aliases} +by_group: dict[tuple[str, str], list[dict]] = defaultdict(list) +by_store_etag: dict[tuple[str, str, str], list[dict]] = defaultdict(list) +for obj in exacts: + by_group[(obj["kind"], obj["os"])] .append(obj) + if obj["kind"] == "store": + by_store_etag[(obj["os"], obj["codec"], obj["etag"])] .append(obj) + +for alias in latest_aliases: + matches = by_store_etag.get((alias["os"], alias["codec"], alias["etag"]), []) + if matches: + newest = max(matches, key=lambda o: o["last_modified"]) + protected.add(newest["key"]) + +for group, items in by_group.items(): + items.sort(key=lambda o: o["last_modified"], reverse=True) + for obj in items[:keep_per_os]: + protected.add(obj["key"]) + for obj in items: + if obj["last_modified"] >= cutoff: + protected.add(obj["key"]) + +eligible = [obj for obj in exacts if obj["key"] not in protected] +eligible.sort(key=lambda o: o["last_modified"]) +aged = [obj for obj in eligible if obj["last_modified"] < cutoff] + +selected: list[dict] = list(aged) +selected_keys = {obj["key"] for obj in selected} +need_reclaim_kib = max(0, used_kib - (total_kib * target_pct // 100)) if used_pct >= prune_trigger else 0 +reclaimed_kib = sum(obj["size"] // 1024 for obj in selected) +if need_reclaim_kib > reclaimed_kib: + for obj in eligible: + if obj["key"] in selected_keys: + continue + selected.append(obj) + selected_keys.add(obj["key"]) + reclaimed_kib += obj["size"] // 1024 + if reclaimed_kib >= need_reclaim_kib: + break + +selected.sort(key=lambda o: (o["kind"], o["os"], o["last_modified"], o["key"])) +delete_total = sum(obj["size"] for obj in selected) + +print(f"bun-cache objects: exact={len(exacts)} latest={len(latest_aliases)} other-prefix={len(other_prefix)} total={human(sum(o['size'] for o in objs))}") +print(f"policy: max_age_days={max_age_days} keep_exact_per_os={keep_per_os} prune_trigger={prune_trigger}% target={target_pct}%") +print(f"eligible exact objects: {len(eligible)} | selected for deletion: {len(selected)} | reclaimable: {human(delete_total)}") +for obj in selected[:20]: + age_days = int((now - obj['last_modified']).total_seconds() // 86400) + print(f" delete {obj['key']} age={age_days}d size={human(obj['size'])}") +if len(selected) > 20: + print(f" ... {len(selected) - 20} more") + +if mode == "prune": + for obj in selected: + if dry_run: + continue + curl("DELETE", f"{base}/{obj['key']}") + +print("JSON_SUMMARY=" + json.dumps({ + "exact": len(exacts), + "latest": len(latest_aliases), + "other": len(other_prefix), + "eligible": len(eligible), + "selected": len(selected), + "delete_bytes": delete_total, + "dry_run": dry_run, +})) +PY +)" + +echo "$summary" + +after="$(usage_line)" +AFTER_USED_PCT_RAW="$(awk '{print $5}' <<<"$after")" +AFTER_USED_PCT="${AFTER_USED_PCT_RAW%%%}" + +echo "==> RustFS PVC after" +echo "$after" + +if [ "$AFTER_USED_PCT" -ge "$CRITICAL_PERCENT" ]; then + echo "CRITICAL: rustfs-data usage ${AFTER_USED_PCT}% >= ${CRITICAL_PERCENT}%" >&2 + exit 2 +fi +if [ "$AFTER_USED_PCT" -ge "$WARN_PERCENT" ]; then + echo "WARNING: rustfs-data usage ${AFTER_USED_PCT}% >= ${WARN_PERCENT}%" >&2 + exit 1 +fi + +echo "OK: rustfs-data usage ${AFTER_USED_PCT}%" +REMOTE diff --git a/infra/tune-kata-runtime.sh b/infra/tune-kata-runtime.sh new file mode 100755 index 000000000..8c782a523 --- /dev/null +++ b/infra/tune-kata-runtime.sh @@ -0,0 +1,81 @@ +#!/usr/bin/env bash +# Patch the live Kata QEMU config on the CI host to match the runner pod's +# guaranteed boot shape and a larger virtiofsd worker pool, then smoke-test that +# a new kata-qemu pod still boots. Driven over SSH from this repo so the desired +# values stay version-controlled. +# +# Usage: +# CI_HOST=my-ci-host ./infra/tune-kata-runtime.sh +# +# Env knobs: +# CI_HOST ssh target of the CI host (required) +# KATA_CONFIG_REMOTE remote Kata config file [/opt/kata/share/defaults/kata-containers/configuration-qemu.toml] +# KUBECONFIG_REMOTE kubeconfig path on the host [/etc/rancher/k3s/k3s.yaml] +# ARC_RELEASE runner scale set name [omp-kata] +# ARC_NAMESPACE runner namespace [arc-runners] +# BOOT_VCPUS Kata default_vcpus [2] +# BOOT_MEMORY_MIB Kata default_memory (MiB) [4096] +# VIRTIOFSD_THREAD_POOL virtiofsd --thread-pool-size [4] +set -euo pipefail + +: "${CI_HOST:?set CI_HOST to the ssh target of your CI host, e.g. CI_HOST=my-ci-host}" +KATA_CONFIG_REMOTE="${KATA_CONFIG_REMOTE:-/opt/kata/share/defaults/kata-containers/configuration-qemu.toml}" +KUBECONFIG_REMOTE="${KUBECONFIG_REMOTE:-/etc/rancher/k3s/k3s.yaml}" +ARC_RELEASE="${ARC_RELEASE:-omp-kata}" +ARC_NAMESPACE="${ARC_NAMESPACE:-arc-runners}" +BOOT_VCPUS="${BOOT_VCPUS:-2}" +BOOT_MEMORY_MIB="${BOOT_MEMORY_MIB:-4096}" +VIRTIOFSD_THREAD_POOL="${VIRTIOFSD_THREAD_POOL:-4}" + +ssh "$CI_HOST" bash -s -- \ + "$KATA_CONFIG_REMOTE" "$KUBECONFIG_REMOTE" "$ARC_RELEASE" "$ARC_NAMESPACE" \ + "$BOOT_VCPUS" "$BOOT_MEMORY_MIB" "$VIRTIOFSD_THREAD_POOL" <<'REMOTE' +set -euo pipefail +KATA_CONFIG="$1" +export KUBECONFIG="$2" +ARC_RELEASE="$3" +ARC_NAMESPACE="$4" +BOOT_VCPUS="$5" +BOOT_MEMORY_MIB="$6" +THREAD_POOL="$7" + +backup="${KATA_CONFIG}.bak.$(date +%Y%m%d-%H%M%S)" +cp "$KATA_CONFIG" "$backup" +echo "==> backup: $backup" + +python3 - "$KATA_CONFIG" "$BOOT_VCPUS" "$BOOT_MEMORY_MIB" "$THREAD_POOL" <<'PY' +from pathlib import Path +import re +import sys +path = Path(sys.argv[1]) +boot_vcpus = sys.argv[2] +boot_mem = sys.argv[3] +thread_pool = sys.argv[4] +text = path.read_text() +replacements = [ + (r'(^\s*default_vcpus\s*=\s*)\d+', rf'\g<1>{boot_vcpus}'), + (r'(^\s*default_memory\s*=\s*)\d+', rf'\g<1>{boot_mem}'), + (r'(^\s*virtio_fs_extra_args\s*=\s*)\[[^\]]*\]', rf'\g<1>["--thread-pool-size={thread_pool}", "--announce-submounts"]'), +] +for pattern, replacement in replacements: + text, n = re.subn(pattern, replacement, text, count=1, flags=re.MULTILINE) + if n != 1: + raise SystemExit(f"failed to patch {pattern}") +path.write_text(text) +PY + +echo "==> active Kata knobs" +grep -nE 'default_vcpus|default_memory|virtio_fs_extra_args' "$KATA_CONFIG" + +image="$(kubectl get autoscalingrunnerset "$ARC_RELEASE" -n "$ARC_NAMESPACE" -o jsonpath='{.spec.template.spec.containers[0].image}')" +pod="kata-runtime-smoke-$(date +%H%M%S)" +trap 'kubectl delete pod "$pod" -n "$ARC_NAMESPACE" --ignore-not-found >/dev/null 2>&1 || true' EXIT + +echo "==> smoke boot via kata-qemu using $image" +kubectl run "$pod" -n "$ARC_NAMESPACE" --restart=Never --image="$image" \ + --overrides='{"spec":{"runtimeClassName":"kata-qemu"}}' \ + --command -- bash -lc 'sleep 120' >/dev/null +kubectl wait --for=condition=Ready "pod/$pod" -n "$ARC_NAMESPACE" --timeout=120s >/dev/null +kubectl exec -n "$ARC_NAMESPACE" "$pod" -- bash -lc 'bun --version; rustc --version | head -1' +echo "OK: kata-qemu still boots after tuning" +REMOTE