Merge remote-tracking branch 'upstream/main' into feat/profiles-and-alias
# Conflicts: # .github/actions/build-native/action.yml # .github/workflows/ci.yml
This commit is contained in:
@@ -0,0 +1,134 @@
|
||||
name: Build native addon (omp-kata)
|
||||
description: >
|
||||
Build the pi_natives cdylib on the preloaded omp-kata runner image, using
|
||||
baked toolchains and the shared RustFS-backed sccache instead of per-job tool
|
||||
setup downloads.
|
||||
|
||||
inputs:
|
||||
hash:
|
||||
description: Rust source hash used in the artifact name
|
||||
required: true
|
||||
platform:
|
||||
description: Target platform (linux, darwin, win32)
|
||||
required: true
|
||||
arch:
|
||||
description: Target arch (x64, arm64)
|
||||
required: true
|
||||
variant:
|
||||
description: Optional build variant (baseline, modern); required for native x64 builds.
|
||||
required: false
|
||||
default: ""
|
||||
target:
|
||||
description: Optional rustc target triple for cross-compilation
|
||||
required: false
|
||||
default: ""
|
||||
rust_checks:
|
||||
description: Run clippy/rustfmt checks (only one matrix entry should set this)
|
||||
required: false
|
||||
default: "false"
|
||||
save_cache:
|
||||
description: Kept for interface parity with the GitHub-hosted action; unused here.
|
||||
required: false
|
||||
default: "false"
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- uses: ./.github/actions/ensure-rust-toolchain
|
||||
with:
|
||||
toolchain: nightly-2026-04-29
|
||||
components: ${{ inputs.rust_checks == 'true' && 'clippy,rustfmt' || '' }}
|
||||
target: ${{ inputs.target }}
|
||||
- name: Configure native Rust flags
|
||||
if: inputs.target == ''
|
||||
shell: bash
|
||||
env:
|
||||
TARGET_ARCH: ${{ inputs.arch }}
|
||||
TARGET_VARIANT: ${{ inputs.variant }}
|
||||
run: |
|
||||
case "$TARGET_ARCH:$TARGET_VARIANT" in
|
||||
x64:modern)
|
||||
rustflags="-C target-cpu=x86-64-v3"
|
||||
;;
|
||||
x64:baseline)
|
||||
rustflags="-C target-cpu=x86-64-v2"
|
||||
;;
|
||||
x64:*)
|
||||
echo "::error::x64 native builds require variant=modern or variant=baseline"
|
||||
exit 1
|
||||
;;
|
||||
*)
|
||||
if [ -n "${RUSTFLAGS:-}" ]; then
|
||||
echo "Using caller-provided RUSTFLAGS=$RUSTFLAGS"
|
||||
exit 0
|
||||
fi
|
||||
rustflags="-C target-cpu=native"
|
||||
;;
|
||||
esac
|
||||
|
||||
echo "RUSTFLAGS=$rustflags" >> "$GITHUB_ENV"
|
||||
echo "Configured RUSTFLAGS=$rustflags"
|
||||
- uses: ./.github/actions/ensure-sccache
|
||||
with:
|
||||
version: "0.15.0"
|
||||
- name: Enable sccache for cargo
|
||||
shell: bash
|
||||
run: |
|
||||
{
|
||||
echo "RUSTC_WRAPPER=sccache"
|
||||
echo "CARGO_INCREMENTAL=0"
|
||||
} >> "$GITHUB_ENV"
|
||||
echo "sccache backend: shared S3 ($SCCACHE_BUCKET @ $SCCACHE_ENDPOINT)"
|
||||
- uses: ./.github/actions/ensure-cargo-tool
|
||||
if: inputs.target == ''
|
||||
with:
|
||||
binary: cargo-nextest
|
||||
crate: cargo-nextest
|
||||
- uses: ./.github/actions/bun-install
|
||||
- uses: ./.github/actions/ensure-zig
|
||||
if: inputs.target != '' && !endsWith(inputs.target, '-msvc')
|
||||
with:
|
||||
version: "0.16.0"
|
||||
- uses: ./.github/actions/ensure-cargo-tool
|
||||
if: inputs.target != '' && !endsWith(inputs.target, '-msvc')
|
||||
with:
|
||||
binary: cargo-zigbuild
|
||||
crate: cargo-zigbuild
|
||||
- uses: ./.github/actions/ensure-cargo-tool
|
||||
if: endsWith(inputs.target, '-msvc')
|
||||
with:
|
||||
binary: cargo-xwin
|
||||
crate: cargo-xwin
|
||||
- name: Cache cargo-xwin Windows SDK
|
||||
if: endsWith(inputs.target, '-msvc')
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.cache/cargo-xwin
|
||||
key: cargo-xwin-${{ runner.os }}-v1
|
||||
- name: Accept xwin license
|
||||
if: endsWith(inputs.target, '-msvc')
|
||||
shell: bash
|
||||
run: echo "XWIN_ACCEPT_LICENSE=1" >> "$GITHUB_ENV"
|
||||
- name: Rust checks
|
||||
if: inputs.rust_checks == 'true'
|
||||
shell: bash
|
||||
run: bun run check:rs
|
||||
- name: Test workspace (Rust)
|
||||
if: inputs.target == '' && inputs.platform != 'darwin'
|
||||
shell: bash
|
||||
run: bun run test:rs
|
||||
- name: Build native addon(s)
|
||||
shell: bash
|
||||
env:
|
||||
CROSS_TARGET: ${{ inputs.target }}
|
||||
TARGET_PLATFORM: ${{ inputs.platform }}
|
||||
TARGET_ARCH: ${{ inputs.arch }}
|
||||
TARGET_VARIANTS: ${{ inputs.variant }}
|
||||
run: bun run ci:build:native
|
||||
- name: Upload native addon(s)
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: pi-natives-${{ inputs.platform }}-${{ inputs.arch }}${{ inputs.variant && format('-{0}', inputs.variant) || '' }}-h${{ inputs.hash }}
|
||||
path: packages/natives/native/pi_natives.${{ inputs.platform }}-${{ inputs.arch }}*.node
|
||||
if-no-files-found: error
|
||||
retention-days: 90
|
||||
@@ -53,18 +53,19 @@ runs:
|
||||
echo "$toolchain_bin" >> "$GITHUB_PATH"
|
||||
echo "Prepended $toolchain_bin to PATH"
|
||||
# `Swatinem/rust-cache` keys target/ off its restore-time environment, so
|
||||
# set RUSTFLAGS before restoring it. If x64 target-cpu is only selected
|
||||
# inside ci-build-native.ts/build-native.ts, cargo invalidates the restored
|
||||
# target/ but rust-cache sees an exact key and refuses to save the rebuilt
|
||||
# artifacts, causing macOS x64 baseline to rebuild forever.
|
||||
# set RUSTFLAGS before deciding whether to restore it. If x64 target-cpu is
|
||||
# only selected inside ci-build-native.ts/build-native.ts, cargo invalidates
|
||||
# the restored target/ but rust-cache sees an exact key and refuses to save
|
||||
# the rebuilt artifacts, causing macOS x64 baseline to rebuild forever.
|
||||
#
|
||||
# Include the native source hash in the shared key as well: rust-cache's
|
||||
# lockfile scan misses the workspace root Cargo.toml version that Cargo
|
||||
# fingerprints for workspace crates. Without it, release version bumps can
|
||||
# get an exact hit for artifacts Cargo must rebuild.
|
||||
#
|
||||
# sccache is still layered on top of rust-cache: target/ wins when warm,
|
||||
# sccache fills the gaps when target/ is cold.
|
||||
# On GitHub-hosted runners, rust-cache can still warm target/. On omp-kata,
|
||||
# the shared RustFS-backed sccache is the primary reuse layer and avoids a
|
||||
# second GitHub-cache restore for Cargo dirs/target.
|
||||
- name: Configure native Rust flags
|
||||
if: inputs.target == ''
|
||||
shell: bash
|
||||
@@ -94,7 +95,19 @@ runs:
|
||||
|
||||
echo "RUSTFLAGS=$rustflags" >> "$GITHUB_ENV"
|
||||
echo "Configured RUSTFLAGS=$rustflags"
|
||||
- name: Decide Rust artifact cache
|
||||
id: rust_artifact_cache
|
||||
shell: bash
|
||||
run: |
|
||||
if [ -n "${SCCACHE_BUCKET:-}" ]; then
|
||||
echo "use_rust_cache=false" >> "$GITHUB_OUTPUT"
|
||||
echo "Rust target cache: skipped on shared-sccache runners"
|
||||
else
|
||||
echo "use_rust_cache=true" >> "$GITHUB_OUTPUT"
|
||||
echo "Rust target cache: GitHub Actions"
|
||||
fi
|
||||
- uses: Swatinem/rust-cache@v2
|
||||
if: steps.rust_artifact_cache.outputs.use_rust_cache == 'true'
|
||||
with:
|
||||
shared-key: native-${{ inputs.platform }}-${{ inputs.arch }}-${{ inputs.variant || 'default' }}-h${{ inputs.hash }}
|
||||
cache-on-failure: true
|
||||
@@ -126,14 +139,7 @@ runs:
|
||||
if: inputs.target == ''
|
||||
with:
|
||||
tool: nextest
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3"
|
||||
env:
|
||||
SCCACHE_BUCKET: ""
|
||||
AWS_ACCESS_KEY_ID: ""
|
||||
- shell: bash
|
||||
run: bun install --frozen-lockfile
|
||||
- uses: ./.github/actions/bun-install
|
||||
# Cross-compile toolchain selection: non-MSVC targets (e.g.
|
||||
# `aarch64-unknown-linux-gnu`) build with `cargo-zigbuild`; MSVC targets
|
||||
# (e.g. `x86_64-pc-windows-msvc`) build with `cargo-xwin`. The napi CLI's
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
name: "bun install (shared cache)"
|
||||
description: >
|
||||
Ensure bun is on PATH, then run `bun install --frozen-lockfile` with a shared
|
||||
dependency cache. bun setup is skipped when the runner image already ships it
|
||||
(the preloaded omp-kata image), and only fetched on runners that lack it
|
||||
(e.g. GitHub-hosted). The cache uses the in-cluster RustFS S3 when the sccache
|
||||
credentials are present, and the stock actions/cache backend otherwise.
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- name: Detect environment
|
||||
id: env
|
||||
shell: bash
|
||||
run: |
|
||||
# bun is baked into the omp-kata runner image — only fetch it when the
|
||||
# runner doesn't already provide it (GitHub-hosted), so we don't
|
||||
# re-download bun from GitHub releases on every job.
|
||||
if command -v bun >/dev/null 2>&1; then
|
||||
echo "setup_bun=false" >> "$GITHUB_OUTPUT"
|
||||
echo "bun present: $(bun --version) (preloaded image)"
|
||||
else
|
||||
echo "setup_bun=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
# The repo keys "are we on can.internal infra?" off $SCCACHE_BUCKET (see
|
||||
# actions/build-native). RUNNER_ENVIRONMENT is empty on ARC pods, so it
|
||||
# is not a usable signal here.
|
||||
if [ -n "${SCCACHE_BUCKET:-}" ] && [ -n "${AWS_ACCESS_KEY_ID:-}" ]; then
|
||||
echo "cache=rustfs" >> "$GITHUB_OUTPUT"
|
||||
echo "bun cache backend: RustFS S3 ($SCCACHE_BUCKET @ $SCCACHE_ENDPOINT)"
|
||||
else
|
||||
echo "cache=gha" >> "$GITHUB_OUTPUT"
|
||||
echo "bun cache backend: GitHub Actions cache"
|
||||
fi
|
||||
- name: Install bun when absent
|
||||
if: steps.env.outputs.setup_bun == 'true'
|
||||
shell: bash
|
||||
run: |
|
||||
export BUN_INSTALL="${HOME}/.bun"
|
||||
curl -fsSL https://bun.sh/install | bash -s "bun-v1.3.14"
|
||||
echo "${BUN_INSTALL}/bin" >> "$GITHUB_PATH"
|
||||
|
||||
# Off-infra (GitHub-hosted): stock actions/cache for the bun store.
|
||||
- name: Cache bun store (GitHub cache)
|
||||
if: steps.env.outputs.cache == 'gha'
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.bun/install/cache
|
||||
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
|
||||
|
||||
# On-infra (omp-kata): RustFS store + node_modules over the LAN.
|
||||
- name: Restore bun caches (RustFS)
|
||||
if: steps.env.outputs.cache == 'rustfs'
|
||||
shell: bash
|
||||
run: |
|
||||
bash "$GITHUB_ACTION_PATH/rustfs-cache.sh" restore || {
|
||||
echo "::warning::RustFS bun cache restore failed; continuing without shared cache"
|
||||
}
|
||||
|
||||
- name: Install dependencies
|
||||
shell: bash
|
||||
run: bun install --frozen-lockfile
|
||||
|
||||
- name: Save bun caches (RustFS)
|
||||
if: steps.env.outputs.cache == 'rustfs'
|
||||
shell: bash
|
||||
run: |
|
||||
bash "$GITHUB_ACTION_PATH/rustfs-cache.sh" save || {
|
||||
echo "::warning::RustFS bun cache save failed; continuing without shared cache"
|
||||
}
|
||||
Executable
+126
@@ -0,0 +1,126 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared bun dependency cache backed by the in-cluster RustFS (S3) object store.
|
||||
#
|
||||
# Used by .github/actions/bun-install on the self-hosted omp-kata runners. There
|
||||
# the stock `actions/cache` restore of ~/.bun/install/cache costs 130-186s per
|
||||
# job because GitHub's cache backend is only reachable over the node's NAT
|
||||
# egress, and ~9 jobs contend on it at once. RustFS lives in the same k3s node
|
||||
# (svc :9000, already allowed by the runner egress NetworkPolicy), so the same
|
||||
# payload moves at LAN speed.
|
||||
#
|
||||
# Credentials are the ones sccache already gets via the `sccache-s3` secret
|
||||
# (envFrom on every runner pod): AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY /
|
||||
# SCCACHE_ENDPOINT / SCCACHE_BUCKET / SCCACHE_REGION / SCCACHE_S3_USE_SSL.
|
||||
#
|
||||
# Two objects per lockfile, under the bun-cache/ key prefix of the sccache
|
||||
# bucket:
|
||||
# store-<os>-<lockhash> the bun global package store (~/.bun/install/cache)
|
||||
# nm-<os>-<lockhash> the installed node_modules trees (root + workspaces)
|
||||
# The store additionally publishes a rolling store-<os>-latest alias, so a
|
||||
# changed lockfile still warm-starts from the previous store and `bun install`
|
||||
# only fetches the delta. A node_modules hit short-circuits everything: the
|
||||
# subsequent `bun install --frozen-lockfile` is a no-op, so the store is neither
|
||||
# fetched nor saved.
|
||||
set -euo pipefail
|
||||
|
||||
mode="${1:?usage: rustfs-cache.sh restore|save}"
|
||||
|
||||
: "${SCCACHE_BUCKET:?SCCACHE_BUCKET required}"
|
||||
: "${SCCACHE_ENDPOINT:?SCCACHE_ENDPOINT required}"
|
||||
: "${AWS_ACCESS_KEY_ID:?AWS_ACCESS_KEY_ID required}"
|
||||
: "${AWS_SECRET_ACCESS_KEY:?AWS_SECRET_ACCESS_KEY required}"
|
||||
|
||||
region="${SCCACHE_REGION:-us-east-1}"
|
||||
if [ "${SCCACHE_S3_USE_SSL:-false}" = "true" ]; then scheme=https; else scheme=http; fi
|
||||
base="${scheme}://${SCCACHE_ENDPOINT}/${SCCACHE_BUCKET}/bun-cache"
|
||||
os="${RUNNER_OS:-$(uname -s)}"
|
||||
store_dir="${BUN_INSTALL_CACHE_DIR:-${HOME}/.bun/install/cache}"
|
||||
work="${RUNNER_TEMP:-/tmp}/bun-rustfs-cache"
|
||||
mkdir -p "$work"
|
||||
|
||||
# Prefer multi-threaded zstd (baked into the omp-kata runner image); fall back to
|
||||
# gzip so the action still works on an image that predates the zstd addition. The
|
||||
# object suffix records the codec, and restore only inflates archives this host
|
||||
# can actually decompress.
|
||||
if command -v zstd >/dev/null 2>&1; then
|
||||
tar_c=(-I "zstd -3 -T0"); ext="tzst"; alt_ext="tgz"
|
||||
else
|
||||
tar_c=(-I "gzip -6"); ext="tgz"; alt_ext="tzst"
|
||||
fi
|
||||
|
||||
lock_hash="$(sha256sum bun.lock | cut -c1-32)"
|
||||
store_key="store-${os}-${lock_hash}"
|
||||
store_latest="store-${os}-latest"
|
||||
nm_key="nm-${os}-${lock_hash}"
|
||||
|
||||
auth=(--aws-sigv4 "aws:amz:${region}:s3" --user "${AWS_ACCESS_KEY_ID}:${AWS_SECRET_ACCESS_KEY}")
|
||||
# 404 (-f) and connection errors are non-zero; transient errors retry, 4xx do not.
|
||||
s3_get() { curl -fsS --retry 3 --retry-connrefused "${auth[@]}" "${base}/$1" -o "$2"; }
|
||||
s3_exists() { curl -fsS -I --retry 3 --retry-connrefused "${auth[@]}" "${base}/$1" -o /dev/null >/dev/null 2>&1; }
|
||||
s3_put() { curl -fsS --retry 3 --retry-connrefused "${auth[@]}" -T "$2" "${base}/$1" -o /dev/null; }
|
||||
|
||||
# Download <name>.<ext> (then the alternate codec) and extract into dir $2.
|
||||
# tar auto-detects the codec from the archive; a present-but-uninflatable archive
|
||||
# (codec mismatch with this host) is treated as a miss.
|
||||
fetch_extract() { # name dest
|
||||
local name="$1" dest="$2" e f
|
||||
for e in "$ext" "$alt_ext"; do
|
||||
f="${work}/${name}.${e}"
|
||||
if s3_get "${name}.${e}" "$f" 2>/dev/null; then
|
||||
mkdir -p "$dest"
|
||||
if tar -xf "$f" -C "$dest" 2>/dev/null; then rm -f "$f"; return 0; fi
|
||||
rm -f "$f"
|
||||
fi
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
case "$mode" in
|
||||
restore)
|
||||
if fetch_extract "$nm_key" "$PWD"; then
|
||||
echo "bun cache: node_modules HIT ($nm_key) — install becomes a no-op"
|
||||
: > "${work}/nm_hit"
|
||||
exit 0
|
||||
fi
|
||||
echo "bun cache: node_modules miss ($nm_key)"
|
||||
if fetch_extract "$store_key" "$store_dir"; then
|
||||
echo "bun cache: store HIT ($store_key)"
|
||||
elif fetch_extract "$store_latest" "$store_dir"; then
|
||||
echo "bun cache: store warm-start ($store_latest)"
|
||||
else
|
||||
echo "bun cache: store miss — cold install"
|
||||
fi
|
||||
;;
|
||||
save)
|
||||
if [ -f "${work}/nm_hit" ]; then
|
||||
echo "bun cache: node_modules was a hit — nothing to save"
|
||||
exit 0
|
||||
fi
|
||||
# Store (+ rolling latest): save when this exact lockfile has none yet.
|
||||
if [ -d "$store_dir" ] && ! s3_exists "${store_key}.${ext}"; then
|
||||
tar "${tar_c[@]}" -cf "${work}/store.${ext}" -C "$store_dir" .
|
||||
s3_put "${store_key}.${ext}" "${work}/store.${ext}"
|
||||
s3_put "${store_latest}.${ext}" "${work}/store.${ext}"
|
||||
rm -f "${work}/store.${ext}"
|
||||
echo "bun cache: saved store ($store_key + $store_latest)"
|
||||
fi
|
||||
# node_modules: save the installed trees for this exact lockfile.
|
||||
if ! s3_exists "${nm_key}.${ext}"; then
|
||||
shopt -s nullglob
|
||||
nm_paths=()
|
||||
for p in node_modules packages/*/node_modules python/robomp/web/node_modules; do
|
||||
[ -d "$p" ] && nm_paths+=("$p")
|
||||
done
|
||||
if [ ${#nm_paths[@]} -gt 0 ]; then
|
||||
tar "${tar_c[@]}" -cf "${work}/nm.${ext}" "${nm_paths[@]}"
|
||||
s3_put "${nm_key}.${ext}" "${work}/nm.${ext}"
|
||||
rm -f "${work}/nm.${ext}"
|
||||
echo "bun cache: saved node_modules ($nm_key, ${#nm_paths[@]} trees)"
|
||||
fi
|
||||
fi
|
||||
;;
|
||||
*)
|
||||
echo "rustfs-cache.sh: unknown mode '$mode' (want restore|save)" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
@@ -0,0 +1,38 @@
|
||||
name: "ensure cargo tool"
|
||||
description: Ensure a cargo-installed CLI is present.
|
||||
|
||||
inputs:
|
||||
binary:
|
||||
required: true
|
||||
description: Binary name expected on PATH
|
||||
crate:
|
||||
required: false
|
||||
default: ""
|
||||
description: Crate name to cargo install; defaults to the binary name
|
||||
version:
|
||||
required: false
|
||||
default: ""
|
||||
description: Optional crate version
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- shell: bash
|
||||
env:
|
||||
BINARY: ${{ inputs.binary }}
|
||||
CRATE: ${{ inputs.crate }}
|
||||
VERSION: ${{ inputs.version }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if command -v "$BINARY" >/dev/null 2>&1; then
|
||||
echo "Using baked cargo tool: $BINARY"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
crate="${CRATE:-$BINARY}"
|
||||
install_args=(install --locked "$crate")
|
||||
if [ -n "$VERSION" ]; then
|
||||
install_args+=(--version "$VERSION")
|
||||
fi
|
||||
cargo "${install_args[@]}"
|
||||
echo "Installed cargo tool: $BINARY"
|
||||
@@ -0,0 +1,59 @@
|
||||
name: "ensure rust toolchain"
|
||||
description: >
|
||||
Ensure a pinned rustup toolchain, optional components, and an optional target
|
||||
are present, then prepend the real toolchain bin dir to PATH.
|
||||
|
||||
inputs:
|
||||
toolchain:
|
||||
required: true
|
||||
description: Rust toolchain name (for example nightly-2026-04-29)
|
||||
components:
|
||||
required: false
|
||||
default: ""
|
||||
description: Optional comma-separated rustup components
|
||||
target:
|
||||
required: false
|
||||
default: ""
|
||||
description: Optional rustup target triple
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- shell: bash
|
||||
env:
|
||||
TOOLCHAIN: ${{ inputs.toolchain }}
|
||||
COMPONENTS: ${{ inputs.components }}
|
||||
TARGET: ${{ inputs.target }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if ! command -v rustup >/dev/null 2>&1; then
|
||||
curl --proto '=https' --tlsv1.2 -fsSL https://sh.rustup.rs \
|
||||
| sh -s -- -y --default-toolchain "$TOOLCHAIN" --profile minimal
|
||||
fi
|
||||
|
||||
if ! rustc +"$TOOLCHAIN" --version >/dev/null 2>&1; then
|
||||
rustup toolchain install "$TOOLCHAIN" --profile minimal --no-self-update
|
||||
fi
|
||||
rustup default "$TOOLCHAIN"
|
||||
|
||||
missing_components=()
|
||||
if [ -n "$COMPONENTS" ]; then
|
||||
IFS=',' read -r -a wanted_components <<< "$COMPONENTS"
|
||||
for component in "${wanted_components[@]}"; do
|
||||
[ -z "$component" ] && continue
|
||||
if ! rustup component list --toolchain "$TOOLCHAIN" --installed | grep -qE "^${component}(-|$)"; then
|
||||
missing_components+=("$component")
|
||||
fi
|
||||
done
|
||||
fi
|
||||
if [ ${#missing_components[@]} -gt 0 ]; then
|
||||
rustup component add --toolchain "$TOOLCHAIN" "${missing_components[@]}"
|
||||
fi
|
||||
|
||||
if [ -n "$TARGET" ] && ! rustup target list --toolchain "$TOOLCHAIN" --installed | grep -qx "$TARGET"; then
|
||||
rustup target add --toolchain "$TOOLCHAIN" "$TARGET"
|
||||
fi
|
||||
|
||||
toolchain_bin="$(dirname "$(rustup which cargo --toolchain "$TOOLCHAIN")")"
|
||||
echo "$toolchain_bin" >> "$GITHUB_PATH"
|
||||
echo "Using Rust toolchain: $(rustc +"$TOOLCHAIN" --version)"
|
||||
@@ -0,0 +1,47 @@
|
||||
name: "ensure sccache"
|
||||
description: Ensure a pinned sccache binary is on PATH.
|
||||
|
||||
inputs:
|
||||
version:
|
||||
required: false
|
||||
default: "0.15.0"
|
||||
description: sccache release version without the leading v
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- shell: bash
|
||||
env:
|
||||
SCCACHE_VERSION: ${{ inputs.version }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if command -v sccache >/dev/null 2>&1; then
|
||||
current="$(sccache --version | awk '{print $2}')"
|
||||
if [ "$current" = "$SCCACHE_VERSION" ]; then
|
||||
echo "Using baked sccache $current"
|
||||
exit 0
|
||||
fi
|
||||
fi
|
||||
|
||||
case "$(uname -s)-$(uname -m)" in
|
||||
Linux-x86_64) triple=x86_64-unknown-linux-musl ;;
|
||||
Darwin-arm64) triple=aarch64-apple-darwin ;;
|
||||
Darwin-x86_64) triple=x86_64-apple-darwin ;;
|
||||
*) triple="" ;;
|
||||
esac
|
||||
|
||||
if [ -n "$triple" ]; then
|
||||
url="https://github.com/mozilla/sccache/releases/download/v${SCCACHE_VERSION}/sccache-v${SCCACHE_VERSION}-${triple}.tar.gz"
|
||||
tmpdir="${RUNNER_TEMP:-/tmp}/sccache-${SCCACHE_VERSION}"
|
||||
bindir="${HOME}/.local/bin"
|
||||
rm -rf "$tmpdir"
|
||||
mkdir -p "$tmpdir" "$bindir"
|
||||
curl -fsSL "$url" | tar -xz -C "$tmpdir"
|
||||
install -m755 "$tmpdir"/sccache-v${SCCACHE_VERSION}-${triple}/sccache "$bindir/sccache"
|
||||
echo "$bindir" >> "$GITHUB_PATH"
|
||||
echo "Installed sccache $("$bindir/sccache" --version)"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
cargo install --locked sccache --version "$SCCACHE_VERSION"
|
||||
echo "Installed sccache $(sccache --version)"
|
||||
@@ -0,0 +1,39 @@
|
||||
name: "ensure zig"
|
||||
description: Ensure a pinned Zig binary is on PATH.
|
||||
|
||||
inputs:
|
||||
version:
|
||||
required: true
|
||||
description: Zig release version
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- shell: bash
|
||||
env:
|
||||
ZIG_VERSION: ${{ inputs.version }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if command -v zig >/dev/null 2>&1 && [ "$(zig version)" = "$ZIG_VERSION" ]; then
|
||||
echo "Using baked zig $ZIG_VERSION"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
case "$(uname -s)-$(uname -m)" in
|
||||
Linux-x86_64) archive="zig-x86_64-linux-${ZIG_VERSION}" ;;
|
||||
Darwin-arm64) archive="zig-aarch64-macos-${ZIG_VERSION}" ;;
|
||||
Darwin-x86_64) archive="zig-x86_64-macos-${ZIG_VERSION}" ;;
|
||||
*)
|
||||
echo "Unsupported zig host: $(uname -s)-$(uname -m)" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
destdir="${HOME}/.local"
|
||||
mkdir -p "$destdir"
|
||||
rm -rf "${destdir:?}/${archive}"
|
||||
curl -fsSL "https://ziglang.org/download/${ZIG_VERSION}/${archive}.tar.xz" -o "${destdir}/zig.tar.xz"
|
||||
tar -xJf "${destdir}/zig.tar.xz" -C "$destdir"
|
||||
rm -f "${destdir}/zig.tar.xz"
|
||||
echo "${destdir}/${archive}" >> "$GITHUB_PATH"
|
||||
echo "Installed zig $("${destdir}/${archive}/zig" version)"
|
||||
+43
-126
@@ -190,18 +190,7 @@ jobs:
|
||||
runs-on: omp-kata
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3"
|
||||
env:
|
||||
SCCACHE_BUCKET: ""
|
||||
AWS_ACCESS_KEY_ID: ""
|
||||
- name: Cache bun dependencies
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.bun/install/cache
|
||||
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
|
||||
- run: bun install --frozen-lockfile
|
||||
- uses: ./.github/actions/bun-install
|
||||
- name: Type check workspace
|
||||
run: bun run ci:check:full
|
||||
- name: Build collab web
|
||||
@@ -223,7 +212,7 @@ jobs:
|
||||
- { variant: modern }
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: ./.github/actions/build-native
|
||||
- uses: ./.github/actions/build-native-kata
|
||||
with:
|
||||
hash: ${{ needs.native_artifact_lookup.outputs.source-hash }}
|
||||
platform: linux
|
||||
@@ -235,7 +224,7 @@ jobs:
|
||||
# Pre-warm the cross-platform native build cache on `main`, in addition to
|
||||
# building the artifacts that ship in releases. Skipped on main when
|
||||
# native_artifact_lookup already found a recent run with all artifacts intact.
|
||||
native_cross_platform:
|
||||
native_cross_platform_kata:
|
||||
name: "Native: ${{ matrix.platform }} ${{ matrix.arch }}"
|
||||
needs: [release_metadata, native_artifact_lookup]
|
||||
if: ${{ needs.release_metadata.outputs.is-release == 'true' || (github.event_name == 'push' && github.ref == 'refs/heads/main' && needs.native_artifact_lookup.outputs.cross-platform-run-id == '') }}
|
||||
@@ -244,9 +233,29 @@ jobs:
|
||||
matrix:
|
||||
include:
|
||||
- { os: omp-kata, platform: linux, arch: arm64, target: aarch64-unknown-linux-gnu }
|
||||
- { os: omp-kata, platform: win32, arch: x64, target: x86_64-pc-windows-msvc, variant: baseline }
|
||||
runs-on: ${{ matrix.os }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: ./.github/actions/build-native-kata
|
||||
with:
|
||||
hash: ${{ needs.native_artifact_lookup.outputs.source-hash }}
|
||||
platform: ${{ matrix.platform }}
|
||||
arch: ${{ matrix.arch }}
|
||||
variant: ${{ matrix.variant }}
|
||||
target: ${{ matrix.target }}
|
||||
save_cache: ${{ github.event_name == 'push' && github.ref == 'refs/heads/main' }}
|
||||
|
||||
native_cross_platform_macos:
|
||||
name: "Native: ${{ matrix.platform }} ${{ matrix.arch }}"
|
||||
needs: [release_metadata, native_artifact_lookup]
|
||||
if: ${{ needs.release_metadata.outputs.is-release == 'true' || (github.event_name == 'push' && github.ref == 'refs/heads/main' && needs.native_artifact_lookup.outputs.cross-platform-run-id == '') }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- { os: macos-15-intel, platform: darwin, arch: x64, variant: baseline }
|
||||
- { os: macos-14, platform: darwin, arch: arm64 }
|
||||
- { os: omp-kata, platform: win32, arch: x64, target: x86_64-pc-windows-msvc, variant: baseline }
|
||||
runs-on: ${{ matrix.os }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
@@ -258,7 +267,6 @@ jobs:
|
||||
variant: ${{ matrix.variant }}
|
||||
target: ${{ matrix.target }}
|
||||
save_cache: ${{ github.event_name == 'push' && github.ref == 'refs/heads/main' }}
|
||||
|
||||
test_workspace:
|
||||
name: Test TS workspace fast
|
||||
runs-on: omp-kata
|
||||
@@ -267,18 +275,7 @@ jobs:
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3"
|
||||
env:
|
||||
SCCACHE_BUCKET: ""
|
||||
AWS_ACCESS_KEY_ID: ""
|
||||
- name: Cache bun dependencies
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.bun/install/cache
|
||||
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
|
||||
- run: bun install --frozen-lockfile
|
||||
- uses: ./.github/actions/bun-install
|
||||
- name: Resolve Linux x64 native artifact run
|
||||
id: source
|
||||
shell: bash
|
||||
@@ -307,18 +304,7 @@ jobs:
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3"
|
||||
env:
|
||||
SCCACHE_BUCKET: ""
|
||||
AWS_ACCESS_KEY_ID: ""
|
||||
- name: Cache bun dependencies
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.bun/install/cache
|
||||
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
|
||||
- run: bun install --frozen-lockfile
|
||||
- uses: ./.github/actions/bun-install
|
||||
- name: Resolve Linux x64 native artifact run
|
||||
id: source
|
||||
shell: bash
|
||||
@@ -349,19 +335,9 @@ jobs:
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3"
|
||||
env:
|
||||
SCCACHE_BUCKET: ""
|
||||
AWS_ACCESS_KEY_ID: ""
|
||||
- name: Cache bun dependencies
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.bun/install/cache
|
||||
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
|
||||
|
||||
- uses: ./.github/actions/setup-system-deps
|
||||
- run: bun install --frozen-lockfile
|
||||
- uses: ./.github/actions/bun-install
|
||||
- name: Resolve Linux x64 native artifact run
|
||||
id: source
|
||||
shell: bash
|
||||
@@ -390,19 +366,9 @@ jobs:
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3"
|
||||
env:
|
||||
SCCACHE_BUCKET: ""
|
||||
AWS_ACCESS_KEY_ID: ""
|
||||
- name: Cache bun dependencies
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.bun/install/cache
|
||||
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
|
||||
|
||||
- uses: ./.github/actions/setup-system-deps
|
||||
- run: bun install --frozen-lockfile
|
||||
- uses: ./.github/actions/bun-install
|
||||
- name: Resolve Linux x64 native artifact run
|
||||
id: source
|
||||
shell: bash
|
||||
@@ -431,18 +397,7 @@ jobs:
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3"
|
||||
env:
|
||||
SCCACHE_BUCKET: ""
|
||||
AWS_ACCESS_KEY_ID: ""
|
||||
- name: Cache bun dependencies
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.bun/install/cache
|
||||
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
|
||||
- run: bun install --frozen-lockfile
|
||||
- uses: ./.github/actions/bun-install
|
||||
- name: Resolve Linux x64 native artifact run
|
||||
id: source
|
||||
shell: bash
|
||||
@@ -473,19 +428,9 @@ jobs:
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3"
|
||||
env:
|
||||
SCCACHE_BUCKET: ""
|
||||
AWS_ACCESS_KEY_ID: ""
|
||||
- name: Cache bun dependencies
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.bun/install/cache
|
||||
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
|
||||
|
||||
- uses: ./.github/actions/setup-system-deps
|
||||
- run: bun install --frozen-lockfile
|
||||
- uses: ./.github/actions/bun-install
|
||||
- name: Resolve Linux x64 native artifact run
|
||||
id: source
|
||||
shell: bash
|
||||
@@ -514,19 +459,9 @@ jobs:
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3"
|
||||
env:
|
||||
SCCACHE_BUCKET: ""
|
||||
AWS_ACCESS_KEY_ID: ""
|
||||
- name: Cache bun dependencies
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.bun/install/cache
|
||||
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
|
||||
|
||||
- uses: ./.github/actions/setup-system-deps
|
||||
- run: bun install --frozen-lockfile
|
||||
- uses: ./.github/actions/bun-install
|
||||
- name: Resolve Linux x64 native artifact run
|
||||
id: source
|
||||
shell: bash
|
||||
@@ -552,26 +487,12 @@ jobs:
|
||||
runs-on: omp-kata
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: "1.3"
|
||||
env:
|
||||
SCCACHE_BUCKET: ""
|
||||
AWS_ACCESS_KEY_ID: ""
|
||||
- uses: dtolnay/rust-toolchain@nightly
|
||||
- uses: ./.github/actions/ensure-rust-toolchain
|
||||
with:
|
||||
toolchain: nightly-2026-04-29
|
||||
- uses: Swatinem/rust-cache@v2
|
||||
- uses: ./.github/actions/ensure-sccache
|
||||
with:
|
||||
shared-key: install-methods-linux-x64
|
||||
cache-on-failure: true
|
||||
save-if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/main' }}
|
||||
cache-workspace-crates: true
|
||||
# Layer sccache on top of rust-cache for the same reason as the
|
||||
# build-native action: release version bumps bust the target/ cache,
|
||||
# but sccache hits at the rustc-unit level survive.
|
||||
- name: Setup sccache
|
||||
uses: mozilla-actions/sccache-action@v0.0.10
|
||||
version: "0.15.0"
|
||||
- name: Enable sccache for cargo
|
||||
# Conditional backend: self-hosted omp-kata injects a shared S3
|
||||
# (RustFS) sccache via pod env; GitHub-hosted runners keep the GHA
|
||||
@@ -588,21 +509,17 @@ jobs:
|
||||
echo "SCCACHE_GHA_ENABLED=true" >> "$GITHUB_ENV"
|
||||
echo "sccache backend: GitHub Actions cache"
|
||||
fi
|
||||
- name: Cache bun dependencies
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: ~/.bun/install/cache
|
||||
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
|
||||
- uses: ./.github/actions/setup-system-deps
|
||||
- run: bun install --frozen-lockfile
|
||||
- uses: ./.github/actions/bun-install
|
||||
- name: Install method smoke tests
|
||||
run: bun run ci:test:install-methods
|
||||
|
||||
release_binary:
|
||||
name: "Release binary: ${{ matrix.target_id }}"
|
||||
if: ${{ needs.release_metadata.outputs.is-release == 'true' && !cancelled() &&
|
||||
needs.native_linux_x64.result == 'success' && needs.native_cross_platform.result ==
|
||||
'success' && needs.test_workspace.result == 'success' &&
|
||||
needs.native_linux_x64.result == 'success' && needs.native_cross_platform_kata.result ==
|
||||
'success' && needs.native_cross_platform_macos.result == 'success' &&
|
||||
needs.test_workspace.result == 'success' &&
|
||||
needs.test_coding_agent_singleton.result == 'success' &&
|
||||
needs.test_ts_native.result == 'success' &&
|
||||
needs.test_coding_agent_ui.result == 'success' &&
|
||||
@@ -610,7 +527,7 @@ jobs:
|
||||
needs.test_coding_agent_native.result == 'success' &&
|
||||
needs.test_smoke.result == 'success' && needs.check.result == 'success' &&
|
||||
needs.install_methods.result == 'success' }}
|
||||
needs: [release_metadata, check, native_linux_x64, native_cross_platform, test_workspace, test_coding_agent_singleton, test_ts_native, test_coding_agent_ui, test_coding_agent_runtime, test_coding_agent_native, test_smoke, install_methods, native_artifact_lookup]
|
||||
needs: [release_metadata, check, native_linux_x64, native_cross_platform_kata, native_cross_platform_macos, test_workspace, test_coding_agent_singleton, test_ts_native, test_coding_agent_ui, test_coding_agent_runtime, test_coding_agent_native, test_smoke, install_methods, native_artifact_lookup]
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
|
||||
+1
-1
@@ -417,7 +417,7 @@ Resolution precedence for exact selectors:
|
||||
|
||||
Supported model roles:
|
||||
|
||||
- `default`, `smol`, `slow`, `vision`, `plan`, `designer`, `commit`, `task`
|
||||
- `default`, `smol`, `slow`, `vision`, `plan`, `designer`, `commit`, `title`, `task`
|
||||
|
||||
Role aliases like `pi/smol` expand through `settings.modelRoles`. Each role value can also append a thinking selector such as `:minimal`, `:low`, `:medium`, or `:high`.
|
||||
|
||||
|
||||
+1
-1
@@ -299,7 +299,7 @@ enabledModels:
|
||||
|
||||
| Key | Type | Default | Notes |
|
||||
|---|---|---|---|
|
||||
| `modelRoles` | record | `{}` | Map of role name -> model id. Built-in roles: `default`, `smol`, `slow`, `vision`, `plan`, `designer`, `commit`, `task`. Per-role env/flags: `--model`/`--smol`/`--slow`/`--plan`. |
|
||||
| `modelRoles` | record | `{}` | Map of role name -> model id. Built-in roles: `default`, `smol`, `slow`, `vision`, `plan`, `designer`, `commit`, `title`, `task`. Per-role env/flags: `--model`/`--smol`/`--slow`/`--plan`. |
|
||||
| `modelTags` | record | `{}` | Custom role/tag metadata; can introduce additional roles. |
|
||||
| `modelProviderOrder` | array | `[]` | Preferred provider order when a model id is ambiguous. |
|
||||
| `cycleOrder` | array | `["smol","default","slow"]` | Roles cycled by the model switcher. |
|
||||
|
||||
@@ -0,0 +1,448 @@
|
||||
# 01 — Host preparation & single-node k3s cluster
|
||||
|
||||
This guide takes a fresh Linux host from bare OS to a **working single-node [k3s](https://k3s.io) cluster** that is ready to run Kata-isolated CI runners. It covers host prerequisites (hardware virtualization, kernel modules, time sync, the nginx port constraint), the exact k3s install, kubeconfig setup, cluster networking (Flannel CNI, CIDRs, CoreDNS), and pod-to-internet egress via host firewalld NAT.
|
||||
|
||||
Read [README.md](README.md) first for the overall architecture and the placeholder/redaction table. When this guide is done, continue with **[02-kata-runtime.md](02-kata-runtime.md)** to install the Kata Containers runtime and register the `kata-qemu` RuntimeClass.
|
||||
|
||||
All configs below are copied from the live reference host and then redacted. Substitute the placeholders from the [README redaction table](README.md#redaction--placeholders) (notably `<CI_HOST>`, `<PUBLIC_IP>`, `<EXT_IFACE>`, `<TAILNET_IP>`) with your own values. CIDRs (`10.42.0.0/16`, `10.43.0.0/16`), the CoreDNS IP (`10.43.0.10`), and version numbers are kept as-is.
|
||||
|
||||
The reference host is a bare-metal **CentOS Stream 10** box, 32 vCPU / 125 GiB RAM, AMD CPU, running **k3s `v1.35.5+k3s1`** (bundled containerd `v2.2.3-k3s1`). Commands are shown for RHEL-family (`dnf` / `firewalld`); adapt package and firewall commands for your distro.
|
||||
|
||||
---
|
||||
|
||||
## 1. Host prerequisites
|
||||
|
||||
### 1.1 Hardware virtualization / KVM
|
||||
|
||||
Every CI job boots its own QEMU/KVM microVM, so the host **must** expose working KVM. On bare metal this means VT-x (Intel) or AMD-V (AMD) enabled in firmware; on a VM you need working *nested* virtualization.
|
||||
|
||||
Check the CPU virtualization flag (`vmx` = Intel, `svm` = AMD) and that the KVM device and modules are present:
|
||||
|
||||
```bash
|
||||
# CPU supports virtualization? (non-zero count = yes)
|
||||
grep -E -c '(vmx|svm)' /proc/cpuinfo
|
||||
|
||||
# which flavor
|
||||
grep -E -om1 '(vmx|svm)' /proc/cpuinfo # reference host prints: svm (AMD)
|
||||
lscpu | grep -i virtualization
|
||||
|
||||
# /dev/kvm must exist and be accessible
|
||||
ls -l /dev/kvm # crw-rw-rw-. 1 root kvm 10, 232 ... /dev/kvm
|
||||
|
||||
# KVM kernel modules loaded
|
||||
lsmod | grep -E '^kvm' # kvm_amd ... kvm (or kvm_intel on Intel)
|
||||
```
|
||||
|
||||
On the reference host this yields:
|
||||
|
||||
```
|
||||
$ ls -l /dev/kvm
|
||||
crw-rw-rw-. 1 root kvm 10, 232 /dev/kvm
|
||||
|
||||
$ lsmod | grep -E '^kvm'
|
||||
kvm_amd 237568 99
|
||||
kvm 1470464 78 kvm_amd
|
||||
```
|
||||
|
||||
The module loads automatically when the CPU flag is present; if `/dev/kvm` is missing, load it explicitly and persist it:
|
||||
|
||||
```bash
|
||||
modprobe kvm_amd # or: modprobe kvm_intel
|
||||
echo kvm_amd > /etc/modules-load.d/kvm.conf
|
||||
```
|
||||
|
||||
On Debian/Ubuntu you can instead run `kvm-ok` (from the `cpu-checker` package); on RHEL-family the checks above are the equivalent.
|
||||
|
||||
> Kata also needs the `vhost_vsock` and `vhost_net` modules for its agent vsock channel and VM networking. Those are part of the Kata runtime setup and are covered in [02-kata-runtime.md](02-kata-runtime.md); the KVM availability above is the only virtualization prerequisite for this guide.
|
||||
|
||||
### 1.2 Kernel modules and sysctls for k3s networking
|
||||
|
||||
k3s needs the `br_netfilter` and `overlay` modules and a couple of sysctls so that bridged pod traffic is seen by iptables and so the host can route/NAT pod traffic. The k3s systemd unit loads the modules on start (`ExecStartPre=-/sbin/modprobe br_netfilter` / `overlay`) and the installer sets the sysctls, but set them explicitly so they survive reboots and are correct before install:
|
||||
|
||||
```bash
|
||||
cat >/etc/modules-load.d/k3s.conf <<'EOF'
|
||||
br_netfilter
|
||||
overlay
|
||||
EOF
|
||||
modprobe br_netfilter overlay
|
||||
|
||||
cat >/etc/sysctl.d/90-k3s.conf <<'EOF'
|
||||
net.ipv4.ip_forward = 1
|
||||
net.bridge.bridge-nf-call-iptables = 1
|
||||
EOF
|
||||
sysctl --system
|
||||
```
|
||||
|
||||
Verify (these are the live values on the reference host):
|
||||
|
||||
```
|
||||
$ sysctl net.ipv4.ip_forward net.bridge.bridge-nf-call-iptables
|
||||
net.ipv4.ip_forward = 1
|
||||
net.bridge.bridge-nf-call-iptables = 1
|
||||
|
||||
$ lsmod | grep -E 'br_netfilter|overlay'
|
||||
br_netfilter 36864 0
|
||||
bridge 409600 1 br_netfilter
|
||||
overlay 229376 49
|
||||
```
|
||||
|
||||
`net.ipv4.ip_forward = 1` is what lets the host route (and NAT — see [section 5](#5-cluster-networking-cni-cidrs--nat-egress)) pod traffic out to the internet.
|
||||
|
||||
### 1.3 SELinux
|
||||
|
||||
The reference host runs SELinux in **Permissive** mode:
|
||||
|
||||
```
|
||||
$ getenforce
|
||||
Permissive
|
||||
```
|
||||
|
||||
The k3s installer installs an SELinux policy (`k3s-selinux`) when SELinux is Enforcing on RHEL-family hosts, so Enforcing also works; Permissive is used here to keep the Kata/QEMU + virtio-fs path unencumbered during bring-up. Pick one consistently — if you run Enforcing, make sure `container-selinux` / `k3s-selinux` are installed (the k3s installer pulls them).
|
||||
|
||||
### 1.4 Base packages
|
||||
|
||||
The k3s install script needs only `curl`; everything else (its own containerd, CNI, kubectl) is bundled. Make sure the host has current packages and the basics:
|
||||
|
||||
```bash
|
||||
dnf -y update
|
||||
dnf -y install curl tar iptables
|
||||
```
|
||||
|
||||
> Do **not** pre-install a separate containerd/Docker for k3s to use — k3s ships and manages its own containerd v2. (A separate Docker install can coexist for *building* the runner image; that is covered in [03-runner-image.md](03-runner-image.md).)
|
||||
|
||||
### 1.5 Time synchronization
|
||||
|
||||
Clock skew breaks TLS to the Kubernetes API and to GitHub. Keep an NTP client running. The reference host uses `chrony`:
|
||||
|
||||
```bash
|
||||
dnf -y install chrony
|
||||
systemctl enable --now chronyd
|
||||
timedatectl # "System clock synchronized: yes", "NTP service: active"
|
||||
```
|
||||
|
||||
### 1.6 The nginx port constraint (why Traefik and servicelb are disabled)
|
||||
|
||||
The reference host **also runs nginx**, which owns ports 80 and 443:
|
||||
|
||||
```
|
||||
$ systemctl is-active nginx
|
||||
active
|
||||
|
||||
$ ss -tlnp | grep -E ':80 |:443 '
|
||||
LISTEN 0 511 0.0.0.0:80 0.0.0.0:* users:(("nginx",...))
|
||||
LISTEN 0 511 0.0.0.0:443 0.0.0.0:* users:(("nginx",...))
|
||||
```
|
||||
|
||||
A default k3s install would deploy **Traefik** (an ingress controller that wants :80/:443) and **servicelb** (the Klipper load-balancer, which binds `LoadBalancer` service ports directly on the host). Both would collide with nginx. We therefore disable both at install time (next section). k3s' own API server listens on **:6443**, which does not conflict with nginx, so the cluster is fully functional without those two add-ons.
|
||||
|
||||
> Note on swap: the reference host has swap enabled (`/dev/md1`, 16 GiB) and k3s runs fine with it. If you prefer the upstream-Kubernetes convention of swap-off, disabling it is also supported — it is not required here.
|
||||
|
||||
---
|
||||
|
||||
## 2. Install k3s
|
||||
|
||||
Install k3s as a single-node server, pinning the version and disabling Traefik and servicelb. This is the exact configuration baked into the reference host's systemd unit:
|
||||
|
||||
```bash
|
||||
curl -sfL https://get.k3s.io | \
|
||||
INSTALL_K3S_VERSION=v1.35.5+k3s1 \
|
||||
INSTALL_K3S_EXEC="server --disable=traefik --disable=servicelb" \
|
||||
sh -
|
||||
```
|
||||
|
||||
What each piece does:
|
||||
|
||||
| Token | Meaning |
|
||||
| --- | --- |
|
||||
| `INSTALL_K3S_VERSION=v1.35.5+k3s1` | Pin the exact k3s release (reproducible installs; omit to track the stable channel). |
|
||||
| `server` | Run this node as a **control-plane + worker** (single-node cluster — it both schedules and runs pods). |
|
||||
| `--disable=traefik` | Do **not** deploy the bundled Traefik ingress controller, so nothing tries to bind host :80/:443 (owned by nginx — see [1.6](#16-the-nginx-port-constraint-why-traefik-and-servicelb-are-disabled)). |
|
||||
| `--disable=servicelb` | Do **not** deploy Klipper servicelb, so `LoadBalancer` services do not bind host ports. Runners need no inbound `LoadBalancer`; the ARC listener reaches GitHub via **outbound** long-poll. |
|
||||
|
||||
Everything else is left at k3s defaults *on purpose* — those defaults are what the rest of this doc set relies on:
|
||||
|
||||
| Default (not overridden) | Value | Why we keep it |
|
||||
| --- | --- | --- |
|
||||
| CNI | **Flannel**, VXLAN backend | Simple single-node overlay; see [section 5](#5-cluster-networking-cni-cidrs--nat-egress). |
|
||||
| `--cluster-cidr` (pod network) | `10.42.0.0/16` | Pod IP range. |
|
||||
| `--service-cidr` (service network) | `10.43.0.0/16` | ClusterIP range. |
|
||||
| `--cluster-dns` (CoreDNS) | `10.43.0.10` | In-cluster DNS resolver. |
|
||||
| Container runtime | bundled **containerd v2** | Kata is wired into *this* containerd in [02-kata-runtime.md](02-kata-runtime.md). |
|
||||
|
||||
Because we do not pass `--node-ip` / `--flannel-iface`, k3s auto-detects the host's primary interface and uses its address as the node IP (the public IPv4 on the reference host). If your host has multiple NICs, set `--node-ip` / `--flannel-iface` explicitly.
|
||||
|
||||
The installer writes the systemd unit `/etc/systemd/system/k3s.service`. On the reference host its `ExecStart` is exactly:
|
||||
|
||||
```ini
|
||||
ExecStartPre=-/sbin/modprobe br_netfilter
|
||||
ExecStartPre=-/sbin/modprobe overlay
|
||||
ExecStart=/usr/local/bin/k3s \
|
||||
server \
|
||||
'--disable=traefik' \
|
||||
'--disable=servicelb' \
|
||||
```
|
||||
|
||||
There is **no** `/etc/rancher/k3s/config.yaml` on the host — the two `--disable` flags above are the *only* customization; everything else is the default set listed above.
|
||||
|
||||
Enable and check the service:
|
||||
|
||||
```bash
|
||||
systemctl enable --now k3s
|
||||
systemctl status k3s --no-pager
|
||||
journalctl -u k3s -f # follow startup logs until the node is Ready
|
||||
```
|
||||
|
||||
Confirm the version:
|
||||
|
||||
```
|
||||
$ k3s --version
|
||||
k3s version v1.35.5+k3s1 (6a4781ad)
|
||||
go version go1.25.9
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. kubeconfig
|
||||
|
||||
k3s writes an admin kubeconfig to `/etc/rancher/k3s/k3s.yaml`. Point `kubectl` at it:
|
||||
|
||||
```bash
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
# persist for future shells:
|
||||
echo 'export KUBECONFIG=/etc/rancher/k3s/k3s.yaml' >> ~/.bashrc
|
||||
```
|
||||
|
||||
The file targets the local API server over loopback (TLS material redacted):
|
||||
|
||||
```yaml
|
||||
apiVersion: v1
|
||||
clusters:
|
||||
- cluster:
|
||||
certificate-authority-data: <REDACTED>
|
||||
server: https://127.0.0.1:6443
|
||||
name: default
|
||||
contexts:
|
||||
- context:
|
||||
cluster: default
|
||||
user: default
|
||||
name: default
|
||||
current-context: default
|
||||
kind: Config
|
||||
users:
|
||||
- name: default
|
||||
user:
|
||||
client-certificate-data: <REDACTED>
|
||||
client-key-data: <REDACTED>
|
||||
```
|
||||
|
||||
> This kubeconfig embeds cluster-admin credentials. Treat the file as a secret (`chmod 600`, root-only). To administer the cluster from another machine, copy the file and replace `127.0.0.1` with the host's reachable address — on the reference host that is done over **Tailscale** (`tailscale0`), so the API server is never exposed on the public interface. Do not commit this file.
|
||||
|
||||
Quick check that the client can talk to the server:
|
||||
|
||||
```bash
|
||||
kubectl version # client + server versions
|
||||
kubectl cluster-info
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 4. Verify the node and system pods
|
||||
|
||||
Wait until the node reports `Ready`:
|
||||
|
||||
```bash
|
||||
kubectl get nodes -o wide
|
||||
```
|
||||
|
||||
Reference output (redacted — `<CI_HOST>` is the hostname, `<PUBLIC_IP>` the auto-detected node IP):
|
||||
|
||||
```
|
||||
NAME STATUS ROLES AGE VERSION INTERNAL-IP EXTERNAL-IP OS-IMAGE KERNEL-VERSION CONTAINER-RUNTIME
|
||||
<CI_HOST> Ready control-plane,master ... v1.35.5+k3s1 <PUBLIC_IP> <none> CentOS Stream 10 (Coughlan) 7.0.10-1.el10.elrepo.x86_64 containerd://2.2.3-k3s1
|
||||
```
|
||||
|
||||
Then confirm the core system pods are running:
|
||||
|
||||
```bash
|
||||
kubectl get pods -A
|
||||
```
|
||||
|
||||
You should see the k3s base set in `kube-system` (these are what remain after disabling Traefik and servicelb):
|
||||
|
||||
```
|
||||
NAMESPACE NAME READY STATUS RESTARTS AGE
|
||||
kube-system coredns-<hash> 1/1 Running 0 ...
|
||||
kube-system local-path-provisioner-<hash> 1/1 Running 0 ...
|
||||
kube-system metrics-server-<hash> 1/1 Running 0 ...
|
||||
```
|
||||
|
||||
`local-path-provisioner` is the default storage class (used later for the RustFS cache PVC in [04-arc-and-caching.md](04-arc-and-caching.md)); `coredns` is cluster DNS; `metrics-server` backs `kubectl top`. There is intentionally **no** `traefik` or `svclb-*` pod.
|
||||
|
||||
---
|
||||
|
||||
## 5. Cluster networking (CNI, CIDRs & NAT egress)
|
||||
|
||||
### 5.1 Flannel CNI
|
||||
|
||||
k3s installs Flannel and writes its CNI config to `/var/lib/rancher/k3s/agent/etc/cni/net.d/10-flannel.conflist`. This file is generated by k3s — copied here verbatim (no secrets; nothing to redact):
|
||||
|
||||
```json
|
||||
{
|
||||
"name":"cbr0",
|
||||
"cniVersion":"1.0.0",
|
||||
"plugins":[
|
||||
{
|
||||
"type":"flannel",
|
||||
"delegate":{
|
||||
"hairpinMode":true,
|
||||
"forceAddress":true,
|
||||
"isDefaultGateway":true
|
||||
}
|
||||
},
|
||||
{
|
||||
"type":"portmap",
|
||||
"capabilities":{
|
||||
"portMappings":true
|
||||
}
|
||||
},
|
||||
{
|
||||
"type":"bandwidth",
|
||||
"capabilities":{
|
||||
"bandwidth":true
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
- `flannel` delegates to the `bridge` plugin (`cbr0`), with `isDefaultGateway` so the pod's default route points at the node — this is the path pod egress takes to reach the host's NAT.
|
||||
- `portmap` and `bandwidth` are standard chained plugins (host-port mapping and per-pod bandwidth shaping).
|
||||
- The backend is Flannel's default **VXLAN** (we did not override it at install). On a single node, pod-to-pod traffic stays on the local bridge.
|
||||
|
||||
### 5.2 Address ranges
|
||||
|
||||
The cluster uses the k3s defaults — keep these as-is (they are referenced throughout the doc set):
|
||||
|
||||
| Range | CIDR | Notes |
|
||||
| --- | --- | --- |
|
||||
| Pod network (cluster-cidr) | `10.42.0.0/16` | Single node carves a `/24` from this: `kubectl get node -o jsonpath='{.items[0].spec.podCIDR}'` → `10.42.0.0/24`. |
|
||||
| Service network (service-cidr) | `10.43.0.0/16` | ClusterIP services. |
|
||||
| CoreDNS service IP | `10.43.0.10` | Cluster DNS resolver (`kube-dns` Service). |
|
||||
|
||||
Confirm CoreDNS:
|
||||
|
||||
```
|
||||
$ kubectl -n kube-system get svc kube-dns
|
||||
NAME TYPE CLUSTER-IP EXTERNAL-IP PORT(S) AGE
|
||||
kube-dns ClusterIP 10.43.0.10 <none> 53/UDP,53/TCP,9153/TCP ...
|
||||
```
|
||||
|
||||
### 5.3 Pod-to-internet egress via host firewalld NAT
|
||||
|
||||
Runner microVMs need outbound internet (to reach GitHub, fetch toolchains, etc.) but the host must **not** expose the cluster on its public interface. The path is:
|
||||
|
||||
```
|
||||
pod (10.42.0.0/16) --default route--> cbr0/flannel --> host routing --> firewalld masquerade (SNAT) --> <EXT_IFACE> --> internet (as <PUBLIC_IP>)
|
||||
```
|
||||
|
||||
This relies on `net.ipv4.ip_forward = 1` (set in [1.2](#12-kernel-modules-and-sysctls-for-k3s-networking)) plus firewalld **masquerade** (SNAT) and **forwarding** on the public zone. On the reference host the external interface lives in the default `public` zone with masquerade enabled:
|
||||
|
||||
```
|
||||
$ firewall-cmd --list-all
|
||||
public (default, active)
|
||||
target: default
|
||||
icmp-block-inversion: no
|
||||
interfaces: <EXT_IFACE>
|
||||
sources:
|
||||
services: cockpit dhcpv6-client http https ssh
|
||||
ports:
|
||||
protocols:
|
||||
forward: yes
|
||||
masquerade: yes
|
||||
forward-ports:
|
||||
source-ports:
|
||||
icmp-blocks:
|
||||
rich rules:
|
||||
```
|
||||
|
||||
The two lines that make pod egress work are **`masquerade: yes`** (SNAT pod source IPs to `<PUBLIC_IP>` on the way out) and **`forward: yes`** (allow routing between interfaces). The `services` list (`ssh`, `http`, `https`, `cockpit`, `dhcpv6-client`) is the host's own inbound allow-list and is unrelated to pod egress — note that `http`/`https` here are for the **host nginx**, not k3s.
|
||||
|
||||
The pod and service CIDRs (plus the Tailscale admin interface) are placed in the **trusted** zone so intra-cluster and admin traffic is accepted without per-rule firewalling:
|
||||
|
||||
```
|
||||
$ firewall-cmd --zone=trusted --list-all
|
||||
trusted (active)
|
||||
target: ACCEPT
|
||||
interfaces: tailscale0
|
||||
sources: 10.42.0.0/16 10.43.0.0/16
|
||||
forward: yes
|
||||
masquerade: no
|
||||
...
|
||||
```
|
||||
|
||||
```
|
||||
$ firewall-cmd --get-active-zones
|
||||
public (default)
|
||||
interfaces: <EXT_IFACE>
|
||||
trusted
|
||||
interfaces: tailscale0
|
||||
sources: 10.42.0.0/16 10.43.0.0/16
|
||||
docker # br-* bridges from the separate Docker stack (unrelated to k3s)
|
||||
interfaces: ...
|
||||
```
|
||||
|
||||
To reproduce this NAT setup on a fresh host (replace `<EXT_IFACE>` with your public NIC, e.g. `eth0`):
|
||||
|
||||
```bash
|
||||
# external interface in the public zone with NAT + forwarding
|
||||
firewall-cmd --permanent --zone=public --change-interface=<EXT_IFACE>
|
||||
firewall-cmd --permanent --zone=public --add-masquerade
|
||||
firewall-cmd --permanent --zone=public --add-forward # firewalld >= 0.9
|
||||
|
||||
# trust intra-cluster traffic (pod + service CIDRs) and the Tailscale admin iface
|
||||
firewall-cmd --permanent --zone=trusted --add-source=10.42.0.0/16
|
||||
firewall-cmd --permanent --zone=trusted --add-source=10.43.0.0/16
|
||||
firewall-cmd --permanent --zone=trusted --change-interface=tailscale0
|
||||
|
||||
firewall-cmd --reload
|
||||
```
|
||||
|
||||
Verify masquerade and forwarding are live:
|
||||
|
||||
```
|
||||
$ firewall-cmd --query-masquerade
|
||||
yes
|
||||
$ firewall-cmd --query-forward
|
||||
yes
|
||||
```
|
||||
|
||||
> This is host-level NAT only. A second, finer-grained layer — the `runner-egress-lockdown` Kubernetes **NetworkPolicy** — restricts *which* destinations runner pods may reach (it blocks `<PUBLIC_IP>`, the tailnet `100.64.0.0/10`, RFC-1918 ranges, etc., while allowing the public internet and cluster DNS). That policy is part of the runner setup and is documented in [04-arc-and-caching.md](04-arc-and-caching.md).
|
||||
|
||||
---
|
||||
|
||||
## 6. Verification: DNS & egress smoke test
|
||||
|
||||
The system pods being `Running` (section 4) already prove in-cluster networking. To prove **DNS resolution** and **pod-to-internet egress** end to end, run a throwaway pod (delete it afterward):
|
||||
|
||||
```bash
|
||||
# in-cluster DNS: resolve the kubernetes Service via CoreDNS (10.43.0.10)
|
||||
kubectl run dns-test --image=busybox:1.36 --restart=Never --rm -it -- \
|
||||
nslookup kubernetes.default.svc.cluster.local
|
||||
|
||||
# external DNS + egress: resolve and reach the internet through host NAT
|
||||
kubectl run egress-test --image=busybox:1.36 --restart=Never --rm -it -- \
|
||||
sh -c 'nslookup github.com && wget -qO- https://api.github.com/zen'
|
||||
```
|
||||
|
||||
Expected: the first command resolves to a `10.43.x.x` ClusterIP; the second resolves a public name and prints a line of text fetched from the internet (proving SNAT/masquerade works). If DNS fails, recheck CoreDNS (`kubectl -n kube-system get pods`); if egress fails, recheck `ip_forward`, `masquerade`, and `forward` from [section 5.3](#53-pod-to-internet-egress-via-host-firewalld-nat).
|
||||
|
||||
> On the live reference host, the ARC scale-set **listener** pod (in `arc-systems`) is itself continuous proof of working egress: it stays `Running` only because it can reach GitHub outbound through this exact NAT path.
|
||||
|
||||
---
|
||||
|
||||
## Next steps
|
||||
|
||||
The host now runs a healthy single-node k3s cluster with working networking and NAT egress, and Traefik/servicelb disabled so nginx keeps :80/:443. Continue with:
|
||||
|
||||
- **[02-kata-runtime.md](02-kata-runtime.md)** — install Kata Containers, wire it into k3s' bundled containerd, and register the `kata-qemu` RuntimeClass.
|
||||
- [README.md](README.md) — architecture overview, full component map, and the placeholder/redaction reference.
|
||||
@@ -0,0 +1,471 @@
|
||||
# 02 — Kata Containers runtime for k3s
|
||||
|
||||
This guide installs **Kata Containers 3.31.0** and wires it into the k3s-bundled
|
||||
containerd as a named runtime, `kata-qemu`, so that any pod carrying
|
||||
`runtimeClassName: kata-qemu` boots inside its own QEMU/KVM microVM with a
|
||||
**separate guest kernel** from the host.
|
||||
|
||||
- **Previous:** [`01-host-and-cluster.md`](./01-host-and-cluster.md) — host prep, k3s install, networking. You need a working single-node k3s, KVM enabled (`/dev/kvm` present), and nested-virt off (this is bare metal).
|
||||
- **Next:** [`03-runner-image.md`](./03-runner-image.md) — the preloaded GitHub Actions runner image that runs inside these microVMs.
|
||||
|
||||
Why a microVM per pod: the CI jobs run untrusted code (third-party deps, fork
|
||||
PRs). A `runc` container shares the host kernel; a Kata pod gets its own guest
|
||||
kernel and a hardware-virtualization boundary (Intel VT-x / AMD-V), so a kernel
|
||||
exploit inside a job does not reach the host. On `<CI_HOST>` the host runs the
|
||||
CentOS Stream 10 kernel `7.0.10-1.el10.elrepo.x86_64`, while every Kata guest
|
||||
runs `6.18.28-194` — the verification in the last section turns that gap into a
|
||||
one-line proof.
|
||||
|
||||
All host paths and commands below were taken from the live host (read-only) and
|
||||
redacted per the doc-set redaction map. Run them on **your** reproduction host;
|
||||
on the live host only the read-only inspections (`kata-runtime check`,
|
||||
`kata-runtime env`, `kubectl get …`) are safe.
|
||||
|
||||
---
|
||||
|
||||
## Step 1 — Install the Kata static release into `/opt/kata`
|
||||
|
||||
Kata ships a self-contained **static tarball**: a pinned QEMU, the guest kernel,
|
||||
the guest rootfs image, virtiofsd, the runtime, and the containerd shim, all
|
||||
rooted at `/opt/kata`. Nothing links against host libraries, so it coexists
|
||||
cleanly with the host's own QEMU/libvirt and survives OS upgrades.
|
||||
|
||||
```bash
|
||||
# amd64 host; pin the exact version so the kernel/image/QEMU triple is reproducible.
|
||||
KATA_VER=3.31.0
|
||||
curl -fsSL -o kata-static.tar.xz \
|
||||
"https://github.com/kata-containers/kata-containers/releases/download/${KATA_VER}/kata-static-${KATA_VER}-amd64.tar.xz"
|
||||
|
||||
# The archive is rooted at ./opt/kata, so extracting at / lands everything in /opt/kata.
|
||||
sudo tar -xf kata-static.tar.xz -C /
|
||||
```
|
||||
|
||||
Put the **shim** and the **CLI** on `PATH`. The shim symlink is what containerd
|
||||
resolves at launch time (see Step 2); the `kata-runtime` symlink is for
|
||||
host-side inspection and `kata-runtime check`/`env`:
|
||||
|
||||
```bash
|
||||
sudo ln -sf /opt/kata/bin/containerd-shim-kata-v2 /usr/local/bin/containerd-shim-kata-v2
|
||||
sudo ln -sf /opt/kata/bin/kata-runtime /usr/local/bin/kata-runtime
|
||||
```
|
||||
|
||||
On the live host both symlinks are in place:
|
||||
|
||||
```text
|
||||
/usr/local/bin/containerd-shim-kata-v2 -> /opt/kata/bin/containerd-shim-kata-v2
|
||||
/usr/local/bin/kata-runtime -> /opt/kata/bin/kata-runtime
|
||||
```
|
||||
|
||||
### Inspect the install layout
|
||||
|
||||
```text
|
||||
/opt/kata
|
||||
├── VERSION # "3.31.0"
|
||||
├── versions.yaml # pinned component versions (QEMU, kernel, rootfs)
|
||||
├── bin/ # qemu-system-x86_64, containerd-shim-kata-v2, kata-runtime, ...
|
||||
├── libexec/ # virtiofsd
|
||||
└── share/
|
||||
├── defaults/kata-containers/
|
||||
│ ├── configuration-qemu.toml # the active config for the kata-qemu runtime
|
||||
│ └── configuration.toml -> configuration-qemu.toml
|
||||
└── kata-containers/
|
||||
├── vmlinux.container -> vmlinux-6.18.28-194 # guest kernel
|
||||
└── kata-containers.img -> kata-ubuntu-noble.image # guest rootfs
|
||||
```
|
||||
|
||||
`bin/` ships several hypervisors (`qemu-system-x86_64`, `cloud-hypervisor`,
|
||||
`firecracker`, `jailer`) and the QEMU confidential-computing variants
|
||||
(`-snp-experimental`, `-tdx-experimental`); this setup uses plain
|
||||
`qemu-system-x86_64`. `share/defaults/kata-containers/` also carries
|
||||
`configuration-clh.toml`, `configuration-fc.toml`, etc. — one per hypervisor.
|
||||
We only use `configuration-qemu.toml`.
|
||||
|
||||
### Confirm version and capability
|
||||
|
||||
```console
|
||||
$ /opt/kata/bin/kata-runtime --version
|
||||
kata-runtime : 3.31.0
|
||||
commit : ddb8a5de89891f12e1ce0013eb066a330b2988b9
|
||||
OCI specs: 1.2.1
|
||||
```
|
||||
|
||||
The component pins live in `/opt/kata/versions.yaml`. The two that matter for
|
||||
the guest are the QEMU and kernel versions:
|
||||
|
||||
```yaml
|
||||
assets:
|
||||
hypervisor:
|
||||
qemu:
|
||||
version: "v10.2.1"
|
||||
kernel:
|
||||
version: "v6.18.28"
|
||||
```
|
||||
|
||||
`kata-runtime check` confirms the host can actually start a microVM (KVM
|
||||
present, CPU virtualization usable). Run it as a host-side smoke test before
|
||||
touching k3s:
|
||||
|
||||
```console
|
||||
$ /opt/kata/bin/kata-runtime check
|
||||
level=warning msg="Not running network checks as super user" arch=amd64 ...
|
||||
System is capable of running Kata Containers
|
||||
System can currently create Kata Containers
|
||||
```
|
||||
|
||||
`kata-runtime env` cross-checks the resolved kernel/image/hypervisor and the
|
||||
host kernel — note the guest kernel (`6.18.28-194`) versus the host kernel
|
||||
(`7.0.10`):
|
||||
|
||||
```console
|
||||
$ /opt/kata/bin/kata-runtime env | grep -iE 'Kernel|Image|MachineType|Path|Hypervisor'
|
||||
[Kernel]
|
||||
Path = "/opt/kata/share/kata-containers/vmlinux-6.18.28-194"
|
||||
[Image]
|
||||
Path = "/opt/kata/share/kata-containers/kata-ubuntu-noble.image"
|
||||
[Hypervisor]
|
||||
MachineType = "q35"
|
||||
Version = "QEMU emulator version 10.2.1 (kata-static) ..."
|
||||
Path = "/opt/kata/bin/qemu-system-x86_64"
|
||||
[Host]
|
||||
Kernel = "7.0.10-1.el10.elrepo.x86_64"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Step 2 — Register `kata-qemu` with k3s's containerd
|
||||
|
||||
k3s embeds its **own** containerd (v2 here) — not the host's. On every start it
|
||||
**regenerates** `/var/lib/rancher/k3s/agent/etc/containerd/config.toml` from a
|
||||
template; the header says so:
|
||||
|
||||
```toml
|
||||
# File generated by k3s. DO NOT EDIT. Use config-v3.toml.tmpl instead.
|
||||
version = 3
|
||||
imports = ["/var/lib/rancher/k3s/agent/etc/containerd/config-v3.toml.d/*.toml"]
|
||||
root = "/var/lib/rancher/k3s/agent/containerd"
|
||||
state = "/run/k3s/containerd"
|
||||
...
|
||||
[plugins.'io.containerd.cri.v1.runtime'.containerd.runtimes.runc]
|
||||
runtime_type = "io.containerd.runc.v2"
|
||||
```
|
||||
|
||||
Editing `config.toml` directly is pointless — k3s overwrites it on the next
|
||||
restart. The durable hook is the generated `imports` line: k3s merges any
|
||||
`*.toml` under `config-v3.toml.d/` into the final config. That directory is the
|
||||
**supported drop-in path** and survives k3s upgrades.
|
||||
|
||||
> k3s picks the drop-in directory from the containerd config schema version. With
|
||||
> the generated `version = 3` config the path is `config-v3.toml.d/`. (On older
|
||||
> k3s that emitted a v2 config it was `config.toml.d/`.) Match whatever your
|
||||
> generated `config.toml` declares.
|
||||
|
||||
Create the drop-in. This is the real file from the host, verbatim:
|
||||
|
||||
```toml
|
||||
# /var/lib/rancher/k3s/agent/etc/containerd/config-v3.toml.d/kata.toml
|
||||
# Kata Containers (QEMU/KVM microVM) runtime for k3s containerd.
|
||||
# Added out-of-band; merged via the generated config's `imports`.
|
||||
[plugins.'io.containerd.cri.v1.runtime'.containerd.runtimes.kata-qemu]
|
||||
runtime_type = "io.containerd.kata.v2"
|
||||
runtime_path = "/opt/kata/bin/containerd-shim-kata-v2"
|
||||
[plugins.'io.containerd.cri.v1.runtime'.containerd.runtimes.kata-qemu.options]
|
||||
ConfigPath = "/opt/kata/share/defaults/kata-containers/configuration-qemu.toml"
|
||||
```
|
||||
|
||||
What each line does:
|
||||
|
||||
- The table key `…runtimes.kata-qemu` defines a CRI runtime **named**
|
||||
`kata-qemu`. A RuntimeClass whose `handler` is `kata-qemu` (Step 3) selects
|
||||
exactly this entry.
|
||||
- `runtime_type = "io.containerd.kata.v2"` tells containerd this is a v2 (shim)
|
||||
runtime. By the v2 naming convention containerd would look for
|
||||
`containerd-shim-kata-v2` on `PATH` — which is why we symlinked it in Step 1.
|
||||
- `runtime_path = "/opt/kata/bin/containerd-shim-kata-v2"` pins the **exact**
|
||||
shim binary regardless of `PATH`, so the runtime is unambiguous even if the
|
||||
symlink is missing or another shim shadows it.
|
||||
- `options.ConfigPath` points the shim at the hypervisor configuration analyzed
|
||||
in Step 4. This is how a single shim binary can back multiple runtimes (e.g. a
|
||||
second `kata-clh` runtime pointing at `configuration-clh.toml`).
|
||||
|
||||
Restart k3s so the bundled containerd reloads and merges the drop-in:
|
||||
|
||||
```bash
|
||||
sudo systemctl restart k3s
|
||||
```
|
||||
|
||||
Verify containerd now knows the runtime (CRI reports the registered runtime
|
||||
handlers):
|
||||
|
||||
```bash
|
||||
sudo k3s crictl info | grep -A2 '"kata-qemu"'
|
||||
```
|
||||
|
||||
The `runc` default is untouched: pods without a `runtimeClassName` keep running
|
||||
as ordinary host-kernel containers. Only pods that opt into the RuntimeClass
|
||||
below get a microVM.
|
||||
|
||||
---
|
||||
|
||||
## Step 3 — Create the `kata-qemu` RuntimeClass
|
||||
|
||||
A Kubernetes [RuntimeClass](https://kubernetes.io/docs/concepts/containers/runtime-class/)
|
||||
maps a friendly name a pod can request to the containerd runtime handler
|
||||
registered in Step 2. The `handler` value **must** equal the runtime name in the
|
||||
drop-in (`kata-qemu`).
|
||||
|
||||
```yaml
|
||||
# kata-qemu-runtimeclass.yaml
|
||||
apiVersion: node.k8s.io/v1
|
||||
kind: RuntimeClass
|
||||
metadata:
|
||||
name: kata-qemu
|
||||
handler: kata-qemu
|
||||
```
|
||||
|
||||
```bash
|
||||
kubectl apply -f kata-qemu-runtimeclass.yaml
|
||||
```
|
||||
|
||||
Confirm it exists (live host, read-only):
|
||||
|
||||
```console
|
||||
$ kubectl get runtimeclass kata-qemu -o yaml
|
||||
apiVersion: node.k8s.io/v1
|
||||
handler: kata-qemu
|
||||
kind: RuntimeClass
|
||||
metadata:
|
||||
annotations:
|
||||
kubectl.kubernetes.io/last-applied-configuration: |
|
||||
{"apiVersion":"node.k8s.io/v1","handler":"kata-qemu","kind":"RuntimeClass","metadata":{"annotations":{},"name":"kata-qemu"}}
|
||||
creationTimestamp: "2026-06-14T19:18:16Z"
|
||||
name: kata-qemu
|
||||
resourceVersion: "580"
|
||||
uid: 50a199ac-3936-4ae8-9deb-be389ce9f042
|
||||
```
|
||||
|
||||
From here, any pod with `spec.runtimeClassName: kata-qemu` is scheduled onto the
|
||||
Kata shim. The ARC runner pod template sets exactly this — see
|
||||
[`04-arc-and-caching.md`](./04-arc-and-caching.md).
|
||||
|
||||
---
|
||||
|
||||
## Step 4 — Kata hypervisor configuration (`configuration-qemu.toml`)
|
||||
|
||||
The shim reads `/opt/kata/share/defaults/kata-containers/configuration-qemu.toml`
|
||||
(via `ConfigPath`). The file is long and mostly defaults; below are the **active
|
||||
settings that shape this deployment**, copied from the host with the noise
|
||||
stripped. Each is explained afterward.
|
||||
|
||||
```toml
|
||||
[hypervisor.qemu]
|
||||
path = "/opt/kata/bin/qemu-system-x86_64"
|
||||
kernel = "/opt/kata/share/kata-containers/vmlinux.container"
|
||||
image = "/opt/kata/share/kata-containers/kata-containers.img"
|
||||
machine_type = "q35"
|
||||
rootfs_type = "ext4"
|
||||
cpu_features = "pmu=off"
|
||||
kernel_params = "cgroup_no_v1=all systemd.unified_cgroup_hierarchy=1"
|
||||
|
||||
default_vcpus = 2
|
||||
default_maxvcpus = 0
|
||||
default_memory = 4096
|
||||
default_maxmemory = 0
|
||||
memory_slots = 10
|
||||
|
||||
shared_fs = "virtio-fs"
|
||||
virtio_fs_daemon = "/opt/kata/libexec/virtiofsd"
|
||||
virtio_fs_cache = "auto"
|
||||
virtio_fs_extra_args = ["--thread-pool-size=4", "--announce-submounts"]
|
||||
|
||||
disable_block_device_use = true
|
||||
block_device_driver = "virtio-scsi"
|
||||
block_device_aio = "io_uring"
|
||||
|
||||
[factory]
|
||||
enable_template = false
|
||||
vm_cache_number = 0
|
||||
|
||||
[runtime]
|
||||
internetworking_model = "tcfilter"
|
||||
emptydir_mode = "shared-fs"
|
||||
static_sandbox_resource_mgmt = false
|
||||
sandbox_cgroup_only = false
|
||||
```
|
||||
|
||||
### Boot artifacts: `path` / `kernel` / `image`
|
||||
|
||||
`path` is the bundled QEMU 10.2.1; `kernel` is the guest kernel
|
||||
(`vmlinux.container -> vmlinux-6.18.28-194`); `image` is the guest rootfs
|
||||
(`kata-containers.img -> kata-ubuntu-noble.image`, an Ubuntu Noble rootfs with
|
||||
`kata-agent` baked in as PID 1's manager). These three are the entire guest —
|
||||
none of them is the host kernel, which is the whole point. `machine_type =
|
||||
"q35"` is the modern PCIe QEMU machine (needed for PCIe hotplug);
|
||||
`cpu_features = "pmu=off"` disables the virtual perf-monitoring unit (avoids
|
||||
spurious PMU passthrough issues). `kernel_params` forces cgroup v2-only in the
|
||||
guest, matching a modern systemd userspace.
|
||||
|
||||
### vCPU / memory sizing — hotplug from pod requests/limits
|
||||
|
||||
This is the most important block to understand for a CI runner.
|
||||
|
||||
- `default_vcpus = 2` and `default_memory = 4096` (MiB) are the **boot-time**
|
||||
size. Runner microVMs now start at the runner pod's guaranteed request: 2 vCPU,
|
||||
4 GiB, rather than booting tiny and immediately hotplugging to that floor.
|
||||
- `default_maxvcpus = 0` means "no fixed ceiling — use the host's physical CPU
|
||||
count" (32 on this box). `default_maxmemory = 0` likewise means "host total
|
||||
RAM". `memory_slots = 10` is the number of ACPI DIMM hotplug slots, i.e. how
|
||||
many memory-grow operations the guest can accept.
|
||||
- `static_sandbox_resource_mgmt = false` still enables **dynamic** sizing: Kata
|
||||
reads the pod's CPU/memory **limits** that the kubelet/CRI hands the shim and
|
||||
hotplugs beyond the boot floor as needed. So a runner pod requesting `2 CPU /
|
||||
4Gi` with limits `8 CPU / 12Gi` now boots at 2 vCPU/4 GiB and grows toward
|
||||
8 vCPU / 12 GiB. If a pod sets no limits, the VM stays at the defaults.
|
||||
|
||||
The practical rule here is simple: align the defaults to the runner pod's
|
||||
requests when every job creates a fresh VM and immediately needs that baseline
|
||||
anyway; let `resources.limits` remain the hotplug ceiling. The repo ships
|
||||
[`infra/tune-kata-runtime.sh`](../tune-kata-runtime.sh) to apply exactly this
|
||||
change (plus the virtiofsd worker-pool tuning below) over SSH to the host.
|
||||
|
||||
### `shared_fs = "virtio-fs"` — sharing the container rootfs into the VM
|
||||
|
||||
The container's rootfs is prepared on the **host** by containerd's overlayfs
|
||||
snapshotter (see below). Rather than repackaging it as a virtual disk, Kata runs
|
||||
**virtiofsd** (`/opt/kata/libexec/virtiofsd`) on the host to export that
|
||||
directory over **virtio-fs**, and the guest mounts it as the container root.
|
||||
This is why `disable_block_device_use = true`: the rootfs travels in over the
|
||||
shared filesystem, not as a block device. Benefits: no image-to-block
|
||||
conversion, near-instant rootfs availability, and host/guest can both see the
|
||||
files. `virtio_fs_cache = "auto"` keeps the conservative page-cache behavior, but
|
||||
the active worker pool is now `--thread-pool-size=4` rather than `1` so the
|
||||
metadata-heavy Bun cache restore / extract path has a few host workers to fan out
|
||||
across. `--announce-submounts` keeps nested mounts visible to the guest.
|
||||
|
||||
`emptydir_mode = "shared-fs"` extends the same mechanism to Kubernetes
|
||||
`emptyDir` volumes — they are shared into the guest over virtio-fs instead of
|
||||
being block devices.
|
||||
|
||||
### `block_device_driver = "virtio-scsi"` — for the volumes that *are* blocks
|
||||
|
||||
Even with `disable_block_device_use = true` for the rootfs, any genuine block
|
||||
volume (e.g. a `local-path` PVC presented as a device) is attached over a
|
||||
**virtio-scsi** controller, with `block_device_aio = "io_uring"` for efficient
|
||||
async I/O. virtio-scsi (vs virtio-blk) supports more disks per controller and
|
||||
hotplug, which matters when volumes attach after boot.
|
||||
|
||||
### vsock agent channel
|
||||
|
||||
The shim on the host talks to `kata-agent` inside the guest over a
|
||||
**VIRTIO-VSOCK** channel — a host↔guest socket transport that needs no guest IP
|
||||
or network. All container lifecycle operations (create/start/exec/IO/metrics)
|
||||
are ttRPC calls over that vsock link. In Kata 3.x vsock is the default and only
|
||||
agent transport, so there is no `use_vsock` toggle to set; the config instead
|
||||
shows `use_legacy_serial = false`, confirming the guest console/agent path is on
|
||||
the modern virtio channel rather than a legacy serial port. The upshot: the
|
||||
agent control plane is isolated from the pod's data-plane networking entirely.
|
||||
|
||||
### Networking into the guest
|
||||
|
||||
`internetworking_model = "tcfilter"` is how the pod's CNI veth reaches the VM:
|
||||
Kata creates a TAP device for the guest NIC and installs a TC (traffic-control)
|
||||
filter that mirrors packets between the CNI-provided veth and the TAP. The pod
|
||||
keeps the IP Flannel assigned it; the VM transparently sits behind it. Egress
|
||||
restrictions are enforced one layer up by a NetworkPolicy — see
|
||||
[`04-arc-and-caching.md`](./04-arc-and-caching.md).
|
||||
|
||||
### `factory.enable_template = false` — a fresh VM per job, deliberately
|
||||
|
||||
Kata's **VM factory/templating** can pre-create a paused "template" VM and fork
|
||||
new microVMs from it via copy-on-write memory, shaving boot time. It is **off**
|
||||
here (`enable_template = false`, `vm_cache_number = 0`) on purpose: CI jobs must
|
||||
be mutually isolated and reproducible, so each job gets a **pristine VM built
|
||||
from scratch** with no memory state inherited from a previous job. The boot cost
|
||||
(a second or two) is an acceptable price for clean isolation, and the preloaded
|
||||
runner image ([`03-runner-image.md`](./03-runner-image.md)) is what removes the
|
||||
*real* per-job cost (dependency installs), not VM templating.
|
||||
|
||||
### Interaction with the overlayfs snapshotter
|
||||
|
||||
k3s's containerd uses the default **overlayfs** snapshotter. For a `runc` pod
|
||||
that overlay mount *is* the container root. For a Kata pod, containerd still
|
||||
builds the same overlayfs rootfs on the host, but because `shared_fs =
|
||||
"virtio-fs"` it is **exported into the guest by virtiofsd** rather than used
|
||||
directly. So the two cooperate cleanly: the snapshotter assembles image layers
|
||||
on the host (image pulls, layer caching, dedup all work normally), and virtio-fs
|
||||
projects the result into the microVM. No special snapshotter (devmapper /
|
||||
blockfile) is needed — that would only be required if you wanted the rootfs
|
||||
delivered as a block device instead of a shared filesystem.
|
||||
|
||||
---
|
||||
|
||||
## Step 5 — Verify microVM isolation
|
||||
|
||||
The defining test: a pod under `kata-qemu` must report a **different kernel**
|
||||
than the host. Run a throwaway pod with `--rm` so nothing is left behind. On
|
||||
your reproduction host:
|
||||
|
||||
```bash
|
||||
# Inside the microVM (Kata): guest kernel.
|
||||
kubectl run kata-smoke --rm -it --restart=Never \
|
||||
--image=busybox \
|
||||
--overrides='{"spec":{"runtimeClassName":"kata-qemu"}}' \
|
||||
-- uname -r
|
||||
```
|
||||
|
||||
Expected — the **guest** kernel:
|
||||
|
||||
```text
|
||||
6.18.28-194
|
||||
```
|
||||
|
||||
Compare with the host kernel:
|
||||
|
||||
```console
|
||||
$ uname -r
|
||||
7.0.10-1.el10.elrepo.x86_64
|
||||
```
|
||||
|
||||
Different kernel string == the workload is genuinely inside a separate guest
|
||||
kernel, not a namespaced host process. As a control, the same pod **without**
|
||||
the RuntimeClass runs on `runc` and prints the **host** kernel
|
||||
(`7.0.10-1.el10.elrepo.x86_64`) — proving the difference comes from Kata, not
|
||||
the image.
|
||||
|
||||
For a closer look at the VM's resources (confirming the hotplug sizing from Step
|
||||
4), use an image with more tools:
|
||||
|
||||
```bash
|
||||
kubectl run kata-smoke --rm -it --restart=Never \
|
||||
--image=ubuntu:24.04 \
|
||||
--overrides='{"spec":{"runtimeClassName":"kata-qemu"}}' \
|
||||
-- bash -lc 'uname -r; nproc; grep MemTotal /proc/meminfo'
|
||||
```
|
||||
|
||||
This prints the guest kernel, the hotplugged vCPU count, and guest RAM (MiB) —
|
||||
which track the pod's `resources.limits`, not the host's 32c/125G.
|
||||
|
||||
> **On the live host, do not run throwaway pods.** It is production CI. Use the
|
||||
> read-only host checks instead: `kata-runtime check` and `kata-runtime env`
|
||||
> (Step 1) prove the runtime is healthy without scheduling anything. The
|
||||
> equivalent VM-boot proof for the actual runner image is the `preload-verify`
|
||||
> recipe documented in [`03-runner-image.md`](./03-runner-image.md).
|
||||
|
||||
---
|
||||
|
||||
## Recap
|
||||
|
||||
1. Static tarball extracted to `/opt/kata` (QEMU + guest kernel + rootfs +
|
||||
virtiofsd + shim), shim and CLI symlinked onto `PATH`.
|
||||
2. Drop-in `config-v3.toml.d/kata.toml` registers the `kata-qemu` runtime
|
||||
(`io.containerd.kata.v2`, pinned `runtime_path`, `ConfigPath`), merged via
|
||||
k3s's generated `imports`; picked up on `systemctl restart k3s`.
|
||||
3. `RuntimeClass/kata-qemu` (`handler: kata-qemu`) lets pods opt in.
|
||||
4. `configuration-qemu.toml` boots a small QEMU q35 VM (1 vCPU / 2 GiB) that
|
||||
hotplugs up to the pod's limits, shares the container rootfs in over
|
||||
virtio-fs, talks to the agent over vsock, and builds a fresh VM per job
|
||||
(templating off).
|
||||
5. A `kata-qemu` pod reports guest kernel `6.18.28-194` vs host
|
||||
`7.0.10-1.el10.elrepo.x86_64` — isolation confirmed.
|
||||
|
||||
Continue to [`03-runner-image.md`](./03-runner-image.md) to build the runner
|
||||
image that boots inside these microVMs.
|
||||
@@ -0,0 +1,466 @@
|
||||
# 03 - Preloaded runner image
|
||||
|
||||
Each ephemeral CI job boots a fresh Kata microVM and runs inside a single
|
||||
runner container. To keep that cold microVM **warm** (no per-job dependency
|
||||
fetches), the container does not use the stock GitHub Actions runner image
|
||||
directly: it uses a locally built image that bakes every dependency CI installs
|
||||
on every job into the snapshot. This document covers building that image,
|
||||
importing it into k3s containerd, pointing ARC at it, and verifying it.
|
||||
|
||||
Navigation: previous - [02-kata-runtime.md](./02-kata-runtime.md) (Kata +
|
||||
containerd + RuntimeClass) | next - [04-arc-and-caching.md](./04-arc-and-caching.md)
|
||||
(ARC runners, RustFS cache, egress policy).
|
||||
|
||||
> The image built here is referenced by the ARC runner pod template
|
||||
> (`template.spec.containers[0].image` + `imagePullPolicy: IfNotPresent`),
|
||||
> documented in [04-arc-and-caching.md](./04-arc-and-caching.md). The pinned
|
||||
> `runtimeClassName: kata-qemu` on that same template comes from
|
||||
> [02-kata-runtime.md](./02-kata-runtime.md).
|
||||
|
||||
All host commands below run on `<CI_HOST>` (the single k3s node) as root.
|
||||
|
||||
---
|
||||
|
||||
## 1. Why a preloaded image
|
||||
|
||||
The base `ghcr.io/actions/actions-runner:latest` is a clean Ubuntu 24.04 runner.
|
||||
On a normal (GitHub-hosted-style) runner, the CI workflow installs its system
|
||||
dependencies at the start of every job: the cairo/pango native stack for canvas
|
||||
builds, `fd`/`ripgrep`/`imagemagick`, `bun`, `sccache`, Zig, the cargo-native
|
||||
helper CLIs (`cargo-nextest`, `cargo-zigbuild`, `cargo-xwin`), and a pinned Rust
|
||||
nightly with the cross targets/components. Inside a Kata microVM that is
|
||||
destroyed after a single job, paying that apt/bun/rustup/tool-download cost on
|
||||
**every** job is pure latency - the microVM starts cold each time.
|
||||
|
||||
The preloaded image moves that work to build time. Every ephemeral runner then
|
||||
starts with the toolchain already present: apt deps are not re-fetched, `bun`,
|
||||
`cargo`/`rustc`, `sccache`, `zig`, and the cargo helper CLIs are on `PATH`, and
|
||||
the pinned Rust toolchain is already the default so target/component installs in
|
||||
CI become no-ops.
|
||||
|
||||
### Stay in sync with `setup-system-deps`
|
||||
|
||||
The apt set baked into the image **must** match the repo's
|
||||
`.github/actions/setup-system-deps` composite action. That action is the
|
||||
self-healing counterpart: it probes for the baked tools and skips the apt
|
||||
round-trip when they are present (preloaded image), but installs the exact same
|
||||
set on a stock runner so CI still works anywhere. Its detection probes are:
|
||||
|
||||
```bash
|
||||
command -v fd && command -v rg && command -v magick \
|
||||
&& pkg-config --exists cairo pango
|
||||
```
|
||||
|
||||
If you add a dependency that the action probes for, add it in **both** the
|
||||
Dockerfile apt line and `setup-system-deps`. If they drift, either the action
|
||||
re-installs deps the image already has (slow) or CI breaks on a tool the image
|
||||
forgot to bake.
|
||||
|
||||
---
|
||||
|
||||
## 2. The Dockerfile
|
||||
|
||||
Build context lives at `/root/omp-kata-runner-image/`. The Dockerfile below is
|
||||
reproduced verbatim (it contains no secrets or redactable host identifiers; the
|
||||
`ARG` pins, the full apt set, and the toolchain steps are real).
|
||||
|
||||
> The canonical copy of this Dockerfile is version-controlled at
|
||||
> [`infra/runner.Dockerfile`](../runner.Dockerfile);
|
||||
> the `/root/omp-kata-runner-image/` copy is overwritten from it by the
|
||||
> repo-driven reload script (see section 3 below).
|
||||
|
||||
```dockerfile
|
||||
# syntax=docker/dockerfile:1
|
||||
# Preloaded omp-kata runner image.
|
||||
#
|
||||
# Stock GitHub Actions runner (Ubuntu 24.04) with the dependencies CI installs
|
||||
# on every job baked in, so each ephemeral Kata microVM boots with them already
|
||||
# present instead of re-fetching them per job:
|
||||
# - APT system deps (canvas/cairo stack + fd/ripgrep/imagemagick) + fd/magick shims
|
||||
# - GitHub CLI (gh) — present on GitHub-hosted runners; the coding-agent github
|
||||
# tool and release workflows expect it
|
||||
# - C/build toolchain the native + canvas builds need
|
||||
# - bun (system-wide, on PATH)
|
||||
# - sccache + Zig + cargo-nextest/cargo-zigbuild/cargo-xwin for native builds
|
||||
# - rust nightly (pinned) + clippy/rustfmt/rust-analyzer + linux-arm64/windows-msvc targets
|
||||
#
|
||||
# Rebuild + reimport (see /root/omp-kata-runner.md) after bumping the ARGs below
|
||||
# or the apt set. Keep the apt set in sync with .github/actions/setup-system-deps.
|
||||
FROM ghcr.io/actions/actions-runner:latest
|
||||
|
||||
ARG RUST_NIGHTLY=nightly-2026-04-29
|
||||
ARG BUN_VERSION=1.3.14
|
||||
ARG SCCACHE_VERSION=0.15.0
|
||||
ARG ZIG_VERSION=0.16.0
|
||||
|
||||
USER root
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
# Mirrors the "Install system deps" block in .github/workflows/ci.yml plus the
|
||||
# native/cross toolchain (clang/lld/llvm), the baked cache/tooling binaries, and
|
||||
# the GitHub CLI. The gh apt repo is added first so `gh` installs in the same apt
|
||||
# transaction.
|
||||
RUN curl -fsSL https://cli.github.com/packages/githubcli-archive-keyring.gpg -o /usr/share/keyrings/githubcli-archive-keyring.gpg \
|
||||
&& chmod go+r /usr/share/keyrings/githubcli-archive-keyring.gpg \
|
||||
&& echo "deb [arch=$(dpkg --print-architecture) signed-by=/usr/share/keyrings/githubcli-archive-keyring.gpg] https://cli.github.com/packages stable main" > /etc/apt/sources.list.d/github-cli.list \
|
||||
&& apt-get update \
|
||||
&& apt-get install -y \
|
||||
build-essential pkg-config curl ca-certificates git unzip xz-utils zstd gh \
|
||||
clang lld llvm \
|
||||
libcairo2-dev libpango1.0-dev libjpeg-dev libgif-dev librsvg2-dev \
|
||||
fd-find ripgrep imagemagick \
|
||||
&& ln -sf "$(command -v fdfind)" /usr/local/bin/fd \
|
||||
&& ln -sf /usr/bin/convert /usr/local/bin/magick \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# bun, system-wide (BUN_INSTALL/bin == /usr/local/bin, already on PATH).
|
||||
ENV BUN_INSTALL=/usr/local
|
||||
RUN curl -fsSL https://bun.sh/install | bash -s "bun-v${BUN_VERSION}" \
|
||||
&& bun --version
|
||||
|
||||
# Pinned native-build helpers, system-wide.
|
||||
RUN curl -fsSL "https://github.com/mozilla/sccache/releases/download/v${SCCACHE_VERSION}/sccache-v${SCCACHE_VERSION}-x86_64-unknown-linux-musl.tar.gz" \
|
||||
| tar -xz -C /tmp \
|
||||
&& install -m755 "/tmp/sccache-v${SCCACHE_VERSION}-x86_64-unknown-linux-musl/sccache" /usr/local/bin/sccache \
|
||||
&& rm -rf "/tmp/sccache-v${SCCACHE_VERSION}-x86_64-unknown-linux-musl"
|
||||
RUN curl -fsSL "https://ziglang.org/download/${ZIG_VERSION}/zig-x86_64-linux-${ZIG_VERSION}.tar.xz" -o /tmp/zig.tar.xz \
|
||||
&& tar -xJf /tmp/zig.tar.xz -C /opt \
|
||||
&& ln -sf "/opt/zig-x86_64-linux-${ZIG_VERSION}/zig" /usr/local/bin/zig \
|
||||
&& rm -f /tmp/zig.tar.xz
|
||||
|
||||
# rust toolchain + cargo helpers for the runner user; rustup default == pinned
|
||||
# nightly so Rust setup becomes a no-op on the preloaded image.
|
||||
USER runner
|
||||
ENV RUSTUP_HOME=/home/runner/.rustup \
|
||||
CARGO_HOME=/home/runner/.cargo \
|
||||
PATH=/home/runner/.cargo/bin:/usr/local/bin:${PATH}
|
||||
RUN curl --proto '=https' --tlsv1.2 -fsSL https://sh.rustup.rs \
|
||||
| sh -s -- -y --default-toolchain "${RUST_NIGHTLY}" --profile minimal \
|
||||
&& rustup component add clippy rustfmt rust-analyzer \
|
||||
&& rustup target add aarch64-unknown-linux-gnu x86_64-pc-windows-msvc \
|
||||
&& cargo install --locked cargo-nextest cargo-zigbuild cargo-xwin \
|
||||
&& cargo --version \
|
||||
&& rustc --version \
|
||||
&& sccache --version \
|
||||
&& zig version \
|
||||
&& cargo-nextest --version \
|
||||
&& cargo-zigbuild --help >/dev/null \
|
||||
&& cargo-xwin --help >/dev/null
|
||||
```
|
||||
|
||||
### Stage-by-stage annotation
|
||||
|
||||
**`# syntax=docker/dockerfile:1` + `FROM ghcr.io/actions/actions-runner:latest`.**
|
||||
BuildKit frontend pin, then the stock Actions runner base (Ubuntu 24.04). The
|
||||
base already ships the runner agent and its `/home/runner/run.sh` entrypoint, the
|
||||
non-root `runner` user, and `sudo`. Everything below layers onto that base; the
|
||||
runner agent itself is never modified.
|
||||
|
||||
**`ARG RUST_NIGHTLY` / `ARG BUN_VERSION`.** The two version knobs you bump. They
|
||||
are build args so you can also override them ad hoc with
|
||||
`docker build --build-arg RUST_NIGHTLY=... --build-arg BUN_VERSION=...` without
|
||||
editing the file. `RUST_NIGHTLY` must match what the repo's
|
||||
`dtolnay/rust-toolchain@nightly` step expects so the toolchain install in CI is a
|
||||
no-op (see step 4 below).
|
||||
|
||||
**`USER root` + `ENV DEBIAN_FRONTEND=noninteractive`.** Switch to root for the
|
||||
apt and bun system installs; `noninteractive` suppresses debconf/tzdata prompts
|
||||
during `apt-get install`.
|
||||
|
||||
**The apt `RUN` block.** This is the set that must mirror `setup-system-deps`.
|
||||
In order:
|
||||
- The first three lines add the **GitHub CLI apt repository** (keyring + signed
|
||||
source list) *before* `apt-get update`, so `gh` resolves and installs in the
|
||||
same apt transaction as everything else. `gh` is present on GitHub-hosted
|
||||
runners and is expected by the release workflows and the coding-agent `github`
|
||||
tool.
|
||||
- `apt-get install` pulls three groups:
|
||||
- **build toolchain / utilities:** `build-essential pkg-config curl
|
||||
ca-certificates git unzip xz-utils zstd gh clang lld llvm`.
|
||||
`build-essential` + `pkg-config` are needed by the native and canvas builds;
|
||||
`zstd` is the codec the bun and sccache cache tarballs use (see
|
||||
[04-arc-and-caching.md](./04-arc-and-caching.md)); `clang lld llvm` are the
|
||||
MSVC-cross prerequisites that used to be apt-installed per job.
|
||||
- **canvas / cairo native stack:** `libcairo2-dev libpango1.0-dev libjpeg-dev
|
||||
libgif-dev librsvg2-dev` - the `-dev` headers the canvas/rsvg native modules
|
||||
compile against.
|
||||
- **CLI tools:** `fd-find ripgrep imagemagick`, used by the agent and tests.
|
||||
- **The two shims** normalize Debian's binary names to what callers expect:
|
||||
Debian ships `fd` as `fdfind`, so `ln -sf "$(command -v fdfind)"
|
||||
/usr/local/bin/fd` exposes it as `fd`; ImageMagick installs `convert`, so
|
||||
`ln -sf /usr/bin/convert /usr/local/bin/magick` exposes the v7-style `magick`
|
||||
name. These two shims are exactly what `setup-system-deps` recreates on a stock
|
||||
runner.
|
||||
- `rm -rf /var/lib/apt/lists/*` drops the apt index to keep the layer smaller.
|
||||
|
||||
**bun (`ENV BUN_INSTALL=/usr/local` + install `RUN`).** Setting
|
||||
`BUN_INSTALL=/usr/local` makes the official installer drop the binary at
|
||||
`/usr/local/bin/bun`, which is already on `PATH` for every user - so bun is
|
||||
**system-wide** with no per-user shell init. The version is pinned via
|
||||
`bun-v${BUN_VERSION}`, and `bun --version` fails the build if the install is
|
||||
broken.
|
||||
|
||||
**Pinned native-build helpers (two root `RUN`s).** `sccache` is downloaded as a
|
||||
version-pinned GitHub release tarball and installed to `/usr/local/bin`; Zig is
|
||||
downloaded as the pinned release archive, unpacked under `/opt`, and symlinked
|
||||
into `/usr/local/bin/zig`. Baking these two removes the per-job
|
||||
`mozilla-actions/sccache-action` and `mlugg/setup-zig` downloads from the
|
||||
self-hosted path.
|
||||
|
||||
**Rust toolchain (`USER runner` + rustup `RUN`).** The toolchain is installed as
|
||||
the **`runner` user** - the UID jobs execute as - so cargo/rustc are owned by and
|
||||
visible to the job without sudo. `RUSTUP_HOME`/`CARGO_HOME` are pinned under
|
||||
`/home/runner`, and `~/.cargo/bin` is prepended to `PATH`. rustup installs the
|
||||
pinned nightly as the **default toolchain** (`--profile minimal`), then adds the
|
||||
`clippy`, `rustfmt`, and `rust-analyzer` components plus the
|
||||
`aarch64-unknown-linux-gnu` (Linux arm64) and `x86_64-pc-windows-msvc` (Windows
|
||||
cross) targets. The same layer also `cargo install`s the Rust-native helper CLIs
|
||||
`cargo-nextest`, `cargo-zigbuild`, and `cargo-xwin`, so the self-hosted native
|
||||
build path no longer fetches those tools job-by-job. Because the default toolchain
|
||||
already *is* the pinned nightly with these components/targets, the corresponding
|
||||
Rust setup steps in CI become no-ops - the warm-start payoff.
|
||||
|
||||
---
|
||||
|
||||
## 3. Build, import, and roll out (`reload.sh`)
|
||||
|
||||
`/root/omp-kata-runner-image/reload.sh` does the whole cycle: build, in-image
|
||||
smoke test, import into k3s containerd, point ARC at the new tag, and verify the
|
||||
rollout. It is idempotent and cache-fast on an unchanged rebuild. Reproduced
|
||||
verbatim (no secrets; the `/root` and kubeconfig paths are the real host paths):
|
||||
|
||||
```bash
|
||||
#!/usr/bin/env bash
|
||||
# Rebuild the preloaded omp-kata runner image, import it into k3s containerd,
|
||||
# point the ARC runner scale set at it, and roll it out. Idempotent: safe to
|
||||
# re-run after editing ./Dockerfile. Docker layer cache makes an unchanged
|
||||
# rebuild near-instant.
|
||||
#
|
||||
# ./reload.sh # build tag omp-kata-runner:YYYY-MM-DD-HHMMSS
|
||||
# ./reload.sh 2026-06-20 # build tag omp-kata-runner:2026-06-20
|
||||
# ./reload.sh foo:bar # build an explicit repo:tag
|
||||
set -euo pipefail
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
cd "$(dirname "$0")"
|
||||
|
||||
arg="${1:-$(date +%Y-%m-%d-%H%M%S)}"
|
||||
case "$arg" in *:*) IMAGE="$arg";; *) IMAGE="omp-kata-runner:$arg";; esac
|
||||
|
||||
echo "==> [1/5] building $IMAGE"
|
||||
DOCKER_BUILDKIT=1 docker build -t "$IMAGE" -t omp-kata-runner:preloaded .
|
||||
|
||||
echo "==> [2/5] verifying baked tools"
|
||||
docker run --rm --entrypoint bash "$IMAGE" -lc '
|
||||
set -e
|
||||
for b in gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do
|
||||
command -v "$b" >/dev/null || { echo "MISSING: $b"; exit 1; }
|
||||
done
|
||||
echo "tools OK | bun $(bun --version) | rust $(rustc --version) | sccache $(sccache --version | awk '\''{print $2}'\'') | zig $(zig version) | gh $(gh --version | head -1 | cut -d\" \" -f3)"
|
||||
'
|
||||
|
||||
echo "==> [3/5] importing into k3s containerd (k8s.io namespace)"
|
||||
docker save "$IMAGE" | k3s ctr -n k8s.io images import --platform linux/amd64 -
|
||||
|
||||
echo "==> [4/5] pointing ARC runner scale set at $IMAGE"
|
||||
sed -i "s#image: omp-kata-runner:.*#image: $IMAGE#" /root/arc-omp-values.yaml
|
||||
helm upgrade omp-kata --namespace arc-runners --version 0.14.2 \
|
||||
-f /root/arc-omp-values.yaml \
|
||||
oci://ghcr.io/actions/actions-runner-controller-charts/gha-runner-scale-set >/dev/null
|
||||
|
||||
echo "==> [5/5] verifying rollout"
|
||||
live="$(kubectl get autoscalingrunnerset omp-kata -n arc-runners -o jsonpath='{.spec.template.spec.containers[0].image}')"
|
||||
echo "ARC runner image is now: $live"
|
||||
[ "$live" = "$IMAGE" ] && echo "OK: reloaded $IMAGE" || { echo "MISMATCH: expected $IMAGE"; exit 1; }
|
||||
```
|
||||
|
||||
Run it with no argument for an auto-dated tag:
|
||||
|
||||
```bash
|
||||
cd /root/omp-kata-runner-image
|
||||
./reload.sh
|
||||
```
|
||||
|
||||
### What each step does
|
||||
|
||||
**Preamble.** `set -euo pipefail` aborts on the first error; `KUBECONFIG` points
|
||||
at the k3s admin config; `cd` into the build context. The tag is resolved from
|
||||
`$1`: no arg gives a timestamped `omp-kata-runner:YYYY-MM-DD-HHMMSS`; an argument
|
||||
containing a colon (`foo:bar`) is used as an explicit `repo:tag`; anything else
|
||||
is treated as a tag suffix on `omp-kata-runner:`.
|
||||
|
||||
**[1/5] build.** `DOCKER_BUILDKIT=1 docker build` tags the result twice: the
|
||||
immutable `$IMAGE` (dated) and the moving `omp-kata-runner:preloaded` alias.
|
||||
BuildKit + the docker layer cache make an unchanged rebuild near-instant.
|
||||
|
||||
**[2/5] verify baked tools.** Runs the freshly built image with a bash entrypoint
|
||||
and asserts every expected binary is on `PATH`
|
||||
(`gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin`),
|
||||
failing the whole script if any is missing, then prints the key version tuple
|
||||
(bun / rust / sccache / zig / gh). This catches a broken apt set, missing shim,
|
||||
or bad toolchain pin **before** anything touches the cluster.
|
||||
|
||||
**[3/5] import into k3s containerd.**
|
||||
`docker save "$IMAGE" | k3s ctr -n k8s.io images import --platform linux/amd64 -`
|
||||
streams the image tarball straight from the docker daemon into k3s's **own**
|
||||
containerd, in the `k8s.io` namespace - the namespace the kubelet pulls from.
|
||||
`--platform linux/amd64` matches the host architecture. Note `docker save` is
|
||||
given only the dated `$IMAGE`, so **only the dated tag is imported**; the
|
||||
`:preloaded` alias stays a docker-local convenience and is never imported or
|
||||
referenced by ARC.
|
||||
|
||||
**[4/5] point ARC at the new tag.** `sed -i` rewrites the single
|
||||
`image: omp-kata-runner:...` line in `/root/arc-omp-values.yaml` to the new tag,
|
||||
then `helm upgrade` re-renders the runner scale set with the chart pinned to
|
||||
`0.14.2`. (That values file is the runner pod template, documented in
|
||||
[04-arc-and-caching.md](./04-arc-and-caching.md).)
|
||||
|
||||
**[5/5] verify rollout.** Reads the image back off the live
|
||||
`autoscalingrunnerset` via `kubectl ... jsonpath` and asserts it equals `$IMAGE`,
|
||||
printing `OK` or exiting non-zero with `MISMATCH`. New ephemeral runner pods
|
||||
created after this point boot from the new image; in-flight jobs finish on the
|
||||
old one (the scale set is scale-to-zero, so this drains quickly).
|
||||
|
||||
### Running it from the repo (over SSH)
|
||||
|
||||
You do not have to keep `reload.sh` on the host. The repo ships the version-
|
||||
controlled Dockerfile plus an SSH-driven wrapper that performs the rollout
|
||||
remotely from a checkout:
|
||||
|
||||
- [`infra/runner.Dockerfile`](../runner.Dockerfile) - the image definition (source of truth).
|
||||
- [`infra/reload-runner.sh`](../reload-runner.sh) - copies that Dockerfile to the host, then prefers a **direct containerd build path**: bootstrap pinned `buildkitd` + `buildctl` + `nerdctl` under the remote build dir if needed, build straight into the k3s `k8s.io` namespace, smoke-test from that image store, then `helm upgrade` ARC. Set `BUILD_BACKEND=docker` to force the legacy `docker build` + `docker save | ctr images import` path.
|
||||
|
||||
The host is never hardcoded; point it at your node with `CI_HOST`:
|
||||
|
||||
```bash
|
||||
CI_HOST=<CI_HOST> ./infra/reload-runner.sh # dated tag
|
||||
CI_HOST=<CI_HOST> ./infra/reload-runner.sh 2026-06-20 # explicit tag
|
||||
```
|
||||
|
||||
It honors the same defaults as the host script (remote build dir, ARC values
|
||||
path, release name, namespace, chart version), each overridable via the
|
||||
environment variables documented in the script header.
|
||||
|
||||
---
|
||||
|
||||
## 4. Why import into k3s containerd instead of using a registry
|
||||
|
||||
This is a single-node cluster, and the **only** consumer of the runner image is
|
||||
the kubelet/containerd on that same node. A registry would add a service to run,
|
||||
secure, and authenticate against, for zero benefit. Instead:
|
||||
|
||||
- `docker save | k3s ctr -n k8s.io images import` places the image directly into
|
||||
the containerd instance k3s schedules from. containerd normalizes the short
|
||||
reference `omp-kata-runner:<tag>` to `docker.io/library/omp-kata-runner:<tag>`
|
||||
in its store (verified: the imported tags appear under that prefix).
|
||||
- The ARC pod template sets `imagePullPolicy: IfNotPresent`. Because the image is
|
||||
already present locally, the kubelet **uses the local copy and never attempts a
|
||||
pull** - no registry, no pull credentials, no registry egress (which the
|
||||
runner egress lockdown in [04-arc-and-caching.md](./04-arc-and-caching.md)
|
||||
would block anyway).
|
||||
|
||||
Trade-off: the image must be (re-)imported on every node that schedules runners.
|
||||
Here that is exactly one node, so a re-roll is simply rebuild + re-import, and the
|
||||
next job's microVM starts cold but with warm dependencies from the local store.
|
||||
|
||||
---
|
||||
|
||||
## 5. Tag conventions
|
||||
|
||||
| Tag | Mutability | Imported into containerd? | Referenced by ARC? | Purpose |
|
||||
| --- | --- | --- | --- | --- |
|
||||
| `omp-kata-runner:YYYY-MM-DD-HHMMSS` | immutable | yes | yes | the build of record; what runners actually boot |
|
||||
| `omp-kata-runner:preloaded` | moving | no | no | docker-local alias to the most recent build |
|
||||
|
||||
- The default `reload.sh` tag is timestamped (`date +%Y-%m-%d-%H%M%S`). You can
|
||||
also pass a date-only tag (`./reload.sh 2026-06-20`) or an explicit `repo:tag`.
|
||||
- ARC always pins the **immutable dated tag**, never `:preloaded`. That keeps a
|
||||
rollout reproducible and makes rollback trivial: `sed` the image line back to
|
||||
the prior dated tag (it is still in the local store) and `helm upgrade`.
|
||||
- Live example at time of writing: the scale set references
|
||||
`omp-kata-runner:2026-06-15-002621`, held in containerd as
|
||||
`docker.io/library/omp-kata-runner:2026-06-15-002621`.
|
||||
|
||||
---
|
||||
|
||||
## 6. Bumping bun / Rust / the apt set and re-rolling
|
||||
|
||||
1. **bun:** edit `ARG BUN_VERSION=` in the Dockerfile (or pass
|
||||
`--build-arg BUN_VERSION=...`).
|
||||
2. **Rust:** edit `ARG RUST_NIGHTLY=` to the new pinned nightly. Keep it equal to
|
||||
what the repo's `dtolnay/rust-toolchain@nightly` step resolves, so the CI
|
||||
toolchain install stays a no-op.
|
||||
3. **apt set:** edit the `apt-get install` line. You **must** mirror the change in
|
||||
`.github/actions/setup-system-deps` (and, if you add a tool the action probes
|
||||
for, in its detection block - currently `fd`, `rg`, `magick`,
|
||||
`pkg-config --exists cairo pango`).
|
||||
4. Re-roll:
|
||||
|
||||
```bash
|
||||
cd /root/omp-kata-runner-image
|
||||
./reload.sh
|
||||
```
|
||||
|
||||
The rebuild is cache-fast for unchanged layers, the baked-tools check guards
|
||||
the change, and the import/helm-upgrade/verify steps roll the new tag out.
|
||||
The next job's microVM boots warm with the updated toolchain.
|
||||
|
||||
---
|
||||
|
||||
## 7. Verification
|
||||
|
||||
### Baked-tools check (any tag)
|
||||
|
||||
`reload.sh` step 2 already runs this on every build. To re-check an existing tag
|
||||
standalone:
|
||||
|
||||
```bash
|
||||
docker run --rm --entrypoint bash omp-kata-runner:preloaded -lc '
|
||||
set -e
|
||||
for b in gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do
|
||||
command -v "$b" >/dev/null || { echo "MISSING: $b"; exit 1; }
|
||||
done
|
||||
echo "tools OK | bun $(bun --version) | rust $(rustc --version) | sccache $(set -- $(sccache --version); echo "$2") | zig $(zig version) | gh $(set -- $(gh --version | head -1); echo "$3")"
|
||||
'
|
||||
```
|
||||
|
||||
Confirm the live ARC reference and that the tag exists in the k3s image store
|
||||
(both read-only):
|
||||
|
||||
```bash
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
kubectl get autoscalingrunnerset omp-kata -n arc-runners \
|
||||
-o jsonpath='{.spec.template.spec.containers[0].image}{"\n"}'
|
||||
k3s ctr -n k8s.io images ls | grep omp-kata-runner
|
||||
```
|
||||
|
||||
### Kata microVM boot check
|
||||
|
||||
The baked-tools check above runs the image under plain docker; it does **not**
|
||||
prove the image boots inside a Kata QEMU/KVM microVM. To verify that, launch a
|
||||
throwaway pod with `runtimeClassName: kata-qemu` (the RuntimeClass set up in
|
||||
[02-kata-runtime.md](./02-kata-runtime.md)) and confirm both the deps and the
|
||||
**guest** kernel:
|
||||
|
||||
```bash
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
kubectl run preload-verify -n arc-runners --restart=Never \
|
||||
--image=omp-kata-runner:2026-06-15-002621 \
|
||||
--overrides='{"spec":{"runtimeClassName":"kata-qemu"}}' \
|
||||
--command -- bash -lc 'uname -r; bun --version; rustc --version; magick -version | head -1'
|
||||
|
||||
kubectl logs preload-verify -n arc-runners
|
||||
kubectl delete pod preload-verify -n arc-runners
|
||||
```
|
||||
|
||||
Use the tag that is currently live (or any imported tag). `uname -r` should show
|
||||
the Kata **guest** kernel (a 6.x `vmlinux.container` build), not the host kernel -
|
||||
confirming the image really booted in its own microVM - and the `bun`/`rustc`/
|
||||
`magick` lines confirm the baked toolchain is present inside the VM. This `kubectl
|
||||
run` is the only step here that creates a cluster object; delete the pod
|
||||
afterward as shown.
|
||||
|
||||
---
|
||||
|
||||
Continue to [04-arc-and-caching.md](./04-arc-and-caching.md) for how ARC
|
||||
references this image in the runner pod template, wires in the RustFS shared
|
||||
cache, and locks down runner egress.
|
||||
@@ -0,0 +1,683 @@
|
||||
# 04 - ARC runners, shared cache, and egress policy
|
||||
|
||||
This is the last setup step. By now the node runs k3s with the `kata-qemu`
|
||||
RuntimeClass ([02-kata-runtime.md](02-kata-runtime.md)) and the preloaded runner
|
||||
image has been imported into the cluster containerd ([03-runner-image.md](03-runner-image.md)).
|
||||
Here we install **actions-runner-controller (ARC)**, register an ephemeral
|
||||
**scale set** whose pods each boot inside their own Kata microVM, stand up the
|
||||
in-cluster **RustFS (S3)** shared cache, and lock down runner egress with a
|
||||
NetworkPolicy. See [README.md](README.md) for the architecture overview.
|
||||
|
||||
Everything below is read against the live cluster; set the kubeconfig once:
|
||||
|
||||
```bash
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
```
|
||||
|
||||
ARC's `gha-runner-scale-set` flavour has three moving parts:
|
||||
|
||||
- **Controller** (`arc` release, ns `arc-systems`) - watches `AutoscalingRunnerSet`
|
||||
custom resources and reconciles them.
|
||||
- **Listener** (one pod per scale set, ns `arc-systems`) - long-polls the GitHub
|
||||
Actions service for jobs targeting the scale set's `runs-on` label.
|
||||
- **Scale set** (`omp-kata` release, ns `arc-runners`) - the `AutoscalingRunnerSet`
|
||||
plus the pod template; the controller turns assigned jobs into ephemeral runner
|
||||
pods here.
|
||||
|
||||
---
|
||||
|
||||
## 1. GitHub App and the `arc-github` secret
|
||||
|
||||
The listener authenticates to GitHub. The durable option is a **GitHub App**
|
||||
(no expiring user token, scoped to exactly the repos you install it on).
|
||||
|
||||
1. Create the App at **GitHub - Settings - Developer settings - GitHub Apps - New GitHub App**.
|
||||
- **Repository permissions**: `Administration: Read and write` (register/remove
|
||||
self-hosted runners) and `Metadata: Read-only` (granted automatically).
|
||||
- No webhook is needed for the scale-set flavour; uncheck **Active** under Webhook.
|
||||
- Generate and download a **private key** (`.pem`).
|
||||
2. **Install** the App on the target repo or org (App page - **Install App** -
|
||||
pick `<OWNER>/<REPO>` or "All repositories"). Note the **App ID** and the
|
||||
**Installation ID** (the trailing number in the install settings URL,
|
||||
`.../installations/<id>`).
|
||||
3. Create the secret in the runners namespace. The three key names below are
|
||||
exactly what the chart reads:
|
||||
|
||||
```bash
|
||||
kubectl create namespace arc-runners
|
||||
|
||||
kubectl -n arc-runners create secret generic arc-github \
|
||||
--from-literal=github_app_id=<GITHUB_APP_ID> \
|
||||
--from-literal=github_app_installation_id=<GITHUB_APP_INSTALLATION_ID> \
|
||||
--from-literal=github_app_private_key=<GITHUB_APP_PRIVATE_KEY>
|
||||
```
|
||||
|
||||
`<GITHUB_APP_PRIVATE_KEY>` is the full PEM body (use `--from-file=github_app_private_key=key.pem`
|
||||
to avoid shell-quoting the multi-line value).
|
||||
|
||||
Verify the live secret carries those three keys (names only - never print values):
|
||||
|
||||
```bash
|
||||
kubectl -n arc-runners get secret arc-github \
|
||||
-o go-template='{{range $k,$v := .data}}{{$k}}{{"\n"}}{{end}}'
|
||||
# github_app_id
|
||||
# github_app_installation_id
|
||||
# github_app_private_key
|
||||
```
|
||||
|
||||
**Token alternative.** ARC also accepts a single-key secret with a classic PAT
|
||||
(scope `repo`) or a fine-grained PAT (`Administration: RW` + `Metadata: R`):
|
||||
|
||||
```bash
|
||||
kubectl -n arc-runners create secret generic arc-github \
|
||||
--from-literal=github_token=<GITHUB_PAT>
|
||||
```
|
||||
|
||||
The App is preferred: it does not expire, it is scoped per-installation, and one
|
||||
installation covers every repo you grant it (useful for [adding another repo](#7-operate)).
|
||||
Whichever you choose, the `githubConfigSecret` value in step 3's chart points at
|
||||
this secret by name.
|
||||
|
||||
---
|
||||
|
||||
## 2. Install ARC (controller + scale set)
|
||||
|
||||
ARC ships as OCI Helm charts; no `helm repo add` is required. Both the controller
|
||||
and the scale set are pinned to the same chart version, **0.14.2** (matches the
|
||||
live `helm list -A`).
|
||||
|
||||
**Controller** (installed with chart defaults - `helm get values arc` is empty):
|
||||
|
||||
```bash
|
||||
helm install arc \
|
||||
--namespace arc-systems --create-namespace \
|
||||
--version 0.14.2 \
|
||||
oci://ghcr.io/actions/actions-runner-controller-charts/gha-runner-scale-set-controller
|
||||
```
|
||||
|
||||
**Scale set** (`omp-kata`), using the values file from step 3:
|
||||
|
||||
```bash
|
||||
helm install omp-kata \
|
||||
--namespace arc-runners --create-namespace \
|
||||
--version 0.14.2 \
|
||||
-f arc-omp-values.yaml \
|
||||
oci://ghcr.io/actions/actions-runner-controller-charts/gha-runner-scale-set
|
||||
```
|
||||
|
||||
Confirm both releases and the running controller image:
|
||||
|
||||
```bash
|
||||
helm list -A
|
||||
# arc arc-systems deployed gha-runner-scale-set-controller-0.14.2 0.14.2
|
||||
# omp-kata arc-runners deployed gha-runner-scale-set-0.14.2 0.14.2
|
||||
|
||||
kubectl -n arc-systems get deploy arc-gha-rs-controller \
|
||||
-o jsonpath='{.spec.template.spec.containers[0].image}{"\n"}'
|
||||
# ghcr.io/actions/gha-runner-scale-set-controller:0.14.2
|
||||
```
|
||||
|
||||
Within a few seconds the controller spawns the listener in `arc-systems`:
|
||||
|
||||
```bash
|
||||
kubectl -n arc-systems get pods
|
||||
# arc-gha-rs-controller-xxxxxxxxxx-xxxxx 1/1 Running
|
||||
# omp-kata-<hash>-listener 1/1 Running
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. Scale-set values (`arc-omp-values.yaml`)
|
||||
|
||||
This is the live `arc-omp-values.yaml` verbatim, with only the repo owner/name in
|
||||
`githubConfigUrl` redacted:
|
||||
|
||||
```yaml
|
||||
githubConfigUrl: "https://github.com/<OWNER>/<REPO>"
|
||||
githubConfigSecret: arc-github
|
||||
runnerScaleSetName: omp-kata
|
||||
minRunners: 0
|
||||
maxRunners: 10
|
||||
# none: each job runs inside the runner container, which itself lives in a Kata microVM
|
||||
containerMode:
|
||||
type: ""
|
||||
template:
|
||||
spec:
|
||||
runtimeClassName: kata-qemu # <-- every runner pod boots its own KVM microVM
|
||||
containers:
|
||||
- name: runner
|
||||
# Preloaded image: stock ghcr.io/actions/actions-runner + CI deps baked in
|
||||
# (apt cairo/pango/jpeg/gif/rsvg stack, fd/ripgrep/imagemagick, bun, rust
|
||||
# nightly + clippy/rustfmt + arm64/msvc targets). Built + imported locally;
|
||||
# see /root/omp-kata-runner-image/. IfNotPresent uses the local image.
|
||||
image: omp-kata-runner:2026-06-15-002621
|
||||
imagePullPolicy: IfNotPresent
|
||||
command: ["/home/runner/run.sh"]
|
||||
# Shared sccache backend (in-cluster RustFS S3). Exposes SCCACHE_BUCKET/
|
||||
# ENDPOINT/REGION/USE_SSL + AWS creds to every job; CI flips RUSTC_WRAPPER
|
||||
# on for rust builds only. GitHub-hosted runners lack this env and keep the
|
||||
# GHA cache backend. See /root/sccache-rustfs/.
|
||||
envFrom:
|
||||
- secretRef:
|
||||
name: sccache-s3
|
||||
resources:
|
||||
requests:
|
||||
cpu: "2"
|
||||
memory: "4Gi"
|
||||
limits:
|
||||
cpu: "8"
|
||||
memory: "12Gi"
|
||||
```
|
||||
|
||||
Field by field:
|
||||
|
||||
- **`githubConfigUrl`** - the repo (or org) the scale set serves. Jobs reach it
|
||||
with `runs-on: omp-kata`.
|
||||
- **`githubConfigSecret: arc-github`** - the auth secret from [step 1](#1-github-app-and-the-arc-github-secret).
|
||||
- **`runnerScaleSetName: omp-kata`** - the runner label. This is the string that
|
||||
goes in a workflow's `runs-on:`.
|
||||
- **`minRunners: 0` / `maxRunners: 10`** - **scale-to-zero**. With no queued jobs
|
||||
there are zero runner pods (and zero microVMs) consuming the node; the listener
|
||||
scales up to ten concurrent runners on demand. (The older ops notes capped this
|
||||
at 3; the live value is 10.)
|
||||
- **`containerMode.type: ""`** - **none**. The default chart offers `dind`
|
||||
(Docker-in-Docker sidecar) or `kubernetes` mode for job-container isolation;
|
||||
both are unnecessary here because the *whole runner pod* is already isolated in
|
||||
a microVM. The job runs directly in the runner container - no privileged dind
|
||||
sidecar, no extra attack surface.
|
||||
- **`template.spec.runtimeClassName: kata-qemu`** - the critical line. It binds
|
||||
the pod to the Kata QEMU runtime ([02-kata-runtime.md](02-kata-runtime.md)), so
|
||||
every runner boots its own KVM microVM with a guest kernel distinct from the host.
|
||||
- **`image` / `imagePullPolicy: IfNotPresent`** - the locally built, dependency-baked
|
||||
runner image ([03-runner-image.md](03-runner-image.md)). `IfNotPresent` uses the
|
||||
copy already imported into cluster containerd; there is no registry. Bump the tag
|
||||
here when you rebuild the image (see [Operate](#7-operate)).
|
||||
- **`command: ["/home/runner/run.sh"]`** - the stock actions-runner entrypoint;
|
||||
overridden explicitly because the custom image keeps the upstream layout.
|
||||
- **`envFrom.secretRef.name: sccache-s3`** - injects the shared-cache S3
|
||||
configuration into every job's environment ([step 5](#5-shared-cache-rustfs-s3)).
|
||||
- **`resources`** - requests `2` CPU / `4Gi`, limits `8` CPU / `12Gi`. Kata reads
|
||||
these and sizes the guest accordingly: the VM now boots at the same
|
||||
guaranteed floor (`default_vcpus: 2`, `default_memory: 4096`) and only
|
||||
hotplugs beyond that toward the limits, with `default_maxvcpus: 0` allowing up
|
||||
to all host CPUs. Effectively the **requests are the boot-time VM size** and
|
||||
the **limits are the hotplug ceiling**. See [02-kata-runtime.md](02-kata-runtime.md)
|
||||
for the runtime knobs and [`infra/tune-kata-runtime.sh`](../tune-kata-runtime.sh)
|
||||
for the SSH-driven patch helper.
|
||||
|
||||
---
|
||||
|
||||
## 4. Job lifecycle and the no-permission ServiceAccount
|
||||
|
||||
One job runs in one fresh microVM that is destroyed afterward:
|
||||
|
||||
1. The **listener** (ns `arc-systems`) long-polls the GitHub Actions service for
|
||||
jobs whose `runs-on` matches `omp-kata`.
|
||||
2. When jobs are assigned, the controller reconciles the `AutoscalingRunnerSet`
|
||||
and creates an **`EphemeralRunnerSet`** sized to the demand (bounded by
|
||||
`minRunners`/`maxRunners`).
|
||||
3. Each replica becomes an **ephemeral runner pod** registered **just-in-time
|
||||
(JIT)** with GitHub - a per-runner registration secret is minted, not a
|
||||
long-lived token.
|
||||
4. Because the pod's `runtimeClassName` is `kata-qemu`, it **boots a microVM**,
|
||||
pulls the one assigned job, runs it, and exits.
|
||||
5. ARC **deletes the pod** (and its microVM); a clean VM is created for the next
|
||||
job. There is no VM templating - state never leaks between jobs.
|
||||
|
||||
Observe the chain live:
|
||||
|
||||
```bash
|
||||
kubectl -n arc-runners get autoscalingrunnerset omp-kata
|
||||
kubectl -n arc-runners get ephemeralrunnerset
|
||||
kubectl -n arc-runners get pods -o wide # one pod per in-flight job; empty when idle
|
||||
```
|
||||
|
||||
**No-permission ServiceAccount.** The scale-set chart runs every runner pod under
|
||||
a ServiceAccount with no RBAC bindings:
|
||||
|
||||
```bash
|
||||
kubectl -n arc-runners get sa
|
||||
# default
|
||||
# omp-kata-gha-rs-no-permission
|
||||
```
|
||||
|
||||
Job code therefore has no Kubernetes API rights - it cannot read secrets, list
|
||||
pods, or touch the cluster, even though it executes inside the cluster. Combined
|
||||
with microVM isolation and the egress policy ([step 6](#6-runner-egress-lockdown)),
|
||||
a compromised job is boxed into a throwaway VM with no cluster reach.
|
||||
|
||||
---
|
||||
|
||||
## 5. Shared cache (RustFS S3)
|
||||
|
||||
GitHub's hosted cache backend is only reachable over the node's NAT egress, so on
|
||||
a busy matrix (many concurrent jobs) it becomes the bottleneck. Instead an
|
||||
**S3-compatible object store, RustFS, runs inside the cluster** and serves the
|
||||
cache at LAN speed over `rustfs.sccache.svc.cluster.local:9000`.
|
||||
|
||||
### 5a. Deploy RustFS
|
||||
|
||||
The store lives in its own `sccache` namespace: a `local-path` PVC for durability,
|
||||
a single-replica Deployment, and a ClusterIP Service. This is `rustfs.yaml`
|
||||
verbatim (no secrets inline - credentials come from a separate secret):
|
||||
|
||||
```yaml
|
||||
apiVersion: v1
|
||||
kind: Namespace
|
||||
metadata:
|
||||
name: sccache
|
||||
labels:
|
||||
kubernetes.io/metadata.name: sccache
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: PersistentVolumeClaim
|
||||
metadata:
|
||||
name: rustfs-data
|
||||
namespace: sccache
|
||||
spec:
|
||||
accessModes: [ReadWriteOnce]
|
||||
storageClassName: local-path
|
||||
resources:
|
||||
requests:
|
||||
storage: 100Gi
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: rustfs
|
||||
namespace: sccache
|
||||
labels: { app: rustfs }
|
||||
spec:
|
||||
replicas: 1
|
||||
strategy: { type: Recreate }
|
||||
selector:
|
||||
matchLabels: { app: rustfs }
|
||||
template:
|
||||
metadata:
|
||||
labels: { app: rustfs }
|
||||
spec:
|
||||
containers:
|
||||
- name: rustfs
|
||||
image: rustfs/rustfs:latest
|
||||
imagePullPolicy: IfNotPresent
|
||||
env:
|
||||
- name: RUSTFS_ACCESS_KEY
|
||||
valueFrom: { secretKeyRef: { name: rustfs-creds, key: RUSTFS_ACCESS_KEY } }
|
||||
- name: RUSTFS_SECRET_KEY
|
||||
valueFrom: { secretKeyRef: { name: rustfs-creds, key: RUSTFS_SECRET_KEY } }
|
||||
- name: RUSTFS_VOLUMES
|
||||
value: "/data"
|
||||
- name: RUSTFS_ADDRESS
|
||||
value: ":9000"
|
||||
- name: RUSTFS_CONSOLE_ENABLE
|
||||
value: "false"
|
||||
- name: RUSTFS_OBS_LOG_DIRECTORY
|
||||
value: "/logs"
|
||||
ports:
|
||||
- { name: s3, containerPort: 9000 }
|
||||
volumeMounts:
|
||||
- { name: data, mountPath: /data }
|
||||
- { name: logs, mountPath: /logs }
|
||||
readinessProbe:
|
||||
tcpSocket: { port: 9000 }
|
||||
initialDelaySeconds: 5
|
||||
periodSeconds: 5
|
||||
livenessProbe:
|
||||
tcpSocket: { port: 9000 }
|
||||
initialDelaySeconds: 15
|
||||
periodSeconds: 20
|
||||
resources:
|
||||
requests: { cpu: "200m", memory: "256Mi" }
|
||||
limits: { cpu: "2", memory: "2Gi" }
|
||||
volumes:
|
||||
- name: data
|
||||
persistentVolumeClaim: { claimName: rustfs-data }
|
||||
- name: logs
|
||||
emptyDir: {}
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: rustfs
|
||||
namespace: sccache
|
||||
spec:
|
||||
selector: { app: rustfs }
|
||||
ports:
|
||||
- { name: s3, port: 9000, targetPort: 9000, protocol: TCP }
|
||||
```
|
||||
|
||||
Notes:
|
||||
|
||||
- **`strategy: Recreate`** with a single replica and an RWO `local-path` PVC: the
|
||||
data is node-local and only one pod ever mounts it.
|
||||
- **`RUSTFS_CONSOLE_ENABLE: "false"`** - only the S3 API on `:9000` is exposed;
|
||||
no admin console.
|
||||
- The `kubernetes.io/metadata.name: sccache` namespace label is what the egress
|
||||
NetworkPolicy's `namespaceSelector` matches ([step 6](#6-runner-egress-lockdown)).
|
||||
|
||||
The RustFS pod credentials come from a two-key secret in the `sccache` namespace
|
||||
(values are the object-store root credentials - use placeholders):
|
||||
|
||||
```bash
|
||||
kubectl -n sccache create secret generic rustfs-creds \
|
||||
--from-literal=RUSTFS_ACCESS_KEY=<S3_ACCESS_KEY> \
|
||||
--from-literal=RUSTFS_SECRET_KEY=<S3_SECRET_KEY>
|
||||
```
|
||||
|
||||
Apply and verify:
|
||||
|
||||
```bash
|
||||
kubectl apply -f rustfs.yaml
|
||||
kubectl -n sccache get deploy,svc,pvc
|
||||
# deployment.apps/rustfs 1/1
|
||||
# service/rustfs ClusterIP 10.43.x.x 9000/TCP
|
||||
# persistentvolumeclaim/rustfs-data Bound 100Gi local-path
|
||||
```
|
||||
|
||||
Create the `sccache` bucket once (any S3 client - e.g. the `aws` CLI or `mc`
|
||||
pointed at the endpoint with the root creds): `mb s3://sccache`.
|
||||
|
||||
### 5b. The `sccache-s3` secret (injected into every runner)
|
||||
|
||||
Every runner pod gets the cache configuration via `envFrom` ([step 3](#3-scale-set-values-arc-omp-valuesyaml)).
|
||||
The secret lives in `arc-runners` (the runners' namespace) and has six keys:
|
||||
|
||||
```bash
|
||||
kubectl -n arc-runners get secret sccache-s3 \
|
||||
-o go-template='{{range $k,$v := .data}}{{$k}}{{"\n"}}{{end}}'
|
||||
# AWS_ACCESS_KEY_ID
|
||||
# AWS_SECRET_ACCESS_KEY
|
||||
# SCCACHE_BUCKET
|
||||
# SCCACHE_ENDPOINT
|
||||
# SCCACHE_REGION
|
||||
# SCCACHE_S3_USE_SSL
|
||||
```
|
||||
|
||||
Recreate it (the two credential values must equal the `rustfs-creds` above; the
|
||||
rest are non-sensitive cluster-local config):
|
||||
|
||||
```bash
|
||||
kubectl -n arc-runners create secret generic sccache-s3 \
|
||||
--from-literal=AWS_ACCESS_KEY_ID=<S3_ACCESS_KEY> \
|
||||
--from-literal=AWS_SECRET_ACCESS_KEY=<S3_SECRET_KEY> \
|
||||
--from-literal=SCCACHE_BUCKET=sccache \
|
||||
--from-literal=SCCACHE_ENDPOINT=rustfs.sccache.svc.cluster.local:9000 \
|
||||
--from-literal=SCCACHE_REGION=us-east-1 \
|
||||
--from-literal=SCCACHE_S3_USE_SSL=false
|
||||
```
|
||||
|
||||
- `AWS_ACCESS_KEY_ID` / `AWS_SECRET_ACCESS_KEY` - the **only sensitive entries**;
|
||||
RustFS's root credentials (S3 SigV4 auth).
|
||||
- `SCCACHE_BUCKET: sccache` - bucket name.
|
||||
- `SCCACHE_ENDPOINT` - the in-cluster Service DNS + port.
|
||||
- `SCCACHE_REGION: us-east-1` - arbitrary region label SigV4 requires.
|
||||
- `SCCACHE_S3_USE_SSL: false` - the endpoint is plain HTTP on the cluster network.
|
||||
|
||||
### 5c. The two consumers
|
||||
|
||||
The presence of `$SCCACHE_BUCKET` in the environment is the repo's single signal
|
||||
for "am I on the self-hosted infra?". Both consumers branch on it and fall back
|
||||
to GitHub-hosted cache backends off-infra (GitHub-hosted macOS/arm runners never
|
||||
get the secret and so cannot reach the private RustFS).
|
||||
|
||||
**(a) sccache for Rust** - [`.github/actions/build-native`](../../.github/actions/build-native/action.yml).
|
||||
It installs `sccache`, then sets `RUSTC_WRAPPER=sccache` and `CARGO_INCREMENTAL=0`
|
||||
(sccache silently no-ops with incremental enabled). The backend is conditional:
|
||||
|
||||
- `$SCCACHE_BUCKET` set - sccache reads `SCCACHE_BUCKET/ENDPOINT/REGION` and the
|
||||
AWS creds straight from the inherited pod env and uses the **shared S3 (RustFS)**.
|
||||
- otherwise - it exports `SCCACHE_GHA_ENABLED=true` and uses the **GitHub Actions
|
||||
cache**.
|
||||
|
||||
`Swatinem/rust-cache` still caches `target/` on top; sccache fills the gaps when
|
||||
`target/` is cold.
|
||||
|
||||
**(b) bun dependency cache** - [`.github/actions/bun-install`](../../.github/actions/bun-install/action.yml),
|
||||
a composite action wrapping `bun install --frozen-lockfile`. A "Detect cache
|
||||
backend" step checks `$SCCACHE_BUCKET` + `$AWS_ACCESS_KEY_ID`:
|
||||
|
||||
- on-infra - it runs `rustfs-cache.sh restore` before install and `... save` after;
|
||||
- off-infra - it uses stock `actions/cache@v4` for the bun store.
|
||||
|
||||
`rustfs-cache.sh` talks to RustFS directly with `curl --aws-sigv4` (S3 SigV4), no
|
||||
SDK. It keys two objects per lockfile under the `bun-cache/` prefix of the
|
||||
`sccache` bucket, derived from `sha256(bun.lock)`:
|
||||
|
||||
- `store-<os>-<lockhash>` - the bun global package store (`~/.bun/install/cache`),
|
||||
plus a rolling `store-<os>-latest` alias so a changed lockfile still warm-starts
|
||||
from the previous store and `bun install` fetches only the delta.
|
||||
- `nm-<os>-<lockhash>` - the installed `node_modules` trees (repo root, every
|
||||
`packages/*`, and `python/robomp/web`).
|
||||
|
||||
On **restore**, a `node_modules` hit short-circuits everything - the subsequent
|
||||
`bun install --frozen-lockfile` is a no-op, so the store is neither fetched nor
|
||||
saved. Archives are multi-threaded **zstd** (`.tzst`, baked into the runner image)
|
||||
with a **gzip** (`.tgz`) fallback so the action still works on an older image; the
|
||||
suffix records the codec and restore only inflates what the host can decompress.
|
||||
On **save**, it writes the store/`node_modules` objects only when this exact
|
||||
lockfile has none yet (`s3_exists` check), avoiding redundant uploads.
|
||||
|
||||
### 5d. Retention and PVC pressure
|
||||
|
||||
The Bun cache intentionally keeps exact lockfile objects once written:
|
||||
|
||||
- `store-<os>-<lockhash>` and `nm-<os>-<lockhash>` are immutable warm caches;
|
||||
- only `store-<os>-latest` is overwritten.
|
||||
|
||||
That means `bun.lock` churn will accumulate old exact objects on the `rustfs-data`
|
||||
PVC. The repo ships [`infra/rustfs-cache-maintenance.sh`](../rustfs-cache-maintenance.sh)
|
||||
to keep that under control from an ops checkout:
|
||||
|
||||
```bash
|
||||
CI_HOST=<CI_HOST> ./infra/rustfs-cache-maintenance.sh report
|
||||
CI_HOST=<CI_HOST> ./infra/rustfs-cache-maintenance.sh prune
|
||||
```
|
||||
|
||||
Its policy is deliberately conservative:
|
||||
|
||||
- always keep every `store-<os>-latest` alias;
|
||||
- always keep the newest exact `store-*` object that matches each `latest` alias
|
||||
by ETag;
|
||||
- always keep the newest `KEEP_EXACT_PER_OS` exact objects per OS (`3` by default)
|
||||
for both `store-*` and `nm-*`;
|
||||
- never delete exact objects newer than `MAX_AGE_DAYS` (`30` by default);
|
||||
- once the PVC reaches `PRUNE_TRIGGER_PERCENT` (`80` by default), prune oldest
|
||||
remaining exact objects toward `TARGET_PERCENT` (`70` by default), even if they
|
||||
are newer than the age threshold.
|
||||
|
||||
The same script is the PVC-pressure alert: after `report` or `prune` it reads
|
||||
`df -P /data` from the live RustFS pod and exits `1` at `WARN_PERCENT` (`80`)
|
||||
and `2` at `CRITICAL_PERCENT` (`90`). Wire that into cron / systemd / your
|
||||
monitoring runner; a failing exit is the signal that the PVC is too full.
|
||||
|
||||
This caching is why RustFS sits inside the egress allow-list on `tcp/9000`
|
||||
([step 6](#6-runner-egress-lockdown)).
|
||||
|
||||
---
|
||||
|
||||
## 6. Runner egress lockdown
|
||||
|
||||
Runner pods reach the public internet (GitHub, package registries, crates.io,
|
||||
npm) but must **not** reach the host's own services, the LAN, the tailnet, or
|
||||
arbitrary cluster workloads. A single NetworkPolicy in `arc-runners` enforces
|
||||
this. Because the pod template sets no special labels, the policy uses
|
||||
`podSelector: {}` to cover **every** pod in the namespace.
|
||||
|
||||
> k3s ships a built-in NetworkPolicy controller (kube-router based) that enforces
|
||||
> policies even though the CNI is Flannel - so this policy actually takes effect.
|
||||
> Do not start k3s with `--disable-network-policy` ([01-host-and-cluster.md](01-host-and-cluster.md)),
|
||||
> or the lockdown silently becomes a no-op.
|
||||
|
||||
Live spec (captured with `kubectl get networkpolicy -n arc-runners runner-egress-lockdown -o yaml`;
|
||||
server-managed metadata omitted, host public IP redacted):
|
||||
|
||||
```yaml
|
||||
apiVersion: networking.k8s.io/v1
|
||||
kind: NetworkPolicy
|
||||
metadata:
|
||||
name: runner-egress-lockdown
|
||||
namespace: arc-runners
|
||||
spec:
|
||||
podSelector: {}
|
||||
policyTypes:
|
||||
- Ingress
|
||||
- Egress
|
||||
egress:
|
||||
# 1. Cluster DNS only (CoreDNS + kube-system).
|
||||
- to:
|
||||
- ipBlock:
|
||||
cidr: 10.43.0.10/32
|
||||
- namespaceSelector:
|
||||
matchLabels:
|
||||
kubernetes.io/metadata.name: kube-system
|
||||
ports:
|
||||
- port: 53
|
||||
protocol: UDP
|
||||
- port: 53
|
||||
protocol: TCP
|
||||
# 2. Public internet, MINUS all private/infra ranges and the host's own public IP.
|
||||
- to:
|
||||
- ipBlock:
|
||||
cidr: 0.0.0.0/0
|
||||
except:
|
||||
- 10.0.0.0/8
|
||||
- 172.16.0.0/12
|
||||
- 192.168.0.0/16
|
||||
- 169.254.0.0/16
|
||||
- 100.64.0.0/10
|
||||
- <PUBLIC_IP>/32
|
||||
# 3. RustFS shared cache (S3) over the cluster network.
|
||||
- to:
|
||||
- ipBlock:
|
||||
cidr: 10.43.0.0/16
|
||||
- namespaceSelector:
|
||||
matchLabels:
|
||||
kubernetes.io/metadata.name: sccache
|
||||
ports:
|
||||
- port: 9000
|
||||
protocol: TCP
|
||||
```
|
||||
|
||||
The allow-list, rule by rule:
|
||||
|
||||
- **Rule 1 - DNS.** UDP/TCP 53 to CoreDNS (`10.43.0.10/32`) and the `kube-system`
|
||||
namespace. Without this, name resolution breaks and rule 2 is useless.
|
||||
- **Rule 2 - public internet only.** `0.0.0.0/0` with an `except` list that
|
||||
carves out every range a job has no business reaching: RFC1918 private space
|
||||
(`10/8`, `172.16/12`, `192.168/16`), link-local (`169.254/16`), the CGNAT range
|
||||
used by the **tailnet** (`100.64.0.0/10`), and the **host's own public IP**
|
||||
(`<PUBLIC_IP>/32`). Note `10.0.0.0/8` covers the pod CIDR (`10.42.0.0/16`) and
|
||||
service CIDR (`10.43.0.0/16`), so this rule alone gives a job **zero** in-cluster
|
||||
reach - rules 1 and 3 punch the only two holes the job legitimately needs.
|
||||
- **Rule 3 - RustFS cache.** TCP 9000 to the service CIDR (`10.43.0.0/16`) and the
|
||||
`sccache` namespace - the shared cache from [step 5](#5-shared-cache-rustfs-s3).
|
||||
- **Ingress.** `policyTypes` lists `Ingress` but no ingress rule is defined, which
|
||||
is a **default-deny**: nothing can open a connection *into* a runner pod.
|
||||
|
||||
Egress that survives rule 2 leaves the node via the host's firewalld masquerade
|
||||
(SNAT to the public IP) over the default interface - see
|
||||
[01-host-and-cluster.md](01-host-and-cluster.md) for the host firewall side.
|
||||
|
||||
### Security model
|
||||
|
||||
- **Kernel isolation.** Each job runs in a Kata microVM with its own guest kernel
|
||||
(6.x), separate from the host kernel (7.0.x) - a kernel exploit hits a throwaway
|
||||
VM, not the host. See [02-kata-runtime.md](02-kata-runtime.md).
|
||||
- **No cluster rights.** Jobs run under `omp-kata-gha-rs-no-permission` with no
|
||||
RBAC ([step 4](#4-job-lifecycle-and-the-no-permission-serviceaccount)).
|
||||
- **Constrained network.** The policy above blocks the host, LAN, tailnet, and
|
||||
arbitrary cluster pods; only DNS, the public internet, and RustFS are reachable.
|
||||
- **Ephemeral.** One job per VM, destroyed afterward - no state, secret, or
|
||||
artifact survives into the next job.
|
||||
- **Public-repo recommendation.** For a public repo, require approval for fork
|
||||
PRs so untrusted code cannot auto-run on the infra: **repo - Settings - Actions
|
||||
- General - Fork pull request workflows from outside collaborators - Require
|
||||
approval for all outside collaborators**.
|
||||
|
||||
---
|
||||
|
||||
## 7. Operate
|
||||
|
||||
```bash
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
```
|
||||
|
||||
**Status / scale**
|
||||
|
||||
```bash
|
||||
kubectl -n arc-runners get autoscalingrunnerset omp-kata # min/max/current runners
|
||||
kubectl -n arc-runners get ephemeralrunnerset # desired vs current replicas
|
||||
kubectl -n arc-runners get pods -o wide # live runner VMs (empty when idle)
|
||||
```
|
||||
|
||||
**Logs**
|
||||
|
||||
```bash
|
||||
# Listener (job dispatch / scaling decisions)
|
||||
kubectl -n arc-systems logs -l app.kubernetes.io/component=runner-scale-set-listener -f
|
||||
# Controller (reconciliation)
|
||||
kubectl -n arc-systems logs deploy/arc-gha-rs-controller -f
|
||||
# A specific runner / its job
|
||||
kubectl -n arc-runners logs <runner-pod>
|
||||
```
|
||||
|
||||
**Verify the cache is being used.** A warm job logs `bun cache: ... HIT` and
|
||||
`sccache backend: shared S3 (sccache @ rustfs.sccache.svc.cluster.local:9000)` in
|
||||
its step output. To inspect objects directly, point any S3 client at the endpoint
|
||||
(with the RustFS root creds) and list `s3://sccache/bun-cache/`.
|
||||
|
||||
**Resize a job's VM** - edit the `resources` block in `arc-omp-values.yaml`
|
||||
([step 3](#3-scale-set-values-arc-omp-valuesyaml); requests = guaranteed VM size,
|
||||
limits = hotplug ceiling) and roll out:
|
||||
|
||||
```bash
|
||||
helm upgrade omp-kata \
|
||||
--namespace arc-runners --version 0.14.2 \
|
||||
-f arc-omp-values.yaml \
|
||||
oci://ghcr.io/actions/actions-runner-controller-charts/gha-runner-scale-set
|
||||
```
|
||||
|
||||
**Change scale-to-zero bounds** - edit `minRunners` / `maxRunners` in the same
|
||||
file and `helm upgrade` as above. (Keep `maxRunners` within the node's CPU/RAM
|
||||
budget: each runner can hotplug up to its `limits`.)
|
||||
|
||||
**Update the runner image** - bump `template.spec.containers[0].image` to the new
|
||||
tag, then `helm upgrade` as above; confirm with:
|
||||
|
||||
```bash
|
||||
kubectl -n arc-runners get autoscalingrunnerset omp-kata \
|
||||
-o jsonpath='{.spec.template.spec.containers[0].image}{"\n"}'
|
||||
```
|
||||
|
||||
See [03-runner-image.md](03-runner-image.md) for building and importing the image.
|
||||
|
||||
**Add another repo.** Because the GitHub App installation can cover multiple repos,
|
||||
reuse the same `arc-github` secret and install a second scale set with its own
|
||||
`githubConfigUrl`, `runnerScaleSetName` (the new `runs-on:` label), and release
|
||||
name:
|
||||
|
||||
```bash
|
||||
helm install <release> \
|
||||
--namespace arc-runners --version 0.14.2 \
|
||||
--set githubConfigUrl=https://github.com/<OWNER>/<OTHER_REPO> \
|
||||
--set githubConfigSecret=arc-github \
|
||||
--set runnerScaleSetName=<other-repo>-kata \
|
||||
-f arc-omp-values.yaml \
|
||||
oci://ghcr.io/actions/actions-runner-controller-charts/gha-runner-scale-set
|
||||
```
|
||||
|
||||
Jobs in the other repo then target `runs-on: <other-repo>-kata`. (On this host a
|
||||
convenience wrapper, `omp-add-repo-runner <OWNER>/<REPO> [label]`, performs exactly
|
||||
this install.)
|
||||
|
||||
**Uninstall** (leaves k3s/Kata in place):
|
||||
|
||||
```bash
|
||||
helm uninstall omp-kata -n arc-runners
|
||||
helm uninstall arc -n arc-systems
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
**Previous:** [03-runner-image.md](03-runner-image.md) - the preloaded runner image.
|
||||
**Overview:** [README.md](README.md) - architecture and the full doc set.
|
||||
@@ -0,0 +1,110 @@
|
||||
# Self-hosted Kata CI
|
||||
|
||||
This is a self-hosted GitHub Actions setup where **every CI job runs inside its own throwaway Kata Containers QEMU/KVM microVM**. A single bare-metal Linux host runs a one-node [k3s](https://k3s.io) cluster; [actions-runner-controller (ARC)](https://github.com/actions/actions-runner-controller) watches GitHub for queued jobs and, for each one, creates a just-in-time ephemeral runner pod that boots a fresh microVM (its own guest kernel, isolated from the host), runs exactly one job, and is then destroyed. Runners share an in-cluster **RustFS** (S3-compatible) object store for `sccache` and Bun dependency caching, and egress the public internet through host NAT under a restrictive NetworkPolicy. The result is hardware-isolated, scale-to-zero CI on hardware you control.
|
||||
|
||||
These docs are written as a **from-scratch setup guide**: read this overview first, then follow the numbered guides in order to reproduce the system on your own host.
|
||||
|
||||
## Architecture
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
GH["GitHub Actions<br/>repo <OWNER>/<REPO> + GitHub App"]
|
||||
|
||||
subgraph HOST["CI host (<CI_HOST>) — CentOS Stream 10, KVM/bare-metal"]
|
||||
NAT["firewalld masquerade<br/>egress NAT → public IP <PUBLIC_IP>"]
|
||||
subgraph K3S["single-node k3s v1.35.5 (own containerd v2)"]
|
||||
subgraph SYS["ns: arc-systems"]
|
||||
LIS["Runner scale-set listener<br/>long-polls GitHub"]
|
||||
CTRL["ARC controller 0.14.2<br/>scales EphemeralRunnerSet"]
|
||||
end
|
||||
subgraph RUN["ns: arc-runners"]
|
||||
NP["NetworkPolicy<br/>runner-egress-lockdown"]
|
||||
POD["Ephemeral runner pod (JIT)<br/>runtimeClassName: kata-qemu"]
|
||||
subgraph VM["Kata QEMU/KVM microVM — separate guest kernel"]
|
||||
JOB["actions/runner + one job's steps"]
|
||||
end
|
||||
end
|
||||
subgraph CACHE["ns: sccache"]
|
||||
RUSTFS["RustFS (S3)<br/>svc rustfs:9000 · PVC rustfs-data 100Gi"]
|
||||
end
|
||||
SEC["Secret sccache-s3<br/>S3 creds + endpoint"]
|
||||
end
|
||||
end
|
||||
|
||||
GH <-->|"long-poll / JIT registration"| LIS
|
||||
LIS --> CTRL
|
||||
CTRL -->|"creates 1 pod per job"| POD
|
||||
POD --> VM
|
||||
SEC -.->|"envFrom"| POD
|
||||
NP -.->|"filters egress"| POD
|
||||
JOB -->|"sccache + Bun cache (S3 SigV4)"| RUSTFS
|
||||
POD -->|"allowed egress"| NAT
|
||||
NAT -->|"checkout / API / internet"| GH
|
||||
```
|
||||
|
||||
Key properties baked into this design:
|
||||
|
||||
- **One job = one VM.** Runner pods are ephemeral and JIT-registered; there is no VM templating or pooling, so a job never inherits state from a previous job.
|
||||
- **Scale-to-zero.** `minRunners: 0` / `maxRunners: 10` — when no jobs are queued, zero runner pods (and zero microVMs) exist.
|
||||
- **Host-kernel isolation.** Jobs see the microVM's guest kernel, not the host kernel, so a kernel exploit in a job does not reach the host.
|
||||
- **No external registry.** The runner image is built on the host and imported straight into k3s' containerd.
|
||||
- **Shared, in-cluster cache.** `sccache` and the Bun dependency cache both target RustFS over the cluster network; nothing cache-related leaves the host.
|
||||
|
||||
## End-to-end job lifecycle
|
||||
|
||||
1. A workflow job targeting the self-hosted label (`runs-on:`) is **queued** on GitHub.
|
||||
2. The **scale-set listener** in `arc-systems` is long-polling the GitHub Actions service and receives the job-assignment message.
|
||||
3. The listener signals demand to the **ARC controller**, which scales the **EphemeralRunnerSet** up by one.
|
||||
4. The controller creates a single **JIT-registered ephemeral runner pod** in `arc-runners`, with `runtimeClassName: kata-qemu` and the `sccache-s3` secret injected via `envFrom`.
|
||||
5. containerd hands the pod to the Kata shim, which **boots a fresh QEMU/KVM microVM** (own guest kernel; the container rootfs is shared in over virtio-fs). No templating — every job gets a clean VM.
|
||||
6. The runner agent inside the microVM **registers just-in-time and picks up exactly one job**. Steps run isolated from the host, using RustFS over S3 for `sccache`/Bun caching and NAT egress for the public internet, all constrained by the `runner-egress-lockdown` NetworkPolicy.
|
||||
7. The job finishes; the ephemeral runner **deregisters and the pod (and its microVM) is destroyed** — never reused.
|
||||
8. When no jobs remain queued, the EphemeralRunnerSet **scales back to zero**, leaving no idle runners or VMs.
|
||||
|
||||
## Component map (bill of materials)
|
||||
|
||||
| Component | What it is | Version | Documented in |
|
||||
| --- | --- | --- | --- |
|
||||
| Host + k3s cluster | Bare-metal CentOS Stream 10 node running single-node k3s (own containerd v2, Flannel CNI; Traefik + servicelb disabled so host nginx keeps :80/:443); firewalld provides NAT egress | k3s `v1.35.5+k3s1` | [01-host-and-cluster.md](01-host-and-cluster.md) |
|
||||
| Kata Containers runtime | QEMU/KVM microVM runtime: containerd drop-in registering `kata-qemu` + the `kata-qemu` RuntimeClass | Kata `3.31.0` | [02-kata-runtime.md](02-kata-runtime.md) |
|
||||
| Preloaded runner image | Custom `actions/runner` image (build toolchain, Bun, Rust nightly + cross targets, native-build deps) built on the host and imported into k3s containerd — no registry | local dated tag | [03-runner-image.md](03-runner-image.md) |
|
||||
| ARC (runner scale set) | actions-runner-controller, `gha-runner-scale-set` flavor: controller in `arc-systems`, one scale set + listener, GitHub App auth | ARC `0.14.2` | [04-arc-and-caching.md](04-arc-and-caching.md) |
|
||||
| RustFS shared cache | In-cluster S3-compatible store (`svc rustfs:9000`, 100Gi PVC) backing `sccache` and the Bun cache, plus the `sccache-s3` secret and the egress NetworkPolicy | in-cluster service | [04-arc-and-caching.md](04-arc-and-caching.md) |
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Before starting, you need:
|
||||
|
||||
- **A Linux host with hardware virtualization.** Intel VT-x or AMD-V enabled, KVM available (`/dev/kvm` present and accessible). Bare metal is simplest; on a VM you need working nested virtualization. The reference host is 32 vCPU / 125 GiB RAM — size to roughly `maxRunners × per-job resources` plus cluster overhead.
|
||||
- **Root (or full sudo)** on that host: you will install k3s, Kata, kernel modules, firewall rules, and a container image.
|
||||
- **A public-ish egress path.** The host must reach GitHub; runner microVMs NAT out through the host's public IP. No inbound ports are required for the runners (the listener uses outbound long-poll).
|
||||
- **A GitHub repository** to attach runners to, and a **GitHub App** (recommended) or PAT installed on it with permissions to manage self-hosted runners. You will record the App ID, installation ID, and private key as a Kubernetes secret.
|
||||
- **CLI tooling on the host:** `kubectl` and `helm` (k3s bundles a kubectl), plus Docker/buildkit for building the runner image (see [03-runner-image.md](03-runner-image.md)).
|
||||
|
||||
## Redaction & placeholders
|
||||
|
||||
The configs in this doc set are copied verbatim from the live host and then redacted. Wherever you see one of these tokens, substitute your own value:
|
||||
|
||||
| Placeholder | Substitute with |
|
||||
| --- | --- |
|
||||
| `<CI_HOST>` | Your CI host's hostname / SSH target |
|
||||
| `<PUBLIC_IP>` | The host's public IPv4 address |
|
||||
| `<TAILNET_IP>` | Your Tailscale/tailnet admin IP(s) (the generic CGNAT range `100.64.0.0/10` is kept as-is) |
|
||||
| `<OWNER>/<REPO>` | Your GitHub repository owner and name |
|
||||
| `<GITHUB_APP_ID>` | Your GitHub App ID |
|
||||
| `<GITHUB_APP_INSTALLATION_ID>` | Your GitHub App installation ID |
|
||||
| `<GITHUB_APP_PRIVATE_KEY>` | Your GitHub App private key (PEM) |
|
||||
| `<S3_ACCESS_KEY>` | RustFS/S3 access key ID |
|
||||
| `<S3_SECRET_KEY>` | RustFS/S3 secret access key |
|
||||
| `<PLACEHOLDER>` | Any other password/key/token (named in context where it appears) |
|
||||
|
||||
Secret **values** never appear in these docs — only key names and placeholders. The following are intentionally **kept as-is** because they are not sensitive and are needed to follow along: the pod CIDR `10.42.0.0/16`, the service CIDR `10.43.0.0/16`, the CoreDNS service IP `10.43.0.10`, the bucket name `sccache`, in-cluster service DNS names and ports, and all version numbers.
|
||||
|
||||
## Recommended setup order
|
||||
|
||||
Work through the numbered guides in order — each builds on the previous:
|
||||
|
||||
1. **[01-host-and-cluster.md](01-host-and-cluster.md)** — Host prep (KVM, firewall/NAT) and the single-node k3s install, networking, and CNI.
|
||||
2. **[02-kata-runtime.md](02-kata-runtime.md)** — Install Kata Containers, wire it into k3s' containerd, and register the `kata-qemu` RuntimeClass.
|
||||
3. **[03-runner-image.md](03-runner-image.md)** — Build the preloaded runner image and import it into k3s containerd.
|
||||
4. **[04-arc-and-caching.md](04-arc-and-caching.md)** — Install ARC and the runner scale set, deploy the RustFS shared cache, wire up the `sccache-s3` secret, and apply the egress NetworkPolicy.
|
||||
Executable
+230
@@ -0,0 +1,230 @@
|
||||
#!/usr/bin/env bash
|
||||
# Build + roll the preloaded omp-kata runner image onto the self-hosted CI host,
|
||||
# driven over SSH from this repo. The Dockerfile next to this script is the
|
||||
# source of truth: it is copied to the host, built there, and the ARC runner
|
||||
# scale set is pointed at the new tag and rolled.
|
||||
#
|
||||
# By default the remote side prefers a direct containerd build path: bootstrap a
|
||||
# pinned BuildKit + nerdctl toolchain under the remote build dir, build straight
|
||||
# into the k3s containerd `k8s.io` namespace, and smoke-test from that image
|
||||
# store. That avoids the old `docker save | ctr images import` tarball hop and
|
||||
# cuts duplicate layer I/O on large reloads. If the direct path is unavailable
|
||||
# or you set BUILD_BACKEND=docker, it falls back to the legacy Docker-daemon
|
||||
# build + import path.
|
||||
#
|
||||
# The host is intentionally NOT hardcoded (this repo is public). Set CI_HOST to
|
||||
# your ssh target; the remaining knobs default to the reference deployment.
|
||||
#
|
||||
# Usage:
|
||||
# CI_HOST=my-ci-host ./infra/reload-runner.sh # tag: omp-kata-runner:YYYY-MM-DD-HHMMSS
|
||||
# CI_HOST=my-ci-host ./infra/reload-runner.sh 2026-06-20 # tag: omp-kata-runner:2026-06-20
|
||||
# CI_HOST=my-ci-host ./infra/reload-runner.sh my/repo:tag # explicit repo:tag
|
||||
#
|
||||
# Env knobs (defaults match the reference deployment):
|
||||
# CI_HOST ssh target of the CI host (required)
|
||||
# REMOTE_CTX remote build dir for the Dockerfile [/root/omp-kata-runner-image]
|
||||
# ARC_VALUES remote ARC scale-set helm values file [/root/arc-omp-values.yaml]
|
||||
# ARC_RELEASE helm release name of the runner scale set [omp-kata]
|
||||
# ARC_NAMESPACE namespace the runner scale set lives in [arc-runners]
|
||||
# ARC_CHART_VERSION gha-runner-scale-set chart version [0.14.2]
|
||||
# KUBECONFIG_REMOTE kubeconfig path on the host [/etc/rancher/k3s/k3s.yaml]
|
||||
# BUILD_BACKEND auto | containerd | docker [auto]
|
||||
# CONTAINERD_SOCKET_REMOTE remote containerd socket [/run/k3s/containerd/containerd.sock]
|
||||
# NERDCTL_VERSION nerdctl release to bootstrap on demand [2.1.6]
|
||||
# BUILDKIT_VERSION BuildKit release to bootstrap on demand [0.25.1]
|
||||
set -euo pipefail
|
||||
|
||||
: "${CI_HOST:?set CI_HOST to the ssh target of your CI host, e.g. CI_HOST=my-ci-host}"
|
||||
REMOTE_CTX="${REMOTE_CTX:-/root/omp-kata-runner-image}"
|
||||
ARC_VALUES="${ARC_VALUES:-/root/arc-omp-values.yaml}"
|
||||
ARC_RELEASE="${ARC_RELEASE:-omp-kata}"
|
||||
ARC_NAMESPACE="${ARC_NAMESPACE:-arc-runners}"
|
||||
ARC_CHART_VERSION="${ARC_CHART_VERSION:-0.14.2}"
|
||||
KUBECONFIG_REMOTE="${KUBECONFIG_REMOTE:-/etc/rancher/k3s/k3s.yaml}"
|
||||
BUILD_BACKEND="${BUILD_BACKEND:-auto}"
|
||||
CONTAINERD_SOCKET_REMOTE="${CONTAINERD_SOCKET_REMOTE:-/run/k3s/containerd/containerd.sock}"
|
||||
NERDCTL_VERSION="${NERDCTL_VERSION:-2.1.6}"
|
||||
BUILDKIT_VERSION="${BUILDKIT_VERSION:-0.25.1}"
|
||||
|
||||
arg="${1:-$(date +%Y-%m-%d-%H%M%S)}"
|
||||
case "$arg" in *:*) IMAGE="$arg";; *) IMAGE="omp-kata-runner:$arg";; esac
|
||||
|
||||
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
[ -f "$here/runner.Dockerfile" ] || { echo "no runner.Dockerfile next to $0" >&2; exit 1; }
|
||||
|
||||
echo "==> [0/5] copying Dockerfile to ${CI_HOST}:${REMOTE_CTX}"
|
||||
ssh "$CI_HOST" "mkdir -p '$REMOTE_CTX'"
|
||||
scp -q "$here/runner.Dockerfile" "${CI_HOST}:${REMOTE_CTX}/Dockerfile"
|
||||
|
||||
# All build/import/rollout steps run on the host. Config is passed as positional
|
||||
# args (no secrets, no spaces) so it survives the ssh command-string re-parse
|
||||
# regardless of the host's login shell.
|
||||
ssh "$CI_HOST" bash -s -- \
|
||||
"$IMAGE" "$REMOTE_CTX" "$ARC_VALUES" "$ARC_RELEASE" "$ARC_NAMESPACE" "$ARC_CHART_VERSION" \
|
||||
"$KUBECONFIG_REMOTE" "$BUILD_BACKEND" "$CONTAINERD_SOCKET_REMOTE" "$NERDCTL_VERSION" "$BUILDKIT_VERSION" <<'REMOTE'
|
||||
set -euo pipefail
|
||||
IMAGE="$1"; REMOTE_CTX="$2"; ARC_VALUES="$3"; ARC_RELEASE="$4"; ARC_NAMESPACE="$5"; ARC_CHART_VERSION="$6"
|
||||
export KUBECONFIG="$7"
|
||||
BUILD_BACKEND="$8"; CONTAINERD_SOCKET="$9"; NERDCTL_VERSION="${10}"; BUILDKIT_VERSION="${11}"
|
||||
cd "$REMOTE_CTX"
|
||||
|
||||
TOOLS_DIR="$REMOTE_CTX/.containerd-build-tools"
|
||||
BIN_DIR="$TOOLS_DIR/bin"
|
||||
RUN_DIR="$TOOLS_DIR/run"
|
||||
ROOT_DIR="$TOOLS_DIR/root"
|
||||
LOG_DIR="$TOOLS_DIR/log"
|
||||
NERDCTL_BIN="$BIN_DIR/nerdctl"
|
||||
BUILDKITD_BIN="$BIN_DIR/buildkitd"
|
||||
BUILDKITCTL_BIN="$BIN_DIR/buildctl"
|
||||
BUILDKIT_ADDR="unix://$RUN_DIR/buildkitd.sock"
|
||||
|
||||
BUILDKITD_PID=""
|
||||
cleanup_buildkitd() {
|
||||
if [ -n "${BUILDKITD_PID:-}" ]; then
|
||||
kill "$BUILDKITD_PID" >/dev/null 2>&1 || true
|
||||
fi
|
||||
}
|
||||
trap cleanup_buildkitd EXIT
|
||||
extract_bin() {
|
||||
local archive="$1" needle="$2" out="$3" tmp
|
||||
tmp="$(mktemp -d)"
|
||||
tar -C "$tmp" -xf "$archive"
|
||||
cp "$(find "$tmp" -type f -name "$needle" | head -1)" "$out"
|
||||
chmod +x "$out"
|
||||
rm -rf "$tmp"
|
||||
}
|
||||
|
||||
bootstrap_containerd_tools() {
|
||||
mkdir -p "$BIN_DIR" "$RUN_DIR" "$ROOT_DIR" "$LOG_DIR"
|
||||
if [ ! -S "$CONTAINERD_SOCKET" ]; then
|
||||
echo "containerd socket missing: $CONTAINERD_SOCKET" >&2
|
||||
return 1
|
||||
fi
|
||||
if [ ! -x "$NERDCTL_BIN" ]; then
|
||||
local archive="$TOOLS_DIR/nerdctl-${NERDCTL_VERSION}.tar.gz"
|
||||
echo "==> [1/5] bootstrapping nerdctl ${NERDCTL_VERSION}"
|
||||
curl -fsSL "https://github.com/containerd/nerdctl/releases/download/v${NERDCTL_VERSION}/nerdctl-${NERDCTL_VERSION}-linux-amd64.tar.gz" -o "$archive"
|
||||
extract_bin "$archive" nerdctl "$NERDCTL_BIN"
|
||||
fi
|
||||
if [ ! -x "$BUILDKITD_BIN" ] || [ ! -x "$BUILDKITCTL_BIN" ]; then
|
||||
local archive="$TOOLS_DIR/buildkit-v${BUILDKIT_VERSION}.tar.gz"
|
||||
echo "==> [1/5] bootstrapping BuildKit ${BUILDKIT_VERSION}"
|
||||
curl -fsSL "https://github.com/moby/buildkit/releases/download/v${BUILDKIT_VERSION}/buildkit-v${BUILDKIT_VERSION}.linux-amd64.tar.gz" -o "$archive"
|
||||
extract_bin "$archive" buildkitd "$BUILDKITD_BIN"
|
||||
extract_bin "$archive" buildctl "$BUILDKITCTL_BIN"
|
||||
fi
|
||||
}
|
||||
|
||||
start_buildkitd() {
|
||||
rm -f "$RUN_DIR/buildkitd.sock"
|
||||
"$BUILDKITD_BIN" \
|
||||
--addr "$BUILDKIT_ADDR" \
|
||||
--root "$ROOT_DIR" \
|
||||
--containerd-worker=true \
|
||||
--containerd-worker-namespace k8s.io \
|
||||
--containerd-worker-addr "$CONTAINERD_SOCKET" \
|
||||
--oci-worker=false >"$LOG_DIR/buildkitd.log" 2>&1 &
|
||||
BUILDKITD_PID="$!"
|
||||
for _ in $(seq 1 120); do
|
||||
[ -S "$RUN_DIR/buildkitd.sock" ] && return 0
|
||||
sleep 0.25
|
||||
done
|
||||
echo "buildkitd did not create $RUN_DIR/buildkitd.sock" >&2
|
||||
sed -n '1,120p' "$LOG_DIR/buildkitd.log" >&2 || true
|
||||
return 1
|
||||
}
|
||||
|
||||
verify_baked_tools() {
|
||||
local runner="$1"
|
||||
"$runner" --namespace k8s.io run --rm --entrypoint bash "$IMAGE" -lc '
|
||||
set -e
|
||||
for b in gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do
|
||||
command -v "$b" >/dev/null || { echo "MISSING: $b"; exit 1; }
|
||||
done
|
||||
echo "tools OK | bun $(bun --version) | rust $(rustc --version) | sccache $(set -- $(sccache --version); echo "$2") | zig $(zig version) | gh $(set -- $(gh --version | head -1); echo "$3")"
|
||||
'
|
||||
}
|
||||
|
||||
build_with_containerd() {
|
||||
bootstrap_containerd_tools
|
||||
start_buildkitd
|
||||
|
||||
echo "==> [2/5] building $IMAGE directly into k3s containerd (k8s.io namespace)"
|
||||
"$BUILDKITCTL_BIN" --addr "$BUILDKIT_ADDR" build \
|
||||
--progress=plain \
|
||||
--frontend dockerfile.v0 \
|
||||
--local context=. \
|
||||
--local dockerfile=. \
|
||||
--opt filename=Dockerfile \
|
||||
--output "type=image,name=$IMAGE,store=true"
|
||||
k3s ctr -n k8s.io images tag "$IMAGE" omp-kata-runner:preloaded >/dev/null 2>&1 || true
|
||||
|
||||
echo "==> [3/5] verifying baked tools from k3s containerd"
|
||||
verify_baked_tools "$NERDCTL_BIN"
|
||||
}
|
||||
|
||||
build_with_docker() {
|
||||
echo "==> [1/5] building $IMAGE with docker"
|
||||
DOCKER_BUILDKIT=1 docker build -t "$IMAGE" -t omp-kata-runner:preloaded .
|
||||
|
||||
echo "==> [2/5] verifying baked tools"
|
||||
docker run --rm --entrypoint bash "$IMAGE" -lc '
|
||||
set -e
|
||||
for b in gh fd rg magick bun cargo rustc pkg-config zstd clang lld sccache zig cargo-nextest cargo-zigbuild cargo-xwin; do
|
||||
command -v "$b" >/dev/null || { echo "MISSING: $b"; exit 1; }
|
||||
done
|
||||
echo "tools OK | bun $(bun --version) | rust $(rustc --version) | sccache $(set -- $(sccache --version); echo "$2") | zig $(zig version) | gh $(set -- $(gh --version | head -1); echo "$3")"
|
||||
'
|
||||
|
||||
echo "==> [3/5] importing into k3s containerd (k8s.io namespace)"
|
||||
docker save "$IMAGE" | k3s ctr -n k8s.io images import --platform linux/amd64 -
|
||||
}
|
||||
|
||||
selected_backend="$BUILD_BACKEND"
|
||||
case "$BUILD_BACKEND" in
|
||||
auto)
|
||||
if bootstrap_containerd_tools >/dev/null 2>&1; then
|
||||
selected_backend=containerd
|
||||
else
|
||||
selected_backend=docker
|
||||
fi
|
||||
;;
|
||||
containerd|docker)
|
||||
;;
|
||||
*)
|
||||
echo "BUILD_BACKEND must be auto, containerd, or docker (got: $BUILD_BACKEND)" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
echo "==> selected build backend: $selected_backend"
|
||||
case "$selected_backend" in
|
||||
containerd)
|
||||
if ! build_with_containerd; then
|
||||
if [ "$BUILD_BACKEND" = auto ]; then
|
||||
echo "==> containerd build path failed; falling back to docker" >&2
|
||||
build_with_docker
|
||||
else
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
;;
|
||||
docker)
|
||||
build_with_docker
|
||||
;;
|
||||
esac
|
||||
|
||||
echo "==> [4/5] pointing ARC runner scale set at $IMAGE"
|
||||
sed -i "s#image: omp-kata-runner:.*#image: $IMAGE#" "$ARC_VALUES"
|
||||
helm upgrade "$ARC_RELEASE" --namespace "$ARC_NAMESPACE" --version "$ARC_CHART_VERSION" \
|
||||
-f "$ARC_VALUES" \
|
||||
oci://ghcr.io/actions/actions-runner-controller-charts/gha-runner-scale-set >/dev/null
|
||||
|
||||
echo "==> [5/5] verifying rollout"
|
||||
live="$(kubectl get autoscalingrunnerset "$ARC_RELEASE" -n "$ARC_NAMESPACE" \
|
||||
-o jsonpath='{.spec.template.spec.containers[0].image}')"
|
||||
echo "ARC runner image is now: $live"
|
||||
[ "$live" = "$IMAGE" ] && echo "OK: reloaded $IMAGE" || { echo "MISMATCH: expected $IMAGE"; exit 1; }
|
||||
REMOTE
|
||||
|
||||
echo "OK: $IMAGE built on ${CI_HOST}, stored in k3s containerd, and rolled out to ARC."
|
||||
@@ -0,0 +1,76 @@
|
||||
# syntax=docker/dockerfile:1
|
||||
# Preloaded omp-kata runner image.
|
||||
#
|
||||
# Stock GitHub Actions runner (Ubuntu 24.04) with the dependencies CI installs
|
||||
# on every job baked in, so each ephemeral Kata microVM boots with them already
|
||||
# present instead of re-fetching them per job:
|
||||
# - APT system deps (canvas/cairo stack + fd/ripgrep/imagemagick) + fd/magick shims
|
||||
# - GitHub CLI (gh) — present on GitHub-hosted runners; the coding-agent github
|
||||
# tool and release workflows expect it
|
||||
# - C/build toolchain the native + canvas builds need
|
||||
# - bun (system-wide, on PATH)
|
||||
# - sccache + Zig + cargo-nextest/cargo-zigbuild/cargo-xwin for native builds
|
||||
# - rust nightly (pinned) + clippy/rustfmt/rust-analyzer + linux-arm64/windows-msvc targets
|
||||
#
|
||||
# Rebuild + reimport (see /root/omp-kata-runner.md) after bumping the ARGs below
|
||||
# or the apt set. Keep the apt set in sync with .github/actions/setup-system-deps.
|
||||
FROM ghcr.io/actions/actions-runner:latest
|
||||
|
||||
ARG RUST_NIGHTLY=nightly-2026-04-29
|
||||
ARG BUN_VERSION=1.3.14
|
||||
ARG SCCACHE_VERSION=0.15.0
|
||||
ARG ZIG_VERSION=0.16.0
|
||||
|
||||
USER root
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
# Mirrors the "Install system deps" block in .github/workflows/ci.yml plus the
|
||||
# native/cross toolchain (clang/lld/llvm), the baked cache/tooling binaries, and
|
||||
# the GitHub CLI. The gh apt repo is added first so `gh` installs in the same apt
|
||||
# transaction.
|
||||
RUN curl -fsSL https://cli.github.com/packages/githubcli-archive-keyring.gpg -o /usr/share/keyrings/githubcli-archive-keyring.gpg \
|
||||
&& chmod go+r /usr/share/keyrings/githubcli-archive-keyring.gpg \
|
||||
&& echo "deb [arch=$(dpkg --print-architecture) signed-by=/usr/share/keyrings/githubcli-archive-keyring.gpg] https://cli.github.com/packages stable main" > /etc/apt/sources.list.d/github-cli.list \
|
||||
&& apt-get update \
|
||||
&& apt-get install -y \
|
||||
build-essential pkg-config curl ca-certificates git unzip xz-utils zstd gh \
|
||||
clang lld llvm \
|
||||
libcairo2-dev libpango1.0-dev libjpeg-dev libgif-dev librsvg2-dev \
|
||||
fd-find ripgrep imagemagick \
|
||||
&& ln -sf "$(command -v fdfind)" /usr/local/bin/fd \
|
||||
&& ln -sf /usr/bin/convert /usr/local/bin/magick \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# bun, system-wide (BUN_INSTALL/bin == /usr/local/bin, already on PATH).
|
||||
ENV BUN_INSTALL=/usr/local
|
||||
RUN curl -fsSL https://bun.sh/install | bash -s "bun-v${BUN_VERSION}" \
|
||||
&& bun --version
|
||||
|
||||
# Pinned native-build helpers, system-wide.
|
||||
RUN curl -fsSL "https://github.com/mozilla/sccache/releases/download/v${SCCACHE_VERSION}/sccache-v${SCCACHE_VERSION}-x86_64-unknown-linux-musl.tar.gz" \
|
||||
| tar -xz -C /tmp \
|
||||
&& install -m755 "/tmp/sccache-v${SCCACHE_VERSION}-x86_64-unknown-linux-musl/sccache" /usr/local/bin/sccache \
|
||||
&& rm -rf "/tmp/sccache-v${SCCACHE_VERSION}-x86_64-unknown-linux-musl"
|
||||
RUN curl -fsSL "https://ziglang.org/download/${ZIG_VERSION}/zig-x86_64-linux-${ZIG_VERSION}.tar.xz" -o /tmp/zig.tar.xz \
|
||||
&& tar -xJf /tmp/zig.tar.xz -C /opt \
|
||||
&& ln -sf "/opt/zig-x86_64-linux-${ZIG_VERSION}/zig" /usr/local/bin/zig \
|
||||
&& rm -f /tmp/zig.tar.xz
|
||||
|
||||
# rust toolchain + cargo helpers for the runner user; rustup default == pinned
|
||||
# nightly so Rust setup becomes a no-op on the preloaded image.
|
||||
USER runner
|
||||
ENV RUSTUP_HOME=/home/runner/.rustup \
|
||||
CARGO_HOME=/home/runner/.cargo \
|
||||
PATH=/home/runner/.cargo/bin:/usr/local/bin:${PATH}
|
||||
RUN curl --proto '=https' --tlsv1.2 -fsSL https://sh.rustup.rs \
|
||||
| sh -s -- -y --default-toolchain "${RUST_NIGHTLY}" --profile minimal \
|
||||
&& rustup component add clippy rustfmt rust-analyzer \
|
||||
&& rustup target add aarch64-unknown-linux-gnu x86_64-pc-windows-msvc \
|
||||
&& cargo install --locked cargo-nextest cargo-zigbuild cargo-xwin \
|
||||
&& cargo --version \
|
||||
&& rustc --version \
|
||||
&& sccache --version \
|
||||
&& zig version \
|
||||
&& cargo-nextest --version \
|
||||
&& cargo-zigbuild --help >/dev/null \
|
||||
&& cargo-xwin --help >/dev/null
|
||||
Executable
+276
@@ -0,0 +1,276 @@
|
||||
#!/usr/bin/env bash
|
||||
# Report/prune Bun cache objects in the shared RustFS bucket and alert on PVC
|
||||
# pressure. The script runs on the CI host over SSH because the RustFS endpoint
|
||||
# and the S3 credentials live there (inside k8s secrets).
|
||||
#
|
||||
# Modes:
|
||||
# report list usage + object summary, exit 1/2 on warn/critical thresholds
|
||||
# prune delete stale exact-lockfile Bun cache objects, then report/alert
|
||||
#
|
||||
# Usage:
|
||||
# CI_HOST=my-ci-host ./infra/rustfs-cache-maintenance.sh report
|
||||
# CI_HOST=my-ci-host ./infra/rustfs-cache-maintenance.sh prune
|
||||
#
|
||||
# Env knobs:
|
||||
# CI_HOST ssh target of the CI host (required)
|
||||
# KUBECONFIG_REMOTE kubeconfig path on the host [/etc/rancher/k3s/k3s.yaml]
|
||||
# SECRET_NAMESPACE namespace of sccache-s3 secret [arc-runners]
|
||||
# SECRET_NAME secret with RustFS client creds [sccache-s3]
|
||||
# RUSTFS_NAMESPACE namespace of the RustFS deployment [sccache]
|
||||
# RUSTFS_DEPLOYMENT deployment name of RustFS [rustfs]
|
||||
# CACHE_PREFIX Bun cache prefix inside the bucket [bun-cache/]
|
||||
# MAX_AGE_DAYS keep exact-lock objects newer than this [30]
|
||||
# KEEP_EXACT_PER_OS always keep this many exact objects per OS [3]
|
||||
# PRUNE_TRIGGER_PERCENT if PVC >= this %, prune oldest extra caches [80]
|
||||
# TARGET_PERCENT when above trigger, prune toward this % [70]
|
||||
# WARN_PERCENT report warning exit code at/above this % [80]
|
||||
# CRITICAL_PERCENT report critical exit code at/above this % [90]
|
||||
# DRY_RUN true = print deletes but do not delete [false]
|
||||
set -euo pipefail
|
||||
|
||||
: "${CI_HOST:?set CI_HOST to the ssh target of your CI host, e.g. CI_HOST=my-ci-host}"
|
||||
mode="${1:-report}"
|
||||
case "$mode" in report|prune) ;; *) echo "usage: $0 report|prune" >&2; exit 2 ;; esac
|
||||
|
||||
KUBECONFIG_REMOTE="${KUBECONFIG_REMOTE:-/etc/rancher/k3s/k3s.yaml}"
|
||||
SECRET_NAMESPACE="${SECRET_NAMESPACE:-arc-runners}"
|
||||
SECRET_NAME="${SECRET_NAME:-sccache-s3}"
|
||||
RUSTFS_NAMESPACE="${RUSTFS_NAMESPACE:-sccache}"
|
||||
RUSTFS_DEPLOYMENT="${RUSTFS_DEPLOYMENT:-rustfs}"
|
||||
CACHE_PREFIX="${CACHE_PREFIX:-bun-cache/}"
|
||||
MAX_AGE_DAYS="${MAX_AGE_DAYS:-30}"
|
||||
KEEP_EXACT_PER_OS="${KEEP_EXACT_PER_OS:-3}"
|
||||
PRUNE_TRIGGER_PERCENT="${PRUNE_TRIGGER_PERCENT:-80}"
|
||||
TARGET_PERCENT="${TARGET_PERCENT:-70}"
|
||||
WARN_PERCENT="${WARN_PERCENT:-80}"
|
||||
CRITICAL_PERCENT="${CRITICAL_PERCENT:-90}"
|
||||
DRY_RUN="${DRY_RUN:-false}"
|
||||
|
||||
ssh "$CI_HOST" bash -s -- \
|
||||
"$mode" "$KUBECONFIG_REMOTE" "$SECRET_NAMESPACE" "$SECRET_NAME" "$RUSTFS_NAMESPACE" "$RUSTFS_DEPLOYMENT" \
|
||||
"$CACHE_PREFIX" "$MAX_AGE_DAYS" "$KEEP_EXACT_PER_OS" "$PRUNE_TRIGGER_PERCENT" "$TARGET_PERCENT" \
|
||||
"$WARN_PERCENT" "$CRITICAL_PERCENT" "$DRY_RUN" <<'REMOTE'
|
||||
set -euo pipefail
|
||||
MODE="$1"
|
||||
export KUBECONFIG="$2"
|
||||
SECRET_NAMESPACE="$3"
|
||||
SECRET_NAME="$4"
|
||||
RUSTFS_NAMESPACE="$5"
|
||||
RUSTFS_DEPLOYMENT="$6"
|
||||
CACHE_PREFIX="$7"
|
||||
MAX_AGE_DAYS="$8"
|
||||
KEEP_EXACT_PER_OS="$9"
|
||||
PRUNE_TRIGGER_PERCENT="${10}"
|
||||
TARGET_PERCENT="${11}"
|
||||
WARN_PERCENT="${12}"
|
||||
CRITICAL_PERCENT="${13}"
|
||||
DRY_RUN="${14}"
|
||||
|
||||
ACCESS_KEY="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.AWS_ACCESS_KEY_ID}' | base64 -d)"
|
||||
SECRET_KEY="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.AWS_SECRET_ACCESS_KEY}' | base64 -d)"
|
||||
BUCKET="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.SCCACHE_BUCKET}' | base64 -d)"
|
||||
ENDPOINT="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.SCCACHE_ENDPOINT}' | base64 -d)"
|
||||
REGION="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.SCCACHE_REGION}' | base64 -d)"
|
||||
USE_SSL="$(kubectl get secret "$SECRET_NAME" -n "$SECRET_NAMESPACE" -o jsonpath='{.data.SCCACHE_S3_USE_SSL}' | base64 -d)"
|
||||
if [ "$USE_SSL" = "true" ]; then SCHEME=https; else SCHEME=http; fi
|
||||
|
||||
# The secret intentionally stores the in-cluster Service DNS because runner pods
|
||||
# consume it. This host-side maintenance script runs outside cluster DNS, so talk
|
||||
# to the same Service by ClusterIP instead when the endpoint points at `.svc`.
|
||||
if [[ "$ENDPOINT" == *.svc.*:* || "$ENDPOINT" == *.svc:* ]]; then
|
||||
svc_ip="$(kubectl get svc "$RUSTFS_DEPLOYMENT" -n "$RUSTFS_NAMESPACE" -o jsonpath='{.spec.clusterIP}')"
|
||||
svc_port="${ENDPOINT##*:}"
|
||||
ENDPOINT="${svc_ip}:${svc_port}"
|
||||
fi
|
||||
|
||||
usage_line() {
|
||||
kubectl exec -n "$RUSTFS_NAMESPACE" deploy/"$RUSTFS_DEPLOYMENT" -- df -P /data | tail -1
|
||||
}
|
||||
|
||||
before="$(usage_line)"
|
||||
TOTAL_KIB="$(awk '{print $2}' <<<"$before")"
|
||||
USED_KIB="$(awk '{print $3}' <<<"$before")"
|
||||
AVAIL_KIB="$(awk '{print $4}' <<<"$before")"
|
||||
USED_PCT_RAW="$(awk '{print $5}' <<<"$before")"
|
||||
USED_PCT="${USED_PCT_RAW%%%}"
|
||||
|
||||
echo "==> RustFS PVC before"
|
||||
echo "$before"
|
||||
|
||||
summary="$(python3 - "$MODE" "$SCHEME" "$ENDPOINT" "$BUCKET" "$REGION" "$ACCESS_KEY" "$SECRET_KEY" "$CACHE_PREFIX" "$MAX_AGE_DAYS" "$KEEP_EXACT_PER_OS" "$PRUNE_TRIGGER_PERCENT" "$TARGET_PERCENT" "$DRY_RUN" "$TOTAL_KIB" "$USED_KIB" "$USED_PCT" <<'PY'
|
||||
from __future__ import annotations
|
||||
from collections import defaultdict
|
||||
from datetime import datetime, timedelta, timezone
|
||||
import json
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import urllib.parse
|
||||
import xml.etree.ElementTree as ET
|
||||
|
||||
mode, scheme, endpoint, bucket, region, access, secret, prefix, max_age_days, keep_per_os, prune_trigger, target_pct, dry_run, total_kib, used_kib, used_pct = sys.argv[1:]
|
||||
max_age_days = int(max_age_days)
|
||||
keep_per_os = int(keep_per_os)
|
||||
prune_trigger = int(prune_trigger)
|
||||
target_pct = int(target_pct)
|
||||
dry_run = dry_run.lower() == "true"
|
||||
total_kib = int(total_kib)
|
||||
used_kib = int(used_kib)
|
||||
used_pct = int(used_pct)
|
||||
|
||||
prefix = prefix.rstrip("/") + "/"
|
||||
base = f"{scheme}://{endpoint}/{bucket}"
|
||||
now = datetime.now(timezone.utc)
|
||||
cutoff = now - timedelta(days=max_age_days)
|
||||
|
||||
|
||||
def curl(method: str, url: str, extra: list[str] | None = None) -> str:
|
||||
cmd = [
|
||||
"curl", "-fsS",
|
||||
"--aws-sigv4", f"aws:amz:{region}:s3",
|
||||
"--user", f"{access}:{secret}",
|
||||
"-X", method,
|
||||
]
|
||||
if extra:
|
||||
cmd.extend(extra)
|
||||
cmd.append(url)
|
||||
return subprocess.check_output(cmd, text=True)
|
||||
|
||||
|
||||
def list_objects() -> list[dict]:
|
||||
objs: list[dict] = []
|
||||
token = None
|
||||
while True:
|
||||
q = {"list-type": "2", "prefix": prefix}
|
||||
if token:
|
||||
q["continuation-token"] = token
|
||||
url = f"{base}/?{urllib.parse.urlencode(q)}"
|
||||
root = ET.fromstring(curl("GET", url))
|
||||
for node in root.findall(".//{*}Contents"):
|
||||
objs.append({
|
||||
"key": node.findtext("{*}Key", default=""),
|
||||
"size": int(node.findtext("{*}Size", default="0")),
|
||||
"etag": node.findtext("{*}ETag", default="").strip('"'),
|
||||
"last_modified": datetime.fromisoformat(node.findtext("{*}LastModified", default="1970-01-01T00:00:00+00:00")),
|
||||
})
|
||||
token = root.findtext(".//{*}NextContinuationToken")
|
||||
if not token:
|
||||
return objs
|
||||
|
||||
|
||||
def human(n: int) -> str:
|
||||
units = ["B", "KiB", "MiB", "GiB", "TiB"]
|
||||
value = float(n)
|
||||
for unit in units:
|
||||
if value < 1024 or unit == units[-1]:
|
||||
return f"{value:.1f}{unit}"
|
||||
value /= 1024
|
||||
return f"{n}B"
|
||||
|
||||
objs = list_objects()
|
||||
exact_re = re.compile(rf"^{re.escape(prefix)}(store|nm)-([^-]+)-([0-9a-f]{{32}})\.(tzst|tgz)$")
|
||||
latest_re = re.compile(rf"^{re.escape(prefix)}store-([^-]+)-latest\.(tzst|tgz)$")
|
||||
|
||||
exacts: list[dict] = []
|
||||
latest_aliases: list[dict] = []
|
||||
other_prefix: list[dict] = []
|
||||
for obj in objs:
|
||||
if m := exact_re.match(obj["key"]):
|
||||
obj = obj | {"kind": m.group(1), "os": m.group(2), "lock": m.group(3), "codec": m.group(4)}
|
||||
exacts.append(obj)
|
||||
elif m := latest_re.match(obj["key"]):
|
||||
obj = obj | {"kind": "store", "os": m.group(1), "lock": "latest", "codec": m.group(2)}
|
||||
latest_aliases.append(obj)
|
||||
else:
|
||||
other_prefix.append(obj)
|
||||
|
||||
protected = {obj["key"] for obj in latest_aliases}
|
||||
by_group: dict[tuple[str, str], list[dict]] = defaultdict(list)
|
||||
by_store_etag: dict[tuple[str, str, str], list[dict]] = defaultdict(list)
|
||||
for obj in exacts:
|
||||
by_group[(obj["kind"], obj["os"])] .append(obj)
|
||||
if obj["kind"] == "store":
|
||||
by_store_etag[(obj["os"], obj["codec"], obj["etag"])] .append(obj)
|
||||
|
||||
for alias in latest_aliases:
|
||||
matches = by_store_etag.get((alias["os"], alias["codec"], alias["etag"]), [])
|
||||
if matches:
|
||||
newest = max(matches, key=lambda o: o["last_modified"])
|
||||
protected.add(newest["key"])
|
||||
|
||||
for group, items in by_group.items():
|
||||
items.sort(key=lambda o: o["last_modified"], reverse=True)
|
||||
for obj in items[:keep_per_os]:
|
||||
protected.add(obj["key"])
|
||||
for obj in items:
|
||||
if obj["last_modified"] >= cutoff:
|
||||
protected.add(obj["key"])
|
||||
|
||||
eligible = [obj for obj in exacts if obj["key"] not in protected]
|
||||
eligible.sort(key=lambda o: o["last_modified"])
|
||||
aged = [obj for obj in eligible if obj["last_modified"] < cutoff]
|
||||
|
||||
selected: list[dict] = list(aged)
|
||||
selected_keys = {obj["key"] for obj in selected}
|
||||
need_reclaim_kib = max(0, used_kib - (total_kib * target_pct // 100)) if used_pct >= prune_trigger else 0
|
||||
reclaimed_kib = sum(obj["size"] // 1024 for obj in selected)
|
||||
if need_reclaim_kib > reclaimed_kib:
|
||||
for obj in eligible:
|
||||
if obj["key"] in selected_keys:
|
||||
continue
|
||||
selected.append(obj)
|
||||
selected_keys.add(obj["key"])
|
||||
reclaimed_kib += obj["size"] // 1024
|
||||
if reclaimed_kib >= need_reclaim_kib:
|
||||
break
|
||||
|
||||
selected.sort(key=lambda o: (o["kind"], o["os"], o["last_modified"], o["key"]))
|
||||
delete_total = sum(obj["size"] for obj in selected)
|
||||
|
||||
print(f"bun-cache objects: exact={len(exacts)} latest={len(latest_aliases)} other-prefix={len(other_prefix)} total={human(sum(o['size'] for o in objs))}")
|
||||
print(f"policy: max_age_days={max_age_days} keep_exact_per_os={keep_per_os} prune_trigger={prune_trigger}% target={target_pct}%")
|
||||
print(f"eligible exact objects: {len(eligible)} | selected for deletion: {len(selected)} | reclaimable: {human(delete_total)}")
|
||||
for obj in selected[:20]:
|
||||
age_days = int((now - obj['last_modified']).total_seconds() // 86400)
|
||||
print(f" delete {obj['key']} age={age_days}d size={human(obj['size'])}")
|
||||
if len(selected) > 20:
|
||||
print(f" ... {len(selected) - 20} more")
|
||||
|
||||
if mode == "prune":
|
||||
for obj in selected:
|
||||
if dry_run:
|
||||
continue
|
||||
curl("DELETE", f"{base}/{obj['key']}")
|
||||
|
||||
print("JSON_SUMMARY=" + json.dumps({
|
||||
"exact": len(exacts),
|
||||
"latest": len(latest_aliases),
|
||||
"other": len(other_prefix),
|
||||
"eligible": len(eligible),
|
||||
"selected": len(selected),
|
||||
"delete_bytes": delete_total,
|
||||
"dry_run": dry_run,
|
||||
}))
|
||||
PY
|
||||
)"
|
||||
|
||||
echo "$summary"
|
||||
|
||||
after="$(usage_line)"
|
||||
AFTER_USED_PCT_RAW="$(awk '{print $5}' <<<"$after")"
|
||||
AFTER_USED_PCT="${AFTER_USED_PCT_RAW%%%}"
|
||||
|
||||
echo "==> RustFS PVC after"
|
||||
echo "$after"
|
||||
|
||||
if [ "$AFTER_USED_PCT" -ge "$CRITICAL_PERCENT" ]; then
|
||||
echo "CRITICAL: rustfs-data usage ${AFTER_USED_PCT}% >= ${CRITICAL_PERCENT}%" >&2
|
||||
exit 2
|
||||
fi
|
||||
if [ "$AFTER_USED_PCT" -ge "$WARN_PERCENT" ]; then
|
||||
echo "WARNING: rustfs-data usage ${AFTER_USED_PCT}% >= ${WARN_PERCENT}%" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "OK: rustfs-data usage ${AFTER_USED_PCT}%"
|
||||
REMOTE
|
||||
Executable
+81
@@ -0,0 +1,81 @@
|
||||
#!/usr/bin/env bash
|
||||
# Patch the live Kata QEMU config on the CI host to match the runner pod's
|
||||
# guaranteed boot shape and a larger virtiofsd worker pool, then smoke-test that
|
||||
# a new kata-qemu pod still boots. Driven over SSH from this repo so the desired
|
||||
# values stay version-controlled.
|
||||
#
|
||||
# Usage:
|
||||
# CI_HOST=my-ci-host ./infra/tune-kata-runtime.sh
|
||||
#
|
||||
# Env knobs:
|
||||
# CI_HOST ssh target of the CI host (required)
|
||||
# KATA_CONFIG_REMOTE remote Kata config file [/opt/kata/share/defaults/kata-containers/configuration-qemu.toml]
|
||||
# KUBECONFIG_REMOTE kubeconfig path on the host [/etc/rancher/k3s/k3s.yaml]
|
||||
# ARC_RELEASE runner scale set name [omp-kata]
|
||||
# ARC_NAMESPACE runner namespace [arc-runners]
|
||||
# BOOT_VCPUS Kata default_vcpus [2]
|
||||
# BOOT_MEMORY_MIB Kata default_memory (MiB) [4096]
|
||||
# VIRTIOFSD_THREAD_POOL virtiofsd --thread-pool-size [4]
|
||||
set -euo pipefail
|
||||
|
||||
: "${CI_HOST:?set CI_HOST to the ssh target of your CI host, e.g. CI_HOST=my-ci-host}"
|
||||
KATA_CONFIG_REMOTE="${KATA_CONFIG_REMOTE:-/opt/kata/share/defaults/kata-containers/configuration-qemu.toml}"
|
||||
KUBECONFIG_REMOTE="${KUBECONFIG_REMOTE:-/etc/rancher/k3s/k3s.yaml}"
|
||||
ARC_RELEASE="${ARC_RELEASE:-omp-kata}"
|
||||
ARC_NAMESPACE="${ARC_NAMESPACE:-arc-runners}"
|
||||
BOOT_VCPUS="${BOOT_VCPUS:-2}"
|
||||
BOOT_MEMORY_MIB="${BOOT_MEMORY_MIB:-4096}"
|
||||
VIRTIOFSD_THREAD_POOL="${VIRTIOFSD_THREAD_POOL:-4}"
|
||||
|
||||
ssh "$CI_HOST" bash -s -- \
|
||||
"$KATA_CONFIG_REMOTE" "$KUBECONFIG_REMOTE" "$ARC_RELEASE" "$ARC_NAMESPACE" \
|
||||
"$BOOT_VCPUS" "$BOOT_MEMORY_MIB" "$VIRTIOFSD_THREAD_POOL" <<'REMOTE'
|
||||
set -euo pipefail
|
||||
KATA_CONFIG="$1"
|
||||
export KUBECONFIG="$2"
|
||||
ARC_RELEASE="$3"
|
||||
ARC_NAMESPACE="$4"
|
||||
BOOT_VCPUS="$5"
|
||||
BOOT_MEMORY_MIB="$6"
|
||||
THREAD_POOL="$7"
|
||||
|
||||
backup="${KATA_CONFIG}.bak.$(date +%Y%m%d-%H%M%S)"
|
||||
cp "$KATA_CONFIG" "$backup"
|
||||
echo "==> backup: $backup"
|
||||
|
||||
python3 - "$KATA_CONFIG" "$BOOT_VCPUS" "$BOOT_MEMORY_MIB" "$THREAD_POOL" <<'PY'
|
||||
from pathlib import Path
|
||||
import re
|
||||
import sys
|
||||
path = Path(sys.argv[1])
|
||||
boot_vcpus = sys.argv[2]
|
||||
boot_mem = sys.argv[3]
|
||||
thread_pool = sys.argv[4]
|
||||
text = path.read_text()
|
||||
replacements = [
|
||||
(r'(^\s*default_vcpus\s*=\s*)\d+', rf'\g<1>{boot_vcpus}'),
|
||||
(r'(^\s*default_memory\s*=\s*)\d+', rf'\g<1>{boot_mem}'),
|
||||
(r'(^\s*virtio_fs_extra_args\s*=\s*)\[[^\]]*\]', rf'\g<1>["--thread-pool-size={thread_pool}", "--announce-submounts"]'),
|
||||
]
|
||||
for pattern, replacement in replacements:
|
||||
text, n = re.subn(pattern, replacement, text, count=1, flags=re.MULTILINE)
|
||||
if n != 1:
|
||||
raise SystemExit(f"failed to patch {pattern}")
|
||||
path.write_text(text)
|
||||
PY
|
||||
|
||||
echo "==> active Kata knobs"
|
||||
grep -nE 'default_vcpus|default_memory|virtio_fs_extra_args' "$KATA_CONFIG"
|
||||
|
||||
image="$(kubectl get autoscalingrunnerset "$ARC_RELEASE" -n "$ARC_NAMESPACE" -o jsonpath='{.spec.template.spec.containers[0].image}')"
|
||||
pod="kata-runtime-smoke-$(date +%H%M%S)"
|
||||
trap 'kubectl delete pod "$pod" -n "$ARC_NAMESPACE" --ignore-not-found >/dev/null 2>&1 || true' EXIT
|
||||
|
||||
echo "==> smoke boot via kata-qemu using $image"
|
||||
kubectl run "$pod" -n "$ARC_NAMESPACE" --restart=Never --image="$image" \
|
||||
--overrides='{"spec":{"runtimeClassName":"kata-qemu"}}' \
|
||||
--command -- bash -lc 'sleep 120' >/dev/null
|
||||
kubectl wait --for=condition=Ready "pod/$pod" -n "$ARC_NAMESPACE" --timeout=120s >/dev/null
|
||||
kubectl exec -n "$ARC_NAMESPACE" "$pod" -- bash -lc 'bun --version; rustc --version | head -1'
|
||||
echo "OK: kata-qemu still boots after tuning"
|
||||
REMOTE
|
||||
+156
-153
@@ -8,9 +8,165 @@
|
||||
- Fixed paste and image placeholders crashing when the editor renders before theme initialization.
|
||||
- Added `ModelRegistry.create(authStorage, modelsPath?)` async factory that runs the JSON → YAML migration step on `models.{yml,yaml}` asynchronously ahead of the sync constructor's bundled-model load. The sync `new ModelRegistry(...)` constructor still works (tests rely on it); production boot paths now use the factory so the migration's I/O lands off the event-loop hot path.
|
||||
- Added `ConfigFile.tryLoadAsync()`, `ConfigFile.loadAsync()`, `ConfigFile.loadOrDefaultAsync()`, `ConfigFile.getMtimeMsAsync()`, and `ConfigFile.warmup(file)` so the rest of the codebase can migrate config reads off the sync path.
|
||||
- Added mouse-driven interaction to `/settings`, including tab and setting row hover highlighting, wheel scrolling, and left-click activation for entries and submenus
|
||||
- Added fullscreen `/settings` mouse-event handling so scrolling and clicks work in an alternate-screen overlay
|
||||
- `ModelRegistry.resolver` now accepts a model directly — `resolver(model, sessionId)` — deriving `provider`, `baseUrl`, and `modelId` from it; all model-scoped call sites migrated from the verbose `resolver(model.provider, { sessionId, baseUrl, modelId })` form.
|
||||
- Added experimental `snapcompact.systemPrompt` and `snapcompact.toolResults` settings (off by default, `/settings` → Context → Experimental) that render the system prompt and large historical tool results as dense snapcompact PNG frames on vision-capable models to cut token cost. Frames are built per-request in the provider-context transform, cached across turns, capped by a per-provider image budget, and gated on a token-savings estimate — they never reach `session.jsonl`.
|
||||
- Added a Personality selector to `/settings` (Model → Prompt): `default` (the previous built-in reply style), `friendly`, `pragmatic`, or `none`. The selected spec renders into a dedicated `<personality>` system-prompt block (extracted from the former `<reply-guidelines>` section) and applies to the live session immediately; subagents always omit the block.
|
||||
- Added `mnemopi.polyphonicRecall` and `mnemopi.enhancedRecall` config.yml settings (off by default, `/settings` → Memory → Mnemopi) that enable the mnemopi 4-voice polyphonic recall engine and the tiered query result cache without environment variables; `MNEMOPI_POLYPHONIC_RECALL` / `MNEMOPI_ENHANCED_RECALL` still override the configured values when set ([#2323](https://github.com/can1357/oh-my-pi/issues/2323)).
|
||||
- Added the Expert Elixir language server (`expert`, invoked as `expert --stdio`) to the built-in LSP server list, auto-detected for Mix projects (`mix.exs`/`mix.lock`). When both are installed, `elixir-ls` remains the primary navigation server (Expert is ordered after it).
|
||||
- Added `magicKeywords.enabled` and per-keyword `magicKeywords.ultrathink`, `magicKeywords.orchestrate`, and `magicKeywords.workflow` settings to disable hidden magic-keyword notices and ultrathink auto-thinking escalation ([#1796](https://github.com/can1357/oh-my-pi/issues/1796)).
|
||||
- Added external-editor support for Plan Review section annotations, preserving multiline feedback for Refine plan ([#2305](https://github.com/can1357/oh-my-pi/issues/2305)).
|
||||
- Added plain-RPC slash command discovery with command source metadata and startup/update notifications ([#2261](https://github.com/can1357/oh-my-pi/issues/2261)).
|
||||
- Added the `statusLine.transparent` appearance setting (default off): when enabled, the status line skips the theme's `statusLineBg` fill and powerline end caps so the bar inherits the terminal's default background — useful in Ghostty and other terminals whose theme background does not match the theme's hardcoded status-line color ([#2306](https://github.com/can1357/oh-my-pi/issues/2306))
|
||||
- Snapcompact compaction now passes the session model so frames render in the provider-optimal shape (unscii `8x8r-bw` for Anthropic-family/unknown APIs, `8x8r-sent` for Google, Lanczos-stretched `6x6u-sent` with `detail: "original"` for OpenAI), per the snapcompact 200k-token evals
|
||||
- Added per-turn supersede pruning of stale `read` results: when a file is re-read, older copies of the same path/selector are pruned from context at cache-favorable moments (small suffix, idle gap, or alongside overflow pruning). Gated by the new `compaction.supersedeReads` setting (default on)
|
||||
- Added soft request budgets for task subagents (explore/quick_task 40, others 90, configurable via `task.softRequestBudget`, 0 disables): crossing the budget injects a one-time wrap-up steer into the child; crossing 1.5× aborts the run gracefully
|
||||
- Added cancelled/aborted subagent salvage: instead of `(no output)`, merged task results now carry the child's last activity snippet plus request/token stats, and per-child stats lines include request counts
|
||||
- Added a repeat-read notice to the `read` tool: the third and later reads of the same file in a session append a one-line note suggesting range re-reads or the context echoed in edit results
|
||||
- Added a hard inline byte cap (~50KB) at the bash and browser tool-result boundaries with head/tail elision and an `artifact://` footer for the full output, closing paths that previously let 100KB+ results land inline
|
||||
- Added the Agent Hub overlay (`ctrl+s`, `alt+a`, or double-tap left arrow on an empty editor): a live table of registered subagents (status, unread IRC count, current task, last activity) with per-agent chat — Enter opens a transcript + input line that steers a running agent, prompts an idle one, and revives a parked one; `r` revives and `x` aborts/releases the selected agent
|
||||
- Added the `snapcompact` compaction strategy (`compaction.strategy: "snapcompact"`): history is archived onto dense bitmap "snapcompact" frames a vision model reads back directly, instead of an LLM-generated summary — instant, free, and verbatim. Auto compaction (including overflow recovery) and manual `/compact` both honor it; falls back to context-full with a visible warning notice when the current model is text-only (e.g. Codex API surfaces) or when `/compact` is given custom instructions. Frames survive context rebuilds and later compactions (budget eviction is middle-out: the session-head frame is pinned); the expanded compaction message notes the attached frame count
|
||||
- Added a persistent subagent lifecycle: finished subagents stay live as `idle`, are parked to disk after `task.agentIdleTtlMs` (default 7 minutes; `0` keeps them live until exit), and are revived automatically when messaged or prompted from the Agent Hub
|
||||
- Added the `history://` protocol: `history://` lists every registered agent and `history://<agentId>` renders a concise markdown transcript (tool calls collapsed to one line each, thinking elided) for live and parked agents alike
|
||||
- Added an IRC mailbox bus with bounded per-agent inboxes: `irc` `wait` blocks until a matching message arrives, `inbox` drains or peeks pending messages, and sending to an idle or parked agent wakes or revives it for a real turn
|
||||
- Added a dedicated TUI renderer for the `irc` tool: directional send/receive headers with delivery-outcome coloring, quoted message bodies with expand-aware truncation, per-recipient receipt trees for broadcasts and failures, and status-badged peer listings with unread counts
|
||||
- Added the `task.batch` setting (default on): the task tool's batch shape `{ agent, context, tasks[] }` spawns one subagent per item — each its own independent background job with the normal idle/parked lifecycle and optional per-item isolation — and prepends the required shared `context` to every spawned subagent's system prompt; disabling it restores the flat single-spawn schema
|
||||
- Added RPC subagent subscription frames, snapshots, and transcript catch-up APIs for desktop clients embedding `omp --mode rpc`.
|
||||
- Added opt-in `shellMinimizer.sourceOutlineLevel` and `shellMinimizer.legacyFilters` settings so shell minimization can tune source outlining and selectively fall back to conservative legacy routing.
|
||||
- Added repeatable `--config <path>` CLI overlays for temporary `config.yml`-style settings without editing the persistent global config ([#1733](https://github.com/can1357/oh-my-pi/issues/1733)).
|
||||
- Added `python.interpreter` to pin eval's Python backend to an explicit interpreter and skip automatic runtime discovery ([#1802](https://github.com/can1357/oh-my-pi/issues/1802)).
|
||||
- Added `!command` resolution for `models.yml` provider `apiKey` values and provider/model headers ([#1888](https://github.com/can1357/oh-my-pi/issues/1888)).
|
||||
- Documented the oMLX setup path through existing OpenAI-compatible local discovery ([#1957](https://github.com/can1357/oh-my-pi/issues/1957)).
|
||||
- Added `TITLE_SYSTEM.md` discovery so users can override the automatic session-title generation prompt for online and local tiny title models without patching installed prompt files. The override is re-discovered when the session working directory changes via `/cwd`.
|
||||
- Added a structured memory runtime surface for extensions and UI integrations to query backend status, search memories, and save explicit memories across the configured memory backend.
|
||||
- Added support for Git repositories using the `reftable` storage format by detecting `extensions.refStorage = reftable` in the repository configuration and falling back to shelling out to Git commands (`git symbolic-ref`, `git rev-parse`) for reference and HEAD resolution.
|
||||
- Added `/setup providers` (also available as `/setup` or `/providers`) to reopen the interactive provider setup scene from an active TUI session, letting users sign in and choose a web search provider without rerunning the full onboarding flow.
|
||||
- Added `supportsReasoningParams`, `alwaysSendMaxTokens`, `strictResponsesPairing`, and a recursive `whenThinking` overlay (alongside `streamIdleTimeoutMs`/`supportsLongPromptCacheRetention`/`requiresToolResultId`/`replayUnsignedThinking`) to the OpenAI/Anthropic `compat` schema so custom model entries can configure those provider-specific capabilities
|
||||
- Custom model `thinking` config now uses the catalog's explicit vocabulary: `efforts` (ordered list) plus optional `defaultLevel`, `effortMap`, and `supportsDisplay` overrides; the legacy `minLevel`/`maxLevel`/`levels` range shape is still accepted and normalized at parse time. Wire facts (`effortMap`/`supportsDisplay`) are backfilled from model identity when not set, so existing claude-proxy configs keep the 5-tier adaptive scale and summarized display without changes.
|
||||
- New `omp usage` command: a detailed per-account breakdown of provider usage limits (bars, windows, reset times, plan metadata) covering every stored credential — accounts with no usage endpoint are listed as "no usage data" rows. Each provider section ends with per-window capacity stats ("capacity: 5h → 2.40/5 accounts used (2.60× quota left)"). Flags: `--provider` to filter, `--json` for the broker-shaped report payload, and `--redact` to mask account emails/ids down to a two-char anchor plus a minimal middle-out differentiator (`ca*9*`) for screenshot-safe sharing.
|
||||
- Startup hangs are now self-diagnosing (speculative fix for the "zero output, hangs even on `omp -h`" report class): a watchdog prints a stderr line every 10s naming the deepest in-flight startup phase (via `logger.openSpanPath()`) until a mode runner takes over, pausing around legitimate interactive waits (fork/move prompts, the `--resume` session picker); `PI_DEBUG_STARTUP` is restored as streaming synchronous `[startup]` phase markers covering command-module imports and the native addon load, which the post-startup `PI_TIMING` tree structurally cannot show for a hang; and waiting on piped-stdin EOF announces itself after 1s instead of blocking silently.
|
||||
- npm installs now execute a prebundled single-file entry: the published `bin.omp` points at `dist/cli.js` (built by `scripts/bundle-dist.ts` during `prepack`, ~18MB minified, natives/transformers/mupdf external), cutting npm-install cold start by roughly 3x versus transpiling the raw TypeScript graph per launch; `src/**` stays published for SDK consumers and worker fallbacks. The on-repo manifest keeps `bin.omp` at `src/cli.ts` — release rewrites it via the `publishBin` override in `scripts/ci-release-publish.ts` — so source installs (`bun link`, `install.sh --source`) keep working without a build step
|
||||
- Plain interactive TTY launches print a dim two-line startup splash (`omp <version>` / `Initializing session…`) before session construction so first pixels appear immediately; suppressed for resume/fork/continue flows, quiet mode, `PI_TIMING`, and non-TTY stdio
|
||||
- Added `/stats` to launch the local stats dashboard from an active session, syncing session files first and opening the same browser dashboard as `omp stats`.
|
||||
- `/settings` now supports type-to-search filtering on setting labels, paths, descriptions, and values; Escape clears an active search before closing the panel.
|
||||
- Added a read-only `view` op to the `todo` tool that echoes the current list without mutating state, so the agent can recover exact task text instead of guessing it from memory.
|
||||
- Added an optional `fetch` option to `CustomToolContext` so custom tools can use a caller-provided HTTP implementation
|
||||
- Added optional `fetch` overrides to `ModelRegistry` construction and MCP/web search/tool network calls, enabling callers to inject custom HTTP clients instead of relying on global `fetch`
|
||||
- Added a `bash.enabled` setting to disable the model-facing bash tool while leaving user-initiated bang/RPC bash commands available.
|
||||
- Added an `@<upstream>` model-selector suffix to pin an aggregator model to a single upstream provider per invocation, e.g. `--model openrouter/z-ai/glm-4.7@cerebras` (sets OpenRouter `provider.only`; Vercel AI Gateway models map to `vercelGatewayRouting.only`). Resolved through `parseModelPattern`, so it works for `--model`/`--smol`, model roles, and the SDK, and composes with a trailing thinking level (`...@cerebras:high`). The base must resolve to an aggregator (`openrouter.ai` / `ai-gateway.vercel.sh`); otherwise the `@` stays part of the id, so ids that legitimately contain `@` (`claude-opus-4-8@default`, `workers-ai/@cf/...`) are unaffected.
|
||||
- Added a `/plan-review` command that manually (re-)opens the plan-review overlay while plan mode is active. Since there is no fixed plan filename, it reviews the newest `local://<slug>-plan.md` the agent wrote — useful for pulling the review back up after dismissing it, or reviewing a plan the agent wrote without calling `resolve`.
|
||||
- Added Homebrew and mise package-manager update paths to the self-update command so installations launched from those tools are updated through their native workflows
|
||||
- Added detection of Homebrew and mise install locations so self-update chooses the manager-specific updater when the active `omp` binary comes from a package-manager-managed path
|
||||
- Added `astCondition` to TTSR rule frontmatter as a syntax-aware alternative to regex `condition`, enabling AST-based matching for edit/write tool snapshots
|
||||
- Added a built-in `ts-redundant-clear-guard` rule that flags redundant guards around `clearTimeout`, `clearInterval`, and `clearImmediate` calls
|
||||
- Added a built-in `ts-no-test-timers` rule that flags real timers (`Bun.sleep`, `setTimeout`, `setInterval`) in `*.test.ts` files, steering toward fake timers (`vi.useFakeTimers()` / `vi.advanceTimersByTime()`)
|
||||
- Added support for paste marker highlighting with accent styling (`[Paste #N, +X lines]`/`[Paste #N, Y chars]`) in the prompt editor, matching the visual treatment of image references
|
||||
- Added pixel dimensions to pasted/loaded image placeholders in the prompt — the marker now reads `[Image #N, WxH]` (falling back to `[Image #N]` when the header can't be decoded).
|
||||
- The bundled shell now treats `nohup` as a builtin: `nohup … &` runs the command without masking `SIGHUP` or detaching it, so agent-started daemons stay tied to this agent's lifetime instead of leaking as orphans when the agent exits. Updated the bash tool prompt's daemon guidance to match (dropped the `nohup … & / setsid … & / disown` detach recommendation in favor of a large `timeout` plus the persistent session).
|
||||
- Added per-tool `tool.*` theme symbol keys (nerd/unicode/ascii presets) plus a quiet `status.done` glyph, so each tool's result header can carry a signature icon instead of a generic status mark
|
||||
- macOS release binaries are now signed with a Developer ID Application identity (hardened runtime + secure timestamp + JIT/library-validation entitlements) and notarized in CI when the `APPLE_*` signing secrets are configured; releases auto-fall back to ad-hoc signing until then. This makes the shipped binaries Gatekeeper-acceptable, unblocking an official Homebrew submission ([#776](https://github.com/can1357/oh-my-pi/issues/776)). See `docs/macos-signing-notarization.md`.
|
||||
- Added a Homebrew install path: `brew install can1357/tap/omp`. The [can1357/homebrew-tap](https://github.com/can1357/homebrew-tap) formula installs the prebuilt release binary, and a `release_brew` CI job regenerates it (version + per-asset sha256) from each published release via `scripts/ci-update-brew-formula.ts` ([#776](https://github.com/can1357/oh-my-pi/issues/776)).
|
||||
- Added clickable file path hyperlinks to read tool outputs (read-call rows, grouped summaries, and inline previews) using resolved or absolute file targets with selector-based line anchors for quick navigation
|
||||
- Added a resolved-span echo to `replace block`/`delete block` edits: a successful block op now prints `replace block N → resolved lines A-B (K lines)` between the section header and the diff preview, so the model can confirm tree-sitter matched the construct it intended (e.g. catch a decorator left outside the block) instead of inferring the span from the diff after the fact.
|
||||
- Added `raw-sse.txt` to debug report bundles, exporting recent raw provider SSE diagnostics when captured
|
||||
- Added `/model` visibility for auto-selected role defaults: inferred `pi/smol`/`pi/slow`/designer choices now show as compact `[ROLE auto]` badges, while explicitly configured roles keep the existing solid badges and thinking labels.
|
||||
- Added credential provenance to the `/login` and `/logout` provider picker: each authenticated provider now shows where its credential comes from — `(login)`, `(api key)`, `(env: VAR_NAME)`, `(config)`, `(--api-key)`, or `(custom provider)` — so a real OAuth login is distinguishable from an env var that merely aliases the provider (e.g. `COPILOT_GITHUB_TOKEN`). The origin is also matched by the picker's type-to-search filter.
|
||||
- Added `display.smoothStreaming` setting (default `true`) to let users enable or disable smooth assistant-stream text reveal
|
||||
- Added `/tan <work>` slash command to fork the current conversation into a background agent so tangential work can continue asynchronously while your main session stays active
|
||||
- Added a background `/tan` dispatch message that records the handoff in the transcript and marks the delegated work as non-blocking
|
||||
- Added `providerPromptCacheKey` support to `CreateAgentSessionOptions` so `/tan` background sessions can reuse the parent session’s prompt-cache lineage
|
||||
- Added session cloning for `/tan` runs with copied artifacts and shared MCP proxy tools
|
||||
- Added `SessionManager.forkFrom`’s optional `suppressBreadcrumb` mode to avoid breadcrumb updates when forking background `/tan` sessions
|
||||
- Added OSC 5522 enhanced paste handling in `InputController`, so terminal clipboard events are decoded as image or text payloads and inserted without passing raw paste sequences to the editor
|
||||
- Added bracketed image-path paste support in `CustomEditor` so a single pasted image file path (PNG/JPEG/GIF/WEBP) is loaded from disk and inserted as an image candidate
|
||||
- Added direct support for `Image #N` insertion from pasted local image paths by routing successful image-path pastes through the same image normalization and resize flow as clipboard image pastes
|
||||
- Added `/fresh` to rotate the provider-facing session id and clear in-memory provider stream/cache state without changing the local session file.
|
||||
- Added a `ChatBlock` transcript primitive (`modes/components/chat-block.ts`) and a single `ctx.present(...)` sink (with `ctx.resetTranscript()`) so chat output is mounted in one place instead of the repeated `chatContainer.addChild(...)` + `ui.requestRender()` pattern scattered across controllers. `ChatBlock` carries a React/Svelte-style lifecycle — `onMount` starts effects, `onCleanup` registers teardown, `finish()` self-completes (stops timers and freezes the block at its final content), and `dispose()`/`resetTranscript()` tears everything down — so animated blocks own their own resources instead of leaking `setInterval`/`requestRender` bookkeeping into callers. The MCP "Connecting…" spinner is now such a block.
|
||||
- Added a `framedBlock` output-block helper (`tui/output-block.ts`) plus a `borderColor` override and `applyBg: false` (no background fill) on output blocks, a `renderStatusLine` `iconOverride`, and an `icon.search` (magnifier) theme symbol — so tool renderers can draw self-contained muted-outline frames and search-family tools can show a magnifier instead of a checkmark.
|
||||
- Added a GitHub Actions read handler to the `read`/web-fetch GitHub scraper. Fetching `github.com/{owner}/{repo}/actions/runs/{id}` renders the run metadata plus a per-job breakdown (steps listed for any job that did not succeed), and `…/actions/runs/{id}/job/{id}` (also the API-style `…/jobs/{id}`) renders a single job's metadata, step table, and full plain-text logs. Logs are fetched via the `actions/jobs/{id}/logs` redirect using `GITHUB_TOKEN`/`GH_TOKEN` when present, with the per-line ISO timestamp prefix and leading BOM stripped; the section degrades to an explicit notice when logs are unavailable (no token, private repo, or expired/unfinalized run).
|
||||
- Added anonymous fallback for Perplexity web search, allowing `web_search` and explicit Perplexity provider usage when no Perplexity credentials are configured
|
||||
- Added `gallery` CLI command to render built-in tool renderer output across streaming, in-progress, success, and failure states
|
||||
- Added `omp gallery` filtering and rendering options (`--tool`, `--state`, `--width`, `--expanded`, and `--plain`) for focused renderer previews and plain-text output
|
||||
- Added `omp gallery` fidelity for tools whose renderers are attached on the tool instance (`lsp`, `task`): the gallery now drives them through the same custom-tool render branch production uses, so regressions in that path surface in the gallery rather than only in a live session.
|
||||
- Added `omp gallery --screenshot`, which renders the gallery through a real virtual terminal (VHS) and writes PNG screenshot(s) instead of ANSI, so agents (and anything that can only read raw bytes) can actually see the rendered output. The capture forces truecolor and matches the active theme/symbol preset; tall galleries split across multiple images (whole renderers are never cut). Tune with `--out`, `--font`, and `--font-size`; requires `vhs` on `PATH` and fails with install guidance when absent.
|
||||
- Added `app.display.reset`, bound to `Ctrl+L` by default, to force an immediate terminal display reset/redraw without resizing the window.
|
||||
- Added `timeout-pause` and `timeout-resume` eval bridge status events emitted around `agent()`/`llm()` operations
|
||||
- Added a `/copy` picker: `/copy` now opens a fullscreen, outlined tree of recent assistant messages with their code blocks nested beneath (like `/tree`). Navigate with ↑↓, and Enter copies the highlighted node — a whole message, an individual code block, "All N blocks", or a bash/eval command interleaved with the assistant turn that issued it. A live preview pane shows the selected target, wrapping prose and syntax-highlighting code/commands.
|
||||
- Added a persistent error banner pinned above the editor when an assistant turn ends on a provider error (e.g. Anthropic's "Output blocked by content filtering policy"). The transcript `Error: …` line scrolls away as the conversation grows, so terminal turns that ended on a stream error could pass unnoticed; the banner stays in the fixed region above the input and is cleared when the next turn starts.
|
||||
- Added bold, underlined, clickable `[Image #N]` placeholders in the draft editor and sent user-message bubbles, backed by extension-bearing blob-store sidecar files so terminal `file://` links open in image viewers.
|
||||
- Added the active model identifier (`provider/id`) to the system prompt's `<workstation>` block so the agent knows which model it is running as. Gated by the new `includeModelInPrompt` setting (default on); the base prompt is rebuilt on a mid-session model switch so the surfaced identifier stays current.
|
||||
- Added `OLLAMA_HOST` support for implicit local Ollama discovery when `OLLAMA_BASE_URL` is unset, so OMP picks up the same host setting used by Ollama.
|
||||
- Added `OLLAMA_CONTEXT_LENGTH` as a positive-integer context-window override for implicit local Ollama discovery, so users can correct OMP context budgeting without writing per-model overrides.
|
||||
- Added an encrypted local auth-broker snapshot cache for `discoverAuthStorage`, with `OMP_AUTH_BROKER_SNAPSHOT_TTL_MS` and `OMP_AUTH_BROKER_SNAPSHOT_CACHE`, so fresh cached broker credentials can boot without a blocking `/v1/snapshot` fetch and survive broker-down startup windows.
|
||||
- Added `dry-balance` CLI command to perform a dry-run OAuth account balancing check across configurable random session IDs, with sample and concurrency options, JSON output, and success/failure summary reporting
|
||||
- Added `--json` output mode and machine-readable result format to `omp dry-balance` for automated use
|
||||
- Added `omitMaxOutputTokens` to `models.yml` model definitions and `modelOverrides`, so users can opt a model out of the on-the-wire `max_output_tokens` / `max_tokens` cap while keeping the catalog `maxTokens` for local budgeting. Intended for Ollama-style proxies whose upstream output limit OMP cannot discover. ([#1881](https://github.com/can1357/oh-my-pi/issues/1881))
|
||||
- Added deferred session-title generation so greetings no longer become the session title. A first user message that is only a greeting / acknowledgement / filler ("hi", "thanks", "ok", a bare number, emoji-only, etc.) is now detected deterministically and skips titling entirely — no title model is invoked. Title generation then retries on each subsequent user message while the session stays unnamed, so the title is deduced from the first message that actually describes work. A capable online title model may additionally answer `none` to decline a non-greeting taskless message (normalized to "no title").
|
||||
- Added env-driven OpenTelemetry trace export. When `OTEL_EXPORTER_OTLP_ENDPOINT` (or `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT`) is set, `omp` registers a global OTLP/proto trace exporter and switches on the agent loop's telemetry, so the `invoke_agent` / `chat` / `execute_tool` spans actually reach a collector instead of a no-op tracer. Honors the standard `OTEL_*` env contract (endpoint, headers, `OTEL_SERVICE_NAME`, `OTEL_SDK_DISABLED` and `OTEL_TRACES_EXPORTER=none` parsed case-insensitively) and the `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT` capture toggle; it is a no-op when no endpoint is configured. Only the `http/protobuf` transport is supported — a `grpc` or `http/json` `OTEL_EXPORTER_OTLP*_PROTOCOL` declines rather than misrouting spans. This makes the existing telemetry usable from headless hosts that run `omp` as a spawned child process, where an in-process `TracerProvider` registered by the parent can't reach the child. Uses the `@opentelemetry/exporter-trace-otlp-proto` 2.x line, which exports cleanly under Bun.
|
||||
## Fixed
|
||||
|
||||
- Fixed the status line session name (and the editor border / status-line gap fill) being nearly illegible on light themes.
|
||||
- Added `IndexedSessionStorage` and `SessionStorageBackend` exports to support shared metadata-indexed session backends
|
||||
- Added the `tui.maxInlineImages` setting (default `8`) capping how many inline images render as live terminal graphics. Once a new image pushes the count past the cap, the oldest images are hidden via a full redraw — replaced by their `[Image: …]` text placeholder and purged from the terminal's graphics store — so long sessions with many screenshots/diagrams stop piling up images (and, on Kitty, stop leaving scrollback ghosts). Set to `0` to keep every image inline.
|
||||
- Added a "View: terminal state" item to the `/debug` menu that prints the detected terminal, live geometry and cell size, multiplexer, and the negotiated subprotocols actually in use — graphics (Kitty/iTerm2/Sixel), desktop notifications (BEL/OSC 9/OSC 99, plus whether OSC 99 was confirmed via a device-attributes probe), OSC 8 hyperlinks, 24-bit color, DECCARA rectangular-SGR background fills, and DEC 2026 synchronized output — alongside the scrollback-clear strategy (`CSI 22 J` vs `CSI 2 J` redraw / ED3 eager-erase risk) and the raw `TERM`/`TERM_PROGRAM`/`COLORTERM` detection signals.
|
||||
- Added a "Test: terminal protocols" item to the `/debug` menu that renders one live sample of every special escape protocol the renderer can emit — SGR text attributes (bold/italic/underline/strikethrough/inverse/dim), themed and 24-bit truecolor, OSC 8 hyperlinks, OSC 66 text sizing (large text), and an inline graphics swatch via the active image protocol (Kitty/iTerm2/Sixel, with a text fallback) — and fires a desktop notification, so you can eyeball which protocols the current terminal actually honors. The sample image is a gradient PNG generated in-process, so the graphics test needs no asset on disk.
|
||||
- Added the `tui.textSizing` setting (default off) that renders Markdown H1 headings at 2x scale via Kitty's OSC 66 text-sizing protocol. It replaces the undocumented `PI_TUI_TEXT_SIZING` env var with a real setting, and only takes effect on Kitty terminals (where OSC 66 is implemented) — it is ignored everywhere else so headings never emit raw escape bytes.
|
||||
- Added a lifecycle status to the `/resume` session picker. Each session's tail (last 32 KiB) is now read alongside the existing header window in a single pass, and its final message classified as `done` (the agent ended its turn and yielded control back), `interrupted` (a trailing tool call or tool result the loop never continued from), `aborted`, `error`, or `pending` (a trailing user message with no reply). The status renders as a colored segment on each session's metadata line. When the final message is larger than the tail window the status is omitted rather than guessed.
|
||||
- Added support for `disable-model-invocation: true` frontmatter field from the [Agent Skills standard](https://agentskills.io/specification). Skills using this field are now hidden from the system prompt listing, matching the behavior of `hide: true`.
|
||||
- Added a bundled TypeScript rule that warns against leaving `@deprecated` compatibility shims behind instead of finishing a refactor.
|
||||
- Added an all-projects scope to the session picker (`pi --resume` / `/resume`). Press `Tab` to toggle between the current folder's sessions and every session across all projects; the all-projects list is loaded lazily and shows each session's directory. When the current folder has no sessions the picker now opens straight into all-projects scope instead of printing "No sessions found".
|
||||
- Migrated the Kagi web search provider to Kagi's V1 Search API (`POST /api/v1/search`), replacing the sunset V0 endpoint while keeping the `kagi` provider id, `KAGI_API_KEY` credential, and `/login kagi` flow unchanged ([#1272](https://github.com/can1357/oh-my-pi/pull/1272) by [@thismat](https://github.com/thismat))
|
||||
- Added Anthropic `anthropic-ratelimit-unified-*` response-header warming for `/usage` and the status-line usage segment, throttled to reduce direct OAuth `/usage` probes during active use.
|
||||
- Added `ask` option descriptions so agents can keep short labels and render explanatory text as separate muted rows in the selector.
|
||||
- Added an extension API for rendering supplemental UI below visible assistant thinking blocks.
|
||||
- Added default-on `lsp.diagnosticsDeduplicate` support so post-edit LSP diagnostics already shown for a file are suppressed within the session and only new or changed diagnostics are surfaced.
|
||||
- Added support for decimal and `k`/`m` suffix turn-budget directives, enabling budgets like `+1.5k` and `+2m` in eval message parsing
|
||||
- Changed eval budget resolution to honor a user `+Nk` directive over an active Goal Mode limit while falling back to Goal Mode when no per-turn ceiling is set
|
||||
- Added `agent()` eval options `agent_type`/`agentType`, `model`, `context`, and `label`, and returned structured JSON when `schema` is provided in JS and Python eval cells
|
||||
- Added a live, Task-tool-style progress tree for eval `agent()` calls, drawn below the notebook (code cell) box. Each subagent surfaces as a status line (icon · id · tool count · context · cost, plus duration on completion) with its current tool/intent while running, and updates mid-execution rather than only at the cell's final result. Progress events coalesce per subagent id so the persisted event list stays bounded across many throttled ticks.
|
||||
- Added `agent()` to the `eval` runtime so JS and Python cells can spawn one subagent through the existing task executor; JS eval also gained bounded `parallel()` and `pipeline()` helpers for orchestrating subagent calls.
|
||||
- Added a `workflow` magic keyword (mirrors `orchestrate`/`ultrathink`): the standalone word glows amber→green in the editor and appends a hidden notice steering the model to author deterministic multi-subagent fan-outs in `eval` (agent/parallel/pipeline). Matching is whitespace-delimited and case-sensitive (lowercase only); the singular and plural both trigger, but capitalized forms, inflections like `workflowed`, and path-embedded occurrences like `workflow.ts` do not.
|
||||
- Added `parallel()` and `pipeline()` to the Python `eval` runtime (thread-pool over the synchronous `agent()` bridge), mirroring the JS helpers: bounded pool (default 4, max 16), input-order preservation, a barrier between every `pipeline` stage, and contextvar propagation so `agent()` works inside worker threads.
|
||||
- Added `log()`, `phase()`, and a `budget` object to both `eval` runtimes (Python and JS). `log`/`phase` emit progress/phase status lines; `budget.total`/`budget.spent()`/`budget.remaining()`/`budget.hard` expose a real per-turn output-token budget. A `+Nk` directive in the user's message sets an advisory budget (the model self-limits via `budget.remaining()`); `+Nk!` (or an active Goal Mode budget) makes it a hard ceiling that blocks further eval `agent()` spawns once reached. `budget.spent()` counts output tokens spent this turn across the main loop and all eval-spawned subagents.
|
||||
- Added search support for virtual internal URLs (including `omp://` roots) by resolving and scanning in-memory internal resources as search targets alongside filesystem paths
|
||||
- Added expansion of virtual internal URL search targets so `search` can match multiple internal documents when given `omp://`
|
||||
- Added `/omfg <complaint>` slash command that drafts a TTSR rule from a complaint, validates it against the current conversation, saves it to project or `~/.omp/agent/rules`, and registers it live.
|
||||
- Added `/shake` slash command and the `shake` / `shake-summary` compaction strategies that reduce context by mechanically dropping heavy content instead of LLM summarization. `/shake` (alias `/shake elide`) strips heavy tool-call results and large fenced/XML blocks, offloads the originals to one session artifact, and leaves a recoverable `artifact://<id>` placeholder; `/shake summary` compresses the same regions with a local on-device model (`providers.shakeSummaryModel`, default `qwen3-1.7b`) and falls back to elide per region when the model is unavailable; `/shake images` strips image blocks. Auto-maintenance honors the `shake` / `shake-summary` strategies (16k protect window); on context overflow a shake that reclaims nothing falls back to context-full summarization.
|
||||
- Added `providers.shakeSummaryModel` setting selecting the local on-device model used by `/shake summary` and the `shake-summary` compaction strategy. Runs entirely on-device (downloads on first use) and never calls a remote/cloud LLM.
|
||||
- Added `providers.autoThinkingModel` setting so users can choose the `auto` thinking classifier backend (online smol or local tiny-memory model)
|
||||
- Added an `auto` thinking level that classifies each real user turn and resolves to a concrete low-through-xhigh effort, with online smol classification by default and an opt-in local on-device classifier.
|
||||
- Added a `Web search` setup tab that lets users choose the preferred `providers.webSearch` provider during onboarding
|
||||
- Added manual authorization-code/redirect URL prompts for OAuth providers that require non-callback login in the setup wizard
|
||||
- Added an `omp completions <bash|zsh|fish>` command that prints a shell completion script generated from the live command/flag metadata, so completions never drift from the actual CLI. Subcommands, flags, and enum values complete statically; `--model`/`--smol`/`--slow`/`--plan` resolve against the bundled model catalog and `--resume` against on-disk sessions via a hidden `__complete` helper.
|
||||
- Added a `/switch` slash command that opens the temporary model selector for the current session, mirroring the `alt+p` keybinding.
|
||||
- Added `replace block N:` and `delete block N` operators to the `edit` tool: they resolve the syntactic block beginning on line N via tree-sitter (native `blockRangeAt`) and replace or delete its full line span, so a construct can be rewritten or removed without counting its closing line. Unresolvable blocks (unsupported language, blank/closing-delimiter line, or a parse error) are rejected with guidance to use an explicit `replace N..M:` / `delete N..M` range.
|
||||
- Added an animated pending border for `bash` and `eval` execution blocks: while a command/cell is running, a single dark segment glides clockwise around the block's outer edge (top → right → bottom → left), replacing the previous static accent border. Motion is eased per edge (decelerating into each corner) and timed against a fixed lap duration mapped onto the live perimeter, so streaming a new output line or resizing the terminal nudges the segment proportionally instead of resetting its position. Driven by the existing spinner cadence and gated on the `display.shimmer` setting (no motion when `disabled`).
|
||||
- Added `providers.tinyModelDevice` and `providers.tinyModelDtype` settings (Providers tab) controlling local tiny-model acceleration for session titles and Mnemopi memory tasks. `providers.tinyModelDevice` selects the ONNX execution provider (`default` keeps the platform pick — DirectML on Windows, CUDA on Linux x64, CPU elsewhere); `providers.tinyModelDtype` selects quantization/precision (`default` keeps each model's shipped `q4`, e.g. `fp16` trades speed for fidelity). The `PI_TINY_DEVICE` / `PI_TINY_DTYPE` env vars override the matching setting. Also added `PI_TINY_DTYPE` as the env counterpart to `PI_TINY_DEVICE`; an unrecognized device/precision fails loudly at worker startup instead of silently loading a different one.
|
||||
- Added a bundled set of default rules shipped with the agent (TypeScript/Rust convention rules registered as TTSR conditions). They load via the new lowest-priority `builtin-defaults` discovery provider, so any user/project/tool rule of the same name overrides the bundled copy. Disable the whole set with `ttsr.builtinRules: false`, or drop individual rules (bundled or your own) by name via `ttsr.disabledRules`.
|
||||
- Added a `symbols.spinnerFrames` field to custom theme JSON so themes can override the loader/tool-execution spinner. Accepts either a flat `string[]` (used for both spinner types) or `{ "status"?: string[], "activity"?: string[] }` to override each independently; anything not specified falls back to the symbol preset. Documented in `docs/theme.md` and validated by `theme-schema.json`. ([#1553](https://github.com/can1357/oh-my-pi/issues/1553))
|
||||
- Added prompt-mode autocomplete for supported internal URL schemes (`skill://`, `rule://`, `agent://`, `artifact://`, `local://`, `memory://`, and `omp://`) so typing those tokens now suggests existing resources as completion candidates
|
||||
- Added fuzzy matching and ranked suggestion ordering for internal URL completion, including rule and skill descriptions, with accepted completion replacing just the typed token and inserting the chosen URL followed by a space
|
||||
- Changed internal URL completions now include nested `local://` path suggestions from the configured local workspace
|
||||
- Added Mnemopi memory inference model selection with an online mode or local transformers.js options (`qwen3-1.7b`, `gemma-3-1b`, `qwen2.5-1.5b`, `lfm2-1.2b`) so memory extraction and consolidation can run via the shared tiny-model worker
|
||||
- Changed memory tiny-model handling to route local memory prompts through the same queueed tiny-model worker pipeline with bounded completion output
|
||||
- Added a Providers → Tiny Model setting for session titles, defaulting to the online `pi/smol` path with five optional local CPU transformers.js models. A local model — and the one-time `@huggingface/transformers` runtime install in compiled binaries — is downloaded and loaded only when explicitly selected (or via `omp tiny-models download`); the default online path never spawns the title worker for inference. Selecting a local model adds a delayed `pi/smol` fallback so titles never block, plus in-chat download progress.
|
||||
- Added a persistent live agent roster pinned below the editor (focus it with `Ctrl+S` or `Alt+Down`), including view-as switching into delegated agent sessions with human-readable delegate names and UI pinning to suppress idle reaping while viewed. The roster stays hidden until at least one delegated agent exists and releases focus back to the editor once the last one is gone.
|
||||
- Recorded the originating session ID alongside each prompt in `history.db` (new `session_id` column, surfaced as `HistoryEntry.sessionId`), so recalled prompts can be traced back to the session they came from. Existing history databases gain the column automatically on next launch.
|
||||
- Added compact inline TUI renderers for the `retain`, `recall`, and `reflect` memory tools. `retain` now shows one themed bullet line per stored item (truncated to width) under a status header with the stored/queued count, and `recall`/`reflect` collapse to a single query header (recall reports the match count and hides recalled memories until expanded) instead of dumping the raw JSON argument tree.
|
||||
- Added a randomly picked tip beneath the welcome screen, sourced from an embedded `tips.txt` (one tip per line). The line is italicized with a purple `Tip:` label and a dimmed light-blue body, and the tip is chosen once per welcome instance so intro-animation and LSP re-renders don't shuffle it.
|
||||
- Added a Mnemopi-only `memory_edit` agent tool for updating, forgetting, or invalidating recalled memories by id, and added `/memory stats` plus `/memory diagnose` slash commands for backend maintenance visibility.
|
||||
- Added an `orchestrate` magic keyword that mirrors `ultrathink`: dropping the standalone word in a message paints it with a cool teal→violet gradient in the editor and appends a hidden system notice that switches the model into the multi-phase, parallel-subagent orchestration contract. Matching is word-bounded and case-insensitive, so `orchestrated`/`orchestrating` never trigger it.
|
||||
- Added a model-tier slider to the plan-approval prompt ("Plan mode - next step"). Left/right arrows move it from any list position to pick which configured role model (`cycleOrder`, e.g. `smol › default › slow`) executes the approved plan, with each tier colored by its role and the resolved model name shown beneath the track. The chosen tier is applied before dispatch and carries through the fresh/compacted execution session; the slider is hidden when fewer than two role models resolve.
|
||||
- `omp plugin install` now accepts GitHub/GitLab/Bitbucket shorthand (`github:user/repo`, `gitlab:user/repo`, …) and full git URLs (`https://github.com/user/repo`, `git@github.com:user/repo`, …) in addition to npm specs and marketplace refs.
|
||||
|
||||
### Changed
|
||||
|
||||
- Replaced the `omp bench` default prompt with a concrete query-planning trace that requires deriving selectivities, cardinalities, and I/O/CPU costs from given schema and data. The old prompt was open-ended prose recall, which rewarded not-thinking: adaptive-thinking models (Opus 4.6+/Sonnet 4.6+) minimized reasoning on the trivial task and streamed faster, skewing throughput comparisons. The new task forces multi-step reasoning so adaptive thinking engages and the benchmark measures generation under real cognitive load. The prompt also demands explicit upfront deliberation and exhaustive enumeration/costing of every join order, so adaptive-thinking models cannot short-circuit to a quick answer and each model is measured over a sustained generation up to the token cap.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Kokoro TTS setup loading the workspace/global `@huggingface/transformers` runtime before the side-installed Kokoro runtime, which could leave `onnxruntime-node@1.26.0` bound to an older `libonnxruntime.so.1` and fail with `VERS_1.26.0` missing ([#2591](https://github.com/can1357/oh-my-pi/issues/2591)).
|
||||
- Fixed `Test & smoke (TS)` CI timeouts caused by parallel test files racing on the process-global Settings singleton. `CustomEditor` now accepts a `magicKeywordsEnabledOverride` injection point so the shimmer-gate test can assert behaviour without calling `resetSettingsForTest()` / `Settings.init()`; the "streaming tool call preview height" describe drops its gratuitous Settings reset+init. Production wiring is unchanged ([#2582](https://github.com/can1357/oh-my-pi/issues/2582))
|
||||
- Fixed MCP OAuth fallback rendering to show a short terminal hyperlink and keep the raw authorization URL on one unwrapped copy line ([#2121](https://github.com/can1357/oh-my-pi/issues/2121)).
|
||||
- Fixed `omp dry-balance --bench` to recover from 401 token failures by re-minting the failing OAuth credential in place before switching accounts
|
||||
@@ -343,159 +499,6 @@
|
||||
- `/context` (TUI panel and ACP report) now shows estimated snapcompact wire savings when `snapcompact.systemPrompt` or `snapcompact.toolResults` is enabled — per-feature text → frames token deltas, the reason a swap does not apply (savings margin, image budget, or text-only model), and the estimated size of the next request. The estimate and the live provider-request transform share one planner (`planInlineSwaps`) so displayed numbers cannot drift from wire behavior.
|
||||
- Added `/debug dump-request` and `/debug next-request` as aliases for `/debug dump-next-request` when arming a one-shot AI provider request dump
|
||||
- Added `/debug dump-next-request <path>` to dump the next AI provider HTTP request JSON to a chosen file.
|
||||
- Added mouse-driven interaction to `/settings`, including tab and setting row hover highlighting, wheel scrolling, and left-click activation for entries and submenus
|
||||
- Added fullscreen `/settings` mouse-event handling so scrolling and clicks work in an alternate-screen overlay
|
||||
- `ModelRegistry.resolver` now accepts a model directly — `resolver(model, sessionId)` — deriving `provider`, `baseUrl`, and `modelId` from it; all model-scoped call sites migrated from the verbose `resolver(model.provider, { sessionId, baseUrl, modelId })` form.
|
||||
- Added experimental `snapcompact.systemPrompt` and `snapcompact.toolResults` settings (off by default, `/settings` → Context → Experimental) that render the system prompt and large historical tool results as dense snapcompact PNG frames on vision-capable models to cut token cost. Frames are built per-request in the provider-context transform, cached across turns, capped by a per-provider image budget, and gated on a token-savings estimate — they never reach `session.jsonl`.
|
||||
- Added a Personality selector to `/settings` (Model → Prompt): `default` (the previous built-in reply style), `friendly`, `pragmatic`, or `none`. The selected spec renders into a dedicated `<personality>` system-prompt block (extracted from the former `<reply-guidelines>` section) and applies to the live session immediately; subagents always omit the block.
|
||||
- Added `mnemopi.polyphonicRecall` and `mnemopi.enhancedRecall` config.yml settings (off by default, `/settings` → Memory → Mnemopi) that enable the mnemopi 4-voice polyphonic recall engine and the tiered query result cache without environment variables; `MNEMOPI_POLYPHONIC_RECALL` / `MNEMOPI_ENHANCED_RECALL` still override the configured values when set ([#2323](https://github.com/can1357/oh-my-pi/issues/2323)).
|
||||
- Added the Expert Elixir language server (`expert`, invoked as `expert --stdio`) to the built-in LSP server list, auto-detected for Mix projects (`mix.exs`/`mix.lock`). When both are installed, `elixir-ls` remains the primary navigation server (Expert is ordered after it).
|
||||
- Added `magicKeywords.enabled` and per-keyword `magicKeywords.ultrathink`, `magicKeywords.orchestrate`, and `magicKeywords.workflow` settings to disable hidden magic-keyword notices and ultrathink auto-thinking escalation ([#1796](https://github.com/can1357/oh-my-pi/issues/1796)).
|
||||
- Added external-editor support for Plan Review section annotations, preserving multiline feedback for Refine plan ([#2305](https://github.com/can1357/oh-my-pi/issues/2305)).
|
||||
- Added plain-RPC slash command discovery with command source metadata and startup/update notifications ([#2261](https://github.com/can1357/oh-my-pi/issues/2261)).
|
||||
- Added the `statusLine.transparent` appearance setting (default off): when enabled, the status line skips the theme's `statusLineBg` fill and powerline end caps so the bar inherits the terminal's default background — useful in Ghostty and other terminals whose theme background does not match the theme's hardcoded status-line color ([#2306](https://github.com/can1357/oh-my-pi/issues/2306))
|
||||
- Snapcompact compaction now passes the session model so frames render in the provider-optimal shape (unscii `8x8r-bw` for Anthropic-family/unknown APIs, `8x8r-sent` for Google, Lanczos-stretched `6x6u-sent` with `detail: "original"` for OpenAI), per the snapcompact 200k-token evals
|
||||
- Added per-turn supersede pruning of stale `read` results: when a file is re-read, older copies of the same path/selector are pruned from context at cache-favorable moments (small suffix, idle gap, or alongside overflow pruning). Gated by the new `compaction.supersedeReads` setting (default on)
|
||||
- Added soft request budgets for task subagents (explore/quick_task 40, others 90, configurable via `task.softRequestBudget`, 0 disables): crossing the budget injects a one-time wrap-up steer into the child; crossing 1.5× aborts the run gracefully
|
||||
- Added cancelled/aborted subagent salvage: instead of `(no output)`, merged task results now carry the child's last activity snippet plus request/token stats, and per-child stats lines include request counts
|
||||
- Added a repeat-read notice to the `read` tool: the third and later reads of the same file in a session append a one-line note suggesting range re-reads or the context echoed in edit results
|
||||
- Added a hard inline byte cap (~50KB) at the bash and browser tool-result boundaries with head/tail elision and an `artifact://` footer for the full output, closing paths that previously let 100KB+ results land inline
|
||||
- Added the Agent Hub overlay (`ctrl+s`, `alt+a`, or double-tap left arrow on an empty editor): a live table of registered subagents (status, unread IRC count, current task, last activity) with per-agent chat — Enter opens a transcript + input line that steers a running agent, prompts an idle one, and revives a parked one; `r` revives and `x` aborts/releases the selected agent
|
||||
- Added the `snapcompact` compaction strategy (`compaction.strategy: "snapcompact"`): history is archived onto dense bitmap "snapcompact" frames a vision model reads back directly, instead of an LLM-generated summary — instant, free, and verbatim. Auto compaction (including overflow recovery) and manual `/compact` both honor it; falls back to context-full with a visible warning notice when the current model is text-only (e.g. Codex API surfaces) or when `/compact` is given custom instructions. Frames survive context rebuilds and later compactions (budget eviction is middle-out: the session-head frame is pinned); the expanded compaction message notes the attached frame count
|
||||
- Added a persistent subagent lifecycle: finished subagents stay live as `idle`, are parked to disk after `task.agentIdleTtlMs` (default 7 minutes; `0` keeps them live until exit), and are revived automatically when messaged or prompted from the Agent Hub
|
||||
- Added the `history://` protocol: `history://` lists every registered agent and `history://<agentId>` renders a concise markdown transcript (tool calls collapsed to one line each, thinking elided) for live and parked agents alike
|
||||
- Added an IRC mailbox bus with bounded per-agent inboxes: `irc` `wait` blocks until a matching message arrives, `inbox` drains or peeks pending messages, and sending to an idle or parked agent wakes or revives it for a real turn
|
||||
- Added a dedicated TUI renderer for the `irc` tool: directional send/receive headers with delivery-outcome coloring, quoted message bodies with expand-aware truncation, per-recipient receipt trees for broadcasts and failures, and status-badged peer listings with unread counts
|
||||
- Added the `task.batch` setting (default on): the task tool's batch shape `{ agent, context, tasks[] }` spawns one subagent per item — each its own independent background job with the normal idle/parked lifecycle and optional per-item isolation — and prepends the required shared `context` to every spawned subagent's system prompt; disabling it restores the flat single-spawn schema
|
||||
- Added RPC subagent subscription frames, snapshots, and transcript catch-up APIs for desktop clients embedding `omp --mode rpc`.
|
||||
- Added opt-in `shellMinimizer.sourceOutlineLevel` and `shellMinimizer.legacyFilters` settings so shell minimization can tune source outlining and selectively fall back to conservative legacy routing.
|
||||
- Added repeatable `--config <path>` CLI overlays for temporary `config.yml`-style settings without editing the persistent global config ([#1733](https://github.com/can1357/oh-my-pi/issues/1733)).
|
||||
- Added `python.interpreter` to pin eval's Python backend to an explicit interpreter and skip automatic runtime discovery ([#1802](https://github.com/can1357/oh-my-pi/issues/1802)).
|
||||
- Added `!command` resolution for `models.yml` provider `apiKey` values and provider/model headers ([#1888](https://github.com/can1357/oh-my-pi/issues/1888)).
|
||||
- Documented the oMLX setup path through existing OpenAI-compatible local discovery ([#1957](https://github.com/can1357/oh-my-pi/issues/1957)).
|
||||
- Added `TITLE_SYSTEM.md` discovery so users can override the automatic session-title generation prompt for online and local tiny title models without patching installed prompt files. The override is re-discovered when the session working directory changes via `/cwd`.
|
||||
- Added a structured memory runtime surface for extensions and UI integrations to query backend status, search memories, and save explicit memories across the configured memory backend.
|
||||
- Added support for Git repositories using the `reftable` storage format by detecting `extensions.refStorage = reftable` in the repository configuration and falling back to shelling out to Git commands (`git symbolic-ref`, `git rev-parse`) for reference and HEAD resolution.
|
||||
- Added `/setup providers` (also available as `/setup` or `/providers`) to reopen the interactive provider setup scene from an active TUI session, letting users sign in and choose a web search provider without rerunning the full onboarding flow.
|
||||
- Added `supportsReasoningParams`, `alwaysSendMaxTokens`, `strictResponsesPairing`, and a recursive `whenThinking` overlay (alongside `streamIdleTimeoutMs`/`supportsLongPromptCacheRetention`/`requiresToolResultId`/`replayUnsignedThinking`) to the OpenAI/Anthropic `compat` schema so custom model entries can configure those provider-specific capabilities
|
||||
- Custom model `thinking` config now uses the catalog's explicit vocabulary: `efforts` (ordered list) plus optional `defaultLevel`, `effortMap`, and `supportsDisplay` overrides; the legacy `minLevel`/`maxLevel`/`levels` range shape is still accepted and normalized at parse time. Wire facts (`effortMap`/`supportsDisplay`) are backfilled from model identity when not set, so existing claude-proxy configs keep the 5-tier adaptive scale and summarized display without changes.
|
||||
- New `omp usage` command: a detailed per-account breakdown of provider usage limits (bars, windows, reset times, plan metadata) covering every stored credential — accounts with no usage endpoint are listed as "no usage data" rows. Each provider section ends with per-window capacity stats ("capacity: 5h → 2.40/5 accounts used (2.60× quota left)"). Flags: `--provider` to filter, `--json` for the broker-shaped report payload, and `--redact` to mask account emails/ids down to a two-char anchor plus a minimal middle-out differentiator (`ca*9*`) for screenshot-safe sharing.
|
||||
- Startup hangs are now self-diagnosing (speculative fix for the "zero output, hangs even on `omp -h`" report class): a watchdog prints a stderr line every 10s naming the deepest in-flight startup phase (via `logger.openSpanPath()`) until a mode runner takes over, pausing around legitimate interactive waits (fork/move prompts, the `--resume` session picker); `PI_DEBUG_STARTUP` is restored as streaming synchronous `[startup]` phase markers covering command-module imports and the native addon load, which the post-startup `PI_TIMING` tree structurally cannot show for a hang; and waiting on piped-stdin EOF announces itself after 1s instead of blocking silently.
|
||||
- npm installs now execute a prebundled single-file entry: the published `bin.omp` points at `dist/cli.js` (built by `scripts/bundle-dist.ts` during `prepack`, ~18MB minified, natives/transformers/mupdf external), cutting npm-install cold start by roughly 3x versus transpiling the raw TypeScript graph per launch; `src/**` stays published for SDK consumers and worker fallbacks. The on-repo manifest keeps `bin.omp` at `src/cli.ts` — release rewrites it via the `publishBin` override in `scripts/ci-release-publish.ts` — so source installs (`bun link`, `install.sh --source`) keep working without a build step
|
||||
- Plain interactive TTY launches print a dim two-line startup splash (`omp <version>` / `Initializing session…`) before session construction so first pixels appear immediately; suppressed for resume/fork/continue flows, quiet mode, `PI_TIMING`, and non-TTY stdio
|
||||
- Added `/stats` to launch the local stats dashboard from an active session, syncing session files first and opening the same browser dashboard as `omp stats`.
|
||||
- `/settings` now supports type-to-search filtering on setting labels, paths, descriptions, and values; Escape clears an active search before closing the panel.
|
||||
- Added a read-only `view` op to the `todo` tool that echoes the current list without mutating state, so the agent can recover exact task text instead of guessing it from memory.
|
||||
- Added an optional `fetch` option to `CustomToolContext` so custom tools can use a caller-provided HTTP implementation
|
||||
- Added optional `fetch` overrides to `ModelRegistry` construction and MCP/web search/tool network calls, enabling callers to inject custom HTTP clients instead of relying on global `fetch`
|
||||
- Added a `bash.enabled` setting to disable the model-facing bash tool while leaving user-initiated bang/RPC bash commands available.
|
||||
- Added an `@<upstream>` model-selector suffix to pin an aggregator model to a single upstream provider per invocation, e.g. `--model openrouter/z-ai/glm-4.7@cerebras` (sets OpenRouter `provider.only`; Vercel AI Gateway models map to `vercelGatewayRouting.only`). Resolved through `parseModelPattern`, so it works for `--model`/`--smol`, model roles, and the SDK, and composes with a trailing thinking level (`...@cerebras:high`). The base must resolve to an aggregator (`openrouter.ai` / `ai-gateway.vercel.sh`); otherwise the `@` stays part of the id, so ids that legitimately contain `@` (`claude-opus-4-8@default`, `workers-ai/@cf/...`) are unaffected.
|
||||
- Added a `/plan-review` command that manually (re-)opens the plan-review overlay while plan mode is active. Since there is no fixed plan filename, it reviews the newest `local://<slug>-plan.md` the agent wrote — useful for pulling the review back up after dismissing it, or reviewing a plan the agent wrote without calling `resolve`.
|
||||
- Added Homebrew and mise package-manager update paths to the self-update command so installations launched from those tools are updated through their native workflows
|
||||
- Added detection of Homebrew and mise install locations so self-update chooses the manager-specific updater when the active `omp` binary comes from a package-manager-managed path
|
||||
- Added `astCondition` to TTSR rule frontmatter as a syntax-aware alternative to regex `condition`, enabling AST-based matching for edit/write tool snapshots
|
||||
- Added a built-in `ts-redundant-clear-guard` rule that flags redundant guards around `clearTimeout`, `clearInterval`, and `clearImmediate` calls
|
||||
- Added a built-in `ts-no-test-timers` rule that flags real timers (`Bun.sleep`, `setTimeout`, `setInterval`) in `*.test.ts` files, steering toward fake timers (`vi.useFakeTimers()` / `vi.advanceTimersByTime()`)
|
||||
- Added support for paste marker highlighting with accent styling (`[Paste #N, +X lines]`/`[Paste #N, Y chars]`) in the prompt editor, matching the visual treatment of image references
|
||||
- Added pixel dimensions to pasted/loaded image placeholders in the prompt — the marker now reads `[Image #N, WxH]` (falling back to `[Image #N]` when the header can't be decoded).
|
||||
- The bundled shell now treats `nohup` as a builtin: `nohup … &` runs the command without masking `SIGHUP` or detaching it, so agent-started daemons stay tied to this agent's lifetime instead of leaking as orphans when the agent exits. Updated the bash tool prompt's daemon guidance to match (dropped the `nohup … & / setsid … & / disown` detach recommendation in favor of a large `timeout` plus the persistent session).
|
||||
- Added per-tool `tool.*` theme symbol keys (nerd/unicode/ascii presets) plus a quiet `status.done` glyph, so each tool's result header can carry a signature icon instead of a generic status mark
|
||||
- macOS release binaries are now signed with a Developer ID Application identity (hardened runtime + secure timestamp + JIT/library-validation entitlements) and notarized in CI when the `APPLE_*` signing secrets are configured; releases auto-fall back to ad-hoc signing until then. This makes the shipped binaries Gatekeeper-acceptable, unblocking an official Homebrew submission ([#776](https://github.com/can1357/oh-my-pi/issues/776)). See `docs/macos-signing-notarization.md`.
|
||||
- Added a Homebrew install path: `brew install can1357/tap/omp`. The [can1357/homebrew-tap](https://github.com/can1357/homebrew-tap) formula installs the prebuilt release binary, and a `release_brew` CI job regenerates it (version + per-asset sha256) from each published release via `scripts/ci-update-brew-formula.ts` ([#776](https://github.com/can1357/oh-my-pi/issues/776)).
|
||||
- Added clickable file path hyperlinks to read tool outputs (read-call rows, grouped summaries, and inline previews) using resolved or absolute file targets with selector-based line anchors for quick navigation
|
||||
- Added a resolved-span echo to `replace block`/`delete block` edits: a successful block op now prints `replace block N → resolved lines A-B (K lines)` between the section header and the diff preview, so the model can confirm tree-sitter matched the construct it intended (e.g. catch a decorator left outside the block) instead of inferring the span from the diff after the fact.
|
||||
- Added `raw-sse.txt` to debug report bundles, exporting recent raw provider SSE diagnostics when captured
|
||||
- Added `/model` visibility for auto-selected role defaults: inferred `pi/smol`/`pi/slow`/designer choices now show as compact `[ROLE auto]` badges, while explicitly configured roles keep the existing solid badges and thinking labels.
|
||||
- Added credential provenance to the `/login` and `/logout` provider picker: each authenticated provider now shows where its credential comes from — `(login)`, `(api key)`, `(env: VAR_NAME)`, `(config)`, `(--api-key)`, or `(custom provider)` — so a real OAuth login is distinguishable from an env var that merely aliases the provider (e.g. `COPILOT_GITHUB_TOKEN`). The origin is also matched by the picker's type-to-search filter.
|
||||
- Added `display.smoothStreaming` setting (default `true`) to let users enable or disable smooth assistant-stream text reveal
|
||||
- Added `/tan <work>` slash command to fork the current conversation into a background agent so tangential work can continue asynchronously while your main session stays active
|
||||
- Added a background `/tan` dispatch message that records the handoff in the transcript and marks the delegated work as non-blocking
|
||||
- Added `providerPromptCacheKey` support to `CreateAgentSessionOptions` so `/tan` background sessions can reuse the parent session’s prompt-cache lineage
|
||||
- Added session cloning for `/tan` runs with copied artifacts and shared MCP proxy tools
|
||||
- Added `SessionManager.forkFrom`’s optional `suppressBreadcrumb` mode to avoid breadcrumb updates when forking background `/tan` sessions
|
||||
- Added OSC 5522 enhanced paste handling in `InputController`, so terminal clipboard events are decoded as image or text payloads and inserted without passing raw paste sequences to the editor
|
||||
- Added bracketed image-path paste support in `CustomEditor` so a single pasted image file path (PNG/JPEG/GIF/WEBP) is loaded from disk and inserted as an image candidate
|
||||
- Added direct support for `Image #N` insertion from pasted local image paths by routing successful image-path pastes through the same image normalization and resize flow as clipboard image pastes
|
||||
- Added `/fresh` to rotate the provider-facing session id and clear in-memory provider stream/cache state without changing the local session file.
|
||||
- Added a `ChatBlock` transcript primitive (`modes/components/chat-block.ts`) and a single `ctx.present(...)` sink (with `ctx.resetTranscript()`) so chat output is mounted in one place instead of the repeated `chatContainer.addChild(...)` + `ui.requestRender()` pattern scattered across controllers. `ChatBlock` carries a React/Svelte-style lifecycle — `onMount` starts effects, `onCleanup` registers teardown, `finish()` self-completes (stops timers and freezes the block at its final content), and `dispose()`/`resetTranscript()` tears everything down — so animated blocks own their own resources instead of leaking `setInterval`/`requestRender` bookkeeping into callers. The MCP "Connecting…" spinner is now such a block.
|
||||
- Added a `framedBlock` output-block helper (`tui/output-block.ts`) plus a `borderColor` override and `applyBg: false` (no background fill) on output blocks, a `renderStatusLine` `iconOverride`, and an `icon.search` (magnifier) theme symbol — so tool renderers can draw self-contained muted-outline frames and search-family tools can show a magnifier instead of a checkmark.
|
||||
- Added a GitHub Actions read handler to the `read`/web-fetch GitHub scraper. Fetching `github.com/{owner}/{repo}/actions/runs/{id}` renders the run metadata plus a per-job breakdown (steps listed for any job that did not succeed), and `…/actions/runs/{id}/job/{id}` (also the API-style `…/jobs/{id}`) renders a single job's metadata, step table, and full plain-text logs. Logs are fetched via the `actions/jobs/{id}/logs` redirect using `GITHUB_TOKEN`/`GH_TOKEN` when present, with the per-line ISO timestamp prefix and leading BOM stripped; the section degrades to an explicit notice when logs are unavailable (no token, private repo, or expired/unfinalized run).
|
||||
- Added anonymous fallback for Perplexity web search, allowing `web_search` and explicit Perplexity provider usage when no Perplexity credentials are configured
|
||||
- Added `gallery` CLI command to render built-in tool renderer output across streaming, in-progress, success, and failure states
|
||||
- Added `omp gallery` filtering and rendering options (`--tool`, `--state`, `--width`, `--expanded`, and `--plain`) for focused renderer previews and plain-text output
|
||||
- Added `omp gallery` fidelity for tools whose renderers are attached on the tool instance (`lsp`, `task`): the gallery now drives them through the same custom-tool render branch production uses, so regressions in that path surface in the gallery rather than only in a live session.
|
||||
- Added `omp gallery --screenshot`, which renders the gallery through a real virtual terminal (VHS) and writes PNG screenshot(s) instead of ANSI, so agents (and anything that can only read raw bytes) can actually see the rendered output. The capture forces truecolor and matches the active theme/symbol preset; tall galleries split across multiple images (whole renderers are never cut). Tune with `--out`, `--font`, and `--font-size`; requires `vhs` on `PATH` and fails with install guidance when absent.
|
||||
- Added `app.display.reset`, bound to `Ctrl+L` by default, to force an immediate terminal display reset/redraw without resizing the window.
|
||||
- Added `timeout-pause` and `timeout-resume` eval bridge status events emitted around `agent()`/`llm()` operations
|
||||
- Added a `/copy` picker: `/copy` now opens a fullscreen, outlined tree of recent assistant messages with their code blocks nested beneath (like `/tree`). Navigate with ↑↓, and Enter copies the highlighted node — a whole message, an individual code block, "All N blocks", or a bash/eval command interleaved with the assistant turn that issued it. A live preview pane shows the selected target, wrapping prose and syntax-highlighting code/commands.
|
||||
- Added a persistent error banner pinned above the editor when an assistant turn ends on a provider error (e.g. Anthropic's "Output blocked by content filtering policy"). The transcript `Error: …` line scrolls away as the conversation grows, so terminal turns that ended on a stream error could pass unnoticed; the banner stays in the fixed region above the input and is cleared when the next turn starts.
|
||||
- Added bold, underlined, clickable `[Image #N]` placeholders in the draft editor and sent user-message bubbles, backed by extension-bearing blob-store sidecar files so terminal `file://` links open in image viewers.
|
||||
- Added the active model identifier (`provider/id`) to the system prompt's `<workstation>` block so the agent knows which model it is running as. Gated by the new `includeModelInPrompt` setting (default on); the base prompt is rebuilt on a mid-session model switch so the surfaced identifier stays current.
|
||||
- Added `OLLAMA_HOST` support for implicit local Ollama discovery when `OLLAMA_BASE_URL` is unset, so OMP picks up the same host setting used by Ollama.
|
||||
- Added `OLLAMA_CONTEXT_LENGTH` as a positive-integer context-window override for implicit local Ollama discovery, so users can correct OMP context budgeting without writing per-model overrides.
|
||||
- Added an encrypted local auth-broker snapshot cache for `discoverAuthStorage`, with `OMP_AUTH_BROKER_SNAPSHOT_TTL_MS` and `OMP_AUTH_BROKER_SNAPSHOT_CACHE`, so fresh cached broker credentials can boot without a blocking `/v1/snapshot` fetch and survive broker-down startup windows.
|
||||
- Added `dry-balance` CLI command to perform a dry-run OAuth account balancing check across configurable random session IDs, with sample and concurrency options, JSON output, and success/failure summary reporting
|
||||
- Added `--json` output mode and machine-readable result format to `omp dry-balance` for automated use
|
||||
- Added `omitMaxOutputTokens` to `models.yml` model definitions and `modelOverrides`, so users can opt a model out of the on-the-wire `max_output_tokens` / `max_tokens` cap while keeping the catalog `maxTokens` for local budgeting. Intended for Ollama-style proxies whose upstream output limit OMP cannot discover. ([#1881](https://github.com/can1357/oh-my-pi/issues/1881))
|
||||
- Added deferred session-title generation so greetings no longer become the session title. A first user message that is only a greeting / acknowledgement / filler ("hi", "thanks", "ok", a bare number, emoji-only, etc.) is now detected deterministically and skips titling entirely — no title model is invoked. Title generation then retries on each subsequent user message while the session stays unnamed, so the title is deduced from the first message that actually describes work. A capable online title model may additionally answer `none` to decline a non-greeting taskless message (normalized to "no title").
|
||||
- Added env-driven OpenTelemetry trace export. When `OTEL_EXPORTER_OTLP_ENDPOINT` (or `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT`) is set, `omp` registers a global OTLP/proto trace exporter and switches on the agent loop's telemetry, so the `invoke_agent` / `chat` / `execute_tool` spans actually reach a collector instead of a no-op tracer. Honors the standard `OTEL_*` env contract (endpoint, headers, `OTEL_SERVICE_NAME`, `OTEL_SDK_DISABLED` and `OTEL_TRACES_EXPORTER=none` parsed case-insensitively) and the `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT` capture toggle; it is a no-op when no endpoint is configured. Only the `http/protobuf` transport is supported — a `grpc` or `http/json` `OTEL_EXPORTER_OTLP*_PROTOCOL` declines rather than misrouting spans. This makes the existing telemetry usable from headless hosts that run `omp` as a spawned child process, where an in-process `TracerProvider` registered by the parent can't reach the child. Uses the `@opentelemetry/exporter-trace-otlp-proto` 2.x line, which exports cleanly under Bun.
|
||||
## Fixed
|
||||
|
||||
- Fixed the status line session name (and the editor border / status-line gap fill) being nearly illegible on light themes.
|
||||
- Added `IndexedSessionStorage` and `SessionStorageBackend` exports to support shared metadata-indexed session backends
|
||||
- Added the `tui.maxInlineImages` setting (default `8`) capping how many inline images render as live terminal graphics. Once a new image pushes the count past the cap, the oldest images are hidden via a full redraw — replaced by their `[Image: …]` text placeholder and purged from the terminal's graphics store — so long sessions with many screenshots/diagrams stop piling up images (and, on Kitty, stop leaving scrollback ghosts). Set to `0` to keep every image inline.
|
||||
- Added a "View: terminal state" item to the `/debug` menu that prints the detected terminal, live geometry and cell size, multiplexer, and the negotiated subprotocols actually in use — graphics (Kitty/iTerm2/Sixel), desktop notifications (BEL/OSC 9/OSC 99, plus whether OSC 99 was confirmed via a device-attributes probe), OSC 8 hyperlinks, 24-bit color, DECCARA rectangular-SGR background fills, and DEC 2026 synchronized output — alongside the scrollback-clear strategy (`CSI 22 J` vs `CSI 2 J` redraw / ED3 eager-erase risk) and the raw `TERM`/`TERM_PROGRAM`/`COLORTERM` detection signals.
|
||||
- Added a "Test: terminal protocols" item to the `/debug` menu that renders one live sample of every special escape protocol the renderer can emit — SGR text attributes (bold/italic/underline/strikethrough/inverse/dim), themed and 24-bit truecolor, OSC 8 hyperlinks, OSC 66 text sizing (large text), and an inline graphics swatch via the active image protocol (Kitty/iTerm2/Sixel, with a text fallback) — and fires a desktop notification, so you can eyeball which protocols the current terminal actually honors. The sample image is a gradient PNG generated in-process, so the graphics test needs no asset on disk.
|
||||
- Added the `tui.textSizing` setting (default off) that renders Markdown H1 headings at 2x scale via Kitty's OSC 66 text-sizing protocol. It replaces the undocumented `PI_TUI_TEXT_SIZING` env var with a real setting, and only takes effect on Kitty terminals (where OSC 66 is implemented) — it is ignored everywhere else so headings never emit raw escape bytes.
|
||||
- Added a lifecycle status to the `/resume` session picker. Each session's tail (last 32 KiB) is now read alongside the existing header window in a single pass, and its final message classified as `done` (the agent ended its turn and yielded control back), `interrupted` (a trailing tool call or tool result the loop never continued from), `aborted`, `error`, or `pending` (a trailing user message with no reply). The status renders as a colored segment on each session's metadata line. When the final message is larger than the tail window the status is omitted rather than guessed.
|
||||
- Added support for `disable-model-invocation: true` frontmatter field from the [Agent Skills standard](https://agentskills.io/specification). Skills using this field are now hidden from the system prompt listing, matching the behavior of `hide: true`.
|
||||
- Added a bundled TypeScript rule that warns against leaving `@deprecated` compatibility shims behind instead of finishing a refactor.
|
||||
- Added an all-projects scope to the session picker (`pi --resume` / `/resume`). Press `Tab` to toggle between the current folder's sessions and every session across all projects; the all-projects list is loaded lazily and shows each session's directory. When the current folder has no sessions the picker now opens straight into all-projects scope instead of printing "No sessions found".
|
||||
- Migrated the Kagi web search provider to Kagi's V1 Search API (`POST /api/v1/search`), replacing the sunset V0 endpoint while keeping the `kagi` provider id, `KAGI_API_KEY` credential, and `/login kagi` flow unchanged ([#1272](https://github.com/can1357/oh-my-pi/pull/1272) by [@thismat](https://github.com/thismat))
|
||||
- Added Anthropic `anthropic-ratelimit-unified-*` response-header warming for `/usage` and the status-line usage segment, throttled to reduce direct OAuth `/usage` probes during active use.
|
||||
- Added `ask` option descriptions so agents can keep short labels and render explanatory text as separate muted rows in the selector.
|
||||
- Added an extension API for rendering supplemental UI below visible assistant thinking blocks.
|
||||
- Added default-on `lsp.diagnosticsDeduplicate` support so post-edit LSP diagnostics already shown for a file are suppressed within the session and only new or changed diagnostics are surfaced.
|
||||
- Added support for decimal and `k`/`m` suffix turn-budget directives, enabling budgets like `+1.5k` and `+2m` in eval message parsing
|
||||
- Changed eval budget resolution to honor a user `+Nk` directive over an active Goal Mode limit while falling back to Goal Mode when no per-turn ceiling is set
|
||||
- Added `agent()` eval options `agent_type`/`agentType`, `model`, `context`, and `label`, and returned structured JSON when `schema` is provided in JS and Python eval cells
|
||||
- Added a live, Task-tool-style progress tree for eval `agent()` calls, drawn below the notebook (code cell) box. Each subagent surfaces as a status line (icon · id · tool count · context · cost, plus duration on completion) with its current tool/intent while running, and updates mid-execution rather than only at the cell's final result. Progress events coalesce per subagent id so the persisted event list stays bounded across many throttled ticks.
|
||||
- Added `agent()` to the `eval` runtime so JS and Python cells can spawn one subagent through the existing task executor; JS eval also gained bounded `parallel()` and `pipeline()` helpers for orchestrating subagent calls.
|
||||
- Added a `workflow` magic keyword (mirrors `orchestrate`/`ultrathink`): the standalone word glows amber→green in the editor and appends a hidden notice steering the model to author deterministic multi-subagent fan-outs in `eval` (agent/parallel/pipeline). Matching is whitespace-delimited and case-sensitive (lowercase only); the singular and plural both trigger, but capitalized forms, inflections like `workflowed`, and path-embedded occurrences like `workflow.ts` do not.
|
||||
- Added `parallel()` and `pipeline()` to the Python `eval` runtime (thread-pool over the synchronous `agent()` bridge), mirroring the JS helpers: bounded pool (default 4, max 16), input-order preservation, a barrier between every `pipeline` stage, and contextvar propagation so `agent()` works inside worker threads.
|
||||
- Added `log()`, `phase()`, and a `budget` object to both `eval` runtimes (Python and JS). `log`/`phase` emit progress/phase status lines; `budget.total`/`budget.spent()`/`budget.remaining()`/`budget.hard` expose a real per-turn output-token budget. A `+Nk` directive in the user's message sets an advisory budget (the model self-limits via `budget.remaining()`); `+Nk!` (or an active Goal Mode budget) makes it a hard ceiling that blocks further eval `agent()` spawns once reached. `budget.spent()` counts output tokens spent this turn across the main loop and all eval-spawned subagents.
|
||||
- Added search support for virtual internal URLs (including `omp://` roots) by resolving and scanning in-memory internal resources as search targets alongside filesystem paths
|
||||
- Added expansion of virtual internal URL search targets so `search` can match multiple internal documents when given `omp://`
|
||||
- Added `/omfg <complaint>` slash command that drafts a TTSR rule from a complaint, validates it against the current conversation, saves it to project or `~/.omp/agent/rules`, and registers it live.
|
||||
- Added `/shake` slash command and the `shake` / `shake-summary` compaction strategies that reduce context by mechanically dropping heavy content instead of LLM summarization. `/shake` (alias `/shake elide`) strips heavy tool-call results and large fenced/XML blocks, offloads the originals to one session artifact, and leaves a recoverable `artifact://<id>` placeholder; `/shake summary` compresses the same regions with a local on-device model (`providers.shakeSummaryModel`, default `qwen3-1.7b`) and falls back to elide per region when the model is unavailable; `/shake images` strips image blocks. Auto-maintenance honors the `shake` / `shake-summary` strategies (16k protect window); on context overflow a shake that reclaims nothing falls back to context-full summarization.
|
||||
- Added `providers.shakeSummaryModel` setting selecting the local on-device model used by `/shake summary` and the `shake-summary` compaction strategy. Runs entirely on-device (downloads on first use) and never calls a remote/cloud LLM.
|
||||
- Added `providers.autoThinkingModel` setting so users can choose the `auto` thinking classifier backend (online smol or local tiny-memory model)
|
||||
- Added an `auto` thinking level that classifies each real user turn and resolves to a concrete low-through-xhigh effort, with online smol classification by default and an opt-in local on-device classifier.
|
||||
- Added a `Web search` setup tab that lets users choose the preferred `providers.webSearch` provider during onboarding
|
||||
- Added manual authorization-code/redirect URL prompts for OAuth providers that require non-callback login in the setup wizard
|
||||
- Added an `omp completions <bash|zsh|fish>` command that prints a shell completion script generated from the live command/flag metadata, so completions never drift from the actual CLI. Subcommands, flags, and enum values complete statically; `--model`/`--smol`/`--slow`/`--plan` resolve against the bundled model catalog and `--resume` against on-disk sessions via a hidden `__complete` helper.
|
||||
- Added a `/switch` slash command that opens the temporary model selector for the current session, mirroring the `alt+p` keybinding.
|
||||
- Added `replace block N:` and `delete block N` operators to the `edit` tool: they resolve the syntactic block beginning on line N via tree-sitter (native `blockRangeAt`) and replace or delete its full line span, so a construct can be rewritten or removed without counting its closing line. Unresolvable blocks (unsupported language, blank/closing-delimiter line, or a parse error) are rejected with guidance to use an explicit `replace N..M:` / `delete N..M` range.
|
||||
- Added an animated pending border for `bash` and `eval` execution blocks: while a command/cell is running, a single dark segment glides clockwise around the block's outer edge (top → right → bottom → left), replacing the previous static accent border. Motion is eased per edge (decelerating into each corner) and timed against a fixed lap duration mapped onto the live perimeter, so streaming a new output line or resizing the terminal nudges the segment proportionally instead of resetting its position. Driven by the existing spinner cadence and gated on the `display.shimmer` setting (no motion when `disabled`).
|
||||
- Added `providers.tinyModelDevice` and `providers.tinyModelDtype` settings (Providers tab) controlling local tiny-model acceleration for session titles and Mnemopi memory tasks. `providers.tinyModelDevice` selects the ONNX execution provider (`default` keeps the platform pick — DirectML on Windows, CUDA on Linux x64, CPU elsewhere); `providers.tinyModelDtype` selects quantization/precision (`default` keeps each model's shipped `q4`, e.g. `fp16` trades speed for fidelity). The `PI_TINY_DEVICE` / `PI_TINY_DTYPE` env vars override the matching setting. Also added `PI_TINY_DTYPE` as the env counterpart to `PI_TINY_DEVICE`; an unrecognized device/precision fails loudly at worker startup instead of silently loading a different one.
|
||||
- Added a bundled set of default rules shipped with the agent (TypeScript/Rust convention rules registered as TTSR conditions). They load via the new lowest-priority `builtin-defaults` discovery provider, so any user/project/tool rule of the same name overrides the bundled copy. Disable the whole set with `ttsr.builtinRules: false`, or drop individual rules (bundled or your own) by name via `ttsr.disabledRules`.
|
||||
- Added a `symbols.spinnerFrames` field to custom theme JSON so themes can override the loader/tool-execution spinner. Accepts either a flat `string[]` (used for both spinner types) or `{ "status"?: string[], "activity"?: string[] }` to override each independently; anything not specified falls back to the symbol preset. Documented in `docs/theme.md` and validated by `theme-schema.json`. ([#1553](https://github.com/can1357/oh-my-pi/issues/1553))
|
||||
- Added prompt-mode autocomplete for supported internal URL schemes (`skill://`, `rule://`, `agent://`, `artifact://`, `local://`, `memory://`, and `omp://`) so typing those tokens now suggests existing resources as completion candidates
|
||||
- Added fuzzy matching and ranked suggestion ordering for internal URL completion, including rule and skill descriptions, with accepted completion replacing just the typed token and inserting the chosen URL followed by a space
|
||||
- Changed internal URL completions now include nested `local://` path suggestions from the configured local workspace
|
||||
- Added Mnemopi memory inference model selection with an online mode or local transformers.js options (`qwen3-1.7b`, `gemma-3-1b`, `qwen2.5-1.5b`, `lfm2-1.2b`) so memory extraction and consolidation can run via the shared tiny-model worker
|
||||
- Changed memory tiny-model handling to route local memory prompts through the same queueed tiny-model worker pipeline with bounded completion output
|
||||
- Added a Providers → Tiny Model setting for session titles, defaulting to the online `pi/smol` path with five optional local CPU transformers.js models. A local model — and the one-time `@huggingface/transformers` runtime install in compiled binaries — is downloaded and loaded only when explicitly selected (or via `omp tiny-models download`); the default online path never spawns the title worker for inference. Selecting a local model adds a delayed `pi/smol` fallback so titles never block, plus in-chat download progress.
|
||||
- Added a persistent live agent roster pinned below the editor (focus it with `Ctrl+S` or `Alt+Down`), including view-as switching into delegated agent sessions with human-readable delegate names and UI pinning to suppress idle reaping while viewed. The roster stays hidden until at least one delegated agent exists and releases focus back to the editor once the last one is gone.
|
||||
- Recorded the originating session ID alongside each prompt in `history.db` (new `session_id` column, surfaced as `HistoryEntry.sessionId`), so recalled prompts can be traced back to the session they came from. Existing history databases gain the column automatically on next launch.
|
||||
- Added compact inline TUI renderers for the `retain`, `recall`, and `reflect` memory tools. `retain` now shows one themed bullet line per stored item (truncated to width) under a status header with the stored/queued count, and `recall`/`reflect` collapse to a single query header (recall reports the match count and hides recalled memories until expanded) instead of dumping the raw JSON argument tree.
|
||||
- Added a randomly picked tip beneath the welcome screen, sourced from an embedded `tips.txt` (one tip per line). The line is italicized with a purple `Tip:` label and a dimmed light-blue body, and the tip is chosen once per welcome instance so intro-animation and LSP re-renders don't shuffle it.
|
||||
- Added a Mnemopi-only `memory_edit` agent tool for updating, forgetting, or invalidating recalled memories by id, and added `/memory stats` plus `/memory diagnose` slash commands for backend maintenance visibility.
|
||||
- Added an `orchestrate` magic keyword that mirrors `ultrathink`: dropping the standalone word in a message paints it with a cool teal→violet gradient in the editor and appends a hidden system notice that switches the model into the multi-phase, parallel-subagent orchestration contract. Matching is word-bounded and case-insensitive, so `orchestrated`/`orchestrating` never trigger it.
|
||||
- Added a model-tier slider to the plan-approval prompt ("Plan mode - next step"). Left/right arrows move it from any list position to pick which configured role model (`cycleOrder`, e.g. `smol › default › slow`) executes the approved plan, with each tier colored by its role and the resolved model name shown beneath the track. The chosen tier is applied before dispatch and carries through the fresh/compacted execution session; the slider is hidden when fewer than two role models resolve.
|
||||
- `omp plugin install` now accepts GitHub/GitLab/Bitbucket shorthand (`github:user/repo`, `gitlab:user/repo`, …) and full git URLs (`https://github.com/user/repo`, `git@github.com:user/repo`, …) in addition to npm specs and marketplace refs.
|
||||
- Added `ModelRegistry.create(authStorage, modelsPath?)` async factory that runs the JSON → YAML migration step on `models.{yml,yaml}` asynchronously ahead of the sync constructor's bundled-model load. The sync `new ModelRegistry(...)` constructor still works (tests rely on it); production boot paths now use the factory so the migration's I/O lands off the event-loop hot path.
|
||||
- Added `ConfigFile.tryLoadAsync()`, `ConfigFile.loadAsync()`, `ConfigFile.loadOrDefaultAsync()`, `ConfigFile.getMtimeMsAsync()`, and `ConfigFile.warmup(file)` so the rest of the codebase can migrate config reads off the sync path.
|
||||
|
||||
### Changed
|
||||
|
||||
|
||||
@@ -5,12 +5,14 @@
|
||||
import { isValidThemeColor, type ThemeColor } from "../modes/theme/theme";
|
||||
import type { Settings } from "./settings";
|
||||
|
||||
export type ModelRole = "default" | "smol" | "slow" | "vision" | "plan" | "designer" | "commit" | "task";
|
||||
export type ModelRole = "default" | "smol" | "slow" | "vision" | "plan" | "designer" | "commit" | "title" | "task";
|
||||
|
||||
export interface ModelRoleInfo {
|
||||
tag?: string;
|
||||
name: string;
|
||||
color?: ThemeColor;
|
||||
/** If true, the role is functional but not shown in the model selector UI. */
|
||||
hidden?: boolean;
|
||||
}
|
||||
|
||||
export const MODEL_ROLES: Record<ModelRole, ModelRoleInfo> = {
|
||||
@@ -21,12 +23,22 @@ export const MODEL_ROLES: Record<ModelRole, ModelRoleInfo> = {
|
||||
plan: { tag: "PLAN", name: "Architect", color: "muted" },
|
||||
designer: { tag: "DESIGNER", name: "Designer", color: "muted" },
|
||||
commit: { tag: "COMMIT", name: "Commit", color: "dim" },
|
||||
title: { tag: "TITLE", name: "Title", color: "dim", hidden: true },
|
||||
task: { tag: "TASK", name: "Subtask", color: "muted" },
|
||||
};
|
||||
|
||||
export const MODEL_ROLE_IDS: ModelRole[] = ["default", "smol", "slow", "vision", "plan", "designer", "commit", "task"];
|
||||
export const MODEL_ROLE_IDS: ModelRole[] = [
|
||||
"default",
|
||||
"smol",
|
||||
"slow",
|
||||
"vision",
|
||||
"plan",
|
||||
"designer",
|
||||
"commit",
|
||||
"title",
|
||||
"task",
|
||||
];
|
||||
|
||||
/** Alias for ModelRoleInfo - used for both built-in and custom roles */
|
||||
export type RoleInfo = ModelRoleInfo;
|
||||
|
||||
/**
|
||||
@@ -37,7 +49,7 @@ export type RoleInfo = ModelRoleInfo;
|
||||
* entries across settings.
|
||||
*/
|
||||
export function getKnownRoleIds(settings: Settings): string[] {
|
||||
const roles = [...MODEL_ROLE_IDS] as string[];
|
||||
const roles = MODEL_ROLE_IDS.filter(role => !MODEL_ROLES[role as ModelRole]?.hidden) as string[];
|
||||
const seen = new Set<string>(roles);
|
||||
const addRole = (role: string) => {
|
||||
if (seen.has(role)) return;
|
||||
@@ -65,6 +77,7 @@ export function getRoleInfo(role: string, settings: Settings): RoleInfo {
|
||||
tag: builtIn?.tag,
|
||||
name: configured.name || builtIn?.name || role,
|
||||
color: configured.color && isValidThemeColor(configured.color) ? configured.color : builtIn?.color,
|
||||
hidden: configured.hidden ?? builtIn?.hidden,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -258,6 +258,8 @@ type SettingDef =
|
||||
export interface ModelTagDef {
|
||||
name: string;
|
||||
color?: string;
|
||||
/** If true, the role is functional but not shown in the model selector UI. */
|
||||
hidden?: boolean;
|
||||
}
|
||||
|
||||
export interface ModelTagsSettings {
|
||||
|
||||
@@ -679,13 +679,14 @@ describe("agent() through eval runtimes", () => {
|
||||
cost: 0,
|
||||
durationMs: i * 10,
|
||||
});
|
||||
await Bun.sleep(5);
|
||||
await Bun.sleep(40);
|
||||
}
|
||||
return singleResult(options, { output: "done" });
|
||||
});
|
||||
|
||||
const ops: string[] = [];
|
||||
using idle = new IdleTimeout(40);
|
||||
// Timing invariant (keep, do not re-tighten): total mock work (20*40ms = 800ms) > idle window (250ms) > scheduling jitter (~tens of ms).
|
||||
using idle = new IdleTimeout(250);
|
||||
const result = await runEvalAgent(
|
||||
{ prompt: "investigate" },
|
||||
{
|
||||
|
||||
@@ -936,7 +936,8 @@ export class ModelSelectorComponent extends Container {
|
||||
// Build role badges. Solid badges are configured; outlined badges are auto-selected defaults.
|
||||
const roleBadgeTokens: string[] = [];
|
||||
for (const role of MODEL_ROLE_IDS) {
|
||||
const { tag, color } = getRoleInfo(role, this.#settings);
|
||||
const { tag, color, hidden } = getRoleInfo(role, this.#settings);
|
||||
if (hidden) continue;
|
||||
const assigned = this.#roles[role];
|
||||
if (!tag || !assigned || !modelsAreEqual(assigned.model, item.model)) continue;
|
||||
|
||||
@@ -1053,6 +1054,10 @@ export class ModelSelectorComponent extends Container {
|
||||
this.#menuStep = "role";
|
||||
this.#menuSelectedRole = null;
|
||||
this.#menuSelectedIndex = 0;
|
||||
// Collapse the model list while the action/thinking menu is open so the
|
||||
// menu owns the full viewport instead of stacking below a now-irrelevant
|
||||
// (and often off-screen) list.
|
||||
this.#listContainer.clear();
|
||||
this.#updateMenu();
|
||||
}
|
||||
|
||||
@@ -1061,6 +1066,8 @@ export class ModelSelectorComponent extends Container {
|
||||
this.#menuStep = "role";
|
||||
this.#menuSelectedRole = null;
|
||||
this.#menuContainer.clear();
|
||||
// Restore the model list that #openMenu collapsed.
|
||||
this.#updateList();
|
||||
}
|
||||
|
||||
#updateMenu(): void {
|
||||
@@ -1088,11 +1095,21 @@ export class ModelSelectorComponent extends Container {
|
||||
? ` Thinking for: ${selectedRoleName} (${selectedItem.id})`
|
||||
: ` Action for: ${selectedItem.id}`;
|
||||
const hintText = showingThinking ? " Enter: confirm Esc: back" : " Enter: continue Esc: cancel";
|
||||
const menuWidth = Math.max(
|
||||
// Window the option list so a long action/thinking menu scrolls inside the
|
||||
// viewport instead of running off the bottom of the screen.
|
||||
const maxVisible = this.#getMenuVisibleCount(optionLines.length);
|
||||
const needsScroll = optionLines.length > maxVisible;
|
||||
const startIndex = needsScroll
|
||||
? Math.max(0, Math.min(this.#menuSelectedIndex - Math.floor(maxVisible / 2), optionLines.length - maxVisible))
|
||||
: 0;
|
||||
const endIndex = needsScroll ? startIndex + maxVisible : optionLines.length;
|
||||
const contentWidth = Math.max(
|
||||
visibleWidth(headerText),
|
||||
visibleWidth(hintText),
|
||||
...optionLines.map(line => visibleWidth(line)),
|
||||
);
|
||||
// Reserve one column for the scrollbar when the list overflows.
|
||||
const menuWidth = contentWidth + (needsScroll ? 1 : 0);
|
||||
|
||||
this.#menuContainer.addChild(new Spacer(1));
|
||||
this.#menuContainer.addChild(new Text(theme.fg("border", theme.boxSharp.horizontal.repeat(menuWidth)), 0, 0));
|
||||
@@ -1109,12 +1126,28 @@ export class ModelSelectorComponent extends Container {
|
||||
}
|
||||
this.#menuContainer.addChild(new Spacer(1));
|
||||
|
||||
for (let i = 0; i < optionLines.length; i++) {
|
||||
const visibleRows: string[] = [];
|
||||
for (let i = startIndex; i < endIndex; i++) {
|
||||
const lineText = optionLines[i];
|
||||
if (!lineText) continue;
|
||||
if (lineText === undefined) continue;
|
||||
const isSelected = i === this.#menuSelectedIndex;
|
||||
const line = isSelected ? theme.fg("accent", lineText) : theme.fg("muted", lineText);
|
||||
this.#menuContainer.addChild(new Text(line, 0, 0));
|
||||
visibleRows.push(isSelected ? theme.fg("accent", lineText) : theme.fg("muted", lineText));
|
||||
}
|
||||
if (needsScroll) {
|
||||
const sv = new ScrollView(visibleRows, {
|
||||
height: visibleRows.length,
|
||||
scrollbar: "auto",
|
||||
totalRows: optionLines.length,
|
||||
theme: { track: t => theme.fg("muted", t), thumb: t => theme.fg("accent", t) },
|
||||
});
|
||||
sv.setScrollOffset(startIndex);
|
||||
for (const row of sv.render(menuWidth)) {
|
||||
this.#menuContainer.addChild(new Text(row, 0, 0));
|
||||
}
|
||||
} else {
|
||||
for (const row of visibleRows) {
|
||||
this.#menuContainer.addChild(new Text(row, 0, 0));
|
||||
}
|
||||
}
|
||||
|
||||
this.#menuContainer.addChild(new Spacer(1));
|
||||
@@ -1122,6 +1155,17 @@ export class ModelSelectorComponent extends Container {
|
||||
this.#menuContainer.addChild(new Text(theme.fg("border", theme.boxSharp.horizontal.repeat(menuWidth)), 0, 0));
|
||||
}
|
||||
|
||||
#getMenuVisibleCount(optionCount: number): number {
|
||||
// Rows the selector chrome and the menu's own header/hint/borders/spacers
|
||||
// consume, leaving the remainder of the viewport for the scrollable option
|
||||
// window. Without a known terminal height (e.g. tests) show every option.
|
||||
const MENU_CHROME_ROWS = 19;
|
||||
const MIN_VISIBLE_OPTIONS = 4;
|
||||
const terminalRows = this.#tui.terminal?.rows ?? 0;
|
||||
if (!Number.isFinite(terminalRows) || terminalRows <= 0) return optionCount;
|
||||
return Math.max(MIN_VISIBLE_OPTIONS, Math.min(optionCount, terminalRows - MENU_CHROME_ROWS));
|
||||
}
|
||||
|
||||
handleInput(keyData: string): void {
|
||||
if (this.#isMenuOpen) {
|
||||
this.#handleMenuInput(keyData);
|
||||
|
||||
@@ -1,7 +1,12 @@
|
||||
Write a continuous, plain-prose technical explanation of how a relational database executes a SQL query: lexing and parsing, semantic analysis, logical plan construction, cost-based optimization, physical operator selection, and row-by-row execution through the iterator model.
|
||||
You are given a relational schema and a multi-way analytical query, and you must work out from first principles the execution plan a cost-based optimizer should choose. This is a hard estimation problem with a large search space, so think it all the way through before you settle on anything and reason your way to each number instead of answering from intuition. Do not recite how query optimization works in general — actually do the analysis for this query, deriving every estimate.
|
||||
|
||||
Schema and statistics: orders(id, customer_id, status, total) holds 50,000,000 rows with 5 distinct status values; customers(id, country, segment) holds 4,000,000 rows across 200 countries; line_items(order_id, product_id, qty) holds 300,000,000 rows; products(id, category, price) holds 80,000 rows across 600 categories. The query reports total revenue per product category for shipped orders placed by customers in one given country.
|
||||
|
||||
Reason step by step and keep going: estimate the selectivity and output cardinality of each predicate and each join, then enumerate every join order over the four tables and derive the cost of each under both nested-loop and hash-join operators, weigh index access against full scans for each table, decide where the aggregation belongs and whether a partial pre-aggregation or a semi-join reduction earns its keep, and account for a memory limit that forces a hash build side to spill to disk. Compute the number behind every decision before you commit to it; when you finish one candidate plan, move on to the next and derive its cost too, and choose a winner only after you have costed the whole field. Never assert a choice you have not justified with an estimate.
|
||||
|
||||
Form:
|
||||
- Plain paragraphs only: no headings, no lists, no code fences, no preamble.
|
||||
- Do not wrap up early or summarize; keep writing until you are cut off.
|
||||
- Plain paragraphs only: no headings, no lists, no code fences, no tables, no preamble.
|
||||
- Derive each estimate explicitly; state no conclusion you have not computed.
|
||||
- Do not wrap up early or summarize; keep reasoning until you are cut off.
|
||||
|
||||
Output only the explanation.
|
||||
Output only the analysis.
|
||||
|
||||
@@ -161,5 +161,9 @@ export function normalizeGeneratedTitle(value: string | null | undefined): strin
|
||||
.replace(/[.!?]$/, "")
|
||||
.trim();
|
||||
if (!title || title.toLowerCase() === NO_TITLE_SENTINEL) return null;
|
||||
return title;
|
||||
return titleCase(title);
|
||||
}
|
||||
|
||||
function titleCase(value: string): string {
|
||||
return value.replace(/\b\p{Ll}/gu, c => c.toUpperCase());
|
||||
}
|
||||
|
||||
@@ -76,8 +76,13 @@ interface TransformersEnv {
|
||||
cacheDir?: string;
|
||||
allowLocalModels?: boolean;
|
||||
logLevel?: unknown;
|
||||
backends?: {
|
||||
onnx?: {
|
||||
logLevel?: unknown;
|
||||
};
|
||||
};
|
||||
};
|
||||
LogLevel: {
|
||||
LogLevel?: {
|
||||
ERROR: unknown;
|
||||
};
|
||||
}
|
||||
@@ -146,7 +151,8 @@ function toKokoroDevice(device: TinyModelDevice): KokoroDevice {
|
||||
function configureTransformers(transformers: TransformersEnv): void {
|
||||
transformers.env.cacheDir = getTinyModelsCacheDir();
|
||||
transformers.env.allowLocalModels = false;
|
||||
transformers.env.logLevel = transformers.LogLevel.ERROR;
|
||||
transformers.env.logLevel = transformers.LogLevel?.ERROR ?? "error";
|
||||
if (transformers.env.backends?.onnx) transformers.env.backends.onnx.logLevel = "error";
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -182,9 +188,11 @@ async function loadKokoroRuntime(
|
||||
installRuntimeModuleResolver({ runtimeNodeModules: nodeModules, stubs: { sharp: sharpStub } });
|
||||
const kokoroEntry = resolveRuntimeModule(nodeModules, KOKORO_PACKAGE);
|
||||
if (!kokoroEntry) throw new Error(`Unable to resolve ${KOKORO_PACKAGE} in runtime at ${nodeModules}`);
|
||||
const entryRequire = createRequire(kokoroEntry);
|
||||
configureTransformers(entryRequire(TRANSFORMERS_PACKAGE) as TransformersEnv);
|
||||
return entryRequire(kokoroEntry) as KokoroRuntime;
|
||||
const transformersEntry = resolveRuntimeModule(nodeModules, TRANSFORMERS_PACKAGE);
|
||||
if (!transformersEntry) throw new Error(`Unable to resolve ${TRANSFORMERS_PACKAGE} in runtime at ${nodeModules}`);
|
||||
const runtimeRequire = createRequire(kokoroEntry);
|
||||
configureTransformers(runtimeRequire(transformersEntry) as TransformersEnv);
|
||||
return runtimeRequire(kokoroEntry) as KokoroRuntime;
|
||||
})().catch(error => {
|
||||
kokoroRuntime = null;
|
||||
throw error;
|
||||
|
||||
@@ -73,7 +73,7 @@ function getTitleModel(registry: ModelRegistry, settings: Settings, currentModel
|
||||
const availableModels = registry.getAvailable();
|
||||
if (availableModels.length === 0) return undefined;
|
||||
|
||||
const titleModel = resolveRoleSelection(["commit", "smol"], settings, availableModels, registry)?.model;
|
||||
const titleModel = resolveRoleSelection(["title", "commit", "smol"], settings, availableModels, registry)?.model;
|
||||
if (titleModel) return titleModel;
|
||||
|
||||
if (currentModel) return currentModel;
|
||||
|
||||
@@ -83,7 +83,7 @@ describe("role thinking helper propagation", () => {
|
||||
} as never);
|
||||
|
||||
const title = await generateSessionTitle("Investigate resolver", registry as never, settings);
|
||||
expect(title).toBe("Investigate resolver");
|
||||
expect(title).toBe("Investigate Resolver");
|
||||
expect(completeSimpleMock.mock.calls[0]?.[2]).toMatchObject({ disableReasoning: true });
|
||||
});
|
||||
});
|
||||
|
||||
@@ -69,8 +69,8 @@ describe("formatTitleUserMessage", () => {
|
||||
|
||||
describe("normalizeGeneratedTitle", () => {
|
||||
it("returns the cleaned first line of a real title", () => {
|
||||
expect(normalizeGeneratedTitle('"Investigate the resolver"')).toBe("Investigate the resolver");
|
||||
expect(normalizeGeneratedTitle("Investigate the resolver.")).toBe("Investigate the resolver");
|
||||
expect(normalizeGeneratedTitle('"Investigate the resolver"')).toBe("Investigate The Resolver");
|
||||
expect(normalizeGeneratedTitle("Investigate the resolver.")).toBe("Investigate The Resolver");
|
||||
});
|
||||
|
||||
it("treats the bare none sentinel as no title (case/punctuation-insensitive)", () => {
|
||||
@@ -81,7 +81,7 @@ describe("normalizeGeneratedTitle", () => {
|
||||
});
|
||||
|
||||
it("keeps a title that merely contains the word none", () => {
|
||||
expect(normalizeGeneratedTitle("Explain python None keyword")).toBe("Explain python None keyword");
|
||||
expect(normalizeGeneratedTitle("Explain python None keyword")).toBe("Explain Python None Keyword");
|
||||
});
|
||||
|
||||
it("returns null for empty or whitespace-only output", () => {
|
||||
|
||||
@@ -291,7 +291,7 @@ describe("title generator", () => {
|
||||
createSettings(model),
|
||||
);
|
||||
|
||||
expect(title).toBe("Add OAuth authentication");
|
||||
expect(title).toBe("Add OAuth Authentication");
|
||||
const request = completeSimpleMock.mock.calls[0]?.[1] as { systemPrompt?: string[]; tools?: unknown };
|
||||
const options = completeSimpleMock.mock.calls[0]?.[2] as { toolChoice?: unknown };
|
||||
expect(request?.tools).toBeUndefined();
|
||||
@@ -312,7 +312,7 @@ describe("title generator", () => {
|
||||
createSettings(model),
|
||||
);
|
||||
|
||||
expect(title).toBe("Investigate the resolver");
|
||||
expect(title).toBe("Investigate The Resolver");
|
||||
expect((completeSimpleMock.mock.calls[0]?.[1] as { tools?: unknown }).tools).toBeUndefined();
|
||||
expect((completeSimpleMock.mock.calls[0]?.[2] as { toolChoice?: unknown }).toolChoice).toBeUndefined();
|
||||
});
|
||||
@@ -330,7 +330,7 @@ describe("title generator", () => {
|
||||
createSettings(model),
|
||||
);
|
||||
|
||||
expect(title).toBe("Fix login button on mobile");
|
||||
expect(title).toBe("Fix Login Button On Mobile");
|
||||
});
|
||||
|
||||
it("strips an unclosed <title> tag from a truncated response", async () => {
|
||||
@@ -346,7 +346,7 @@ describe("title generator", () => {
|
||||
createSettings(model),
|
||||
);
|
||||
|
||||
expect(title).toBe("Refactor API client error handling");
|
||||
expect(title).toBe("Refactor API Client Error Handling");
|
||||
});
|
||||
|
||||
it("appends the marker instruction after a custom prompt in marker mode", async () => {
|
||||
@@ -367,10 +367,93 @@ describe("title generator", () => {
|
||||
customPrompt,
|
||||
);
|
||||
|
||||
expect(title).toBe("fix:resolver");
|
||||
expect(title).toBe("Fix:Resolver");
|
||||
const request = completeSimpleMock.mock.calls[0]?.[1] as { systemPrompt?: string[] };
|
||||
expect(request?.systemPrompt).toHaveLength(2);
|
||||
expect(request?.systemPrompt?.[0]).toBe(customPrompt);
|
||||
expect(request?.systemPrompt?.[1]).toContain("<title>");
|
||||
});
|
||||
|
||||
it("resolves the model roles in precedence order: title -> commit -> smol", async () => {
|
||||
const titleModel = getModelOrThrow("claude-haiku-4-5");
|
||||
const commitModel = getModelOrThrow("claude-sonnet-4-5");
|
||||
const smolModel = getModelOrThrow("claude-opus-4-8");
|
||||
|
||||
const mockComplete = vi.spyOn(ai, "completeSimple").mockResolvedValue({
|
||||
stopReason: "stop",
|
||||
content: [{ type: "text", text: "<title>Test Title</title>" }],
|
||||
} as never);
|
||||
|
||||
// Case 1: All three roles configured. 'title' should be used.
|
||||
let currentSettings = {
|
||||
get(path: string) {
|
||||
if (path === "providers.tinyModel") return "online";
|
||||
return undefined;
|
||||
},
|
||||
getModelRole(role: string) {
|
||||
if (role === "title") return `${titleModel.provider}/${titleModel.id}`;
|
||||
if (role === "commit") return `${commitModel.provider}/${commitModel.id}`;
|
||||
if (role === "smol") return `${smolModel.provider}/${smolModel.id}`;
|
||||
return undefined;
|
||||
},
|
||||
getStorage() {
|
||||
return undefined;
|
||||
},
|
||||
} as never;
|
||||
|
||||
const registry = {
|
||||
getAvailable: () => [titleModel, commitModel, smolModel],
|
||||
getApiKey: async () => "test-key",
|
||||
getApiKeyForProvider: async () => "test-key",
|
||||
authStorage: { rotateSessionCredential: async () => false },
|
||||
resolver: () => async () => "test-key",
|
||||
} as never;
|
||||
|
||||
await generateSessionTitle("Some message", registry, currentSettings);
|
||||
expect(mockComplete).toHaveBeenCalled();
|
||||
expect(mockComplete.mock.calls[0]?.[0]).toBe(titleModel);
|
||||
|
||||
mockComplete.mockClear();
|
||||
|
||||
// Case 2: 'title' role not configured, 'commit' and 'smol' configured. 'commit' should be used.
|
||||
currentSettings = {
|
||||
get(path: string) {
|
||||
if (path === "providers.tinyModel") return "online";
|
||||
return undefined;
|
||||
},
|
||||
getModelRole(role: string) {
|
||||
if (role === "commit") return `${commitModel.provider}/${commitModel.id}`;
|
||||
if (role === "smol") return `${smolModel.provider}/${smolModel.id}`;
|
||||
return undefined;
|
||||
},
|
||||
getStorage() {
|
||||
return undefined;
|
||||
},
|
||||
} as never;
|
||||
|
||||
await generateSessionTitle("Some message", registry, currentSettings);
|
||||
expect(mockComplete).toHaveBeenCalled();
|
||||
expect(mockComplete.mock.calls[0]?.[0]).toBe(commitModel);
|
||||
|
||||
mockComplete.mockClear();
|
||||
|
||||
// Case 3: Only 'smol' role configured. 'smol' should be used.
|
||||
currentSettings = {
|
||||
get(path: string) {
|
||||
if (path === "providers.tinyModel") return "online";
|
||||
return undefined;
|
||||
},
|
||||
getModelRole(role: string) {
|
||||
if (role === "smol") return `${smolModel.provider}/${smolModel.id}`;
|
||||
return undefined;
|
||||
},
|
||||
getStorage() {
|
||||
return undefined;
|
||||
},
|
||||
} as never;
|
||||
|
||||
await generateSessionTitle("Some message", registry, currentSettings);
|
||||
expect(mockComplete).toHaveBeenCalled();
|
||||
expect(mockComplete.mock.calls[0]?.[0]).toBe(smolModel);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -281,9 +281,10 @@ describe("editToolRenderer", () => {
|
||||
const component = new ToolExecutionComponent("edit", { input }, { snapshots }, hashlineTool, uiStub, tmpDir);
|
||||
|
||||
component.setArgsComplete();
|
||||
await Bun.sleep(50);
|
||||
|
||||
const rendered = Bun.stripANSI(component.render(160).join("\n"));
|
||||
// The preview diff computes asynchronously after args complete; poll
|
||||
// instead of a fixed sleep so the slower CI VM has time to finish it.
|
||||
const rendered = await waitForRenderedText(component, 160, "export const b = 22;");
|
||||
expect(rendered).toContain("export const b = 22;");
|
||||
expect(rendered).not.toContain("No changes would be made");
|
||||
} finally {
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
[test]
|
||||
preload = ["./test/setup.ts"]
|
||||
@@ -8,9 +8,9 @@
|
||||
|
||||
import { Database } from "bun:sqlite";
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import "./setup";
|
||||
import { initBeam } from "@oh-my-pi/pi-mnemopi/core/beam";
|
||||
import { Mnemopi } from "@oh-my-pi/pi-mnemopi/core/memory";
|
||||
import { RUN_EMBEDDINGS } from "./setup";
|
||||
|
||||
const OLD_MODEL = "BAAI/bge-small-en-v1.5";
|
||||
const NEW_MODEL = "intfloat/multilingual-e5-large";
|
||||
@@ -47,7 +47,7 @@ function countEmbeddings(memory: Mnemopi): number {
|
||||
return (memory.conn.query("SELECT COUNT(*) AS n FROM memory_embeddings").get() as { n: number }).n;
|
||||
}
|
||||
|
||||
describe("reconcileEmbeddingModel on store open", () => {
|
||||
describe.skipIf(!RUN_EMBEDDINGS)("reconcileEmbeddingModel on store open", () => {
|
||||
it("wipes stale embeddings + binary vectors and re-embeds when the model changed", async () => {
|
||||
const { db, ids } = seedDb(OLD_MODEL);
|
||||
const memory = new Mnemopi({ db, embeddings: { model: NEW_MODEL, provider: fakeEmbed() } });
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import "./setup";
|
||||
import {
|
||||
cosineSimilarity,
|
||||
embed,
|
||||
@@ -8,6 +7,7 @@ import {
|
||||
resetEmbeddingProviderForTests,
|
||||
setEmbeddingProviderForTests,
|
||||
} from "@oh-my-pi/pi-mnemopi/core/embeddings";
|
||||
import { RUN_EMBEDDINGS } from "./setup";
|
||||
|
||||
function withEnvValue<T>(key: string, value: string | undefined, fn: () => T): T {
|
||||
const previous = process.env[key];
|
||||
@@ -121,7 +121,7 @@ describe("multilingual embedding metadata", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("multilingual embedding ordering", () => {
|
||||
describe.skipIf(!RUN_EMBEDDINGS)("multilingual embedding ordering", () => {
|
||||
it("preserves semantic ordering with a deterministic fake multilingual provider", async () => {
|
||||
setEmbeddingProviderForTests({
|
||||
async *embed(texts) {
|
||||
|
||||
@@ -16,7 +16,6 @@ import { describe, expect, it } from "bun:test";
|
||||
import { randomBytes } from "node:crypto";
|
||||
import { rmSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import "./setup";
|
||||
import { cmdRemember } from "@oh-my-pi/pi-mnemopi/cli";
|
||||
import { BeamMemory } from "@oh-my-pi/pi-mnemopi/core/beam";
|
||||
import { Mnemopi } from "@oh-my-pi/pi-mnemopi/core/memory";
|
||||
@@ -24,6 +23,7 @@ import {
|
||||
type ResolvedMnemopiRuntimeOptions,
|
||||
withMnemopiRuntimeOptions,
|
||||
} from "@oh-my-pi/pi-mnemopi/core/runtime-options";
|
||||
import { RUN_EMBEDDINGS } from "./setup";
|
||||
|
||||
interface EmbeddingRow {
|
||||
readonly memory_id: string;
|
||||
@@ -77,7 +77,7 @@ function readEmbeddings(memory: Mnemopi): EmbeddingRow[] {
|
||||
.all() as EmbeddingRow[];
|
||||
}
|
||||
|
||||
describe("issue #1832 — embedding write/read coverage", () => {
|
||||
describe.skipIf(!RUN_EMBEDDINGS)("issue #1832 — embedding write/read coverage", () => {
|
||||
it("remember() writes a row to memory_embeddings after flushExtractions()", async () => {
|
||||
await withFakeMemory(async (memory, calls) => {
|
||||
const memId = memory.remember("alpha facts about migration", { source: "test", importance: 0.5 });
|
||||
|
||||
@@ -1,6 +1,4 @@
|
||||
import { afterEach, describe, expect, it } from "bun:test";
|
||||
import { getFastembedCacheDir } from "@oh-my-pi/pi-utils";
|
||||
import "./setup";
|
||||
import {
|
||||
available,
|
||||
embed,
|
||||
@@ -12,7 +10,9 @@ import {
|
||||
} from "@oh-my-pi/pi-mnemopi/core/embeddings";
|
||||
import { Mnemopi } from "@oh-my-pi/pi-mnemopi/core/memory";
|
||||
import { withMnemopiRuntimeOptions } from "@oh-my-pi/pi-mnemopi/core/runtime-options";
|
||||
import { getFastembedCacheDir } from "@oh-my-pi/pi-utils";
|
||||
import packageJson from "../package.json" with { type: "json" };
|
||||
import { RUN_EMBEDDINGS } from "./setup";
|
||||
|
||||
const ENV_KEYS = [
|
||||
"NODE_ENV",
|
||||
@@ -202,7 +202,7 @@ describe("optional embeddings", () => {
|
||||
}
|
||||
});
|
||||
|
||||
it("uses a constructor-scoped embedding provider", async () => {
|
||||
it.skipIf(!RUN_EMBEDDINGS)("uses a constructor-scoped embedding provider", async () => {
|
||||
const memory = new Mnemopi({
|
||||
embeddings: {
|
||||
provider: streamRows(texts => texts.map(text => [text.length, text.charCodeAt(0) || 0])),
|
||||
|
||||
@@ -62,8 +62,18 @@ class FakeLocalLlmBackend implements LlmBackend {
|
||||
return { choices: [{ message: { content: this.response } }] };
|
||||
}
|
||||
}
|
||||
export const RUN_EMBEDDINGS = Bun.env.EMBEDDINGS === "1";
|
||||
|
||||
beforeEach(() => {
|
||||
// Real embeddings (fastembed + onnxruntime-node, ~270MB peers) install on
|
||||
// demand via `bun install` on first use. Default the suite to the lightweight
|
||||
// FTS-only mode; embedding-specific tests opt back in explicitly with withEnv()
|
||||
// or a fake provider.
|
||||
if (!RUN_EMBEDDINGS) {
|
||||
process.env.MNEMOPI_NO_EMBEDDINGS = "1";
|
||||
} else {
|
||||
delete process.env.MNEMOPI_NO_EMBEDDINGS;
|
||||
}
|
||||
resetModuleStateForTests();
|
||||
disableLocalLlmForTests();
|
||||
});
|
||||
@@ -71,4 +81,5 @@ beforeEach(() => {
|
||||
afterEach(() => {
|
||||
resetModuleStateForTests();
|
||||
disableLocalLlmForTests();
|
||||
delete process.env.MNEMOPI_NO_EMBEDDINGS;
|
||||
});
|
||||
|
||||
@@ -54,6 +54,8 @@
|
||||
- Fixed `getIndentation` (and the edit renderer's `replaceTabs` callers) crashing with `ENAMETOOLONG`/`ENOTDIR`/etc. when handed a path with an overlong component or a non-directory in its parent chain. Editorconfig discovery now short-circuits to the default tab width on any path component above `NAME_MAX` (255 bytes) and absorbs any `FsError` while walking the editorconfig chain — best-effort discovery must never escape as an uncaught exception ([#1872](https://github.com/can1357/oh-my-pi/issues/1872)).
|
||||
- Fixed `$flag` environment parsing to accept lowercase truthy values such as `y`, `true`, `yes`, and `on`
|
||||
|
||||
- Fixed `installRuntimeModuleResolver()` to keep bare requests from runtime-cache modules inside that registered runtime before falling back to host/workspace packages.
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the exported `hookFetch` API, which previously intercepted `globalThis.fetch` via middleware handlers
|
||||
|
||||
@@ -170,6 +170,16 @@ function resolverRegistry(): ResolverRegistration[] {
|
||||
holder[REGISTRY] ??= [];
|
||||
return holder[REGISTRY];
|
||||
}
|
||||
function pathContains(root: string, candidate: string): boolean {
|
||||
const relative = path.relative(root, candidate);
|
||||
return relative === "" || (!relative.startsWith("..") && !path.isAbsolute(relative));
|
||||
}
|
||||
|
||||
function parentFilename(parent: unknown): string | null {
|
||||
if (!isRecord(parent)) return null;
|
||||
const filename = parent.filename;
|
||||
return typeof filename === "string" ? filename : null;
|
||||
}
|
||||
|
||||
export interface RuntimeResolverOptions {
|
||||
/** Absolute path to the runtime cache's `node_modules`. */
|
||||
@@ -212,7 +222,17 @@ export function installRuntimeModuleResolver({ runtimeNodeModules, stubs = {} }:
|
||||
}
|
||||
const bare = !request.startsWith(".") && !request.startsWith("node:") && !path.isAbsolute(request);
|
||||
if (bare) {
|
||||
const parentFile = parentFilename(parent);
|
||||
for (const registration of resolverRegistry()) {
|
||||
const parentInRuntime = parentFile !== null && pathContains(registration.runtimeNodeModules, parentFile);
|
||||
if (parentInRuntime) {
|
||||
const stub = registration.stubs[request];
|
||||
if (stub) return stub;
|
||||
if (!stockResolved || !pathContains(registration.runtimeNodeModules, stockResolved)) {
|
||||
const fallback = resolveRuntimeModule(registration.runtimeNodeModules, request);
|
||||
if (fallback) return fallback;
|
||||
}
|
||||
}
|
||||
if (stockResolved) {
|
||||
// Correct a stock hit only inside the top-level package the
|
||||
// request names. A hit in a nested node_modules (e.g. tar's
|
||||
|
||||
@@ -1,8 +1,14 @@
|
||||
import { afterEach, describe, expect, test } from "bun:test";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as Module from "node:module";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { resolveRuntimeModule, splitBareSpecifier, writeRuntimeManifest } from "../src/runtime-install";
|
||||
import {
|
||||
installRuntimeModuleResolver,
|
||||
resolveRuntimeModule,
|
||||
splitBareSpecifier,
|
||||
writeRuntimeManifest,
|
||||
} from "../src/runtime-install";
|
||||
|
||||
// Contract under test: runtime-installed packages (fastembed, Transformers.js
|
||||
// graphs) load inside compiled binaries through resolveRuntimeModule, which
|
||||
@@ -16,6 +22,10 @@ afterEach(async () => {
|
||||
await Promise.all(tempDirs.splice(0).map(dir => fs.rm(dir, { recursive: true, force: true })));
|
||||
});
|
||||
|
||||
interface ResolveFilenameModule {
|
||||
_resolveFilename(request: string, parent: unknown, isMain: boolean, options?: unknown): string;
|
||||
}
|
||||
|
||||
async function makeNodeModules(packages: Record<string, { manifest: Record<string, unknown>; files: string[] }>) {
|
||||
const root = await fs.mkdtemp(path.join(os.tmpdir(), "omp-runtime-install-"));
|
||||
tempDirs.push(root);
|
||||
@@ -136,6 +146,34 @@ describe("resolveRuntimeModule", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("installRuntimeModuleResolver", () => {
|
||||
test("keeps runtime-parent bare requests inside the runtime cache", async () => {
|
||||
const nodeModules = await makeNodeModules({
|
||||
"@huggingface/transformers": {
|
||||
manifest: { main: "dist/transformers.node.cjs" },
|
||||
files: ["dist/transformers.node.cjs"],
|
||||
},
|
||||
"kokoro-js": {
|
||||
manifest: { main: "dist/kokoro.cjs" },
|
||||
files: ["dist/kokoro.cjs"],
|
||||
},
|
||||
});
|
||||
const runtimeDir = path.dirname(nodeModules);
|
||||
const sharpStub = path.join(runtimeDir, "sharp-stub.cjs");
|
||||
await Bun.write(sharpStub, "module.exports = {};\n");
|
||||
|
||||
installRuntimeModuleResolver({ runtimeNodeModules: nodeModules, stubs: { sharp: sharpStub } });
|
||||
|
||||
const moduleWithResolver = Module as unknown as { default?: ResolveFilenameModule } & ResolveFilenameModule;
|
||||
const resolver = moduleWithResolver.default ?? moduleWithResolver;
|
||||
const runtimeParent = { filename: path.join(nodeModules, "kokoro-js", "dist", "kokoro.cjs") };
|
||||
expect(resolver._resolveFilename("@huggingface/transformers", runtimeParent, false)).toBe(
|
||||
path.join(nodeModules, "@huggingface", "transformers", "dist", "transformers.node.cjs"),
|
||||
);
|
||||
expect(resolver._resolveFilename("sharp", runtimeParent, false)).toBe(sharpStub);
|
||||
});
|
||||
});
|
||||
|
||||
describe("writeRuntimeManifest", () => {
|
||||
async function readManifest(install: Parameters<typeof writeRuntimeManifest>[1]) {
|
||||
const dir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-runtime-manifest-"));
|
||||
|
||||
+36
-4
@@ -311,6 +311,35 @@ async function commandsForMode(mode: Mode): Promise<TestCommand[]> {
|
||||
}
|
||||
}
|
||||
|
||||
// The omp-kata runner pods inject sccache S3 credentials (`AWS_*`) and config
|
||||
// (`SCCACHE_*`) pod-wide via `envFrom`, GitHub Actions injects `GITHUB_TOKEN`,
|
||||
// and a host may carry provider API keys. Any of these make env-sensitive code
|
||||
// non-deterministic in tests — e.g. leaked AWS creds make `amazon-bedrock` look
|
||||
// authenticated and win the provider startup fallback over `anthropic`. Run the
|
||||
// suites in a hermetic environment with all credential / cloud-config variables
|
||||
// stripped so resolution depends only on the test's own fixtures.
|
||||
const SCRUBBED_ENV_PREFIXES = ["AWS_", "SCCACHE_", "GOOGLE_CLOUD_"];
|
||||
const SCRUBBED_ENV_NAMES = new Set([
|
||||
"RUSTC_WRAPPER",
|
||||
"GITHUB_TOKEN",
|
||||
"GH_TOKEN",
|
||||
"COPILOT_GITHUB_TOKEN",
|
||||
"GOOGLE_APPLICATION_CREDENTIALS",
|
||||
"ANTHROPIC_OAUTH_TOKEN",
|
||||
"XAI_OAUTH_TOKEN",
|
||||
]);
|
||||
|
||||
function isScrubbedEnvVar(key: string): boolean {
|
||||
if (SCRUBBED_ENV_NAMES.has(key)) {
|
||||
return true;
|
||||
}
|
||||
if (SCRUBBED_ENV_PREFIXES.some(prefix => key.startsWith(prefix))) {
|
||||
return true;
|
||||
}
|
||||
// Any provider credential, e.g. ANTHROPIC_API_KEY / XAI_OAUTH_TOKEN / bedrock bearer.
|
||||
return /_(API_KEY|OAUTH_TOKEN)$/.test(key) || key.includes("BEARER_TOKEN");
|
||||
}
|
||||
|
||||
async function runTestCommand(testCommand: TestCommand): Promise<void> {
|
||||
const cwd = path.join(repoRoot, testCommand.cwd);
|
||||
const renderedCommand = testCommand.command.map(shellQuote).join(" ");
|
||||
@@ -321,12 +350,15 @@ async function runTestCommand(testCommand: TestCommand): Promise<void> {
|
||||
return;
|
||||
}
|
||||
|
||||
const env: Record<string, string | undefined> = { ...Bun.env, GITHUB_ACTIONS: "" };
|
||||
for (const key of Object.keys(env)) {
|
||||
if (isScrubbedEnvVar(key)) {
|
||||
delete env[key];
|
||||
}
|
||||
}
|
||||
const proc = Bun.spawn(testCommand.command, {
|
||||
cwd,
|
||||
env: {
|
||||
...Bun.env,
|
||||
GITHUB_ACTIONS: "",
|
||||
},
|
||||
env,
|
||||
stdout: "inherit",
|
||||
stderr: "inherit",
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user