Skip to content

chore(agents): bump codex to 0.154.0 #617

chore(agents): bump codex to 0.154.0

chore(agents): bump codex to 0.154.0 #617

Workflow file for this run

name: E2E Tests
on:
push:
branches: [ main ]
pull_request:
branches: [ main ]
env:
CARGO_TERM_COLOR: always
RUST_BACKTRACE: 1
# Read-only token: this workflow only runs tests; nothing it does
# requires write access. Declared explicitly so a future change to the
# org-level default workflow permissions can't silently widen it.
permissions:
contents: read
jobs:
e2e:
name: E2E KVM (${{ matrix.name }})
runs-on: ${{ matrix.runner }}
timeout-minutes: 60
strategy:
fail-fast: false
matrix:
include:
- name: Ubuntu x86_64
runner: ubuntu-latest
musl-target: x86_64-unknown-linux-musl
steps:
- uses: actions/checkout@v4
# Reclaim ~28 GB by removing GitHub's preinstalled Android SDK,
# .NET, Haskell, Docker images, and swapfile. Required because the
# default ubuntu-latest image leaves only ~14 GB free, and the
# combined cargo target + initramfs + kernel copy + multiple
# `cargo test --test <name>` rebuilds blow past that ceiling
# (every E2E run on 2026-04-28 failed with "No space left on
# device" until this step was added). `large-packages: false`
# skips the slow apt-remove path; the Android delete alone
# buys ~14 GB in seconds.
- name: Free runner disk space
uses: jlumbroso/free-disk-space@54081f138730dfa15788a46383842cd2f914a1be # v1.3.1
with:
android: true
dotnet: true
haskell: true
large-packages: false
docker-images: true
swap-storage: true
tool-cache: false
# ---- KVM setup ----
- name: Enable KVM access
run: |
# Make /dev/kvm and /dev/vhost-vsock accessible to the runner user.
# Without chmod on /dev/vhost-vsock the vsock-preflight silently
# skips every KVM e2e test — masking real failures.
sudo chmod a+rw /dev/kvm
sudo modprobe vhost_vsock || true
sudo chmod a+rw /dev/vhost-vsock 2>/dev/null || true
ls -la /dev/kvm
ls -la /dev/vhost-vsock 2>/dev/null || echo "/dev/vhost-vsock not present (may be built-in)"
# ---- System dependencies ----
- name: Install system packages
# apt on a hosted runner can hang; cap this step so an infra stall fails
# fast instead of consuming the whole job budget and looking like a test
# cancellation.
timeout-minutes: 10
run: |
# apt on a hosted runner can block indefinitely and silently: the
# periodic unattended-upgrades job holds the dpkg lock, and apt
# without DPkg::Lock::Timeout waits on it forever, printing nothing —
# the step then dies at its cap with an empty log. Bound the lock
# wait, cap each call, and retry, so one stall costs seconds instead
# of the whole budget. Observed failing 3 runs in 5 without this.
apt_get() {
for attempt in 1 2 3; do
if sudo timeout 150 apt-get -o DPkg::Lock::Timeout=60 "$@"; then
return 0
fi
echo "::warning::apt-get $1 attempt $attempt stalled or failed; retrying"
sleep 5
done
echo "::error::apt-get $1 failed after 3 attempts"
return 1
}
apt_get update -qq
# busybox-static is REQUIRED — `scripts/build_test_image.sh`
# silently builds an initramfs with no `/bin/sh` when BUSYBOX
# is unset (see `scripts/lib/guest_common.sh::install_busybox`).
# Guests without `/bin/sh`+`ip` fail in two ways:
# - `Command::new("ip")` from guest-agent's setup_network()
# hangs PID 1 (see AGENTS.md "Known issues" — vsock
# control-channel timeout) → persistent_channel handshake
# deadline.
# - `vm.exec("echo", …)` returns ENOENT in the guest agent
# → snapshot suite asserts fail, pty `sh -c "exit 42"`
# hits the execvp(127) child path.
# All three failures share this single missing dep.
apt_get install -y -qq cpio gzip zstd musl-tools busybox-static
# `truncate` and `mkfs.ext4` back the OCI block-rootfs path: the
# oci_integration suite shells out to both to build the ext4 image it
# attaches as /dev/vda. Assert rather than install. Their packages
# (coreutils, e2fsprogs) are preinstalled, so installing buys nothing —
# and naming coreutils is actively harmful: it is an Essential package,
# and any apt plan touching one demands an interactive
# "Yes, do as I say!" that -y does not satisfy, so the step hangs
# silently under -qq until its timeout. A presence check gives the same
# protection against a slimmer runner with no apt involvement.
for tool in truncate mkfs.ext4; do
command -v "$tool" >/dev/null \
|| { echo "::error::$tool is missing; oci_integration cannot build its ext4 rootfs"; exit 1; }
done
# Ensure kernel modules are available for the running kernel
apt_get install -y -qq linux-modules-$(uname -r) || true
# Resolve the installed busybox path and export it so the build
# step (and any future step) gets `BUSYBOX=<actual path>` from
# the workflow env. Avoids mutating system paths via symlinks
# and works regardless of whether the package lands the binary
# at /bin/busybox or /usr/bin/busybox.
BUSYBOX_PATH="$(command -v busybox)"
test -n "$BUSYBOX_PATH" \
|| { echo "::error::busybox not on PATH after busybox-static install"; exit 1; }
echo "BUSYBOX=$BUSYBOX_PATH" >> "$GITHUB_ENV"
# ---- Rust toolchain ----
- name: Install Rust stable + musl target
uses: dtolnay/rust-toolchain@stable
with:
targets: ${{ matrix.musl-target }}
# cargo-nextest runs the VM suites with bounded cross-suite parallelism
# (the `vm` test group in .config/nextest.toml), replacing the strictly
# sequential `cargo test --test-threads=1` lane.
- name: Install cargo-nextest
uses: taiki-e/install-action@82cd3e7658a6f96c86c0234aeeda1748937cb0a1 # v2
with:
tool: nextest
# ---- Cargo cache ----
# Swatinem/rust-cache is Rust-aware: it caches `~/.cargo/registry`
# and `~/.cargo/git` whole, but prunes `target/` to drop workspace
# crates and incremental build artifacts before saving — which is
# exactly the unbounded-growth portion that filled the disk on the
# raw `actions/cache@v4` setup. Cache key is auto-derived from
# Cargo.lock + rustc version + job + matrix.
- name: Cache cargo registry + build
uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1
with:
shared-key: e2e-${{ matrix.musl-target }}
# The required lane makes three anonymous registry pulls: alpine:3.20 twice
# from Docker Hub and the guest image once from ghcr. Docker Hub rate-limits
# per source IP, which hosted runners share, and `retries = 2` in nextest
# turns one 429 into three requests.
#
# This cache covers exactly one of the three. Only `vm_oci_alpine_os_release`
# resolves through the shared `~/.voidbox/oci`; the other two
# (`oci_client_resolves_alpine_rootfs_into_cache` and
# `guest_image_pull_and_extract`) each build a client on a fresh TempDir by
# design, so their pulls stay cold on every run and the ~115 MB ghcr pull is
# not mitigated at all. `restore-keys` lets a bumped image tag start from
# the previous entry instead of cold.
- name: Cache extracted OCI images
uses: actions/cache@v4
with:
path: ~/.voidbox/oci
key: oci-${{ runner.os }}-${{ runner.arch }}-alpine3.20-${{ github.run_id }}
restore-keys: |
oci-${{ runner.os }}-${{ runner.arch }}-alpine3.20-
# ---- Build test initramfs ----
- name: Build test initramfs (guest-agent + claudio)
# `install_busybox` (scripts/lib/guest_common.sh) requires the
# BUSYBOX env var to be set — file existence alone is not
# enough. Without it the build silently produces an initramfs
# with no `/bin/sh`, see commit 004190b. The earlier
# "Install system packages" step writes BUSYBOX=$(command -v
# busybox) to $GITHUB_ENV so this step inherits the resolved
# path automatically.
run: scripts/build_test_image.sh
# ---- Unit + integration tests ----
- name: Run unit tests
run: cargo test --workspace --all-features --verbose
# ---- E2E KVM tests ----
- name: Run E2E KVM tests
# Was ~7 sequential suites plus a separate snapshot step; now a single
# nextest run. The whole VM lane finishes well inside this budget when
# healthy, and the timeout keeps a hung boot from consuming the job.
timeout-minutes: 30
env:
VOID_BOX_KERNEL: /boot/vmlinuz-${{ env.KERNEL_VERSION }}
VOID_BOX_INITRAMFS: /tmp/void-box-test-rootfs.cpio.gz
# This runner is meant to boot VMs: past the vsock early-bail below,
# an incapable machine must fail, not skip silently.
VOID_BOX_REQUIRE_VM: "1"
# oci_integration's guest_image_pull_and_extract defaults to a
# localhost:5555 registry and skips when none is running, which is
# every machine. Pointing it at the published guest image turns that
# skip into real coverage of resolve_guest_files (pull, extract
# vmlinuz + rootfs.cpio.gz, second-call cache hit). A version tag
# rather than `latest` so the lane does not follow every publish, but
# a tag is mutable — repointing v0.2.0 changes what this pulls. Pin by
# digest if that becomes a problem. This and the alpine:3.20 pull in
# the same suite put a registry dependency on the required gate: a
# registry outage or rate limit fails the lane for reasons unrelated
# to the change under review.
VOIDBOX_TEST_GUEST_IMAGE: ghcr.io/the-void-ia/voidbox-guest:v0.2.0
# #149: multiplex establishment intermittently exceeds the 30 s default
# on this runner class, and doubling the suites doubles the draws. The
# knob only ever lengthens the deadline (never shortens it), so a
# healthy boot is unaffected and a slow one stops reading as a step
# cancellation.
VOID_BOX_CONNECT_DEADLINE_SECS: "120"
run: |
# Do not exit green when /dev/vhost-vsock is missing: VOID_BOX_REQUIRE_VM=1
# asserts this runner is capable, so a lost device must fail the suites
# that need it, not skip. snapshot_integration uses userspace vsock and
# runs regardless. Warn for the log, then proceed.
if [ ! -e /dev/vhost-vsock ]; then
echo "::warning::/dev/vhost-vsock is not available; vhost-vsock suites will fail (REQUIRE_VM=1)."
fi
# Detect kernel version at runtime (env context can't run commands)
KERNEL_SRC="/boot/vmlinuz-$(uname -r)"
KERNEL_COPY="/tmp/void-box-kernel-$(uname -r)"
sudo cp "$KERNEL_SRC" "$KERNEL_COPY"
sudo chown "$USER:$USER" "$KERNEL_COPY"
chmod 644 "$KERNEL_COPY"
export VOID_BOX_KERNEL="$KERNEL_COPY"
echo "Kernel: $VOID_BOX_KERNEL"
echo "Initramfs: $VOID_BOX_INITRAMFS"
# Verify artifacts exist
ls -la "$VOID_BOX_KERNEL"
ls -la "$VOID_BOX_INITRAMFS"
# One nextest run for every deterministic VM suite: the backend
# contract, direct MicroVm/Sandbox boots, the OCI root switch and its
# read-only invariant, host-directory mounts, the guest→host telemetry
# stream and sidecar gateway, the PTY and credential-proxy paths, the
# persistent control channel, and snapshot/restore. The `vm` test group
# (.config/nextest.toml) caps how many VMs boot at once, so the suites
# run with bounded cross-suite parallelism rather than serially, and
# --no-fail-fast collects every result instead of stopping at the first
# failure.
#
# The exclusions, each deliberate:
#
# pty_command_not_allowed asserts the guest rejects a non-allowlisted
# program, but the allowlist loads at guest boot from a file only the
# agent-box provisioning path writes, which this direct attach_pty path
# never stages — so the guest allows all commands (#99).
#
# The agent suites (e2e_service_mode, e2e_agent_mcp) need a real agent
# credential, which no pull-request runner holds, so here they would
# detect its absence and report green without starting an agent. They
# run in `e2e-agent.yml` instead, under VOID_BOX_REQUIRE_AGENT_CREDS=1,
# and on contributors' machines.
cargo nextest run --run-ignored only --no-fail-fast -E \
'binary(/^(conformance|telemetry|kvm_integration|oci_integration|e2e_sidecar|e2e_telemetry|e2e_mount|persistent_channel|e2e_skill_pipeline|e2e_pty|e2e_credential_proxy|snapshot_integration)$/) & not test(pty_command_not_allowed)'