chore(agents): bump codex to 0.154.0 #617
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: E2E Tests | |
| on: | |
| push: | |
| branches: [ main ] | |
| pull_request: | |
| branches: [ main ] | |
| env: | |
| CARGO_TERM_COLOR: always | |
| RUST_BACKTRACE: 1 | |
| # Read-only token: this workflow only runs tests; nothing it does | |
| # requires write access. Declared explicitly so a future change to the | |
| # org-level default workflow permissions can't silently widen it. | |
| permissions: | |
| contents: read | |
| jobs: | |
| e2e: | |
| name: E2E KVM (${{ matrix.name }}) | |
| runs-on: ${{ matrix.runner }} | |
| timeout-minutes: 60 | |
| strategy: | |
| fail-fast: false | |
| matrix: | |
| include: | |
| - name: Ubuntu x86_64 | |
| runner: ubuntu-latest | |
| musl-target: x86_64-unknown-linux-musl | |
| steps: | |
| - uses: actions/checkout@v4 | |
| # Reclaim ~28 GB by removing GitHub's preinstalled Android SDK, | |
| # .NET, Haskell, Docker images, and swapfile. Required because the | |
| # default ubuntu-latest image leaves only ~14 GB free, and the | |
| # combined cargo target + initramfs + kernel copy + multiple | |
| # `cargo test --test <name>` rebuilds blow past that ceiling | |
| # (every E2E run on 2026-04-28 failed with "No space left on | |
| # device" until this step was added). `large-packages: false` | |
| # skips the slow apt-remove path; the Android delete alone | |
| # buys ~14 GB in seconds. | |
| - name: Free runner disk space | |
| uses: jlumbroso/free-disk-space@54081f138730dfa15788a46383842cd2f914a1be # v1.3.1 | |
| with: | |
| android: true | |
| dotnet: true | |
| haskell: true | |
| large-packages: false | |
| docker-images: true | |
| swap-storage: true | |
| tool-cache: false | |
| # ---- KVM setup ---- | |
| - name: Enable KVM access | |
| run: | | |
| # Make /dev/kvm and /dev/vhost-vsock accessible to the runner user. | |
| # Without chmod on /dev/vhost-vsock the vsock-preflight silently | |
| # skips every KVM e2e test — masking real failures. | |
| sudo chmod a+rw /dev/kvm | |
| sudo modprobe vhost_vsock || true | |
| sudo chmod a+rw /dev/vhost-vsock 2>/dev/null || true | |
| ls -la /dev/kvm | |
| ls -la /dev/vhost-vsock 2>/dev/null || echo "/dev/vhost-vsock not present (may be built-in)" | |
| # ---- System dependencies ---- | |
| - name: Install system packages | |
| # apt on a hosted runner can hang; cap this step so an infra stall fails | |
| # fast instead of consuming the whole job budget and looking like a test | |
| # cancellation. | |
| timeout-minutes: 10 | |
| run: | | |
| # apt on a hosted runner can block indefinitely and silently: the | |
| # periodic unattended-upgrades job holds the dpkg lock, and apt | |
| # without DPkg::Lock::Timeout waits on it forever, printing nothing — | |
| # the step then dies at its cap with an empty log. Bound the lock | |
| # wait, cap each call, and retry, so one stall costs seconds instead | |
| # of the whole budget. Observed failing 3 runs in 5 without this. | |
| apt_get() { | |
| for attempt in 1 2 3; do | |
| if sudo timeout 150 apt-get -o DPkg::Lock::Timeout=60 "$@"; then | |
| return 0 | |
| fi | |
| echo "::warning::apt-get $1 attempt $attempt stalled or failed; retrying" | |
| sleep 5 | |
| done | |
| echo "::error::apt-get $1 failed after 3 attempts" | |
| return 1 | |
| } | |
| apt_get update -qq | |
| # busybox-static is REQUIRED — `scripts/build_test_image.sh` | |
| # silently builds an initramfs with no `/bin/sh` when BUSYBOX | |
| # is unset (see `scripts/lib/guest_common.sh::install_busybox`). | |
| # Guests without `/bin/sh`+`ip` fail in two ways: | |
| # - `Command::new("ip")` from guest-agent's setup_network() | |
| # hangs PID 1 (see AGENTS.md "Known issues" — vsock | |
| # control-channel timeout) → persistent_channel handshake | |
| # deadline. | |
| # - `vm.exec("echo", …)` returns ENOENT in the guest agent | |
| # → snapshot suite asserts fail, pty `sh -c "exit 42"` | |
| # hits the execvp(127) child path. | |
| # All three failures share this single missing dep. | |
| apt_get install -y -qq cpio gzip zstd musl-tools busybox-static | |
| # `truncate` and `mkfs.ext4` back the OCI block-rootfs path: the | |
| # oci_integration suite shells out to both to build the ext4 image it | |
| # attaches as /dev/vda. Assert rather than install. Their packages | |
| # (coreutils, e2fsprogs) are preinstalled, so installing buys nothing — | |
| # and naming coreutils is actively harmful: it is an Essential package, | |
| # and any apt plan touching one demands an interactive | |
| # "Yes, do as I say!" that -y does not satisfy, so the step hangs | |
| # silently under -qq until its timeout. A presence check gives the same | |
| # protection against a slimmer runner with no apt involvement. | |
| for tool in truncate mkfs.ext4; do | |
| command -v "$tool" >/dev/null \ | |
| || { echo "::error::$tool is missing; oci_integration cannot build its ext4 rootfs"; exit 1; } | |
| done | |
| # Ensure kernel modules are available for the running kernel | |
| apt_get install -y -qq linux-modules-$(uname -r) || true | |
| # Resolve the installed busybox path and export it so the build | |
| # step (and any future step) gets `BUSYBOX=<actual path>` from | |
| # the workflow env. Avoids mutating system paths via symlinks | |
| # and works regardless of whether the package lands the binary | |
| # at /bin/busybox or /usr/bin/busybox. | |
| BUSYBOX_PATH="$(command -v busybox)" | |
| test -n "$BUSYBOX_PATH" \ | |
| || { echo "::error::busybox not on PATH after busybox-static install"; exit 1; } | |
| echo "BUSYBOX=$BUSYBOX_PATH" >> "$GITHUB_ENV" | |
| # ---- Rust toolchain ---- | |
| - name: Install Rust stable + musl target | |
| uses: dtolnay/rust-toolchain@stable | |
| with: | |
| targets: ${{ matrix.musl-target }} | |
| # cargo-nextest runs the VM suites with bounded cross-suite parallelism | |
| # (the `vm` test group in .config/nextest.toml), replacing the strictly | |
| # sequential `cargo test --test-threads=1` lane. | |
| - name: Install cargo-nextest | |
| uses: taiki-e/install-action@82cd3e7658a6f96c86c0234aeeda1748937cb0a1 # v2 | |
| with: | |
| tool: nextest | |
| # ---- Cargo cache ---- | |
| # Swatinem/rust-cache is Rust-aware: it caches `~/.cargo/registry` | |
| # and `~/.cargo/git` whole, but prunes `target/` to drop workspace | |
| # crates and incremental build artifacts before saving — which is | |
| # exactly the unbounded-growth portion that filled the disk on the | |
| # raw `actions/cache@v4` setup. Cache key is auto-derived from | |
| # Cargo.lock + rustc version + job + matrix. | |
| - name: Cache cargo registry + build | |
| uses: Swatinem/rust-cache@c19371144df3bb44fab255c43d04cbc2ab54d1c4 # v2.9.1 | |
| with: | |
| shared-key: e2e-${{ matrix.musl-target }} | |
| # The required lane makes three anonymous registry pulls: alpine:3.20 twice | |
| # from Docker Hub and the guest image once from ghcr. Docker Hub rate-limits | |
| # per source IP, which hosted runners share, and `retries = 2` in nextest | |
| # turns one 429 into three requests. | |
| # | |
| # This cache covers exactly one of the three. Only `vm_oci_alpine_os_release` | |
| # resolves through the shared `~/.voidbox/oci`; the other two | |
| # (`oci_client_resolves_alpine_rootfs_into_cache` and | |
| # `guest_image_pull_and_extract`) each build a client on a fresh TempDir by | |
| # design, so their pulls stay cold on every run and the ~115 MB ghcr pull is | |
| # not mitigated at all. `restore-keys` lets a bumped image tag start from | |
| # the previous entry instead of cold. | |
| - name: Cache extracted OCI images | |
| uses: actions/cache@v4 | |
| with: | |
| path: ~/.voidbox/oci | |
| key: oci-${{ runner.os }}-${{ runner.arch }}-alpine3.20-${{ github.run_id }} | |
| restore-keys: | | |
| oci-${{ runner.os }}-${{ runner.arch }}-alpine3.20- | |
| # ---- Build test initramfs ---- | |
| - name: Build test initramfs (guest-agent + claudio) | |
| # `install_busybox` (scripts/lib/guest_common.sh) requires the | |
| # BUSYBOX env var to be set — file existence alone is not | |
| # enough. Without it the build silently produces an initramfs | |
| # with no `/bin/sh`, see commit 004190b. The earlier | |
| # "Install system packages" step writes BUSYBOX=$(command -v | |
| # busybox) to $GITHUB_ENV so this step inherits the resolved | |
| # path automatically. | |
| run: scripts/build_test_image.sh | |
| # ---- Unit + integration tests ---- | |
| - name: Run unit tests | |
| run: cargo test --workspace --all-features --verbose | |
| # ---- E2E KVM tests ---- | |
| - name: Run E2E KVM tests | |
| # Was ~7 sequential suites plus a separate snapshot step; now a single | |
| # nextest run. The whole VM lane finishes well inside this budget when | |
| # healthy, and the timeout keeps a hung boot from consuming the job. | |
| timeout-minutes: 30 | |
| env: | |
| VOID_BOX_KERNEL: /boot/vmlinuz-${{ env.KERNEL_VERSION }} | |
| VOID_BOX_INITRAMFS: /tmp/void-box-test-rootfs.cpio.gz | |
| # This runner is meant to boot VMs: past the vsock early-bail below, | |
| # an incapable machine must fail, not skip silently. | |
| VOID_BOX_REQUIRE_VM: "1" | |
| # oci_integration's guest_image_pull_and_extract defaults to a | |
| # localhost:5555 registry and skips when none is running, which is | |
| # every machine. Pointing it at the published guest image turns that | |
| # skip into real coverage of resolve_guest_files (pull, extract | |
| # vmlinuz + rootfs.cpio.gz, second-call cache hit). A version tag | |
| # rather than `latest` so the lane does not follow every publish, but | |
| # a tag is mutable — repointing v0.2.0 changes what this pulls. Pin by | |
| # digest if that becomes a problem. This and the alpine:3.20 pull in | |
| # the same suite put a registry dependency on the required gate: a | |
| # registry outage or rate limit fails the lane for reasons unrelated | |
| # to the change under review. | |
| VOIDBOX_TEST_GUEST_IMAGE: ghcr.io/the-void-ia/voidbox-guest:v0.2.0 | |
| # #149: multiplex establishment intermittently exceeds the 30 s default | |
| # on this runner class, and doubling the suites doubles the draws. The | |
| # knob only ever lengthens the deadline (never shortens it), so a | |
| # healthy boot is unaffected and a slow one stops reading as a step | |
| # cancellation. | |
| VOID_BOX_CONNECT_DEADLINE_SECS: "120" | |
| run: | | |
| # Do not exit green when /dev/vhost-vsock is missing: VOID_BOX_REQUIRE_VM=1 | |
| # asserts this runner is capable, so a lost device must fail the suites | |
| # that need it, not skip. snapshot_integration uses userspace vsock and | |
| # runs regardless. Warn for the log, then proceed. | |
| if [ ! -e /dev/vhost-vsock ]; then | |
| echo "::warning::/dev/vhost-vsock is not available; vhost-vsock suites will fail (REQUIRE_VM=1)." | |
| fi | |
| # Detect kernel version at runtime (env context can't run commands) | |
| KERNEL_SRC="/boot/vmlinuz-$(uname -r)" | |
| KERNEL_COPY="/tmp/void-box-kernel-$(uname -r)" | |
| sudo cp "$KERNEL_SRC" "$KERNEL_COPY" | |
| sudo chown "$USER:$USER" "$KERNEL_COPY" | |
| chmod 644 "$KERNEL_COPY" | |
| export VOID_BOX_KERNEL="$KERNEL_COPY" | |
| echo "Kernel: $VOID_BOX_KERNEL" | |
| echo "Initramfs: $VOID_BOX_INITRAMFS" | |
| # Verify artifacts exist | |
| ls -la "$VOID_BOX_KERNEL" | |
| ls -la "$VOID_BOX_INITRAMFS" | |
| # One nextest run for every deterministic VM suite: the backend | |
| # contract, direct MicroVm/Sandbox boots, the OCI root switch and its | |
| # read-only invariant, host-directory mounts, the guest→host telemetry | |
| # stream and sidecar gateway, the PTY and credential-proxy paths, the | |
| # persistent control channel, and snapshot/restore. The `vm` test group | |
| # (.config/nextest.toml) caps how many VMs boot at once, so the suites | |
| # run with bounded cross-suite parallelism rather than serially, and | |
| # --no-fail-fast collects every result instead of stopping at the first | |
| # failure. | |
| # | |
| # The exclusions, each deliberate: | |
| # | |
| # pty_command_not_allowed asserts the guest rejects a non-allowlisted | |
| # program, but the allowlist loads at guest boot from a file only the | |
| # agent-box provisioning path writes, which this direct attach_pty path | |
| # never stages — so the guest allows all commands (#99). | |
| # | |
| # The agent suites (e2e_service_mode, e2e_agent_mcp) need a real agent | |
| # credential, which no pull-request runner holds, so here they would | |
| # detect its absence and report green without starting an agent. They | |
| # run in `e2e-agent.yml` instead, under VOID_BOX_REQUIRE_AGENT_CREDS=1, | |
| # and on contributors' machines. | |
| cargo nextest run --run-ignored only --no-fail-fast -E \ | |
| 'binary(/^(conformance|telemetry|kvm_integration|oci_integration|e2e_sidecar|e2e_telemetry|e2e_mount|persistent_channel|e2e_skill_pipeline|e2e_pty|e2e_credential_proxy|snapshot_integration)$/) & not test(pty_command_not_allowed)' |