Behavioral Eval · PR #2516 #3229
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: Behavioral Eval | ||
|
Check warning on line 1 in .github/workflows/behavioral-evals.yml
|
||
| run-name: >- | ||
| Behavioral Eval · ${{ github.event_name == 'push' && 'base update' || | ||
| format('PR #{0}', github.event.pull_request.number) }} | ||
| on: | ||
| pull_request_target: | ||
| branches: [main] | ||
| types: [labeled, unlabeled, synchronize, edited, reopened] | ||
| push: | ||
| branches: [main] | ||
| concurrency: | ||
| group: >- | ||
| ${{ | ||
| github.event_name == 'pull_request_target' && | ||
| ((github.event.action == 'labeled' && github.event.label.name == 'pre-release') || | ||
| (github.event.action == 'unlabeled' && github.event.label.name == 'pre-release') || | ||
| ((github.event.action == 'synchronize' || github.event.action == 'reopened' || | ||
| (github.event.action == 'edited' && github.event.changes.base)) && | ||
| contains(github.event.pull_request.labels.*.name, 'pre-release'))) && | ||
| format('prime-agent-behavioral-{0}', github.event.pull_request.number) || | ||
| format('prime-agent-behavioral-skip-{0}', github.run_id) | ||
| }} | ||
| cancel-in-progress: true | ||
| permissions: {} | ||
| jobs: | ||
| invalidate-on-pr-change: | ||
| if: >- | ||
| github.event_name == 'pull_request_target' && | ||
| ((github.event.action == 'unlabeled' && github.event.label.name == 'pre-release') || | ||
| ((github.event.action == 'synchronize' || github.event.action == 'reopened' || | ||
| (github.event.action == 'edited' && github.event.changes.base)) && | ||
| contains(github.event.pull_request.labels.*.name, 'pre-release'))) | ||
| runs-on: ubuntu-24.04 | ||
| permissions: | ||
| statuses: write | ||
| steps: | ||
| - name: Revoke the durable release status | ||
| uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v8.0.0 | ||
| with: | ||
| script: | | ||
| const action = context.payload.action; | ||
| const description = action === 'unlabeled' | ||
| ? 'pre-release approval removed' | ||
| : action === 'synchronize' | ||
| ? 'Head changed; reapply pre-release' | ||
| : 'PR identity changed; reapply pre-release'; | ||
| for (const statusContext of [ | ||
| 'Behavioral Eval / pre-release approval', | ||
| 'Behavioral Eval / pre-release', | ||
| ]) { | ||
| await github.rest.repos.createCommitStatus({ | ||
| owner: context.repo.owner, | ||
| repo: context.repo.repo, | ||
| sha: context.payload.pull_request.head.sha, | ||
| state: 'failure', | ||
| context: statusContext, | ||
| description, | ||
| target_url: `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`, | ||
| }); | ||
| } | ||
| invalidate-on-base-change: | ||
| if: github.event_name == 'push' | ||
| runs-on: ubuntu-24.04 | ||
| permissions: | ||
| pull-requests: read | ||
| statuses: write | ||
| steps: | ||
| - name: Revoke results bound to the previous base | ||
| uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v8.0.0 | ||
| with: | ||
| script: | | ||
| const pulls = await github.paginate(github.rest.pulls.list, { | ||
| owner: context.repo.owner, | ||
| repo: context.repo.repo, | ||
| state: 'open', | ||
| base: context.ref.replace('refs/heads/', ''), | ||
| per_page: 100, | ||
| }); | ||
| for (const pull of pulls) { | ||
| const statuses = await github.paginate(github.rest.repos.listCommitStatusesForRef, { | ||
| owner: context.repo.owner, | ||
| repo: context.repo.repo, | ||
| ref: pull.head.sha, | ||
| per_page: 100, | ||
| }); | ||
| const contexts = [ | ||
| 'Behavioral Eval / pre-release approval', | ||
| 'Behavioral Eval / pre-release', | ||
| ]; | ||
| if (!statuses.some(status => contexts.includes(status.context))) continue; | ||
| for (const statusContext of contexts) { | ||
| await github.rest.repos.createCommitStatus({ | ||
| owner: context.repo.owner, | ||
| repo: context.repo.repo, | ||
| sha: pull.head.sha, | ||
| state: 'failure', | ||
| context: statusContext, | ||
| description: 'Base advanced; reapply pre-release', | ||
| target_url: `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`, | ||
| }); | ||
| } | ||
| } | ||
| authorize: | ||
| name: Authorize the pre-release label actor | ||
| if: >- | ||
| github.event_name == 'pull_request_target' && github.event.action == 'labeled' && | ||
| github.event.label.name == 'pre-release' | ||
| runs-on: ubuntu-24.04 | ||
| timeout-minutes: 5 | ||
| permissions: {} | ||
| steps: | ||
| - name: Validate the labeling actor's repository permission | ||
| uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v8.0.0 | ||
| with: | ||
| script: | | ||
| const actor = context.payload.sender; | ||
| if (!actor) { | ||
| core.setFailed('No labeling actor recorded for this event.'); | ||
| return; | ||
| } | ||
| const { data: permission } = await github.rest.repos.getCollaboratorPermissionLevel({ | ||
| owner: context.repo.owner, | ||
| repo: context.repo.repo, | ||
| username: actor.login, | ||
| }); | ||
| const allowed = ['admin', 'maintain', 'write']; | ||
| if (!allowed.includes(permission.permission)) { | ||
| core.setFailed( | ||
| `Actor ${actor.login} has '${permission.permission}' access; ` + | ||
| 'the pre-release label requires write, maintain, or admin.' | ||
| ); | ||
| } | ||
| mark-pending: | ||
| name: Mark the requested head pending | ||
| if: >- | ||
| github.event_name == 'pull_request_target' && github.event.action == 'labeled' && | ||
| github.event.label.name == 'pre-release' | ||
| runs-on: ubuntu-24.04 | ||
| timeout-minutes: 5 | ||
| permissions: | ||
| statuses: write | ||
| needs: authorize | ||
| steps: | ||
| - name: Mark the requested head pending | ||
| uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v8.0.0 | ||
| with: | ||
| script: | | ||
| await github.rest.repos.createCommitStatus({ | ||
| owner: context.repo.owner, | ||
| repo: context.repo.repo, | ||
| sha: context.payload.pull_request.head.sha, | ||
| state: 'pending', | ||
| context: 'Behavioral Eval / pre-release', | ||
| description: 'Starting exact base and head validation', | ||
| target_url: `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`, | ||
| }); | ||
| await github.rest.repos.createCommitStatus({ | ||
| owner: context.repo.owner, | ||
| repo: context.repo.repo, | ||
| sha: context.payload.pull_request.head.sha, | ||
| state: 'success', | ||
| context: 'Behavioral Eval / pre-release approval', | ||
| description: 'pre-release label approves this exact head', | ||
| target_url: `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`, | ||
| }); | ||
| evaluate: | ||
| if: >- | ||
| github.event_name == 'pull_request_target' && github.event.action == 'labeled' && | ||
| github.event.label.name == 'pre-release' | ||
| name: runner | ||
| runs-on: ubuntu-24.04 | ||
| timeout-minutes: 360 | ||
| needs: [authorize, mark-pending] | ||
| permissions: | ||
| actions: read | ||
| contents: read | ||
| steps: | ||
| - name: Record the job budget | ||
| # Mirrors timeout-minutes above so the orchestrator fits its waits inside the job. | ||
| run: echo "BEHAVIORAL_JOB_DEADLINE_EPOCH=$(( $(date +%s) + 360 * 60 ))" >> "$GITHUB_ENV" | ||
| - name: Check out trusted evaluator | ||
| uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 | ||
| with: | ||
| ref: ${{ github.sha }} | ||
| persist-credentials: false | ||
| - name: Resolve exact base and head | ||
| id: resolve | ||
| run: >- | ||
| python3 scripts/evals/short_swe/ci.py | ||
| --output request --harness-sha "$(git rev-parse HEAD)" | ||
| - name: Set up uv | ||
| uses: astral-sh/setup-uv@94527f2e458b27549849d47d273a16bec83a01e9 # v7 | ||
| with: | ||
| version: 0.12.9 | ||
| enable-cache: false | ||
| - name: Install controller tooling | ||
| run: uv sync --locked --project scripts/benchmarks | ||
| - name: Check out pinned Verifiers | ||
| uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 | ||
| with: | ||
| repository: PrimeIntellect-ai/verifiers | ||
| ref: 9df6a3c01bb640c34380f4f6dff63a1b168dee22 | ||
| path: vendor/verifiers | ||
| persist-credentials: false | ||
| - name: Check out pinned environments | ||
| uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 | ||
| with: | ||
| repository: PrimeIntellect-ai/prime-envs | ||
| ref: a41660014674ba43d56ed054199dde25ddf69946 | ||
| path: vendor/environments | ||
| persist-credentials: false | ||
| - name: Install pinned evaluator | ||
| id: evaluator | ||
| continue-on-error: true | ||
| run: | | ||
| uv sync --locked --project vendor/verifiers --extra harbor | ||
| uv pip install --python vendor/verifiers/.venv/bin/python --no-deps \ | ||
| -e vendor/environments/environments/swe/swebench_verified \ | ||
| -e vendor/environments/environments/swe/swebench_pro \ | ||
| -e vendor/environments/environments/swe/scaleswe | ||
| - name: Build exact base and head in isolated sandboxes | ||
| id: build | ||
| continue-on-error: true | ||
| if: steps.evaluator.outcome == 'success' | ||
| run: | | ||
| runner=(uv run --locked --project scripts/benchmarks python scripts/evals/short_swe/build_controller.py) | ||
| "${runner[@]}" --repository "$GITHUB_REPOSITORY" --source-repository "$GITHUB_REPOSITORY" \ | ||
| --sha "${{ steps.resolve.outputs.base_sha }}" --side base \ | ||
| --run-id "$GITHUB_RUN_ID" --attempt "$GITHUB_RUN_ATTEMPT" --output artifacts/base & | ||
| base_pid=$! | ||
| "${runner[@]}" --repository "$GITHUB_REPOSITORY" \ | ||
| --source-repository "${{ steps.resolve.outputs.head_repository }}" \ | ||
| --sha "${{ steps.resolve.outputs.head_sha }}" --side head \ | ||
| --run-id "$GITHUB_RUN_ID" --attempt "$GITHUB_RUN_ATTEMPT" --output artifacts/head & | ||
| head_pid=$! | ||
| status=0 | ||
| wait "$base_pid" || status=1 | ||
| wait "$head_pid" || status=1 | ||
| exit "$status" | ||
| env: | ||
| PRIME_SANDBOX_API_KEY: ${{ secrets.PRIME_SANDBOX_API_KEY }} | ||
| PRIME_TEAM_ID: ${{ vars.PRIME_BENCHMARK_TEAM_ID }} | ||
| - name: Generate the fixed 15/8/5 configs | ||
| id: prepare | ||
| continue-on-error: true | ||
| if: steps.build.outcome == 'success' | ||
| run: | | ||
| python3 scripts/evals/short_swe/prepare.py \ | ||
| --verifiers vendor/verifiers --environments vendor/environments \ | ||
| --base-artifacts artifacts/base --base-sha "${{ steps.resolve.outputs.base_sha }}" \ | ||
| --head-artifacts artifacts/head --head-sha "${{ steps.resolve.outputs.head_sha }}" \ | ||
| --output generated-configs | ||
| - name: Run the deterministic gold-patch oracle | ||
| id: oracle | ||
| continue-on-error: true | ||
| if: steps.prepare.outcome == 'success' | ||
| run: | | ||
| vendor/verifiers/.venv/bin/python scripts/evals/short_swe/evaluate.py \ | ||
| --eval vendor/verifiers/.venv/bin/eval --configs generated-configs \ | ||
| --output oracle-eval --request request/request.json --result results/oracle.json \ | ||
| --oracle-only | ||
| env: | ||
| PRIME_API_KEY: ${{ secrets.PRIME_BEHAVIORAL_API_KEY }} | ||
| PRIME_SANDBOX_API_KEY: ${{ secrets.PRIME_SANDBOX_API_KEY }} | ||
| PRIME_TEAM_ID: ${{ vars.PRIME_BENCHMARK_TEAM_ID }} | ||
| - name: Publish the base candidate tarballs | ||
| id: upload-base | ||
| continue-on-error: true | ||
| if: steps.oracle.outcome == 'success' | ||
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | ||
| with: | ||
| name: candidate-tarballs-base | ||
| path: artifacts/base/*.tgz | ||
| if-no-files-found: error | ||
| retention-days: 7 | ||
| - name: Publish the head candidate tarballs | ||
| id: upload-head | ||
| continue-on-error: true | ||
| if: steps.upload-base.outcome == 'success' | ||
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | ||
| with: | ||
| name: candidate-tarballs-head | ||
| path: artifacts/head/*.tgz | ||
| if-no-files-found: error | ||
| retention-days: 7 | ||
| - name: Build the hosted candidate sources | ||
| id: sources | ||
| continue-on-error: true | ||
| if: steps.upload-head.outcome == 'success' | ||
| run: | | ||
| python3 - <<'PY' | ||
| import json | ||
| import os | ||
| from pathlib import Path | ||
| def side(name: str, sha: str, artifact_id: str) -> dict: | ||
| manifest = json.loads(Path(f"artifacts/{name}/artifact-manifest.json").read_text()) | ||
| if manifest["sha"] != sha: | ||
| raise ValueError(f"{name}: artifact manifest does not match the request") | ||
| return { | ||
| "tarballs_url": ( | ||
| f"https://api.github.com/repos/{os.environ['GITHUB_REPOSITORY']}" | ||
| f"/actions/artifacts/{artifact_id}/zip" | ||
| ), | ||
| "commit": sha, | ||
| "checksums": {row["name"]: row["sha256"] for row in manifest["artifacts"]}, | ||
| "token": os.environ["GH_TOKEN"], | ||
| } | ||
| request = json.loads(Path("request/request.json").read_text()) | ||
| sources = { | ||
| "base": side("base", request["base_sha"], "${{ steps.upload-base.outputs.artifact-id }}"), | ||
| "head": side("head", request["head_sha"], "${{ steps.upload-head.outputs.artifact-id }}"), | ||
| } | ||
| Path("request/sources.json").write_text(json.dumps(sources)) | ||
| PY | ||
| env: | ||
| GH_TOKEN: ${{ github.token }} | ||
| - name: Install the Prime CLI | ||
| id: prime-cli | ||
| continue-on-error: true | ||
| if: steps.sources.outcome == 'success' | ||
| run: uv tool install prime==0.6.31 | ||
| - name: Launch the six hosted evaluations | ||
| id: hosted | ||
| continue-on-error: true | ||
| if: steps.prime-cli.outcome == 'success' | ||
| run: | | ||
| uv run --locked --project scripts/benchmarks python \ | ||
| scripts/evals/short_swe/hosted_eval.py launch \ | ||
| --request request/request.json --sources request/sources.json --output hosted | ||
| env: | ||
| PRIME_API_KEY: ${{ secrets.PRIME_BEHAVIORAL_API_KEY }} | ||
| PRIME_TEAM_ID: ${{ vars.PRIME_BENCHMARK_TEAM_ID }} | ||
| - name: Wait for the hosted evaluations | ||
| id: hosted-wait | ||
| continue-on-error: true | ||
| if: steps.hosted.outcome == 'success' | ||
| run: | | ||
| uv run --locked --project scripts/benchmarks python \ | ||
| scripts/evals/short_swe/hosted_eval.py wait --output hosted | ||
| env: | ||
| PRIME_API_KEY: ${{ secrets.PRIME_BEHAVIORAL_API_KEY }} | ||
| - name: Collect the hosted episodes | ||
| id: hosted-collect | ||
| continue-on-error: true | ||
| if: steps.hosted-wait.outcome == 'success' | ||
| run: | | ||
| uv run --locked --project scripts/benchmarks python \ | ||
| scripts/evals/short_swe/hosted_eval.py collect --output hosted | ||
| env: | ||
| PRIME_API_KEY: ${{ secrets.PRIME_BEHAVIORAL_API_KEY }} | ||
| - name: Gate the paired hosted result | ||
| id: evaluate | ||
| continue-on-error: true | ||
| if: steps.hosted-collect.outcome == 'success' | ||
| run: | | ||
| vendor/verifiers/.venv/bin/python scripts/evals/short_swe/evaluate.py \ | ||
| --eval vendor/verifiers/.venv/bin/eval --configs generated-configs \ | ||
| --output hosted/raw-eval --request request/request.json \ | ||
| --result results/paired.json --from-existing | ||
| - name: Stop hosted evaluations and clean up sandboxes | ||
| if: always() | ||
| continue-on-error: true | ||
| run: | | ||
| uv run --locked --project scripts/benchmarks python \ | ||
| scripts/evals/short_swe/hosted_eval.py stop --output hosted | ||
| uv run --locked --project scripts/benchmarks python scripts/evals/short_swe/cleanup.py \ | ||
| --repository "$GITHUB_REPOSITORY" --run-id "$GITHUB_RUN_ID" \ | ||
| --attempt "$GITHUB_RUN_ATTEMPT" | ||
| env: | ||
| PRIME_API_KEY: ${{ secrets.PRIME_BEHAVIORAL_API_KEY }} | ||
| PRIME_SANDBOX_API_KEY: ${{ secrets.PRIME_SANDBOX_API_KEY }} | ||
| PRIME_TEAM_ID: ${{ vars.PRIME_BENCHMARK_TEAM_ID }} | ||
| - name: Render trusted report | ||
| if: always() | ||
| run: | | ||
| python3 scripts/evals/short_swe/report.py \ | ||
| --request request/request.json --result results/paired.json \ | ||
| --markdown results/report.md --verdict results/verdict \ | ||
| --evaluations hosted/hosted-runs.json | ||
| cat results/report.md >> "$GITHUB_STEP_SUMMARY" | ||
| - name: Save packages, traces, and paired results | ||
| if: always() | ||
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | ||
| with: | ||
| name: behavioral-results-${{ github.run_attempt }} | ||
| path: | | ||
| artifacts | ||
| generated-configs | ||
| oracle-eval | ||
| hosted | ||
| request | ||
| !request/sources.json | ||
| results | ||
| if-no-files-found: warn | ||
| retention-days: 30 | ||
| - name: Enforce complete paired result and regression thresholds | ||
| if: always() | ||
| run: test -f results/verdict && test "$(cat results/verdict)" = pass | ||
| report: | ||
| name: Publish PR report | ||
| needs: evaluate | ||
| if: always() && needs.evaluate.result != 'skipped' | ||
| runs-on: ubuntu-24.04 | ||
| permissions: | ||
| contents: read | ||
| pull-requests: write | ||
| statuses: write | ||
| steps: | ||
| - name: Download evaluation results | ||
| uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7 | ||
| with: | ||
| name: behavioral-results-${{ github.run_attempt }} | ||
| path: . | ||
| - name: Set exact-head status | ||
| if: always() | ||
| uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v8.0.0 | ||
| with: | ||
| script: | | ||
| const fs = require('node:fs'); | ||
| let request; | ||
| try { | ||
| request = JSON.parse(fs.readFileSync('request/request.json', 'utf8')); | ||
| } catch { | ||
| request = null; | ||
| } | ||
| const {data: pull} = await github.rest.pulls.get({ | ||
| owner: context.repo.owner, | ||
| repo: context.repo.repo, | ||
| pull_number: context.payload.pull_request.number, | ||
| }); | ||
| const rules = await github.paginate( | ||
| 'GET /repos/{owner}/{repo}/rules/branches/{branch}', | ||
| { | ||
| owner: context.repo.owner, | ||
| repo: context.repo.repo, | ||
| branch: pull.base.ref, | ||
| per_page: 100, | ||
| }, | ||
| ); | ||
| const requiredContexts = new Set([ | ||
| 'Behavioral Eval / pre-release approval', | ||
| 'Behavioral Eval / pre-release', | ||
| ]); | ||
| const strictContexts = new Set( | ||
| rules | ||
| .filter(rule => rule.type === 'required_status_checks' && | ||
| rule.parameters?.strict_required_status_checks_policy === true) | ||
| .flatMap(rule => rule.parameters.required_status_checks.map(check => check.context)), | ||
| ); | ||
| const strictRequired = [...requiredContexts].every( | ||
| contextName => strictContexts.has(contextName), | ||
| ); | ||
| const labelPresent = pull.labels.some(label => label.name === 'pre-release'); | ||
| const identityCurrent = request && | ||
| request.head_sha === pull.head.sha && request.base_sha === pull.base.sha; | ||
| const passed = strictRequired && identityCurrent && labelPresent && | ||
| fs.existsSync('results/verdict') && | ||
| fs.readFileSync('results/verdict', 'utf8').trim() === 'pass'; | ||
| await github.rest.repos.createCommitStatus({ | ||
| owner: context.repo.owner, | ||
| repo: context.repo.repo, | ||
| sha: context.payload.pull_request.head.sha, | ||
| state: passed ? 'success' : 'failure', | ||
| context: 'Behavioral Eval / pre-release', | ||
| description: passed ? 'Paired Short SWE gate passed' : 'Paired Short SWE gate failed or stale', | ||
| target_url: `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`, | ||
| }); | ||
| - name: Publish PR report | ||
| if: always() && hashFiles('results/report.md') != '' | ||
| uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v8.0.0 | ||
| with: | ||
| script: | | ||
| const fs = require('node:fs'); | ||
| const marker = '<!-- prime-agent-behavioral-eval:v1 -->'; | ||
| const body = fs.readFileSync('results/report.md', 'utf8'); | ||
| const comments = await github.paginate(github.rest.issues.listComments, { | ||
| owner: context.repo.owner, | ||
| repo: context.repo.repo, | ||
| issue_number: context.issue.number, | ||
| per_page: 100, | ||
| }); | ||
| const prior = comments.find(comment => | ||
| comment.user.type === 'Bot' && comment.body?.includes(marker)); | ||
| if (prior) { | ||
| await github.rest.issues.updateComment({ | ||
| owner: context.repo.owner, | ||
| repo: context.repo.repo, | ||
| comment_id: prior.id, | ||
| body, | ||
| }); | ||
| } else { | ||
| await github.rest.issues.createComment({ | ||
| owner: context.repo.owner, | ||
| repo: context.repo.repo, | ||
| issue_number: context.issue.number, | ||
| body, | ||
| }); | ||
| } | ||