Skip to content

Behavioral Eval · PR #2516 #3229

Behavioral Eval · PR #2516

Behavioral Eval · PR #2516 #3229

name: Behavioral Eval

Check warning on line 1 in .github/workflows/behavioral-evals.yml

View workflow run for this annotation

GitHub Actions / Behavioral Eval

Workflow execution policy warning (evaluate mode)

On November 2, 2026, GitHub will restrict `pull_request_target` on public repositories by default. To continue allowing the event trigger, configure an Actions policy. Learn more: https://gh.io/securely-using-pull_request_target#default-policy-for-pull_request_target
run-name: >-
Behavioral Eval · ${{ github.event_name == 'push' && 'base update' ||
format('PR #{0}', github.event.pull_request.number) }}
on:
pull_request_target:
branches: [main]
types: [labeled, unlabeled, synchronize, edited, reopened]
push:
branches: [main]
concurrency:
group: >-
${{
github.event_name == 'pull_request_target' &&
((github.event.action == 'labeled' && github.event.label.name == 'pre-release') ||
(github.event.action == 'unlabeled' && github.event.label.name == 'pre-release') ||
((github.event.action == 'synchronize' || github.event.action == 'reopened' ||
(github.event.action == 'edited' && github.event.changes.base)) &&
contains(github.event.pull_request.labels.*.name, 'pre-release'))) &&
format('prime-agent-behavioral-{0}', github.event.pull_request.number) ||
format('prime-agent-behavioral-skip-{0}', github.run_id)
}}
cancel-in-progress: true
permissions: {}
jobs:
invalidate-on-pr-change:
if: >-
github.event_name == 'pull_request_target' &&
((github.event.action == 'unlabeled' && github.event.label.name == 'pre-release') ||
((github.event.action == 'synchronize' || github.event.action == 'reopened' ||
(github.event.action == 'edited' && github.event.changes.base)) &&
contains(github.event.pull_request.labels.*.name, 'pre-release')))
runs-on: ubuntu-24.04
permissions:
statuses: write
steps:
- name: Revoke the durable release status
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v8.0.0
with:
script: |
const action = context.payload.action;
const description = action === 'unlabeled'
? 'pre-release approval removed'
: action === 'synchronize'
? 'Head changed; reapply pre-release'
: 'PR identity changed; reapply pre-release';
for (const statusContext of [
'Behavioral Eval / pre-release approval',
'Behavioral Eval / pre-release',
]) {
await github.rest.repos.createCommitStatus({
owner: context.repo.owner,
repo: context.repo.repo,
sha: context.payload.pull_request.head.sha,
state: 'failure',
context: statusContext,
description,
target_url: `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`,
});
}
invalidate-on-base-change:
if: github.event_name == 'push'
runs-on: ubuntu-24.04
permissions:
pull-requests: read
statuses: write
steps:
- name: Revoke results bound to the previous base
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v8.0.0
with:
script: |
const pulls = await github.paginate(github.rest.pulls.list, {
owner: context.repo.owner,
repo: context.repo.repo,
state: 'open',
base: context.ref.replace('refs/heads/', ''),
per_page: 100,
});
for (const pull of pulls) {
const statuses = await github.paginate(github.rest.repos.listCommitStatusesForRef, {
owner: context.repo.owner,
repo: context.repo.repo,
ref: pull.head.sha,
per_page: 100,
});
const contexts = [
'Behavioral Eval / pre-release approval',
'Behavioral Eval / pre-release',
];
if (!statuses.some(status => contexts.includes(status.context))) continue;
for (const statusContext of contexts) {
await github.rest.repos.createCommitStatus({
owner: context.repo.owner,
repo: context.repo.repo,
sha: pull.head.sha,
state: 'failure',
context: statusContext,
description: 'Base advanced; reapply pre-release',
target_url: `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`,
});
}
}
authorize:
name: Authorize the pre-release label actor
if: >-
github.event_name == 'pull_request_target' && github.event.action == 'labeled' &&
github.event.label.name == 'pre-release'
runs-on: ubuntu-24.04
timeout-minutes: 5
permissions: {}
steps:
- name: Validate the labeling actor's repository permission
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v8.0.0
with:
script: |
const actor = context.payload.sender;
if (!actor) {
core.setFailed('No labeling actor recorded for this event.');
return;
}
const { data: permission } = await github.rest.repos.getCollaboratorPermissionLevel({
owner: context.repo.owner,
repo: context.repo.repo,
username: actor.login,
});
const allowed = ['admin', 'maintain', 'write'];
if (!allowed.includes(permission.permission)) {
core.setFailed(
`Actor ${actor.login} has '${permission.permission}' access; ` +
'the pre-release label requires write, maintain, or admin.'
);
}
mark-pending:
name: Mark the requested head pending
if: >-
github.event_name == 'pull_request_target' && github.event.action == 'labeled' &&
github.event.label.name == 'pre-release'
runs-on: ubuntu-24.04
timeout-minutes: 5
permissions:
statuses: write
needs: authorize
steps:
- name: Mark the requested head pending
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v8.0.0
with:
script: |
await github.rest.repos.createCommitStatus({
owner: context.repo.owner,
repo: context.repo.repo,
sha: context.payload.pull_request.head.sha,
state: 'pending',
context: 'Behavioral Eval / pre-release',
description: 'Starting exact base and head validation',
target_url: `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`,
});
await github.rest.repos.createCommitStatus({
owner: context.repo.owner,
repo: context.repo.repo,
sha: context.payload.pull_request.head.sha,
state: 'success',
context: 'Behavioral Eval / pre-release approval',
description: 'pre-release label approves this exact head',
target_url: `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`,
});
evaluate:
if: >-
github.event_name == 'pull_request_target' && github.event.action == 'labeled' &&
github.event.label.name == 'pre-release'
name: runner
runs-on: ubuntu-24.04
timeout-minutes: 360
needs: [authorize, mark-pending]
permissions:
actions: read
contents: read
steps:
- name: Record the job budget
# Mirrors timeout-minutes above so the orchestrator fits its waits inside the job.
run: echo "BEHAVIORAL_JOB_DEADLINE_EPOCH=$(( $(date +%s) + 360 * 60 ))" >> "$GITHUB_ENV"
- name: Check out trusted evaluator
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
ref: ${{ github.sha }}
persist-credentials: false
- name: Resolve exact base and head
id: resolve
run: >-
python3 scripts/evals/short_swe/ci.py
--output request --harness-sha "$(git rev-parse HEAD)"
- name: Set up uv
uses: astral-sh/setup-uv@94527f2e458b27549849d47d273a16bec83a01e9 # v7
with:
version: 0.12.9
enable-cache: false
- name: Install controller tooling
run: uv sync --locked --project scripts/benchmarks
- name: Check out pinned Verifiers
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
repository: PrimeIntellect-ai/verifiers
ref: 9df6a3c01bb640c34380f4f6dff63a1b168dee22
path: vendor/verifiers
persist-credentials: false
- name: Check out pinned environments
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
repository: PrimeIntellect-ai/prime-envs
ref: a41660014674ba43d56ed054199dde25ddf69946
path: vendor/environments
persist-credentials: false
- name: Install pinned evaluator
id: evaluator
continue-on-error: true
run: |
uv sync --locked --project vendor/verifiers --extra harbor
uv pip install --python vendor/verifiers/.venv/bin/python --no-deps \
-e vendor/environments/environments/swe/swebench_verified \
-e vendor/environments/environments/swe/swebench_pro \
-e vendor/environments/environments/swe/scaleswe
- name: Build exact base and head in isolated sandboxes
id: build
continue-on-error: true
if: steps.evaluator.outcome == 'success'
run: |
runner=(uv run --locked --project scripts/benchmarks python scripts/evals/short_swe/build_controller.py)
"${runner[@]}" --repository "$GITHUB_REPOSITORY" --source-repository "$GITHUB_REPOSITORY" \
--sha "${{ steps.resolve.outputs.base_sha }}" --side base \
--run-id "$GITHUB_RUN_ID" --attempt "$GITHUB_RUN_ATTEMPT" --output artifacts/base &
base_pid=$!
"${runner[@]}" --repository "$GITHUB_REPOSITORY" \
--source-repository "${{ steps.resolve.outputs.head_repository }}" \
--sha "${{ steps.resolve.outputs.head_sha }}" --side head \
--run-id "$GITHUB_RUN_ID" --attempt "$GITHUB_RUN_ATTEMPT" --output artifacts/head &
head_pid=$!
status=0
wait "$base_pid" || status=1
wait "$head_pid" || status=1
exit "$status"
env:
PRIME_SANDBOX_API_KEY: ${{ secrets.PRIME_SANDBOX_API_KEY }}
PRIME_TEAM_ID: ${{ vars.PRIME_BENCHMARK_TEAM_ID }}
- name: Generate the fixed 15/8/5 configs
id: prepare
continue-on-error: true
if: steps.build.outcome == 'success'
run: |
python3 scripts/evals/short_swe/prepare.py \
--verifiers vendor/verifiers --environments vendor/environments \
--base-artifacts artifacts/base --base-sha "${{ steps.resolve.outputs.base_sha }}" \
--head-artifacts artifacts/head --head-sha "${{ steps.resolve.outputs.head_sha }}" \
--output generated-configs
- name: Run the deterministic gold-patch oracle
id: oracle
continue-on-error: true
if: steps.prepare.outcome == 'success'
run: |
vendor/verifiers/.venv/bin/python scripts/evals/short_swe/evaluate.py \
--eval vendor/verifiers/.venv/bin/eval --configs generated-configs \
--output oracle-eval --request request/request.json --result results/oracle.json \
--oracle-only
env:
PRIME_API_KEY: ${{ secrets.PRIME_BEHAVIORAL_API_KEY }}
PRIME_SANDBOX_API_KEY: ${{ secrets.PRIME_SANDBOX_API_KEY }}
PRIME_TEAM_ID: ${{ vars.PRIME_BENCHMARK_TEAM_ID }}
- name: Publish the base candidate tarballs
id: upload-base
continue-on-error: true
if: steps.oracle.outcome == 'success'
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: candidate-tarballs-base
path: artifacts/base/*.tgz
if-no-files-found: error
retention-days: 7
- name: Publish the head candidate tarballs
id: upload-head
continue-on-error: true
if: steps.upload-base.outcome == 'success'
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: candidate-tarballs-head
path: artifacts/head/*.tgz
if-no-files-found: error
retention-days: 7
- name: Build the hosted candidate sources
id: sources
continue-on-error: true
if: steps.upload-head.outcome == 'success'
run: |
python3 - <<'PY'
import json
import os
from pathlib import Path
def side(name: str, sha: str, artifact_id: str) -> dict:
manifest = json.loads(Path(f"artifacts/{name}/artifact-manifest.json").read_text())
if manifest["sha"] != sha:
raise ValueError(f"{name}: artifact manifest does not match the request")
return {
"tarballs_url": (
f"https://api.github.com/repos/{os.environ['GITHUB_REPOSITORY']}"
f"/actions/artifacts/{artifact_id}/zip"
),
"commit": sha,
"checksums": {row["name"]: row["sha256"] for row in manifest["artifacts"]},
"token": os.environ["GH_TOKEN"],
}
request = json.loads(Path("request/request.json").read_text())
sources = {
"base": side("base", request["base_sha"], "${{ steps.upload-base.outputs.artifact-id }}"),
"head": side("head", request["head_sha"], "${{ steps.upload-head.outputs.artifact-id }}"),
}
Path("request/sources.json").write_text(json.dumps(sources))
PY
env:
GH_TOKEN: ${{ github.token }}
- name: Install the Prime CLI
id: prime-cli
continue-on-error: true
if: steps.sources.outcome == 'success'
run: uv tool install prime==0.6.31
- name: Launch the six hosted evaluations
id: hosted
continue-on-error: true
if: steps.prime-cli.outcome == 'success'
run: |
uv run --locked --project scripts/benchmarks python \
scripts/evals/short_swe/hosted_eval.py launch \
--request request/request.json --sources request/sources.json --output hosted
env:
PRIME_API_KEY: ${{ secrets.PRIME_BEHAVIORAL_API_KEY }}
PRIME_TEAM_ID: ${{ vars.PRIME_BENCHMARK_TEAM_ID }}
- name: Wait for the hosted evaluations
id: hosted-wait
continue-on-error: true
if: steps.hosted.outcome == 'success'
run: |
uv run --locked --project scripts/benchmarks python \
scripts/evals/short_swe/hosted_eval.py wait --output hosted
env:
PRIME_API_KEY: ${{ secrets.PRIME_BEHAVIORAL_API_KEY }}
- name: Collect the hosted episodes
id: hosted-collect
continue-on-error: true
if: steps.hosted-wait.outcome == 'success'
run: |
uv run --locked --project scripts/benchmarks python \
scripts/evals/short_swe/hosted_eval.py collect --output hosted
env:
PRIME_API_KEY: ${{ secrets.PRIME_BEHAVIORAL_API_KEY }}
- name: Gate the paired hosted result
id: evaluate
continue-on-error: true
if: steps.hosted-collect.outcome == 'success'
run: |
vendor/verifiers/.venv/bin/python scripts/evals/short_swe/evaluate.py \
--eval vendor/verifiers/.venv/bin/eval --configs generated-configs \
--output hosted/raw-eval --request request/request.json \
--result results/paired.json --from-existing
- name: Stop hosted evaluations and clean up sandboxes
if: always()
continue-on-error: true
run: |
uv run --locked --project scripts/benchmarks python \
scripts/evals/short_swe/hosted_eval.py stop --output hosted
uv run --locked --project scripts/benchmarks python scripts/evals/short_swe/cleanup.py \
--repository "$GITHUB_REPOSITORY" --run-id "$GITHUB_RUN_ID" \
--attempt "$GITHUB_RUN_ATTEMPT"
env:
PRIME_API_KEY: ${{ secrets.PRIME_BEHAVIORAL_API_KEY }}
PRIME_SANDBOX_API_KEY: ${{ secrets.PRIME_SANDBOX_API_KEY }}
PRIME_TEAM_ID: ${{ vars.PRIME_BENCHMARK_TEAM_ID }}
- name: Render trusted report
if: always()
run: |
python3 scripts/evals/short_swe/report.py \
--request request/request.json --result results/paired.json \
--markdown results/report.md --verdict results/verdict \
--evaluations hosted/hosted-runs.json
cat results/report.md >> "$GITHUB_STEP_SUMMARY"
- name: Save packages, traces, and paired results
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: behavioral-results-${{ github.run_attempt }}
path: |
artifacts
generated-configs
oracle-eval
hosted
request
!request/sources.json
results
if-no-files-found: warn
retention-days: 30
- name: Enforce complete paired result and regression thresholds
if: always()
run: test -f results/verdict && test "$(cat results/verdict)" = pass
report:
name: Publish PR report
needs: evaluate
if: always() && needs.evaluate.result != 'skipped'
runs-on: ubuntu-24.04
permissions:
contents: read
pull-requests: write
statuses: write
steps:
- name: Download evaluation results
uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7
with:
name: behavioral-results-${{ github.run_attempt }}
path: .
- name: Set exact-head status
if: always()
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v8.0.0
with:
script: |
const fs = require('node:fs');
let request;
try {
request = JSON.parse(fs.readFileSync('request/request.json', 'utf8'));
} catch {
request = null;
}
const {data: pull} = await github.rest.pulls.get({
owner: context.repo.owner,
repo: context.repo.repo,
pull_number: context.payload.pull_request.number,
});
const rules = await github.paginate(
'GET /repos/{owner}/{repo}/rules/branches/{branch}',
{
owner: context.repo.owner,
repo: context.repo.repo,
branch: pull.base.ref,
per_page: 100,
},
);
const requiredContexts = new Set([
'Behavioral Eval / pre-release approval',
'Behavioral Eval / pre-release',
]);
const strictContexts = new Set(
rules
.filter(rule => rule.type === 'required_status_checks' &&
rule.parameters?.strict_required_status_checks_policy === true)
.flatMap(rule => rule.parameters.required_status_checks.map(check => check.context)),
);
const strictRequired = [...requiredContexts].every(
contextName => strictContexts.has(contextName),
);
const labelPresent = pull.labels.some(label => label.name === 'pre-release');
const identityCurrent = request &&
request.head_sha === pull.head.sha && request.base_sha === pull.base.sha;
const passed = strictRequired && identityCurrent && labelPresent &&
fs.existsSync('results/verdict') &&
fs.readFileSync('results/verdict', 'utf8').trim() === 'pass';
await github.rest.repos.createCommitStatus({
owner: context.repo.owner,
repo: context.repo.repo,
sha: context.payload.pull_request.head.sha,
state: passed ? 'success' : 'failure',
context: 'Behavioral Eval / pre-release',
description: passed ? 'Paired Short SWE gate passed' : 'Paired Short SWE gate failed or stale',
target_url: `https://github.com/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`,
});
- name: Publish PR report
if: always() && hashFiles('results/report.md') != ''
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v8.0.0
with:
script: |
const fs = require('node:fs');
const marker = '<!-- prime-agent-behavioral-eval:v1 -->';
const body = fs.readFileSync('results/report.md', 'utf8');
const comments = await github.paginate(github.rest.issues.listComments, {
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: context.issue.number,
per_page: 100,
});
const prior = comments.find(comment =>
comment.user.type === 'Bot' && comment.body?.includes(marker));
if (prior) {
await github.rest.issues.updateComment({
owner: context.repo.owner,
repo: context.repo.repo,
comment_id: prior.id,
body,
});
} else {
await github.rest.issues.createComment({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: context.issue.number,
body,
});
}