Skip to content

Benchmark

Benchmark #90

Workflow file for this run

name: Benchmark
on:
schedule:
- cron: '47 9 * * *'
workflow_dispatch:
inputs:
bifrost_ref:
description: 'Bifrost ref or commit to benchmark'
required: false
default: 'master'
type: string
include_unsupported:
description: 'Run cases marked unsupported'
required: false
default: false
type: boolean
notify_slack:
description: 'Post the benchmark summary to Slack'
required: false
default: false
type: boolean
permissions: {}
concurrency:
group: usagebench-${{ github.ref }}
cancel-in-progress: true
jobs:
benchmark:
if: github.ref == 'refs/heads/main'
runs-on: ubuntu-latest
permissions:
contents: read
# Keep an explicit overall ceiling while reserving 130 minutes around the
# benchmark step for cold checkout/build work and post-timeout artifact
# publication.
timeout-minutes: 300
outputs:
slack_payload: ${{ steps.slack_payload.outputs.payload }}
env:
BIFROST_REF: ${{ inputs.bifrost_ref || 'master' }}
steps:
- name: Checkout usagebench
uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5.1.0
with:
fetch-depth: 0
lfs: true
persist-credentials: false
- name: Validate Bifrost ref
shell: bash
run: |
set -euo pipefail
git check-ref-format --allow-onelevel "$BIFROST_REF" >/dev/null || {
echo "Bifrost ref is not a valid Git ref: $BIFROST_REF" >&2
exit 1
}
- name: Checkout Bifrost
uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5.1.0
with:
repository: BrokkAi/bifrost
# This path is already ignored by UsageBench, so the analyzer checkout
# does not make report provenance appear dirty.
path: .bifrost
ref: ${{ env.BIFROST_REF }}
fetch-depth: 0
lfs: false
persist-credentials: false
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable
with:
toolchain: stable
- name: Cache cargo + target
uses: swatinem/rust-cache@e18b497796c12c097a38f9edb9d0641fb99eee32 # v2
- name: Build usagebench
run: cargo build --locked --bin usagebench
- name: Validate benchmark cases
run: ./target/debug/usagebench validate benchmarks/cases
- name: Run Bifrost usage benchmark
id: run_usagebench
continue-on-error: true
# Expire well before the job so checkpoint discovery, artifact upload,
# and summaries still run after a slow benchmark is terminated.
timeout-minutes: 170
shell: bash
env:
INCLUDE_UNSUPPORTED: ${{ inputs.include_unsupported || false }}
run: |
set -euo pipefail
mkdir -p benchmark-output
args=(
run-bifrost
benchmarks/cases
--bifrost-repo .bifrost
--bifrost-commit "$BIFROST_REF"
--bifrost-working-tree
--work-dir target/usagebench
--scan-usages-max-duration-secs 300
--expected-passes benchmarks/expectations/bifrost-expected-passes.yaml
--output "benchmark-output/run-${GITHUB_RUN_ID}.json"
)
if [ "$INCLUDE_UNSUPPORTED" = "true" ]; then
args+=(--include-unsupported)
fi
./target/debug/usagebench "${args[@]}"
- name: Resolve benchmark paths
id: paths
if: always()
shell: bash
run: |
set -euo pipefail
report_path="benchmark-output/run-${GITHUB_RUN_ID}.json"
report_complete="true"
partial_path="benchmark-output/run-${GITHUB_RUN_ID}.partial.json"
if [ -f "$report_path" ]; then
report_complete="$(jq -r 'if has("completed") then .completed else true end' "$report_path")"
elif [ -f "$partial_path" ] && jq -e '.completed == false' "$partial_path" >/dev/null; then
report_path="$partial_path"
report_complete="false"
elif [ ! -f "$report_path" ]; then
report_path=""
report_complete="false"
fi
{
echo "report_path=${report_path}"
echo "report_complete=${report_complete}"
} >> "$GITHUB_OUTPUT"
- name: Record UsageBench provenance
id: usagebench_provenance
if: always()
shell: bash
env:
REPORT_PATH: ${{ steps.paths.outputs.report_path }}
run: |
set -euo pipefail
report_path="$REPORT_PATH"
revision="${GITHUB_SHA}"
release=""
if [ -n "$report_path" ] && [ -f "$report_path" ]; then
revision="$(jq -r '.usagebenchRevision // ""' "$report_path")"
release="$(jq -r '.usagebenchRelease // ""' "$report_path")"
fi
if [[ ! "$revision" =~ ^[0-9a-fA-F]{40}(-dirty)?$ ]]; then
echo "Ignoring invalid UsageBench revision from report" >&2
revision="$GITHUB_SHA"
release=""
fi
if [[ -n "$release" && ! "$release" =~ ^v[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
echo "Ignoring invalid UsageBench release from report" >&2
release=""
fi
artifact_label="${release:-${revision:0:12}}"
{
echo "revision=$revision"
echo "release=$release"
echo "artifact_label=$artifact_label"
} >> "$GITHUB_OUTPUT"
- name: Upload benchmark artifacts
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: usagebench-${{ steps.usagebench_provenance.outputs.artifact_label }}-${{ github.run_id }}
path: benchmark-output
if-no-files-found: warn
retention-days: 14
- name: Publish benchmark summary
if: always()
shell: bash
env:
RUN_OUTCOME: ${{ steps.run_usagebench.outcome }}
USAGEBENCH_REVISION: ${{ steps.usagebench_provenance.outputs.revision }}
USAGEBENCH_RELEASE: ${{ steps.usagebench_provenance.outputs.release }}
REPORT_PATH: ${{ steps.paths.outputs.report_path }}
REPORT_COMPLETE: ${{ steps.paths.outputs.report_complete }}
run: |
set -euo pipefail
{
echo "# Usage Benchmark"
echo
echo "- Run step: $RUN_OUTCOME"
echo "- UsageBench revision: \`$USAGEBENCH_REVISION\`"
if [ -n "$USAGEBENCH_RELEASE" ]; then
echo "- UsageBench release: \`$USAGEBENCH_RELEASE\`"
fi
echo "- Bifrost ref: \`${BIFROST_REF}\`"
echo "- Complete report: \`${REPORT_COMPLETE}\`"
if [ -n "$REPORT_PATH" ]; then
echo "- Report: \`$REPORT_PATH\`"
echo
scripts/render-benchmark-summary.sh "$REPORT_PATH"
else
echo "- Report: none written"
fi
} >> "$GITHUB_STEP_SUMMARY"
- name: Prepare Slack payload
id: slack_payload
if: always()
continue-on-error: true
shell: bash
env:
RUN_OUTCOME: ${{ steps.run_usagebench.outcome }}
REPORT_PATH: ${{ steps.paths.outputs.report_path }}
REPORT_COMPLETE: ${{ steps.paths.outputs.report_complete }}
WORKFLOW_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
USAGEBENCH_REVISION: ${{ steps.usagebench_provenance.outputs.revision }}
USAGEBENCH_RELEASE: ${{ steps.usagebench_provenance.outputs.release }}
run: |
set -euo pipefail
run_outcome="$RUN_OUTCOME"
report_path="$REPORT_PATH"
report_complete="$REPORT_COMPLETE"
head_sha_short="$(git rev-parse --short "$GITHUB_SHA")"
bifrost_resolved_commit=""
bifrost_sha_short=""
cases_count="0"
processed_documents_count="0"
requested_documents_count="0"
processed_cases_count="0"
requested_authored_cases_count="0"
requested_planned_cases_count="0"
passed_count="0"
near_misses_count="0"
position_unverified_count="0"
improved_count="0"
total_passed_count="0"
failed_count="0"
expected_failures_count="0"
not_planned_count="0"
unsupported_count="0"
unsupported_navigation_lookups_count="0"
skipped_count="0"
errors_count="0"
failing_cases=""
if [ -n "$report_path" ] && [ -f "$report_path" ]; then
bifrost_resolved_commit="$(jq -r '.bifrostResolvedCommit // ""' "$report_path")"
bifrost_sha_short="$(printf '%s' "$bifrost_resolved_commit" | cut -c1-7)"
report_counts="$(scripts/render-benchmark-summary.sh --counts "$report_path")"
report_complete="$(jq -r '.report_complete' <<< "$report_counts")"
processed_documents_count="$(jq -r '.processed_documents_count' <<< "$report_counts")"
requested_documents_count="$(jq -r '.requested_documents_count' <<< "$report_counts")"
processed_cases_count="$(jq -r '.processed_cases_count' <<< "$report_counts")"
requested_authored_cases_count="$(jq -r '.requested_authored_cases_count' <<< "$report_counts")"
requested_planned_cases_count="$(jq -r '.requested_planned_cases_count' <<< "$report_counts")"
cases_count="$requested_planned_cases_count"
passed_count="$(jq -r '.totals.passed // 0' "$report_path")"
near_misses_count="$(jq -r '.totals.nearMisses // 0' "$report_path")"
position_unverified_count="$(jq -r '.totals.positionUnverified // 0' "$report_path")"
improved_count="$(jq -r '.totals.improved // 0' "$report_path")"
total_passed_count="$((passed_count + improved_count))"
failed_count="$(jq -r '.totals.failed // 0' "$report_path")"
expected_failures_count="$(jq -r '.totals.expectedFailures // 0' "$report_path")"
not_planned_count="$(jq -r '.totals.notPlanned // 0' "$report_path")"
unsupported_count="$(jq -r '.totals.unsupported // 0' "$report_path")"
unsupported_navigation_lookups_count="$(
jq '[.documents[]?.cases[]?.usageToDeclaration[]? | select(.status == "unsupported")] | length' "$report_path"
)"
skipped_count="$(jq -r '.totals.skipped // 0' "$report_path")"
errors_count="$(jq -r '.totals.errors // 0' "$report_path")"
failing_cases="$(
scripts/render-benchmark-summary.sh --failing-cases "$report_path"
)"
fi
ok="true"
error_text=""
if [ -z "$report_path" ] || [ ! -f "$report_path" ]; then
ok="false"
error_text="usage benchmark run did not write a report"
elif [ "$report_complete" != "true" ]; then
ok="false"
requested_planned_display="$requested_planned_cases_count"
if [ "$requested_planned_display" = "null" ]; then
requested_planned_display="unknown"
fi
error_text="usage benchmark report is INCOMPLETE: processed ${processed_documents_count}/${requested_documents_count} documents and ${processed_cases_count}/${requested_planned_display} planned cases"
elif [ "$errors_count" != "0" ]; then
ok="false"
if [ -n "$failing_cases" ]; then
error_text="$(printf 'Usage benchmark completed with runner errors:\n%s' "$failing_cases")"
else
error_text="usage benchmark completed with runner errors"
fi
elif [ "$run_outcome" != "success" ] || [ "$failed_count" != "0" ]; then
ok="false"
if [ -n "$failing_cases" ]; then
error_text="$(printf 'Usage benchmark recorded semantic failures:\n%s' "$failing_cases")"
else
error_text="usage benchmark recorded semantic failures"
fi
fi
payload="$(
jq -nc \
--arg error_text "$error_text" \
--arg workflow_run_url "$WORKFLOW_RUN_URL" \
--arg head_sha_short "$head_sha_short" \
--arg usagebench_revision "$USAGEBENCH_REVISION" \
--arg usagebench_release "$USAGEBENCH_RELEASE" \
--arg bifrost_ref "$BIFROST_REF" \
--arg bifrost_sha_short "$bifrost_sha_short" \
--arg run_outcome "$run_outcome" \
--argjson report_complete "$report_complete" \
--argjson ok "$ok" \
--argjson cases_count "$cases_count" \
--argjson processed_documents_count "$processed_documents_count" \
--argjson requested_documents_count "$requested_documents_count" \
--argjson processed_cases_count "$processed_cases_count" \
--argjson requested_authored_cases_count "$requested_authored_cases_count" \
--argjson requested_planned_cases_count "$requested_planned_cases_count" \
--argjson passed_count "$passed_count" \
--argjson near_misses_count "$near_misses_count" \
--argjson position_unverified_count "$position_unverified_count" \
--argjson improved_count "$improved_count" \
--argjson total_passed_count "$total_passed_count" \
--argjson failed_count "$failed_count" \
--argjson expected_failures_count "$expected_failures_count" \
--argjson not_planned_count "$not_planned_count" \
--argjson unsupported_count "$unsupported_count" \
--argjson unsupported_navigation_lookups_count "$unsupported_navigation_lookups_count" \
--argjson skipped_count "$skipped_count" \
--argjson errors_count "$errors_count" \
'{
ok: $ok,
error_text: $error_text,
workflow_run_url: $workflow_run_url,
head_sha_short: $head_sha_short,
usagebench_revision: $usagebench_revision,
usagebench_release: $usagebench_release,
bifrost_ref: $bifrost_ref,
bifrost_sha_short: $bifrost_sha_short,
run_outcome: $run_outcome,
report_complete: $report_complete,
cases_count: $cases_count,
processed_documents_count: $processed_documents_count,
requested_documents_count: $requested_documents_count,
processed_cases_count: $processed_cases_count,
requested_authored_cases_count: $requested_authored_cases_count,
requested_planned_cases_count: $requested_planned_cases_count,
passed_count: $passed_count,
near_misses_count: $near_misses_count,
position_unverified_count: $position_unverified_count,
improved_count: $improved_count,
total_passed_count: $total_passed_count,
failed_count: $failed_count,
expected_failures_count: $expected_failures_count,
not_planned_count: $not_planned_count,
unsupported_count: $unsupported_count,
unsupported_navigation_lookups_count: $unsupported_navigation_lookups_count,
skipped_count: $skipped_count,
errors_count: $errors_count
}'
)"
{
echo 'payload<<EOF'
echo "$payload"
echo 'EOF'
} >> "$GITHUB_OUTPUT"
- name: Enforce report completeness
if: always()
shell: bash
env:
REPORT_PATH: ${{ steps.paths.outputs.report_path }}
REPORT_COMPLETE: ${{ steps.paths.outputs.report_complete }}
run: |
set -euo pipefail
if [ -z "$REPORT_PATH" ] || [ ! -f "$REPORT_PATH" ]; then
echo "usage benchmark run did not write a report" >&2
exit 1
fi
if [ "$REPORT_COMPLETE" != "true" ]; then
echo "usage benchmark report is incomplete" >&2
exit 1
fi
- name: Enforce benchmark outcome
if: always()
shell: bash
env:
RUN_OUTCOME: ${{ steps.run_usagebench.outcome }}
REPORT_PATH: ${{ steps.paths.outputs.report_path }}
REPORT_COMPLETE: ${{ steps.paths.outputs.report_complete }}
run: |
set -euo pipefail
if [ -z "$REPORT_PATH" ] || [ ! -f "$REPORT_PATH" ] || [ "$REPORT_COMPLETE" != "true" ]; then
echo "skipping benchmark outcome enforcement because no complete report exists"
exit 0
fi
errors_count="$(jq -r '.totals.errors // 0' "$REPORT_PATH")"
if [ "$errors_count" != "0" ]; then
echo "usage benchmark completed with ${errors_count} runner error(s)" >&2
exit 1
fi
if [ "$RUN_OUTCOME" != "success" ]; then
echo "usage benchmark completed with semantic failures" >&2
exit 1
fi
notify:
name: notify Slack
needs: benchmark
if: >-
always() &&
needs.benchmark.outputs.slack_payload != '' &&
(github.event_name == 'schedule' || inputs.notify_slack)
runs-on: ubuntu-latest
permissions: {}
timeout-minutes: 5
steps:
- name: Send benchmark data to Slack Workflow
continue-on-error: true
uses: slackapi/slack-github-action@0d95c9a7becc1e6e297d76df9bc735c44f4cbcbc # v3.0.5
env:
SLACK_WEBHOOK: ${{ secrets.SLACK_DAILY_USAGEBENCH_WEBHOOK_URL }}
with:
webhook: ${{ env.SLACK_WEBHOOK }}
webhook-type: webhook-trigger
payload: ${{ needs.benchmark.outputs.slack_payload }}