Benchmark #90
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: Benchmark | |
| on: | |
| schedule: | |
| - cron: '47 9 * * *' | |
| workflow_dispatch: | |
| inputs: | |
| bifrost_ref: | |
| description: 'Bifrost ref or commit to benchmark' | |
| required: false | |
| default: 'master' | |
| type: string | |
| include_unsupported: | |
| description: 'Run cases marked unsupported' | |
| required: false | |
| default: false | |
| type: boolean | |
| notify_slack: | |
| description: 'Post the benchmark summary to Slack' | |
| required: false | |
| default: false | |
| type: boolean | |
| permissions: {} | |
| concurrency: | |
| group: usagebench-${{ github.ref }} | |
| cancel-in-progress: true | |
| jobs: | |
| benchmark: | |
| if: github.ref == 'refs/heads/main' | |
| runs-on: ubuntu-latest | |
| permissions: | |
| contents: read | |
| # Keep an explicit overall ceiling while reserving 130 minutes around the | |
| # benchmark step for cold checkout/build work and post-timeout artifact | |
| # publication. | |
| timeout-minutes: 300 | |
| outputs: | |
| slack_payload: ${{ steps.slack_payload.outputs.payload }} | |
| env: | |
| BIFROST_REF: ${{ inputs.bifrost_ref || 'master' }} | |
| steps: | |
| - name: Checkout usagebench | |
| uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5.1.0 | |
| with: | |
| fetch-depth: 0 | |
| lfs: true | |
| persist-credentials: false | |
| - name: Validate Bifrost ref | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| git check-ref-format --allow-onelevel "$BIFROST_REF" >/dev/null || { | |
| echo "Bifrost ref is not a valid Git ref: $BIFROST_REF" >&2 | |
| exit 1 | |
| } | |
| - name: Checkout Bifrost | |
| uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5.1.0 | |
| with: | |
| repository: BrokkAi/bifrost | |
| # This path is already ignored by UsageBench, so the analyzer checkout | |
| # does not make report provenance appear dirty. | |
| path: .bifrost | |
| ref: ${{ env.BIFROST_REF }} | |
| fetch-depth: 0 | |
| lfs: false | |
| persist-credentials: false | |
| - name: Install Rust toolchain | |
| uses: dtolnay/rust-toolchain@4cda84d5c5c54efe2404f9d843567869ab1699d4 # stable | |
| with: | |
| toolchain: stable | |
| - name: Cache cargo + target | |
| uses: swatinem/rust-cache@e18b497796c12c097a38f9edb9d0641fb99eee32 # v2 | |
| - name: Build usagebench | |
| run: cargo build --locked --bin usagebench | |
| - name: Validate benchmark cases | |
| run: ./target/debug/usagebench validate benchmarks/cases | |
| - name: Run Bifrost usage benchmark | |
| id: run_usagebench | |
| continue-on-error: true | |
| # Expire well before the job so checkpoint discovery, artifact upload, | |
| # and summaries still run after a slow benchmark is terminated. | |
| timeout-minutes: 170 | |
| shell: bash | |
| env: | |
| INCLUDE_UNSUPPORTED: ${{ inputs.include_unsupported || false }} | |
| run: | | |
| set -euo pipefail | |
| mkdir -p benchmark-output | |
| args=( | |
| run-bifrost | |
| benchmarks/cases | |
| --bifrost-repo .bifrost | |
| --bifrost-commit "$BIFROST_REF" | |
| --bifrost-working-tree | |
| --work-dir target/usagebench | |
| --scan-usages-max-duration-secs 300 | |
| --expected-passes benchmarks/expectations/bifrost-expected-passes.yaml | |
| --output "benchmark-output/run-${GITHUB_RUN_ID}.json" | |
| ) | |
| if [ "$INCLUDE_UNSUPPORTED" = "true" ]; then | |
| args+=(--include-unsupported) | |
| fi | |
| ./target/debug/usagebench "${args[@]}" | |
| - name: Resolve benchmark paths | |
| id: paths | |
| if: always() | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| report_path="benchmark-output/run-${GITHUB_RUN_ID}.json" | |
| report_complete="true" | |
| partial_path="benchmark-output/run-${GITHUB_RUN_ID}.partial.json" | |
| if [ -f "$report_path" ]; then | |
| report_complete="$(jq -r 'if has("completed") then .completed else true end' "$report_path")" | |
| elif [ -f "$partial_path" ] && jq -e '.completed == false' "$partial_path" >/dev/null; then | |
| report_path="$partial_path" | |
| report_complete="false" | |
| elif [ ! -f "$report_path" ]; then | |
| report_path="" | |
| report_complete="false" | |
| fi | |
| { | |
| echo "report_path=${report_path}" | |
| echo "report_complete=${report_complete}" | |
| } >> "$GITHUB_OUTPUT" | |
| - name: Record UsageBench provenance | |
| id: usagebench_provenance | |
| if: always() | |
| shell: bash | |
| env: | |
| REPORT_PATH: ${{ steps.paths.outputs.report_path }} | |
| run: | | |
| set -euo pipefail | |
| report_path="$REPORT_PATH" | |
| revision="${GITHUB_SHA}" | |
| release="" | |
| if [ -n "$report_path" ] && [ -f "$report_path" ]; then | |
| revision="$(jq -r '.usagebenchRevision // ""' "$report_path")" | |
| release="$(jq -r '.usagebenchRelease // ""' "$report_path")" | |
| fi | |
| if [[ ! "$revision" =~ ^[0-9a-fA-F]{40}(-dirty)?$ ]]; then | |
| echo "Ignoring invalid UsageBench revision from report" >&2 | |
| revision="$GITHUB_SHA" | |
| release="" | |
| fi | |
| if [[ -n "$release" && ! "$release" =~ ^v[0-9]+\.[0-9]+\.[0-9]+$ ]]; then | |
| echo "Ignoring invalid UsageBench release from report" >&2 | |
| release="" | |
| fi | |
| artifact_label="${release:-${revision:0:12}}" | |
| { | |
| echo "revision=$revision" | |
| echo "release=$release" | |
| echo "artifact_label=$artifact_label" | |
| } >> "$GITHUB_OUTPUT" | |
| - name: Upload benchmark artifacts | |
| if: always() | |
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | |
| with: | |
| name: usagebench-${{ steps.usagebench_provenance.outputs.artifact_label }}-${{ github.run_id }} | |
| path: benchmark-output | |
| if-no-files-found: warn | |
| retention-days: 14 | |
| - name: Publish benchmark summary | |
| if: always() | |
| shell: bash | |
| env: | |
| RUN_OUTCOME: ${{ steps.run_usagebench.outcome }} | |
| USAGEBENCH_REVISION: ${{ steps.usagebench_provenance.outputs.revision }} | |
| USAGEBENCH_RELEASE: ${{ steps.usagebench_provenance.outputs.release }} | |
| REPORT_PATH: ${{ steps.paths.outputs.report_path }} | |
| REPORT_COMPLETE: ${{ steps.paths.outputs.report_complete }} | |
| run: | | |
| set -euo pipefail | |
| { | |
| echo "# Usage Benchmark" | |
| echo | |
| echo "- Run step: $RUN_OUTCOME" | |
| echo "- UsageBench revision: \`$USAGEBENCH_REVISION\`" | |
| if [ -n "$USAGEBENCH_RELEASE" ]; then | |
| echo "- UsageBench release: \`$USAGEBENCH_RELEASE\`" | |
| fi | |
| echo "- Bifrost ref: \`${BIFROST_REF}\`" | |
| echo "- Complete report: \`${REPORT_COMPLETE}\`" | |
| if [ -n "$REPORT_PATH" ]; then | |
| echo "- Report: \`$REPORT_PATH\`" | |
| echo | |
| scripts/render-benchmark-summary.sh "$REPORT_PATH" | |
| else | |
| echo "- Report: none written" | |
| fi | |
| } >> "$GITHUB_STEP_SUMMARY" | |
| - name: Prepare Slack payload | |
| id: slack_payload | |
| if: always() | |
| continue-on-error: true | |
| shell: bash | |
| env: | |
| RUN_OUTCOME: ${{ steps.run_usagebench.outcome }} | |
| REPORT_PATH: ${{ steps.paths.outputs.report_path }} | |
| REPORT_COMPLETE: ${{ steps.paths.outputs.report_complete }} | |
| WORKFLOW_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} | |
| USAGEBENCH_REVISION: ${{ steps.usagebench_provenance.outputs.revision }} | |
| USAGEBENCH_RELEASE: ${{ steps.usagebench_provenance.outputs.release }} | |
| run: | | |
| set -euo pipefail | |
| run_outcome="$RUN_OUTCOME" | |
| report_path="$REPORT_PATH" | |
| report_complete="$REPORT_COMPLETE" | |
| head_sha_short="$(git rev-parse --short "$GITHUB_SHA")" | |
| bifrost_resolved_commit="" | |
| bifrost_sha_short="" | |
| cases_count="0" | |
| processed_documents_count="0" | |
| requested_documents_count="0" | |
| processed_cases_count="0" | |
| requested_authored_cases_count="0" | |
| requested_planned_cases_count="0" | |
| passed_count="0" | |
| near_misses_count="0" | |
| position_unverified_count="0" | |
| improved_count="0" | |
| total_passed_count="0" | |
| failed_count="0" | |
| expected_failures_count="0" | |
| not_planned_count="0" | |
| unsupported_count="0" | |
| unsupported_navigation_lookups_count="0" | |
| skipped_count="0" | |
| errors_count="0" | |
| failing_cases="" | |
| if [ -n "$report_path" ] && [ -f "$report_path" ]; then | |
| bifrost_resolved_commit="$(jq -r '.bifrostResolvedCommit // ""' "$report_path")" | |
| bifrost_sha_short="$(printf '%s' "$bifrost_resolved_commit" | cut -c1-7)" | |
| report_counts="$(scripts/render-benchmark-summary.sh --counts "$report_path")" | |
| report_complete="$(jq -r '.report_complete' <<< "$report_counts")" | |
| processed_documents_count="$(jq -r '.processed_documents_count' <<< "$report_counts")" | |
| requested_documents_count="$(jq -r '.requested_documents_count' <<< "$report_counts")" | |
| processed_cases_count="$(jq -r '.processed_cases_count' <<< "$report_counts")" | |
| requested_authored_cases_count="$(jq -r '.requested_authored_cases_count' <<< "$report_counts")" | |
| requested_planned_cases_count="$(jq -r '.requested_planned_cases_count' <<< "$report_counts")" | |
| cases_count="$requested_planned_cases_count" | |
| passed_count="$(jq -r '.totals.passed // 0' "$report_path")" | |
| near_misses_count="$(jq -r '.totals.nearMisses // 0' "$report_path")" | |
| position_unverified_count="$(jq -r '.totals.positionUnverified // 0' "$report_path")" | |
| improved_count="$(jq -r '.totals.improved // 0' "$report_path")" | |
| total_passed_count="$((passed_count + improved_count))" | |
| failed_count="$(jq -r '.totals.failed // 0' "$report_path")" | |
| expected_failures_count="$(jq -r '.totals.expectedFailures // 0' "$report_path")" | |
| not_planned_count="$(jq -r '.totals.notPlanned // 0' "$report_path")" | |
| unsupported_count="$(jq -r '.totals.unsupported // 0' "$report_path")" | |
| unsupported_navigation_lookups_count="$( | |
| jq '[.documents[]?.cases[]?.usageToDeclaration[]? | select(.status == "unsupported")] | length' "$report_path" | |
| )" | |
| skipped_count="$(jq -r '.totals.skipped // 0' "$report_path")" | |
| errors_count="$(jq -r '.totals.errors // 0' "$report_path")" | |
| failing_cases="$( | |
| scripts/render-benchmark-summary.sh --failing-cases "$report_path" | |
| )" | |
| fi | |
| ok="true" | |
| error_text="" | |
| if [ -z "$report_path" ] || [ ! -f "$report_path" ]; then | |
| ok="false" | |
| error_text="usage benchmark run did not write a report" | |
| elif [ "$report_complete" != "true" ]; then | |
| ok="false" | |
| requested_planned_display="$requested_planned_cases_count" | |
| if [ "$requested_planned_display" = "null" ]; then | |
| requested_planned_display="unknown" | |
| fi | |
| error_text="usage benchmark report is INCOMPLETE: processed ${processed_documents_count}/${requested_documents_count} documents and ${processed_cases_count}/${requested_planned_display} planned cases" | |
| elif [ "$errors_count" != "0" ]; then | |
| ok="false" | |
| if [ -n "$failing_cases" ]; then | |
| error_text="$(printf 'Usage benchmark completed with runner errors:\n%s' "$failing_cases")" | |
| else | |
| error_text="usage benchmark completed with runner errors" | |
| fi | |
| elif [ "$run_outcome" != "success" ] || [ "$failed_count" != "0" ]; then | |
| ok="false" | |
| if [ -n "$failing_cases" ]; then | |
| error_text="$(printf 'Usage benchmark recorded semantic failures:\n%s' "$failing_cases")" | |
| else | |
| error_text="usage benchmark recorded semantic failures" | |
| fi | |
| fi | |
| payload="$( | |
| jq -nc \ | |
| --arg error_text "$error_text" \ | |
| --arg workflow_run_url "$WORKFLOW_RUN_URL" \ | |
| --arg head_sha_short "$head_sha_short" \ | |
| --arg usagebench_revision "$USAGEBENCH_REVISION" \ | |
| --arg usagebench_release "$USAGEBENCH_RELEASE" \ | |
| --arg bifrost_ref "$BIFROST_REF" \ | |
| --arg bifrost_sha_short "$bifrost_sha_short" \ | |
| --arg run_outcome "$run_outcome" \ | |
| --argjson report_complete "$report_complete" \ | |
| --argjson ok "$ok" \ | |
| --argjson cases_count "$cases_count" \ | |
| --argjson processed_documents_count "$processed_documents_count" \ | |
| --argjson requested_documents_count "$requested_documents_count" \ | |
| --argjson processed_cases_count "$processed_cases_count" \ | |
| --argjson requested_authored_cases_count "$requested_authored_cases_count" \ | |
| --argjson requested_planned_cases_count "$requested_planned_cases_count" \ | |
| --argjson passed_count "$passed_count" \ | |
| --argjson near_misses_count "$near_misses_count" \ | |
| --argjson position_unverified_count "$position_unverified_count" \ | |
| --argjson improved_count "$improved_count" \ | |
| --argjson total_passed_count "$total_passed_count" \ | |
| --argjson failed_count "$failed_count" \ | |
| --argjson expected_failures_count "$expected_failures_count" \ | |
| --argjson not_planned_count "$not_planned_count" \ | |
| --argjson unsupported_count "$unsupported_count" \ | |
| --argjson unsupported_navigation_lookups_count "$unsupported_navigation_lookups_count" \ | |
| --argjson skipped_count "$skipped_count" \ | |
| --argjson errors_count "$errors_count" \ | |
| '{ | |
| ok: $ok, | |
| error_text: $error_text, | |
| workflow_run_url: $workflow_run_url, | |
| head_sha_short: $head_sha_short, | |
| usagebench_revision: $usagebench_revision, | |
| usagebench_release: $usagebench_release, | |
| bifrost_ref: $bifrost_ref, | |
| bifrost_sha_short: $bifrost_sha_short, | |
| run_outcome: $run_outcome, | |
| report_complete: $report_complete, | |
| cases_count: $cases_count, | |
| processed_documents_count: $processed_documents_count, | |
| requested_documents_count: $requested_documents_count, | |
| processed_cases_count: $processed_cases_count, | |
| requested_authored_cases_count: $requested_authored_cases_count, | |
| requested_planned_cases_count: $requested_planned_cases_count, | |
| passed_count: $passed_count, | |
| near_misses_count: $near_misses_count, | |
| position_unverified_count: $position_unverified_count, | |
| improved_count: $improved_count, | |
| total_passed_count: $total_passed_count, | |
| failed_count: $failed_count, | |
| expected_failures_count: $expected_failures_count, | |
| not_planned_count: $not_planned_count, | |
| unsupported_count: $unsupported_count, | |
| unsupported_navigation_lookups_count: $unsupported_navigation_lookups_count, | |
| skipped_count: $skipped_count, | |
| errors_count: $errors_count | |
| }' | |
| )" | |
| { | |
| echo 'payload<<EOF' | |
| echo "$payload" | |
| echo 'EOF' | |
| } >> "$GITHUB_OUTPUT" | |
| - name: Enforce report completeness | |
| if: always() | |
| shell: bash | |
| env: | |
| REPORT_PATH: ${{ steps.paths.outputs.report_path }} | |
| REPORT_COMPLETE: ${{ steps.paths.outputs.report_complete }} | |
| run: | | |
| set -euo pipefail | |
| if [ -z "$REPORT_PATH" ] || [ ! -f "$REPORT_PATH" ]; then | |
| echo "usage benchmark run did not write a report" >&2 | |
| exit 1 | |
| fi | |
| if [ "$REPORT_COMPLETE" != "true" ]; then | |
| echo "usage benchmark report is incomplete" >&2 | |
| exit 1 | |
| fi | |
| - name: Enforce benchmark outcome | |
| if: always() | |
| shell: bash | |
| env: | |
| RUN_OUTCOME: ${{ steps.run_usagebench.outcome }} | |
| REPORT_PATH: ${{ steps.paths.outputs.report_path }} | |
| REPORT_COMPLETE: ${{ steps.paths.outputs.report_complete }} | |
| run: | | |
| set -euo pipefail | |
| if [ -z "$REPORT_PATH" ] || [ ! -f "$REPORT_PATH" ] || [ "$REPORT_COMPLETE" != "true" ]; then | |
| echo "skipping benchmark outcome enforcement because no complete report exists" | |
| exit 0 | |
| fi | |
| errors_count="$(jq -r '.totals.errors // 0' "$REPORT_PATH")" | |
| if [ "$errors_count" != "0" ]; then | |
| echo "usage benchmark completed with ${errors_count} runner error(s)" >&2 | |
| exit 1 | |
| fi | |
| if [ "$RUN_OUTCOME" != "success" ]; then | |
| echo "usage benchmark completed with semantic failures" >&2 | |
| exit 1 | |
| fi | |
| notify: | |
| name: notify Slack | |
| needs: benchmark | |
| if: >- | |
| always() && | |
| needs.benchmark.outputs.slack_payload != '' && | |
| (github.event_name == 'schedule' || inputs.notify_slack) | |
| runs-on: ubuntu-latest | |
| permissions: {} | |
| timeout-minutes: 5 | |
| steps: | |
| - name: Send benchmark data to Slack Workflow | |
| continue-on-error: true | |
| uses: slackapi/slack-github-action@0d95c9a7becc1e6e297d76df9bc735c44f4cbcbc # v3.0.5 | |
| env: | |
| SLACK_WEBHOOK: ${{ secrets.SLACK_DAILY_USAGEBENCH_WEBHOOK_URL }} | |
| with: | |
| webhook: ${{ env.SLACK_WEBHOOK }} | |
| webhook-type: webhook-trigger | |
| payload: ${{ needs.benchmark.outputs.slack_payload }} |