feat: attach raw agent transcripts to Braintrust eval runs #1858
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: Refresh eval results | |
| on: | |
| workflow_dispatch: | |
| inputs: | |
| experiments: | |
| description: "Comma-separated experiment names to run (blank to auto-discover from experiment_suite)" | |
| required: false | |
| default: "" | |
| eval: | |
| description: "Optional comma-separated eval ids to run" | |
| required: false | |
| default: "" | |
| suite: | |
| description: "Comma-separated eval suites to run" | |
| required: true | |
| default: "benchmark" | |
| experiment_suite: | |
| description: "Comma-separated experiment suites to run" | |
| required: true | |
| default: "benchmark,no-skills" | |
| runs: | |
| description: "Independent scored runs per experiment/eval pair" | |
| required: true | |
| default: "3" | |
| timeout_sec: | |
| description: "Timeout per run in seconds" | |
| required: true | |
| default: "720" | |
| sandbox_concurrency: | |
| description: "Maximum Vercel Sandboxes running at once" | |
| required: true | |
| default: "250" | |
| merge: | |
| description: "Merge into existing results instead of overwriting (graft new experiment/eval pairs)" | |
| type: boolean | |
| required: false | |
| default: false | |
| commit_to_branch: | |
| description: "Commit exported results to the dispatched branch instead of opening a PR" | |
| type: boolean | |
| required: false | |
| default: false | |
| cli_stable_version: | |
| description: "Pin the cli suite's stable CLI version instead of resolving npm's latest dist-tag (blank to resolve)" | |
| required: false | |
| default: "" | |
| cli_beta_version: | |
| description: "Pin the cli suite's beta CLI version instead of resolving npm's beta dist-tag (blank to resolve)" | |
| required: false | |
| default: "" | |
| schedule: | |
| - cron: '15 6 * * *' | |
| pull_request: | |
| # Run whenever a PR carrying the run-evals label is opened, pushed to, or | |
| # receives the run-evals label. The job-level `if` gates on those cases. | |
| types: [opened, synchronize, labeled] | |
| permissions: | |
| contents: write | |
| pull-requests: write | |
| actions: read | |
| concurrency: | |
| group: eval-refresh-${{ github.event_name == 'pull_request' && github.event.pull_request.number || github.ref }} | |
| # Supersede an in-flight run when a new commit is pushed to the same PR, but | |
| # never cancel a workflow_dispatch run (those open the refresh PR). | |
| cancel-in-progress: ${{ github.event_name == 'pull_request' }} | |
| jobs: | |
| prepare: | |
| if: >- | |
| github.event_name == 'workflow_dispatch' || | |
| github.event_name == 'schedule' || | |
| (github.event_name == 'pull_request' && | |
| (contains(github.event.pull_request.labels.*.name, 'run-evals') || | |
| contains(github.event.pull_request.labels.*.name, 'run-evals-changed')) && | |
| github.event.pull_request.head.repo.full_name == github.repository && | |
| (github.event.action != 'labeled' || | |
| github.event.label.name == 'run-evals' || | |
| github.event.label.name == 'run-evals-changed')) | |
| runs-on: ubuntu-latest | |
| outputs: | |
| pairs: ${{ steps.discover.outputs.pairs }} | |
| runs: ${{ steps.inputs.outputs.runs }} | |
| timeout_sec: ${{ steps.inputs.outputs.timeout_sec }} | |
| sandbox_concurrency: ${{ steps.inputs.outputs.sandbox_concurrency }} | |
| filter_changed: ${{ steps.inputs.outputs.filter_changed }} | |
| do_merge: ${{ steps.inputs.outputs.do_merge }} | |
| suite: ${{ steps.inputs.outputs.suite }} | |
| cli_stable_version: ${{ steps.inputs.outputs.cli_stable_version }} | |
| cli_beta_version: ${{ steps.inputs.outputs.cli_beta_version }} | |
| steps: | |
| - name: Prepare inputs | |
| id: inputs | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then | |
| experiments_override="${{ inputs.experiments }}" | |
| eval_id="${{ inputs.eval }}" | |
| suite="${{ inputs.suite }}" | |
| experiment_suite="${{ inputs.experiment_suite }}" | |
| runs="${{ inputs.runs }}" | |
| timeout_sec="${{ inputs.timeout_sec }}" | |
| sandbox_concurrency="${{ inputs.sandbox_concurrency }}" | |
| cli_stable_version="${{ inputs.cli_stable_version }}" | |
| cli_beta_version="${{ inputs.cli_beta_version }}" | |
| elif [ "${{ github.event_name }}" = "schedule" ]; then | |
| experiments_override="" | |
| eval_id="" | |
| suite="regression,cli" | |
| experiment_suite="regression,cli" | |
| runs="3" | |
| timeout_sec="720" | |
| sandbox_concurrency="250" | |
| cli_stable_version="" | |
| cli_beta_version="" | |
| else | |
| experiments_override="" | |
| eval_id="" | |
| suite="benchmark,regression,docs,cli" | |
| experiment_suite="benchmark,no-skills,regression,docs,cli" | |
| runs="3" | |
| timeout_sec="720" | |
| sandbox_concurrency="250" | |
| cli_stable_version="" | |
| cli_beta_version="" | |
| fi | |
| suite_json="$(jq -Rc 'split(",") | map(gsub("^\\s+|\\s+$"; "")) | map(select(length > 0))' <<< "$suite")" | |
| experiment_suite_json="$(jq -Rc 'split(",") | map(gsub("^\\s+|\\s+$"; "")) | map(select(length > 0))' <<< "$experiment_suite")" | |
| # run-evals takes priority; run-evals-changed only filters when run-evals is absent. | |
| # filter_changed drives the changed-eval filter (PR-only, needs a PR diff). | |
| filter_changed="false" | |
| if [ "${{ github.event_name }}" = "pull_request" ] && \ | |
| [ "${{ contains(github.event.pull_request.labels.*.name, 'run-evals-changed') }}" = "true" ] && \ | |
| [ "${{ contains(github.event.pull_request.labels.*.name, 'run-evals') }}" = "false" ]; then | |
| filter_changed="true" | |
| fi | |
| # do_merge drives the export --merge (graft into existing results). It's | |
| # always on for the changed path, and opt-in for manual dispatch. | |
| do_merge="$filter_changed" | |
| if [ "${{ github.event_name }}" = "workflow_dispatch" ] && [ "${{ inputs.merge }}" = "true" ]; then | |
| do_merge="true" | |
| fi | |
| { | |
| echo "experiments_override=$experiments_override" | |
| echo "eval=$eval_id" | |
| echo "suite=$suite_json" | |
| echo "experiment_suite=$experiment_suite_json" | |
| echo "runs=$runs" | |
| echo "timeout_sec=$timeout_sec" | |
| echo "sandbox_concurrency=$sandbox_concurrency" | |
| echo "filter_changed=$filter_changed" | |
| echo "do_merge=$do_merge" | |
| echo "cli_stable_version=$cli_stable_version" | |
| echo "cli_beta_version=$cli_beta_version" | |
| } >> "$GITHUB_OUTPUT" | |
| - name: Checkout | |
| uses: actions/checkout@9f698171ed81b15d1823a05fc7211befd50c8ae0 # v6.0.3 | |
| with: | |
| ref: ${{ github.event_name == 'pull_request' && github.head_ref || github.ref }} | |
| - name: Install pnpm | |
| uses: pnpm/action-setup@fc06bc1257f339d1d5d8b3a19a8cae5388b55320 # v5.0.0 | |
| - name: Setup Node.js | |
| uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 | |
| with: | |
| node-version-file: .node-version | |
| cache: pnpm | |
| - name: Install dependencies | |
| run: pnpm install --frozen-lockfile | |
| - name: Discover eval pairs | |
| id: discover | |
| env: | |
| GH_TOKEN: ${{ github.token }} | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| touch .env | |
| suite_json='${{ steps.inputs.outputs.suite }}' | |
| eval_ids="${{ steps.inputs.outputs.eval }}" | |
| if [ -n "$eval_ids" ]; then | |
| matching=() | |
| while IFS= read -r id; do | |
| compgen -G "evals/*/$id" > /dev/null && matching+=("$id") | |
| done < <(jq -Rr 'split(",") | map(gsub("^\\s+|\\s+$"; "")) | .[]' <<< "$eval_ids") | |
| else | |
| matching=() | |
| for dir in evals/*/*/; do | |
| [ -d "$dir" ] || continue | |
| id=$(basename "$dir") | |
| [ -f "$dir/PROMPT.md" ] || continue | |
| suite_val=$(basename "$(dirname "$dir")") | |
| if jq -e --arg s "$suite_val" 'index($s) != null' <<< "$suite_json" > /dev/null 2>&1; then | |
| matching+=("$id") | |
| fi | |
| done | |
| fi | |
| if [ "${{ steps.inputs.outputs.filter_changed }}" = "true" ]; then | |
| changed_evals=$(gh pr diff ${{ github.event.pull_request.number }} --name-only \ | |
| | grep '^evals/' | cut -d/ -f3 | sort -u || true) | |
| if [ -z "$changed_evals" ]; then | |
| echo "No eval directories changed in this PR — nothing to run" | |
| echo "pairs=[]" >> "$GITHUB_OUTPUT" | |
| exit 0 | |
| fi | |
| echo "Changed eval dirs: $changed_evals" | |
| filtered=() | |
| for id in "${matching[@]}"; do | |
| if echo "$changed_evals" | grep -qx "$id"; then | |
| filtered+=("$id") | |
| fi | |
| done | |
| matching=("${filtered[@]+"${filtered[@]}"}") | |
| fi | |
| if [ "${#matching[@]}" -eq 0 ]; then | |
| echo "Changed evals don't match requested suites: $suite_json" | |
| echo "pairs=[]" >> "$GITHUB_OUTPUT" | |
| exit 0 | |
| fi | |
| pairs='[]' | |
| experiments_override="${{ steps.inputs.outputs.experiments_override }}" | |
| for id in "${matching[@]}"; do | |
| eval_suite=$(basename "$(dirname "$(compgen -G "evals/*/$id")")") | |
| case "$eval_suite" in | |
| benchmark) experiment_suites=(benchmark no-skills) ;; | |
| regression) experiment_suites=(regression) ;; | |
| docs) experiment_suites=(docs) ;; | |
| cli) experiment_suites=(cli) ;; | |
| *) continue ;; | |
| esac | |
| for experiment_suite in "${experiment_suites[@]}"; do | |
| if ! jq -e --arg suite "$experiment_suite" 'index($suite) != null' \ | |
| <<< '${{ steps.inputs.outputs.experiment_suite }}' > /dev/null; then | |
| continue | |
| fi | |
| if [ -n "$experiments_override" ]; then | |
| experiments_json="$(jq -Rc 'split(",") | map(gsub("^\\s+|\\s+$"; "")) | map(select(length > 0))' <<< "$experiments_override")" | |
| else | |
| experiments_json="$(pnpm --silent eval -- list --experiment-suite "$experiment_suite" --eval "$id")" | |
| fi | |
| while IFS= read -r experiment; do | |
| pairs="$(jq -c \ | |
| --arg eval_id "$id" \ | |
| --arg experiment "$experiment" \ | |
| --arg experiment_suite "$experiment_suite" \ | |
| --arg eval_suite "$eval_suite" \ | |
| '. + [{eval_id: $eval_id, experiment: $experiment, experiment_suite: $experiment_suite, eval_suite: $eval_suite}]' \ | |
| <<< "$pairs")" | |
| done < <(jq -r '.[]' <<< "$experiments_json") | |
| done | |
| done | |
| if [ "$(jq 'length' <<< "$pairs")" -eq 0 ]; then | |
| echo "No experiment and eval pairs matched" >&2 | |
| exit 1 | |
| fi | |
| echo "pairs=$pairs" >> "$GITHUB_OUTPUT" | |
| run-evals: | |
| needs: prepare | |
| if: needs.prepare.outputs.pairs != '[]' | |
| runs-on: ubuntu-latest | |
| # 60 min gives ~2x margin over the slowest observed run: | |
| # https://github.com/supabase/evals/actions/runs/31802584549 | |
| timeout-minutes: 60 | |
| env: | |
| ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} | |
| OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} | |
| AI_GATEWAY_API_KEY: ${{ secrets.AI_GATEWAY_API_KEY }} | |
| XAI_API_KEY: ${{ secrets.XAI_API_KEY }} | |
| VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }} | |
| VERCEL_TEAM_ID: ${{ secrets.VERCEL_TEAM_ID }} | |
| VERCEL_PROJECT_ID: ${{ secrets.VERCEL_PROJECT_ID }} | |
| SUPABASE_CLI_STABLE_VERSION: ${{ needs.prepare.outputs.cli_stable_version }} | |
| SUPABASE_CLI_BETA_VERSION: ${{ needs.prepare.outputs.cli_beta_version }} | |
| steps: | |
| - name: Checkout | |
| uses: actions/checkout@9f698171ed81b15d1823a05fc7211befd50c8ae0 # v6.0.3 | |
| with: | |
| ref: ${{ github.event_name == 'pull_request' && github.head_ref || github.ref }} | |
| - name: Install pnpm | |
| uses: pnpm/action-setup@fc06bc1257f339d1d5d8b3a19a8cae5388b55320 # v5.0.0 | |
| - name: Setup Node.js | |
| uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 | |
| with: | |
| node-version-file: .node-version | |
| cache: pnpm | |
| - name: Install dependencies | |
| run: pnpm install --frozen-lockfile | |
| - name: Write eval environment | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| { | |
| echo "ANTHROPIC_API_KEY=${ANTHROPIC_API_KEY}" | |
| echo "OPENAI_API_KEY=${OPENAI_API_KEY}" | |
| echo "AI_GATEWAY_API_KEY=${AI_GATEWAY_API_KEY}" | |
| echo "XAI_API_KEY=${XAI_API_KEY}" | |
| echo "SUPABASE_CLI_STABLE_VERSION=${SUPABASE_CLI_STABLE_VERSION}" | |
| echo "SUPABASE_CLI_BETA_VERSION=${SUPABASE_CLI_BETA_VERSION}" | |
| } > .env | |
| - name: Run evals | |
| env: | |
| EVAL_PAIRS: ${{ needs.prepare.outputs.pairs }} | |
| EVAL_REVISION: ${{ github.event_name == 'pull_request' && github.event.pull_request.head.sha || github.sha }} | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| pnpm --filter @supabase-evals/framework eval:vercel -- \ | |
| --pairs-json "$EVAL_PAIRS" \ | |
| --revision "$EVAL_REVISION" \ | |
| --runs "${{ needs.prepare.outputs.runs }}" \ | |
| --timeout-sec "${{ needs.prepare.outputs.timeout_sec }}" \ | |
| --concurrency "${{ needs.prepare.outputs.sandbox_concurrency }}" | |
| - name: Upload raw results | |
| if: always() | |
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | |
| with: | |
| name: raw-results | |
| path: results/downloaded/ | |
| overwrite: true | |
| if-no-files-found: warn | |
| retention-days: 3 | |
| publish-results: | |
| needs: [prepare, run-evals] | |
| # Runs even after run-evals times out or partially fails, since results | |
| # are written per pair and export-results skips any pair that never finished. | |
| if: "always() && needs.run-evals.result != 'skipped' && needs.prepare.outputs.pairs != '[]'" | |
| runs-on: ubuntu-latest | |
| steps: | |
| - name: Generate GitHub App token | |
| id: generate-token | |
| # A GitHub App token is needed so the push below can trigger the | |
| # gh-pages workflow: GITHUB_TOKEN pushes don't trigger other workflows. | |
| # https://docs.github.com/en/actions/how-tos/write-workflows/choose-when-workflows-run/trigger-a-workflow#triggering-a-workflow-from-a-workflow | |
| if: >- | |
| github.event_name == 'schedule' || | |
| (github.event_name == 'workflow_dispatch' && !inputs.commit_to_branch) | |
| uses: actions/create-github-app-token@bcd2ba49218906704ab6c1aa796996da409d3eb1 # v3.2.0 | |
| with: | |
| app-id: ${{ secrets.GH_APP_ID }} | |
| private-key: ${{ secrets.GH_APP_PRIVATE_KEY }} | |
| - name: Checkout | |
| uses: actions/checkout@9f698171ed81b15d1823a05fc7211befd50c8ae0 # v6.0.3 | |
| with: | |
| ref: ${{ github.event_name == 'pull_request' && github.head_ref || github.ref }} | |
| token: ${{ steps.generate-token.outputs.token || github.token }} | |
| - name: Install pnpm | |
| uses: pnpm/action-setup@fc06bc1257f339d1d5d8b3a19a8cae5388b55320 # v5.0.0 | |
| - name: Setup Node.js | |
| uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 | |
| with: | |
| node-version-file: .node-version | |
| cache: pnpm | |
| - name: Install dependencies | |
| run: pnpm install --frozen-lockfile | |
| - name: Download raw results | |
| # If every pair failed, run-evals never uploads a raw-results artifact | |
| # at all, so this action errors. Audit pair results below already | |
| # handles a missing/empty results/downloaded correctly (every pair | |
| # reports missing), so don't let a total failure hide behind this. | |
| continue-on-error: true | |
| uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 | |
| with: | |
| name: raw-results | |
| path: results/downloaded | |
| - name: Extract raw workspaces | |
| # Workspaces arrive tarred so upload-artifact never sees agent-written filenames. | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| if [ -d results/downloaded ]; then | |
| find results/downloaded -name 'workspace.tgz' -print0 | while IFS= read -r -d '' archive; do | |
| workspace="$(dirname "$archive")/workspace" | |
| mkdir -p "$workspace" | |
| tar -xzf "$archive" -C "$workspace" | |
| rm "$archive" | |
| done | |
| fi | |
| - name: Audit pair results | |
| id: audit | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| pairs='${{ needs.prepare.outputs.pairs }}' | |
| total="$(jq 'length' <<< "$pairs")" | |
| # A pair needs every requested run, not just a directory, since a | |
| # partial pair would otherwise skew the average against its neighbours. | |
| expected="${{ needs.prepare.outputs.runs }}" | |
| required="$(jq -cn --argjson expected "$expected" '[range(1; $expected + 1)]')" | |
| missing=() | |
| while IFS= read -r key; do | |
| dir="results/downloaded/raw-results-$key" | |
| actual='[]' | |
| if [ -d "$dir" ]; then | |
| actual="$(find "$dir" -mindepth 3 -maxdepth 3 -path '*/run-*/result.json' -print \ | |
| | jq -Rsc '[splits("\n") | select(length > 0) | capture("/run-(?<run>[0-9]+)/result.json$").run | tonumber] | sort')" | |
| fi | |
| found="$(jq 'length' <<< "$actual")" | |
| [ "$actual" = "$required" ] || missing+=("$key ($found/$expected runs)") | |
| done < <(jq -r '.[] | "\(.experiment)__\(.eval_id)"' <<< "$pairs") | |
| echo "count=${#missing[@]}" >> "$GITHUB_OUTPUT" | |
| if [ "${#missing[@]}" -eq "$total" ]; then | |
| echo "::warning::No results published: all $total eval pairs are missing runs (run-evals likely hit its timeout or failed partway). This run needs a follow-up to complete the refresh." | |
| { | |
| echo "## :warning: No results published" | |
| echo "" | |
| echo "All $total eval pairs never finished, likely because \`run-evals\` hit its timeout or failed partway through:" | |
| echo "" | |
| echo '```' | |
| printf '%s\n' "${missing[@]}" | |
| echo '```' | |
| } >> "$GITHUB_STEP_SUMMARY" | |
| elif [ "${#missing[@]}" -gt 0 ]; then | |
| echo "::warning::Publishing PARTIAL results: ${#missing[@]} of $total eval pairs are missing runs (run-evals likely hit its timeout or failed partway). This run needs a follow-up to complete the refresh." | |
| { | |
| echo "## :warning: Partial results published" | |
| echo "" | |
| echo "${#missing[@]} of $total eval pairs are missing runs, likely because \`run-evals\` hit its timeout or failed partway through. Only the pairs with a complete set of runs were published; these were not:" | |
| echo "" | |
| echo '```' | |
| printf '%s\n' "${missing[@]}" | |
| echo '```' | |
| echo "" | |
| echo "Re-run this workflow to fill in the missing pairs." | |
| } >> "$GITHUB_STEP_SUMMARY" | |
| fi | |
| - name: Export results | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| mkdir -p results | |
| pairs='${{ needs.prepare.outputs.pairs }}' | |
| while IFS= read -r experiment; do | |
| mkdir -p "results/$experiment" | |
| for artifact_dir in "results/downloaded/raw-results-${experiment}__"*/; do | |
| [ -d "$artifact_dir" ] && cp -R "$artifact_dir"/. "results/$experiment"/ | |
| done | |
| done < <(jq -r '.[].experiment' <<< "$pairs" | sort -u) | |
| if jq -e 'any(.[]; .eval_suite == "benchmark")' <<< "$pairs" > /dev/null; then | |
| export_args=(--suite benchmark --runs "${{ needs.prepare.outputs.runs }}" --output apps/web/src/data/eval-results.json) | |
| if [ "${{ needs.prepare.outputs.do_merge }}" = "true" ]; then | |
| export_args+=(--merge) | |
| fi | |
| pnpm --filter @supabase-evals/framework export-results -- "${export_args[@]}" | |
| fi | |
| for eval_suite in regression docs cli; do | |
| if jq -e --arg s "$eval_suite" 'any(.[]; .eval_suite == $s)' <<< "$pairs" > /dev/null; then | |
| export_args=(--suite "$eval_suite" --runs "${{ needs.prepare.outputs.runs }}" --output "apps/web/src/data/${eval_suite}-eval-results.json") | |
| if [ "${{ needs.prepare.outputs.do_merge }}" = "true" ]; then | |
| export_args+=(--merge) | |
| fi | |
| pnpm --filter @supabase-evals/framework export-results -- "${export_args[@]}" | |
| fi | |
| done | |
| - name: Upload results to Braintrust | |
| # A Braintrust outage shouldn't block publishing results. | |
| continue-on-error: true | |
| env: | |
| BRAINTRUST_API_KEY: ${{ secrets.BRAINTRUST_API_KEY }} | |
| BRAINTRUST_PROJECT_ID: ${{ secrets.BRAINTRUST_PROJECT_ID }} | |
| run: pnpm upload-braintrust | |
| - name: Upload exported results | |
| if: github.event_name == 'pull_request' || github.event_name == 'workflow_dispatch' | |
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | |
| with: | |
| name: eval-results-json | |
| path: apps/web/src/data/*eval-results.json | |
| if-no-files-found: error | |
| retention-days: 7 | |
| - name: Commit exported results to branch | |
| # PR runs commit to the PR head branch. Manual dispatch commits to the | |
| # selected branch only when commit_to_branch is enabled. Scheduled runs | |
| # go through a PR instead (see "Create results pull request" below): | |
| # main's branch protection rejects a direct push. | |
| if: >- | |
| github.event_name == 'pull_request' || | |
| (github.event_name == 'workflow_dispatch' && inputs.commit_to_branch) | |
| shell: bash | |
| run: | | |
| set -euo pipefail | |
| git config user.name "github-actions[bot]" | |
| # github-actions[bot]'s noreply email uses its public user ID: https://github.com/actions/checkout#push-a-commit-using-the-built-in-token | |
| git config user.email "41898282+github-actions[bot]@users.noreply.github.com" | |
| for result_file in apps/web/src/data/*eval-results.json; do | |
| [ -f "$result_file" ] && git add "$result_file" | |
| done | |
| if git diff --cached --quiet; then | |
| echo "No eval result changes to commit" | |
| exit 0 | |
| fi | |
| # [skip ci] stops the bot push from re-triggering PR workflows (they'd need maintainer approval and loop the refresh): | |
| # https://docs.github.com/en/actions/managing-workflow-runs-and-deployments/managing-workflow-runs/skipping-workflow-runs | |
| git commit -m "chore: refresh eval results [skip ci]" | |
| git push | |
| - name: Create results pull request | |
| if: >- | |
| github.event_name == 'schedule' || | |
| (github.event_name == 'workflow_dispatch' && !inputs.commit_to_branch) | |
| id: cpr | |
| uses: peter-evans/create-pull-request@5f6978faf089d4d20b00c7766989d076bb2fc7f1 # v8.1.1 | |
| with: | |
| token: ${{ steps.generate-token.outputs.token }} | |
| add-paths: apps/web/src/data/*eval-results.json | |
| # Per ref, event, and suite so concurrent refreshes open separate PRs. | |
| branch: chore/refresh-eval-results-${{ github.ref_name }}-${{ github.event_name }}-${{ join(fromJSON(needs.prepare.outputs.suite), '-') }} | |
| base: ${{ github.ref_name }} | |
| commit-message: "chore: refresh eval results" | |
| title: "chore: refresh eval results" | |
| body: | | |
| Refreshes `apps/web/src/data/eval-results.json` from the latest automated eval run. | |
| # Draft PRs can't be merged, and the scheduled path merges itself. | |
| draft: ${{ github.event_name != 'schedule' }} | |
| delete-branch: true | |
| - name: Merge scheduled results pull request | |
| # --admin merges without the required review, which the bypass list | |
| # entitles the app to. | |
| # https://cli.github.com/manual/gh_pr_merge | |
| # Skips the merge when results are partial, so it stays open for review. | |
| if: >- | |
| github.event_name == 'schedule' && | |
| steps.cpr.outputs.pull-request-number && | |
| steps.audit.outputs.count == '0' | |
| env: | |
| GH_TOKEN: ${{ steps.generate-token.outputs.token }} | |
| run: gh pr merge "${{ steps.cpr.outputs.pull-request-number }}" --squash --delete-branch --admin | |
| - name: Require complete results | |
| # Runs last so partial results still publish, but the job still fails. | |
| if: >- | |
| always() && | |
| steps.audit.outcome == 'success' && | |
| steps.audit.outputs.count != '0' | |
| run: | | |
| echo "::error::${{ steps.audit.outputs.count }} eval pair(s) are missing runs. This run needs a follow-up to complete the refresh." | |
| exit 1 |