Skip to content

feat(evals): check the database functions design accounts for the caller #1824

feat(evals): check the database functions design accounts for the caller

feat(evals): check the database functions design accounts for the caller #1824

Workflow file for this run

name: Refresh eval results
on:
workflow_dispatch:
inputs:
experiments:
description: "Comma-separated experiment names to run (blank to auto-discover from experiment_suite)"
required: false
default: ""
eval:
description: "Optional comma-separated eval ids to run"
required: false
default: ""
suite:
description: "Comma-separated eval suites to run"
required: true
default: "benchmark"
experiment_suite:
description: "Comma-separated experiment suites to run"
required: true
default: "benchmark,no-skills"
runs:
description: "Independent scored runs per experiment/eval pair"
required: true
default: "3"
timeout_sec:
description: "Timeout per run in seconds"
required: true
default: "720"
sandbox_concurrency:
description: "Maximum Vercel Sandboxes running at once"
required: true
default: "250"
merge:
description: "Merge into existing results instead of overwriting (graft new experiment/eval pairs)"
type: boolean
required: false
default: false
commit_to_branch:
description: "Commit exported results to the dispatched branch instead of opening a PR"
type: boolean
required: false
default: false
cli_stable_version:
description: "Pin the cli suite's stable CLI version instead of resolving npm's latest dist-tag (blank to resolve)"
required: false
default: ""
cli_beta_version:
description: "Pin the cli suite's beta CLI version instead of resolving npm's beta dist-tag (blank to resolve)"
required: false
default: ""
schedule:
- cron: '15 6 * * *'
pull_request:
# Run whenever a PR carrying the run-evals label is opened, pushed to, or
# receives the run-evals label. The job-level `if` gates on those cases.
types: [opened, synchronize, labeled]
permissions:
contents: write
pull-requests: write
actions: read
concurrency:
group: eval-refresh-${{ github.event_name == 'pull_request' && github.event.pull_request.number || github.ref }}
# Supersede an in-flight run when a new commit is pushed to the same PR, but
# never cancel a workflow_dispatch run (those open the refresh PR).
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
jobs:
prepare:
if: >-
github.event_name == 'workflow_dispatch' ||
github.event_name == 'schedule' ||
(github.event_name == 'pull_request' &&
(contains(github.event.pull_request.labels.*.name, 'run-evals') ||
contains(github.event.pull_request.labels.*.name, 'run-evals-changed')) &&
github.event.pull_request.head.repo.full_name == github.repository &&
(github.event.action != 'labeled' ||
github.event.label.name == 'run-evals' ||
github.event.label.name == 'run-evals-changed'))
runs-on: ubuntu-latest
outputs:
pairs: ${{ steps.discover.outputs.pairs }}
runs: ${{ steps.inputs.outputs.runs }}
timeout_sec: ${{ steps.inputs.outputs.timeout_sec }}
sandbox_concurrency: ${{ steps.inputs.outputs.sandbox_concurrency }}
filter_changed: ${{ steps.inputs.outputs.filter_changed }}
do_merge: ${{ steps.inputs.outputs.do_merge }}
suite: ${{ steps.inputs.outputs.suite }}
cli_stable_version: ${{ steps.inputs.outputs.cli_stable_version }}
cli_beta_version: ${{ steps.inputs.outputs.cli_beta_version }}
steps:
- name: Prepare inputs
id: inputs
shell: bash
run: |
set -euo pipefail
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
experiments_override="${{ inputs.experiments }}"
eval_id="${{ inputs.eval }}"
suite="${{ inputs.suite }}"
experiment_suite="${{ inputs.experiment_suite }}"
runs="${{ inputs.runs }}"
timeout_sec="${{ inputs.timeout_sec }}"
sandbox_concurrency="${{ inputs.sandbox_concurrency }}"
cli_stable_version="${{ inputs.cli_stable_version }}"
cli_beta_version="${{ inputs.cli_beta_version }}"
elif [ "${{ github.event_name }}" = "schedule" ]; then
experiments_override=""
eval_id=""
suite="regression,cli"
experiment_suite="regression,cli"
runs="3"
timeout_sec="720"
sandbox_concurrency="250"
cli_stable_version=""
cli_beta_version=""
else
experiments_override=""
eval_id=""
suite="benchmark,regression,docs,cli"
experiment_suite="benchmark,no-skills,regression,docs,cli"
runs="3"
timeout_sec="720"
sandbox_concurrency="250"
cli_stable_version=""
cli_beta_version=""
fi
suite_json="$(jq -Rc 'split(",") | map(gsub("^\\s+|\\s+$"; "")) | map(select(length > 0))' <<< "$suite")"
experiment_suite_json="$(jq -Rc 'split(",") | map(gsub("^\\s+|\\s+$"; "")) | map(select(length > 0))' <<< "$experiment_suite")"
# run-evals takes priority; run-evals-changed only filters when run-evals is absent.
# filter_changed drives the changed-eval filter (PR-only, needs a PR diff).
filter_changed="false"
if [ "${{ github.event_name }}" = "pull_request" ] && \
[ "${{ contains(github.event.pull_request.labels.*.name, 'run-evals-changed') }}" = "true" ] && \
[ "${{ contains(github.event.pull_request.labels.*.name, 'run-evals') }}" = "false" ]; then
filter_changed="true"
fi
# do_merge drives the export --merge (graft into existing results). It's
# always on for the changed path, and opt-in for manual dispatch.
do_merge="$filter_changed"
if [ "${{ github.event_name }}" = "workflow_dispatch" ] && [ "${{ inputs.merge }}" = "true" ]; then
do_merge="true"
fi
{
echo "experiments_override=$experiments_override"
echo "eval=$eval_id"
echo "suite=$suite_json"
echo "experiment_suite=$experiment_suite_json"
echo "runs=$runs"
echo "timeout_sec=$timeout_sec"
echo "sandbox_concurrency=$sandbox_concurrency"
echo "filter_changed=$filter_changed"
echo "do_merge=$do_merge"
echo "cli_stable_version=$cli_stable_version"
echo "cli_beta_version=$cli_beta_version"
} >> "$GITHUB_OUTPUT"
- name: Checkout
uses: actions/checkout@9f698171ed81b15d1823a05fc7211befd50c8ae0 # v6.0.3
with:
ref: ${{ github.event_name == 'pull_request' && github.head_ref || github.ref }}
- name: Install pnpm
uses: pnpm/action-setup@fc06bc1257f339d1d5d8b3a19a8cae5388b55320 # v5.0.0
- name: Setup Node.js
uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version-file: .node-version
cache: pnpm
- name: Install dependencies
run: pnpm install --frozen-lockfile
- name: Discover eval pairs
id: discover
env:
GH_TOKEN: ${{ github.token }}
shell: bash
run: |
set -euo pipefail
touch .env
suite_json='${{ steps.inputs.outputs.suite }}'
eval_ids="${{ steps.inputs.outputs.eval }}"
if [ -n "$eval_ids" ]; then
matching=()
while IFS= read -r id; do
compgen -G "evals/*/$id" > /dev/null && matching+=("$id")
done < <(jq -Rr 'split(",") | map(gsub("^\\s+|\\s+$"; "")) | .[]' <<< "$eval_ids")
else
matching=()
for dir in evals/*/*/; do
[ -d "$dir" ] || continue
id=$(basename "$dir")
[ -f "$dir/PROMPT.md" ] || continue
suite_val=$(basename "$(dirname "$dir")")
if jq -e --arg s "$suite_val" 'index($s) != null' <<< "$suite_json" > /dev/null 2>&1; then
matching+=("$id")
fi
done
fi
if [ "${{ steps.inputs.outputs.filter_changed }}" = "true" ]; then
changed_evals=$(gh pr diff ${{ github.event.pull_request.number }} --name-only \
| grep '^evals/' | cut -d/ -f3 | sort -u || true)
if [ -z "$changed_evals" ]; then
echo "No eval directories changed in this PR — nothing to run"
echo "pairs=[]" >> "$GITHUB_OUTPUT"
exit 0
fi
echo "Changed eval dirs: $changed_evals"
filtered=()
for id in "${matching[@]}"; do
if echo "$changed_evals" | grep -qx "$id"; then
filtered+=("$id")
fi
done
matching=("${filtered[@]+"${filtered[@]}"}")
fi
if [ "${#matching[@]}" -eq 0 ]; then
echo "Changed evals don't match requested suites: $suite_json"
echo "pairs=[]" >> "$GITHUB_OUTPUT"
exit 0
fi
pairs='[]'
experiments_override="${{ steps.inputs.outputs.experiments_override }}"
for id in "${matching[@]}"; do
eval_suite=$(basename "$(dirname "$(compgen -G "evals/*/$id")")")
case "$eval_suite" in
benchmark) experiment_suites=(benchmark no-skills) ;;
regression) experiment_suites=(regression) ;;
docs) experiment_suites=(docs) ;;
cli) experiment_suites=(cli) ;;
*) continue ;;
esac
for experiment_suite in "${experiment_suites[@]}"; do
if ! jq -e --arg suite "$experiment_suite" 'index($suite) != null' \
<<< '${{ steps.inputs.outputs.experiment_suite }}' > /dev/null; then
continue
fi
if [ -n "$experiments_override" ]; then
experiments_json="$(jq -Rc 'split(",") | map(gsub("^\\s+|\\s+$"; "")) | map(select(length > 0))' <<< "$experiments_override")"
else
experiments_json="$(pnpm --silent eval -- list --experiment-suite "$experiment_suite" --eval "$id")"
fi
while IFS= read -r experiment; do
pairs="$(jq -c \
--arg eval_id "$id" \
--arg experiment "$experiment" \
--arg experiment_suite "$experiment_suite" \
--arg eval_suite "$eval_suite" \
'. + [{eval_id: $eval_id, experiment: $experiment, experiment_suite: $experiment_suite, eval_suite: $eval_suite}]' \
<<< "$pairs")"
done < <(jq -r '.[]' <<< "$experiments_json")
done
done
if [ "$(jq 'length' <<< "$pairs")" -eq 0 ]; then
echo "No experiment and eval pairs matched" >&2
exit 1
fi
echo "pairs=$pairs" >> "$GITHUB_OUTPUT"
run-evals:
needs: prepare
if: needs.prepare.outputs.pairs != '[]'
runs-on: ubuntu-latest
# 60 min gives ~2x margin over the slowest observed run:
# https://github.com/supabase/evals/actions/runs/31802584549
timeout-minutes: 60
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
AI_GATEWAY_API_KEY: ${{ secrets.AI_GATEWAY_API_KEY }}
XAI_API_KEY: ${{ secrets.XAI_API_KEY }}
VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }}
VERCEL_TEAM_ID: ${{ secrets.VERCEL_TEAM_ID }}
VERCEL_PROJECT_ID: ${{ secrets.VERCEL_PROJECT_ID }}
SUPABASE_CLI_STABLE_VERSION: ${{ needs.prepare.outputs.cli_stable_version }}
SUPABASE_CLI_BETA_VERSION: ${{ needs.prepare.outputs.cli_beta_version }}
steps:
- name: Checkout
uses: actions/checkout@9f698171ed81b15d1823a05fc7211befd50c8ae0 # v6.0.3
with:
ref: ${{ github.event_name == 'pull_request' && github.head_ref || github.ref }}
- name: Install pnpm
uses: pnpm/action-setup@fc06bc1257f339d1d5d8b3a19a8cae5388b55320 # v5.0.0
- name: Setup Node.js
uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version-file: .node-version
cache: pnpm
- name: Install dependencies
run: pnpm install --frozen-lockfile
- name: Write eval environment
shell: bash
run: |
set -euo pipefail
{
echo "ANTHROPIC_API_KEY=${ANTHROPIC_API_KEY}"
echo "OPENAI_API_KEY=${OPENAI_API_KEY}"
echo "AI_GATEWAY_API_KEY=${AI_GATEWAY_API_KEY}"
echo "XAI_API_KEY=${XAI_API_KEY}"
echo "SUPABASE_CLI_STABLE_VERSION=${SUPABASE_CLI_STABLE_VERSION}"
echo "SUPABASE_CLI_BETA_VERSION=${SUPABASE_CLI_BETA_VERSION}"
} > .env
- name: Run evals
env:
EVAL_PAIRS: ${{ needs.prepare.outputs.pairs }}
EVAL_REVISION: ${{ github.event_name == 'pull_request' && github.event.pull_request.head.sha || github.sha }}
shell: bash
run: |
set -euo pipefail
pnpm --filter @supabase-evals/framework eval:vercel -- \
--pairs-json "$EVAL_PAIRS" \
--revision "$EVAL_REVISION" \
--runs "${{ needs.prepare.outputs.runs }}" \
--timeout-sec "${{ needs.prepare.outputs.timeout_sec }}" \
--concurrency "${{ needs.prepare.outputs.sandbox_concurrency }}"
- name: Upload raw results
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: raw-results
path: results/downloaded/
overwrite: true
if-no-files-found: warn
retention-days: 3
publish-results:
needs: [prepare, run-evals]
# Runs even after run-evals times out or partially fails, since results
# are written per pair and export-results skips any pair that never finished.
if: "always() && needs.run-evals.result != 'skipped' && needs.prepare.outputs.pairs != '[]'"
runs-on: ubuntu-latest
steps:
- name: Generate GitHub App token
id: generate-token
# A GitHub App token is needed so the push below can trigger the
# gh-pages workflow: GITHUB_TOKEN pushes don't trigger other workflows.
# https://docs.github.com/en/actions/how-tos/write-workflows/choose-when-workflows-run/trigger-a-workflow#triggering-a-workflow-from-a-workflow
if: >-
github.event_name == 'schedule' ||
(github.event_name == 'workflow_dispatch' && !inputs.commit_to_branch)
uses: actions/create-github-app-token@bcd2ba49218906704ab6c1aa796996da409d3eb1 # v3.2.0
with:
app-id: ${{ secrets.GH_APP_ID }}
private-key: ${{ secrets.GH_APP_PRIVATE_KEY }}
- name: Checkout
uses: actions/checkout@9f698171ed81b15d1823a05fc7211befd50c8ae0 # v6.0.3
with:
ref: ${{ github.event_name == 'pull_request' && github.head_ref || github.ref }}
token: ${{ steps.generate-token.outputs.token || github.token }}
- name: Install pnpm
uses: pnpm/action-setup@fc06bc1257f339d1d5d8b3a19a8cae5388b55320 # v5.0.0
- name: Setup Node.js
uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version-file: .node-version
cache: pnpm
- name: Install dependencies
run: pnpm install --frozen-lockfile
- name: Download raw results
# If every pair failed, run-evals never uploads a raw-results artifact
# at all, so this action errors. Audit pair results below already
# handles a missing/empty results/downloaded correctly (every pair
# reports missing), so don't let a total failure hide behind this.
continue-on-error: true
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
with:
name: raw-results
path: results/downloaded
- name: Extract raw workspaces
# Workspaces arrive tarred so upload-artifact never sees agent-written filenames.
shell: bash
run: |
set -euo pipefail
if [ -d results/downloaded ]; then
find results/downloaded -name 'workspace.tgz' -print0 | while IFS= read -r -d '' archive; do
workspace="$(dirname "$archive")/workspace"
mkdir -p "$workspace"
tar -xzf "$archive" -C "$workspace"
rm "$archive"
done
fi
- name: Audit pair results
id: audit
shell: bash
run: |
set -euo pipefail
pairs='${{ needs.prepare.outputs.pairs }}'
total="$(jq 'length' <<< "$pairs")"
# A pair needs every requested run, not just a directory, since a
# partial pair would otherwise skew the average against its neighbours.
expected="${{ needs.prepare.outputs.runs }}"
required="$(jq -cn --argjson expected "$expected" '[range(1; $expected + 1)]')"
missing=()
while IFS= read -r key; do
dir="results/downloaded/raw-results-$key"
actual='[]'
if [ -d "$dir" ]; then
actual="$(find "$dir" -mindepth 3 -maxdepth 3 -path '*/run-*/result.json' -print \
| jq -Rsc '[splits("\n") | select(length > 0) | capture("/run-(?<run>[0-9]+)/result.json$").run | tonumber] | sort')"
fi
found="$(jq 'length' <<< "$actual")"
[ "$actual" = "$required" ] || missing+=("$key ($found/$expected runs)")
done < <(jq -r '.[] | "\(.experiment)__\(.eval_id)"' <<< "$pairs")
echo "count=${#missing[@]}" >> "$GITHUB_OUTPUT"
if [ "${#missing[@]}" -eq "$total" ]; then
echo "::warning::No results published: all $total eval pairs are missing runs (run-evals likely hit its timeout or failed partway). This run needs a follow-up to complete the refresh."
{
echo "## :warning: No results published"
echo ""
echo "All $total eval pairs never finished, likely because \`run-evals\` hit its timeout or failed partway through:"
echo ""
echo '```'
printf '%s\n' "${missing[@]}"
echo '```'
} >> "$GITHUB_STEP_SUMMARY"
elif [ "${#missing[@]}" -gt 0 ]; then
echo "::warning::Publishing PARTIAL results: ${#missing[@]} of $total eval pairs are missing runs (run-evals likely hit its timeout or failed partway). This run needs a follow-up to complete the refresh."
{
echo "## :warning: Partial results published"
echo ""
echo "${#missing[@]} of $total eval pairs are missing runs, likely because \`run-evals\` hit its timeout or failed partway through. Only the pairs with a complete set of runs were published; these were not:"
echo ""
echo '```'
printf '%s\n' "${missing[@]}"
echo '```'
echo ""
echo "Re-run this workflow to fill in the missing pairs."
} >> "$GITHUB_STEP_SUMMARY"
fi
- name: Export results
shell: bash
run: |
set -euo pipefail
mkdir -p results
pairs='${{ needs.prepare.outputs.pairs }}'
while IFS= read -r experiment; do
mkdir -p "results/$experiment"
for artifact_dir in "results/downloaded/raw-results-${experiment}__"*/; do
[ -d "$artifact_dir" ] && cp -R "$artifact_dir"/. "results/$experiment"/
done
done < <(jq -r '.[].experiment' <<< "$pairs" | sort -u)
if jq -e 'any(.[]; .eval_suite == "benchmark")' <<< "$pairs" > /dev/null; then
export_args=(--suite benchmark --runs "${{ needs.prepare.outputs.runs }}" --output apps/web/src/data/eval-results.json)
if [ "${{ needs.prepare.outputs.do_merge }}" = "true" ]; then
export_args+=(--merge)
fi
pnpm --filter @supabase-evals/framework export-results -- "${export_args[@]}"
fi
for eval_suite in regression docs cli; do
if jq -e --arg s "$eval_suite" 'any(.[]; .eval_suite == $s)' <<< "$pairs" > /dev/null; then
export_args=(--suite "$eval_suite" --runs "${{ needs.prepare.outputs.runs }}" --output "apps/web/src/data/${eval_suite}-eval-results.json")
if [ "${{ needs.prepare.outputs.do_merge }}" = "true" ]; then
export_args+=(--merge)
fi
pnpm --filter @supabase-evals/framework export-results -- "${export_args[@]}"
fi
done
- name: Upload exported results
if: github.event_name == 'pull_request' || github.event_name == 'workflow_dispatch'
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: eval-results-json
path: apps/web/src/data/*eval-results.json
if-no-files-found: error
retention-days: 7
- name: Commit exported results to branch
# PR runs commit to the PR head branch. Manual dispatch commits to the
# selected branch only when commit_to_branch is enabled. Scheduled runs
# go through a PR instead (see "Create results pull request" below):
# main's branch protection rejects a direct push.
if: >-
github.event_name == 'pull_request' ||
(github.event_name == 'workflow_dispatch' && inputs.commit_to_branch)
shell: bash
run: |
set -euo pipefail
git config user.name "github-actions[bot]"
# github-actions[bot]'s noreply email uses its public user ID: https://github.com/actions/checkout#push-a-commit-using-the-built-in-token
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
for result_file in apps/web/src/data/*eval-results.json; do
[ -f "$result_file" ] && git add "$result_file"
done
if git diff --cached --quiet; then
echo "No eval result changes to commit"
exit 0
fi
git commit -m "chore: refresh eval results"
git push
- name: Create results pull request
if: >-
github.event_name == 'schedule' ||
(github.event_name == 'workflow_dispatch' && !inputs.commit_to_branch)
id: cpr
uses: peter-evans/create-pull-request@5f6978faf089d4d20b00c7766989d076bb2fc7f1 # v8.1.1
with:
token: ${{ steps.generate-token.outputs.token }}
add-paths: apps/web/src/data/*eval-results.json
# Per ref, event, and suite so concurrent refreshes open separate PRs.
branch: chore/refresh-eval-results-${{ github.ref_name }}-${{ github.event_name }}-${{ join(fromJSON(needs.prepare.outputs.suite), '-') }}
base: ${{ github.ref_name }}
commit-message: "chore: refresh eval results"
title: "chore: refresh eval results"
body: |
Refreshes `apps/web/src/data/eval-results.json` from the latest automated eval run.
# Draft PRs can't be merged, and the scheduled path merges itself.
draft: ${{ github.event_name != 'schedule' }}
delete-branch: true
- name: Merge scheduled results pull request
# --admin merges without the required review, which the bypass list
# entitles the app to.
# https://cli.github.com/manual/gh_pr_merge
# Skips the merge when results are partial, so it stays open for review.
if: >-
github.event_name == 'schedule' &&
steps.cpr.outputs.pull-request-number &&
steps.audit.outputs.count == '0'
env:
GH_TOKEN: ${{ steps.generate-token.outputs.token }}
run: gh pr merge "${{ steps.cpr.outputs.pull-request-number }}" --squash --delete-branch --admin
- name: Require complete results
# Runs last so partial results still publish, but the job still fails.
if: >-
always() &&
steps.audit.outcome == 'success' &&
steps.audit.outputs.count != '0'
run: |
echo "::error::${{ steps.audit.outputs.count }} eval pair(s) are missing runs. This run needs a follow-up to complete the refresh."
exit 1