Skip to content

React: Prebundle react/jsx-runtime to avoid a reload when plugins import it #7341

React: Prebundle react/jsx-runtime to avoid a reload when plugins import it

React: Prebundle react/jsx-runtime to avoid a reload when plugins import it #7341

Workflow file for this run

name: Agent eval
on:
schedule:
# Weekly on `next`, Monday 08:00 UTC. Conditions with `"schedule": true` in
# agent-eval/eval-conditions.json (the full 8xx/82x line) are on for this run.
- cron: '0 8 * * 1'
pull_request:
types:
- opened
- reopened
- synchronize
- labeled
- unlabeled
workflow_dispatch:
inputs:
pr_number:
description: 'PR number for the eval gate thread (optional; inferred from the branch if omitted)'
type: string
default: ''
evals:
description: 'Comma-separated eval names to run (overrides all_evals), e.g. 803-edit-component; only 9xx evals with storybook_latest'
type: string
default: ''
all_evals:
description: 'Run the full 8xx eval line instead of only the default eval'
type: boolean
default: false
storybook_latest:
description: 'Install every Storybook package from the npm latest tag instead of this checkout'
type: boolean
default: false
concurrency:
# One run per PR at a time. A push, adding an `agent-eval:` label, or removing `agent-eval:eval`
# cancels the running eval. Other label changes also trigger this workflow, and GitHub applies
# the concurrency group before the job `if` can skip the run.
# The run ID suffix gives that run its own group, so it cannot cancel the running eval.
group: agent-eval-${{ github.workflow }}-${{ inputs.pr_number || github.event.pull_request.number || github.ref }}${{ ((github.event.action == 'labeled' && !startsWith(github.event.label.name, 'agent-eval:')) || (github.event.action == 'unlabeled' && github.event.label.name != 'agent-eval:eval')) && format('-{0}', github.run_id) || '' }}
cancel-in-progress: true
env:
VERCEL_PROJECT_NAME: storybook-evals
jobs:
eval:
name: Run agent eval experiments
# Evals start when an `agent-eval:` label is added, or when a PR with `agent-eval:eval` is
# opened or reopened. Pushes and other labels do not start them, because every run costs money.
# A push reopens the gate thread instead (see reopen-eval-gate below).
if: >-
github.event_name == 'schedule' ||
github.event_name == 'workflow_dispatch' ||
(github.event_name == 'pull_request' &&
contains(fromJSON('["opened","reopened","labeled"]'), github.event.action) &&
(github.event.action != 'labeled' || startsWith(github.event.label.name, 'agent-eval:')) &&
!github.event.pull_request.head.repo.fork &&
contains(github.event.pull_request.labels.*.name, 'agent-eval:eval'))
runs-on: ubuntu-latest
environment:
# Keep this rule in sync with the Resolve Vercel target step below.
# GitHub evaluates environment.name before step outputs exist.
name: ${{ (github.event_name == 'workflow_dispatch' || github.event_name == 'schedule') && github.ref == 'refs/heads/next' && 'production' || 'preview' }}
url: ${{ steps.deploy_playground.outputs.url }}
permissions:
# resolveReviewThread/unresolveReviewThread need contents: write on top
# of pull-requests: write (GitHub returns 403 with pull-requests alone).
contents: write
deployments: write
pull-requests: write
steps:
- name: Checkout
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
# The PR head rather than GitHub's temporary merge commit, so the commit each result
# records can still be looked up later.
ref: ${{ github.event.pull_request.head.sha || github.sha }}
persist-credentials: false
- name: Setup Node.js
uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version-file: '.nvmrc'
cache: 'yarn'
- name: Resolve PR for the eval gate thread
id: gate_pr
if: ${{ github.event_name == 'pull_request' || github.event_name == 'workflow_dispatch' }}
env:
GH_TOKEN: ${{ github.token }}
PR_NUMBER_INPUT: ${{ github.event.pull_request.number || inputs.pr_number || '' }}
run: |
pr="$PR_NUMBER_INPUT"
if [[ -z "$pr" && "$GITHUB_EVENT_NAME" == "workflow_dispatch" ]]; then
pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --head "$GITHUB_REF_NAME" --json number,isCrossRepository --jq '[.[] | select(.isCrossRepository | not)][0].number // empty')"
fi
if [[ -z "$pr" ]]; then
echo "pr_number=" >> "$GITHUB_OUTPUT"
echo "No PR for the eval gate thread (ok for next/schedule-style runs)."
exit 0
fi
labels="$(gh pr view "$pr" --repo "$GITHUB_REPOSITORY" --json labels --jq '[.labels[].name] | tojson')"
echo "labels=$labels" >> "$GITHUB_OUTPUT"
if ! jq -e 'any(. == "agent-eval:eval")' <<< "$labels" > /dev/null; then
echo "pr_number=" >> "$GITHUB_OUTPUT"
echo "PR #$pr does not have agent-eval:eval, so its gate thread is left alone."
exit 0
fi
echo "pr_number=$pr" >> "$GITHUB_OUTPUT"
# The result step compares this with the live PR head, so a run that finishes after a new
# push does not resolve the gate thread for a commit it did not evaluate.
- name: Capture evaluated head SHA
id: gate_head
env:
PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }}
run: echo "evaluated_head_sha=${PR_HEAD_SHA:-$GITHUB_SHA}" >> "$GITHUB_OUTPUT"
# Conditions are declared in agent-eval/eval-conditions.json.
- name: Resolve eval conditions
id: conditions
env:
PR_LABELS: ${{ steps.gate_pr.outputs.labels || toJSON(github.event.pull_request.labels.*.name) }}
DISPATCH_INPUTS: ${{ toJSON(inputs) }}
run: node agent-eval/scripts/resolve-conditions.ts --labels "$PR_LABELS" --inputs "$DISPATCH_INPUTS"
- name: Create or update the eval gate thread
id: gate_thread
if: ${{ steps.gate_pr.outputs.pr_number != '' }}
env:
GH_TOKEN: ${{ github.token }}
PR_NUMBER: ${{ steps.gate_pr.outputs.pr_number }}
EVALUATED_HEAD_SHA: ${{ steps.gate_head.outputs.evaluated_head_sha }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
SCOPE_LINE: ${{ steps.conditions.outputs.scope_line }}
run: >-
node agent-eval/scripts/eval-gate-thread.ts ensure
--pr "$PR_NUMBER" --sha "$EVALUATED_HEAD_SHA" --run-url "$RUN_URL" --scope "$SCOPE_LINE"
- name: Install dependencies
run: yarn install --immutable
# Sandbox setup compiles these again, but inside each eval's timeout, so warm the nx cache here.
- name: Compile the Storybook packages the sandboxes install
if: ${{ steps.conditions.outputs.storybook_latest != '1' }}
env:
NX_CLOUD_ACCESS_TOKEN: ${{ secrets.NX_CLOUD_ACCESS_TOKEN }}
run: yarn workspace agent-eval run compile:checkout
- name: Type check agent eval config
run: yarn workspace agent-eval run typecheck
- name: Check playground route shims
run: yarn workspace agent-eval run playground:check-routes
- name: Validate Vercel credentials
shell: bash
env:
VERCEL_PROJECT_ID: ${{ secrets.VERCEL_PROJECT_ID }}
VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }}
VERCEL_TEAM_ID: ${{ secrets.VERCEL_TEAM_ID }}
run: |
missing=0
for name in VERCEL_PROJECT_ID VERCEL_TOKEN VERCEL_TEAM_ID; do
if [[ -z "${!name}" ]]; then
echo "::error title=Missing Vercel secret::$name is required for Vercel Sandbox evals and the preview playground deployment."
missing=1
fi
done
exit "$missing"
- name: Run all experiments
id: run_evals
run: |
for key in ANTHROPIC_API_KEY OPENAI_API_KEY; do
if [[ -z "${!key}" ]]; then
echo "::error title=Missing eval secret::$key is required; agent-eval would skip those experiments."
exit 1
fi
done
yarn workspace agent-eval run eval
env:
EVAL_ONLY: ${{ inputs.evals || '' }}
EVAL_ALL: ${{ steps.conditions.outputs.all_evals }}
EVAL_STORYBOOK_LATEST: ${{ steps.conditions.outputs.storybook_latest }}
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
VERCEL_PROJECT_ID: ${{ secrets.VERCEL_PROJECT_ID }}
VERCEL_TEAM_ID: ${{ secrets.VERCEL_TEAM_ID }}
VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }}
- name: Check eval results
if: ${{ always() }}
id: check_results
shell: bash
run: |
if [[ ! -d agent-eval/results ]]; then
echo "::error title=Missing eval results::Expected agent-eval/results to exist before deploying the playground."
exit 1
fi
if ! find agent-eval/results -type f -print -quit | grep -q .; then
echo "::error title=No eval result files::Expected agent-eval/results to contain files before deploying the playground."
exit 1
fi
echo "has_result_files=true" >> "$GITHUB_OUTPUT"
- name: Compute offline metrics
if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }}
# Write analysis.json next to each run's result.json so the metrics ride
# along in the uploaded artifact.
continue-on-error: true
run: yarn workspace agent-eval run results:analyze
- name: Archive eval results
if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }}
# Tar before upload: result files may contain characters (e.g. the colon
# in test:stories output names) that actions/upload-artifact rejects.
run: tar -czf agent-eval-results.tgz -C agent-eval results
- name: Upload eval results
if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }}
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
with:
name: agent-eval-results
path: agent-eval-results.tgz
retention-days: 30
- name: Resolve Vercel target
if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }}
id: vercel_target
shell: bash
run: |
# Keep this rule in sync with jobs.eval.environment.name above.
if [[ ("$GITHUB_EVENT_NAME" == "workflow_dispatch" || "$GITHUB_EVENT_NAME" == "schedule") && "$GITHUB_REF" == "refs/heads/next" ]]; then
echo "target=production" >> "$GITHUB_OUTPUT"
else
echo "target=preview" >> "$GITHUB_OUTPUT"
fi
# Vercel CLI commands run from the repository root: the storybook-evals
# project's Root Directory setting is agent-eval, and in a monorepo
# the CLI must not be invoked from the subdirectory (vercel.com/docs/monorepos)
# or `deploy --prebuilt` resolves trace paths against the wrong base.
- name: Link Vercel project
if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }}
env:
VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }}
VERCEL_TEAM_ID: ${{ secrets.VERCEL_TEAM_ID }}
run: ./agent-eval/node_modules/.bin/vercel link --yes --team "$VERCEL_TEAM_ID" --project "$VERCEL_PROJECT_NAME" --token "$VERCEL_TOKEN"
- name: Pull Vercel environment
if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }}
shell: bash
env:
VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }}
VERCEL_TARGET: ${{ steps.vercel_target.outputs.target }}
run: |
if [[ "$VERCEL_TARGET" == "production" ]]; then
./agent-eval/node_modules/.bin/vercel pull --yes --environment=production --token "$VERCEL_TOKEN"
else
branch="${GITHUB_HEAD_REF:-$GITHUB_REF_NAME}"
./agent-eval/node_modules/.bin/vercel pull --yes --environment=preview --git-branch "$branch" --token "$VERCEL_TOKEN"
fi
- name: Build playground
if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }}
id: build_playground
env:
VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }}
VERCEL_TARGET: ${{ steps.vercel_target.outputs.target }}
run: ./agent-eval/node_modules/.bin/vercel build --yes --target="$VERCEL_TARGET" --token "$VERCEL_TOKEN"
- name: Deploy playground
if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' && steps.build_playground.outcome == 'success' }}
id: deploy_playground
shell: bash
env:
VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }}
VERCEL_TARGET: ${{ steps.vercel_target.outputs.target }}
run: |
deploy_output="$(./agent-eval/node_modules/.bin/vercel deploy --prebuilt --yes --target="$VERCEL_TARGET" --token "$VERCEL_TOKEN")"
printf '%s\n' "$deploy_output"
url="$(
printf '%s\n' "$deploy_output" |
awk '/https:\/\/[^[:space:]]+\.vercel\.app/ { match($0, /https:\/\/[^[:space:]]+\.vercel\.app[^[:space:]]*/); if (RSTART) found=substr($0,RSTART,RLENGTH) } END { print found }'
)"
if [[ -z "$url" ]]; then
echo "::error title=Missing deployment URL::Vercel deploy did not print a deployment URL."
exit 1
fi
echo "url=$url" >> "$GITHUB_OUTPUT"
- name: Update PR description with eval results
if: ${{ always() && github.event_name == 'pull_request' && steps.check_results.outputs.has_result_files == 'true' }}
shell: bash
env:
GH_TOKEN: ${{ github.token }}
PR_NUMBER: ${{ github.event.pull_request.number }}
PLAYGROUND_URL: ${{ steps.deploy_playground.outputs.url }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
run: |
summary="$(node agent-eval/scripts/render-results-summary.mjs)"
body="$(gh api "repos/$GITHUB_REPOSITORY/pulls/$PR_NUMBER" --jq '.body // ""')"
printf '%s' "$body" |
SUMMARY="$summary" node agent-eval/scripts/upsert-pr-eval-results.mjs |
gh api --method PATCH "repos/$GITHUB_REPOSITORY/pulls/$PR_NUMBER" --field body=@- --silent
- name: Write deployment summary
if: ${{ always() && steps.deploy_playground.outcome == 'success' }}
env:
VERCEL_TARGET: ${{ steps.vercel_target.outputs.target }}
PLAYGROUND_URL: ${{ steps.deploy_playground.outputs.url }}
run: |
{
echo "### Agent eval playground"
echo
echo "- Target: \`$VERCEL_TARGET\`"
echo "- Project: \`$VERCEL_PROJECT_NAME\`"
echo "- URL: $PLAYGROUND_URL"
} >> "$GITHUB_STEP_SUMMARY"
- name: Update the eval gate thread with the result
if: >-
always() &&
!cancelled() &&
steps.gate_thread.outcome == 'success'
env:
GH_TOKEN: ${{ github.token }}
PR_NUMBER: ${{ steps.gate_pr.outputs.pr_number }}
EVALUATED_HEAD_SHA: ${{ steps.gate_head.outputs.evaluated_head_sha }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
PLAYGROUND_URL: ${{ steps.deploy_playground.outputs.url }}
EVAL_OUTCOME: ${{ steps.check_results.outcome == 'success' && steps.run_evals.outcome || 'failure' }}
run: |
summary_file="$(mktemp)"
node agent-eval/scripts/render-results-summary.mjs > "$summary_file" || true
node agent-eval/scripts/eval-gate-thread.ts result \
--pr "$PR_NUMBER" --sha "$EVALUATED_HEAD_SHA" --outcome "$EVAL_OUTCOME" \
--run-url "$RUN_URL" --playground-url "$PLAYGROUND_URL" --summary-file "$summary_file"
# Weekly / next-dispatch summary for the #sb-monitoring channel.
- name: Notify Slack
if: >-
always() &&
!cancelled() &&
(github.event_name == 'schedule' ||
(github.event_name == 'workflow_dispatch' && github.ref == 'refs/heads/next'))
shell: bash
env:
SLACK_WEBHOOK_URL: ${{ secrets.SLACK_AGENT_EVAL_WEBHOOK_URL }}
PLAYGROUND_URL: ${{ steps.deploy_playground.outputs.url }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
EVAL_OUTCOME: ${{ steps.run_evals.outcome }}
run: |
if [[ -z "$SLACK_WEBHOOK_URL" ]]; then
echo "::warning title=Missing Slack webhook::SLACK_AGENT_EVAL_WEBHOOK_URL is not set; skipping #sb-monitoring notify."
exit 0
fi
text="$(FORMAT=slack node agent-eval/scripts/render-results-summary.mjs)"
payload="$(jq -n --arg text "$text" '{text: $text}')"
curl --fail --silent --show-error --connect-timeout 10 --max-time 30 \
-X POST -H 'Content-type: application/json' --data "$payload" "$SLACK_WEBHOOK_URL"
# New push on an agent-eval:eval PR: previous eval results no longer cover
# the head. Pushes do not auto-reopen resolved review threads, so explicitly
# reopen the gate thread without rerunning evals.
reopen-eval-gate:
name: Reopen or resolve the eval gate thread
if: >-
github.event_name == 'pull_request' &&
!github.event.pull_request.head.repo.fork &&
((github.event.action == 'synchronize' &&
contains(github.event.pull_request.labels.*.name, 'agent-eval:eval')) ||
(github.event.action == 'unlabeled' && github.event.label.name == 'agent-eval:eval'))
runs-on: ubuntu-latest
permissions:
# unresolveReviewThread needs contents: write + pull-requests: write.
contents: write
pull-requests: write
steps:
- name: Checkout eval gate script
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
sparse-checkout: |
agent-eval/scripts
scripts/utils
- name: Setup Node.js
uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version-file: '.nvmrc'
- name: Reopen the eval gate thread for the new head
if: ${{ github.event.action == 'synchronize' }}
env:
GH_TOKEN: ${{ github.token }}
PR_NUMBER: ${{ github.event.pull_request.number }}
NEW_HEAD_SHA: ${{ github.event.pull_request.head.sha }}
run: node agent-eval/scripts/eval-gate-thread.ts reopen --pr "$PR_NUMBER" --sha "$NEW_HEAD_SHA"
- name: Resolve the eval gate thread after opting out
if: ${{ github.event.action == 'unlabeled' }}
env:
GH_TOKEN: ${{ github.token }}
PR_NUMBER: ${{ github.event.pull_request.number }}
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
run: >-
node agent-eval/scripts/eval-gate-thread.ts opt-out
--pr "$PR_NUMBER" --sha "$HEAD_SHA" --actor "$GITHUB_ACTOR"