Repository navigation
React: Prebundle react/jsx-runtime to avoid a reload when plugins import it #7341
Workflow file for this run
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: Agent eval | |
| on: | |
| schedule: | |
| # Weekly on `next`, Monday 08:00 UTC. Conditions with `"schedule": true` in | |
| # agent-eval/eval-conditions.json (the full 8xx/82x line) are on for this run. | |
| - cron: '0 8 * * 1' | |
| pull_request: | |
| types: | |
| - opened | |
| - reopened | |
| - synchronize | |
| - labeled | |
| - unlabeled | |
| workflow_dispatch: | |
| inputs: | |
| pr_number: | |
| description: 'PR number for the eval gate thread (optional; inferred from the branch if omitted)' | |
| type: string | |
| default: '' | |
| evals: | |
| description: 'Comma-separated eval names to run (overrides all_evals), e.g. 803-edit-component; only 9xx evals with storybook_latest' | |
| type: string | |
| default: '' | |
| all_evals: | |
| description: 'Run the full 8xx eval line instead of only the default eval' | |
| type: boolean | |
| default: false | |
| storybook_latest: | |
| description: 'Install every Storybook package from the npm latest tag instead of this checkout' | |
| type: boolean | |
| default: false | |
| concurrency: | |
| # One run per PR at a time. A push, adding an `agent-eval:` label, or removing `agent-eval:eval` | |
| # cancels the running eval. Other label changes also trigger this workflow, and GitHub applies | |
| # the concurrency group before the job `if` can skip the run. | |
| # The run ID suffix gives that run its own group, so it cannot cancel the running eval. | |
| group: agent-eval-${{ github.workflow }}-${{ inputs.pr_number || github.event.pull_request.number || github.ref }}${{ ((github.event.action == 'labeled' && !startsWith(github.event.label.name, 'agent-eval:')) || (github.event.action == 'unlabeled' && github.event.label.name != 'agent-eval:eval')) && format('-{0}', github.run_id) || '' }} | |
| cancel-in-progress: true | |
| env: | |
| VERCEL_PROJECT_NAME: storybook-evals | |
| jobs: | |
| eval: | |
| name: Run agent eval experiments | |
| # Evals start when an `agent-eval:` label is added, or when a PR with `agent-eval:eval` is | |
| # opened or reopened. Pushes and other labels do not start them, because every run costs money. | |
| # A push reopens the gate thread instead (see reopen-eval-gate below). | |
| if: >- | |
| github.event_name == 'schedule' || | |
| github.event_name == 'workflow_dispatch' || | |
| (github.event_name == 'pull_request' && | |
| contains(fromJSON('["opened","reopened","labeled"]'), github.event.action) && | |
| (github.event.action != 'labeled' || startsWith(github.event.label.name, 'agent-eval:')) && | |
| !github.event.pull_request.head.repo.fork && | |
| contains(github.event.pull_request.labels.*.name, 'agent-eval:eval')) | |
| runs-on: ubuntu-latest | |
| environment: | |
| # Keep this rule in sync with the Resolve Vercel target step below. | |
| # GitHub evaluates environment.name before step outputs exist. | |
| name: ${{ (github.event_name == 'workflow_dispatch' || github.event_name == 'schedule') && github.ref == 'refs/heads/next' && 'production' || 'preview' }} | |
| url: ${{ steps.deploy_playground.outputs.url }} | |
| permissions: | |
| # resolveReviewThread/unresolveReviewThread need contents: write on top | |
| # of pull-requests: write (GitHub returns 403 with pull-requests alone). | |
| contents: write | |
| deployments: write | |
| pull-requests: write | |
| steps: | |
| - name: Checkout | |
| uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 | |
| with: | |
| # The PR head rather than GitHub's temporary merge commit, so the commit each result | |
| # records can still be looked up later. | |
| ref: ${{ github.event.pull_request.head.sha || github.sha }} | |
| persist-credentials: false | |
| - name: Setup Node.js | |
| uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 | |
| with: | |
| node-version-file: '.nvmrc' | |
| cache: 'yarn' | |
| - name: Resolve PR for the eval gate thread | |
| id: gate_pr | |
| if: ${{ github.event_name == 'pull_request' || github.event_name == 'workflow_dispatch' }} | |
| env: | |
| GH_TOKEN: ${{ github.token }} | |
| PR_NUMBER_INPUT: ${{ github.event.pull_request.number || inputs.pr_number || '' }} | |
| run: | | |
| pr="$PR_NUMBER_INPUT" | |
| if [[ -z "$pr" && "$GITHUB_EVENT_NAME" == "workflow_dispatch" ]]; then | |
| pr="$(gh pr list --repo "$GITHUB_REPOSITORY" --state open --head "$GITHUB_REF_NAME" --json number,isCrossRepository --jq '[.[] | select(.isCrossRepository | not)][0].number // empty')" | |
| fi | |
| if [[ -z "$pr" ]]; then | |
| echo "pr_number=" >> "$GITHUB_OUTPUT" | |
| echo "No PR for the eval gate thread (ok for next/schedule-style runs)." | |
| exit 0 | |
| fi | |
| labels="$(gh pr view "$pr" --repo "$GITHUB_REPOSITORY" --json labels --jq '[.labels[].name] | tojson')" | |
| echo "labels=$labels" >> "$GITHUB_OUTPUT" | |
| if ! jq -e 'any(. == "agent-eval:eval")' <<< "$labels" > /dev/null; then | |
| echo "pr_number=" >> "$GITHUB_OUTPUT" | |
| echo "PR #$pr does not have agent-eval:eval, so its gate thread is left alone." | |
| exit 0 | |
| fi | |
| echo "pr_number=$pr" >> "$GITHUB_OUTPUT" | |
| # The result step compares this with the live PR head, so a run that finishes after a new | |
| # push does not resolve the gate thread for a commit it did not evaluate. | |
| - name: Capture evaluated head SHA | |
| id: gate_head | |
| env: | |
| PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }} | |
| run: echo "evaluated_head_sha=${PR_HEAD_SHA:-$GITHUB_SHA}" >> "$GITHUB_OUTPUT" | |
| # Conditions are declared in agent-eval/eval-conditions.json. | |
| - name: Resolve eval conditions | |
| id: conditions | |
| env: | |
| PR_LABELS: ${{ steps.gate_pr.outputs.labels || toJSON(github.event.pull_request.labels.*.name) }} | |
| DISPATCH_INPUTS: ${{ toJSON(inputs) }} | |
| run: node agent-eval/scripts/resolve-conditions.ts --labels "$PR_LABELS" --inputs "$DISPATCH_INPUTS" | |
| - name: Create or update the eval gate thread | |
| id: gate_thread | |
| if: ${{ steps.gate_pr.outputs.pr_number != '' }} | |
| env: | |
| GH_TOKEN: ${{ github.token }} | |
| PR_NUMBER: ${{ steps.gate_pr.outputs.pr_number }} | |
| EVALUATED_HEAD_SHA: ${{ steps.gate_head.outputs.evaluated_head_sha }} | |
| RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} | |
| SCOPE_LINE: ${{ steps.conditions.outputs.scope_line }} | |
| run: >- | |
| node agent-eval/scripts/eval-gate-thread.ts ensure | |
| --pr "$PR_NUMBER" --sha "$EVALUATED_HEAD_SHA" --run-url "$RUN_URL" --scope "$SCOPE_LINE" | |
| - name: Install dependencies | |
| run: yarn install --immutable | |
| # Sandbox setup compiles these again, but inside each eval's timeout, so warm the nx cache here. | |
| - name: Compile the Storybook packages the sandboxes install | |
| if: ${{ steps.conditions.outputs.storybook_latest != '1' }} | |
| env: | |
| NX_CLOUD_ACCESS_TOKEN: ${{ secrets.NX_CLOUD_ACCESS_TOKEN }} | |
| run: yarn workspace agent-eval run compile:checkout | |
| - name: Type check agent eval config | |
| run: yarn workspace agent-eval run typecheck | |
| - name: Check playground route shims | |
| run: yarn workspace agent-eval run playground:check-routes | |
| - name: Validate Vercel credentials | |
| shell: bash | |
| env: | |
| VERCEL_PROJECT_ID: ${{ secrets.VERCEL_PROJECT_ID }} | |
| VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }} | |
| VERCEL_TEAM_ID: ${{ secrets.VERCEL_TEAM_ID }} | |
| run: | | |
| missing=0 | |
| for name in VERCEL_PROJECT_ID VERCEL_TOKEN VERCEL_TEAM_ID; do | |
| if [[ -z "${!name}" ]]; then | |
| echo "::error title=Missing Vercel secret::$name is required for Vercel Sandbox evals and the preview playground deployment." | |
| missing=1 | |
| fi | |
| done | |
| exit "$missing" | |
| - name: Run all experiments | |
| id: run_evals | |
| run: | | |
| for key in ANTHROPIC_API_KEY OPENAI_API_KEY; do | |
| if [[ -z "${!key}" ]]; then | |
| echo "::error title=Missing eval secret::$key is required; agent-eval would skip those experiments." | |
| exit 1 | |
| fi | |
| done | |
| yarn workspace agent-eval run eval | |
| env: | |
| EVAL_ONLY: ${{ inputs.evals || '' }} | |
| EVAL_ALL: ${{ steps.conditions.outputs.all_evals }} | |
| EVAL_STORYBOOK_LATEST: ${{ steps.conditions.outputs.storybook_latest }} | |
| ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} | |
| OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} | |
| VERCEL_PROJECT_ID: ${{ secrets.VERCEL_PROJECT_ID }} | |
| VERCEL_TEAM_ID: ${{ secrets.VERCEL_TEAM_ID }} | |
| VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }} | |
| - name: Check eval results | |
| if: ${{ always() }} | |
| id: check_results | |
| shell: bash | |
| run: | | |
| if [[ ! -d agent-eval/results ]]; then | |
| echo "::error title=Missing eval results::Expected agent-eval/results to exist before deploying the playground." | |
| exit 1 | |
| fi | |
| if ! find agent-eval/results -type f -print -quit | grep -q .; then | |
| echo "::error title=No eval result files::Expected agent-eval/results to contain files before deploying the playground." | |
| exit 1 | |
| fi | |
| echo "has_result_files=true" >> "$GITHUB_OUTPUT" | |
| - name: Compute offline metrics | |
| if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }} | |
| # Write analysis.json next to each run's result.json so the metrics ride | |
| # along in the uploaded artifact. | |
| continue-on-error: true | |
| run: yarn workspace agent-eval run results:analyze | |
| - name: Archive eval results | |
| if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }} | |
| # Tar before upload: result files may contain characters (e.g. the colon | |
| # in test:stories output names) that actions/upload-artifact rejects. | |
| run: tar -czf agent-eval-results.tgz -C agent-eval results | |
| - name: Upload eval results | |
| if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }} | |
| uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 | |
| with: | |
| name: agent-eval-results | |
| path: agent-eval-results.tgz | |
| retention-days: 30 | |
| - name: Resolve Vercel target | |
| if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }} | |
| id: vercel_target | |
| shell: bash | |
| run: | | |
| # Keep this rule in sync with jobs.eval.environment.name above. | |
| if [[ ("$GITHUB_EVENT_NAME" == "workflow_dispatch" || "$GITHUB_EVENT_NAME" == "schedule") && "$GITHUB_REF" == "refs/heads/next" ]]; then | |
| echo "target=production" >> "$GITHUB_OUTPUT" | |
| else | |
| echo "target=preview" >> "$GITHUB_OUTPUT" | |
| fi | |
| # Vercel CLI commands run from the repository root: the storybook-evals | |
| # project's Root Directory setting is agent-eval, and in a monorepo | |
| # the CLI must not be invoked from the subdirectory (vercel.com/docs/monorepos) | |
| # or `deploy --prebuilt` resolves trace paths against the wrong base. | |
| - name: Link Vercel project | |
| if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }} | |
| env: | |
| VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }} | |
| VERCEL_TEAM_ID: ${{ secrets.VERCEL_TEAM_ID }} | |
| run: ./agent-eval/node_modules/.bin/vercel link --yes --team "$VERCEL_TEAM_ID" --project "$VERCEL_PROJECT_NAME" --token "$VERCEL_TOKEN" | |
| - name: Pull Vercel environment | |
| if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }} | |
| shell: bash | |
| env: | |
| VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }} | |
| VERCEL_TARGET: ${{ steps.vercel_target.outputs.target }} | |
| run: | | |
| if [[ "$VERCEL_TARGET" == "production" ]]; then | |
| ./agent-eval/node_modules/.bin/vercel pull --yes --environment=production --token "$VERCEL_TOKEN" | |
| else | |
| branch="${GITHUB_HEAD_REF:-$GITHUB_REF_NAME}" | |
| ./agent-eval/node_modules/.bin/vercel pull --yes --environment=preview --git-branch "$branch" --token "$VERCEL_TOKEN" | |
| fi | |
| - name: Build playground | |
| if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' }} | |
| id: build_playground | |
| env: | |
| VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }} | |
| VERCEL_TARGET: ${{ steps.vercel_target.outputs.target }} | |
| run: ./agent-eval/node_modules/.bin/vercel build --yes --target="$VERCEL_TARGET" --token "$VERCEL_TOKEN" | |
| - name: Deploy playground | |
| if: ${{ always() && steps.check_results.outputs.has_result_files == 'true' && steps.build_playground.outcome == 'success' }} | |
| id: deploy_playground | |
| shell: bash | |
| env: | |
| VERCEL_TOKEN: ${{ secrets.VERCEL_TOKEN }} | |
| VERCEL_TARGET: ${{ steps.vercel_target.outputs.target }} | |
| run: | | |
| deploy_output="$(./agent-eval/node_modules/.bin/vercel deploy --prebuilt --yes --target="$VERCEL_TARGET" --token "$VERCEL_TOKEN")" | |
| printf '%s\n' "$deploy_output" | |
| url="$( | |
| printf '%s\n' "$deploy_output" | | |
| awk '/https:\/\/[^[:space:]]+\.vercel\.app/ { match($0, /https:\/\/[^[:space:]]+\.vercel\.app[^[:space:]]*/); if (RSTART) found=substr($0,RSTART,RLENGTH) } END { print found }' | |
| )" | |
| if [[ -z "$url" ]]; then | |
| echo "::error title=Missing deployment URL::Vercel deploy did not print a deployment URL." | |
| exit 1 | |
| fi | |
| echo "url=$url" >> "$GITHUB_OUTPUT" | |
| - name: Update PR description with eval results | |
| if: ${{ always() && github.event_name == 'pull_request' && steps.check_results.outputs.has_result_files == 'true' }} | |
| shell: bash | |
| env: | |
| GH_TOKEN: ${{ github.token }} | |
| PR_NUMBER: ${{ github.event.pull_request.number }} | |
| PLAYGROUND_URL: ${{ steps.deploy_playground.outputs.url }} | |
| RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} | |
| run: | | |
| summary="$(node agent-eval/scripts/render-results-summary.mjs)" | |
| body="$(gh api "repos/$GITHUB_REPOSITORY/pulls/$PR_NUMBER" --jq '.body // ""')" | |
| printf '%s' "$body" | | |
| SUMMARY="$summary" node agent-eval/scripts/upsert-pr-eval-results.mjs | | |
| gh api --method PATCH "repos/$GITHUB_REPOSITORY/pulls/$PR_NUMBER" --field body=@- --silent | |
| - name: Write deployment summary | |
| if: ${{ always() && steps.deploy_playground.outcome == 'success' }} | |
| env: | |
| VERCEL_TARGET: ${{ steps.vercel_target.outputs.target }} | |
| PLAYGROUND_URL: ${{ steps.deploy_playground.outputs.url }} | |
| run: | | |
| { | |
| echo "### Agent eval playground" | |
| echo | |
| echo "- Target: \`$VERCEL_TARGET\`" | |
| echo "- Project: \`$VERCEL_PROJECT_NAME\`" | |
| echo "- URL: $PLAYGROUND_URL" | |
| } >> "$GITHUB_STEP_SUMMARY" | |
| - name: Update the eval gate thread with the result | |
| if: >- | |
| always() && | |
| !cancelled() && | |
| steps.gate_thread.outcome == 'success' | |
| env: | |
| GH_TOKEN: ${{ github.token }} | |
| PR_NUMBER: ${{ steps.gate_pr.outputs.pr_number }} | |
| EVALUATED_HEAD_SHA: ${{ steps.gate_head.outputs.evaluated_head_sha }} | |
| RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} | |
| PLAYGROUND_URL: ${{ steps.deploy_playground.outputs.url }} | |
| EVAL_OUTCOME: ${{ steps.check_results.outcome == 'success' && steps.run_evals.outcome || 'failure' }} | |
| run: | | |
| summary_file="$(mktemp)" | |
| node agent-eval/scripts/render-results-summary.mjs > "$summary_file" || true | |
| node agent-eval/scripts/eval-gate-thread.ts result \ | |
| --pr "$PR_NUMBER" --sha "$EVALUATED_HEAD_SHA" --outcome "$EVAL_OUTCOME" \ | |
| --run-url "$RUN_URL" --playground-url "$PLAYGROUND_URL" --summary-file "$summary_file" | |
| # Weekly / next-dispatch summary for the #sb-monitoring channel. | |
| - name: Notify Slack | |
| if: >- | |
| always() && | |
| !cancelled() && | |
| (github.event_name == 'schedule' || | |
| (github.event_name == 'workflow_dispatch' && github.ref == 'refs/heads/next')) | |
| shell: bash | |
| env: | |
| SLACK_WEBHOOK_URL: ${{ secrets.SLACK_AGENT_EVAL_WEBHOOK_URL }} | |
| PLAYGROUND_URL: ${{ steps.deploy_playground.outputs.url }} | |
| RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} | |
| EVAL_OUTCOME: ${{ steps.run_evals.outcome }} | |
| run: | | |
| if [[ -z "$SLACK_WEBHOOK_URL" ]]; then | |
| echo "::warning title=Missing Slack webhook::SLACK_AGENT_EVAL_WEBHOOK_URL is not set; skipping #sb-monitoring notify." | |
| exit 0 | |
| fi | |
| text="$(FORMAT=slack node agent-eval/scripts/render-results-summary.mjs)" | |
| payload="$(jq -n --arg text "$text" '{text: $text}')" | |
| curl --fail --silent --show-error --connect-timeout 10 --max-time 30 \ | |
| -X POST -H 'Content-type: application/json' --data "$payload" "$SLACK_WEBHOOK_URL" | |
| # New push on an agent-eval:eval PR: previous eval results no longer cover | |
| # the head. Pushes do not auto-reopen resolved review threads, so explicitly | |
| # reopen the gate thread without rerunning evals. | |
| reopen-eval-gate: | |
| name: Reopen or resolve the eval gate thread | |
| if: >- | |
| github.event_name == 'pull_request' && | |
| !github.event.pull_request.head.repo.fork && | |
| ((github.event.action == 'synchronize' && | |
| contains(github.event.pull_request.labels.*.name, 'agent-eval:eval')) || | |
| (github.event.action == 'unlabeled' && github.event.label.name == 'agent-eval:eval')) | |
| runs-on: ubuntu-latest | |
| permissions: | |
| # unresolveReviewThread needs contents: write + pull-requests: write. | |
| contents: write | |
| pull-requests: write | |
| steps: | |
| - name: Checkout eval gate script | |
| uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 | |
| with: | |
| persist-credentials: false | |
| sparse-checkout: | | |
| agent-eval/scripts | |
| scripts/utils | |
| - name: Setup Node.js | |
| uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 | |
| with: | |
| node-version-file: '.nvmrc' | |
| - name: Reopen the eval gate thread for the new head | |
| if: ${{ github.event.action == 'synchronize' }} | |
| env: | |
| GH_TOKEN: ${{ github.token }} | |
| PR_NUMBER: ${{ github.event.pull_request.number }} | |
| NEW_HEAD_SHA: ${{ github.event.pull_request.head.sha }} | |
| run: node agent-eval/scripts/eval-gate-thread.ts reopen --pr "$PR_NUMBER" --sha "$NEW_HEAD_SHA" | |
| - name: Resolve the eval gate thread after opting out | |
| if: ${{ github.event.action == 'unlabeled' }} | |
| env: | |
| GH_TOKEN: ${{ github.token }} | |
| PR_NUMBER: ${{ github.event.pull_request.number }} | |
| HEAD_SHA: ${{ github.event.pull_request.head.sha }} | |
| run: >- | |
| node agent-eval/scripts/eval-gate-thread.ts opt-out | |
| --pr "$PR_NUMBER" --sha "$HEAD_SHA" --actor "$GITHUB_ACTOR" |