Repository navigation
266 lines (259 loc) · 10.5 KB
/
Copy pathbenchmark.yml
File metadata and controls
266 lines (259 loc) · 10.5 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
name: Benchmark
# Splits the Online-Mind2Web tasks into shards, one job each, and runs every
# setup on every task of its shard (see src/plan.ts), has WebJudge decide
# success, then publishes results/<date>-run-<id>/ (raw data, meta.json,
# report.md), the results index and the README's "Latest results" section.
#
# Versions accept anything npm does: an exact version, a dist-tag
# (latest, next) or a range. The report records the exact version that ran.
on:
workflow_dispatch:
inputs:
webdriverio:
description: '@wdio/cli version for wdio session (v10 and up)'
default: latest
wdio-mcp:
description: '@wdio/mcp version'
default: latest
playwright-mcp:
description: '@playwright/mcp version'
default: 0.0.83
agent-browser:
description: 'agent-browser version'
default: 0.38.2
playwright-cli:
description: '@playwright/cli version'
default: 0.1.22
stagehand:
description: 'browserbase/stagehand git ref (branch, tag or sha) to build its Claude Code MCP server from; latest = newest release tag'
default: cd7b230778cf92269e4cb90e80d97f5113781c51
model:
description: 'Model for every agent: Claude through Anthropic, others through OpenRouter (needs the OPENROUTER_API_KEY secret)'
type: choice
options: [claude-sonnet-5, deepseek-flash-4-1]
default: claude-sonnet-5
runs:
description: Runs per task and setup
default: '1'
setups:
description: Comma-separated setup ids, or "all"
default: all
tasks:
description: Comma-separated task ids, or "all"
default: all
seed:
description: Seed for the run order
default: '1'
shards:
description: 'Jobs to split the tasks into, each on its own runner; every job runs every setup on its tasks'
default: '10'
concurrency:
description: Runs at the same time in each job
default: '4'
publish:
description: Commit the results to the repository
type: boolean
default: true
concurrency:
group: benchmark
cancel-in-progress: false
permissions:
contents: read
jobs:
plan:
name: Plan
runs-on: ubuntu-latest
outputs:
setups: ${{ steps.plan.outputs.setups }}
matrix: ${{ steps.plan.outputs.matrix }}
result-id: ${{ steps.plan.outputs.result-id }}
steps:
- uses: actions/checkout@v5
- uses: actions/setup-node@v5
with:
node-version: 24
cache: npm
- run: npm ci
- id: plan
env:
SETUPS: ${{ inputs.setups }}
TASKS: ${{ inputs.tasks }}
SHARDS: ${{ inputs.shards }}
run: |
plan=$(node src/plan.ts "$SETUPS" "$TASKS" "$SHARDS")
echo "setups=$(jq -r .setups <<< "$plan")" >> "$GITHUB_OUTPUT"
echo "matrix=$(jq -c .matrix <<< "$plan")" >> "$GITHUB_OUTPUT"
jq -r '.matrix[] | "shard \(.shard): \(.tasks)"' <<< "$plan"
echo "result-id=$(date -u +%Y-%m-%d)-run-${GITHUB_RUN_ID}" >> "$GITHUB_OUTPUT"
bench:
# every setup in each shard's job, so the setups on one task share an IP
# address and a time window (see src/plan.ts)
name: Run and judge (shard ${{ matrix.shard }})
needs: plan
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
include: ${{ fromJSON(needs.plan.outputs.matrix) }}
# GitHub's maximum. A shard of 10 tasks × 7 setups takes 40 to 60
# minutes; the limits below only matter for a run with few shards
timeout-minutes: 360
steps:
- uses: actions/checkout@v5
- uses: actions/setup-node@v5
with:
node-version: 24
cache: npm
- run: npm ci
- uses: actions/setup-python@v5
with:
python-version: '3.11'
- name: Run benchmark
env:
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENROUTER_API_KEY: ${{ secrets.OPENROUTER_API_KEY }}
WDIO_VERSION: ${{ inputs.webdriverio }}
WDIO_MCP_VERSION: ${{ inputs.wdio-mcp }}
PLAYWRIGHT_MCP_VERSION: ${{ inputs.playwright-mcp }}
STAGEHAND_REF: ${{ inputs.stagehand }}
AGENT_BROWSER_VERSION: ${{ inputs.agent-browser }}
PLAYWRIGHT_CLI_VERSION: ${{ inputs.playwright-cli }}
HF_TOKEN: ${{ secrets.HF_TOKEN }}
CONCURRENCY: ${{ inputs.concurrency }}
SETUP: ${{ needs.plan.outputs.setups }}
TASKS: ${{ matrix.tasks }}
SHARD: ${{ matrix.shard }}
RUNS: ${{ inputs.runs }}
MODEL: ${{ inputs.model }}
SEED: ${{ inputs.seed }}
OUT_DIR: results/${{ needs.plan.outputs.result-id }}
# starts no run after 300 min, so the runs going (10 min at most) end
# and judging and uploading still fit into the job's 360; the step's
# own limit is the backstop if the runner hangs. Every finished run is
# already in runs-*.jsonl, so either way the results so far are kept
timeout-minutes: 320
# a virtual display for every job: Stagehand's facade always launches a
# headed browser, and the other setups should run under the same conditions
run: xvfb-run --auto-servernum node src/run.ts --setups "$SETUP" --tasks "$TASKS" --runs "$RUNS" --model "$MODEL" --seed "$SEED" --concurrency "$CONCURRENCY" --shard "$SHARD" --deadline-min 300 --out-dir "$OUT_DIR"
- name: Judge with WebJudge
# a judge from another model family than the agents; runs it cannot
# judge stay pending and are left out of the results, not failed
if: ${{ always() }}
timeout-minutes: 30
env:
HF_TOKEN: ${{ secrets.HF_TOKEN }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
OUT_DIR: results/${{ needs.plan.outputs.result-id }}
run: node src/judge.ts "$OUT_DIR"
- uses: actions/upload-artifact@v4
if: always()
with:
name: results-${{ matrix.shard }}
path: results/${{ needs.plan.outputs.result-id }}
if-no-files-found: warn
- uses: actions/upload-artifact@v4
if: always()
with:
name: transcripts-${{ matrix.shard }}
path: transcripts/${{ needs.plan.outputs.result-id }}
if-no-files-found: ignore
retention-days: 90
publish:
name: Publish results
needs: [plan, bench]
if: ${{ always() && needs.plan.result == 'success' }}
runs-on: ubuntu-latest
permissions:
contents: write
statuses: read
steps:
- uses: actions/checkout@v5
- uses: actions/setup-node@v5
with:
node-version: 24
cache: npm
- run: npm ci
- uses: actions/download-artifact@v5
with:
pattern: results-*
merge-multiple: true
path: results/${{ needs.plan.outputs.result-id }}
- name: Check every shard reported
env:
RESULT_ID: ${{ needs.plan.outputs.result-id }}
MATRIX: ${{ needs.plan.outputs.matrix }}
run: |
expected=$(jq length <<< "$MATRIX")
found=$(find "results/$RESULT_ID" -maxdepth 1 -name 'meta-*.json' | wc -l)
if [ "$found" -lt "$expected" ]; then
echo "::warning::only $found of $expected shards left results; the report covers their tasks only"
fi
- name: Write report
env:
RESULT_ID: ${{ needs.plan.outputs.result-id }}
run: |
node src/publish.ts "results/$RESULT_ID"
cat "results/$RESULT_ID/report.md" >> "$GITHUB_STEP_SUMMARY"
- name: Commit results
if: ${{ inputs.publish }}
env:
RESULT_ID: ${{ needs.plan.outputs.result-id }}
# README.md and results/README.md are generated from every published
# run, so anything pushed to main while this run was going (another
# run's results, a README edit) conflicts with a plain rebase. Instead,
# start each attempt from the current main, put this run's raw results
# back and regenerate the derived files there, then push.
run: |
git config user.name "github-actions[bot]"
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
saved="$RUNNER_TEMP/result-$RESULT_ID"
cp -R "results/$RESULT_ID" "$saved"
lock=$(sha256sum package-lock.json)
for attempt in 1 2 3 4 5; do
git fetch --depth=1 origin main
git reset --hard FETCH_HEAD
if [ "$(sha256sum package-lock.json)" != "$lock" ]; then
npm ci
lock=$(sha256sum package-lock.json)
fi
rm -rf "results/$RESULT_ID"
cp -R "$saved" "results/$RESULT_ID"
node src/publish.ts "results/$RESULT_ID"
git add results README.md
if git diff --cached --quiet; then
echo "results/$RESULT_ID is already published on main"
exit 0
fi
git commit -m "results: $RESULT_ID" -m "$GITHUB_SERVER_URL/$GITHUB_REPOSITORY/actions/runs/$GITHUB_RUN_ID"
git push origin HEAD:main && exit 0
sleep $((attempt * 10))
done
exit 1
- name: Wait for benchmark.webdriver.io
# Vercel's Git integration deploys every push to main; make sure this
# run's commit was deployed and the live site serves its results
if: ${{ inputs.publish }}
env:
GH_TOKEN: ${{ github.token }}
RESULT_ID: ${{ needs.plan.outputs.result-id }}
run: |
sha=$(git rev-parse HEAD)
for _ in $(seq 1 90); do
state=$(gh api "repos/$GITHUB_REPOSITORY/commits/$sha/status" --jq '[.statuses[] | select(.context == "Vercel")][0].state // ""')
if [ "$state" = "failure" ] || [ "$state" = "error" ]; then
echo "::error::Vercel could not deploy $sha"
exit 1
fi
if [ "$state" = "success" ] && curl -fsS "https://benchmark.webdriver.io/data.json?ts=$(date +%s)" | grep -q "\"$RESULT_ID\""; then
echo "Live on [benchmark.webdriver.io](https://benchmark.webdriver.io) (commit \`${sha:0:7}\`)." >> "$GITHUB_STEP_SUMMARY"
exit 0
fi
sleep 10
done
echo "::error::benchmark.webdriver.io does not show $RESULT_ID 15 minutes after the push"
exit 1
- uses: actions/upload-artifact@v4
if: ${{ !inputs.publish }}
with:
name: report
path: results/${{ needs.plan.outputs.result-id }}