diff --git a/.github/workflows/eval-refresh.yml b/.github/workflows/eval-refresh.yml index 4f922f4..b40d931 100644 --- a/.github/workflows/eval-refresh.yml +++ b/.github/workflows/eval-refresh.yml @@ -1,28 +1,35 @@ name: Refresh eval results -# Two schedules, because everything weekly costs $279 a month against a $200 -# budget, and cadence is a better lever than dropping scenarios: the cheap -# experiments carry the signal and the expensive ones carry the headline. +# One schedule, and it does not run the benchmark. # -# Weekly, about $185 a month: the frontier agents plus the weak pair. The weak -# pair is the only source of failures in the suite and the only place a skills -# difference has been observable, and it is the most sensitive regression -# detector because it sits on the pass/fail boundary rather than passing -# everything. +# The benchmark ran weekly and monthly from August to October 2026, about $185 +# a month plus $90-110 for each monthly matrix. Paused on 1 October, because +# the question "what decision does this run inform?" had stopped having an +# answer: eight product findings were open and none had shipped, so the runs +# were widening a queue nobody was consuming. Five weeklies produced one +# harness defect and a great deal of variance data about a delta we already +# know we cannot measure precisely enough (#2). # -# Monthly adds the `-no-skills` twins of the two expensive agents. Their delta -# has measured zero on every scenario, four times, and will not move week to -# week; a month is soon enough to notice if it ever does. +# What stays automated is the regression suite: three scenarios guarding +# mistakes we have already seen and fixed, across all six experiments. Every +# agent passing is the expected state, so a failing check fails the job here — +# unlike the benchmark, where a failure is a score — and the run opens an issue +# instead of going red where nobody looks. The 21 September failure sat +# unnoticed until somebody asked about it four days later. # -# Costs are measured rather than estimated: $20.53 for a fourteen-scenario -# Claude pass, $7.40 for gpt-5.6 (which resolves to gpt-5.6-sol), about $4 for -# the weak model, pennies for judging. +# Cost and duration are not yet measured. The $5-and-ten-minutes figure that +# used to sit here came from `.plans/delivery-plan.md` at planning time, before +# any regression run existed; measured wall clock on the benchmark is about +# 3.4 minutes per cell serialised, which would put eighteen cells nearer an +# hour. Replace this sentence with a measurement after the first run. +# +# **Benchmark runs are dispatched against a bucket of work, not a calendar.** +# Something changed what we measure, or something shipped to the product, and +# the run tells us what that did. See "Releases" and the roadmap (#24). on: schedule: - # Weekly: frontier agents and the weak pair. + # Weekly, regression only. Catches a guarded mistake coming back. - cron: '0 6 * * 1' - # Monthly: the full matrix, adding the no-skills twins. - - cron: '0 8 1 * *' workflow_dispatch: inputs: experiments: @@ -64,6 +71,10 @@ permissions: # run when the provider stops answering, instead of letting every remaining # job fail the same way. actions: write + # The regression alert below opens or updates an issue when the suite fails. + # Work lives in GitHub Issues (AGENTS.md), and a red run in a tab nobody has + # open is not a notification. + issues: write # Scoring runs against one shared Hookdeck project, so two workflow runs must # never overlap. Queue rather than cancel: a cancelled run leaves the project @@ -87,25 +98,33 @@ jobs: run: | set -euo pipefail - # A scheduled run has no inputs, so the two crons carry their own - # experiment sets. Weekly is the frontier agents and the weak pair; - # the monthly one adds the no-skills twins for a full matrix. + # A scheduled run has no inputs, so the cron carries its own. One + # cron now, and it runs the regression suite across every experiment; + # the benchmark is dispatched against a bucket of work instead. if [ "${{ github.event_name }}" = "schedule" ]; then - if [ "${{ github.event.schedule }}" = "0 8 1 * *" ]; then - experiments="claude-code-sonnet-5,claude-code-sonnet-5-no-skills,codex-gpt-5.6,codex-gpt-5.6-no-skills,codex-gpt-5.4-mini,codex-gpt-5.4-mini-no-skills" - else - experiments="claude-code-sonnet-5,codex-gpt-5.6,codex-gpt-5.4-mini,codex-gpt-5.4-mini-no-skills" - fi + # Regression only. No experiment override: the suite decides who + # runs, and naming them here was both redundant and wrong — it + # listed six while only four carried `regression`, so the weak pair + # looked included and was not. + experiments="" eval_id="" - suite="benchmark" - experiment_suite="benchmark,no-skills" - runs="1" + suite="regression" + # `regression`, not `benchmark,no-skills`. A regression eval maps to + # the `regression` experiment suite and nothing else (see the + # discovery step below), so asking for the benchmark suites matched + # no pairs at all: `prepare` exited 1 every week and the alert fired + # on a run that had measured nothing. + experiment_suite="regression" + # Two attempts. A regression scenario guards a mistake we have + # already fixed, so every agent passing is the expected state and a + # single failure is either the mistake returning or noise. Paying + # for a second attempt on the failing cell is cheaper than a person + # deciding which it was. + runs="2" timeout_sec="900" - # Merge, never overwrite. The weekly run covers four experiments and - # the monthly covers six, so an overwriting weekly deletes the - # -no-skills twins the monthly produced and the published page loses - # the skills comparison for three weeks in four. It did exactly that - # on 17 August, dropping the scoreboard from six columns to four. + # Regression results do not reach the published scoreboard — the + # page renders the benchmark suite only — so this writes + # `regression-eval-results.json` and nothing a reader sees moves. do_merge="true" else experiments="${{ inputs.experiments }}" @@ -421,6 +440,33 @@ jobs: exit 1 fi + - name: Check the regression suite passed + # A scenario failing its checks does not fail the eval step: in the + # benchmark that is a score, not an error. On the regression suite it + # is the whole point — every agent passing is the expected state, so a + # red check means either a guarded mistake has returned or the harness + # is broken, and both want somebody's attention. + # + # Keyed on the suite, not on the trigger. The distinction is + # benchmark-versus-regression, and `github.event_name == 'schedule'` + # only encodes that while the single cron happens to be regression-only + # — add a benchmark cron back and every benchmark cell that merely + # scored a failure would fail its job. The schedule clause stays + # because a dispatched run is somebody already reading the output. + if: ${{ matrix.eval_suite == 'regression' && github.event_name == 'schedule' }} + shell: bash + run: | + set -euo pipefail + + result=".eval-runs/${{ matrix.experiment }}/${{ matrix.eval_id }}.json" + [ -f "$result" ] || exit 0 + + if [ "$(jq -r '.passed' "$result")" = "false" ]; then + failed="$(jq -r '[.checks[]? | select(.passed == false) | .name] | join("; ")' "$result")" + echo "::error::${{ matrix.eval_id }} failed for ${{ matrix.experiment }}: ${failed:-no check names recorded}" + exit 1 + fi + - name: Upload raw results uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: @@ -439,9 +485,12 @@ jobs: # questions worth asking. Nearly every real question is comparative # ("this flipped between runs, why?") and answering one needs the # evidence from both. Weekly runs at 3 days meant the previous run's - # evidence expired before its successor even started. The monthly - # -no-skills twins need 35 days to compare two consecutive runs at - # all, which is why this is not 30 either. + # evidence expired before its successor even started, and when the + # `-no-skills` twins refreshed monthly, comparing two consecutive runs + # of them needed 35 days — which is why this is not 30 either. The + # benchmark is dispatched rather than scheduled now, so the gap + # between two comparable runs is nobody's to predict, which argues for + # the ceiling rather than against it. # # Storage is free on a public repository, so the only trade-off here # is against nothing. @@ -450,7 +499,7 @@ jobs: publish-results: needs: [prepare, run-evals] runs-on: ubuntu-latest - # A way to say "run, but do not publish this week". + # A way to say "run, but do not publish". # # Publishing was gated on the matrix succeeding and nothing else, so a run # against a `main` we already knew was wrong would publish anyway. On 24 @@ -548,6 +597,15 @@ jobs: fi - name: Publish snapshot + # Only when this run actually measured benchmark cells. `Export results` + # already skips the benchmark file when there are no benchmark pairs, + # but `publish-snapshot` reads that untouched file and writes a fresh + # `latest.json`, a new `results/runs/.json` and an `index.json` + # entry regardless — so a regression-only run would commit a snapshot + # whose `runId` names a workflow run that produced none of its rows. + # README states the opposite as a contract: a figure can be traced to + # the job that produced it. + if: ${{ contains(needs.prepare.outputs.pairs, '"eval_suite":"benchmark"') }} # `apps/web/src/data/eval-results.json` is the preview app's own input. # Anything outside this repo reading it is coupled to where our app # keeps its fixtures. `results/` is the contract: latest.json for the @@ -631,3 +689,72 @@ jobs: echo "::error::exhausted push attempts" exit 1 + + # A failing scheduled run used to be visible only to somebody who went + # looking. The 21 September failure sat unnoticed until it was asked about + # four days later, and the cause was a scorer publishing a topic the project + # does not have (#79) — a defect that cost a whole run's publishing and + # announced itself nowhere. + # + # Scheduled only. A dispatched run has a person attached to it by definition. + regression-alert: + name: regression-alert + needs: [prepare, run-evals, publish-results] + # `cancelled()` as well as `failure()`: the provider-dead path above calls + # `gh run cancel`, so a zero credit balance produces a cancelled run and + # would otherwise alert nobody. `publish-results` is in `needs` because it + # is a third of the pipeline and has failed twice on its own — a shallow + # clone that could not rebase, and a re-run against a moved branch (#74). + if: ${{ (failure() || cancelled()) && github.event_name == 'schedule' }} + runs-on: ubuntu-latest + steps: + - name: Open or update the alert issue + env: + GH_TOKEN: ${{ github.token }} + # This job has no checkout, so `gh` has no git remote to infer the + # repository from — and it does not fall back to `GITHUB_REPOSITORY`. + # Without this the alert step fails and notifies nobody, which is the + # failure it exists to prevent. + GH_REPO: ${{ github.repository }} + RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + shell: bash + run: | + set -euo pipefail + + # One issue, reopened and commented rather than a new one each week. + # A suite that breaks and stays broken should read as one problem + # getting worse, not as four identical issues nobody closes. + existing="$(gh issue list --label regression-alert --state open \ + --json number --jq '.[0].number // empty')" + + # Indented for the YAML block scalar, then de-indented for GitHub: + # four or more leading spaces after a blank line is a markdown code + # block, so the first version of this rendered the whole alert as + # monospace with literal `**` and `1.` in it. + body="$(sed 's/^ //' <