From d6d1b9b1d12a1bda599ffcd6478ff14a101d8610 Mon Sep 17 00:00:00 2001 From: Phil Leggetter Date: Thu, 1 Oct 2026 08:32:20 +0100 Subject: [PATCH 1/4] Stop running the benchmark on a schedule; keep a regression run that tells us MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The benchmark ran weekly and monthly from August, about $185 a month plus $90-110 a matrix. Paused because the question "what decision does this run inform?" had stopped having an answer: eight product findings were open and none had shipped, so the runs were widening a queue nobody was consuming. Five weeklies produced one harness defect and a great deal of variance data about a delta we already know we cannot measure precisely enough. What stays automated is the regression suite — three scenarios guarding mistakes already seen and fixed, every experiment, no environment to stand up. Under $5 and under ten minutes. Two attempts per cell, because a regression scenario that fails once is either the mistake returning or noise, and paying for the second attempt is cheaper than a person deciding which. And it now says something when it fails. A failing scheduled run was visible only to somebody who went looking: the 21 September failure sat unnoticed for four days until it was asked about. On failure the run opens an issue labelled `regression-alert`, or comments on the open one, so a suite that breaks and stays broken reads as one problem getting worse rather than four identical issues nobody closes. The body says what a regression failure usually is — the harness, not a regression — and to read the transcript first. Benchmark runs are now dispatched against a bucket of work: an eval change, or a product change this benchmark found. A run belonging to neither is buying data nobody has a decision waiting on. AGENTS.md carries that, and the note that the workflow is disabled so dispatching needs enabling first. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01MQzUoMAwEBJWpEGVvVzSjK --- .github/workflows/eval-refresh.yml | 117 ++++++++++++++++++++++------- AGENTS.md | 31 ++++++++ 2 files changed, 119 insertions(+), 29 deletions(-) diff --git a/.github/workflows/eval-refresh.yml b/.github/workflows/eval-refresh.yml index 4f922f4..fca6c43 100644 --- a/.github/workflows/eval-refresh.yml +++ b/.github/workflows/eval-refresh.yml @@ -1,28 +1,29 @@ name: Refresh eval results -# Two schedules, because everything weekly costs $279 a month against a $200 -# budget, and cadence is a better lever than dropping scenarios: the cheap -# experiments carry the signal and the expensive ones carry the headline. +# One schedule, and it does not run the benchmark. # -# Weekly, about $185 a month: the frontier agents plus the weak pair. The weak -# pair is the only source of failures in the suite and the only place a skills -# difference has been observable, and it is the most sensitive regression -# detector because it sits on the pass/fail boundary rather than passing -# everything. +# The benchmark ran weekly and monthly from August to October 2026, about $185 +# a month plus $90-110 for each monthly matrix. Paused on 1 October, because +# the question "what decision does this run inform?" had stopped having an +# answer: eight product findings were open and none had shipped, so the runs +# were widening a queue nobody was consuming. Five weeklies produced one +# harness defect and a great deal of variance data about a delta we already +# know we cannot measure precisely enough (#2). # -# Monthly adds the `-no-skills` twins of the two expensive agents. Their delta -# has measured zero on every scenario, four times, and will not move week to -# week; a month is soon enough to notice if it ever does. +# What stays automated is the regression suite: three scenarios guarding +# mistakes we have already seen and fixed, across all six experiments, with no +# environment to stand up. Under $5 and under ten minutes. All agents passing +# is the expected state, so a failure is a signal rather than a score, and this +# workflow now says so out loud instead of going red where nobody looks — the +# 21 September failure sat unnoticed until somebody asked about it. # -# Costs are measured rather than estimated: $20.53 for a fourteen-scenario -# Claude pass, $7.40 for gpt-5.6 (which resolves to gpt-5.6-sol), about $4 for -# the weak model, pennies for judging. +# **Benchmark runs are dispatched against a bucket of work, not a calendar.** +# Something changed what we measure, or something shipped to the product, and +# the run tells us what that did. See "Releases" and the roadmap (#24). on: schedule: - # Weekly: frontier agents and the weak pair. + # Weekly, regression only. Catches a guarded mistake coming back. - cron: '0 6 * * 1' - # Monthly: the full matrix, adding the no-skills twins. - - cron: '0 8 1 * *' workflow_dispatch: inputs: experiments: @@ -64,6 +65,10 @@ permissions: # run when the provider stops answering, instead of letting every remaining # job fail the same way. actions: write + # The regression alert below opens or updates an issue when the suite fails. + # Work lives in GitHub Issues (AGENTS.md), and a red run in a tab nobody has + # open is not a notification. + issues: write # Scoring runs against one shared Hookdeck project, so two workflow runs must # never overlap. Queue rather than cancel: a cancelled run leaves the project @@ -91,21 +96,23 @@ jobs: # experiment sets. Weekly is the frontier agents and the weak pair; # the monthly one adds the no-skills twins for a full matrix. if [ "${{ github.event_name }}" = "schedule" ]; then - if [ "${{ github.event.schedule }}" = "0 8 1 * *" ]; then - experiments="claude-code-sonnet-5,claude-code-sonnet-5-no-skills,codex-gpt-5.6,codex-gpt-5.6-no-skills,codex-gpt-5.4-mini,codex-gpt-5.4-mini-no-skills" - else - experiments="claude-code-sonnet-5,codex-gpt-5.6,codex-gpt-5.4-mini,codex-gpt-5.4-mini-no-skills" - fi + # Regression only, every experiment. Three scenarios with no + # environment to stand up: under $5 and under ten minutes, against + # $185 a month for the benchmark weekly it replaces. + experiments="claude-code-sonnet-5,claude-code-sonnet-5-no-skills,codex-gpt-5.6,codex-gpt-5.6-no-skills,codex-gpt-5.4-mini,codex-gpt-5.4-mini-no-skills" eval_id="" - suite="benchmark" + suite="regression" experiment_suite="benchmark,no-skills" - runs="1" + # Two attempts. A regression scenario guards a mistake we have + # already fixed, so every agent passing is the expected state and a + # single failure is either the mistake returning or noise. Paying + # for a second attempt on the failing cell is cheaper than a person + # deciding which it was. + runs="2" timeout_sec="900" - # Merge, never overwrite. The weekly run covers four experiments and - # the monthly covers six, so an overwriting weekly deletes the - # -no-skills twins the monthly produced and the published page loses - # the skills comparison for three weeks in four. It did exactly that - # on 17 August, dropping the scoreboard from six columns to four. + # Regression results do not reach the published scoreboard — the + # page renders the benchmark suite only — so this writes + # `regression-eval-results.json` and nothing a reader sees moves. do_merge="true" else experiments="${{ inputs.experiments }}" @@ -631,3 +638,55 @@ jobs: echo "::error::exhausted push attempts" exit 1 + + # A failing scheduled run used to be visible only to somebody who went + # looking. The 21 September failure sat unnoticed until it was asked about + # four days later, and the cause was a scorer publishing a topic the project + # does not have (#79) — a defect that cost a whole run's publishing and + # announced itself nowhere. + # + # Scheduled only. A dispatched run has a person attached to it by definition. + regression-alert: + name: regression-alert + needs: [prepare, run-evals] + if: failure() && github.event_name == 'schedule' + runs-on: ubuntu-latest + steps: + - name: Open or update the alert issue + env: + GH_TOKEN: ${{ github.token }} + RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + shell: bash + run: | + set -euo pipefail + + # One issue, reopened and commented rather than a new one each week. + # A suite that breaks and stays broken should read as one problem + # getting worse, not as four identical issues nobody closes. + existing="$(gh issue list --label regression-alert --state open \ + --json number --jq '.[0].number // empty')" + + body="The weekly regression run failed: ${RUN_URL} + + Regression scenarios guard mistakes we have already seen and fixed, so + every agent passing is the expected state. A failure here is one of + three things, in rough order of likelihood: + + 1. **The harness broke.** A scorer, a seed or a credential — the most + common cause by some distance, and the one to rule out first. + 2. **A guarded mistake came back.** The documentation or skill change + that fixed it has been undone or outgrown. + 3. **Noise.** The run already took two attempts per cell, so this is + the least likely of the three. + + Read the failing cell's transcript before anything else." + + if [ -n "$existing" ]; then + gh issue comment "$existing" --body "$body" + else + gh issue create \ + --title "Weekly regression run is failing" \ + --label regression-alert \ + --label harness \ + --body "$body" + fi diff --git a/AGENTS.md b/AGENTS.md index 1877702..e21746a 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -162,6 +162,37 @@ Two of those exist because the cheap version was tried first and did not work. `/v1/models` answers 200 with a zero credit balance, so a liveness check built on it passes while every real call fails; only an actual completion sees it. +## Runs + +**Nothing runs the benchmark on a schedule.** One cron remains, weekly, and it runs +the regression suite only: three scenarios guarding mistakes already seen and fixed, +across all six experiments, under $5 and under ten minutes. Every agent passing is the +expected state, so its job is to notice a guarded mistake coming back — and when it +fails it opens or updates an issue labelled `regression-alert`, because a red run in a +tab nobody has open is not a notification. + +**A benchmark run is dispatched against a bucket of work, never a date.** Two buckets, +and a run belongs to one of them: + +- **Eval changes** — a scenario, a scorer, the base prompt, the CLI pin, a new model or + experiment. The run measures what the change did to what we measure. +- **Product changes** — a fix, a documentation change or a skill change that this + benchmark found. The run measures whether it worked, which is the loop closing. + +If a proposed run belongs to neither, it is buying data nobody has a decision waiting +on. Paused on 1 October 2026 for exactly that reason: eight product findings were open +and none had shipped, so weekly and monthly matrices were widening a queue nobody was +consuming, at about $185 a month plus $90-110 a matrix. Five weeklies had produced one +harness defect and a lot of variance data about a delta we already know we cannot +measure precisely enough (#2). + +Re-enable the workflow before dispatching: it is disabled, which blocks +`workflow_dispatch` as well as the cron. + +```bash +gh workflow enable eval-refresh.yml +``` + ## What to work on next **GitHub Issues is the source of truth for work, and #24 is the order to do it From b12ef8813cc4e33c837d4e9245b72b59259aa08d Mon Sep 17 00:00:00 2001 From: Phil Leggetter Date: Thu, 1 Oct 2026 09:23:35 +0100 Subject: [PATCH 2/4] Fix the three blockers, and make a failing regression fail the run MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review found the cadence change could never have worked. Correcting it, plus the stale prose it left behind. The scheduled branch asked for `suite=regression` with `experiment_suite=benchmark,no-skills`, and a regression eval maps only to the `regression` experiment suite — so every pair was skipped, `prepare` exited 1, and the alert would have fired every Monday on a run that measured nothing. Verified the fix produces 18 pairs. The weak pair was named in a hardcoded six-experiment override while carrying no `regression` suite, so it looked included and was not. They now carry it: that pair holds nearly every failure in the suite, which is exactly where a guarded mistake returning would show up first. The override is gone — the suite decides who runs. `regression-alert` had no checkout and no `GH_REPO`, so `gh` had no repository to resolve against and the notifier would have failed silently. Its body was indented inside the `run:` block, which GitHub renders as a code block; it is now a heredoc de-indented to column zero, checked by running it. A failing scenario did not fail the job, so the alert could not fire for the thing it exists to detect. On the regression suite every agent passing is the expected state, so a failing check now fails the job — scheduled runs only, and the benchmark's semantics are untouched, where a failure is a score. A regression-only run would also have republished the benchmark snapshot: `publish-snapshot` writes a fresh `latest.json` from an untouched file, so the `runId` would have named a run that produced none of its rows, against a contract README states explicitly. Gated on benchmark pairs being present. And the claims. "Under $5 and under ten minutes" was a planning estimate from before any regression run existed, promoted to fact in AGENTS.md; the measured 3.4 minutes per cell puts eighteen cells nearer an hour, so both places now say it is unmeasured. Eleven stale cadence statements across AGENTS.md, LOOPS.md, the delivery plan and this workflow's own comments are corrected, including the one in Status that an agent reads before anything else. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01MQzUoMAwEBJWpEGVvVzSjK --- .github/workflows/eval-refresh.yml | 123 +++++++++++++++----- .plans/delivery-plan.md | 8 +- AGENTS.md | 40 ++++--- LOOPS.md | 9 +- experiments/codex-gpt-5.4-mini-no-skills.ts | 5 +- experiments/codex-gpt-5.4-mini.ts | 7 +- 6 files changed, 135 insertions(+), 57 deletions(-) diff --git a/.github/workflows/eval-refresh.yml b/.github/workflows/eval-refresh.yml index fca6c43..7f2cf00 100644 --- a/.github/workflows/eval-refresh.yml +++ b/.github/workflows/eval-refresh.yml @@ -11,11 +11,17 @@ name: Refresh eval results # know we cannot measure precisely enough (#2). # # What stays automated is the regression suite: three scenarios guarding -# mistakes we have already seen and fixed, across all six experiments, with no -# environment to stand up. Under $5 and under ten minutes. All agents passing -# is the expected state, so a failure is a signal rather than a score, and this -# workflow now says so out loud instead of going red where nobody looks — the -# 21 September failure sat unnoticed until somebody asked about it. +# mistakes we have already seen and fixed, across all six experiments. Every +# agent passing is the expected state, so a failing check fails the job here — +# unlike the benchmark, where a failure is a score — and the run opens an issue +# instead of going red where nobody looks. The 21 September failure sat +# unnoticed until somebody asked about it four days later. +# +# Cost and duration are not yet measured. The $5-and-ten-minutes figure that +# used to sit here came from `.plans/delivery-plan.md` at planning time, before +# any regression run existed; measured wall clock on the benchmark is about +# 3.4 minutes per cell serialised, which would put eighteen cells nearer an +# hour. Replace this sentence with a measurement after the first run. # # **Benchmark runs are dispatched against a bucket of work, not a calendar.** # Something changed what we measure, or something shipped to the product, and @@ -92,17 +98,23 @@ jobs: run: | set -euo pipefail - # A scheduled run has no inputs, so the two crons carry their own - # experiment sets. Weekly is the frontier agents and the weak pair; - # the monthly one adds the no-skills twins for a full matrix. + # A scheduled run has no inputs, so the cron carries its own. One + # cron now, and it runs the regression suite across every experiment; + # the benchmark is dispatched against a bucket of work instead. if [ "${{ github.event_name }}" = "schedule" ]; then - # Regression only, every experiment. Three scenarios with no - # environment to stand up: under $5 and under ten minutes, against - # $185 a month for the benchmark weekly it replaces. - experiments="claude-code-sonnet-5,claude-code-sonnet-5-no-skills,codex-gpt-5.6,codex-gpt-5.6-no-skills,codex-gpt-5.4-mini,codex-gpt-5.4-mini-no-skills" + # Regression only. No experiment override: the suite decides who + # runs, and naming them here was both redundant and wrong — it + # listed six while only four carried `regression`, so the weak pair + # looked included and was not. + experiments="" eval_id="" suite="regression" - experiment_suite="benchmark,no-skills" + # `regression`, not `benchmark,no-skills`. A regression eval maps to + # the `regression` experiment suite and nothing else (see the + # discovery step below), so asking for the benchmark suites matched + # no pairs at all: `prepare` exited 1 every week and the alert fired + # on a run that had measured nothing. + experiment_suite="regression" # Two attempts. A regression scenario guards a mistake we have # already fixed, so every agent passing is the expected state and a # single failure is either the mistake returning or noise. Paying @@ -428,6 +440,29 @@ jobs: exit 1 fi + - name: Fail on a regression that came back + # A scenario failing its checks does not fail the eval step: in the + # benchmark that is a score, not an error. On the regression suite it + # is the whole point — every agent passing is the expected state, so a + # red check means either a guarded mistake has returned or the harness + # is broken, and both want somebody's attention. + # + # Scheduled runs only. A dispatched regression run is somebody already + # looking at the output. + if: ${{ github.event_name == 'schedule' }} + shell: bash + run: | + set -euo pipefail + + result=".eval-runs/${{ matrix.experiment }}/${{ matrix.eval_id }}.json" + [ -f "$result" ] || exit 0 + + if [ "$(jq -r '.passed' "$result")" = "false" ]; then + failed="$(jq -r '[.checks[]? | select(.passed == false) | .name] | join("; ")' "$result")" + echo "::error::${{ matrix.eval_id }} failed for ${{ matrix.experiment }}: ${failed:-no check names recorded}" + exit 1 + fi + - name: Upload raw results uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: @@ -446,9 +481,12 @@ jobs: # questions worth asking. Nearly every real question is comparative # ("this flipped between runs, why?") and answering one needs the # evidence from both. Weekly runs at 3 days meant the previous run's - # evidence expired before its successor even started. The monthly - # -no-skills twins need 35 days to compare two consecutive runs at - # all, which is why this is not 30 either. + # evidence expired before its successor even started, and when the + # `-no-skills` twins refreshed monthly, comparing two consecutive runs + # of them needed 35 days — which is why this is not 30 either. The + # benchmark is dispatched rather than scheduled now, so the gap + # between two comparable runs is nobody's to predict, which argues for + # the ceiling rather than against it. # # Storage is free on a public repository, so the only trade-off here # is against nothing. @@ -457,7 +495,7 @@ jobs: publish-results: needs: [prepare, run-evals] runs-on: ubuntu-latest - # A way to say "run, but do not publish this week". + # A way to say "run, but do not publish". # # Publishing was gated on the matrix succeeding and nothing else, so a run # against a `main` we already knew was wrong would publish anyway. On 24 @@ -555,6 +593,15 @@ jobs: fi - name: Publish snapshot + # Only when this run actually measured benchmark cells. `Export results` + # already skips the benchmark file when there are no benchmark pairs, + # but `publish-snapshot` reads that untouched file and writes a fresh + # `latest.json`, a new `results/runs/.json` and an `index.json` + # entry regardless — so a regression-only run would commit a snapshot + # whose `runId` names a workflow run that produced none of its rows. + # README states the opposite as a contract: a figure can be traced to + # the job that produced it. + if: ${{ contains(needs.prepare.outputs.pairs, '"eval_suite":"benchmark"') }} # `apps/web/src/data/eval-results.json` is the preview app's own input. # Anything outside this repo reading it is coupled to where our app # keeps its fixtures. `results/` is the contract: latest.json for the @@ -655,6 +702,11 @@ jobs: - name: Open or update the alert issue env: GH_TOKEN: ${{ github.token }} + # This job has no checkout, so `gh` has no git remote to infer the + # repository from — and it does not fall back to `GITHUB_REPOSITORY`. + # Without this the alert step fails and notifies nobody, which is the + # failure it exists to prevent. + GH_REPO: ${{ github.repository }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} shell: bash run: | @@ -666,21 +718,28 @@ jobs: existing="$(gh issue list --label regression-alert --state open \ --json number --jq '.[0].number // empty')" - body="The weekly regression run failed: ${RUN_URL} - - Regression scenarios guard mistakes we have already seen and fixed, so - every agent passing is the expected state. A failure here is one of - three things, in rough order of likelihood: - - 1. **The harness broke.** A scorer, a seed or a credential — the most - common cause by some distance, and the one to rule out first. - 2. **A guarded mistake came back.** The documentation or skill change - that fixed it has been undone or outgrown. - 3. **Noise.** The run already took two attempts per cell, so this is - the least likely of the three. - - Read the failing cell's transcript before anything else." - + # Indented for the YAML block scalar, then de-indented for GitHub: + # four or more leading spaces after a blank line is a markdown code + # block, so the first version of this rendered the whole alert as + # monospace with literal `**` and `1.` in it. + body="$(sed 's/^ \{12\}//' < Date: Thu, 1 Oct 2026 11:46:59 +0100 Subject: [PATCH 3/4] Rename the step to what it does MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit "Fail on a regression that came back" names one of the two things a failure here means, and the less likely one — the step fails for a broken harness far more often than for a guarded mistake returning, which is why the alert body leads with that. It also reads as a wish rather than a check. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01MQzUoMAwEBJWpEGVvVzSjK --- .github/workflows/eval-refresh.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/eval-refresh.yml b/.github/workflows/eval-refresh.yml index 7f2cf00..01b738c 100644 --- a/.github/workflows/eval-refresh.yml +++ b/.github/workflows/eval-refresh.yml @@ -440,7 +440,7 @@ jobs: exit 1 fi - - name: Fail on a regression that came back + - name: Check the regression suite passed # A scenario failing its checks does not fail the eval step: in the # benchmark that is a score, not an error. On the regression suite it # is the whole point — every agent passing is the expected state, so a From e7346733ef187ea56d0d093078a052347ecc0aac Mon Sep 17 00:00:00 2001 From: Phil Leggetter Date: Thu, 1 Oct 2026 12:00:32 +0100 Subject: [PATCH 4/4] Key the check on the suite, de-indent by the right amount, alert on more MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four from the second review, all small, one of them a correction to a claim I made in a commit message. The new check keyed off `github.event_name == 'schedule'` while its own comment said the distinction was benchmark-versus-regression. Those are the same thing only while the single cron happens to be regression-only; add a benchmark cron back and every benchmark cell that merely scored a failure would fail its job. It now keys on `matrix.eval_suite == 'regression'`, which is what the comment meant, with the schedule clause kept because a dispatched run has somebody reading it. The alert body's `sed` stripped twelve spaces. YAML's block scalar had already removed ten, so the lines arrived with two and the `sed` matched nothing. It rendered correctly anyway, because two is below the four that makes a markdown code block — so the mechanism was dead and the output was fine by luck. The earlier commit message said this was "de-indented to column zero, checked by running it": I did run it, in a standalone script where the indent was twelve, which is not what YAML hands to bash. Now `sed 's/^ //'`, checked by parsing the workflow and running the body exactly as the runner would. The alert could not fire for a `publish-results` failure, which is a third of the pipeline and has failed on its own twice — a shallow clone that could not rebase, and a re-run against a moved branch (#74). Nor for a cancelled run, which is what the provider-dead path produces when credits run out. Both are covered now. And the "$5 and under 10 minutes" estimate this branch says it retired was still in the delivery plan, two lines below the bullet that was edited. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01MQzUoMAwEBJWpEGVvVzSjK --- .github/workflows/eval-refresh.yml | 21 +++++++++++++++------ .plans/delivery-plan.md | 8 +++++--- 2 files changed, 20 insertions(+), 9 deletions(-) diff --git a/.github/workflows/eval-refresh.yml b/.github/workflows/eval-refresh.yml index 01b738c..b40d931 100644 --- a/.github/workflows/eval-refresh.yml +++ b/.github/workflows/eval-refresh.yml @@ -447,9 +447,13 @@ jobs: # red check means either a guarded mistake has returned or the harness # is broken, and both want somebody's attention. # - # Scheduled runs only. A dispatched regression run is somebody already - # looking at the output. - if: ${{ github.event_name == 'schedule' }} + # Keyed on the suite, not on the trigger. The distinction is + # benchmark-versus-regression, and `github.event_name == 'schedule'` + # only encodes that while the single cron happens to be regression-only + # — add a benchmark cron back and every benchmark cell that merely + # scored a failure would fail its job. The schedule clause stays + # because a dispatched run is somebody already reading the output. + if: ${{ matrix.eval_suite == 'regression' && github.event_name == 'schedule' }} shell: bash run: | set -euo pipefail @@ -695,8 +699,13 @@ jobs: # Scheduled only. A dispatched run has a person attached to it by definition. regression-alert: name: regression-alert - needs: [prepare, run-evals] - if: failure() && github.event_name == 'schedule' + needs: [prepare, run-evals, publish-results] + # `cancelled()` as well as `failure()`: the provider-dead path above calls + # `gh run cancel`, so a zero credit balance produces a cancelled run and + # would otherwise alert nobody. `publish-results` is in `needs` because it + # is a third of the pipeline and has failed twice on its own — a shallow + # clone that could not rebase, and a re-run against a moved branch (#74). + if: ${{ (failure() || cancelled()) && github.event_name == 'schedule' }} runs-on: ubuntu-latest steps: - name: Open or update the alert issue @@ -722,7 +731,7 @@ jobs: # four or more leading spaces after a blank line is a markdown code # block, so the first version of this rendered the whole alert as # monospace with literal `**` and `1.` in it. - body="$(sed 's/^ \{12\}//' <