diff --git a/apps/framework/lib/docs-reach.ts b/apps/framework/lib/docs-reach.ts new file mode 100644 index 0000000..1755dd3 --- /dev/null +++ b/apps/framework/lib/docs-reach.ts @@ -0,0 +1,49 @@ +/** + * What a documentation call actually tells us, and why that is four things + * rather than two. + * + * Reporting bucketed on `pages` alone, which folded two different kinds of + * ignorance into the same number as a real result: + * + * - **Codex's `web_search` is hosted.** The hits go to the model on the + * provider's side and the CLI receives nothing, so the call arrives with no + * pages, `hasContent` unknown and `resultChars` omitted. That is our blind + * spot, not a failed search. On the 14 September snapshot all 254 of them were + * unobserved: 216 counted as failures and 38 counted as successful page reads, + * the latter because a url-shaped query is recorded as the page it probably + * opened. + * - **Claude's `WebSearch` returns hits, not pages.** Titles and urls, no page + * text — `docs-results.ts` sets `hasContent: false` for exactly this. Counting + * those as "reached the docs" credits an agent for seeing a list of links. Ten + * of the 46 calls that arm was credited with on the same snapshot were hit + * lists. + * + * Neither is a documentation or skills gap, and a delta that moves when either + * moves is not a skills effect. See #83, and #61 for the number that rested on + * the two-bucket version. + * + * The fields do different jobs and both are needed. `resultChars` separates "we + * saw a result and it was empty" (`0`) from "there was no result to see" + * (omitted). `hasContent` separates page text from a list of links. + */ +export type DocsReach = 'read' | 'hits' | 'none' | 'unobserved'; + +export interface DocsReachCall { + pages?: unknown[]; + hasContent?: boolean; + resultChars?: number; +} + +export function docsReach(call: DocsReachCall): DocsReach { + if (call.resultChars === undefined) return 'unobserved'; + if (call.hasContent === false) return 'hits'; + return (call.pages ?? []).length > 0 ? 'read' : 'none'; +} + +/** One line of prose per bucket, for whatever is printing them. */ +export const DOCS_REACH_MEANING: Record = { + read: 'page text reached the agent', + hits: 'a list of links, no page text', + none: 'a result we saw, carrying nothing', + unobserved: 'a hosted search whose hits never reach us', +}; diff --git a/apps/framework/scripts/report-results.ts b/apps/framework/scripts/report-results.ts index e626b77..e5e41fc 100644 --- a/apps/framework/scripts/report-results.ts +++ b/apps/framework/scripts/report-results.ts @@ -1,6 +1,11 @@ import { existsSync, readFileSync } from 'node:fs'; import { join } from 'node:path'; import { discoverEvals, EVALS_ROOT } from '../lib/discovery.js'; +import { + DOCS_REACH_MEANING, + docsReach, + type DocsReach, +} from '../lib/docs-reach.js'; /** * Read a snapshot and say what kind of failures it contains. @@ -27,6 +32,7 @@ import { discoverEvals, EVALS_ROOT } from '../lib/discovery.js'; * ```bash * pnpm --filter @hookdeck-evals/framework report-results * pnpm --filter @hookdeck-evals/framework report-results results/runs/2026-08-21.json + * pnpm --filter @hookdeck-evals/framework report-results --queries * ``` */ @@ -39,7 +45,16 @@ interface Check { interface DocsCall { source?: string; + // Whichever field was the call's "ask": a search term, a url, a shell command. + // For a hosted search this is the only part we ever see, which is why + // `--queries` exists. + query?: string; pages?: unknown[]; + // False when the result was a list of links rather than page text. + hasContent?: boolean; + // Omitted when the trace exposed no result at all, which is what separates a + // hosted search we cannot see from one that genuinely returned nothing. + resultChars?: number; } interface Row { @@ -67,7 +82,11 @@ function load(path: string): Snapshot { } function main() { - const [path = 'results/latest.json'] = process.argv.slice(2); + const args = process.argv.slice(2); + const showQueries = args.includes('--queries'); + const [path = 'results/latest.json'] = args.filter( + (a) => !a.startsWith('--') + ); const snapshot = load(path); // Benchmark only. Regression scenarios are meant to pass everywhere, so @@ -159,6 +178,7 @@ function main() { } reportDocsReach(snapshot.results); + if (showQueries) reportDocsQueries(snapshot.results); } /** @@ -169,46 +189,41 @@ function main() { * not a confound, and not something to design out. That distinction is the whole * reason this prints two numbers instead of one. * - * What is worth watching is the other half, and printing it per experiment moved - * the answer. Aggregated by arm it reads as a skills effect — 296 calls in - * `-no-skills` against 83 in `+skills`, 58% empty against 37%. Split by - * experiment on the 25 August snapshot, most of that is the agent: - * - * claude-code-sonnet-5 21 calls 0 empty (0%) - * claude-code-sonnet-5-no-skills 64 calls 0 empty (0%) - * codex-gpt-5.4-mini 43 calls 26 empty (60%) - * codex-gpt-5.4-mini-no-skills 196 calls 144 empty (73%) - * codex-gpt-5.6 19 calls 5 empty (26%) - * codex-gpt-5.6-no-skills 36 calls 30 empty (83%) + * What is worth watching is the other half, and it has to be split four ways. + * Only `read` means page text reached the agent. `hits` is a list of links, + * `none` is a result we saw that carried nothing, and `unobserved` is a hosted + * search whose results never reach us at all — see `lib/docs-reach.ts` for which + * agent produces which, and why the two-bucket version was wrong in both + * directions. * - * **Both Claude arms are at zero.** Claude Code reaches docs with `web_fetch`, - * which returns a page; Codex leans on `web_search`, which frequently returns - * none. There is still a real skills effect inside each Codex pair — 60 against - * 73, 26 against 83 — but it is the smaller term, and the aggregate hid that by - * pooling two agents with different habits. + * So read this per row, never pooled. `read` is signal about the docs and the + * skills. The other three are mostly facts about how a given agent goes looking, + * and a delta that moves when they move is not a skills effect. * - * So read this per row, never pooled: `reached` is signal about the docs and the - * skills, `empty` is mostly a fact about the agent's search path, and a delta - * that moves when `empty` moves is neither. See #61, and #2 for the scenario - * where this presented as a clean skills win. + * `--queries` prints what was actually searched for. For a hosted search the + * query is the only observable there is, so it is the only way to ask what an + * agent was trying to find out — which is the question a skills delta usually + * turns out to be about. See #83, and #61 for the number that rested on the old + * bucket. */ +const DOCS_REACH_ORDER: DocsReach[] = ['read', 'hits', 'none', 'unobserved']; + function reportDocsReach(rows: Row[]) { const arms = new Map< string, - { reached: number; empty: number; bySource: Map } + { counts: Map; bySource: Map } >(); for (const row of rows) { const calls = row.docs?.calls ?? []; if (calls.length === 0) continue; const arm = arms.get(row.experiment) ?? { - reached: 0, - empty: 0, + counts: new Map(), bySource: new Map(), }; for (const call of calls) { - if ((call.pages ?? []).length > 0) arm.reached += 1; - else arm.empty += 1; + const reach = docsReach(call); + arm.counts.set(reach, (arm.counts.get(reach) ?? 0) + 1); const source = call.source ?? 'unknown'; arm.bySource.set(source, (arm.bySource.get(source) ?? 0) + 1); } @@ -219,24 +234,74 @@ function reportDocsReach(rows: Row[]) { console.log('\n How each arm reached the docs:'); for (const [experiment, arm] of [...arms].sort()) { - const total = arm.reached + arm.empty; - const pct = Math.round((100 * arm.empty) / total); + const total = [...arm.counts.values()].reduce((a, b) => a + b, 0); + const read = arm.counts.get('read') ?? 0; + const pct = Math.round((100 * read) / total); + const buckets = DOCS_REACH_ORDER.map( + (reach) => `${reach} ${String(arm.counts.get(reach) ?? 0).padStart(3)}` + ).join(' '); const mix = [...arm.bySource] .sort((a, b) => b[1] - a[1]) .map(([source, n]) => `${source} ${n}`) .join(', '); console.log( ` ${experiment.padEnd(32)} ${String(total).padStart(4)} calls ` + - `${String(arm.reached).padStart(4)} reached a page ` + - `${String(arm.empty).padStart(4)} empty (${pct}%) [${mix}]` + `${buckets} ${String(pct).padStart(3)}% read [${mix}]` ); } + for (const reach of DOCS_REACH_ORDER) { + console.log(` ${reach.padEnd(11)} ${DOCS_REACH_MEANING[reach]}`); + } console.log( - ' A high empty rate is an external search index returning nothing, not a\n' + - ' documentation or skills gap. Do not read a delta that moves with it as one.' + ' Only `read` is signal about the docs or the skills. Do not read a delta\n' + + ' that moves with the other three as one. `--queries` shows what was asked.' ); } +/** + * What each arm went looking for, and what came back. + * + * The query is the one part of a hosted search we always see, so for Codex it is + * the only evidence of what the agent was trying to find out. That turns out to + * be the interesting half: `verification-002`'s baseline searched twice for an + * ElevenLabs source type, got nothing it could use, and built a generic `WEBHOOK` + * source with hand-rolled HMAC — while the arm with the skill named the preset + * and passed. A pass rate cannot show that. Two queries and their outcome can. + * + * Off by default because it is long. Grouped by scenario rather than by arm, so + * the two arms of a pair sit next to each other and the diff is readable. + */ +function reportDocsQueries(rows: Row[]) { + const byEval = new Map(); + for (const row of rows) { + if ((row.docs?.calls ?? []).length === 0) continue; + byEval.set(row.eval, [...(byEval.get(row.eval) ?? []), row]); + } + if (byEval.size === 0) { + console.log('\n No docs calls recorded in this snapshot.'); + return; + } + + console.log('\n What each arm asked the docs:'); + for (const [evalId, evalRows] of [...byEval].sort()) { + console.log(`\n ${evalId}`); + for (const row of [...evalRows].sort((a, b) => + a.experiment.localeCompare(b.experiment) + )) { + const calls = row.docs?.calls ?? []; + console.log( + ` ${row.experiment} ${row.passed ? 'passed' : 'FAILED'} ${calls.length} calls` + ); + for (const call of calls) { + const query = (call.query ?? '').replace(/\s+/g, ' ').trim(); + console.log( + ` ${docsReach(call).padEnd(11)} ${call.source ?? 'unknown'} ${query}` + ); + } + } + } +} + /** What each scenario's frontmatter says right now, by eval id. */ function currentClassifications(): Map { const map = new Map(); diff --git a/apps/framework/test/docs-reach.test.ts b/apps/framework/test/docs-reach.test.ts new file mode 100644 index 0000000..9be08ee --- /dev/null +++ b/apps/framework/test/docs-reach.test.ts @@ -0,0 +1,63 @@ +import { describe, expect, it } from 'vitest'; +import { docsReach } from '../lib/docs-reach.js'; + +/** + * The defect these guard against was published, and it ran in both directions: + * two buckets credited a list of links as reading the docs, and counted our own + * blind spot as a search that failed. #61 built a headline on the second. + */ +describe('docsReach', () => { + it('counts page text as read', () => { + expect( + docsReach({ + pages: [{ url: 'https://hookdeck.com/docs' }], + hasContent: true, + resultChars: 4208, + }) + ).toBe('read'); + }); + + it('counts a shell fetch with no proof either way as read, because a url was fetched', () => { + expect( + docsReach({ + pages: [{ url: 'https://hookdeck.com/docs' }], + resultChars: 4208, + }) + ).toBe('read'); + }); + + it("counts a search's hit list as hits, not as reading the docs", () => { + // Claude Code's WebSearch: eight titles and urls, no page text. Ten of these + // were credited to `claude-code-sonnet-5-no-skills` as pages on the + // 14 September snapshot. + expect( + docsReach({ + pages: Array.from({ length: 8 }, (_, i) => ({ url: `https://x/${i}` })), + hasContent: false, + resultChars: 2100, + }) + ).toBe('hits'); + }); + + it('counts an observed result with no pages as none', () => { + expect(docsReach({ pages: [], resultChars: 0 })).toBe('none'); + }); + + it('counts a call with no result recorded as unobserved, not as none', () => { + expect(docsReach({ pages: [] })).toBe('unobserved'); + }); + + it('counts a url-shaped query with no result as unobserved, not as a page read', () => { + // The 38 calls on the 14 September snapshot credited as docs reads: the + // "page" is the query the agent typed, not anything we saw come back. + expect( + docsReach({ + pages: [{ url: 'https://hookdeck.com/docs/outpost/llms.txt' }], + }) + ).toBe('unobserved'); + }); + + it('treats a missing pages array as no pages rather than throwing', () => { + expect(docsReach({ resultChars: 12 })).toBe('none'); + }); +});