From 29e56df05e71b30301f3531846981297704d8969 Mon Sep 17 00:00:00 2001 From: Drew Stone Date: Thu, 1 Oct 2026 06:27:25 -0600 Subject: [PATCH 1/2] feat(router): preserve Auto policy authority and response evidence --- .release-notes/router-auto-evidence.md | 7 ++ api-surface.json | 5 +- docs/api/primitive-catalog.md | 3 +- docs/api/runtime.md | 70 +++++++++++++- docs/canonical-api.md | 23 +++++ src/runtime/index.ts | 2 +- src/runtime/router-auto-evidence.ts | 103 +++++++++++++++++++++ src/runtime/router-client.ts | 80 +++++++++++++++- src/runtime/supervise/model-policy.ts | 9 ++ src/runtime/supervise/router-transcript.ts | 11 ++- src/runtime/supervise/runtime.ts | 30 +++++- 11 files changed, 334 insertions(+), 9 deletions(-) create mode 100644 .release-notes/router-auto-evidence.md create mode 100644 src/runtime/router-auto-evidence.ts diff --git a/.release-notes/router-auto-evidence.md b/.release-notes/router-auto-evidence.md new file mode 100644 index 000000000..9eba39858 --- /dev/null +++ b/.release-notes/router-auto-evidence.md @@ -0,0 +1,7 @@ +type: minor +--- +Router tool executors support versioned inline Auto policies without weakening concrete model checks. +Runtime checks every declared strategy and assessment against the allowed models before dispatch. +Buffered HTTP response receipts preserve complete routing metadata and failed responses in native transcripts. +An awaited provider response observer supports durable capture during execution. +Aggregate Auto usage remains unknown because its response reports final-generation tokens only. diff --git a/api-surface.json b/api-surface.json index 2e7d0cbdb..85948f884 100644 --- a/api-surface.json +++ b/api-surface.json @@ -1234,8 +1234,9 @@ "RootProviderModelEvidence": "type 062cb80ee9d2", "RootSignal": "type b819c5206299", "RootStreamReceipt": "type e19150df084b", + "RouterResponseReceipt": "type 4abf1e70a35a", "RouterSeam": "type 3f4c9e0a841e", - "RouterToolsSeam": "type 44e34624f17c", + "RouterToolsSeam": "type c5dc09516fca", "RouterTransportConfig": "type d14ea3416352", "RunAgentRoundsOptions": "type 4caa711f9f2e", "RunAgenticOptions": "type 346e500a1693", @@ -2111,7 +2112,7 @@ "./testing": { "AgentProfileImprovementFixture": "value 8a79d0f5c646", "AgentProfileImprovementProposalFixture": "value 95f42c91bc08", - "DriverAgentOptions": "type 52612582f058", + "DriverAgentOptions": "type 605daae26739", "RunGraphTestOptions": "type c5b5730ba4f0", "SuperviseTestOptions": "type 1fd4a3f892e4", "SupervisorAgentTestDeps": "type f39aa3b16149", diff --git a/docs/api/primitive-catalog.md b/docs/api/primitive-catalog.md index f2cdf7db9..4798c2903 100644 --- a/docs/api/primitive-catalog.md +++ b/docs/api/primitive-catalog.md @@ -449,7 +449,7 @@ Import from `@tangle-network/agent-runtime/intelligence` — 167 exports. ### Execution kernel — recursive atom, supervision, executors, round-synchronous loop -Import from `@tangle-network/agent-runtime/kernel` — 1043 exports. +Import from `@tangle-network/agent-runtime/kernel` — 1044 exports. | Symbol | Kind | Summary | |---|---|---| @@ -1069,6 +1069,7 @@ Import from `@tangle-network/agent-runtime/kernel` — 1043 exports. | `RetainedRunStartMaterial` | interface | Environment, turn, and optional identity needed to replay one retained start. | | `RootHandle` | interface | Live root handle — a chat/pi-viz client uses it to inspect and control one root run. | | `RootStreamReceipt` | interface | The root manager's retained provider stream: `/root-stream.jsonl`, one line per | +| `RouterResponseReceipt` | interface | Exact buffered HTTP evidence. Authentication and cookie headers are excluded. | | `RouterSeam` | interface | Router/inline transport seam. The profile owns model, prompt, and generation behavior. | | `RouterToolsSeam` | interface | Router seam WITH tool use — the tool-using router backend. Same direct | | `RouterTransportConfig` | interface | Connection details for Runtime's Router-backed executors. | diff --git a/docs/api/runtime.md b/docs/api/runtime.md index 6dd5da225..a3d171f8c 100644 --- a/docs/api/runtime.md +++ b/docs/api/runtime.md @@ -1417,7 +1417,7 @@ One flattened node with the journal tree that owns its records. ###### Inherited from -[`NodeSnapshot`](#nodesnapshot).[`status`](#status-20) +[`NodeSnapshot`](#nodesnapshot).[`status`](#status-21) ##### runtime @@ -9591,6 +9591,58 @@ Injectable OpenAI-compatible transport. Optional usage.resources carries measure *** +### RouterResponseReceipt + +Exact buffered HTTP evidence. Authentication and cookie headers are excluded. + +#### Properties + +##### endpoint + +> `readonly` **endpoint**: `string` + +##### callId + +> `readonly` **callId**: `string` + +##### attempt + +> `readonly` **attempt**: `number` + +##### startedAt + +> `readonly` **startedAt**: `string` + +##### endedAt + +> `readonly` **endedAt**: `string` + +##### status + +> `readonly` **status**: `number` \| `null` + +##### error? + +> `readonly` `optional` **error?**: `object` + +###### name + +> `readonly` **name**: `string` + +###### message + +> `readonly` **message**: `string` + +##### headers + +> `readonly` **headers**: `Readonly`\<`Record`\<`string`, `string`\>\> + +##### body + +> `readonly` **body**: `string` \| `null` + +*** + ### ToolSpec #### Properties @@ -20842,6 +20894,22 @@ surfaces (e.g. a gym keyed by task) can dispatch correctly. Exact conversation to continue. Runtime validates its system message against the profile. +##### onProviderResponse? + +> `optional` **onProviderResponse?**: (`receipt`) => `void` \| `Promise`\<`void`\> + +Persist each buffered HTTP attempt, including failed responses, before the next turn. + +###### Parameters + +###### receipt + +[`RouterResponseReceipt`](#routerresponsereceipt) + +###### Returns + +`void` \| `Promise`\<`void`\> + ##### onMessages? > `optional` **onMessages?**: (`messages`) => `void` \| `Promise`\<`void`\> diff --git a/docs/canonical-api.md b/docs/canonical-api.md index d25dc3909..f0e35f34d 100644 --- a/docs/canonical-api.md +++ b/docs/canonical-api.md @@ -14,6 +14,29 @@ Run pnpm docs:freshness after editing this file. --> > > **Read this before writing any orchestration, optimization, or measurement code in this repo.** If you are about to write a persona⟷agent conversation runner, a "skill optimizer", a "profile-seam", a depth-vs-breadth A/B harness, a bootstrap loop, or a `new Sandbox(...)` + stream + read dance: **stop**, it already exists, and a parallel copy will silently break one of the guarantees the existing primitives enforce: equal compute per compared arm ("equal-k"), the attempt-picker never being the grader ("selector≠judge"), complete usage capture, or eval running the same code path as production. +## Router Auto execution and evidence + +Use `createExecutor({ backend: 'router-tools', ... })` with an exact profile. +Declare `model.default: 'tangle/auto'` and an inline policy in `model.metadata.extraBody.auto`. +Use the Router Auto base URL and buffered transport. +Runtime checks strategy and assessment models against `allowedModels` before dispatch. +It checks each reported child model against the declared policy. +Preset references require resolution into an inline policy before execution. + +`onProviderResponse` receives each buffered HTTP attempt before response decoding. +The receipt preserves its full body, endpoint, call identity, times, status, and response headers. +Authentication and cookie headers are excluded. +Network failures have no HTTP status and retain their error. +Native transcripts preserve these receipts outside the model conversation. +Final outputs also retain received HTTP receipts. +A failed attempt still retains its transcript. +Observer failures stop the executor and do not authorize another paid request. + +Auto response usage covers the final generation only. +Runtime marks aggregate token and dollar usage unknown. +Use the returned physical generation identifiers to join authoritative billing records. +A receipt header cost does not prove complete campaign billing. + ## 1. Mental model: the spine > **Legend**: five terms the rest of this doc leans on, in plain terms: diff --git a/src/runtime/index.ts b/src/runtime/index.ts index 6011bfce0..20fed0ca3 100644 --- a/src/runtime/index.ts +++ b/src/runtime/index.ts @@ -481,7 +481,7 @@ export { // Router requests are an internal transport adapter. Public execution always enters through an // exact AgentProfile (`createExecutor` + `streamAgentTurn`); callers may configure only the // endpoint/auth transport used by that path. -export type { RouterTransportConfig } from './router-client' +export type { RouterResponseReceipt, RouterTransportConfig } from './router-client' export { type BenchmarkCell, type BenchmarkConfig, diff --git a/src/runtime/router-auto-evidence.ts b/src/runtime/router-auto-evidence.ts new file mode 100644 index 000000000..f646eeb73 --- /dev/null +++ b/src/runtime/router-auto-evidence.ts @@ -0,0 +1,103 @@ +import { ValidationError } from '../errors' +import { observedModelMatchesDeclared } from './model-identity' + +function object(value: unknown): Record { + if (value === null || typeof value !== 'object' || Array.isArray(value)) { + throw new ValidationError('Router Auto requires an inline policy and routing receipt') + } + return value as Record +} + +/** Router owns policy parsing. Runtime requires visible model authority before dispatch. */ +export function assertRouterAutoDeclaration(value: unknown, stream: boolean | undefined): void { + const policy = object(value) + if (stream === true) throw new ValidationError('Router Auto requires buffered transport') + if ( + policy.version !== 1 || + typeof policy.id !== 'string' || + typeof policy.revision !== 'string' || + !Array.isArray(policy.strategies) || + policy.strategies.length === 0 + ) { + throw new ValidationError( + 'Router Auto requires a versioned inline policy with concrete strategies', + ) + } + for (const raw of policy.strategies) { + const strategy = object(raw) + if ( + typeof strategy.id !== 'string' || + typeof strategy.model !== 'string' || + strategy.model === 'tangle/auto' || + strategy.model.length === 0 + ) { + throw new ValidationError('Router Auto strategy model authority is missing') + } + } + for (const phase of ['selector', 'review']) { + if (policy[phase] !== undefined && typeof object(policy[phase]).model !== 'string') { + throw new ValidationError('Router Auto assessment model authority is missing') + } + } +} + +/** An alias does not waive identity checks. Every reported child must match its declared phase. */ +export function assertRouterAutoResponse( + policyValue: unknown, + response: unknown, + served: string | undefined, +): void { + const policy = object(policyValue) + const receipt = object(object(response).tangle_auto) + const decision = object(receipt.decision) + const strategies = policy.strategies as unknown[] + const selected = strategies.map(object).find((strategy) => strategy.id === decision.strategyId) + if ( + decision.requestedModel !== 'tangle/auto' || + decision.policyId !== policy.id || + decision.policyRevision !== policy.revision || + typeof decision.policyDigest !== 'string' || + !selected || + decision.selectedModel !== selected.model || + served === undefined || + !observedModelMatchesDeclared(served, String(selected.model)) + ) { + throw new ValidationError( + 'Router Auto response does not match AgentProfile policy/model authority', + ) + } + if (!Array.isArray(receipt.calls) || receipt.calls.length === 0) { + throw new ValidationError('Router Auto response is missing physical call receipts') + } + for (const raw of receipt.calls) { + const call = object(raw) + const declared = + call.phase === 'selector' + ? object(policy.selector).model + : call.phase === 'review' + ? object(policy.review).model + : call.phase === 'initial' || call.phase === 'escalation' + ? strategies.map(object).map((strategy) => strategy.model) + : [] + const allowed = Array.isArray(declared) ? declared : [declared] + if ( + typeof call.model !== 'string' || + !allowed.some( + (model) => + typeof model === 'string' && observedModelMatchesDeclared(call.model as string, model), + ) + ) + throw new ValidationError('Router Auto physical call exceeds AgentProfile model authority') + } +} + +/** The concrete authority declared by an inline Auto policy, including paid assessments. */ +export function routerAutoDeclaredModels(value: unknown): readonly string[] { + assertRouterAutoDeclaration(value, undefined) + const policy = object(value) + const models = (policy.strategies as unknown[]).map((raw) => String(object(raw).model)) + for (const phase of ['selector', 'review']) { + if (policy[phase] !== undefined) models.push(String(object(policy[phase]).model)) + } + return [...new Set(models)] +} diff --git a/src/runtime/router-client.ts b/src/runtime/router-client.ts index 15b03bab4..564d5eff7 100644 --- a/src/runtime/router-client.ts +++ b/src/runtime/router-client.ts @@ -293,7 +293,22 @@ export interface RouterToolCall { arguments: string } +/** Exact buffered HTTP evidence. Authentication and cookie headers are excluded. */ +export interface RouterResponseReceipt { + readonly endpoint: string + readonly callId: string + readonly attempt: number + readonly startedAt: string + readonly endedAt: string + readonly status: number | null + readonly error?: { readonly name: string; readonly message: string } + readonly headers: Readonly> + readonly body: string | null +} + export interface RouterChatToolsResult { + /** Complete buffered response, including routing decisions and physical call receipts. */ + response?: unknown content: string | null toolCalls: RouterToolCall[] usage?: { input: number; output: number; reasoning?: number } @@ -349,6 +364,8 @@ export async function routerChatWithTools( function: { name: string; description?: string; parameters: unknown } }>, opts?: { + /** Observe each actual HTTP attempt before decoding or rejecting its response. */ + onResponse?: (receipt: RouterResponseReceipt) => void | Promise temperature?: number signal?: AbortSignal toolChoice?: 'auto' | 'required' | 'none' @@ -390,6 +407,7 @@ export async function routerChatWithTools( }, opts?.signal, retry, + opts?.onResponse, ) transportAttempts = result.attempts return routerResponseJson(result.response, result.attempts) @@ -418,6 +436,7 @@ export async function routerChatWithTools( transportAttempts, ) return { + response: raw, content: msg?.content ?? null, toolCalls, transportAttempts, @@ -873,9 +892,68 @@ async function fetchRouterResponse( init: Omit, callerSignal: AbortSignal | undefined, retry: ReturnType, + onResponse?: (receipt: RouterResponseReceipt) => void | Promise, ): Promise<{ response: Response; attempts: number }> { + let attempt = 0 const result = await retryRouterOperation(retry, callerSignal, async (signal) => { - const candidate = await fetch(url, { ...init, signal }) + attempt += 1 + const startedAt = new Date().toISOString() + let candidate: Response + try { + candidate = await fetch(url, { ...init, signal }) + } catch (error) { + await onResponse?.({ + endpoint: url, + callId: new Headers(init.headers).get('idempotency-key') ?? '', + attempt, + startedAt, + endedAt: new Date().toISOString(), + status: null, + headers: {}, + body: null, + error: { + name: error instanceof Error ? error.name : 'Error', + message: error instanceof Error ? error.message : String(error), + }, + }) + throw error + } + if (onResponse) { + const headers = Object.fromEntries( + [...candidate.headers].filter( + ([name]) => + !['authorization', 'proxy-authorization', 'set-cookie', 'cookie'].includes( + name.toLowerCase(), + ), + ), + ) + let body: string | null = null + let bodyError: unknown + try { + body = await candidate.clone().text() + } catch (error) { + bodyError = error + } + await onResponse({ + endpoint: url, + callId: new Headers(init.headers).get('idempotency-key') ?? '', + attempt, + startedAt, + endedAt: new Date().toISOString(), + status: candidate.status, + headers, + body, + ...(bodyError === undefined + ? {} + : { + error: { + name: bodyError instanceof Error ? bodyError.name : 'Error', + message: bodyError instanceof Error ? bodyError.message : String(bodyError), + }, + }), + }) + if (bodyError !== undefined) throw bodyError + } if (candidate.ok) return candidate const text = await candidate.text().catch(() => '') throw new SDKError(`router ${candidate.status}: ${text.slice(0, 200)}`, { diff --git a/src/runtime/supervise/model-policy.ts b/src/runtime/supervise/model-policy.ts index bdadea2f8..71d2755ed 100644 --- a/src/runtime/supervise/model-policy.ts +++ b/src/runtime/supervise/model-policy.ts @@ -8,6 +8,7 @@ import { HARNESS_NATIVE_MODEL } from '@tangle-network/agent-eval' import type { AgentProfile } from '@tangle-network/agent-interface' import { ConfigError } from '../../errors' import { agentHarness } from '../harness-role' +import { routerAutoDeclaredModels } from '../router-auto-evidence' import { type ResolvedRouterRetryPolicy, resolveRouterRetryPolicy } from '../router-retry-policy' /** @@ -383,6 +384,14 @@ export function assertProfileModelsAllowed( ): void { assertModelAllowed(profile.model?.default, allowed) assertModelAllowed(profile.model?.small, allowed) + if (profile.model?.default === 'tangle/auto') { + const extra = profile.model.metadata?.extraBody + const policy = + extra !== null && typeof extra === 'object' + ? (extra as Record).auto + : undefined + for (const model of routerAutoDeclaredModels(policy)) assertModelAllowed(model, allowed) + } for (const subagent of Object.values(profile.subagents ?? {})) { assertModelAllowed(subagent.model, allowed) } diff --git a/src/runtime/supervise/router-transcript.ts b/src/runtime/supervise/router-transcript.ts index 886962d0d..3acb52c3e 100644 --- a/src/runtime/supervise/router-transcript.ts +++ b/src/runtime/supervise/router-transcript.ts @@ -25,6 +25,7 @@ import { harnessTranscriptFromLines, harnessTranscriptUnavailable, } from '../harness-transcript' +import type { RouterResponseReceipt } from '../router-client' import type { ToolLoopMessageRecord, ToolLoopToolCall } from '../tool-loop' /** The harness name a router-brained agent's transcript carries. */ @@ -48,12 +49,15 @@ export interface RouterTranscript { observe(messages: ReadonlyArray): void /** Keep the model's reply for this turn, and return what this turn added, for the turn event. */ reply(reply: RouterTranscriptReply): ReadonlyArray + /** Keep a physical HTTP response outside the model conversation. */ + response(receipt: RouterResponseReceipt): void /** The whole conversation so far as a harness transcript. */ capture(): HarnessTranscriptCapture } export function createRouterTranscript(): RouterTranscript { const kept: ToolLoopMessageRecord[] = [] + const records: unknown[] = [] const seen = new WeakSet() // Index in `kept` where the current turn's additions start. let turnStart = 0 @@ -72,11 +76,15 @@ export function createRouterTranscript(): RouterTranscript { if (message.role === 'assistant') continue } kept.push(message) + records.push(message) } } return { observe, + response(receipt) { + records.push({ type: 'provider.response', ...receipt }) + }, reply(reply) { const message: ToolLoopMessageRecord = { role: 'assistant', @@ -92,6 +100,7 @@ export function createRouterTranscript(): RouterTranscript { : {}), } kept.push(message) + records.push(message) replyEchoPending = true const added = kept.slice(turnStart).map(forTurnEvent) turnStart = kept.length @@ -103,7 +112,7 @@ export function createRouterTranscript(): RouterTranscript { return harnessTranscriptFromLines( ROUTER_TRANSCRIPT_HARNESS, 'conversation', - kept.map((message) => JSON.stringify(message)), + records.map((record) => JSON.stringify(record)), ) }, } diff --git a/src/runtime/supervise/runtime.ts b/src/runtime/supervise/runtime.ts index b642b7301..02c582364 100644 --- a/src/runtime/supervise/runtime.ts +++ b/src/runtime/supervise/runtime.ts @@ -44,11 +44,13 @@ import { import { agentHarness } from '../harness-role' import { observedModelMatchesDeclared } from '../model-identity' import { selectProviderPlacement } from '../provider-placement' +import { assertRouterAutoDeclaration, assertRouterAutoResponse } from '../router-auto-evidence' import { type PromptCacheUsage, type RouterChatResult, type RouterChatToolsResult, type RouterConfig, + type RouterResponseReceipt, routerChatWithTools, routerChatWithUsage, routerTransportAttemptsFromError, @@ -632,6 +634,8 @@ export interface RouterToolsSeam { executeToolCall: (name: string, args: Record, task: unknown) => Promise /** Exact conversation to continue. Runtime validates its system message against the profile. */ initialMessages?: ReadonlyArray>> + /** Persist each buffered HTTP attempt, including failed responses, before the next turn. */ + onProviderResponse?: (receipt: RouterResponseReceipt) => void | Promise /** Observe the detached final conversation for session persistence. */ onMessages?: (messages: ReadonlyArray>>) => void | Promise /** Online observer of each tool step — the seam a `DetectorMonitor` taps to watch the live pipe @@ -705,6 +709,8 @@ export const routerToolsInlineExecutor: ExecutorFactory = (spec, ctx) = }, { multiTurn: true }, ) + if (model === 'tangle/auto') + assertRouterAutoDeclaration(profileExecution.extraBody?.auto, profileExecution.stream) const enabledToolNames = new Set(seam.tools.map((tool) => tool.function.name)) const maxTurns = profileExecution.maxTurns ?? 0 const requestIdentity = routerRequestIdentity(ctx) @@ -745,6 +751,7 @@ export const routerToolsInlineExecutor: ExecutorFactory = (spec, ctx) = let resources: Spend['resources'] let transportAttempts = 0 let observedModel: string | undefined + const providerResponses: RouterResponseReceipt[] = [] let reasoningTokens = 0 let reasoningKnown = true const promptCache: Record = {} @@ -800,6 +807,11 @@ export const routerToolsInlineExecutor: ExecutorFactory = (spec, ctx) = ? { temperature: profileExecution.temperature } : {}), signal: turnController.signal, + onResponse: async (receipt: RouterResponseReceipt) => { + providerResponses.push(receipt) + transcript.response(receipt) + await seam.onProviderResponse?.(structuredClone(receipt)) + }, ...(profileExecution.toolChoice ? { toolChoice: profileExecution.toolChoice } : {}), @@ -865,7 +877,12 @@ export const routerToolsInlineExecutor: ExecutorFactory = (spec, ctx) = ).resources transportAttempts += res.transportAttempts if (res.model !== undefined) recordRuntimeOwnedProviderModel(executor, res.model) - assertObservedRouterModel(res.model, model, 'routerToolsInlineExecutor') + if (model === 'tangle/auto') { + assertRouterAutoResponse(profileExecution.extraBody?.auto, res.response, res.model) + // Auto reports final-generation usage only. Child usage stays unknown here. + tokensKnown = false + reasoningKnown = false + } else assertObservedRouterModel(res.model, model, 'routerToolsInlineExecutor') if (res.model !== undefined) observedModel = res.model mergePromptCache(promptCache, res.cache) if (res.usage) { @@ -990,6 +1007,7 @@ export const routerToolsInlineExecutor: ExecutorFactory = (spec, ctx) = turns, toolCalls: executedToolCalls, transportAttempts, + providerResponses, ...(estimatedUsd !== undefined ? { estimatedCostUsd: estimatedUsd } : {}), ...(Object.keys(promptCache).length > 0 ? { promptCache } : {}), ...(reasoningKnown && turns > 0 ? { reasoningTokens } : {}), @@ -2063,12 +2081,20 @@ export type ExecutorConfig = export function snapshotExecutorConfig(config: ExecutorConfig): ExecutorConfig { switch (config.backend) { case 'router-tools': { - const { complete, executeToolCall, onMessages, onToolStep, ...decisionData } = config + const { + complete, + executeToolCall, + onMessages, + onToolStep, + onProviderResponse, + ...decisionData + } = config const snapshot = detachedSnapshot(decisionData, 'createExecutor router-tools config') return Object.freeze({ ...snapshot, ...(complete === undefined ? {} : { complete }), executeToolCall, + ...(onProviderResponse === undefined ? {} : { onProviderResponse }), ...(onMessages === undefined ? {} : { onMessages }), ...(onToolStep === undefined ? {} : { onToolStep }), }) From 078de78df5fe1152035f81a79afb10ae966d8db4 Mon Sep 17 00:00:00 2001 From: Drew Stone Date: Thu, 1 Oct 2026 06:43:54 -0600 Subject: [PATCH 2/2] fix(router): stop retries on response capture failure --- docs/canonical-api.md | 3 + src/runtime/router-client.ts | 99 ++++++++++++++++++++++---------- src/runtime/supervise/runtime.ts | 1 + 3 files changed, 72 insertions(+), 31 deletions(-) diff --git a/docs/canonical-api.md b/docs/canonical-api.md index f0e35f34d..84af63c3a 100644 --- a/docs/canonical-api.md +++ b/docs/canonical-api.md @@ -21,6 +21,8 @@ Declare `model.default: 'tangle/auto'` and an inline policy in `model.metadata.e Use the Router Auto base URL and buffered transport. Runtime checks strategy and assessment models against `allowedModels` before dispatch. It checks each reported child model against the declared policy. +It checks policy ID and revision; Router owns normalization and the policy digest. +Consumers must compare that digest with their frozen Router policy before using the result. Preset references require resolution into an inline policy before execution. `onProviderResponse` receives each buffered HTTP attempt before response decoding. @@ -34,6 +36,7 @@ Observer failures stop the executor and do not authorize another paid request. Auto response usage covers the final generation only. Runtime marks aggregate token and dollar usage unknown. +Any observed final-generation dollar subtotal is a lower bound, not a complete Auto total. Use the returned physical generation identifiers to join authoritative billing records. A receipt header cost does not prove complete campaign billing. diff --git a/src/runtime/router-client.ts b/src/runtime/router-client.ts index 564d5eff7..a6e151b1b 100644 --- a/src/runtime/router-client.ts +++ b/src/runtime/router-client.ts @@ -902,20 +902,24 @@ async function fetchRouterResponse( try { candidate = await fetch(url, { ...init, signal }) } catch (error) { - await onResponse?.({ - endpoint: url, - callId: new Headers(init.headers).get('idempotency-key') ?? '', - attempt, - startedAt, - endedAt: new Date().toISOString(), - status: null, - headers: {}, - body: null, - error: { - name: error instanceof Error ? error.name : 'Error', - message: error instanceof Error ? error.message : String(error), + await observeRouterResponse( + onResponse, + { + endpoint: url, + callId: new Headers(init.headers).get('idempotency-key') ?? '', + attempt, + startedAt, + endedAt: new Date().toISOString(), + status: null, + headers: {}, + body: null, + error: { + name: error instanceof Error ? error.name : 'Error', + message: error instanceof Error ? error.message : String(error), + }, }, - }) + signal, + ) throw error } if (onResponse) { @@ -934,24 +938,28 @@ async function fetchRouterResponse( } catch (error) { bodyError = error } - await onResponse({ - endpoint: url, - callId: new Headers(init.headers).get('idempotency-key') ?? '', - attempt, - startedAt, - endedAt: new Date().toISOString(), - status: candidate.status, - headers, - body, - ...(bodyError === undefined - ? {} - : { - error: { - name: bodyError instanceof Error ? bodyError.name : 'Error', - message: bodyError instanceof Error ? bodyError.message : String(bodyError), - }, - }), - }) + await observeRouterResponse( + onResponse, + { + endpoint: url, + callId: new Headers(init.headers).get('idempotency-key') ?? '', + attempt, + startedAt, + endedAt: new Date().toISOString(), + status: candidate.status, + headers, + body, + ...(bodyError === undefined + ? {} + : { + error: { + name: bodyError instanceof Error ? bodyError.name : 'Error', + message: bodyError instanceof Error ? bodyError.message : String(bodyError), + }, + }), + }, + signal, + ) if (bodyError !== undefined) throw bodyError } if (candidate.ok) return candidate @@ -966,6 +974,33 @@ async function fetchRouterResponse( return { response: result.value, attempts: result.attempts } } +/** Evidence failure cannot authorize a second inference, even after the request deadline. */ +async function observeRouterResponse( + observer: ((receipt: RouterResponseReceipt) => void | Promise) | undefined, + receipt: RouterResponseReceipt, + signal: AbortSignal, +): Promise { + if (!observer) return + let onAbort: (() => void) | undefined + try { + const aborted = new Promise((_resolve, reject) => { + onAbort = () => reject(signal.reason ?? new Error('Response observation aborted')) + if (signal.aborted) onAbort() + else signal.addEventListener('abort', onAbort, { once: true }) + }) + await Promise.race([Promise.resolve().then(() => observer(receipt)), aborted]) + } catch (error) { + throw new SDKError('Router response observation failed', { + code: 'UNKNOWN', + retryable: false, + cause: error instanceof Error ? error : new Error(String(error)), + context: { phase: 'response_observation' }, + }) + } finally { + if (onAbort) signal.removeEventListener('abort', onAbort) + } +} + async function retryRouterOperation( retry: ReturnType, callerSignal: AbortSignal | undefined, @@ -980,6 +1015,8 @@ async function retryRouterOperation( try { return await operation(attemptSignal.signal) } catch (error) { + if (error instanceof SDKError && error.context?.phase === 'response_observation') + throw error if (callerSignal?.aborted) throw callerSignal.reason ?? error if (attemptSignal.signal.aborted) { throw new SDKError(`router request timeout after ${retry.requestTimeoutMs}ms`, { diff --git a/src/runtime/supervise/runtime.ts b/src/runtime/supervise/runtime.ts index 02c582364..494d337a6 100644 --- a/src/runtime/supervise/runtime.ts +++ b/src/runtime/supervise/runtime.ts @@ -882,6 +882,7 @@ export const routerToolsInlineExecutor: ExecutorFactory = (spec, ctx) = // Auto reports final-generation usage only. Child usage stays unknown here. tokensKnown = false reasoningKnown = false + usdKnown = false } else assertObservedRouterModel(res.model, model, 'routerToolsInlineExecutor') if (res.model !== undefined) observedModel = res.model mergePromptCache(promptCache, res.cache)