From 06db78077daef3e5669ebfbf9bdd5a4a7574f4b2 Mon Sep 17 00:00:00 2001 From: Michael Ryaboy Date: Tue, 22 Sep 2026 01:46:11 -0700 Subject: [PATCH] Fix agent-reported batch recovery and spending error reporting --- .claude-plugin/marketplace.json | 4 +- e2e/spending.mjs | 11 ++++- plugins/classifier/.claude-plugin/plugin.json | 2 +- plugins/classifier/.codex-plugin/plugin.json | 2 +- .../classifier/skills/bulk-classify/SKILL.md | 14 +++--- src/SKILL.md | 14 +++--- src/alerts.ts | 2 +- src/docs.ts | 11 +++-- src/home.ts | 2 +- src/index.ts | 26 +++++++--- src/jev.ts | 3 ++ src/openapi.ts | 6 +-- src/pages.ts | 30 +++++++----- tests/spending.e2e.test.ts | 47 +++++++++++++++++++ 14 files changed, 126 insertions(+), 48 deletions(-) diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index be6f9a8..0d15288 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -8,14 +8,14 @@ }, "metadata": { "description": "Plugins from classifier.dev: zero-shot text classification with a calibrated confidence per answer, no API key.", - "version": "1.0.0" + "version": "1.0.1" }, "plugins": [ { "name": "classifier", "source": "./plugins/classifier", "description": "Sort up to 1,000 texts into your own labels in one call, with a calibrated confidence per answer. No API key.", - "version": "1.0.0", + "version": "1.0.1", "author": { "name": "Michael Ryaboy", "email": "contact@classifier.dev", "url": "https://classifier.dev" }, "homepage": "https://classifier.dev/mcp-setup", "repository": "https://github.com/mrmps/classifier-dev", diff --git a/e2e/spending.mjs b/e2e/spending.mjs index aeb6ccb..813b50c 100644 --- a/e2e/spending.mjs +++ b/e2e/spending.mjs @@ -5,7 +5,7 @@ import assert from 'node:assert/strict'; const report = { runtime: 'workerd + SQLite Durable Objects', upstream: 'deterministic HTTP provider fixtures; no live inference', results: [] }; await mkdir('captures', { recursive: true }); -let calls = 0, lookups = 0, release; +let calls = 0, lookups = 0, release, jevUnavailable = false; let barrier = Promise.resolve(); const modules = (await readdir('dist/server', { recursive: true })).filter(p => p.endsWith('.js')).sort((a, b) => a === 'index.js' ? -1 : b === 'index.js' ? 1 : a.localeCompare(b)).map(p => ({ type: 'ESModule', path: `dist/server/${p}` })); const mf = new Miniflare(convertV4MiniflareOptions({ workers: [{ name: "spending", modules, modulesRoot: 'dist/server', compatibilityDate: '2026-08-01', compatibilityFlags: ['nodejs_compat'], @@ -15,7 +15,8 @@ const mf = new Miniflare(convertV4MiniflareOptions({ workers: [{ name: "spending if (request.url.startsWith('https://api.spur.us/')) { lookups++; return WorkerResponse.json({}); } calls++; await barrier; const body = await request.json(); - if (request.url.includes('openrouter.ai')) return WorkerResponse.json({ model: body.model, usage: { cost: 0.0007125, prompt_tokens: 200, completion_tokens: 150, prompt_tokens_details: { cached_tokens: 0 } }, choices: [{ message: { content: 'A' } }] }); + if (jevUnavailable && request.url.includes('typesafe.ai')) return WorkerResponse.json({ detail: { error_type: 'insufficient_credits' } }, { status: 402 }); + if (request.url.includes('openrouter.ai')) return WorkerResponse.json({ model: body.model, usage: { cost: jevUnavailable ? 0.000001812 : 0.0007125, prompt_tokens: jevUnavailable ? 100 : 200, completion_tokens: jevUnavailable ? 1 : 150, prompt_tokens_details: { cached_tokens: 0 } }, choices: [{ message: { content: 'A' } }] }); const answers = Object.fromEntries(Object.entries(body.questions).map(([id, q]) => { const keys = Object.keys(q.criteria); return [id, { choice: keys[0], confidence: 0.51, probabilities: Object.fromEntries(keys.map((k, i) => [k, i ? 0.49 : 0.51])) }]; })); return WorkerResponse.json({ model: 'jev-1.13.0', usage: { input_tokens: 100, output_tokens: 0 }, answers }); }, @@ -45,6 +46,12 @@ try { const duplicate = await request('203.0.113.4', undefined, undefined, { 'idempotency-key': 'one' }); assert.equal(duplicate.status, 409); report.results.push({ name: 'durable idempotency', first: first.status, duplicate: duplicate.status }); + jevUnavailable = true; + const recovered = await request('203.0.113.5', { inputs: Array(119).fill('Invoice'), labels: ['billing', 'support'] }); + assert.equal(recovered.status, 200); + const recoveredBody = await recovered.json(); + assert.equal(recoveredBody.results.length, 119); + report.results.push({ name: '119-item batch during primary outage', status: recovered.status, results: recoveredBody.results.length }); await writeFile('captures/spending-e2e.json', JSON.stringify(report, null, 2)); console.log(JSON.stringify(report, null, 2)); } finally { await mf.dispose(); } diff --git a/plugins/classifier/.claude-plugin/plugin.json b/plugins/classifier/.claude-plugin/plugin.json index 8ceaca0..a18a27f 100644 --- a/plugins/classifier/.claude-plugin/plugin.json +++ b/plugins/classifier/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json", "name": "classifier", - "version": "1.0.0", + "version": "1.0.1", "description": "Sort up to 1,000 texts into your own labels in one call, with a calibrated confidence per answer. No API key.", "author": { "name": "Michael Ryaboy", diff --git a/plugins/classifier/.codex-plugin/plugin.json b/plugins/classifier/.codex-plugin/plugin.json index d31c0c2..3032dbb 100644 --- a/plugins/classifier/.codex-plugin/plugin.json +++ b/plugins/classifier/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "classifier", - "version": "1.0.0", + "version": "1.0.1", "description": "Sort up to 1,000 texts into your own labels in one call, with a calibrated confidence per answer. No API key.", "author": { "name": "Michael Ryaboy", diff --git a/plugins/classifier/skills/bulk-classify/SKILL.md b/plugins/classifier/skills/bulk-classify/SKILL.md index e73baaa..171a870 100644 --- a/plugins/classifier/skills/bulk-classify/SKILL.md +++ b/plugins/classifier/skills/bulk-classify/SKILL.md @@ -73,7 +73,7 @@ npm i -g classifier-dev classify bug,feature,praise < feedback.txt # labelconfidencetext, input order classify relevant,"not relevant" --review 0.7 < snippets.txt # only the unsure ones classify db,web,ml --count < titles.txt # a histogram instead of rows -classify a,b --json < items.txt | jq -c 'select(.confidence < 0.8)' +classify a,b --json < items.txt | jq -c 'select(.confidence == null or .confidence < 0.8)' ``` It batches a thousand inputs per request, four requests at a time, and streams @@ -185,8 +185,6 @@ def keep_relevant(question, snippets): data=body, headers={ "content-type": "application/json", - # Send a real User-Agent. Python's stdlib default is a known-bot - # signature and gets a 403 at the edge before it reaches the API. "user-agent": "my-agent/1.0", }, ) @@ -203,10 +201,12 @@ Then read only what comes back. The snippets you dropped never enter context. matters more than precision here. The confidence gate above does that directly; "When in doubt, keep it" in the instructions also measurably helps. -**Always set a `User-Agent`.** Most clients (curl, node, bun, requests, Go, axios) -send a usable one already, but Python's `urllib` default is blocked at the edge -and returns `403` before your request is ever classified. If you get a 403, -this is why. Rate limiting returns `429`. +Python's standard `urllib`, curl and Node fetch work without a custom +`User-Agent`. A descriptive agent name is optional. For a JSON error, read +`code`, `action` and `retryable`: a 403 can mean the free service detected an +anonymous proxy network, which requires a funded workspace key. A 429 carries +`Retry-After`. An HTML error is an edge/network failure; report its status and +request ID rather than assuming classification ran. ## Report a problem with classifier.dev diff --git a/src/SKILL.md b/src/SKILL.md index e73baaa..171a870 100644 --- a/src/SKILL.md +++ b/src/SKILL.md @@ -73,7 +73,7 @@ npm i -g classifier-dev classify bug,feature,praise < feedback.txt # labelconfidencetext, input order classify relevant,"not relevant" --review 0.7 < snippets.txt # only the unsure ones classify db,web,ml --count < titles.txt # a histogram instead of rows -classify a,b --json < items.txt | jq -c 'select(.confidence < 0.8)' +classify a,b --json < items.txt | jq -c 'select(.confidence == null or .confidence < 0.8)' ``` It batches a thousand inputs per request, four requests at a time, and streams @@ -185,8 +185,6 @@ def keep_relevant(question, snippets): data=body, headers={ "content-type": "application/json", - # Send a real User-Agent. Python's stdlib default is a known-bot - # signature and gets a 403 at the edge before it reaches the API. "user-agent": "my-agent/1.0", }, ) @@ -203,10 +201,12 @@ Then read only what comes back. The snippets you dropped never enter context. matters more than precision here. The confidence gate above does that directly; "When in doubt, keep it" in the instructions also measurably helps. -**Always set a `User-Agent`.** Most clients (curl, node, bun, requests, Go, axios) -send a usable one already, but Python's `urllib` default is blocked at the edge -and returns `403` before your request is ever classified. If you get a 403, -this is why. Rate limiting returns `429`. +Python's standard `urllib`, curl and Node fetch work without a custom +`User-Agent`. A descriptive agent name is optional. For a JSON error, read +`code`, `action` and `retryable`: a 403 can mean the free service detected an +anonymous proxy network, which requires a funded workspace key. A 429 carries +`Retry-After`. An HTML error is an edge/network failure; report its status and +request ID rather than assuming classification ran. ## Report a problem with classifier.dev diff --git a/src/alerts.ts b/src/alerts.ts index e71fb36..b727ae1 100644 --- a/src/alerts.ts +++ b/src/alerts.ts @@ -257,7 +257,7 @@ export async function evaluate(env: Env): Promise<{ alerts: Alert[]; checked: bo alerts.push({ id: "dimensions_fallback", severity: "warning", title: "Multidimensional classification is using the LLM fallback", - detail: `${dimensionFallback} fields used the fallback in the last ${WINDOW_MIN}m. Check Jev availability; requests above 20 decisions cannot use this fallback.`, + detail: `${dimensionFallback} fields used the fallback in the last ${WINDOW_MIN}m. Check Jev availability. Fallback uses bounded concurrency and the request spending allowance.`, }); } diff --git a/src/docs.ts b/src/docs.ts index cd6e137..faebee5 100644 --- a/src/docs.ts +++ b/src/docs.ts @@ -314,8 +314,8 @@ MULTIPLE DIMENSIONS Confidence and scores can also be null when the provider returns no score. usage reports items, dimensions, classifications (decisions), escalated, - fallback, and ms. If Jev is unavailable, the LLM fallback accepts at most - 20 decisions; larger requests return 502 batch_unavailable. A failed field + fallback, and ms. If Jev is unavailable, fallback processes the batch with + bounded concurrency inside the same request spending allowance. A failed field fails the whole request rather than returning an incomplete matrix. @@ -428,7 +428,7 @@ TIERS The models are not fixed. They are benchmarked as candidates appear and swapped when a measurement, not a launch post, says to. If the decision - model is unavailable, requests of up to twenty inputs fall back to a chain of + model is unavailable, requests fall back within their spending allowance to a chain of language models on different providers; JSON responses always report which model actually answered. @@ -478,8 +478,9 @@ ERRORS the limit (https://classifier.dev/pricing) 502 typesafe or typesafe_ when the decision model failed; openrouter_, chain_exhausted or timeout when the fallback - chain did; batch_unavailable for more than twenty inputs while the - decision model is down; upstream_other. Retry with backoff. + chain did; upstream_other. Retry with backoff. + 402 request_spending_limit: send fewer or shorter inputs, or use a funded + workspace key. Do not repeatedly retry an unchanged over-budget request. The full list, in the shape a client can validate against, is components.schemas.Error in https://classifier.dev/openapi.json diff --git a/src/home.ts b/src/home.ts index a314130..e030bed 100644 --- a/src/home.ts +++ b/src/home.ts @@ -619,7 +619,7 @@ const JSON_LD = () => { { "@type": "Question", name: "When should an agent call classifier.dev instead of classifying text itself?", - acceptedAnswer: { "@type": "Answer", text: "When reading the input is the expensive part: filtering search results before opening them, bucketing logs or tickets, routing a pipeline branch deterministically. One request classifies up to 1,000 texts in about a second. Under about five items you can already see, just decide yourself." }, + acceptedAnswer: { "@type": "Answer", text: "When reading the input is the expensive part: filtering search results before opening them, bucketing logs or tickets, routing a pipeline branch by label. One request classifies up to 1,000 texts in about a second. Under about five items you can already see, just decide yourself." }, }, { "@type": "Question", diff --git a/src/index.ts b/src/index.ts index 187b4ea..6727c68 100644 --- a/src/index.ts +++ b/src/index.ts @@ -651,6 +651,7 @@ async function callModel( signal: AbortSignal.timeout(cfg.reasoning ? 60_000 : 15_000), }); } catch (e) { + if (e instanceof SpendingError) throw e; const timeout = e instanceof Error && (e.name === "AbortError" || e.name === "TimeoutError"); last = timeout ? "upstream timeout" : "upstream network failure"; if (attempt < 2) await new Promise((r) => setTimeout(r, 300 * 2 ** attempt + Math.random() * 200)); @@ -742,6 +743,7 @@ async function runChain( try { return await callModel(env, cfg, input, labels, instructions, multi, meter); } catch (e) { + if (e instanceof SpendingError) throw e; last = e; } } @@ -959,7 +961,7 @@ async function classifyMany( jev = await (layaRun ?? startLaya(env, layaPlan, meter, layaTiming)); } else jev = await jevClassify(keys!, inputs, labels, instructions, !!multi, meter); } catch (e) { - if (layaPlan) throw e; + if (layaPlan || e instanceof SpendingError || meter?.permit?.error) throw meter?.permit?.error ?? e; console.warn(`jev failed, falling back: ${(e as Error).message}`); } if (jev) { @@ -994,7 +996,7 @@ async function classifyMany( return { results, escalationFailed }; } } - if (inputs.length > FALLBACK_MAX_INPUTS) { + if (!meter?.permit && inputs.length > FALLBACK_MAX_INPUTS) { throw new Error(`batch classification is temporarily unavailable; send up to ${FALLBACK_MAX_INPUTS} inputs or retry shortly`); } return { results: await llmClassifyMany(env, inputs, labels, tier, instructions, multi, meter), escalationFailed: 0 }; @@ -1010,9 +1012,12 @@ async function classifyMatrix(env: Env, inputs: string[], dimensions: Dimension[ jev = inputs.map((_, i) => flat.slice(i * dimensions.length, (i + 1) * dimensions.length)); } else if (keys) { try { jev = await classifyDimensions(keys, batches, meter); } - catch (e) { console.warn(`dimensions Jev failed: ${(e as Error).message}`); } + catch (e) { + if (e instanceof SpendingError || meter.permit?.error) throw meter.permit?.error ?? e; + console.warn(`dimensions Jev failed: ${(e as Error).message}`); + } } - if (!jev && inputs.length * dimensions.length > FALLBACK_MAX_INPUTS) { + if (!jev && !meter.permit && inputs.length * dimensions.length > FALLBACK_MAX_INPUTS) { throw new Error(`batch classification is temporarily unavailable; send up to ${FALLBACK_MAX_INPUTS} decisions or retry shortly`); } const results: Result[][] = inputs.map(() => []); @@ -1873,8 +1878,8 @@ const worker = { // Every API answer, success or not, says which version answered, how much // room is left (IETF RateLimit header fields), and echoes an idempotency - // key if the caller sent one — classification has no side effects, so the - // echo is all a retrying client needs. + // key if the caller sent one. Spending admission rejects duplicate work; + // responses are not cached for replay. const apiHeaders = (remaining = -1): Record => { const isLaya = selectedModel !== "jev"; const limit = isLaya ? Math.min(LAYA_LIMITS[processing].rpm, TIERS[tier].rpm * multiplier) : TIERS[tier].rpm * multiplier; @@ -1893,7 +1898,7 @@ const worker = { if (idem) h["idempotency-key"] = idem.slice(0, 255); return h; }; - const fail = (msg: string, status: number, reason: ErrorCode, extra: Record = {}, ms = 0, remaining = -1, more: Record = {}) => { + const fail = (msg: string, status: number, reason: ErrorCode, extra: Record = {}, ms = 0, remaining = -1, more: Record = {}) => { record(env, ctx, { tier, n: 0, ms, labels, ip, country, status, client, model: selectedModel === "jev" ? "" : layaModel(selectedModel), usd: meter.usd, reason, agent, attempted: inputs.length, escalationFailed: 0, mode, dimensions: dimensions?.length ?? 0 }); const headers = { ...apiHeaders(remaining), ...extra }; @@ -2128,7 +2133,14 @@ const worker = { escalationFailed = r.escalationFailed; fallbackDecisions = r.fallbackDecisions; } else ({ results, escalationFailed } = await classifyMany(env, inputs, labels, tier, instructions, multi, meter, layaPlan, layaTiming, layaRun)); + if (meter.permit?.error) throw meter.permit.error; } catch (e) { + const spending = e instanceof SpendingError ? e : meter.permit?.error; + if (spending) { + const response = errorResponse(spending); + return fail(spending.message, spending.status, spending.code as ErrorCode, + Object.fromEntries(response.headers), Date.now() - started, -1, await response.json() as Record); + } if (e instanceof LayaError) return fail(e.message, e.status, e.status === 400 ? "laya_input" : e.status === 429 ? "laya_rate_limit" : "laya_unavailable", e.status === 400 ? {} : { "retry-after": String(e.retryAfter) }, Date.now() - started); diff --git a/src/jev.ts b/src/jev.ts index 0f66f3d..eb2c776 100644 --- a/src/jev.ts +++ b/src/jev.ts @@ -1,4 +1,5 @@ import { providerFetch } from "./spending/permit"; +import { SpendingError } from "./spending/policy"; /** * TypeSafe's Jev, the model behind both tiers. * @@ -515,6 +516,7 @@ async function postBeam(key: string, body: JevBody, backend: Backend, meter?: Me signal: signal ? AbortSignal.any([signal, AbortSignal.timeout(UPSTREAM_TIMEOUT_MS)]) : AbortSignal.timeout(UPSTREAM_TIMEOUT_MS), }); } catch (e) { + if (e instanceof SpendingError) throw e; const timeout = e instanceof Error && (e.name === "AbortError" || e.name === "TimeoutError"); observe(timeout ? 504 : 0, timeout ? "timeout" : "network"); last = new JevError(`beam ${timeout ? "timeout" : "network failure"}`, timeout ? 504 : 0, timeout ? "timeout" : "network"); @@ -584,6 +586,7 @@ async function postTypesafe(key: string, body: JevBody, meter?: Meter, analytics signal: AbortSignal.timeout(UPSTREAM_TIMEOUT_MS), }); } catch (e) { + if (e instanceof SpendingError) throw e; const timeout = e instanceof Error && (e.name === "AbortError" || e.name === "TimeoutError"); observe(timeout ? 504 : 0, timeout ? "timeout" : "network"); last = new JevError(`typesafe ${timeout ? "timeout" : "network failure"}`, timeout ? 504 : 0, timeout ? "timeout" : "network"); diff --git a/src/openapi.ts b/src/openapi.ts index 536f957..8f2cd98 100644 --- a/src/openapi.ts +++ b/src/openapi.ts @@ -124,7 +124,7 @@ const errors = (plain: boolean) => ({ "Retry-After": { schema: { type: "integer" }, description: "Seconds until the window resets." }, ...RATE_LIMIT_HEADERS, }, plain), - "502": err("The model provider failed after retries; retry with backoff. `code` is typesafe_ or typesafe (the decision model), openrouter_, chain_exhausted or timeout (the fallback chain), batch_unavailable (more than 20 inputs while the decision model is down) or upstream_other.", RATE_LIMIT_HEADERS, plain), + "502": err("The model provider failed after retries; retry with backoff. `code` is typesafe_ or typesafe (the decision model), openrouter_, chain_exhausted or timeout (the fallback chain), batch_unavailable or upstream_other. Fallback remains bounded by the request spending allowance.", RATE_LIMIT_HEADERS, plain), default: err("Any other error, same {error, code} shape.", undefined, plain), }); const ERRORS = errors(false); @@ -1291,8 +1291,8 @@ Each results[i].dimensions[name] has a label, confidence, scores and model. Up to 20 dimensions and 1,000 item × dimension decisions; each decision counts against the quota. Each dimension may instead be {"labels":[...],"instructions":"..."}. Do not combine dimensions with labels, multi or max_labels. Smart escalation is -per field; escalated fields have null confidence and scores. LLM fallback is -limited to 20 decisions; larger requests return 502 if Jev is unavailable. +per field; escalated fields have null confidence and scores. Fallback processes +the batch with bounded concurrency inside the same request spending allowance. ## Multi-label diff --git a/src/pages.ts b/src/pages.ts index 80abbc1..17a9b40 100644 --- a/src/pages.ts +++ b/src/pages.ts @@ -262,9 +262,14 @@ ENDPOINTS GET /api/v1/receipts/{id} Poll whether an agent report landed GET /api/v1/policy Feedback categories, evidence types and limits - Response for POST: {"tier", "model", "results": [{"label", "confidence", "scores"}...], "usage"} - in input order. Multi-label results carry "labels" (every label >= 0.7) and - independent "scores". Full field reference: https://classifier.dev (PARAMETERS). + Single-label response for POST: {"tier", "model", "results": [{"label", "confidence", "scores"}...], "usage"} + + Results preserve input order. Confidence and scores may be null; check for + null before numeric comparisons. Multi-label results instead have labels + (an array), independent scores and model, with no singular label or confidence. + Labels scoring at least 0.7 are returned. Repeated inference can vary; it is + not an exact deterministic computation. Full field reference: + https://classifier.dev (PARAMETERS). AUTHENTICATION @@ -381,12 +386,13 @@ ERRORS bad_json. 404: not_found. 429: rate_limit_minute, rate_limit_day (with Retry-After). 502: typesafe or typesafe_ when the decision model failed; openrouter_, chain_exhausted or timeout when the fallback - chain did; batch_unavailable for more than twenty inputs while the decision - model is down; upstream_other. Retry 502s with backoff. - 401: invalid_api_key for unsupported credentials. Workspace authentication - also returns 401 for invalid, paused or revoked keys; workspace errors carry - an error message without a code. 402: insufficient balance for inference. - 403: the key is inactive or the workspace cannot authorize usage. + chain did; upstream_other. A 402 request_spending_limit means the batch + needs fewer or shorter inputs, or a funded workspace key. Retry 502s with backoff. + 401: invalid_api_key for unsupported credentials. 403: inactive_api_key for + paused or revoked workspace keys; proxy_requires_payment when a free caller + uses anonymous proxy infrastructure. 402: insufficient_balance or + request_spending_limit. Follow action and retryable in spending errors; + retrying an unchanged over-budget request will not make it fit. The list a client can validate against: components.schemas.Error in https://classifier.dev/openapi.json @@ -399,8 +405,10 @@ VERSIONING current major. Every response carries an x-api-version header. A breaking change would ship as /v2 alongside /v1, and /v1 would then carry Deprecation and Sunset headers for at least six months before removal. Classification is - idempotent by nature; an Idempotency-Key header is accepted and echoed so - retry logic that expects one keeps working. + side-effect-free, but inference has a cost. Send Idempotency-Key to prevent + duplicate work: a repeated admitted key returns 409 duplicate_request, not + a cached result. Keep the original response or request ID; changing the key + starts a new billable operation. SOURCE AND SUPPORT diff --git a/tests/spending.e2e.test.ts b/tests/spending.e2e.test.ts index 8bfec47..cf4eb10 100644 --- a/tests/spending.e2e.test.ts +++ b/tests/spending.e2e.test.ts @@ -366,3 +366,50 @@ test("failed Smart reviews charge only base input and a failed request returns i expect(await env.APP_DB.prepare("SELECT balance::text FROM app_accounts WHERE id='local-demo'").first()).toEqual(before); expect(await env.APP_DB.prepare("SELECT status FROM app_usage WHERE id=?").bind(failed!.headers.get("x-request-id")).first()).toEqual({ status: "refunded" }); }); + +test("free admission denials are recorded as quota responses, never provider failures", async () => { + const points: { doubles: number[]; blobs: string[] }[] = []; + const s = setup({ FREE_DAILY_USD: "0.005", AE: { writeDataPoint(point: { doubles: number[]; blobs: string[] }) { points.push(point); } } }); + providers(); + const response = await worker.fetch(request(undefined, undefined, { inputs: Array(119).fill("Invoice"), labels: ["billing", "support"] }), s.env, s.ctx); + expect(response.status).toBe(429); + await s.flush(); + expect(points.length).toBeGreaterThan(0); + expect(points.some(p => Number(p.blobs[3]) >= 500)).toBe(false); + expect(points.some(p => p.blobs.includes("free_daily_budget"))).toBe(true); +}); + +test("a protected 119-item batch can recover from a Jev outage within its spending allowance", async () => { + const s = setup(); let fallbackCalls = 0; + globalThis.fetch = (async (url, init) => { + if (String(url).startsWith("https://api.spur.us/")) return Response.json({}); + if (String(url).includes("typesafe.ai")) return Response.json({ detail: { error_type: "insufficient_credits" } }, { status: 402 }); + fallbackCalls++; + const b = JSON.parse(String(init?.body)); + return Response.json({ model: b.model, choices: [{ message: { content: "A" } }], usage: { prompt_tokens: 100, completion_tokens: 1, cost: 0.000001812 } }); + }) as typeof fetch; + const response = await worker.fetch(request(undefined, undefined, { inputs: Array(119).fill("Invoice"), labels: ["billing", "support"] }), s.env, s.ctx); + expect(response.status).toBe(200); + expect((await response.json() as { results: unknown[] }).results).toHaveLength(119); + expect(fallbackCalls).toBe(119); + await s.flush(); +}); + +test("a Smart request that exhausts its permit records the final 402, not a successful classification", async () => { + const points: { blobs: string[] }[] = []; + const s = setup({ AE: { writeDataPoint(point: { blobs: string[] }) { points.push(point); } } }); + providers(); const primary = globalThis.fetch; + globalThis.fetch = (async (url, init) => { + const response = await primary(url, init); + if (String(url).startsWith("https://api.spur.us/")) return response; + const body = await response.json() as { usage: { input_tokens: number }; answers: Record }; + body.usage.input_tokens = 60000; + for (const answer of Object.values(body.answers)) answer.confidence = 0.51; + return Response.json(body); + }) as typeof fetch; + const response = await worker.fetch(request(undefined, undefined, { input: "Invoice", labels: ["billing", "support"], tier: "smart" }), s.env, s.ctx); + expect(response.status).toBe(402); + await s.flush(); + expect(points.filter(p => p.blobs[3] === "200")).toHaveLength(0); + expect(points.some(p => p.blobs[3] === "402" && p.blobs.includes("request_spending_limit"))).toBe(true); +});