From d9e89a199493c0f66f1aee95fda1d5d223069457 Mon Sep 17 00:00:00 2001 From: myxamediyar Date: Wed, 23 Sep 2026 05:02:38 +0500 Subject: [PATCH] Route documents over 32,000 characters to chunklaya An input over MAX_CHARS used to be refused with input_too_long. When CHUNKLAYA_URL and CHUNKLAYA_TOKEN are set and CHUNKLAYA_ENABLED is "true", it is now answered by chunklaya, our own long-document service (Laya behind a chunk-and-index harness, github.com/myxamediyar/chunklaya, serve/), and the result is labelled chunklaya/multilingual. Nothing at or under 32,000 characters changes; an explicit model "jev" keeps its ceiling; model "chunklaya" selects it for shorter text. Unconfigured, the old 400 stands. The service speaks System One, so it is a fourth Backend in src/jev.ts through the bearer transport Beam already uses, with the URL and token read from the environment, one document per request, a 60 s deadline, and no per-token cost (the pod is billed by the hour). Its refusals are reported as chunklaya_input, chunklaya_busy and chunklaya_unavailable in our own words, and never fall back to Jev or the LLM chain. Ceilings: 4,000,000 characters per input, 20 inputs per request, no smart tier. Dimensions pack against the backend's limits so every question about a document travels in one request. Billing prices the model at zero in the rate card, the spending table and the reservation bound. Docs, OpenAPI, MCP and the CLI render the new model and codes from the same constants. The deploy workflow passes CHUNKLAYA_URL and CHUNKLAYA_TOKEN to the Worker when both exist as repository secrets, and leaves them out otherwise. Co-Authored-By: Claude Fable 5.1 --- .github/workflows/deploy.yml | 9 +- AGENTS.md | 10 ++ cli/classify.js | 8 +- src/cost.ts | 2 +- src/dimensions.ts | 12 ++- src/docs.ts | 38 ++++++- src/http/spending-classification.ts | 2 +- src/index.ts | 79 +++++++++++--- src/jev-observability.ts | 2 +- src/jev.ts | 63 ++++++++--- src/mcp.ts | 6 +- src/openapi.ts | 8 +- src/retail-rates.json | 9 +- src/server/token-pricing.ts | 2 +- src/server/token-reservation.ts | 1 + src/spending/policy.ts | 2 + test/chunklaya.test.ts | 155 ++++++++++++++++++++++++++++ wrangler.example.toml | 6 ++ 18 files changed, 358 insertions(+), 56 deletions(-) create mode 100644 test/chunklaya.test.ts diff --git a/.github/workflows/deploy.yml b/.github/workflows/deploy.yml index 7a5b975..3a5e3c1 100644 --- a/.github/workflows/deploy.yml +++ b/.github/workflows/deploy.yml @@ -118,15 +118,22 @@ jobs: DATABASE_URL: ${{ secrets.DATABASE_URL }} run: bun scripts/check-newsletter-cutover.ts --activate + # The chunklaya pair is optional and travels together: with both set as + # repository secrets the Worker routes long documents there; with either + # missing they are left out of the file, so an existing value survives + # and an unconfigured Worker keeps answering input_too_long. - name: Deploy env: CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }} DATABASE_URL: ${{ secrets.DATABASE_URL }} + CHUNKLAYA_URL: ${{ secrets.CHUNKLAYA_URL }} + CHUNKLAYA_TOKEN: ${{ secrets.CHUNKLAYA_TOKEN }} run: | node --input-type=module -e ' import { writeFileSync } from "node:fs"; + const { DATABASE_URL, CHUNKLAYA_URL, CHUNKLAYA_TOKEN } = process.env; writeFileSync(process.env.RUNNER_TEMP + "/classifier-secrets.json", - JSON.stringify({ DATABASE_URL: process.env.DATABASE_URL }), { mode: 0o600 }); + JSON.stringify({ DATABASE_URL, ...(CHUNKLAYA_URL && CHUNKLAYA_TOKEN ? { CHUNKLAYA_URL, CHUNKLAYA_TOKEN } : {}) }), { mode: 0o600 }); ' npx wrangler deploy --secrets-file "$RUNNER_TEMP/classifier-secrets.json" diff --git a/AGENTS.md b/AGENTS.md index 464cb66..c494846 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -42,6 +42,16 @@ the plain text (`curl classifier.dev`), the HTML and the Markdown never drift. oversized context rather than truncating, so a context refusal is translated to `max_tokens_exceeded` and the batch halves and retries. Unlike Jev, a Beam request is never retried: a lane quota counts attempts. +- chunklaya (`chunklaya/multilingual`) is our own long-document service: Laya + behind a chunk-and-index harness, github.com/myxamediyar/chunklaya under + `serve/`, on a RunPod pod. It speaks System One too, so it is a fourth + transport in `src/jev.ts`, reached through `CHUNKLAYA_URL` and + `CHUNKLAYA_TOKEN` (Worker secrets). The Worker sends an input over + `MAX_CHARS` there when `CHUNKLAYA_ENABLED` is `"true"` and both secrets are + set; otherwise such inputs stay `input_too_long`. One document is one + request; the service refuses rather than truncates, its 4xx become + `chunklaya_input`, and there is no fallback to Jev or the LLM chain. It is + billed by the hour, so no per-token provider cost is metered. - The updates roadmap is one constant, `ROADMAP` in `src/newsletter.ts`; the plain text, the signup form and the Markdown all render from it. Addresses go to the `subscriber` table in the shared application Neon database. Preserve consent, diff --git a/cli/classify.js b/cli/classify.js index efc12fa..f376cec 100755 --- a/cli/classify.js +++ b/cli/classify.js @@ -42,6 +42,7 @@ OPTIONS -k, --max at most n labels (implies --multi) -s, --smart re-ask uncertain answers of a reasoning model (slower) --model laya opt into the automatically routed Laya trial (default: jev) + --model chunklaya the long-document model; also automatic past 32,000 characters --processing bulk Laya bulk lane; default fast is one decision per call -i, --instructions extra criteria: "judge only the service, ignore the food" -r, --review print only inputs with confidence below t @@ -121,8 +122,8 @@ function parseArgs(argv) { if (!(Number.isInteger(o.max) && o.max > 0)) fail("--max takes a whole number above 0, e.g. --max 3"); o.multi = true; // the API reads max_labels as multi-label; a single label cannot be capped } - if (!["jev", "laya", "kev"].includes(o.model)) fail("--model must be jev, laya or kev"); - if (!["fast", "bulk"].includes(o.processing) || o.model === "jev" && o.processing !== "fast") fail("--processing bulk requires --model laya or --model kev"); + if (!["jev", "laya", "kev", "chunklaya"].includes(o.model)) fail("--model must be jev, laya, kev or chunklaya"); + if (!["fast", "bulk"].includes(o.processing) || (o.model === "jev" || o.model === "chunklaya") && o.processing !== "fast") fail("--processing bulk requires --model laya or --model kev"); return o; } @@ -175,7 +176,7 @@ async function readStdin() { async function post(o, inputs) { const body = { inputs, labels: o.labels }; - if (o.model !== "jev") { body.model = o.model; body.processing = o.processing; } + if (o.model !== "jev") { body.model = o.model; if (o.model !== "chunklaya") body.processing = o.processing; } if (o.multi) body.multi = true; if (o.max) body.max_labels = o.max; if (o.smart) body.tier = "smart"; @@ -244,6 +245,7 @@ function configuredBatch() { } function batchSize(o) { + if (o.model === "chunklaya") return Math.min(configuredBatch(), 20); // one document is one upstream request there if (o.model !== "jev") return Math.min(configuredBatch(), o.processing === "fast" ? 1 : Math.floor(1000 / (o.multi ? o.labels.length : 1)), o.smart ? SMART_BATCH : MAX_BATCH); return Math.min(configuredBatch(), o.smart && !o.apiKey ? SMART_BATCH : MAX_BATCH); } diff --git a/src/cost.ts b/src/cost.ts index 3eee0a2..679f5fa 100644 --- a/src/cost.ts +++ b/src/cost.ts @@ -30,7 +30,7 @@ export type TokenCounts = { }; export type ModelTokenUsage = TokenCounts & { - provider: "typesafe" | "vercel" | "openrouter" | "beam"; + provider: "typesafe" | "vercel" | "openrouter" | "beam" | "chunklaya"; model: string; calls: number; }; diff --git a/src/dimensions.ts b/src/dimensions.ts index d2232dc..f61a8b8 100644 --- a/src/dimensions.ts +++ b/src/dimensions.ts @@ -7,6 +7,9 @@ import { type JevQuestionGroup, type JevResult, type Question, + type Limits, + type Backend, + JEV_BACKEND, } from "./jev"; import type { Meter } from "./cost"; @@ -57,7 +60,8 @@ type Cell = { item: number; dimension: number; id: string; question: Extract; /** State is shared across questions. Respect BOTH Jev context limits, with headroom. */ -export function packDimensions(inputs: string[], dimensions: Dimension[], shared?: string): DimensionBatch[] { +/** `limits` packs for another backend; chunklaya takes one document per request and has no token budget to respect. */ +export function packDimensions(inputs: string[], dimensions: Dimension[], shared?: string, limits?: Limits): DimensionBatch[] { const groups: JevQuestionGroup[] = []; inputs.forEach((text, item) => dimensions.forEach((d, dimension) => { const question: Question = { @@ -69,7 +73,7 @@ export function packDimensions(inputs: string[], dimensions: Dimension[], shared groups.push({ state: { id: `i${item}`, text }, questions: { [cell.id]: question }, value: cell }); })); try { - return prepareJevBatches(groups, { rejectOversized: true }); + return prepareJevBatches(groups, { rejectOversized: !limits, limits }); } catch (error) { if (error instanceof JevContextError) { throw new DimensionError("An input and dimension exceed Jev's context budget; shorten the input or dimension instructions"); @@ -79,9 +83,9 @@ export function packDimensions(inputs: string[], dimensions: Dimension[], shared } /** One result per matrix cell. Splitting by question also handles a single wide item. */ -export async function classifyDimensions(keys: JevKeys, batches: DimensionBatch[], meter?: Meter): Promise { +export async function classifyDimensions(keys: JevKeys, batches: DimensionBatch[], meter?: Meter, backend: Backend = JEV_BACKEND): Promise { const results: JevResult[][] = []; - for (const { value: cell, model, answers } of await runJevBatches(keys, batches, meter)) { + for (const { value: cell, model, answers } of await runJevBatches(keys, batches, meter, backend)) { const answer = answers[cell.id]; // The shared runner validates every answer before returning. (results[cell.item] ??= [])[cell.dimension] = { label: answer.choice!, confidence: answer.confidence!, diff --git a/src/docs.ts b/src/docs.ts index faebee5..f9e5a39 100644 --- a/src/docs.ts +++ b/src/docs.ts @@ -140,6 +140,35 @@ LAYA AND KEV node cli/classify.js billing,technical --model kev --processing bulk < tickets.txt +LONG DOCUMENTS + + An input over 32,000 characters is answered by chunklaya, our own + long-document service, when it is configured; otherwise it is refused with + input_too_long as before. chunklaya is Laya behind a harness: the document + is split into passages and indexed once, and every question in the request + is answered off that index, so a request with dimensions asks all of them + in one pass. Up to 4,000,000 characters per input and 20 inputs per + request. The result is labelled chunklaya/multilingual. Nothing at or under + 32,000 characters changes, and an explicit model: "jev" keeps the 32,000 + ceiling; model: "chunklaya" selects it for shorter text too. + + {"input": "", + "dimensions": {"kind": ["lease", "employment", "supply"], + "renews": ["automatically", "on notice", "never"]}} + + What it will not do. tier: "smart" is refused with bad_tier: the reviewing + model cannot read the document. A document with more passages than the + service scores in one request (256 paragraphs), or a request with more + questions than fit, is refused with chunklaya_input rather than answered + from part of the text. A busy service answers 429 chunklaya_busy with + Retry-After; an unreachable one 503 chunklaya_unavailable. There is no + fallback to another model. + + Accuracy on long documents has not been measured against Jev on + classifier.dev traffic; the harness's own results are in its repository, + github.com/myxamediyar/chunklaya. No retail charge during this trial. + + AGAINST THE MODEL IT RUNS ON ${vsJevText(false)} @@ -323,7 +352,9 @@ PARAMETERS labels Two to one hundred categories. Required unless dimensions is supplied. dimensions Named label sets for independent decisions; see MULTIPLE DIMENSIONS. - input The text to classify, up to 32,000 characters. + input The text to classify, up to 32,000 characters. Longer + documents, up to 4,000,000, are answered by chunklaya; see + LONG DOCUMENTS. inputs Up to one thousand strings classified in a single call. tier Either fast (the default) or smart, in any case. Anything else is a 400 with code bad_tier, never a silent fast. @@ -466,14 +497,15 @@ ERRORS and try as fields. 400 bad_json, no_input, too_many_inputs, too_few_labels, too_many_labels, - empty_label, duplicate_labels, empty_input, input_too_long, bad_tier + empty_label, duplicate_labels, empty_input, input_too_long, bad_tier, + chunklaya_input 401 invalid_api_key for unsupported credentials. Workspace authentication also rejects invalid, paused or revoked keys with an error message; workspace errors do not include a code. 402 insufficient workspace balance for inference 403 the key is inactive or the workspace cannot authorize usage 404 not_found - 429 rate_limit_minute, rate_limit_day, with Retry-After; on the free + 429 rate_limit_minute, rate_limit_day, chunklaya_busy, with Retry-After; on the free tier the body also carries upgrade, the URL of the plan that lifts the limit (https://classifier.dev/pricing) 502 typesafe or typesafe_ when the decision model failed; diff --git a/src/http/spending-classification.ts b/src/http/spending-classification.ts index 3d204f5..42859fa 100644 --- a/src/http/spending-classification.ts +++ b/src/http/spending-classification.ts @@ -27,7 +27,7 @@ export async function spendingClassification(request: Request, env: AppEnv & Par const keyHash = await hashToken(request.headers.get("authorization")!.replace(/^Bearer\s+/i, "")); const maxCredits = Math.ceil(limits.paidRequest / 10000); const decisions = items * (body.dimensions && typeof body.dimensions === "object" ? Math.max(1, Object.keys(body.dimensions).length) : 1); - const trial = body.model === "laya" || body.model === "kev"; + const trial = body.model === "laya" || body.model === "kev" || body.model === "chunklaya"; const quote = Number((classificationCharge(trial ? 0 : 65536 * decisions, body.tier === "smart" ? decisions : 0).nanodollars + 9999n) / 10000n); if (quote > maxCredits) throw new SpendingError(402, "request_spending_limit", "This request exceeds the workspace request allowance. Split the batch."); const idempotencyHash = idem ? await fingerprint(env, `account-idempotency:${idem}`) : null; diff --git a/src/index.ts b/src/index.ts index 6727c68..ef7e112 100644 --- a/src/index.ts +++ b/src/index.ts @@ -22,7 +22,7 @@ import { runAlerts } from "./alerts"; import * as feedback from "./feedback"; import * as newsletter from "./newsletter"; import * as skills from "./skills"; -import { jevClassify, jevKeys, MULTI_THRESHOLD } from "./jev"; +import { jevClassify, jevKeys, MULTI_THRESHOLD, JEV_BACKEND, CHUNKLAYA_BACKEND, JevError, type Backend } from "./jev"; import { LAYA_LIMITS, LayaError, planLaya, runLaya, limitLaya, layaModel, readQuotaTiming, type QuotaTiming, type LayaEnv, type LayaPlan, type LayaTiming, type Processing, type LayaModel } from "./laya"; import { readDimensions, packDimensions, classifyDimensions, dimensionInstructions, MAX_DECISIONS, type Dimension, type DimensionBatch } from "./dimensions"; import { newMeter, addUsd, addTokens, type Meter } from "./cost"; @@ -43,6 +43,16 @@ export interface Env extends LayaEnv, SpendingEnv { TYPESAFE_API_KEY?: string; /** Beam workspace token; the credential for the Beam-hosted Laya and Kev. */ BEAM_API_KEY?: string; + /** + * chunklaya, our own long-document service: Laya behind a chunk-and-index + * harness on a pod (github.com/myxamediyar/chunklaya, serve/). An input over + * MAX_CHARS goes there when both secrets are set and CHUNKLAYA_ENABLED is + * "true"; otherwise it keeps answering input_too_long. The URL is the pod's + * proxy address and the token its bearer; both are Worker secrets. + */ + CHUNKLAYA_URL?: string; + CHUNKLAYA_TOKEN?: string; + CHUNKLAYA_ENABLED?: string; /** * Vercel AI Gateway, which serves Jev on a free monthly credit. With it set, * Jev is asked there first and TYPESAFE_API_KEY catches what the gateway @@ -99,6 +109,12 @@ const MAX_LABELS_SINGLE = 26; // The fallback is one upstream call per input, so it cannot take a real batch. const FALLBACK_MAX_INPUTS = 20; const MAX_CHARS = 32_000; +// Past MAX_CHARS a document goes to chunklaya, which indexes it once and answers +// every question in the request off the index. The service refuses more than +// this: a million tokens of prose, roughly. +const CHUNKLAYA_MAX_CHARS = 4_000_000; +// One document is one upstream request there, so a batch is bounded by count. +const CHUNKLAYA_MAX_INPUTS = 20; /** The schedule that runs the alert check rather than the digest. */ const ALERT_CRON = "*/15 * * * *"; // Smart tier: a fast answer below this confidence is re-asked of the reasoning chain. @@ -951,6 +967,7 @@ async function classifyMany( layaPlan?: LayaPlan, layaTiming?: LayaRequestTiming, layaRun?: LayaRun, + backend: Backend = JEV_BACKEND, ): Promise<{ results: Result[]; escalationFailed: number }> { const keys = jevKeys(env); if (keys || layaPlan) { @@ -959,9 +976,11 @@ async function classifyMany( try { if (layaPlan) { jev = await (layaRun ?? startLaya(env, layaPlan, meter, layaTiming)); - } else jev = await jevClassify(keys!, inputs, labels, instructions, !!multi, meter); + } else jev = await jevClassify(keys!, inputs, labels, instructions, !!multi, meter, backend); } catch (e) { - if (layaPlan || e instanceof SpendingError || meter?.permit?.error) throw meter?.permit?.error ?? e; + // A document too long for Jev is too long for the LLM chain as well: a + // chunklaya failure is reported, never quietly answered by another model. + if (layaPlan || backend.id === "chunklaya" || e instanceof SpendingError || meter?.permit?.error) throw meter?.permit?.error ?? e; console.warn(`jev failed, falling back: ${(e as Error).message}`); } if (jev) { @@ -1003,7 +1022,7 @@ async function classifyMany( } /** Keep each field independent, including smart escalation and the bounded LLM fallback. */ -async function classifyMatrix(env: Env, inputs: string[], dimensions: Dimension[], batches: DimensionBatch[], tier: Tier, instructions: string | undefined, meter: Meter, layaPlan?: LayaPlan, layaTiming?: LayaRequestTiming, layaRun?: LayaRun) { +async function classifyMatrix(env: Env, inputs: string[], dimensions: Dimension[], batches: DimensionBatch[], tier: Tier, instructions: string | undefined, meter: Meter, layaPlan?: LayaPlan, layaTiming?: LayaRequestTiming, layaRun?: LayaRun, backend: Backend = JEV_BACKEND) { const started = Date.now(); let jev: Awaited> | undefined; const keys = jevKeys(env); @@ -1011,9 +1030,9 @@ async function classifyMatrix(env: Env, inputs: string[], dimensions: Dimension[ const flat = await (layaRun ?? startLaya(env, layaPlan, meter, layaTiming)); jev = inputs.map((_, i) => flat.slice(i * dimensions.length, (i + 1) * dimensions.length)); } else if (keys) { - try { jev = await classifyDimensions(keys, batches, meter); } + try { jev = await classifyDimensions(keys, batches, meter, backend); } catch (e) { - if (e instanceof SpendingError || meter.permit?.error) throw meter.permit?.error ?? e; + if (backend.id === "chunklaya" || e instanceof SpendingError || meter.permit?.error) throw meter.permit?.error ?? e; console.warn(`dimensions Jev failed: ${(e as Error).message}`); } } @@ -1858,7 +1877,9 @@ const worker = { let inputs: string[] = []; let labels: string[] = []; let tier: Tier = "fast"; - let selectedModel: "jev" | LayaModel = "jev"; + let selectedModel: "jev" | LayaModel | "chunklaya" = "jev"; + // Set when the body named a model, so a long input under an explicit "jev" stays refused rather than rerouted. + let explicitModel = false; let processing: Processing = "fast"; let automaticProcessing = true; let layaPlan: LayaPlan | undefined; @@ -1881,7 +1902,7 @@ const worker = { // key if the caller sent one. Spending admission rejects duplicate work; // responses are not cached for replay. const apiHeaders = (remaining = -1): Record => { - const isLaya = selectedModel !== "jev"; + const isLaya = selectedModel === "laya" || selectedModel === "kev"; const limit = isLaya ? Math.min(LAYA_LIMITS[processing].rpm, TIERS[tier].rpm * multiplier) : TIERS[tier].rpm * multiplier; const daily = isLaya ? Math.min(LAYA_LIMITS[processing].daily, TIERS[tier].daily * multiplier) : TIERS[tier].daily * multiplier; const h: Record = { @@ -1899,7 +1920,7 @@ const worker = { return h; }; const fail = (msg: string, status: number, reason: ErrorCode, extra: Record = {}, ms = 0, remaining = -1, more: Record = {}) => { - record(env, ctx, { tier, n: 0, ms, labels, ip, country, status, client, model: selectedModel === "jev" ? "" : layaModel(selectedModel), + record(env, ctx, { tier, n: 0, ms, labels, ip, country, status, client, model: selectedModel === "jev" ? "" : selectedModel === "chunklaya" ? CHUNKLAYA_BACKEND.model : layaModel(selectedModel), usd: meter.usd, reason, agent, attempted: inputs.length, escalationFailed: 0, mode, dimensions: dimensions?.length ?? 0 }); const headers = { ...apiHeaders(remaining), ...extra }; // A GET that was malformed gets back a URL that would have worked, in the spelling it used. @@ -1927,12 +1948,14 @@ const worker = { return fail('Body must be a JSON object such as {"input":"...","labels":["a","b"]}. See https://classifier.dev', 400, "bad_json"); } const b = body as Record; - if (b.model !== undefined && b.model !== "jev" && b.model !== "laya" && b.model !== "kev") - return fail('model must be "jev", "laya" or "kev"', 400, "bad_model"); + if (b.model !== undefined && b.model !== "jev" && b.model !== "laya" && b.model !== "kev" && b.model !== "chunklaya") + return fail('model must be "jev", "laya", "kev" or "chunklaya"', 400, "bad_model"); + explicitModel = b.model !== undefined; // `processing` without a model still means Laya: it is the lane selector // for the Beam-hosted trial and Jev has no lanes. selectedModel = b.model === "laya" || b.model === "kev" ? (b.model as LayaModel) + : b.model === "chunklaya" ? "chunklaya" : b.model === undefined && b.processing !== undefined ? "laya" : "jev"; if (b.processing !== undefined && b.processing !== "fast" && b.processing !== "bulk") return fail('processing must be "fast" or "bulk"', 400, "bad_processing"); @@ -2008,18 +2031,31 @@ const worker = { if (labels.some((l) => typeof l !== "string" || !l.trim())) return fail("Labels must be non-empty strings", 400, "empty_label"); if (new Set(labels).size !== labels.length) return fail("Labels must be distinct", 400, "duplicate_labels"); if (inputs.some((i) => typeof i !== "string" || !i.trim())) return fail("Inputs must be non-empty strings", 400, "empty_input"); - if (inputs.some((i) => i.length > MAX_CHARS)) return fail(`Each input must be at most ${MAX_CHARS.toLocaleString("en-US")} characters`, 400, "input_too_long"); + // Past MAX_CHARS a document goes to chunklaya, when it is configured and the + // caller did not name a model. Everything at or under the ceiling is + // untouched, and an explicit "jev" keeps its ceiling: the reroute answers + // requests that used to fail, never changes one that used to work. + const longest = inputs.reduce((max, input) => Math.max(max, input.length), 0); + const chunklayaReady = env.CHUNKLAYA_ENABLED === "true" && !!jevKeys(env)?.chunklaya; + if (selectedModel === "jev" && !explicitModel && longest > MAX_CHARS && chunklayaReady) selectedModel = "chunklaya"; + if (selectedModel === "chunklaya") { + if (!chunklayaReady) return fail("Long-document classification is currently unavailable", 503, "chunklaya_unavailable", { "retry-after": "60" }); + if (longest > CHUNKLAYA_MAX_CHARS) return fail(`Each input must be at most ${CHUNKLAYA_MAX_CHARS.toLocaleString("en-US")} characters`, 400, "input_too_long"); + if (inputs.length > CHUNKLAYA_MAX_INPUTS) return fail(`Documents over ${MAX_CHARS.toLocaleString("en-US")} characters are classified at most ${CHUNKLAYA_MAX_INPUTS} per request`, 400, "too_many_inputs"); + // The reviewing model cannot read the document, so there is nothing to escalate to. + if (tier === "smart") return fail(`tier "smart" is not available for documents over ${MAX_CHARS.toLocaleString("en-US")} characters; use fast`, 400, "bad_tier"); + } else if (longest > MAX_CHARS) return fail(`Each input must be at most ${MAX_CHARS.toLocaleString("en-US")} characters`, 400, "input_too_long"); if (dimensions) { if (inputs.length * dimensions.length > MAX_DECISIONS) return fail(`Maximum ${MAX_DECISIONS} decisions (items × dimensions) per request`, 400, "too_many_decisions"); - try { if (selectedModel === "jev") dimensionBatches = packDimensions(inputs, dimensions, instructions); } + try { if (selectedModel === "jev" || selectedModel === "chunklaya") dimensionBatches = packDimensions(inputs, dimensions, instructions, selectedModel === "chunklaya" ? CHUNKLAYA_BACKEND.limits : undefined); } catch (e) { return fail((e as Error).message, 400, "dimension_context_too_large"); } } else mode = multi ? "multi" : "single"; const decisions = inputs.length * (dimensions?.length ?? 1); - if (selectedModel !== "jev" && automaticProcessing) { + if ((selectedModel === "laya" || selectedModel === "kev") && automaticProcessing) { processing = decisions > 1 || (!!multi && labels.length > LAYA_LIMITS.fast.questions) ? "bulk" : "fast"; } - if (selectedModel !== "jev") { + if (selectedModel === "laya" || selectedModel === "kev") { if (!account && !req.headers.has("authorization")) layaTiming = {}; if (env.LAYA_ENABLED !== "true") return fail("Laya trial is currently unavailable", 503, "laya_unavailable"); try { @@ -2121,18 +2157,19 @@ const worker = { } const started = Date.now(); + const backend = selectedModel === "chunklaya" ? CHUNKLAYA_BACKEND : JEV_BACKEND; let results: Result[]; let matrix: Result[][] | undefined; let fallbackDecisions = 0; let escalationFailed = 0; try { if (dimensions) { - const r = await classifyMatrix(env, inputs, dimensions, dimensionBatches, tier, instructions, meter, layaPlan, layaTiming, layaRun); + const r = await classifyMatrix(env, inputs, dimensions, dimensionBatches, tier, instructions, meter, layaPlan, layaTiming, layaRun, backend); matrix = r.results; results = matrix.flat(); escalationFailed = r.escalationFailed; fallbackDecisions = r.fallbackDecisions; - } else ({ results, escalationFailed } = await classifyMany(env, inputs, labels, tier, instructions, multi, meter, layaPlan, layaTiming, layaRun)); + } else ({ results, escalationFailed } = await classifyMany(env, inputs, labels, tier, instructions, multi, meter, layaPlan, layaTiming, layaRun, backend)); if (meter.permit?.error) throw meter.permit.error; } catch (e) { const spending = e instanceof SpendingError ? e : meter.permit?.error; @@ -2144,6 +2181,14 @@ const worker = { if (e instanceof LayaError) return fail(e.message, e.status, e.status === 400 ? "laya_input" : e.status === 429 ? "laya_rate_limit" : "laya_unavailable", e.status === 400 ? {} : { "retry-after": String(e.retryAfter) }, Date.now() - started); + if (selectedModel === "chunklaya" && e instanceof JevError) { + // Only our own words travel outward: the service's `detail` can describe the document. + const elapsed = Date.now() - started; + if (e.status === 400 || e.status === 413 || e.status === 422 || e.errorType === "invalid_request" || e.errorType === "max_tokens_exceeded") + return fail("The long-document model refused the request: the document has more passages, or the request more questions, than it scores at once. Split the document or ask fewer questions", 400, "chunklaya_input", {}, elapsed); + if (e.status === 429) return fail("The long-document model is busy; retry with backoff", 429, "chunklaya_busy", { "retry-after": "2" }, elapsed); + return fail("Long-document classification is currently unavailable", 503, "chunklaya_unavailable", { "retry-after": "10" }, elapsed); + } if (e instanceof ProviderConfigurationError) return fail(e.message, 503, "inference_unavailable", {}, Date.now() - started); const msg = (e as Error).message; return fail(`upstream: ${msg}`, 502, upstreamReason(msg), {}, Date.now() - started); diff --git a/src/jev-observability.ts b/src/jev-observability.ts index 8f99b80..b6694f9 100644 --- a/src/jev-observability.ts +++ b/src/jev-observability.ts @@ -2,7 +2,7 @@ * must remain visible without inflating request counts. Never record payloads, * provider messages, keys, caller identifiers or label sets here. */ export type JevAttempt = { - provider: "gateway" | "typesafe" | "beam"; + provider: "gateway" | "typesafe" | "beam" | "chunklaya"; outcome: "success" | "failure" | "skipped"; reason: string; status: number; diff --git a/src/jev.ts b/src/jev.ts index eb2c776..c1db4c4 100644 --- a/src/jev.ts +++ b/src/jev.ts @@ -75,12 +75,14 @@ export function resetGatewayPause() { } /** Where Jev can be asked. Either key alone works; with both, the gateway goes first and TypeSafe catches what it drops. */ -export type JevKeys = { typesafe?: string; gateway?: string; beam?: string; analytics?: AnalyticsEngineDataset }; +export type JevKeys = { typesafe?: string; gateway?: string; beam?: string; chunklaya?: { url: string; token: string }; analytics?: AnalyticsEngineDataset }; -export const jevKeys = (env: { TYPESAFE_API_KEY?: string; AI_GATEWAY_API_KEY?: string; AI_GATEWAY_DISABLED?: string; BEAM_API_KEY?: string; JEV_AE?: AnalyticsEngineDataset }): JevKeys | null => { +export const jevKeys = (env: { TYPESAFE_API_KEY?: string; AI_GATEWAY_API_KEY?: string; AI_GATEWAY_DISABLED?: string; BEAM_API_KEY?: string; CHUNKLAYA_URL?: string; CHUNKLAYA_TOKEN?: string; JEV_AE?: AnalyticsEngineDataset }): JevKeys | null => { const gateway = env.AI_GATEWAY_DISABLED === "true" ? undefined : env.AI_GATEWAY_API_KEY; - return env.TYPESAFE_API_KEY || gateway || env.BEAM_API_KEY - ? { typesafe: env.TYPESAFE_API_KEY, gateway, beam: env.BEAM_API_KEY, ...(env.JEV_AE ? { analytics: env.JEV_AE } : {}) } + // Both halves or neither: a URL without its token would send documents to a service that refuses them. + const chunklaya = env.CHUNKLAYA_URL && env.CHUNKLAYA_TOKEN ? { url: env.CHUNKLAYA_URL, token: env.CHUNKLAYA_TOKEN } : undefined; + return env.TYPESAFE_API_KEY || gateway || env.BEAM_API_KEY || chunklaya + ? { typesafe: env.TYPESAFE_API_KEY, gateway, beam: env.BEAM_API_KEY, chunklaya, ...(env.JEV_AE ? { analytics: env.JEV_AE } : {}) } : null; }; @@ -122,10 +124,10 @@ const BEAM_API = "https://app.beam.cloud/v1/systemone"; export const BEAM_MAX_QUESTIONS = 32; export type Limits = { tokenBudget: number; stateQuestionBudget: number; maxItems: number; maxQuestions: number }; -export type BackendId = "jev" | "laya" | "kev"; +export type BackendId = "jev" | "laya" | "kev" | "chunklaya"; export type Backend = { id: BackendId; - via: "typesafe" | "beam"; + via: "typesafe" | "beam" | "chunklaya"; url: string; /** Sent as `model`, and the label an answer is attributed to. */ model: string; @@ -139,6 +141,8 @@ export type Backend = { */ attempts: number; limits: Limits; + /** Upstream deadline. The default suits Jev and Beam; a document service that indexes a megabyte first needs longer. */ + timeoutMs?: number; }; export const JEV_BACKEND: Backend = { @@ -161,8 +165,25 @@ export const KEV_BACKEND: Backend = { id: "kev", via: "beam", url: BEAM_API, model: "jev/kev", accountModel: "jev/kev", attempts: 1, limits: { tokenBudget: 7_200, stateQuestionBudget: 7_200, maxItems: BEAM_MAX_QUESTIONS, maxQuestions: BEAM_MAX_QUESTIONS }, }; -export const BACKENDS: Record = { jev: JEV_BACKEND, laya: LAYA_BACKEND, kev: KEV_BACKEND }; -export const isBackendId = (v: unknown): v is BackendId => v === "jev" || v === "laya" || v === "kev"; +/** + * chunklaya: Laya behind a chunk-and-index harness, on a pod of ours + * (github.com/myxamediyar/chunklaya, serve/). Same protocol as Beam, so the + * same transport, with three differences that live here: the URL and bearer + * come from the environment rather than a constant, one document is one + * request (the service indexes it once and answers every question off the + * index, so `maxItems` is 1 and there is no token budget to pack against), + * and it refuses oversized work with a 4xx that must not be retried by + * halving — its 422s never say "tokens", so `beamErrorType` reads them as + * `invalid_request`. The answer is labelled by the service itself. + */ +export const CHUNKLAYA_MAX_QUESTIONS = 32; +export const CHUNKLAYA_BACKEND: Backend = { + id: "chunklaya", via: "chunklaya", url: "", model: "chunklaya/multilingual", accountModel: "chunklaya/multilingual", attempts: 1, + limits: { tokenBudget: Number.MAX_SAFE_INTEGER, stateQuestionBudget: Number.MAX_SAFE_INTEGER, maxItems: 1, maxQuestions: CHUNKLAYA_MAX_QUESTIONS }, + timeoutMs: 60_000, +}; +export const BACKENDS: Record = { jev: JEV_BACKEND, laya: LAYA_BACKEND, kev: KEV_BACKEND, chunklaya: CHUNKLAYA_BACKEND }; +export const isBackendId = (v: unknown): v is BackendId => v === "jev" || v === "laya" || v === "kev" || v === "chunklaya"; /** * Tokens in a string, estimated. ASCII runs at about 3.5 characters a token; @@ -465,6 +486,13 @@ async function post(keys: JevKeys, body: JevBody, meter?: Meter, backend: Backen // Beam hosts its own models; there is no gateway door and nothing to fall // back to, so an unconfigured key is an error rather than a silent reroute // to Jev, which would answer with a different model than the caller asked for. + // Our own document service: the same wire protocol as Beam, reached at the + // address in the environment. Unconfigured is an error for the same reason. + if (backend.via === "chunklaya") { + if (!keys.chunklaya) throw new JevError("chunklaya: not configured", 0, "unconfigured"); + const url = keys.chunklaya.url.replace(/\/+$/, "") + "/v1/systemone"; + return postBeam(keys.chunklaya.token, { ...body, model: backend.model }, { ...backend, url }, meter, keys.analytics, signal, "chunklaya"); + } if (backend.via === "beam") { if (!keys.beam) throw new JevError("beam: no key configured", 0, "unconfigured"); return postBeam(keys.beam, { ...body, model: backend.model }, backend, meter, keys.analytics, signal); @@ -501,25 +529,25 @@ async function post(keys: JevKeys, body: JevBody, meter?: Meter, backend: Backen * `max_tokens_exceeded` so the caller halves the batch, which is the only * recovery that can work when an estimate underran a 512-token window. */ -async function postBeam(key: string, body: JevBody, backend: Backend, meter?: Meter, analytics?: AnalyticsEngineDataset, signal?: AbortSignal): Promise { - let last: Error = new Error("beam: no attempt made"); +async function postBeam(key: string, body: JevBody, backend: Backend, meter?: Meter, analytics?: AnalyticsEngineDataset, signal?: AbortSignal, provider: "beam" | "chunklaya" = "beam"): Promise { + let last: Error = new Error(`${provider}: no attempt made`); for (let attempt = 0; attempt < backend.attempts; attempt++) { - await meter?.beforeCall?.("beam", body.model, 0); + await meter?.beforeCall?.(provider, body.model, 0); const started = Date.now(); - const observe = (status: number, reason = "") => recordJevAttempt(analytics, { provider: "beam", outcome: reason ? "failure" : "success", reason, status, ms: Date.now() - started, items: body.state.length, attempt: attempt + 1 }); + const observe = (status: number, reason = "") => recordJevAttempt(analytics, { provider, outcome: reason ? "failure" : "success", reason, status, ms: Date.now() - started, items: body.state.length, attempt: attempt + 1 }); let res: Response; try { res = await providerFetch(meter, "beam", body.model, 0, backend.url, { method: "POST", headers: { authorization: `Bearer ${key}`, "content-type": "application/json" }, body: JSON.stringify(body), - signal: signal ? AbortSignal.any([signal, AbortSignal.timeout(UPSTREAM_TIMEOUT_MS)]) : AbortSignal.timeout(UPSTREAM_TIMEOUT_MS), + signal: signal ? AbortSignal.any([signal, AbortSignal.timeout(backend.timeoutMs ?? UPSTREAM_TIMEOUT_MS)]) : AbortSignal.timeout(backend.timeoutMs ?? UPSTREAM_TIMEOUT_MS), }); } catch (e) { if (e instanceof SpendingError) throw e; const timeout = e instanceof Error && (e.name === "AbortError" || e.name === "TimeoutError"); observe(timeout ? 504 : 0, timeout ? "timeout" : "network"); - last = new JevError(`beam ${timeout ? "timeout" : "network failure"}`, timeout ? 504 : 0, timeout ? "timeout" : "network"); + last = new JevError(`${provider} ${timeout ? "timeout" : "network failure"}`, timeout ? 504 : 0, timeout ? "timeout" : "network"); // A caller that has withdrawn the request gets no further attempts. if (signal?.aborted || attempt >= backend.attempts - 1) break; await new Promise((r) => setTimeout(r, 300 * 2 ** attempt + Math.random() * 200)); @@ -528,13 +556,14 @@ async function postBeam(key: string, body: JevBody, backend: Backend, meter?: Me const rawPayload = await res.json().catch(() => null); const payload = (isRecord(rawPayload) ? rawPayload : {}) as Partial & { error?: unknown; detail?: unknown }; if (res.ok && !payload.error && validPayload(payload, body)) { - addBeamCost(meter, payload.usage?.input_tokens); - addTokens(meter, "beam", payload.model, { inputTokens: payload.usage?.input_tokens, outputTokens: payload.usage?.output_tokens, cachedInputTokens: 0 }); + // A pod of ours is billed by the hour, not the token; only Beam has a per-token price. + if (provider === "beam") addBeamCost(meter, payload.usage?.input_tokens); + addTokens(meter, provider, payload.model, { inputTokens: payload.usage?.input_tokens, outputTokens: payload.usage?.output_tokens, cachedInputTokens: 0 }); observe(res.status); return payload; } observe(res.status, beamFailureReason(res.status, rawPayload)); - last = new JevError(`beam ${res.status}: ${res.ok ? "malformed response" : beamErrorType(res.status, rawPayload)}`, res.status, beamErrorType(res.status, rawPayload)); + last = new JevError(`${provider} ${res.status}: ${res.ok ? "malformed response" : beamErrorType(res.status, rawPayload)}`, res.status, beamErrorType(res.status, rawPayload)); const retryable = res.status === 408 || res.status === 425 || res.status === 429 || res.status === 529 || res.status >= 500 || res.ok; if (!retryable || attempt >= backend.attempts - 1) break; await new Promise((r) => setTimeout(r, 300 * 2 ** attempt + Math.random() * 200)); diff --git a/src/mcp.ts b/src/mcp.ts index d583b02..e438ade 100644 --- a/src/mcp.ts +++ b/src/mcp.ts @@ -136,7 +136,7 @@ export function productServer(classify: ClassifyFn): McpServer { inputSchema: { type: "object", properties: { inputs: INPUTS_SCHEMA, labels: LABELS_SCHEMA, instructions: INSTRUCTIONS_SCHEMA, tier: TIER_SCHEMA, - model: { type: "string", enum: ["jev", "laya", "kev"], description: "Optional Beam-hosted models: 'laya' (ModernBERT-large, 512-token context) or 'kev' (Qwen2.5-0.5B with a pointer head, 8K context). Jev remains the default." }, + model: { type: "string", enum: ["jev", "laya", "kev", "chunklaya"], description: "Optional models: 'laya' (ModernBERT-large, 512-token context) or 'kev' (Qwen2.5-0.5B with a pointer head, 8K context), or 'chunklaya', our long-document model (up to 4,000,000 characters, 20 inputs per request), which is also chosen automatically for any input over 32,000 characters. Jev remains the default." }, processing: { type: "string", enum: ["fast", "bulk"], description: "Optional. Implies Laya if model is omitted; has no effect with explicit Jev. With Laya, omit to select fast for one decision or bulk for batches automatically. Explicit fast accepts one decision. Shared capacity limits can return 429." } }, required: ["inputs", "labels"], additionalProperties: false, @@ -189,7 +189,7 @@ export function productServer(classify: ClassifyFn): McpServer { inputSchema: { type: "object", required: ["items", "dimensions"], additionalProperties: false, properties: { items: INPUTS_SCHEMA, dimensions: DIMENSIONS_SCHEMA, instructions: { ...INSTRUCTIONS_SCHEMA, maxLength: 4000 }, tier: TIER_SCHEMA, - model: { type: "string", enum: ["jev", "laya", "kev"] }, processing: { type: "string", enum: ["fast", "bulk"], description: "Optional. Implies Laya if model is omitted; has no effect with explicit Jev. Omit for automatic fast/bulk selection based on item × dimension decisions." } }, + model: { type: "string", enum: ["jev", "laya", "kev", "chunklaya"] }, processing: { type: "string", enum: ["fast", "bulk"], description: "Optional. Implies Laya if model is omitted; has no effect with explicit Jev. Omit for automatic fast/bulk selection based on item × dimension decisions." } }, }, annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: false }, async run(a, ctx) { @@ -213,7 +213,7 @@ export function productServer(classify: ClassifyFn): McpServer { labels: LABELS_SCHEMA, instructions: INSTRUCTIONS_SCHEMA, max_labels: { type: "integer", minimum: 1, maximum: 100, description: "At most this many labels per text, most likely first." }, - model: { type: "string", enum: ["jev", "laya", "kev"] }, + model: { type: "string", enum: ["jev", "laya", "kev", "chunklaya"] }, processing: { type: "string", enum: ["fast", "bulk"], description: "Optional. Implies Laya if model is omitted; has no effect with explicit Jev. Omit to select fast for up to four labels on one text, or bulk for larger work automatically." }, }, required: ["inputs", "labels"], diff --git a/src/openapi.ts b/src/openapi.ts index 8f2cd98..37798f9 100644 --- a/src/openapi.ts +++ b/src/openapi.ts @@ -13,6 +13,7 @@ import { MAX_DESIRED_LATENCY_MS, MIN_DESIRED_LATENCY_MS, ROADMAP, ROADMAP_KEYS } export const ERROR_CODES = [ ...SPENDING_ERROR_CODES, "bad_model", "bad_processing", "laya_input", "laya_rate_limit", "laya_unavailable", + "chunklaya_input", "chunklaya_busy", "chunklaya_unavailable", // 400 "bad_dimensions", "too_many_decisions", "dimension_context_too_large", "bad_json", "no_input", "too_many_inputs", "too_few_labels", "too_many_labels", "empty_label", "duplicate_labels", "empty_input", "input_too_long", "bad_tier", "bad_cursor", "invalid_submission", "skill_invalid", "account_route_required", @@ -940,10 +941,10 @@ export const OPENAPI = { { inputs: ["postgres index tuning for ML feature stores"], labels: ["databases", "ml", "frontend"], multi: true, max_labels: 2 }, ], properties: { - model: { type: "string", enum: ["jev", "laya", "kev"], description: "When omitted, uses Jev unless processing is supplied, which implies Laya. Laya and Kev are experimental models hosted by Beam: \"laya\" is ModernBERT-large with a 512-token context, \"kev\" is Qwen2.5-0.5B with a pointer head and an 8,192-token context. Both take 2–16 short labels, text ≤2,000 characters and instructions ≤400 characters. A result is labelled jev/laya or jev/kev; Beam does not report which checkpoint answered. Jev calibration claims do not apply to either." }, + model: { type: "string", enum: ["jev", "laya", "kev", "chunklaya"], description: "When omitted, uses Jev unless processing is supplied, which implies Laya, or an input exceeds 32,000 characters, which routes the request to chunklaya, our long-document model (up to 4,000,000 characters and 20 inputs per request; the result is labelled chunklaya/multilingual; tier smart is not available). \"chunklaya\" selects it explicitly. Laya and Kev are experimental models hosted by Beam: \"laya\" is ModernBERT-large with a 512-token context, \"kev\" is Qwen2.5-0.5B with a pointer head and an 8,192-token context. Both take 2–16 short labels, text ≤2,000 characters and instructions ≤400 characters. A result is labelled jev/laya or jev/kev; Beam does not report which checkpoint answered. Jev calibration claims do not apply to either." }, processing: { type: "string", enum: ["fast", "bulk"], description: "Implies Laya when model is omitted. Accepted but has no effect with explicit model jev, which handles batching automatically. When omitted for Laya, automatically selects fast for one decision with up to 4 yes/no questions, otherwise bulk. Explicit Laya lanes are honored. Fast allows 60 questions/min and 2,000/day per caller. Bulk chunks batches up to 1,000 questions per call, 1,000/min and 20,000/day. These caps also apply to paid/operator keys. Same model weights in both lanes. Overload returns 429; a cold bulk worker returns 503 with Retry-After. Smart review is independent." }, dimensions: DIMENSIONS_SCHEMA, - items: { type: "array", minItems: 1, maxItems: 1000, items: { type: "string", minLength: 1, maxLength: 32000 }, description: "Alias for inputs in dimensions mode. Do not combine with input or inputs." }, + items: { type: "array", minItems: 1, maxItems: 1000, items: { type: "string", minLength: 1, maxLength: 4000000 }, description: "Alias for inputs in dimensions mode; each up to 32,000 characters, or 4,000,000 when routed to chunklaya. Do not combine with input or inputs." }, input: { type: "string", description: "A single text. Provide this or inputs; a string under `inputs` is read as one text too." }, inputs: { type: "array", @@ -1334,7 +1335,8 @@ are limited to 1 MB. Billing settles asynchronously after the response. Free, per IP, counted in classifications: 3,000/minute and 20,000/day on the fast tier, 200/minute and 2,000/day on the smart tier. Inputs cap at 32,000 -characters; free requests accept 1,000 inputs on fast or 200 on smart. +characters, or 4,000,000 for documents routed to chunklaya (at most 20 per +request); free requests accept 1,000 inputs on fast or 200 on smart. Pro workspaces get 10x minute and daily limits shared across keys and agents, and up to 1,000 inputs on either tier. Current plans are at https://classifier.dev/pricing. diff --git a/src/retail-rates.json b/src/retail-rates.json index 3a28a03..26b93b1 100644 --- a/src/retail-rates.json +++ b/src/retail-rates.json @@ -1,5 +1,5 @@ { - "version": "2026-09-21-spending-v2", + "version": "2026-09-23-chunklaya-v3", "models": [ { "provider": "typesafe", @@ -29,6 +29,13 @@ "outputUsdPerMillion": "0", "cachedInputUsdPerMillion": "0" }, + { + "provider": "chunklaya", + "model": "chunklaya/multilingual", + "inputUsdPerMillion": "0", + "outputUsdPerMillion": "0", + "cachedInputUsdPerMillion": "0" + }, { "provider": "openrouter", "model": "ibm-granite/granite-4.0-h-micro", diff --git a/src/server/token-pricing.ts b/src/server/token-pricing.ts index 8df3399..33d8adc 100644 --- a/src/server/token-pricing.ts +++ b/src/server/token-pricing.ts @@ -44,7 +44,7 @@ export function parseTokenRateCard(json: string | undefined): TokenRateCard | nu const seen = new Set(); const models = data.models.map(value => { const row = object(value); - if (!row || !["typesafe", "vercel", "openrouter", "beam"].includes(String(row.provider)) || typeof row.model !== "string" || !row.model.trim() || row.model.length > 200) { + if (!row || !["typesafe", "vercel", "openrouter", "beam", "chunklaya"].includes(String(row.provider)) || typeof row.model !== "string" || !row.model.trim() || row.model.length > 200) { throw new Error("Invalid provider or model in token rate card."); } const key = `${row.provider}:${row.model}`; diff --git a/src/server/token-reservation.ts b/src/server/token-reservation.ts index 309dee3..7764343 100644 --- a/src/server/token-reservation.ts +++ b/src/server/token-reservation.ts @@ -9,6 +9,7 @@ export function providerCallBound(card: TokenRateCard, provider: ModelTokenUsage const inputLimit = provider === "typesafe" && model === JEV_ACCOUNT_MODEL ? 65_536 : provider === "beam" && model === "jev/laya" ? 512 : provider === "beam" && model === "jev/kev" ? 8_192 + : provider === "chunklaya" && model === "chunklaya/multilingual" ? 1_048_576 : provider === "openrouter" && model === "google/gemini-3.8-flash" ? 1_048_576 : null; const rate = card.models.find((row) => row.provider === provider && row.model === model); if (inputLimit === null || !rate || !Number.isSafeInteger(maxOutput) || maxOutput < 0 || maxOutput > 65_536) diff --git a/src/spending/policy.ts b/src/spending/policy.ts index 05f1772..bf36be1 100644 --- a/src/spending/policy.ts +++ b/src/spending/policy.ts @@ -40,6 +40,8 @@ const models: Record "typesafe:jev-1.13.0": { context: 65536, input: 0.042, output: 0 }, "beam:jev/laya": { context: 512, input: 0.021, output: 0 }, "beam:jev/kev": { context: 8192, input: 0.021, output: 0 }, + // Our own pod, billed by the hour: no per-token provider price. The context is the service's document ceiling. + "chunklaya:chunklaya/multilingual": { context: 1048576, input: 0, output: 0 }, "openrouter:google/gemini-3.8-flash": { context: 1048576, input: 0.75, output: 3.75 }, "openrouter:ibm-granite/granite-4.0-h-micro": { context: 131000, input: 0.017, output: 0.112 }, "openrouter:deepseek/deepseek-v4-flash": { context: 1048576, input: 0.056, output: 0.111 }, diff --git a/test/chunklaya.test.ts b/test/chunklaya.test.ts new file mode 100644 index 0000000..cb9d3a1 --- /dev/null +++ b/test/chunklaya.test.ts @@ -0,0 +1,155 @@ +import { test, expect, afterEach } from "bun:test"; +import worker, { type Env } from "../src/index"; +import { CHUNKLAYA_BACKEND } from "../src/jev"; +import { parseTokenRateCard, priceTokens } from "../src/server/token-pricing"; +import { providerCallBound } from "../src/server/token-reservation"; +import rates from "../src/retail-rates.json"; + +/** + * The long-document route. chunklaya speaks System One at the pod's address; + * the Worker sends it any input over MAX_CHARS when it is configured, one + * document per request, and never answers such a request from another model. + */ +const original = globalThis.fetch; +afterEach(() => { globalThis.fetch = original; }); + +const POD = "https://pod.example"; +const env = { + CHUNKLAYA_ENABLED: "true", CHUNKLAYA_URL: POD + "/", CHUNKLAYA_TOKEN: "pod-bearer", TYPESAFE_API_KEY: "test", + STATS: { get: async () => null, put: async () => {} }, + LIMITER: { idFromName: (s: string) => s, get: () => ({ fetch: async () => Response.json({ limited: false, remaining: 59 }) }) }, +} as unknown as Env; +const ctx = { waitUntil: () => {} } as unknown as ExecutionContext; +const LONG = "The invoice was paid twice on Monday.\n\n".repeat(900); // 34,200 characters, over MAX_CHARS +const request = (body: object, bindings = env) => worker.fetch(new Request("https://classifier.dev/v1/classify", { + method: "POST", body: JSON.stringify(body) }), bindings, ctx); + +type Call = { url: string; auth: string | null; body: { model: string; state: { id: string; text: string }[]; questions: Record }> } }; + +/** A pod that answers every question, or refuses with the status and body given. */ +function mockPod(refuse?: { status: number; body?: unknown } | Error) { + const calls: Call[] = []; + globalThis.fetch = (async (url, init) => { + const body = JSON.parse(String(init?.body)) as Call["body"]; + calls.push({ url: String(url), auth: new Headers(init?.headers).get("authorization"), body }); + if (refuse instanceof Error) throw refuse; + if (refuse) return Response.json(refuse.body ?? { error: { code: "busy" } }, { status: refuse.status }); + const answers = Object.fromEntries(Object.entries(body.questions).map(([id, q]) => { + const labels = Object.keys(q.criteria ?? {}); + return [id, q.type === "noul" ? { noul: 0.8 } : { choice: labels[0], confidence: 0.91, + probabilities: Object.fromEntries(labels.map((label, i) => [label, i ? 0.09 / (labels.length - 1) : 0.91])) }]; + })); + return Response.json({ model: "chunklaya/multilingual", answers, usage: { input_tokens: 4200, output_tokens: 0 } }); + }) as typeof fetch; + return calls; +} + +test("an input over 32,000 characters is answered by chunklaya, one document per request, with the pod's bearer", async () => { + const calls = mockPod(); + const response = await request({ input: LONG, labels: ["billing", "technical"] }); + expect(response.status).toBe(200); + const body = await response.json() as { results: { label: string; confidence: number; scores: Record; model: string }[] }; + expect(body.results[0]).toMatchObject({ label: "billing", confidence: 0.91, model: "chunklaya/multilingual" }); + expect(body.results[0].scores).toEqual({ billing: 0.91, technical: 0.09 }); + expect(calls).toHaveLength(1); + expect(calls[0].url).toBe(POD + "/v1/systemone"); + expect(calls[0].auth).toBe("Bearer pod-bearer"); + expect(calls[0].body.model).toBe(CHUNKLAYA_BACKEND.model); + expect(calls[0].body.state).toHaveLength(1); + expect(calls[0].body.state[0].text).toBe(LONG); + // Tier limits, not Laya lane limits, and no lane header. + expect(response.headers.get("ratelimit-limit")).toBe("3000"); + expect(response.headers.get("x-classifier-processing")).toBeNull(); +}); + +test("without the service configured, or under an explicit jev, a long input is still input_too_long", async () => { + const calls = mockPod(); + for (const bindings of [{ ...env, CHUNKLAYA_URL: undefined }, { ...env, CHUNKLAYA_TOKEN: undefined }, { ...env, CHUNKLAYA_ENABLED: "false" }]) { + const response = await request({ input: LONG, labels: ["a", "b"] }, bindings as Env); + expect(response.status).toBe(400); + const body = await response.json() as { code: string; error: string }; + expect(body.code).toBe("input_too_long"); + expect(body.error).toContain("at most 32,000"); + } + const explicit = await request({ model: "jev", input: LONG, labels: ["a", "b"] }); + expect(explicit.status).toBe(400); + expect((await explicit.json() as { code: string }).code).toBe("input_too_long"); + expect(calls).toHaveLength(0); +}); + +test("an explicit model selects chunklaya for short text too, and cannot be used without the service", async () => { + const calls = mockPod(); + const response = await request({ model: "chunklaya", input: "Please refund this charge", labels: ["billing", "technical"] }); + expect(response.status).toBe(200); + expect(calls).toHaveLength(1); + const missing = await request({ model: "chunklaya", input: "short", labels: ["a", "b"] }, { ...env, CHUNKLAYA_URL: undefined } as Env); + expect(missing.status).toBe(503); + expect((await missing.json() as { code: string }).code).toBe("chunklaya_unavailable"); +}); + +test("the long-document ceilings: characters, inputs per request, and no smart tier", async () => { + const calls = mockPod(); + const huge = await request({ input: "x".repeat(4_000_001), labels: ["a", "b"] }); + expect(huge.status).toBe(400); + expect(await huge.json()).toMatchObject({ code: "input_too_long", error: expect.stringContaining("4,000,000") }); + const many = await request({ inputs: Array.from({ length: 21 }, () => LONG), labels: ["a", "b"] }); + expect(many.status).toBe(400); + expect((await many.json() as { code: string }).code).toBe("too_many_inputs"); + const smart = await request({ input: LONG, labels: ["a", "b"], tier: "smart" }); + expect(smart.status).toBe(400); + expect(await smart.json()).toMatchObject({ code: "bad_tier", error: expect.stringContaining("smart") }); + expect(calls).toHaveLength(0); +}); + +test("multi-label asks one noul per label and reads the scores back", async () => { + const calls = mockPod(); + const response = await request({ input: LONG, labels: ["billing", "technical", "legal"], multi: true }); + expect(response.status).toBe(200); + const body = await response.json() as { results: { labels: string[]; scores: Record; model: string }[] }; + expect(body.results[0].labels).toEqual(["billing", "technical", "legal"]); + expect(body.results[0].scores).toEqual({ billing: 0.8, technical: 0.8, legal: 0.8 }); + expect(Object.values(calls[0].body.questions).map((q) => q.type)).toEqual(["noul", "noul", "noul"]); +}); + +test("dimensions send every question about a document in one request, one request per document", async () => { + const calls = mockPod(); + const response = await request({ items: [LONG, LONG + " second"], dimensions: { kind: ["lease", "employment"], renews: ["yes", "no"] } }); + expect(response.status).toBe(200); + const body = await response.json() as { results: { dimensions: Record }[] }; + expect(body.results).toHaveLength(2); + expect(body.results[1].dimensions.kind).toMatchObject({ label: "lease", model: "chunklaya/multilingual" }); + expect(body.results[1].dimensions.renews.label).toBe("yes"); + expect(calls).toHaveLength(2); + for (const call of calls) { + expect(call.body.state).toHaveLength(1); + expect(Object.keys(call.body.questions)).toHaveLength(2); + } +}); + +test("the service's refusals keep their meaning and never fall back to another model", async () => { + let calls = mockPod({ status: 422, body: { error: { code: "too_many_passages", type: "too_many_passages" }, detail: "the document has 6500 passages and scan scores at most 256; use strategy \"locate\"" } }); + let response = await request({ input: LONG, labels: ["a", "b"] }); + expect(response.status).toBe(400); + const refused = await response.json() as { code: string; error: string }; + expect(refused.code).toBe("chunklaya_input"); + expect(refused.error).not.toContain("6500"); // our words, not the service's + expect(calls.every((c) => c.url.startsWith(POD))).toBe(true); + + calls = mockPod({ status: 429 }); + response = await request({ input: LONG, labels: ["a", "b"] }); + expect(response.status).toBe(429); + expect((await response.json() as { code: string }).code).toBe("chunklaya_busy"); + expect(response.headers.get("retry-after")).toBe("2"); + + calls = mockPod(new Error("connect ECONNREFUSED")); + response = await request({ input: LONG, labels: ["a", "b"] }); + expect(response.status).toBe(503); + expect((await response.json() as { code: string }).code).toBe("chunklaya_unavailable"); + expect(calls).toHaveLength(1); // one attempt, no retry, no OpenRouter fallback +}); + +test("account billing prices chunklaya at zero and bounds a call by its document ceiling", () => { + const card = parseTokenRateCard(JSON.stringify(rates))!; + expect(providerCallBound(card, "chunklaya", "chunklaya/multilingual", 0)).toBe(0); + expect(priceTokens(card, [{ provider: "chunklaya", model: "chunklaya/multilingual", calls: 1, inputTokens: 250_000, outputTokens: 0, cachedInputTokens: 0 }])?.nanodollars).toBe(0n); +}); diff --git a/wrangler.example.toml b/wrangler.example.toml index b1bc9a4..dddc134 100644 --- a/wrangler.example.toml +++ b/wrangler.example.toml @@ -67,6 +67,12 @@ SPUR_MONTHLY_LOOKUPS = "45000" # deployment of our own to point at. BEAM_API_KEY is a Worker secret, never a # plaintext var, and it is the only credential either model needs. LAYA_ENABLED = "true" +# Documents over 32,000 characters go to chunklaya, our own long-document +# service (github.com/myxamediyar/chunklaya, serve/), when CHUNKLAYA_URL (the +# pod's proxy address) and CHUNKLAYA_TOKEN (its bearer) are set as Worker +# secrets. Without them such inputs keep answering 400 input_too_long, so this +# flag is safe to leave on. +CHUNKLAYA_ENABLED = "true" # Enable only after account migration, pricing and Autumn reconciliation checks. APP_ACCOUNTS_ENABLED = "true" # Gateway free-tier 429s observed in production. Re-enable only after capacity is verified.