Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions .claude-plugin/marketplace.json
Original file line number Diff line number Diff line change
Expand Up @@ -8,14 +8,14 @@
},
"metadata": {
"description": "Plugins from classifier.dev: zero-shot text classification with a calibrated confidence per answer, no API key.",
"version": "1.0.0"
"version": "1.0.1"
},
"plugins": [
{
"name": "classifier",
"source": "./plugins/classifier",
"description": "Sort up to 1,000 texts into your own labels in one call, with a calibrated confidence per answer. No API key.",
"version": "1.0.0",
"version": "1.0.1",
"author": { "name": "Michael Ryaboy", "email": "contact@classifier.dev", "url": "https://classifier.dev" },
"homepage": "https://classifier.dev/mcp-setup",
"repository": "https://github.com/mrmps/classifier-dev",
Expand Down
11 changes: 9 additions & 2 deletions e2e/spending.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@ import assert from 'node:assert/strict';
const report = { runtime: 'workerd + SQLite Durable Objects', upstream: 'deterministic HTTP provider fixtures; no live inference', results: [] };
await mkdir('captures', { recursive: true });

let calls = 0, lookups = 0, release;
let calls = 0, lookups = 0, release, jevUnavailable = false;
let barrier = Promise.resolve();
const modules = (await readdir('dist/server', { recursive: true })).filter(p => p.endsWith('.js')).sort((a, b) => a === 'index.js' ? -1 : b === 'index.js' ? 1 : a.localeCompare(b)).map(p => ({ type: 'ESModule', path: `dist/server/${p}` }));
const mf = new Miniflare(convertV4MiniflareOptions({ workers: [{ name: "spending", modules, modulesRoot: 'dist/server', compatibilityDate: '2026-08-01', compatibilityFlags: ['nodejs_compat'],
Expand All @@ -15,7 +15,8 @@ const mf = new Miniflare(convertV4MiniflareOptions({ workers: [{ name: "spending
if (request.url.startsWith('https://api.spur.us/')) { lookups++; return WorkerResponse.json({}); }
calls++; await barrier;
const body = await request.json();
if (request.url.includes('openrouter.ai')) return WorkerResponse.json({ model: body.model, usage: { cost: 0.0007125, prompt_tokens: 200, completion_tokens: 150, prompt_tokens_details: { cached_tokens: 0 } }, choices: [{ message: { content: 'A' } }] });
if (jevUnavailable && request.url.includes('typesafe.ai')) return WorkerResponse.json({ detail: { error_type: 'insufficient_credits' } }, { status: 402 });
if (request.url.includes('openrouter.ai')) return WorkerResponse.json({ model: body.model, usage: { cost: jevUnavailable ? 0.000001812 : 0.0007125, prompt_tokens: jevUnavailable ? 100 : 200, completion_tokens: jevUnavailable ? 1 : 150, prompt_tokens_details: { cached_tokens: 0 } }, choices: [{ message: { content: 'A' } }] });
const answers = Object.fromEntries(Object.entries(body.questions).map(([id, q]) => { const keys = Object.keys(q.criteria); return [id, { choice: keys[0], confidence: 0.51, probabilities: Object.fromEntries(keys.map((k, i) => [k, i ? 0.49 : 0.51])) }]; }));
return WorkerResponse.json({ model: 'jev-1.13.0', usage: { input_tokens: 100, output_tokens: 0 }, answers });
},
Expand Down Expand Up @@ -45,6 +46,12 @@ try {
const duplicate = await request('203.0.113.4', undefined, undefined, { 'idempotency-key': 'one' });
assert.equal(duplicate.status, 409);
report.results.push({ name: 'durable idempotency', first: first.status, duplicate: duplicate.status });
jevUnavailable = true;
const recovered = await request('203.0.113.5', { inputs: Array(119).fill('Invoice'), labels: ['billing', 'support'] });
assert.equal(recovered.status, 200);
const recoveredBody = await recovered.json();
assert.equal(recoveredBody.results.length, 119);
report.results.push({ name: '119-item batch during primary outage', status: recovered.status, results: recoveredBody.results.length });
await writeFile('captures/spending-e2e.json', JSON.stringify(report, null, 2));
console.log(JSON.stringify(report, null, 2));
} finally { await mf.dispose(); }
2 changes: 1 addition & 1 deletion plugins/classifier/.claude-plugin/plugin.json
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
{
"$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
"name": "classifier",
"version": "1.0.0",
"version": "1.0.1",
"description": "Sort up to 1,000 texts into your own labels in one call, with a calibrated confidence per answer. No API key.",
"author": {
"name": "Michael Ryaboy",
Expand Down
2 changes: 1 addition & 1 deletion plugins/classifier/.codex-plugin/plugin.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"name": "classifier",
"version": "1.0.0",
"version": "1.0.1",
"description": "Sort up to 1,000 texts into your own labels in one call, with a calibrated confidence per answer. No API key.",
"author": {
"name": "Michael Ryaboy",
Expand Down
14 changes: 7 additions & 7 deletions plugins/classifier/skills/bulk-classify/SKILL.md
Original file line number Diff line number Diff line change
Expand Up @@ -73,7 +73,7 @@ npm i -g classifier-dev
classify bug,feature,praise < feedback.txt # label<TAB>confidence<TAB>text, input order
classify relevant,"not relevant" --review 0.7 < snippets.txt # only the unsure ones
classify db,web,ml --count < titles.txt # a histogram instead of rows
classify a,b --json < items.txt | jq -c 'select(.confidence < 0.8)'
classify a,b --json < items.txt | jq -c 'select(.confidence == null or .confidence < 0.8)'
```

It batches a thousand inputs per request, four requests at a time, and streams
Expand Down Expand Up @@ -185,8 +185,6 @@ def keep_relevant(question, snippets):
data=body,
headers={
"content-type": "application/json",
# Send a real User-Agent. Python's stdlib default is a known-bot
# signature and gets a 403 at the edge before it reaches the API.
"user-agent": "my-agent/1.0",
},
)
Expand All @@ -203,10 +201,12 @@ Then read only what comes back. The snippets you dropped never enter context.
matters more than precision here. The confidence gate above does that
directly; "When in doubt, keep it" in the instructions also measurably helps.

**Always set a `User-Agent`.** Most clients (curl, node, bun, requests, Go, axios)
send a usable one already, but Python's `urllib` default is blocked at the edge
and returns `403` before your request is ever classified. If you get a 403,
this is why. Rate limiting returns `429`.
Python's standard `urllib`, curl and Node fetch work without a custom
`User-Agent`. A descriptive agent name is optional. For a JSON error, read
`code`, `action` and `retryable`: a 403 can mean the free service detected an
anonymous proxy network, which requires a funded workspace key. A 429 carries
`Retry-After`. An HTML error is an edge/network failure; report its status and
request ID rather than assuming classification ran.

## Report a problem with classifier.dev

Expand Down
14 changes: 7 additions & 7 deletions src/SKILL.md
Original file line number Diff line number Diff line change
Expand Up @@ -73,7 +73,7 @@ npm i -g classifier-dev
classify bug,feature,praise < feedback.txt # label<TAB>confidence<TAB>text, input order
classify relevant,"not relevant" --review 0.7 < snippets.txt # only the unsure ones
classify db,web,ml --count < titles.txt # a histogram instead of rows
classify a,b --json < items.txt | jq -c 'select(.confidence < 0.8)'
classify a,b --json < items.txt | jq -c 'select(.confidence == null or .confidence < 0.8)'
```

It batches a thousand inputs per request, four requests at a time, and streams
Expand Down Expand Up @@ -185,8 +185,6 @@ def keep_relevant(question, snippets):
data=body,
headers={
"content-type": "application/json",
# Send a real User-Agent. Python's stdlib default is a known-bot
# signature and gets a 403 at the edge before it reaches the API.
"user-agent": "my-agent/1.0",
},
)
Expand All @@ -203,10 +201,12 @@ Then read only what comes back. The snippets you dropped never enter context.
matters more than precision here. The confidence gate above does that
directly; "When in doubt, keep it" in the instructions also measurably helps.

**Always set a `User-Agent`.** Most clients (curl, node, bun, requests, Go, axios)
send a usable one already, but Python's `urllib` default is blocked at the edge
and returns `403` before your request is ever classified. If you get a 403,
this is why. Rate limiting returns `429`.
Python's standard `urllib`, curl and Node fetch work without a custom
`User-Agent`. A descriptive agent name is optional. For a JSON error, read
`code`, `action` and `retryable`: a 403 can mean the free service detected an
anonymous proxy network, which requires a funded workspace key. A 429 carries
`Retry-After`. An HTML error is an edge/network failure; report its status and
request ID rather than assuming classification ran.

## Report a problem with classifier.dev

Expand Down
2 changes: 1 addition & 1 deletion src/alerts.ts
Original file line number Diff line number Diff line change
Expand Up @@ -257,7 +257,7 @@ export async function evaluate(env: Env): Promise<{ alerts: Alert[]; checked: bo
alerts.push({
id: "dimensions_fallback", severity: "warning",
title: "Multidimensional classification is using the LLM fallback",
detail: `${dimensionFallback} fields used the fallback in the last ${WINDOW_MIN}m. Check Jev availability; requests above 20 decisions cannot use this fallback.`,
detail: `${dimensionFallback} fields used the fallback in the last ${WINDOW_MIN}m. Check Jev availability. Fallback uses bounded concurrency and the request spending allowance.`,
});
}

Expand Down
11 changes: 6 additions & 5 deletions src/docs.ts
Original file line number Diff line number Diff line change
Expand Up @@ -314,8 +314,8 @@ MULTIPLE DIMENSIONS
Confidence and scores can also be null when the provider returns no score.

usage reports items, dimensions, classifications (decisions), escalated,
fallback, and ms. If Jev is unavailable, the LLM fallback accepts at most
20 decisions; larger requests return 502 batch_unavailable. A failed field
fallback, and ms. If Jev is unavailable, fallback processes the batch with
bounded concurrency inside the same request spending allowance. A failed field
fails the whole request rather than returning an incomplete matrix.


Expand Down Expand Up @@ -428,7 +428,7 @@ TIERS

The models are not fixed. They are benchmarked as candidates appear and
swapped when a measurement, not a launch post, says to. If the decision
model is unavailable, requests of up to twenty inputs fall back to a chain of
model is unavailable, requests fall back within their spending allowance to a chain of
language models on different providers; JSON responses always report which
model actually answered.

Expand Down Expand Up @@ -478,8 +478,9 @@ ERRORS
the limit (https://classifier.dev/pricing)
502 typesafe or typesafe_<status> when the decision model failed;
openrouter_<status>, chain_exhausted or timeout when the fallback
chain did; batch_unavailable for more than twenty inputs while the
decision model is down; upstream_other. Retry with backoff.
chain did; upstream_other. Retry with backoff.
402 request_spending_limit: send fewer or shorter inputs, or use a funded
workspace key. Do not repeatedly retry an unchanged over-budget request.

The full list, in the shape a client can validate against, is
components.schemas.Error in https://classifier.dev/openapi.json
Expand Down
2 changes: 1 addition & 1 deletion src/home.ts
Original file line number Diff line number Diff line change
Expand Up @@ -619,7 +619,7 @@ const JSON_LD = () => {
{
"@type": "Question",
name: "When should an agent call classifier.dev instead of classifying text itself?",
acceptedAnswer: { "@type": "Answer", text: "When reading the input is the expensive part: filtering search results before opening them, bucketing logs or tickets, routing a pipeline branch deterministically. One request classifies up to 1,000 texts in about a second. Under about five items you can already see, just decide yourself." },
acceptedAnswer: { "@type": "Answer", text: "When reading the input is the expensive part: filtering search results before opening them, bucketing logs or tickets, routing a pipeline branch by label. One request classifies up to 1,000 texts in about a second. Under about five items you can already see, just decide yourself." },
},
{
"@type": "Question",
Expand Down
26 changes: 19 additions & 7 deletions src/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -651,6 +651,7 @@ async function callModel(
signal: AbortSignal.timeout(cfg.reasoning ? 60_000 : 15_000),
});
} catch (e) {
if (e instanceof SpendingError) throw e;
const timeout = e instanceof Error && (e.name === "AbortError" || e.name === "TimeoutError");
last = timeout ? "upstream timeout" : "upstream network failure";
if (attempt < 2) await new Promise((r) => setTimeout(r, 300 * 2 ** attempt + Math.random() * 200));
Expand Down Expand Up @@ -742,6 +743,7 @@ async function runChain(
try {
return await callModel(env, cfg, input, labels, instructions, multi, meter);
} catch (e) {
if (e instanceof SpendingError) throw e;
last = e;
}
}
Expand Down Expand Up @@ -959,7 +961,7 @@ async function classifyMany(
jev = await (layaRun ?? startLaya(env, layaPlan, meter, layaTiming));
} else jev = await jevClassify(keys!, inputs, labels, instructions, !!multi, meter);
} catch (e) {
if (layaPlan) throw e;
if (layaPlan || e instanceof SpendingError || meter?.permit?.error) throw meter?.permit?.error ?? e;
console.warn(`jev failed, falling back: ${(e as Error).message}`);
}
if (jev) {
Expand Down Expand Up @@ -994,7 +996,7 @@ async function classifyMany(
return { results, escalationFailed };
}
}
if (inputs.length > FALLBACK_MAX_INPUTS) {
if (!meter?.permit && inputs.length > FALLBACK_MAX_INPUTS) {
throw new Error(`batch classification is temporarily unavailable; send up to ${FALLBACK_MAX_INPUTS} inputs or retry shortly`);
}
return { results: await llmClassifyMany(env, inputs, labels, tier, instructions, multi, meter), escalationFailed: 0 };
Expand All @@ -1010,9 +1012,12 @@ async function classifyMatrix(env: Env, inputs: string[], dimensions: Dimension[
jev = inputs.map((_, i) => flat.slice(i * dimensions.length, (i + 1) * dimensions.length));
} else if (keys) {
try { jev = await classifyDimensions(keys, batches, meter); }
catch (e) { console.warn(`dimensions Jev failed: ${(e as Error).message}`); }
catch (e) {
if (e instanceof SpendingError || meter.permit?.error) throw meter.permit?.error ?? e;
console.warn(`dimensions Jev failed: ${(e as Error).message}`);
}
}
if (!jev && inputs.length * dimensions.length > FALLBACK_MAX_INPUTS) {
if (!jev && !meter.permit && inputs.length * dimensions.length > FALLBACK_MAX_INPUTS) {
throw new Error(`batch classification is temporarily unavailable; send up to ${FALLBACK_MAX_INPUTS} decisions or retry shortly`);
}
const results: Result[][] = inputs.map(() => []);
Expand Down Expand Up @@ -1873,8 +1878,8 @@ const worker = {

// Every API answer, success or not, says which version answered, how much
// room is left (IETF RateLimit header fields), and echoes an idempotency
// key if the caller sent one — classification has no side effects, so the
// echo is all a retrying client needs.
// key if the caller sent one. Spending admission rejects duplicate work;
// responses are not cached for replay.
const apiHeaders = (remaining = -1): Record<string, string> => {
const isLaya = selectedModel !== "jev";
const limit = isLaya ? Math.min(LAYA_LIMITS[processing].rpm, TIERS[tier].rpm * multiplier) : TIERS[tier].rpm * multiplier;
Expand All @@ -1893,7 +1898,7 @@ const worker = {
if (idem) h["idempotency-key"] = idem.slice(0, 255);
return h;
};
const fail = (msg: string, status: number, reason: ErrorCode, extra: Record<string, string> = {}, ms = 0, remaining = -1, more: Record<string, string> = {}) => {
const fail = (msg: string, status: number, reason: ErrorCode, extra: Record<string, string> = {}, ms = 0, remaining = -1, more: Record<string, unknown> = {}) => {
record(env, ctx, { tier, n: 0, ms, labels, ip, country, status, client, model: selectedModel === "jev" ? "" : layaModel(selectedModel),
usd: meter.usd, reason, agent, attempted: inputs.length, escalationFailed: 0, mode, dimensions: dimensions?.length ?? 0 });
const headers = { ...apiHeaders(remaining), ...extra };
Expand Down Expand Up @@ -2128,7 +2133,14 @@ const worker = {
escalationFailed = r.escalationFailed;
fallbackDecisions = r.fallbackDecisions;
} else ({ results, escalationFailed } = await classifyMany(env, inputs, labels, tier, instructions, multi, meter, layaPlan, layaTiming, layaRun));
if (meter.permit?.error) throw meter.permit.error;
} catch (e) {
const spending = e instanceof SpendingError ? e : meter.permit?.error;
if (spending) {
const response = errorResponse(spending);
return fail(spending.message, spending.status, spending.code as ErrorCode,
Object.fromEntries(response.headers), Date.now() - started, -1, await response.json() as Record<string, unknown>);
}
if (e instanceof LayaError) return fail(e.message, e.status,
e.status === 400 ? "laya_input" : e.status === 429 ? "laya_rate_limit" : "laya_unavailable",
e.status === 400 ? {} : { "retry-after": String(e.retryAfter) }, Date.now() - started);
Expand Down
3 changes: 3 additions & 0 deletions src/jev.ts
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
import { providerFetch } from "./spending/permit";
import { SpendingError } from "./spending/policy";
/**
* TypeSafe's Jev, the model behind both tiers.
*
Expand Down Expand Up @@ -515,6 +516,7 @@ async function postBeam(key: string, body: JevBody, backend: Backend, meter?: Me
signal: signal ? AbortSignal.any([signal, AbortSignal.timeout(UPSTREAM_TIMEOUT_MS)]) : AbortSignal.timeout(UPSTREAM_TIMEOUT_MS),
});
} catch (e) {
if (e instanceof SpendingError) throw e;
const timeout = e instanceof Error && (e.name === "AbortError" || e.name === "TimeoutError");
observe(timeout ? 504 : 0, timeout ? "timeout" : "network");
last = new JevError(`beam ${timeout ? "timeout" : "network failure"}`, timeout ? 504 : 0, timeout ? "timeout" : "network");
Expand Down Expand Up @@ -584,6 +586,7 @@ async function postTypesafe(key: string, body: JevBody, meter?: Meter, analytics
signal: AbortSignal.timeout(UPSTREAM_TIMEOUT_MS),
});
} catch (e) {
if (e instanceof SpendingError) throw e;
const timeout = e instanceof Error && (e.name === "AbortError" || e.name === "TimeoutError");
observe(timeout ? 504 : 0, timeout ? "timeout" : "network");
last = new JevError(`typesafe ${timeout ? "timeout" : "network failure"}`, timeout ? 504 : 0, timeout ? "timeout" : "network");
Expand Down
6 changes: 3 additions & 3 deletions src/openapi.ts
Original file line number Diff line number Diff line change
Expand Up @@ -124,7 +124,7 @@ const errors = (plain: boolean) => ({
"Retry-After": { schema: { type: "integer" }, description: "Seconds until the window resets." },
...RATE_LIMIT_HEADERS,
}, plain),
"502": err("The model provider failed after retries; retry with backoff. `code` is typesafe_<status> or typesafe (the decision model), openrouter_<status>, chain_exhausted or timeout (the fallback chain), batch_unavailable (more than 20 inputs while the decision model is down) or upstream_other.", RATE_LIMIT_HEADERS, plain),
"502": err("The model provider failed after retries; retry with backoff. `code` is typesafe_<status> or typesafe (the decision model), openrouter_<status>, chain_exhausted or timeout (the fallback chain), batch_unavailable or upstream_other. Fallback remains bounded by the request spending allowance.", RATE_LIMIT_HEADERS, plain),
default: err("Any other error, same {error, code} shape.", undefined, plain),
});
const ERRORS = errors(false);
Expand Down Expand Up @@ -1291,8 +1291,8 @@ Each results[i].dimensions[name] has a label, confidence, scores and model.
Up to 20 dimensions and 1,000 item × dimension decisions; each decision counts
against the quota. Each dimension may instead be {"labels":[...],"instructions":"..."}.
Do not combine dimensions with labels, multi or max_labels. Smart escalation is
per field; escalated fields have null confidence and scores. LLM fallback is
limited to 20 decisions; larger requests return 502 if Jev is unavailable.
per field; escalated fields have null confidence and scores. Fallback processes
the batch with bounded concurrency inside the same request spending allowance.

## Multi-label

Expand Down
Loading
Loading