Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 8 additions & 1 deletion .github/workflows/deploy.yml
Original file line number Diff line number Diff line change
Expand Up @@ -118,15 +118,22 @@ jobs:
DATABASE_URL: ${{ secrets.DATABASE_URL }}
run: bun scripts/check-newsletter-cutover.ts --activate

# The chunklaya pair is optional and travels together: with both set as
# repository secrets the Worker routes long documents there; with either
# missing they are left out of the file, so an existing value survives
# and an unconfigured Worker keeps answering input_too_long.
- name: Deploy
env:
CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }}
DATABASE_URL: ${{ secrets.DATABASE_URL }}
CHUNKLAYA_URL: ${{ secrets.CHUNKLAYA_URL }}
CHUNKLAYA_TOKEN: ${{ secrets.CHUNKLAYA_TOKEN }}
run: |
node --input-type=module -e '
import { writeFileSync } from "node:fs";
const { DATABASE_URL, CHUNKLAYA_URL, CHUNKLAYA_TOKEN } = process.env;
writeFileSync(process.env.RUNNER_TEMP + "/classifier-secrets.json",
JSON.stringify({ DATABASE_URL: process.env.DATABASE_URL }), { mode: 0o600 });
JSON.stringify({ DATABASE_URL, ...(CHUNKLAYA_URL && CHUNKLAYA_TOKEN ? { CHUNKLAYA_URL, CHUNKLAYA_TOKEN } : {}) }), { mode: 0o600 });
'
npx wrangler deploy --secrets-file "$RUNNER_TEMP/classifier-secrets.json"

Expand Down
10 changes: 10 additions & 0 deletions AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,16 @@ the plain text (`curl classifier.dev`), the HTML and the Markdown never drift.
oversized context rather than truncating, so a context refusal is
translated to `max_tokens_exceeded` and the batch halves and retries.
Unlike Jev, a Beam request is never retried: a lane quota counts attempts.
- chunklaya (`chunklaya/multilingual`) is our own long-document service: Laya
behind a chunk-and-index harness, github.com/myxamediyar/chunklaya under
`serve/`, on a RunPod pod. It speaks System One too, so it is a fourth
transport in `src/jev.ts`, reached through `CHUNKLAYA_URL` and
`CHUNKLAYA_TOKEN` (Worker secrets). The Worker sends an input over
`MAX_CHARS` there when `CHUNKLAYA_ENABLED` is `"true"` and both secrets are
set; otherwise such inputs stay `input_too_long`. One document is one
request; the service refuses rather than truncates, its 4xx become
`chunklaya_input`, and there is no fallback to Jev or the LLM chain. It is
billed by the hour, so no per-token provider cost is metered.
- The updates roadmap is one constant, `ROADMAP` in `src/newsletter.ts`; the plain
text, the signup form and the Markdown all render from it. Addresses go to the
`subscriber` table in the shared application Neon database. Preserve consent,
Expand Down
8 changes: 5 additions & 3 deletions cli/classify.js
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,7 @@ OPTIONS
-k, --max <n> at most n labels (implies --multi)
-s, --smart re-ask uncertain answers of a reasoning model (slower)
--model laya opt into the automatically routed Laya trial (default: jev)
--model chunklaya the long-document model; also automatic past 32,000 characters
--processing bulk Laya bulk lane; default fast is one decision per call
-i, --instructions <text> extra criteria: "judge only the service, ignore the food"
-r, --review <t> print only inputs with confidence below t
Expand Down Expand Up @@ -121,8 +122,8 @@ function parseArgs(argv) {
if (!(Number.isInteger(o.max) && o.max > 0)) fail("--max takes a whole number above 0, e.g. --max 3");
o.multi = true; // the API reads max_labels as multi-label; a single label cannot be capped
}
if (!["jev", "laya", "kev"].includes(o.model)) fail("--model must be jev, laya or kev");
if (!["fast", "bulk"].includes(o.processing) || o.model === "jev" && o.processing !== "fast") fail("--processing bulk requires --model laya or --model kev");
if (!["jev", "laya", "kev", "chunklaya"].includes(o.model)) fail("--model must be jev, laya, kev or chunklaya");
if (!["fast", "bulk"].includes(o.processing) || (o.model === "jev" || o.model === "chunklaya") && o.processing !== "fast") fail("--processing bulk requires --model laya or --model kev");
return o;
}

Expand Down Expand Up @@ -175,7 +176,7 @@ async function readStdin() {

async function post(o, inputs) {
const body = { inputs, labels: o.labels };
if (o.model !== "jev") { body.model = o.model; body.processing = o.processing; }
if (o.model !== "jev") { body.model = o.model; if (o.model !== "chunklaya") body.processing = o.processing; }
if (o.multi) body.multi = true;
if (o.max) body.max_labels = o.max;
if (o.smart) body.tier = "smart";
Expand Down Expand Up @@ -244,6 +245,7 @@ function configuredBatch() {
}

function batchSize(o) {
if (o.model === "chunklaya") return Math.min(configuredBatch(), 20); // one document is one upstream request there
if (o.model !== "jev") return Math.min(configuredBatch(), o.processing === "fast" ? 1 : Math.floor(1000 / (o.multi ? o.labels.length : 1)), o.smart ? SMART_BATCH : MAX_BATCH);
return Math.min(configuredBatch(), o.smart && !o.apiKey ? SMART_BATCH : MAX_BATCH);
}
Expand Down
2 changes: 1 addition & 1 deletion src/cost.ts
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,7 @@ export type TokenCounts = {
};

export type ModelTokenUsage = TokenCounts & {
provider: "typesafe" | "vercel" | "openrouter" | "beam";
provider: "typesafe" | "vercel" | "openrouter" | "beam" | "chunklaya";
model: string;
calls: number;
};
Expand Down
12 changes: 8 additions & 4 deletions src/dimensions.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,9 @@ import {
type JevQuestionGroup,
type JevResult,
type Question,
type Limits,
type Backend,
JEV_BACKEND,
} from "./jev";
import type { Meter } from "./cost";

Expand Down Expand Up @@ -57,7 +60,8 @@ type Cell = { item: number; dimension: number; id: string; question: Extract<Que
export type DimensionBatch = JevBatch<Cell>;

/** State is shared across questions. Respect BOTH Jev context limits, with headroom. */
export function packDimensions(inputs: string[], dimensions: Dimension[], shared?: string): DimensionBatch[] {
/** `limits` packs for another backend; chunklaya takes one document per request and has no token budget to respect. */
export function packDimensions(inputs: string[], dimensions: Dimension[], shared?: string, limits?: Limits): DimensionBatch[] {
const groups: JevQuestionGroup<Cell>[] = [];
inputs.forEach((text, item) => dimensions.forEach((d, dimension) => {
const question: Question = {
Expand All @@ -69,7 +73,7 @@ export function packDimensions(inputs: string[], dimensions: Dimension[], shared
groups.push({ state: { id: `i${item}`, text }, questions: { [cell.id]: question }, value: cell });
}));
try {
return prepareJevBatches(groups, { rejectOversized: true });
return prepareJevBatches(groups, { rejectOversized: !limits, limits });
} catch (error) {
if (error instanceof JevContextError) {
throw new DimensionError("An input and dimension exceed Jev's context budget; shorten the input or dimension instructions");
Expand All @@ -79,9 +83,9 @@ export function packDimensions(inputs: string[], dimensions: Dimension[], shared
}

/** One result per matrix cell. Splitting by question also handles a single wide item. */
export async function classifyDimensions(keys: JevKeys, batches: DimensionBatch[], meter?: Meter): Promise<JevResult[][]> {
export async function classifyDimensions(keys: JevKeys, batches: DimensionBatch[], meter?: Meter, backend: Backend = JEV_BACKEND): Promise<JevResult[][]> {
const results: JevResult[][] = [];
for (const { value: cell, model, answers } of await runJevBatches(keys, batches, meter)) {
for (const { value: cell, model, answers } of await runJevBatches(keys, batches, meter, backend)) {
const answer = answers[cell.id]; // The shared runner validates every answer before returning.
(results[cell.item] ??= [])[cell.dimension] = {
label: answer.choice!, confidence: answer.confidence!,
Expand Down
38 changes: 35 additions & 3 deletions src/docs.ts
Original file line number Diff line number Diff line change
Expand Up @@ -140,6 +140,35 @@ LAYA AND KEV
node cli/classify.js billing,technical --model kev --processing bulk < tickets.txt


LONG DOCUMENTS

An input over 32,000 characters is answered by chunklaya, our own
long-document service, when it is configured; otherwise it is refused with
input_too_long as before. chunklaya is Laya behind a harness: the document
is split into passages and indexed once, and every question in the request
is answered off that index, so a request with dimensions asks all of them
in one pass. Up to 4,000,000 characters per input and 20 inputs per
request. The result is labelled chunklaya/multilingual. Nothing at or under
32,000 characters changes, and an explicit model: "jev" keeps the 32,000
ceiling; model: "chunklaya" selects it for shorter text too.

{"input": "<a 300,000-character contract>",
"dimensions": {"kind": ["lease", "employment", "supply"],
"renews": ["automatically", "on notice", "never"]}}

What it will not do. tier: "smart" is refused with bad_tier: the reviewing
model cannot read the document. A document with more passages than the
service scores in one request (256 paragraphs), or a request with more
questions than fit, is refused with chunklaya_input rather than answered
from part of the text. A busy service answers 429 chunklaya_busy with
Retry-After; an unreachable one 503 chunklaya_unavailable. There is no
fallback to another model.

Accuracy on long documents has not been measured against Jev on
classifier.dev traffic; the harness's own results are in its repository,
github.com/myxamediyar/chunklaya. No retail charge during this trial.


AGAINST THE MODEL IT RUNS ON

${vsJevText(false)}
Expand Down Expand Up @@ -327,7 +356,9 @@ PARAMETERS

labels Two to one hundred categories. Required unless dimensions is supplied.
dimensions Named label sets for independent decisions; see MULTIPLE DIMENSIONS.
input The text to classify, up to 32,000 characters.
input The text to classify, up to 32,000 characters. Longer
documents, up to 4,000,000, are answered by chunklaya; see
LONG DOCUMENTS.
inputs Up to one thousand strings classified in a single call.
tier Either fast (the default) or smart, in any case. Anything
else is a 400 with code bad_tier, never a silent fast.
Expand Down Expand Up @@ -470,14 +501,15 @@ ERRORS
and try as fields.

400 bad_json, no_input, too_many_inputs, too_few_labels, too_many_labels,
empty_label, duplicate_labels, empty_input, input_too_long, bad_tier
empty_label, duplicate_labels, empty_input, input_too_long, bad_tier,
chunklaya_input
401 invalid_api_key for unsupported credentials. Workspace authentication
also rejects invalid, paused or revoked keys with an error message;
workspace errors do not include a code.
402 insufficient workspace balance for inference
403 the key is inactive or the workspace cannot authorize usage
404 not_found
429 rate_limit_minute, rate_limit_day, with Retry-After; on the free
429 rate_limit_minute, rate_limit_day, chunklaya_busy, with Retry-After; on the free
tier the body also carries upgrade, the URL of the plan that lifts
the limit (https://classifier.dev/pricing)
502 typesafe or typesafe_<status> when the decision model failed;
Expand Down
2 changes: 1 addition & 1 deletion src/http/spending-classification.ts
Original file line number Diff line number Diff line change
Expand Up @@ -27,7 +27,7 @@ export async function spendingClassification(request: Request, env: AppEnv & Par
const keyHash = await hashToken(request.headers.get("authorization")!.replace(/^Bearer\s+/i, ""));
const maxCredits = Math.ceil(limits.paidRequest / 10000);
const decisions = items * (body.dimensions && typeof body.dimensions === "object" ? Math.max(1, Object.keys(body.dimensions).length) : 1);
const trial = body.model === "laya" || body.model === "kev";
const trial = body.model === "laya" || body.model === "kev" || body.model === "chunklaya";
const quote = Number((classificationCharge(trial ? 0 : 65536 * decisions, body.tier === "smart" ? decisions : 0).nanodollars + 9999n) / 10000n);
if (quote > maxCredits) throw new SpendingError(402, "request_spending_limit", "This request exceeds the workspace request allowance. Split the batch.");
const idempotencyHash = idem ? await fingerprint(env, `account-idempotency:${idem}`) : null;
Expand Down
Loading
Loading