Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
54 changes: 52 additions & 2 deletions e2e/spending.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@ import assert from 'node:assert/strict';
const report = { runtime: 'workerd + SQLite Durable Objects', upstream: 'deterministic HTTP provider fixtures; no live inference', results: [] };
await mkdir('captures', { recursive: true });

let calls = 0, lookups = 0, release, jevUnavailable = false;
let calls = 0, lookups = 0, release, jevUnavailable = false, expensiveSmart = false;
let barrier = Promise.resolve();
const modules = (await readdir('dist/server', { recursive: true })).filter(p => /\.(js|wasm)$/.test(p)).sort((a, b) => a === 'index.js' ? -1 : b === 'index.js' ? 1 : a.localeCompare(b)).map(p => ({ type: p.endsWith('.wasm') ? 'CompiledWasm' : 'ESModule', path: `dist/server/${p}` }));
const mf = new Miniflare(convertV4MiniflareOptions({ workers: [{ name: "spending", modules, modulesRoot: 'dist/server', compatibilityDate: '2026-08-01', compatibilityFlags: ['nodejs_compat'],
Expand All @@ -16,7 +16,7 @@ const mf = new Miniflare(convertV4MiniflareOptions({ workers: [{ name: "spending
calls++; await barrier;
const body = await request.json();
if (jevUnavailable && request.url.includes('typesafe.ai')) return WorkerResponse.json({ detail: { error_type: 'insufficient_credits' } }, { status: 402 });
if (request.url.includes('openrouter.ai')) return WorkerResponse.json({ model: body.model, usage: { cost: jevUnavailable ? 0.000001812 : 0.0007125, prompt_tokens: jevUnavailable ? 100 : 200, completion_tokens: jevUnavailable ? 1 : 150, prompt_tokens_details: { cached_tokens: 0 } }, choices: [{ message: { content: 'A' } }] });
if (request.url.includes('openrouter.ai')) return WorkerResponse.json({ model: body.model, usage: { cost: expensiveSmart ? 0.0072 : jevUnavailable ? 0.000001812 : 0.0007125, prompt_tokens: jevUnavailable ? 100 : 200, completion_tokens: expensiveSmart ? 1900 : jevUnavailable ? 1 : 150, prompt_tokens_details: { cached_tokens: 0 } }, choices: [{ message: { content: 'A' } }] });
const answers = Object.fromEntries(Object.entries(body.questions).map(([id, q]) => { const keys = Object.keys(q.criteria); return [id, { choice: keys[0], confidence: 0.51, probabilities: Object.fromEntries(keys.map((k, i) => [k, i ? 0.49 : 0.51])) }]; }));
return WorkerResponse.json({ model: 'jev-1.13.0', usage: { input_tokens: 100, output_tokens: 0 }, answers });
},
Expand Down Expand Up @@ -91,6 +91,32 @@ try {
assert.equal(smart.status, 200); const result = await smart.json();
assert.ok(JSON.stringify(result).includes('gemini'));
report.results.push({ name: 'free smart escalation', status: smart.status, model: 'google/gemini-3.8-flash' });
const longSmartBody = { inputs: ['Invoice question '.repeat(380)], labels: ['billing', 'support'], tier: 'smart' };
const longSmart = await request('203.0.113.6', longSmartBody);
assert.equal(longSmart.status, 200, await longSmart.text());
report.results.push({ name: 'free smart request beyond former JSON cutoff', bodyBytes: Buffer.byteLength(JSON.stringify(longSmartBody)), status: longSmart.status });
const smartGetUrl = new URL('https://classifier.dev/');
smartGetUrl.searchParams.set('labels', 'billing,support');
smartGetUrl.searchParams.set('text', longSmartBody.inputs[0]);
smartGetUrl.searchParams.set('tier', 'SMART');
const smartGet = await mf.dispatchFetch(smartGetUrl, { headers: { 'cf-connecting-ip': '203.0.113.8', accept: 'application/json' } });
assert.equal(smartGet.status, 200, await smartGet.clone().text());
const smartGetBody = await smartGet.json();
assert.equal(smartGetBody.tier, 'smart');
assert.equal(smartGetBody.escalated, true);
report.results.push({ name: 'free smart GET uses the same provider ceiling', status: smartGet.status });
expensiveSmart = true;
const partial = await request('203.0.113.7', { inputs: Array(20).fill('Invoice question'), labels: ['billing', 'support'], tier: 'smart' });
const partialBody = await partial.json();
assert.equal(partial.status, 200, JSON.stringify(partialBody));
assert.equal(partialBody.results.length, 20);
assert.ok(partialBody.usage.escalated > 0);
assert.ok(partialBody.usage.escalation_failed > 0);
assert.equal(partialBody.usage.escalated + partialBody.usage.escalation_failed, 20);
assert.equal(partialBody.results.filter(result => result.confidence !== null).length, partialBody.usage.escalation_failed);
expensiveSmart = false;
report.results.push({ name: 'smart reviews stop at provider ceiling and retain fast answers', status: partial.status,
escalated: partialBody.usage.escalated, escalationFailed: partialBody.usage.escalation_failed });
for (const path of ['/%76%31/skills', '/skills', '/v1/skills']) {
assert.equal((await request('203.0.113.3', {}, path)).status, 404);
}
Expand Down Expand Up @@ -137,6 +163,25 @@ try {
decisionsAccepted: 200, providerCalls: afterLabels - beforeLabels, dimensionStatus: dimensionDenied.status, sdkStatus: sdkDenied.status, registry: storedLabels });
const budgets = await mf.getDurableObjectNamespace('FREE_BUDGET', 'spending');
const publicBudget = budgets.get(budgets.idFromName('free-spending'));
for (const [tier, amount, ip] of [['fast', 10_000_000, '203.0.113.90'], ['smart', 100_000_000, '203.0.113.91']]) {
const reservation = await publicBudget.fetch('https://budget/reserve', { method: 'POST', body: JSON.stringify({ ip, tier }) });
assert.equal(reservation.status, 200);
const hold = await reservation.json();
assert.equal(hold.amount, amount);
assert.equal((await publicBudget.fetch('https://budget/settle', { method: 'POST', body: JSON.stringify({ id: hold.id, used: 0 }) })).status, 200);
}
report.results.push({ name: 'free request provider ceilings', fastUsd: 0.01, smartUsd: 0.10 });
for (const [tier, count] of [['smart', 4], ['fast', 5]]) {
for (let i = 0; i < count; i++) {
const reservation = await publicBudget.fetch('https://budget/reserve', { method: 'POST', body: JSON.stringify({ ip: '203.0.113.88', tier }) });
assert.equal(reservation.status, 200);
const hold = await reservation.json();
assert.equal((await publicBudget.fetch('https://budget/settle', { method: 'POST', body: JSON.stringify({ id: hold.id, used: hold.amount }) })).status, 200);
}
}
const remainingSmart = await request('203.0.113.88', longSmartBody);
assert.equal(remainingSmart.status, 200, await remainingSmart.text());
report.results.push({ name: 'remaining daily allowance funds a smaller smart reservation', spentBeforeUsd: 0.45, status: remainingSmart.status });
for (let i = 0; i < 50; i++) {
const reserved = await publicBudget.fetch('https://budget/reserve', { method: 'POST', body: JSON.stringify({ ip: '203.0.113.99' }) });
assert.equal(reserved.status, 200);
Expand All @@ -150,6 +195,11 @@ try {
assert.equal(operator.status, 200, 'operator has its own allowance after anonymous exhaustion');
await operator.arrayBuffer();
await new Promise(r => setTimeout(r, 100));
const operatorBudget = budgets.get(budgets.idFromName('operator-spending'));
const operatorRemainder = await operatorBudget.fetch('https://budget/reserve', { method: 'POST', body: JSON.stringify({ ip: '203.0.113.99', operator: true }) });
assert.equal(operatorRemainder.status, 200);
const operatorHold = await operatorRemainder.json();
assert.equal((await operatorBudget.fetch('https://budget/settle', { method: 'POST', body: JSON.stringify({ id: operatorHold.id, used: operatorHold.amount }) })).status, 200);
const operatorExhausted = await request('203.0.113.100', undefined, undefined, { authorization: 'Bearer operator-fixture' });
assert.equal(operatorExhausted.status, 429);
assert.equal((await operatorExhausted.json()).code, 'operator_daily_budget');
Expand Down
11 changes: 6 additions & 5 deletions src/docs.ts
Original file line number Diff line number Diff line change
Expand Up @@ -6,11 +6,12 @@ import { LONG_CONTEXT_JOB_MAX_TOKENS } from "./long-context";

export const SPENDING_LIMITS = `SPENDING LIMITS

Free inference has a $0.01 maximum provider allowance per request, $0.50
per IP per UTC day, and a $100 shared daily ceiling. IPv6 addresses share
a /64 allowance. At most four free requests run concurrently per IP.
Smart mode is available for requests that fit this allowance. Longer
prompts or expensive batches need a funded workspace API key.
Free inference allows up to $0.01 of provider cost per Fast request or $0.10
per Smart request, $0.50 per IP per UTC day, and $100 across everyone per
UTC day. IPv6 addresses share a /64 allowance. At most four free requests
run concurrently per IP. If Smart reviews reach the request ceiling, the
fast answers remain and usage.escalation_failed counts missed reviews.
Longer prompts or expensive batches need a funded workspace API key.
Reservations include in-flight work, retries and fallback models.
Anonymous proxy traffic requires a funded key. Unfunded workspace keys
share the free limits. Funded work uses its workspace balance, outside
Expand Down
11 changes: 5 additions & 6 deletions src/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ import { requestTokens } from "./model-analytics";
import { classificationUsage, classificationPricing } from "./classification-usage";
import { classifyLongContext, LongContextError, longContextInputTokens, LONG_CONTEXT_THRESHOLD, LONG_CONTEXT_MAX_TOKENS, LONG_CONTEXT_MAX_INPUTS, LONG_CONTEXT_MAX_DECISIONS } from "./long-context";
import { recordLongContext } from "./long-context-analytics";
import { withFreeSpending, boundedRequest, freePreflight, type SpendingEnv } from "./spending";
import { withFreeSpending, boundedRequest, type SpendingEnv } from "./spending";
import { SpendingError, errorResponse } from "./spending/policy";
import { providerFetch } from "./spending/permit";
export { FreeBudget } from "./spending";
Expand Down Expand Up @@ -1354,13 +1354,12 @@ const worker = {
try {
req = await boundedRequest(req);
const body = req.method === "POST" ? await req.clone().json().catch(() => undefined) as Record<string, unknown> | undefined : undefined;
const querySmart = new URL(req.url).searchParams.get("tier") === "smart";
if (!execution?.funded) freePreflight(req, env, querySmart ? { tier: "smart" } : body);
const tier = readTier(req.method === "POST" ? body?.tier : new URL(req.url).searchParams.get("tier")) ?? "fast";
const meter = execution?.meter ?? newMeter();
const next = () => worker.fetch(req, env, ctx, { ...execution, meter, spending: true });
const credential = req.headers.get("authorization")?.match(/^Bearer\s+(\S+)$/i)?.[1];
const operator = !!env.AGENT_API_KEY && !!credential && await secretEquals(credential, env.AGENT_API_KEY);
return execution?.funded ? await next() : await withFreeSpending(req, env, ctx, meter, next, operator);
return execution?.funded ? await next() : await withFreeSpending(req, env, ctx, meter, next, { tier, operator });
} catch (error) {
if (error instanceof SpendingError) return errorResponse(error);
throw error;
Expand Down Expand Up @@ -2378,7 +2377,7 @@ const worker = {
escalationFailed = r.escalationFailed;
fallbackDecisions = r.fallbackDecisions;
} else ({ results, escalationFailed } = await classifyMany(env, inputs, labels, tier, instructions, multi, meter, layaPlan, layaTiming, layaRun, backend, longContext));
if (meter.permit?.error && !(execution?.funded && meter.permit.error.code === "request_spending_limit")) throw meter.permit.error;
if (meter.permit?.error && !((execution?.funded || (tier === "smart" && escalationFailed > 0)) && meter.permit.error.code === "request_spending_limit")) throw meter.permit.error;
} catch (e) {
const spending = e instanceof SpendingError ? e : meter.permit?.error;
if (spending) {
Expand Down Expand Up @@ -2457,7 +2456,7 @@ const worker = {
return text(body + "\n", 200, headers);
}
if (req.method === "GET") {
return json({ ...results[0], tier, usage: tokenUsage, pricing }, 200, headers);
return json({ ...results[0], tier, usage: { ...tokenUsage, ...(escalationFailed ? { escalation_failed: escalationFailed } : {}) }, pricing }, 200, headers);
}
return json(
{
Expand Down
7 changes: 4 additions & 3 deletions src/openapi.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1524,9 +1524,10 @@ short \`reporter.agent_description\` describing what kind of agent they are.

## Limits

Free provider spending is capped at $0.01/request, $0.50/IP/UTC day and
$100/day across all free traffic, with four concurrent requests per IP.
IPv6 addresses share a /64. Smart requests must fit the request allowance.
Free provider spending is capped at $0.01/Fast request, $0.10/Smart request,
$0.50/IP/UTC day and $100/day across all free traffic, with four concurrent
requests per IP. IPv6 addresses share a /64. Smart reviews that reach the
request cap retain fast answers and appear in usage.escalation_failed.
Funded workspace keys use their balance and bypass the shared subsidy and
proxy check; the default maximum request allowance is $10. Request bodies
are limited to 1 MB for synchronous classification. One whole Jev document can
Expand Down
10 changes: 6 additions & 4 deletions src/pages.ts
Original file line number Diff line number Diff line change
Expand Up @@ -523,10 +523,12 @@ FREE
request are 1,000 classifications. Multi-label counts once per text, not per
label.

Free provider spending is capped at $0.01 per request, $0.50 per IP network
per UTC day and $100 across everyone per UTC day. Up to four free requests
may run at once per IP; IPv6 addresses share a /64 allowance. Smart requests
must fit the same allowance. Large inputs or batches need a funded key.
Free provider spending is capped at $0.01 per Fast request, $0.10 per Smart
request, $0.50 per IP network per UTC day and $100 across everyone per UTC
day. Up to four free requests may run at once per IP; IPv6 addresses share
a /64 allowance. If Smart reviews reach the request cap, fast answers
remain and usage.escalation_failed counts missed reviews. Large inputs or
batches need a funded key.
Free access pauses when the shared pool or verification capacity is spent;
anonymous proxy networks require a funded key. Signup credit uses these
same free limits. Synchronous request bodies are limited to 1 MB;
Expand Down
4 changes: 3 additions & 1 deletion src/pricingui.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,9 +3,11 @@ import { FOOT, HOME_CSS, META, NAV } from "./home";
import { BILLING_PLANS, formatCreditsUsd } from "./lib/billing";
import { INPUT_PRICE_PER_MILLION, ESCALATION_PRICE_PER_THOUSAND, LONG_CONTEXT_PRICING } from "./lib/classification-pricing";
import { LONG_CONTEXT_JOB_MAX_TOKENS } from "./long-context";
import { policy } from "./spending/policy";
import { esc, page } from "./ui";

const feature = (text: string) => `<li>${esc(text)}</li>`;
const freeLimits = policy({});

export function pricingHtml(signedIn = false) {
const pro = BILLING_PLANS.pro;
Expand Down Expand Up @@ -87,7 +89,7 @@ export function pricingHtml(signedIn = false) {
<tr><th scope="row">Pro</th><td>30,000/min · 200,000/day</td><td>2,000/min · 20,000/day</td></tr>
</tbody></table></div>
<p class="pricing-note">Limits count classifications and are shared across workspace keys and agents. Public access is limited per IP. Laya trial limits apply to every plan.</p>
<p>Free access shares daily capacity and allows four requests at once per IP. A funded workspace has its own allowance and supports larger Smart requests. Signup credit alone uses the free limits.</p>
<p>Free access allows up to $${(freeLimits.fastRequest / 1e9).toFixed(2)} of provider cost per Fast request or $${(freeLimits.smartRequest / 1e9).toFixed(2)} per Smart request, with $${(freeLimits.ipDaily / 1e9).toFixed(2)} per IP per UTC day and four requests at once. Smart reviews beyond the request cap retain their Fast answers. A funded workspace has its own allowance and supports larger Smart requests. Signup credit alone uses the free limits.</p>
<p class="pricing-note">See <a href="/developers">request limits and retry guidance</a>. Usage is reserved before a request and unused funds are released afterward. Anonymous proxy networks require a funded key.</p>
</section>
${FOOT}
Expand Down
14 changes: 8 additions & 6 deletions src/spending/free-budget.ts
Original file line number Diff line number Diff line change
Expand Up @@ -73,7 +73,7 @@ export class FreeBudget {
const [owner, prior] = await Promise.all([fingerprint(this.env, `${day}:${net}`), fingerprint(this.env, `${previous}:${net}`)]);
const limits = policy(this.env);
const id = crypto.randomUUID();
const amount = limits.request;
const ceiling = body.tier === "smart" ? limits.smartRequest : limits.fastRequest;
const daily = operator ? limits.operatorDaily : limits.daily;
const ipDaily = operator ? limits.operatorDaily : limits.ipDaily;
const reserve = async (commit: boolean) => this.state.storage.transaction(async tx => {
Expand All @@ -84,17 +84,18 @@ export class FreeBudget {
const globalPrior = await tx.get<Total>(`total:${previous}`) ?? empty();
const allActive = Object.values(global.holds).filter(t => t > Date.now()).length + Object.values(globalPrior.holds).filter(t => t > Date.now()).length;
const retryAfter = Math.ceil((Date.parse(`${day}T00:00:00Z`) + 86400000 - Date.now()) / 1000);
if (global.spent + amount > daily)
if (global.spent >= daily)
throw new SpendingError(429, operator ? "operator_daily_budget" : "free_daily_budget", `The ${operator ? "operator" : "shared free"} budget is spent or reserved ($${daily / 1e9} per UTC day).`, { retryAfter, limitUsd: daily / 1e9, availableUsd: Math.max(0, daily - global.spent) / 1e9 });
if (mine.spent + amount > ipDaily)
throw new SpendingError(429, "free_ip_daily_budget", `This IP network's free budget is spent or reserved ($${limits.ipDaily / 1e9} per UTC day).`, { retryAfter, limitUsd: limits.ipDaily / 1e9, availableUsd: Math.max(0, limits.ipDaily - mine.spent) / 1e9 });
if (mine.spent >= ipDaily)
throw new SpendingError(429, "free_ip_daily_budget", `This IP network's free budget is spent or reserved ($${ipDaily / 1e9} per UTC day).`, { retryAfter, limitUsd: ipDaily / 1e9, availableUsd: Math.max(0, ipDaily - mine.spent) / 1e9 });
if (active >= limits.ipConcurrency)
throw new SpendingError(429, "free_ip_concurrency", `At most ${limits.ipConcurrency} free requests may run at once per IP network.`, { retryAfter: 2, limit: limits.ipConcurrency });
if (allActive >= limits.concurrency)
throw new SpendingError(429, "free_capacity", "The free inference pool is at capacity.", { retryAfter: 2 });
const idem = typeof body.idempotency === "string" ? await fingerprint(this.env, `${owner}:${body.idempotency}`) : undefined;
if (idem && await tx.get(`idem:${day}:${idem}`)) throw new SpendingError(409, "duplicate_request", "This idempotency key was already admitted today. No additional work was started.");
if (!commit) return;
const amount = Math.min(ceiling, daily - global.spent, ipDaily - mine.spent);
if (!commit) return amount;
for (const total of [global, mine]) for (const [key, expires] of Object.entries(total.holds))
if (expires <= Date.now()) delete total.holds[key];
const expires = Date.now() + 180000;
Expand All @@ -103,10 +104,11 @@ export class FreeBudget {
await tx.put(`total:${day}`, global); await tx.put(`ip:${day}:${owner}`, mine);
await tx.put(`hold:${id}`, { day, owner, amount, expires } satisfies Hold);
if (idem) await tx.put(`idem:${day}:${idem}`, id);
return amount;
});
await reserve(false);
if (!operator && !await this.reputation(ip, `${day}:${owner}`)) throw new SpendingError(403, "proxy_requires_payment", "Anonymous proxy traffic requires a funded API key.");
await reserve(true);
const amount = await reserve(true);
return { id, amount, expires: Date.now() + 90000 };
}
private async settle(id: string, used: number) {
Expand Down
Loading
Loading