From ca8a948a56593d5c5291b78d4ad679e56cd0459d Mon Sep 17 00:00:00 2001 From: Michael Ryaboy Date: Tue, 22 Sep 2026 01:33:40 -0700 Subject: [PATCH 1/2] Price base input at Jev's rate and Smart reviews per escalation --- drizzle/schema.ts | 4 ++ .../postgres/0013_classification_pricing.sql | 2 + src/alerts.ts | 2 +- src/cost.ts | 1 - src/features/billing/plans.tsx | 49 +++------------ src/http/spending-classification.ts | 46 +++++++------- src/lib/classification-pricing.ts | 25 ++++++++ src/openapi.ts | 3 + src/pages.ts | 36 ++++++----- src/pricingui.ts | 27 ++++----- src/spending/index.ts | 2 +- src/spending/permit.ts | 18 +----- tests/dashboard-analytics.test.ts | 4 +- tests/plans-prices.test.ts | 9 +-- tests/public-pricing.test.ts | 4 +- tests/spending.e2e.test.ts | 60 ++++++++++++++++++- 16 files changed, 166 insertions(+), 126 deletions(-) create mode 100644 migrations/postgres/0013_classification_pricing.sql create mode 100644 src/lib/classification-pricing.ts diff --git a/drizzle/schema.ts b/drizzle/schema.ts index 25d5e92..e7ddac6 100644 --- a/drizzle/schema.ts +++ b/drizzle/schema.ts @@ -108,6 +108,8 @@ export const app_usage = pgTable("app_usage", { actual_nano: bigint({ mode: "bigint" }), rate_version: text(), metering_mode: text().default('credits').notNull(), + classifications: integer(), + escalations: integer(), reporting_status: text().default('not_ready').notNull(), }, (table) => [ index("app_usage_account").using("btree", table.account_id.asc().nullsLast().op("text_ops"), table.created_at.asc().nullsLast().op("text_ops")), @@ -124,6 +126,8 @@ export const app_usage = pgTable("app_usage", { name: "app_usage_agent_id_fkey" }), check("app_usage_actual_nano_check", sql`actual_nano >= 0`), + check("app_usage_classifications_check", sql`classifications >= 0`), + check("app_usage_escalations_check", sql`escalations >= 0`), check("app_usage_metering_mode_check", sql`metering_mode = ANY (ARRAY['credits'::text, 'tokens'::text, 'legacy'::text])`), check("app_usage_reporting_status_check", sql`reporting_status = ANY (ARRAY['not_ready'::text, 'review'::text, 'pending'::text, 'reported'::text, 'exempt'::text])`), ]); diff --git a/migrations/postgres/0013_classification_pricing.sql b/migrations/postgres/0013_classification_pricing.sql new file mode 100644 index 0000000..05c84ea --- /dev/null +++ b/migrations/postgres/0013_classification_pricing.sql @@ -0,0 +1,2 @@ +ALTER TABLE app_usage ADD COLUMN classifications INTEGER CHECK(classifications>=0); +ALTER TABLE app_usage ADD COLUMN escalations INTEGER CHECK(escalations>=0); diff --git a/src/alerts.ts b/src/alerts.ts index 00cd0d5..e71fb36 100644 --- a/src/alerts.ts +++ b/src/alerts.ts @@ -140,7 +140,7 @@ export async function evaluate(env: Env): Promise<{ alerts: Alert[]; checked: bo if (app.APP_ACCOUNTS_ENABLED === "true") { const row = await app.APP_DB.prepare("SELECT count(*) AS count FROM app_usage WHERE metering_mode='tokens' AND (reporting_status='review' OR (status='pending' AND created_at::timestamptz(); if (Number(row?.count) > 0) pre.push({ id: "billing_review", severity: "warning", title: "billing reservations need review", - detail: `${row!.count} token reservations have uncertain usage or have remained pending for over five minutes. Funds remain held. Reconcile app_usage request IDs against provider usage before settling or refunding; do not blindly release these reservations.` }); + detail: `${row!.count} usage reservations have uncertain outcomes or have remained pending for over five minutes. Funds remain held. Reconcile app_usage request IDs against provider usage before settling or refunding; do not blindly release these reservations.` }); } } catch { pre.push({ id: "spending_monitor", severity: "critical", title: "spending monitoring is unavailable", detail: "Check the free-budget Durable Object and account database. Admission remains fail-closed; this alert must clear before assuming spending and billing are healthy." }); diff --git a/src/cost.ts b/src/cost.ts index 9d0883d..3eee0a2 100644 --- a/src/cost.ts +++ b/src/cost.ts @@ -37,7 +37,6 @@ export type ModelTokenUsage = TokenCounts & { /** Per-request accumulator. Token counts are upstream usage, not a retail price. */ export type Meter = { - accountAllowance?: { card: import("./server/token-pricing").TokenRateCard; credits: number }; permit?: import("./spending/permit").Permit; usd: number; tokens: ModelTokenUsage[]; diff --git a/src/features/billing/plans.tsx b/src/features/billing/plans.tsx index a65bba2..6fd8f62 100644 --- a/src/features/billing/plans.tsx +++ b/src/features/billing/plans.tsx @@ -21,7 +21,7 @@ import { type BillingPlanId, } from "@/lib/billing"; import type { AppSnapshot } from "@/server/contracts"; -import retailRates from "../../retail-rates.json"; +import { INPUT_PRICE_PER_MILLION, ESCALATION_PRICE_PER_THOUSAND } from "@/lib/classification-pricing"; const date = (value: string) => new Date(value).toLocaleDateString("en-US", { @@ -250,50 +250,19 @@ export function Plans({ ))} -
-

- Token prices -

-

- Prices per million tokens. Default Fast uses Jev at cost. Smart adds Gemini at - cost plus 20% only when it escalates. Laya inference is free during the trial; - Smart reviews are still billed. -

+
+

Simple usage prices

- - - - - - - - - +
ModelInputCached inputOutput
+ - {retailRates.models.map((rate) => ( - - - {[ - rate.inputUsdPerMillion, - rate.cachedInputUsdPerMillion, - rate.outputUsdPerMillion, - ].map((value, index) => ( - - ))} - - ))} + +
UsagePrice
- {rate.provider === "typesafe" ? "Jev" : rate.model === "google/gemini-3.8-flash" ? "Gemini escalation" : rate.model} - - {Number(value) === 0 ? "Free" : `$${Number(value)}`} -
Input tokens${INPUT_PRICE_PER_MILLION.toFixed(3)} / million
Smart escalations+${ESCALATION_PRICE_PER_THOUSAND.toFixed(2)} / 1,000
-

- Smart requests without escalation cost the same as Fast. Gemini output - includes reasoning tokens. Usage stops when your balance is depleted; - there are no automatic top-ups. -

+

Smart reviews uncertain answers. You pay extra only for successful escalations. For example, 1 million input tokens with 50 Smart escalations cost $0.142.

+

Output tokens are free. Input usage includes text, labels and instructions. Retries, fallback routing and Smart model tokens add no separate charges. No automatic top-ups.

diff --git a/src/http/spending-classification.ts b/src/http/spending-classification.ts index d44fbd8..3d204f5 100644 --- a/src/http/spending-classification.ts +++ b/src/http/spending-classification.ts @@ -1,8 +1,7 @@ import worker, { type Env } from "../index"; import { newMeter } from "../cost"; import { AppError, hashToken, now, type AppEnv } from "../server/db"; -import rates from "../retail-rates.json"; -import { parseTokenRateCard, priceTokens } from "../server/token-pricing"; +import { classificationCharge, classificationInputTokens } from "../lib/classification-pricing"; import { refundTokenReservation, settleTokenReservation } from "../server/token-ledger"; import { Permit } from "../spending/permit"; import { boundedRequest } from "../spending"; @@ -10,7 +9,6 @@ import { SpendingError, errorResponse, fingerprint, policy } from "../spending/p import { writeAccountAnalytics } from "../server/analytics/write"; import { typeSafeDecisionCount } from "../typesafe-compat"; -const card = parseTokenRateCard(JSON.stringify(rates))!; export async function spendingClassification(request: Request, env: AppEnv & Partial, source: "API" | "MCP", ctx: ExecutionContext): Promise { try { if (env.APP_ACCOUNTS_ENABLED !== "true") throw new AppError(503, "Account credentials are disabled."); @@ -29,25 +27,25 @@ export async function spendingClassification(request: Request, env: AppEnv & Par const keyHash = await hashToken(request.headers.get("authorization")!.replace(/^Bearer\s+/i, "")); const maxCredits = Math.ceil(limits.paidRequest / 10000); const decisions = items * (body.dimensions && typeof body.dimensions === "object" ? Math.max(1, Object.keys(body.dimensions).length) : 1); - const bytes = new TextEncoder().encode(text.normalize("NFKC")).length; - const smartCredits = Math.ceil(((bytes * 2 + 4096) * 0.9 + 2000 * 4.5) * 0.1) * 3; - const quote = Math.min(maxCredits, Math.ceil(decisions * (body.tier === "smart" ? smartCredits + 1500 : 1500))); + const trial = body.model === "laya" || body.model === "kev"; + const quote = Number((classificationCharge(trial ? 0 : 65536 * decisions, body.tier === "smart" ? decisions : 0).nanodollars + 9999n) / 10000n); + if (quote > maxCredits) throw new SpendingError(402, "request_spending_limit", "This request exceeds the workspace request allowance. Split the batch."); const idempotencyHash = idem ? await fingerprint(env, `account-idempotency:${idem}`) : null; const result = await env.APP_DB.prepare(`WITH owner AS ( SELECT a.id,a.balance,a.paid_balance,(a.paid_balance>0 OR (a.billing_plan IN ('pro','max','scale') AND a.reset_at::timestamptz>now())) AS funded,k.id AS agent_id FROM app_accounts a JOIN app_agents k ON k.account_id=a.id WHERE k.token_hash=? AND k.status IN ('pending','connected') AND NOT a.billing_hold FOR UPDATE OF a ), held AS ( - INSERT INTO app_usage(id,account_id,agent_id,items,credits,status,created_at,paid_credits,usage_type,metering_mode,idempotency_key) - SELECT ?,id,agent_id,?,LEAST(balance,CASE WHEN funded THEN ?::bigint ELSE ?::bigint END),'pending',?, - GREATEST(0,LEAST(balance,CASE WHEN funded THEN ?::bigint ELSE ?::bigint END)-(balance-paid_balance)),?,'tokens',? - FROM owner WHERE balance>0 ON CONFLICT DO NOTHING RETURNING * + INSERT INTO app_usage(id,account_id,agent_id,items,credits,status,created_at,paid_credits,usage_type,metering_mode,idempotency_key,classifications) + SELECT ?,id,agent_id,?,?::bigint,'pending',?, + GREATEST(0,?::bigint-(balance-paid_balance)),?,'tokens',?,? + FROM owner WHERE balance>0 AND balance>=? ON CONFLICT DO NOTHING RETURNING * ), debited AS ( UPDATE app_accounts a SET balance=a.balance-h.credits,paid_balance=a.paid_balance-h.paid_credits FROM held h WHERE a.id=h.account_id RETURNING a.id ), agent AS ( UPDATE app_agents a SET used=a.used+h.credits FROM held h,debited d WHERE a.id=h.agent_id AND d.id=h.account_id RETURNING a.id ) SELECT h.account_id,h.agent_id,h.credits,o.funded FROM held h JOIN owner o ON o.id=h.account_id JOIN agent k ON k.id=h.agent_id`) - .bind(keyHash, id, items, quote, Math.ceil(limits.request * 1.5 / 10000), now(), quote, Math.ceil(limits.request * 1.5 / 10000), `${source} · Classification`, idempotencyHash) + .bind(keyHash, id, items, quote, now(), quote, `${source} · Classification`, idempotencyHash, decisions, quote) .first<{ account_id: string; agent_id: string; credits: number; funded: boolean }>(); if (!result) { const reason = await env.APP_DB.prepare(`SELECT k.status,a.billing_hold, @@ -57,11 +55,10 @@ export async function spendingClassification(request: Request, env: AppEnv & Par if (!reason) throw new SpendingError(401, "invalid_api_key", "This workspace API key is not valid."); if (!["pending", "connected"].includes(reason.status)) throw new SpendingError(403, "inactive_api_key", "This workspace API key is paused or revoked."); if (reason.request_id) throw new SpendingError(409, "duplicate_request", "This workspace already admitted the idempotency key.", { requestId: reason.request_id }); - throw new SpendingError(402, "insufficient_balance", reason.billing_hold ? "Workspace billing is awaiting review; funds remain held." : "The workspace has no unreserved balance for this request."); + throw new SpendingError(402, "insufficient_balance", reason.billing_hold ? "Workspace billing is awaiting review; funds remain held." : "The workspace needs enough available balance for the maximum input tokens and possible Smart escalations. Unused funds are released after the request.", { requiredUsd: quote / 100000 }); } const meter = newMeter(); - meter.accountAllowance = { card, credits: result.credits }; - const permit = result.funded ? new Permit(limits.paidRequest, Date.now() + 90000, { card, credits: result.credits }) : undefined; + const permit = result.funded ? new Permit(limits.paidRequest, Date.now() + 90000) : undefined; if (permit) { meter.permit = permit; meter.beforeCall = async () => {}; } const started = Date.now(); let response: Response; @@ -71,32 +68,41 @@ export async function spendingClassification(request: Request, env: AppEnv & Par } catch (error) { response = error instanceof SpendingError ? errorResponse(error) : Response.json({ error: "Classification failed." }, { status: 502 }); } + const payload = response.ok ? await response.clone().json().catch(() => null) as { usage?: { escalated?: unknown } } | null : null; + const escalations = new URL(request.url).pathname === "/v1/systemone" ? 0 : payload?.usage?.escalated; + const inputTokens = trial ? 0 : classificationInputTokens(meter.tokens); + const charge = response.ok && inputTokens !== null && typeof escalations === "number" && Number.isSafeInteger(escalations) && escalations >= 0 && escalations <= decisions + ? classificationCharge(inputTokens, escalations) : null; const settle = async () => { const actual = meter.permit; actual?.close(); await actual?.drain(); const tokens = actual?.tokens ?? []; - const charge = actual?.unknown ? null : response.ok && tokens.length ? priceTokens(card, tokens) : { version: card.version, nanodollars: 0n }; - // Uncertain attempts remain held for reconciliation, even on an HTTP error. for (let attempt = 0; attempt < 3; attempt++) { try { - if (!actual || actual.used === 0) await refundTokenReservation(env.APP_DB, id); + if (!response.ok || !actual || actual.used === 0) await refundTokenReservation(env.APP_DB, id); else await settleTokenReservation(env.APP_DB, id, charge, { - inputTokens: tokens.length && tokens.every(t => t.inputTokens !== null) ? tokens.reduce((sum, t) => sum + t.inputTokens!, 0) : null, + inputTokens, outputTokens: tokens.length && tokens.every(t => t.outputTokens !== null) ? tokens.reduce((sum, t) => sum + t.outputTokens!, 0) : null, }); + if (charge !== null) await env.APP_DB.prepare("UPDATE app_usage SET escalations=? WHERE id=? AND status='completed'").bind(escalations, id).run(); writeAccountAnalytics(env, { accountId: result.account_id, keyId: result.agent_id, requestId: id, source, tier: body.tier === "smart" ? "smart" : "fast", status: response.ok ? "success" : "error", items, - inputTokens: tokens.every(t => t.inputTokens !== null) ? tokens.reduce((sum, t) => sum + t.inputTokens!, 0) : null, + inputTokens, outputTokens: tokens.every(t => t.outputTokens !== null) ? tokens.reduce((sum, t) => sum + t.outputTokens!, 0) : null, cachedInputTokens: null, model: tokens.map(t => t.model).join(","), providerCostUsd: actual?.unknown ? null : (actual?.used ?? 0) / 1e9, - retailCostUsd: charge ? Number(charge.nanodollars) / 1e9 : null, latencyMs: Date.now() - started, escalations: tokens.filter(t => t.provider === "openrouter").reduce((n, t) => n + t.calls, 0) }); + retailCostUsd: !response.ok ? 0 : charge ? Number(charge.nanodollars) / 1e9 : null, latencyMs: Date.now() - started, escalations: typeof escalations === "number" ? escalations : 0 }); return; } catch { /* Durable pending funds remain unavailable until reconciliation. */ } } }; ctx.waitUntil(settle()); const headers = new Headers(response.headers); + if (charge) { + headers.set("x-billed-input-tokens", String(inputTokens)); + headers.set("x-smart-escalations", String(escalations)); + headers.set("x-usage-cost-usd", (Number(charge.nanodollars) / 1e9).toFixed(9)); + } headers.set("x-request-id", id); headers.set("x-billing-status", "pending"); headers.set("cache-control", "no-store"); return new Response(response.body, { status: response.status, headers }); } catch (error) { diff --git a/src/lib/classification-pricing.ts b/src/lib/classification-pricing.ts new file mode 100644 index 0000000..e8ed260 --- /dev/null +++ b/src/lib/classification-pricing.ts @@ -0,0 +1,25 @@ +import type { ModelTokenUsage } from "../cost"; + +export const CLASSIFICATION_PRICING = { + version: "2026-09-22-input-escalation-v1", + inputNanodollars: 42, + escalationNanodollars: 2_000_000, +} as const; + +export const INPUT_PRICE_PER_MILLION = CLASSIFICATION_PRICING.inputNanodollars / 1000; +export const ESCALATION_PRICE_PER_THOUSAND = CLASSIFICATION_PRICING.escalationNanodollars / 1e6; + +export function classificationCharge(inputTokens: number, escalations: number) { + if (!Number.isSafeInteger(inputTokens) || inputTokens < 0 || !Number.isSafeInteger(escalations) || escalations < 0) + throw new Error("Invalid classification billing counts."); + return { version: CLASSIFICATION_PRICING.version, + nanodollars: BigInt(inputTokens) * BigInt(CLASSIFICATION_PRICING.inputNanodollars) + + BigInt(escalations) * BigInt(CLASSIFICATION_PRICING.escalationNanodollars) }; +} + +export function classificationInputTokens(tokens: ModelTokenUsage[]): number | null { + // Only answered primary calls count. Recovery and Smart provider costs are ours. + const primary = tokens.filter(row => row.provider === "typesafe" || row.provider === "vercel"); + return primary.every(row => row.inputTokens !== null) + ? primary.reduce((sum, row) => sum + row.inputTokens!, 0) : null; +} diff --git a/src/openapi.ts b/src/openapi.ts index d41bdf7..536f957 100644 --- a/src/openapi.ts +++ b/src/openapi.ts @@ -59,6 +59,9 @@ const TYPESAFE_REQUEST_ID_HEADER = { }; const ACCOUNT_BILLING_HEADERS = { "x-request-id": { schema: { type: "string" }, description: "Workspace usage request ID. Present when a classifier_agent_ key is used." }, + "x-billed-input-tokens": { schema: { type: "integer", minimum: 0 }, description: "Successful base-classifier input tokens billed at the published rate; excludes Smart and fallback provider tokens." }, + "x-smart-escalations": { schema: { type: "integer", minimum: 0 }, description: "Successful Smart reviews billed at the flat escalation price." }, + "x-usage-cost-usd": { schema: { type: "string" }, description: "Customer charge in USD before credit rounding. Settlement may still be pending." }, "x-billing-status": { schema: { type: "string", enum: ["pending", "settled", "refunded", "review"] }, description: "Workspace charge result. Present when a classifier_agent_ key is used." }, }; const TYPESAFE_ENTRY = { diff --git a/src/pages.ts b/src/pages.ts index af48bac..80abbc1 100644 --- a/src/pages.ts +++ b/src/pages.ts @@ -12,7 +12,7 @@ import { SPENDING_LIMITS } from "./docs"; import { SITE, SITE_UPDATED } from "./wellknown"; import { codeLang } from "./ui"; import { BILLING_PLANS, formatCreditsUsd } from "./lib/billing"; -import retailRates from "./retail-rates.json"; +import { INPUT_PRICE_PER_MILLION, ESCALATION_PRICE_PER_THOUSAND } from "./lib/classification-pricing"; export const MCP_SETUP = `classifier.dev MCP @@ -349,8 +349,10 @@ KEYS AND LIMITS and MCP APIs. GET /v1/models does not spend quota or credits. TypeSafe-native validation, rate-limit and service errors retain their status and body; x-typesafe-request-id, Retry-After and Retry-After-Ms are preserved. Workspace - responses also expose x-request-id and x-billing-status (settled, refunded or - review). + responses also expose x-request-id and x-billing-status (pending, settled, refunded or + review). Successful workspace responses include x-billed-input-tokens, + x-smart-escalations and x-usage-cost-usd. The price is $${INPUT_PRICE_PER_MILLION.toFixed(3)} per million + base input tokens plus $${(ESCALATION_PRICE_PER_THOUSAND / 1000).toFixed(3)} per successful Smart escalation. SANDBOX @@ -445,14 +447,14 @@ PRO Rate limits 10x Free, shared across workspace keys and agents Billing https://classifier.dev/app/plans - Usage is charged to the workspace balance at the published token prices. + Usage is charged to the workspace balance at the published input-token and escalation prices. Smart costs the same as Fast when no escalation is needed. Usage stops when the balance reaches zero; there are no automatic top-ups. Funded workspaces skip the shared free pool and proxy checks. Each request has a default $10 provider-cost ceiling, bounded by the available workspace - balance at the published retail prices. Unused reservations are released - after inference; uncertain provider usage remains held for billing review. + balance at the published customer prices. Unused reservations are released + after inference; missing input-token measurements remain held for billing review. ENTERPRISE @@ -476,22 +478,18 @@ ENTERPRISE own data before you commit. -TOKEN PRICES +USAGE PRICES - Prices per million tokens. Fast uses Jev at cost. Smart adds Gemini at cost - plus 20% only when it escalates. + Input tokens $${INPUT_PRICE_PER_MILLION.toFixed(3)} per million + Smart escalations +$${ESCALATION_PRICE_PER_THOUSAND.toFixed(2)} per 1,000 - Jev input $${Number(retailRates.models[0].inputUsdPerMillion)} - Jev cached input $${Number(retailRates.models[0].cachedInputUsdPerMillion)} - Jev output free - Gemini input $${Number(retailRates.models[1].inputUsdPerMillion)} - Gemini cached input $${Number(retailRates.models[1].cachedInputUsdPerMillion)} - Gemini output $${Number(retailRates.models[1].outputUsdPerMillion)} + Smart starts with Fast and reviews uncertain answers. You pay extra only + for successful Smart escalations. No escalation means no extra charge. + 1 million input tokens with 50 Smart escalations cost $0.142. - Laya trial lanes free during the trial, subject to shared capacity - - Smart requests without escalation cost the same as Fast. Gemini output - includes reasoning tokens. Usage stops when your balance reaches zero. + Output tokens are free. Input usage includes text, labels and instructions + processed by the base classifier. Retries, fallback routing and Smart model + tokens add no separate charges. Usage stops when your balance runs out. INCLUDED ON EVERY PLAN diff --git a/src/pricingui.ts b/src/pricingui.ts index 15cb2f2..05de578 100644 --- a/src/pricingui.ts +++ b/src/pricingui.ts @@ -1,20 +1,12 @@ import { FOOT, HOME_CSS, META, NAV } from "./home"; import { BILLING_PLANS, formatCreditsUsd } from "./lib/billing"; -import retailRates from "./retail-rates.json"; +import { INPUT_PRICE_PER_MILLION, ESCALATION_PRICE_PER_THOUSAND } from "./lib/classification-pricing"; import { esc, page } from "./ui"; const feature = (text: string) => `
  • ${esc(text)}
  • `; export function pricingHtml(signedIn = false) { const pro = BILLING_PLANS.pro; - const tokenRows = retailRates.models - .map( - (rate) => `${rate.provider === "typesafe" ? "Jev" : rate.model === "google/gemini-3.8-flash" ? "Gemini escalation" : esc(rate.model)} - ${Number(rate.inputUsdPerMillion) === 0 ? "Free" : `$${Number(rate.inputUsdPerMillion)}`} - ${Number(rate.cachedInputUsdPerMillion) === 0 ? "Free" : `$${Number(rate.cachedInputUsdPerMillion)}`} - ${Number(rate.outputUsdPerMillion) === 0 ? "Free" : `$${Number(rate.outputUsdPerMillion)}`}`, - ) - .join(""); const description = "Simple plans with upfront usage for classifier.dev workspaces."; return page({ @@ -68,10 +60,14 @@ export function pricingHtml(signedIn = false) {
      ${feature("Volume-based capacity")}${feature("Dedicated deployment")}${feature("Private inference options")}${feature("Measured accuracy on your data")}
    Contact sales
    -

    Token prices

    -

    Prices per million tokens. Fast uses Jev at cost. Smart adds Gemini at cost plus 20% only when it escalates.

    -
    ${tokenRows}
    ModelInputCached inputOutput
    -

    Smart requests without escalation cost the same as Fast. Gemini output includes reasoning tokens. Usage stops when your balance reaches zero; no automatic top-ups. Laya lanes are free during the trial, subject to shared capacity limits.

    +

    Usage prices

    +
    + + +
    UsagePrice
    Input tokens$${INPUT_PRICE_PER_MILLION.toFixed(3)} / million
    Smart escalation+$${ESCALATION_PRICE_PER_THOUSAND.toFixed(2)} / 1,000
    +

    Smart starts with Fast and reviews uncertain answers. You pay the extra charge only for answers that are successfully reviewed. No escalation means no extra charge.

    +

    For example, 1 million input tokens with 50 Smart escalations cost $0.142.

    +

    Output tokens are free. Input usage includes the text, labels and instructions processed by the base classifier. Retries, fallback routing and Smart model tokens add no separate charges. Usage stops when your balance runs out; no automatic top-ups.

    Included on every plan

    Fast and Smart classification with your own labels through REST, MCP or the CLI.

    @@ -82,9 +78,8 @@ export function pricingHtml(signedIn = false) { Pro30,000/min · 200,000/day2,000/min · 20,000/day

    Limits count classifications and are shared across workspace keys and agents. Public access is limited per IP. Laya trial limits apply to every plan.

    -

    Free inference also has a $0.01 provider-cost allowance per request, $0.50 per IP network per UTC day, and a $100 shared daily pool. Up to four requests may run at once per IP; IPv6 addresses share a /64 allowance. Smart requests must fit the same allowance, so large inputs or batches need a funded key.

    -

    A funded workspace uses its own balance, skips free-pool and proxy checks, and has a default $10 provider-cost ceiling per request. Actual usage is billed at the token prices above. Signup credit alone uses the free limits. Every request body is limited to 1 MB.

    -

    Free access may pause when its shared pool or verification capacity is exhausted. Anonymous proxy networks require a funded key. See spending limits and retry guidance.

    +

    Free access shares daily capacity and allows four requests at once per IP. A funded workspace has its own allowance and supports larger Smart requests. Signup credit alone uses the free limits.

    +

    See request limits and retry guidance. Usage is reserved before a request and unused funds are released afterward. Anonymous proxy networks require a funded key.

    ${FOOT} `, diff --git a/src/spending/index.ts b/src/spending/index.ts index 492a67d..ccbd361 100644 --- a/src/spending/index.ts +++ b/src/spending/index.ts @@ -20,7 +20,7 @@ export async function withFreeSpending(request: Request, env: SpendingEnv, ctx: throw new SpendingError(response.status, error.code, error.error, error); } hold = await response.json(); - meter.permit = new Permit(hold!.amount, hold!.expires, meter.accountAllowance); + meter.permit = new Permit(hold!.amount, hold!.expires); })(); try { const result = await execute(); diff --git a/src/spending/permit.ts b/src/spending/permit.ts index 662944f..4695ec5 100644 --- a/src/spending/permit.ts +++ b/src/spending/permit.ts @@ -1,21 +1,18 @@ import { addTokens, type Meter, type ModelTokenUsage } from "../cost"; -import { priceTokens, type TokenRateCard } from "../server/token-pricing"; import { modelPrice, SpendingError, type Provider } from "./policy"; export class Permit { used = 0; - retailHeld = 0; tokens: ModelTokenUsage[] = []; unknown = false; error?: SpendingError; private closed = false; close() { this.closed = true; } private pending = new Set>(); - constructor(readonly amount: number, readonly expires: number, private retail?: { card: TokenRateCard; credits: number }) {} + constructor(readonly amount: number, readonly expires: number) {} async drain() { while (this.pending.size) await Promise.allSettled([...this.pending]); } async fetch(provider: Provider, model: string, output: number, input: RequestInfo | URL, init: RequestInit): Promise { let bound: number; - let retailBound = 0; try { if (this.closed) throw new SpendingError(402, "request_finished", "The request allowance has closed."); if (this.error) throw this.error; @@ -24,12 +21,6 @@ export class Permit { while (this.used + bound > this.amount && this.pending.size && bound <= this.amount && !this.closed) await Promise.race([...this.pending].map(p => p.catch(() => {}))); if (this.closed || Date.now() >= this.expires || this.used + bound > this.amount) throw new SpendingError(402, "request_spending_limit", "This request cannot fit another provider attempt within its allowance.", { limitUsd: this.amount / 1e9, spentOrReservedUsd: this.used / 1e9, requiredAttemptUsd: bound / 1e9 }); - if (this.retail) { - const priced = priceTokens(this.retail.card, [{ provider, model, calls: 1, inputTokens: price.context, outputTokens: output, cachedInputTokens: 0 }]); - if (!priced) throw new SpendingError(503, "unpriced_model", "This model is not configured for account billing."); - retailBound = Number((priced.nanodollars + 9999n) / 10000n); - if (this.retailHeld + retailBound > this.retail.credits) throw new SpendingError(402, "request_spending_limit", "The request exceeds its reserved balance."); - } if (provider === "openrouter") { const body = JSON.parse(String(init.body)); body.provider = { ...body.provider, max_price: { prompt: price.input, completion: price.output }, require_parameters: true }; @@ -40,7 +31,6 @@ export class Permit { throw this.error; } this.used += bound; - this.retailHeld += retailBound; const execute = async () => { try { const response = await fetch(input, { ...init, signal: init.signal ? AbortSignal.any([init.signal, AbortSignal.timeout(30000)]) : AbortSignal.timeout(30000) }); @@ -53,7 +43,6 @@ export class Permit { // TypeSafe rejects authentication/credit failures before inference. if (provider === "typesafe" && [401, 402, 403].includes(response.status) && payload?.detail && !usage) { this.used -= bound; - this.retailHeld -= retailBound; return response; } const count = (v: unknown) => typeof v === "number" && Number.isSafeInteger(v) && v >= 0 ? v : null; @@ -65,11 +54,6 @@ export class Permit { const details = usage?.prompt_tokens_details as { cached_tokens?: unknown } | undefined; const meter = { usd: 0, tokens: this.tokens }; addTokens(meter, provider, model, { inputTokens, outputTokens, cachedInputTokens: provider === "openrouter" ? count(details?.cached_tokens) ?? 0 : 0 }); - if (this.retail) { - const actual = priceTokens(this.retail.card, [{ provider, model, calls: 1, inputTokens, outputTokens, cachedInputTokens: provider === "openrouter" ? count(details?.cached_tokens) ?? 0 : 0 }]); - if (actual) this.retailHeld -= retailBound - Number((actual.nanodollars + 9999n) / 10000n); - else this.unknown = true; - } } else this.unknown = true; return response; } catch (error) { this.unknown = true; throw error; } diff --git a/tests/dashboard-analytics.test.ts b/tests/dashboard-analytics.test.ts index a8bfe71..be848f8 100644 --- a/tests/dashboard-analytics.test.ts +++ b/tests/dashboard-analytics.test.ts @@ -89,9 +89,9 @@ test("hosted dashboard never scans the usage ledger and renders loading rather t navigate() {}, }), ); - expect(plans).toContain("Token prices"); + expect(plans).toContain("Simple usage prices"); expect(plans).toContain("$0.042"); - expect(plans).toContain("$4.5"); + expect(plans).toContain("+$2.00 / 1,000"); expect(plans).not.toContain("Upgrade to Max"); expect(plans).not.toContain("Upgrade to Scale"); }); diff --git a/tests/plans-prices.test.ts b/tests/plans-prices.test.ts index a959ef5..11030cf 100644 --- a/tests/plans-prices.test.ts +++ b/tests/plans-prices.test.ts @@ -6,12 +6,13 @@ import { getSnapshot } from "../src/server/accounts"; import { provisionTestAccount } from "./support/account"; import { database } from "./support/postgres"; -test("account plans identify the Beam trial rates without advertising free Gemini reviews", async () => { +test("account plans show one input rate and one successful escalation price", async () => { const env = { APP_DB: database(), APP_ACCOUNTS_ENABLED: "true" }; await provisionTestAccount(new Request("http://localhost/login"), env); const html = renderToStaticMarkup(createElement(Plans, { snapshot: await getSnapshot("local-demo", env), navigate: () => {} })); const names = Array.from(html.matchAll(/]*>(.*?)<\/th>/g), match => match[1]); - expect(names).toEqual(["Jev", "Gemini escalation", "jev/laya", "jev/kev", "ibm-granite/granite-4.0-h-micro", "deepseek/deepseek-v4-flash", "inclusionai/ling-3.0-flash", "inception/mercury-2.5", "ibm-granite/granite-4.2-8b"]); - expect(html).toContain("Laya inference is free during the trial"); - expect(html).toContain("Smart reviews are still billed"); + expect(names).toEqual(["Input tokens", "Smart escalations"]); + expect(html).toContain("$0.042"); + expect(html).toContain("+$2.00 / 1,000"); + expect(html).toContain("Output tokens are free"); }); diff --git a/tests/public-pricing.test.ts b/tests/public-pricing.test.ts index 048ed28..9cf7136 100644 --- a/tests/public-pricing.test.ts +++ b/tests/public-pricing.test.ts @@ -83,8 +83,8 @@ test("pricing renders workspace plans and Pro rate limits", () => { expect(html).toContain('class="plan-action" href="/auth/sign-up"'); expect(html).toContain(`$${BILLING_PLANS.pro.priceCents / 100}`); expect(html).toContain(formatCreditsUsd(BILLING_PLANS.pro.includedCredits)); - expect(html).toContain("jev/laya"); - expect(html).toContain("jev/kev"); + expect(html).toContain("$0.042 / million"); + expect(html).toContain("+$2.00 / 1,000"); expect(html).toContain('href="/auth/sign-up?returnTo=/app/plans"'); expect(html).not.toContain("classifier_pro_"); expect(html).toContain("30,000/min · 200,000/day"); diff --git a/tests/spending.e2e.test.ts b/tests/spending.e2e.test.ts index ee24c73..8bfec47 100644 --- a/tests/spending.e2e.test.ts +++ b/tests/spending.e2e.test.ts @@ -159,7 +159,7 @@ test("uncertain attempts consume allowance, retries cannot exceed a penny, and f expect((await worker.fetch(request(), s.env, s.ctx)).status).toBe(429); }); -test("funded HTTP classification skips Spur/free budget, reserves once, returns before settlement, and bills actual tokens", async () => { +test("funded HTTP classification skips Spur/free budget and bills the published input-token price", async () => { const s = setup(); const env = { ...s.env, APP_DB: database(), APP_ACCOUNTS_ENABLED: "true", API_KEY_ENCRYPTION_KEY: "test-only-key-encryption-secret-32-characters" } as Env & AppEnv; await provisionTestAccount(new Request("http://localhost/auth/demo", { headers: { origin: "http://localhost" } }), env); @@ -279,7 +279,7 @@ test("funded requests accept work above the free request ceiling and concurrent expect((await worker.fetch(request(undefined, undefined, body), env, s.ctx)).status).toBe(402); const large = await accountClassification(request(undefined, undefined, body, { authorization: `Bearer ${key.secret}` }), env, "API", s.ctx); expect(large?.status).toBe(200); await s.flush(); - await env.APP_DB.prepare("UPDATE app_accounts SET balance=1000,paid_balance=1000,fractional_spend_nano=0 WHERE id='local-demo'").run(); + await env.APP_DB.prepare("UPDATE app_accounts SET balance=276,paid_balance=276,fractional_spend_nano=0 WHERE id='local-demo'").run(); const second = await performAction("local-demo", { type: "enroll", client: "Claude" }, env); let release!: () => void; const barrier = new Promise(resolve => { release = resolve; }); @@ -292,7 +292,7 @@ test("funded requests accept work above the free request ceiling and concurrent expect(statuses.filter(status => status === 402)).toHaveLength(11); await s.flush(); const balance = await env.APP_DB.prepare("SELECT balance FROM app_accounts WHERE id='local-demo'").first<{ balance: number }>(); - expect(Number(balance!.balance)).toBe(999); + expect(Number(balance!.balance)).toBe(275); expect(s.stored.size).toBe(0); }); @@ -312,3 +312,57 @@ test.skipIf(process.env.LIVE_TOKEN_BILLING !== "true")("live TypeSafe inference expect(Number(row!.nano)).toBeGreaterThan(0); expect(s.stored.size).toBe(0); }); + +test("Smart bills only successful escalations at the fixed price, independent of provider token usage", async () => { + const s = setup(); + const env = { ...s.env, APP_DB: database(), APP_ACCOUNTS_ENABLED: "true", API_KEY_ENCRYPTION_KEY: "test-only-key-encryption-secret-32-characters" } as Env & AppEnv; + await provisionTestAccount(new Request("http://localhost/auth/demo", { headers: { origin: "http://localhost" } }), env); + await env.APP_DB.prepare("UPDATE app_accounts SET paid_balance=balance WHERE id='local-demo'").run(); + const key = await performAction("local-demo", { type: "enroll", client: "Codex" }, env); + globalThis.fetch = (async (_url, init) => { + const b = JSON.parse(String(init?.body)); + if (b.model === "google/gemini-3.8-flash") return Response.json({ model: b.model, choices: [{ message: { content: "A" } }], usage: { prompt_tokens: 700, completion_tokens: 1000, cost: 0.004275 } }); + return Response.json({ model: "jev-1.13.0", usage: { input_tokens: 9000, output_tokens: 0 }, answers: Object.fromEntries(Object.entries(b.questions).map(([id, q], i) => { + const keys = Object.keys((q as { criteria: object }).criteria); const score = i === 0 ? 0.51 : 0.99; + return [id, { choice: keys[0], confidence: score, probabilities: Object.fromEntries(keys.map((k, j) => [k, j ? 1-score : score])) }]; + })) }); + }) as typeof fetch; + const response = await accountClassification(request(undefined, undefined, { inputs: ["Unclear request", "Invoice", "Payment"], labels: ["billing", "support"], tier: "smart" }, { authorization: `Bearer ${key.secret}` }), env, "API", s.ctx); + expect(response?.status).toBe(200); + expect((await response!.json() as { usage: { escalated: number } }).usage.escalated).toBe(1); + await s.flush(); + expect(await env.APP_DB.prepare("SELECT credits::integer AS credits,actual_nano::text AS nano,metering_mode FROM app_usage WHERE id=?").bind(response!.headers.get("x-request-id")).first()).toEqual({ credits: 238, nano: "2378000", metering_mode: "tokens" }); +}); + +test("failed Smart reviews charge only base input and a failed request returns its entire hold", async () => { + const s = setup(); + const env = { ...s.env, APP_DB: database(), APP_ACCOUNTS_ENABLED: "true", API_KEY_ENCRYPTION_KEY: "test-only-key-encryption-secret-32-characters" } as Env & AppEnv; + await provisionTestAccount(new Request("http://localhost/auth/demo", { headers: { origin: "http://localhost" } }), env); + await env.APP_DB.prepare("UPDATE app_accounts SET paid_balance=balance WHERE id='local-demo'").run(); + const key = await performAction("local-demo", { type: "enroll", client: "Codex" }, env); + const primary = providers(); + const successful = globalThis.fetch; + globalThis.fetch = (async (url, init) => { + if (String(url).includes("openrouter.ai")) return Response.json({ error: "unavailable" }, { status: 403 }); + const response = await successful(url, init); + const body = await response.json() as { answers: Record }; + for (const answer of Object.values(body.answers)) answer.confidence = 0.51; + return Response.json(body); + }) as typeof fetch; + const headers = { authorization: `Bearer ${key.secret}` }; + const response = await accountClassification(request(undefined, undefined, { input: "Ambiguous invoice", labels: ["billing", "support"], tier: "smart" }, headers), env, "API", s.ctx); + expect(response?.status).toBe(200); + expect(response!.headers.get("x-usage-cost-usd")).toBe("0.000004200"); + const body = await response!.json() as { usage: { escalated: number; escalation_failed: number } }; + expect(body.usage.escalated).toBe(0); + expect(body.usage.escalation_failed).toBe(1); + await s.flush(); + expect(primary).toHaveLength(1); + const before = await env.APP_DB.prepare("SELECT balance::text FROM app_accounts WHERE id='local-demo'").first(); + globalThis.fetch = (async () => Response.json({ detail: { error_type: "insufficient_credits" } }, { status: 402 })) as typeof fetch; + const failed = await accountClassification(request(undefined, "/v1/systemone", { model: "jev-latest", state: [], questions: {} }, headers), env, "API", s.ctx); + expect(failed?.status).toBe(402); + await s.flush(); + expect(await env.APP_DB.prepare("SELECT balance::text FROM app_accounts WHERE id='local-demo'").first()).toEqual(before); + expect(await env.APP_DB.prepare("SELECT status FROM app_usage WHERE id=?").bind(failed!.headers.get("x-request-id")).first()).toEqual({ status: "refunded" }); +}); From 8078fa759df769bb998b6e0d94ff9607f8eae95e Mon Sep 17 00:00:00 2001 From: Michael Ryaboy Date: Tue, 22 Sep 2026 01:34:06 -0700 Subject: [PATCH 2/2] Expose customer usage headers to browser API clients --- src/http/account-api.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/http/account-api.ts b/src/http/account-api.ts index aff320a..763f94d 100644 --- a/src/http/account-api.ts +++ b/src/http/account-api.ts @@ -25,6 +25,6 @@ export async function accountApi(request: Request, env: AppEnv & Env, ctx: Execu const headers = new Headers(response.headers); headers.set("access-control-allow-origin", "*"); const exposed = headers.get("access-control-expose-headers"); - headers.set("access-control-expose-headers", [exposed, "x-request-id", "x-billing-status"].filter(Boolean).join(", ")); + headers.set("access-control-expose-headers", [exposed, "x-request-id", "x-billing-status", "x-billed-input-tokens", "x-smart-escalations", "x-usage-cost-usd"].filter(Boolean).join(", ")); return new Response(response.body, { status: response.status, headers }); }