Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions drizzle/schema.ts
Original file line number Diff line number Diff line change
Expand Up @@ -108,6 +108,8 @@ export const app_usage = pgTable("app_usage", {
actual_nano: bigint({ mode: "bigint" }),
rate_version: text(),
metering_mode: text().default('credits').notNull(),
classifications: integer(),
escalations: integer(),
reporting_status: text().default('not_ready').notNull(),
}, (table) => [
index("app_usage_account").using("btree", table.account_id.asc().nullsLast().op("text_ops"), table.created_at.asc().nullsLast().op("text_ops")),
Expand All @@ -124,6 +126,8 @@ export const app_usage = pgTable("app_usage", {
name: "app_usage_agent_id_fkey"
}),
check("app_usage_actual_nano_check", sql`actual_nano >= 0`),
check("app_usage_classifications_check", sql`classifications >= 0`),
check("app_usage_escalations_check", sql`escalations >= 0`),
check("app_usage_metering_mode_check", sql`metering_mode = ANY (ARRAY['credits'::text, 'tokens'::text, 'legacy'::text])`),
check("app_usage_reporting_status_check", sql`reporting_status = ANY (ARRAY['not_ready'::text, 'review'::text, 'pending'::text, 'reported'::text, 'exempt'::text])`),
]);
Expand Down
2 changes: 2 additions & 0 deletions migrations/postgres/0013_classification_pricing.sql
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
ALTER TABLE app_usage ADD COLUMN classifications INTEGER CHECK(classifications>=0);
ALTER TABLE app_usage ADD COLUMN escalations INTEGER CHECK(escalations>=0);
2 changes: 1 addition & 1 deletion src/alerts.ts
Original file line number Diff line number Diff line change
Expand Up @@ -140,7 +140,7 @@ export async function evaluate(env: Env): Promise<{ alerts: Alert[]; checked: bo
if (app.APP_ACCOUNTS_ENABLED === "true") {
const row = await app.APP_DB.prepare("SELECT count(*) AS count FROM app_usage WHERE metering_mode='tokens' AND (reporting_status='review' OR (status='pending' AND created_at::timestamptz<now()-interval '5 minutes'))").first<{ count: number }>();
if (Number(row?.count) > 0) pre.push({ id: "billing_review", severity: "warning", title: "billing reservations need review",
detail: `${row!.count} token reservations have uncertain usage or have remained pending for over five minutes. Funds remain held. Reconcile app_usage request IDs against provider usage before settling or refunding; do not blindly release these reservations.` });
detail: `${row!.count} usage reservations have uncertain outcomes or have remained pending for over five minutes. Funds remain held. Reconcile app_usage request IDs against provider usage before settling or refunding; do not blindly release these reservations.` });
}
} catch {
pre.push({ id: "spending_monitor", severity: "critical", title: "spending monitoring is unavailable", detail: "Check the free-budget Durable Object and account database. Admission remains fail-closed; this alert must clear before assuming spending and billing are healthy." });
Expand Down
1 change: 0 additions & 1 deletion src/cost.ts
Original file line number Diff line number Diff line change
Expand Up @@ -37,7 +37,6 @@ export type ModelTokenUsage = TokenCounts & {

/** Per-request accumulator. Token counts are upstream usage, not a retail price. */
export type Meter = {
accountAllowance?: { card: import("./server/token-pricing").TokenRateCard; credits: number };
permit?: import("./spending/permit").Permit;
usd: number;
tokens: ModelTokenUsage[];
Expand Down
49 changes: 9 additions & 40 deletions src/features/billing/plans.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,7 @@ import {
type BillingPlanId,
} from "@/lib/billing";
import type { AppSnapshot } from "@/server/contracts";
import retailRates from "../../retail-rates.json";
import { INPUT_PRICE_PER_MILLION, ESCALATION_PRICE_PER_THOUSAND } from "@/lib/classification-pricing";

const date = (value: string) =>
new Date(value).toLocaleDateString("en-US", {
Expand Down Expand Up @@ -250,50 +250,19 @@ export function Plans({
))}
</section>

<section aria-labelledby="token-prices" className="flex flex-col gap-4">
<h2 id="token-prices" className="text-lg font-semibold">
Token prices
</h2>
<p className="text-sm text-muted-foreground">
Prices per million tokens. Default Fast uses Jev at cost. Smart adds Gemini at
cost plus 20% only when it escalates. Laya inference is free during the trial;
Smart reviews are still billed.
</p>
<section aria-labelledby="usage-prices" className="flex flex-col gap-4">
<h2 id="usage-prices" className="text-lg font-semibold">Simple usage prices</h2>
<div className="overflow-x-auto rounded-xl border border-border">
<table className="w-full min-w-[540px] text-left text-sm">
<thead>
<tr className="bg-muted/20">
<th className="p-4">Model</th>
<th className="p-4">Input</th>
<th className="p-4">Cached input</th>
<th className="p-4">Output</th>
</tr>
</thead>
<table className="w-full text-left text-sm">
<thead><tr><th className="p-4">Usage</th><th className="p-4">Price</th></tr></thead>
<tbody>
{retailRates.models.map((rate) => (
<tr key={rate.model} className="border-t border-border">
<th scope="row" className="p-4 font-medium">
{rate.provider === "typesafe" ? "Jev" : rate.model === "google/gemini-3.8-flash" ? "Gemini escalation" : rate.model}
</th>
{[
rate.inputUsdPerMillion,
rate.cachedInputUsdPerMillion,
rate.outputUsdPerMillion,
].map((value, index) => (
<td key={index} className="p-4 tabular-nums">
{Number(value) === 0 ? "Free" : `$${Number(value)}`}
</td>
))}
</tr>
))}
<tr className="border-t border-border"><th scope="row" className="p-4 font-medium">Input tokens</th><td className="p-4 tabular-nums">${INPUT_PRICE_PER_MILLION.toFixed(3)} / million</td></tr>
<tr className="border-t border-border"><th scope="row" className="p-4 font-medium">Smart escalations</th><td className="p-4 tabular-nums">+${ESCALATION_PRICE_PER_THOUSAND.toFixed(2)} / 1,000</td></tr>
</tbody>
</table>
</div>
<p className="text-xs text-muted-foreground">
Smart requests without escalation cost the same as Fast. Gemini output
includes reasoning tokens. Usage stops when your balance is depleted;
there are no automatic top-ups.
</p>
<p className="text-sm text-muted-foreground">Smart reviews uncertain answers. You pay extra only for successful escalations. For example, 1 million input tokens with 50 Smart escalations cost $0.142.</p>
<p className="text-xs text-muted-foreground">Output tokens are free. Input usage includes text, labels and instructions. Retries, fallback routing and Smart model tokens add no separate charges. No automatic top-ups.</p>
</section>

<Card className="shadow-none">
Expand Down
2 changes: 1 addition & 1 deletion src/http/account-api.ts
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,6 @@ export async function accountApi(request: Request, env: AppEnv & Env, ctx: Execu
const headers = new Headers(response.headers);
headers.set("access-control-allow-origin", "*");
const exposed = headers.get("access-control-expose-headers");
headers.set("access-control-expose-headers", [exposed, "x-request-id", "x-billing-status"].filter(Boolean).join(", "));
headers.set("access-control-expose-headers", [exposed, "x-request-id", "x-billing-status", "x-billed-input-tokens", "x-smart-escalations", "x-usage-cost-usd"].filter(Boolean).join(", "));
return new Response(response.body, { status: response.status, headers });
}
46 changes: 26 additions & 20 deletions src/http/spending-classification.ts
Original file line number Diff line number Diff line change
@@ -1,16 +1,14 @@
import worker, { type Env } from "../index";
import { newMeter } from "../cost";
import { AppError, hashToken, now, type AppEnv } from "../server/db";
import rates from "../retail-rates.json";
import { parseTokenRateCard, priceTokens } from "../server/token-pricing";
import { classificationCharge, classificationInputTokens } from "../lib/classification-pricing";
import { refundTokenReservation, settleTokenReservation } from "../server/token-ledger";
import { Permit } from "../spending/permit";
import { boundedRequest } from "../spending";
import { SpendingError, errorResponse, fingerprint, policy } from "../spending/policy";
import { writeAccountAnalytics } from "../server/analytics/write";
import { typeSafeDecisionCount } from "../typesafe-compat";

const card = parseTokenRateCard(JSON.stringify(rates))!;
export async function spendingClassification(request: Request, env: AppEnv & Partial<Env>, source: "API" | "MCP", ctx: ExecutionContext): Promise<Response> {
try {
if (env.APP_ACCOUNTS_ENABLED !== "true") throw new AppError(503, "Account credentials are disabled.");
Expand All @@ -29,25 +27,25 @@ export async function spendingClassification(request: Request, env: AppEnv & Par
const keyHash = await hashToken(request.headers.get("authorization")!.replace(/^Bearer\s+/i, ""));
const maxCredits = Math.ceil(limits.paidRequest / 10000);
const decisions = items * (body.dimensions && typeof body.dimensions === "object" ? Math.max(1, Object.keys(body.dimensions).length) : 1);
const bytes = new TextEncoder().encode(text.normalize("NFKC")).length;
const smartCredits = Math.ceil(((bytes * 2 + 4096) * 0.9 + 2000 * 4.5) * 0.1) * 3;
const quote = Math.min(maxCredits, Math.ceil(decisions * (body.tier === "smart" ? smartCredits + 1500 : 1500)));
const trial = body.model === "laya" || body.model === "kev";
const quote = Number((classificationCharge(trial ? 0 : 65536 * decisions, body.tier === "smart" ? decisions : 0).nanodollars + 9999n) / 10000n);
if (quote > maxCredits) throw new SpendingError(402, "request_spending_limit", "This request exceeds the workspace request allowance. Split the batch.");
const idempotencyHash = idem ? await fingerprint(env, `account-idempotency:${idem}`) : null;
const result = await env.APP_DB.prepare(`WITH owner AS (
SELECT a.id,a.balance,a.paid_balance,(a.paid_balance>0 OR (a.billing_plan IN ('pro','max','scale') AND a.reset_at::timestamptz>now())) AS funded,k.id AS agent_id FROM app_accounts a JOIN app_agents k ON k.account_id=a.id
WHERE k.token_hash=? AND k.status IN ('pending','connected') AND NOT a.billing_hold FOR UPDATE OF a
), held AS (
INSERT INTO app_usage(id,account_id,agent_id,items,credits,status,created_at,paid_credits,usage_type,metering_mode,idempotency_key)
SELECT ?,id,agent_id,?,LEAST(balance,CASE WHEN funded THEN ?::bigint ELSE ?::bigint END),'pending',?,
GREATEST(0,LEAST(balance,CASE WHEN funded THEN ?::bigint ELSE ?::bigint END)-(balance-paid_balance)),?,'tokens',?
FROM owner WHERE balance>0 ON CONFLICT DO NOTHING RETURNING *
INSERT INTO app_usage(id,account_id,agent_id,items,credits,status,created_at,paid_credits,usage_type,metering_mode,idempotency_key,classifications)
SELECT ?,id,agent_id,?,?::bigint,'pending',?,
GREATEST(0,?::bigint-(balance-paid_balance)),?,'tokens',?,?
FROM owner WHERE balance>0 AND balance>=? ON CONFLICT DO NOTHING RETURNING *
), debited AS (
UPDATE app_accounts a SET balance=a.balance-h.credits,paid_balance=a.paid_balance-h.paid_credits
FROM held h WHERE a.id=h.account_id RETURNING a.id
), agent AS (
UPDATE app_agents a SET used=a.used+h.credits FROM held h,debited d WHERE a.id=h.agent_id AND d.id=h.account_id RETURNING a.id
) SELECT h.account_id,h.agent_id,h.credits,o.funded FROM held h JOIN owner o ON o.id=h.account_id JOIN agent k ON k.id=h.agent_id`)
.bind(keyHash, id, items, quote, Math.ceil(limits.request * 1.5 / 10000), now(), quote, Math.ceil(limits.request * 1.5 / 10000), `${source} · Classification`, idempotencyHash)
.bind(keyHash, id, items, quote, now(), quote, `${source} · Classification`, idempotencyHash, decisions, quote)
.first<{ account_id: string; agent_id: string; credits: number; funded: boolean }>();
if (!result) {
const reason = await env.APP_DB.prepare(`SELECT k.status,a.billing_hold,
Expand All @@ -57,11 +55,10 @@ export async function spendingClassification(request: Request, env: AppEnv & Par
if (!reason) throw new SpendingError(401, "invalid_api_key", "This workspace API key is not valid.");
if (!["pending", "connected"].includes(reason.status)) throw new SpendingError(403, "inactive_api_key", "This workspace API key is paused or revoked.");
if (reason.request_id) throw new SpendingError(409, "duplicate_request", "This workspace already admitted the idempotency key.", { requestId: reason.request_id });
throw new SpendingError(402, "insufficient_balance", reason.billing_hold ? "Workspace billing is awaiting review; funds remain held." : "The workspace has no unreserved balance for this request.");
throw new SpendingError(402, "insufficient_balance", reason.billing_hold ? "Workspace billing is awaiting review; funds remain held." : "The workspace needs enough available balance for the maximum input tokens and possible Smart escalations. Unused funds are released after the request.", { requiredUsd: quote / 100000 });
}
const meter = newMeter();
meter.accountAllowance = { card, credits: result.credits };
const permit = result.funded ? new Permit(limits.paidRequest, Date.now() + 90000, { card, credits: result.credits }) : undefined;
const permit = result.funded ? new Permit(limits.paidRequest, Date.now() + 90000) : undefined;
if (permit) { meter.permit = permit; meter.beforeCall = async () => {}; }
const started = Date.now();
let response: Response;
Expand All @@ -71,32 +68,41 @@ export async function spendingClassification(request: Request, env: AppEnv & Par
} catch (error) {
response = error instanceof SpendingError ? errorResponse(error) : Response.json({ error: "Classification failed." }, { status: 502 });
}
const payload = response.ok ? await response.clone().json().catch(() => null) as { usage?: { escalated?: unknown } } | null : null;
const escalations = new URL(request.url).pathname === "/v1/systemone" ? 0 : payload?.usage?.escalated;
const inputTokens = trial ? 0 : classificationInputTokens(meter.tokens);
const charge = response.ok && inputTokens !== null && typeof escalations === "number" && Number.isSafeInteger(escalations) && escalations >= 0 && escalations <= decisions
? classificationCharge(inputTokens, escalations) : null;
const settle = async () => {
const actual = meter.permit;
actual?.close();
await actual?.drain();
const tokens = actual?.tokens ?? [];
const charge = actual?.unknown ? null : response.ok && tokens.length ? priceTokens(card, tokens) : { version: card.version, nanodollars: 0n };
// Uncertain attempts remain held for reconciliation, even on an HTTP error.
for (let attempt = 0; attempt < 3; attempt++) {
try {
if (!actual || actual.used === 0) await refundTokenReservation(env.APP_DB, id);
if (!response.ok || !actual || actual.used === 0) await refundTokenReservation(env.APP_DB, id);
else await settleTokenReservation(env.APP_DB, id, charge, {
inputTokens: tokens.length && tokens.every(t => t.inputTokens !== null) ? tokens.reduce((sum, t) => sum + t.inputTokens!, 0) : null,
inputTokens,
outputTokens: tokens.length && tokens.every(t => t.outputTokens !== null) ? tokens.reduce((sum, t) => sum + t.outputTokens!, 0) : null,
});
if (charge !== null) await env.APP_DB.prepare("UPDATE app_usage SET escalations=? WHERE id=? AND status='completed'").bind(escalations, id).run();
writeAccountAnalytics(env, { accountId: result.account_id, keyId: result.agent_id, requestId: id, source,
tier: body.tier === "smart" ? "smart" : "fast", status: response.ok ? "success" : "error", items,
inputTokens: tokens.every(t => t.inputTokens !== null) ? tokens.reduce((sum, t) => sum + t.inputTokens!, 0) : null,
inputTokens,
outputTokens: tokens.every(t => t.outputTokens !== null) ? tokens.reduce((sum, t) => sum + t.outputTokens!, 0) : null,
cachedInputTokens: null, model: tokens.map(t => t.model).join(","), providerCostUsd: actual?.unknown ? null : (actual?.used ?? 0) / 1e9,
retailCostUsd: charge ? Number(charge.nanodollars) / 1e9 : null, latencyMs: Date.now() - started, escalations: tokens.filter(t => t.provider === "openrouter").reduce((n, t) => n + t.calls, 0) });
retailCostUsd: !response.ok ? 0 : charge ? Number(charge.nanodollars) / 1e9 : null, latencyMs: Date.now() - started, escalations: typeof escalations === "number" ? escalations : 0 });
return;
} catch { /* Durable pending funds remain unavailable until reconciliation. */ }
}
};
ctx.waitUntil(settle());
const headers = new Headers(response.headers);
if (charge) {
headers.set("x-billed-input-tokens", String(inputTokens));
headers.set("x-smart-escalations", String(escalations));
headers.set("x-usage-cost-usd", (Number(charge.nanodollars) / 1e9).toFixed(9));
}
headers.set("x-request-id", id); headers.set("x-billing-status", "pending"); headers.set("cache-control", "no-store");
return new Response(response.body, { status: response.status, headers });
} catch (error) {
Expand Down
25 changes: 25 additions & 0 deletions src/lib/classification-pricing.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
import type { ModelTokenUsage } from "../cost";

export const CLASSIFICATION_PRICING = {
version: "2026-09-22-input-escalation-v1",
inputNanodollars: 42,
escalationNanodollars: 2_000_000,
} as const;

export const INPUT_PRICE_PER_MILLION = CLASSIFICATION_PRICING.inputNanodollars / 1000;
export const ESCALATION_PRICE_PER_THOUSAND = CLASSIFICATION_PRICING.escalationNanodollars / 1e6;

export function classificationCharge(inputTokens: number, escalations: number) {
if (!Number.isSafeInteger(inputTokens) || inputTokens < 0 || !Number.isSafeInteger(escalations) || escalations < 0)
throw new Error("Invalid classification billing counts.");
return { version: CLASSIFICATION_PRICING.version,
nanodollars: BigInt(inputTokens) * BigInt(CLASSIFICATION_PRICING.inputNanodollars)
+ BigInt(escalations) * BigInt(CLASSIFICATION_PRICING.escalationNanodollars) };
}

export function classificationInputTokens(tokens: ModelTokenUsage[]): number | null {
// Only answered primary calls count. Recovery and Smart provider costs are ours.
const primary = tokens.filter(row => row.provider === "typesafe" || row.provider === "vercel");
return primary.every(row => row.inputTokens !== null)
? primary.reduce((sum, row) => sum + row.inputTokens!, 0) : null;
}
3 changes: 3 additions & 0 deletions src/openapi.ts
Original file line number Diff line number Diff line change
Expand Up @@ -59,6 +59,9 @@ const TYPESAFE_REQUEST_ID_HEADER = {
};
const ACCOUNT_BILLING_HEADERS = {
"x-request-id": { schema: { type: "string" }, description: "Workspace usage request ID. Present when a classifier_agent_ key is used." },
"x-billed-input-tokens": { schema: { type: "integer", minimum: 0 }, description: "Successful base-classifier input tokens billed at the published rate; excludes Smart and fallback provider tokens." },
"x-smart-escalations": { schema: { type: "integer", minimum: 0 }, description: "Successful Smart reviews billed at the flat escalation price." },
"x-usage-cost-usd": { schema: { type: "string" }, description: "Customer charge in USD before credit rounding. Settlement may still be pending." },
"x-billing-status": { schema: { type: "string", enum: ["pending", "settled", "refunded", "review"] }, description: "Workspace charge result. Present when a classifier_agent_ key is used." },
};
const TYPESAFE_ENTRY = {
Expand Down
Loading
Loading