From dd228712ce76b20dc8ab49d3eb2db018913cf5da Mon Sep 17 00:00:00 2001 From: Michael Ryaboy Date: Wed, 23 Sep 2026 20:13:59 -0700 Subject: [PATCH 1/4] Connect image decisions to Beam DiffusionGemma --- .github/workflows/check.yml | 8 ++ e2e/diffusiongemma.live.py | 19 +++++ e2e/diffusiongemma.ts | 109 ++++++++++++++++++++++++++++ package.json | 1 + src/dgemma.ts | 45 +++++++----- src/docs.ts | 91 +++++++++++++++++------ src/http/spending-classification.ts | 2 +- src/index.ts | 8 +- src/openapi.ts | 8 +- src/pages.ts | 12 ++- src/retail-rates.json | 9 ++- src/server/token-reservation.ts | 1 + src/spending/policy.ts | 1 + wrangler.example.toml | 4 +- 14 files changed, 259 insertions(+), 59 deletions(-) create mode 100644 e2e/diffusiongemma.live.py create mode 100644 e2e/diffusiongemma.ts diff --git a/.github/workflows/check.yml b/.github/workflows/check.yml index 1e0123f..b66d467 100644 --- a/.github/workflows/check.yml +++ b/.github/workflows/check.yml @@ -24,6 +24,14 @@ jobs: with: bun-version: latest - run: npm ci + - name: Verify Beam image routing and SDK compatibility + run: npm run test:e2e:images + - name: Save image classification evidence + if: always() + uses: actions/upload-artifact@v4 + with: + name: diffusiongemma-e2e + path: captures/diffusiongemma-fixture.json - run: npm run typecheck - run: npm test env: diff --git a/e2e/diffusiongemma.live.py b/e2e/diffusiongemma.live.py new file mode 100644 index 0000000..4539c17 --- /dev/null +++ b/e2e/diffusiongemma.live.py @@ -0,0 +1,19 @@ +# /// script +# dependencies = ["typesafe-sdk==0.7.1"] +# /// +# First run the JS image E2E to create the fixture, then: +# uv run e2e/diffusiongemma.live.py (against npm run dev on port 3000) +import base64, os +from pathlib import Path +from typesafe_sdk import TypeSafeClient,Choice,Noul,Score +image='data:image/png;base64,'+base64.b64encode(Path('captures/diffusiongemma-red.png').read_bytes()).decode() +with TypeSafeClient(api_key=os.environ.get('CLASSIFIER_API_KEY', 'unused'),base_url=os.environ.get('CLASSIFIER_BASE_URL', 'http://127.0.0.1:3000')) as client: + r=client.system_one(model='jev/diffusiongemma',state='Look at the image.',timeout=60,extra_body={'images':[image]},questions={ + 'color':Choice(instructions='What color is the image?',criteria={'red':None,'blue':None}), + 'red':Noul(instructions='Is the image red?'), + 'intensity':Score(instructions='How red is the image?',criteria=['Not red','Some red','Entirely red'])}) + print('Python TypeSafe SDK image E2E passed') + Path('captures/diffusiongemma-python.json').write_text(r.model_dump_json(indent=2)) + assert r.choices['color'].choice=='red' + assert r.nouls['red'].noul > 0.9 + assert r.scores['intensity'].score > 1.8 diff --git a/e2e/diffusiongemma.ts b/e2e/diffusiongemma.ts new file mode 100644 index 0000000..86cbea8 --- /dev/null +++ b/e2e/diffusiongemma.ts @@ -0,0 +1,109 @@ +import assert from "node:assert/strict"; +import { mkdir, writeFile } from "node:fs/promises"; +import { deflateSync } from "node:zlib"; +import { TypeSafeClient } from "@typesafe-ai/sdk"; +import worker, { type Env } from "../src/index"; +import { newMeter } from "../src/cost"; +import { Permit } from "../src/spending/permit"; +import { parseTokenRateCard, priceTokens } from "../src/server/token-pricing"; +import { providerCallBound } from "../src/server/token-reservation"; +import rates from "../src/retail-rates.json"; + +// bun --env-file=.dev.vars e2e/diffusiongemma.ts --live +// SDK → local HTTP Worker → Beam (or a local HTTP provider fixture). +// Failure modes: dropped images, text-only fallback, wrong model/auth, excess +// images, malformed replies, provider refusal, cancellation, and unmetered spend. +const live = process.argv.includes("--live"); +if (live && !process.env.BEAM_API_KEY) throw new Error("BEAM_API_KEY is required for --live"); +const originalFetch = globalThis.fetch; +let scenario = "success", calls = 0; +const provider = Bun.serve({ hostname: "127.0.0.1", port: 0, async fetch(req) { + calls++; + assert.equal(req.headers.get("authorization"), "Bearer fixture-beam"); + const body = await req.json() as Record; + assert.equal(body.model, "jev/diffusiongemma"); + if (scenario === "busy") return Response.json({ detail: "busy" }, { status: 429 }); + if (scenario === "invalid") return Response.json({ detail: "invalid image" }, { status: 422 }); + if (scenario === "malformed") return Response.json({ model: body.model, answers: {} }); + return Response.json({ model: body.model, answers: { + color: { type: "choice", choice: body.images ? "red" : "blue", confidence: 0.99, probabilities: { red: body.images ? 0.99 : 0.01, blue: body.images ? 0.01 : 0.99 } }, + red: { type: "noul", noul: 0.99 }, + intensity: { type: "score", score: 1.99, confidence: 0.99, legend: { 0: "Not red", 1: "Some red", 2: "Entirely red" }, probabilities: { 0: 0, 1: 0.01, 2: 0.99 } }, + }, usage: { input_tokens: 310, output_tokens: 0 } }); +} }); +if (!live) globalThis.fetch = ((input, init) => { + if (String(input).startsWith("http://127.0.0.1:")) return originalFetch(input, init); + assert.equal(String(input), "https://app.beam.cloud/v1/systemone", "never fall back to a text model or pod"); + return originalFetch(provider.url, init); +}) as typeof fetch; +const env = { DGEMMA_ENABLED: "true", BEAM_API_KEY: live ? process.env.BEAM_API_KEY : "fixture-beam", + DGEMMA_URL: "https://unused-pod.example", DGEMMA_TOKEN: "unused", + STATS: { get: async () => null, put: async () => {} }, + LIMITER: { idFromName: (s: string) => s, get: () => ({ fetch: async () => Response.json({ limited: false, remaining: 59 }) }) }, +} as unknown as Env; +const report: unknown[] = []; +const server = Bun.serve({ hostname: "127.0.0.1", port: 0, async fetch(req) { + const meter = newMeter(); + meter.permit = new Permit(10_000_000, Date.now() + 90000); + const response = await worker.fetch(req, env, { waitUntil: () => {} } as unknown as ExecutionContext, { meter }); + await meter.permit.drain(); + report.push({ status: response.status, response: await response.clone().json(), providerUsd: meter.usd, + tokens: meter.tokens, spending: { used: meter.permit.used, unknown: meter.permit.unknown } }); + if (response.ok) { + assert.equal(meter.tokens[0].provider, "beam"); + assert.equal(meter.tokens[0].model, "jev/diffusiongemma"); + assert.ok(meter.usd > 0); + assert.equal(meter.permit.unknown, false); + assert.ok(Math.abs(meter.permit.used - meter.tokens[0].inputTokens! * 21) <= 1); + } + return response; +} }); +// Deterministic 128×128 red PNG, constructed without external image assets. +function chunk(type: string, data: Buffer) { + const name = Buffer.from(type), crc = Bun.hash.crc32(Buffer.concat([name, data])); + const size = Buffer.alloc(4), checksum = Buffer.alloc(4); + size.writeUInt32BE(data.length); checksum.writeUInt32BE(crc); + return Buffer.concat([size, name, data, checksum]); +} +const header = Buffer.alloc(13); header.writeUInt32BE(128); header.writeUInt32BE(128, 4); header[8] = 8; header[9] = 2; +const row = Buffer.from([0, ...Array.from({ length: 128 }, () => [255, 0, 0]).flat()]); +const png = Buffer.concat([Buffer.from("89504e470d0a1a0a", "hex"), chunk("IHDR", header), chunk("IDAT", deflateSync(Buffer.concat(Array(128).fill(row)))), chunk("IEND", Buffer.alloc(0))]); +const image = `data:image/png;base64,${png.toString("base64")}`; +const questions = { + color: { type: "choice" as const, instructions: "What color is the image or described square?", criteria: { red: null, blue: null } }, + red: { type: "noul" as const, instructions: "Is the image red?" }, + intensity: { type: "score" as const, instructions: "How red is the image?", criteria: ["Not red", "Some red", "Entirely red"] as const }, +}; +const client = new TypeSafeClient({ apiKey: "unused", baseURL: server.url.origin, retry: { maxRetries: 0 }, timeout: 65000 }); +try { + const request = { model: "jev/diffusiongemma", state: "Look at the image.", images: [image], questions }; + const result = await client.systemOne(request); + assert.equal(result.answers.color.choice, "red"); + assert.ok(result.answers.red.noul > 0.9); + assert.ok(result.answers.intensity.score > 1.8); + const alias = await client.systemOne({ ...request, model: "dgemma" }); + assert.equal(alias.model, "jev/diffusiongemma"); + const text = await client.systemOne({ model: "jev/diffusiongemma", state: "The square is blue.", questions }); + assert.equal(text.answers.color.choice, "blue"); + const post = (body: object) => fetch(new URL("/v1/systemone", server.url), { method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify(body) }); + for (const change of [{ images: [image, image] }, { images: [] }, { images: ["https://example.com/a.png"] }, { model: "jev-latest" }]) { + const before = calls; + assert.equal((await post({ ...request, ...change })).status, 400); + assert.equal(calls, before, "invalid requests must not reach the provider"); + } + if (!live) for (const [mode, status] of [["busy", 429], ["invalid", 400], ["malformed", 503]] as const) { + scenario = mode; + const response = await post(request); + assert.equal(response.status, status); + if (mode === "busy") assert.ok(response.headers.get("retry-after")); + } + const card = parseTokenRateCard(JSON.stringify(rates))!; + assert.equal(providerCallBound(card, "beam", "jev/diffusiongemma", 0), 0); + assert.equal(priceTokens(card, [{ provider: "beam", model: "jev/diffusiongemma", calls: 1, inputTokens: 310, outputTokens: 0, cachedInputTokens: 0 }])?.nanodollars, 0n); + console.log(`DiffusionGemma ${live ? "live" : "fixture"} E2E passed (${report.length} HTTP requests)`); +} finally { + await mkdir("captures", { recursive: true }); + await writeFile(`captures/diffusiongemma-${live ? "live" : "fixture"}.json`, JSON.stringify({ live, sdk: "@typesafe-ai/sdk 0.6.0", results: report }, null, 2)); + await writeFile("captures/diffusiongemma-red.png", png); + server.stop(true); provider.stop(true); globalThis.fetch = originalFetch; +} diff --git a/package.json b/package.json index 31253e2..3687f62 100644 --- a/package.json +++ b/package.json @@ -27,6 +27,7 @@ "test:e2e:long-context-billing": "bun e2e/long-context-billing.ts", "test:e2e:long-context-job": "bun e2e/long-context-job.ts --full", "test:e2e:whole-document": "node e2e/whole-document.mjs --full", + "test:e2e:images": "bun e2e/diffusiongemma.ts", "test:e2e:url": "bun e2e/url-classification.ts" }, "keywords": [], diff --git a/src/dgemma.ts b/src/dgemma.ts index 54ebaad..6ee131d 100644 --- a/src/dgemma.ts +++ b/src/dgemma.ts @@ -1,5 +1,5 @@ import { providerFetch } from "./spending/permit"; -import { addTokens, type Meter } from "./cost"; +import { addBeamCost, addTokens, type Meter } from "./cost"; /** * The image door of POST /v1/systemone. @@ -13,6 +13,14 @@ import { addTokens, type Meter } from "./cost"; */ export const DGEMMA_MODEL = "dgemma"; +export const BEAM_DGEMMA_MODEL = "jev/diffusiongemma"; +export type DgemmaService = { url: string; token: string; model?: typeof BEAM_DGEMMA_MODEL }; + +export function dgemmaService(env: { DGEMMA_ENABLED?: string; BEAM_API_KEY?: string; DGEMMA_URL?: string; DGEMMA_TOKEN?: string }): DgemmaService | undefined { + if (env.DGEMMA_ENABLED !== "true") return; + if (env.BEAM_API_KEY) return { url: "https://app.beam.cloud", token: env.BEAM_API_KEY, model: BEAM_DGEMMA_MODEL }; + if (env.DGEMMA_URL && env.DGEMMA_TOKEN) return { url: env.DGEMMA_URL, token: env.DGEMMA_TOKEN }; +} export const DGEMMA_MAX_IMAGES = 4; /** Base64 characters across all images. The route's body is bounded at 1 MB regardless. */ export const DGEMMA_MAX_IMAGE_CHARS = 900_000; @@ -32,7 +40,7 @@ function isRecord(value: unknown): value is Record { const refuse = (code: DgemmaRefusalCode, message: string): DgemmaRoute => ({ kind: "refuse", status: 400, code, message }); /** Which door a System One body goes through. A body that is not clearly ours stays TypeSafe's to validate. */ -export function dgemmaRoute(body: string): DgemmaRoute { +export function dgemmaRoute(body: string, maxImages = DGEMMA_MAX_IMAGES): DgemmaRoute { let parsed: unknown; try { parsed = JSON.parse(body); } catch { return { kind: "typesafe" }; } if (!isRecord(parsed)) return { kind: "typesafe" }; @@ -40,14 +48,14 @@ export function dgemmaRoute(body: string): DgemmaRoute { // Presence selects the door; what the field holds is validated after, so an // empty or malformed images field never travels to TypeSafe as an unknown key. const hasImages = images !== undefined && images !== null; - const named = parsed.model === DGEMMA_MODEL; + const named = parsed.model === DGEMMA_MODEL || parsed.model === BEAM_DGEMMA_MODEL; if (!hasImages && !named) return { kind: "typesafe" }; if (hasImages && parsed.model !== undefined && !named) { return refuse("images_unsupported", `Jev does not accept images; set model to "${DGEMMA_MODEL}" or remove images`); } if (hasImages) { - if (!Array.isArray(images) || images.length === 0 || images.length > DGEMMA_MAX_IMAGES) { - return refuse("dgemma_input", `images must be an array of 1 to ${DGEMMA_MAX_IMAGES} data URLs`); + if (!Array.isArray(images) || images.length === 0 || images.length > maxImages) { + return refuse("dgemma_input", `images must be an array of 1 to ${maxImages} data URLs`); } let chars = 0; for (const image of images) { @@ -82,7 +90,7 @@ export const dgemmaUnconfigured = () => unavailable(60); /** Forward one System One body to the service and answer in the route's shapes. */ export async function dgemmaResponse( - pod: { url: string; token: string }, + pod: DgemmaService, body: string, meter?: Meter, signal?: AbortSignal, @@ -94,26 +102,25 @@ export async function dgemmaResponse( try { origin = new URL(pod.url); } catch { return unavailable(60); } if (origin.protocol !== "https:") return unavailable(60); const url = pod.url.replace(/\/+$/, "") + "/v1/systemone"; - const controller = new AbortController(); - const timer = setTimeout(() => controller.abort(), TIMEOUT_MS); - signal?.addEventListener("abort", () => controller.abort()); - await meter?.beforeCall?.("dgemma", DGEMMA_MODEL, 0); + const model = pod.model ?? DGEMMA_MODEL; + const provider = pod.model ? "beam" : "dgemma"; + const sent = JSON.stringify({ ...JSON.parse(body), model }); + await meter?.beforeCall?.(provider, model, 0); + const deadline = AbortSignal.timeout(TIMEOUT_MS); let res: Response; let payload: unknown = null; try { - res = await providerFetch(meter, "dgemma", DGEMMA_MODEL, 0, url, { + res = await providerFetch(meter, provider, model, 0, url, { method: "POST", headers: { authorization: `Bearer ${pod.token}`, "content-type": "application/json" }, - body, + body: sent, redirect: "manual", - signal: controller.signal, + signal: signal ? AbortSignal.any([signal, deadline]) : deadline, }); // The deadline covers the body too: headers followed by a stalled body is still a dead pod. try { payload = JSON.parse(await res.text()); } catch { /* handled by status below */ } } catch { return unavailable(10); - } finally { - clearTimeout(timer); } if (res.ok) { // A 200 is only a 200 when it is the whole contract: this model, an entry @@ -121,10 +128,12 @@ export async function dgemmaResponse( // and a usage count. Anything less is an outage. const asked = questionIds(body); const usage = isRecord(payload) && isRecord(payload.usage) ? payload.usage : null; - if (!isRecord(payload) || payload.model !== DGEMMA_MODEL || !isRecord(payload.answers) || !usage - || !Number.isInteger(usage.input_tokens) || !Number.isInteger(usage.output_tokens) + if (!isRecord(payload) || payload.model !== model || !isRecord(payload.answers) || !usage + || !Number.isSafeInteger(usage.input_tokens) || (usage.input_tokens as number) < 0 + || !Number.isSafeInteger(usage.output_tokens) || (usage.output_tokens as number) < 0 || asked.some((id) => !Object.prototype.hasOwnProperty.call(payload.answers, id))) return unavailable(10); - if (meter) addTokens(meter, "dgemma", DGEMMA_MODEL, { inputTokens: usage.input_tokens, outputTokens: usage.output_tokens }); + if (provider === "beam") addBeamCost(meter, usage.input_tokens); + addTokens(meter, provider, model, { inputTokens: usage.input_tokens, outputTokens: usage.output_tokens, cachedInputTokens: 0 }); return Response.json(payload, { headers: NO_STORE }); } // The service's validation messages are about the caller's own schema, so they travel; nothing else does. diff --git a/src/docs.ts b/src/docs.ts index 146ff3c..c1ab9b7 100644 --- a/src/docs.ts +++ b/src/docs.ts @@ -64,6 +64,74 @@ export const URL_CLASSIFICATION = `SCRAPE AND CLASSIFY A URL are not stored in the usage ledger. MCP accepts url and include on classify_texts, classify_dimensions and classify_multi_label.`; +export const IMAGE_CLASSIFICATION = `IMAGE CLASSIFICATION + + Classify an inline image at POST /v1/systemone with model "jev/diffusiongemma" + ("dgemma" is an alias). Beam hosts DiffusionGemma 26B-A4B: text and an image + in, typed decisions and probabilities out. It does not generate images. + Text-only requests can select this model too. It is experimental; Jev's + calibration measurements do not apply to its probabilities. + + Send one PNG, JPEG, WebP or GIF as a base64 data URL in images. Remote URLs + are not fetched. The image URL must fit 900,000 characters and the entire + JSON body must fit 1 MB. Beam accepts at most 32 questions in a request, + within its 32,768-token rendered context. Each question uses the existing + Choice, Noul or Score shape: + + {"model": "jev/diffusiongemma", + "state": "Look at the attached image.", + "images": ["data:image/png;base64,iVBORw0KGgo..."], + "questions": {"color": {"type": "choice", + "instructions": "What color is the image?", + "criteria": {"red": null, "blue": null}}}} + + The official TypeSafe SDKs can send images as an extra request field. + JavaScript has no images property in SystemOneRequest; use a variable so + TypeScript accepts the additional field. The SDK forwards it unchanged: + + import { readFileSync } from "node:fs"; + import { TypeSafeClient, choice } from "@typesafe-ai/sdk"; + + const client = new TypeSafeClient({ + apiKey: process.env.CLASSIFIER_API_KEY ?? "unused", + baseURL: "https://classifier.dev", + timeout: 60000, + }); + const request = { + model: "jev/diffusiongemma", + state: "Look at the image.", + images: ["data:image/png;base64," + readFileSync("image.png").toString("base64")], + questions: { color: choice("What color?", { red: null, blue: null }) }, + }; + console.log((await client.systemOne(request)).answers.color.choice); + + Python exposes the extension through extra_body: + + import base64 + from pathlib import Path + from typesafe_sdk import Choice, TypeSafeClient + + image = base64.b64encode(Path("image.png").read_bytes()).decode() + with TypeSafeClient(api_key="unused", base_url="https://classifier.dev") as client: + result = client.system_one( + model="jev/diffusiongemma", state="Look at the image.", timeout=60, + extra_body={"images": ["data:image/png;base64," + image]}, + questions={"color": Choice(instructions="What color?", + criteria={"red": None, "blue": None})}, + ) + print(result.choices["color"].choice) + + Both SDKs return their usual typed answers and token usage. Neither defines + image or audio output types. The default Jev model does not read images; + select DiffusionGemma explicitly when using an SDK that defaults to Jev. + DiffusionGemma is currently a free experimental model for callers; Beam's + provider cost is still counted against the service's spending allowances. + An unavailable model returns 503 dgemma_unavailable, invalid input returns + 400 dgemma_input, and saturation returns 429 dgemma_busy with Retry-After. + Images under another model return 400 images_unsupported. Image requests + never fall back to text-only inference. +`; + export const DOCS = `classifier.dev Zero-shot text classification over plain HTTP. You send text and a list of @@ -133,28 +201,7 @@ TYPESAFE SDK COMPATIBILITY credentials. GET /v1/models is public and does not spend quota or credits. The corresponding HTTP resources are POST /v1/systemone and GET /v1/models. - Images go through the same route. Set model to "dgemma" and add an images - array of data URLs (image/png, image/jpeg, image/webp or image/gif, base64; - at most 4 images and 900,000 characters of base64 in total, within the 1 MB - body). The questions keep the Choice, Noul and Score shapes and are answered - about the images and the state together: - - {"model": "dgemma", - "state": {"note": "Look at the attached image."}, - "images": ["data:image/png;base64,iVBORw0KGgo..."], - "questions": {"red": {"type": "noul", - "instructions": "Does the image contain a red square?"}}} - - "dgemma" is DiffusionGemma 26B-A4B in vLLM's structured-read mode: one - denoise step over a seeded answer template, read as a calibrated - distribution per question, re-read a few times when the first read is - uncertain. Text-only bodies may name it too. A choice question offers at - most 26 options and a request at most 64 questions. Jev never sees these - requests: when the model is down the answer is 503 dgemma_unavailable, not - a text-only guess. A body it refuses is 400 dgemma_input with its reason, a - saturated model is 429 dgemma_busy with Retry-After, and images sent under - another model are 400 images_unsupported. - +${IMAGE_CLASSIFICATION} LAYA AND KEV diff --git a/src/http/spending-classification.ts b/src/http/spending-classification.ts index fe82540..16ac61f 100644 --- a/src/http/spending-classification.ts +++ b/src/http/spending-classification.ts @@ -62,7 +62,7 @@ export async function spendingClassification(request: Request, env: AppEnv & Par } // dgemma is a System One door only; there, images select it as surely as its name. const systemOne = new URL(request.url).pathname === "/v1/systemone"; - const dgemma = systemOne && (body.model === "dgemma" || (body.images !== undefined && body.images !== null)); + const dgemma = systemOne && (body.model === "dgemma" || body.model === "jev/diffusiongemma" || (body.images !== undefined && body.images !== null)); const trial = body.model === "laya" || body.model === "kev" || body.model === "chunklaya" || dgemma; const quotedCharge = contextTokens !== undefined ? longContextCharge(contextTokens) : classificationCharge(trial ? 0 : 65536 * decisions, body.tier === "smart" ? decisions : 0); diff --git a/src/index.ts b/src/index.ts index e29c3e1..0e1e8b0 100644 --- a/src/index.ts +++ b/src/index.ts @@ -38,14 +38,14 @@ export { QuotaCoordinator } from "./admission"; import { admit, type AdmissionResult, type Quota } from "./admission"; import { pricingHtml } from "./pricingui"; import { typeSafeCompatibleResponse, typeSafeDecisionCount, typeSafeLabelSets } from "./typesafe-compat"; -import { dgemmaRoute, dgemmaResponse, dgemmaUnconfigured } from "./dgemma"; +import { dgemmaRoute, dgemmaResponse, dgemmaUnconfigured, dgemmaService } from "./dgemma"; export interface Env extends LayaEnv, SpendingEnv { QUOTAS?: DurableObjectNamespace; QUOTA_COORDINATOR_ENABLED?: string; OPENROUTER_API_KEY: string; TYPESAFE_API_KEY?: string; - /** Beam workspace token; the credential for the Beam-hosted Laya and Kev. */ + /** Beam workspace token; the credential for Beam-hosted Laya, Kev and DiffusionGemma. */ BEAM_API_KEY?: string; /** * chunklaya, our own long-document service: Laya behind a chunk-and-index @@ -1566,14 +1566,14 @@ const worker = { // A body that names model "dgemma" or carries images belongs to the // image-capable service, which speaks the same contract; everything // else stays TypeSafe's to validate. Neither answers for the other. - const door = decisions ? dgemmaRoute(body ?? "") : { kind: "typesafe" as const }; + const pod = dgemmaService(env); + const door = decisions ? dgemmaRoute(body ?? "", pod?.model ? 1 : undefined) : { kind: "typesafe" as const }; if (door.kind === "refuse") { sdkRecord(door.status, door.code, "dgemma"); return json({ error: door.message, code: door.code }, door.status); } let response: Response; if (door.kind === "dgemma") { - const pod = env.DGEMMA_ENABLED === "true" ? jevKeys(env)?.dgemma : undefined; response = pod ? await dgemmaResponse(pod, door.body, meter, req.signal) : dgemmaUnconfigured(); sdkRecord(response.status, response.ok ? "" : `dgemma_${response.status}`, "dgemma"); } else { diff --git a/src/openapi.ts b/src/openapi.ts index a431f4d..e9a8da6 100644 --- a/src/openapi.ts +++ b/src/openapi.ts @@ -577,7 +577,7 @@ export const OPENAPI = { "Wire-compatible with TypeSafe's POST /v1/systemone. The official JavaScript and Python SDKs work unchanged when their base URL is https://classifier.dev. " + "Use any non-empty placeholder API key for anonymous per-IP limits, or a classifier_agent_ workspace key to use workspace quota, credits and usage history. Free workspaces have the public ceilings and Pro workspaces get 10x limits. " + "classifier.dev never forwards caller credentials to TypeSafe. Choice, Noul, Score, structured state, model aliases, usage, validation errors and request IDs retain TypeSafe's shapes. Quota is counted by named questions, not requests. TypeSafe reference: https://docs.typesafe.ai/. " + - "Images: set model to \"dgemma\" and add an images array of data URLs; the same questions are then answered about the images and the state by DiffusionGemma, never by Jev, and a body with images under another model is refused with images_unsupported.", + "Images: set model to \"jev/diffusiongemma\" (or \"dgemma\") and add one base64 data URL in images; the same questions are then answered about the images and the state by DiffusionGemma, never by Jev, and a body with images under another model is refused with images_unsupported.", tags: ["classify"], security: [{ accountKey: [] }, {}], requestBody: { required: true, content: { "application/json": { schema: { $ref: "#/components/schemas/TypeSafeSystemOneRequest" } } } }, @@ -1011,12 +1011,12 @@ export const OPENAPI = { required: ["state", "model", "questions"], properties: { state: TYPESAFE_ENTRY, - model: { type: "string", description: "A name or alias returned by GET /v1/models, or \"dgemma\" for the image-capable DiffusionGemma model (required when images are sent)." }, + model: { type: "string", description: "A name or alias returned by GET /v1/models, or \"jev/diffusiongemma\" (alias \"dgemma\") for the image-capable DiffusionGemma model (required when images are sent)." }, questions: { type: "object", minProperties: 1, additionalProperties: TYPESAFE_QUESTION }, images: { - type: "array", minItems: 1, maxItems: 4, + type: "array", minItems: 1, maxItems: 1, items: { type: "string", pattern: "^data:image/(png|jpeg|webp|gif);base64,", maxLength: 900000 }, - description: "Images the questions are asked about, ahead of the state, as data URLs; at most 4 and 900,000 base64 characters in total. Only model \"dgemma\" reads them.", + description: "One inline image as a base64 data URL, at most 900,000 characters within the 1 MB body. Only model \"jev/diffusiongemma\" (alias \"dgemma\") reads it. Outputs are typed decisions, not images.", }, }, example: { diff --git a/src/pages.ts b/src/pages.ts index dba320b..e174190 100644 --- a/src/pages.ts +++ b/src/pages.ts @@ -9,7 +9,7 @@ import { SCRAPE_PRICE } from "./scrape"; * canonical document and nothing can drift between the three. */ -import { SPENDING_LIMITS, URL_CLASSIFICATION } from "./docs"; +import { SPENDING_LIMITS, URL_CLASSIFICATION, IMAGE_CLASSIFICATION } from "./docs"; import { SITE, SITE_UPDATED } from "./wellknown"; import { codeLang } from "./ui"; import { BILLING_PLANS, formatCreditsUsd } from "./lib/billing"; @@ -207,18 +207,16 @@ QUICKSTART }); -COMING SOON +${IMAGE_CLASSIFICATION} - Two things are being built on the same call shape: +COMING SOON - Image classification Labels in, one calibrated answer out, for images - instead of text. Private inference Zero-knowledge, end-to-end encrypted classification: the input is unreadable in transit and unreadable to the service that classifies it. - If either is on your roadmap, say so now and it gets built against your - case. + If private inference is on your roadmap, say so now and it gets built + against your case. Book a call ${SITE.author.cal} Email ${SITE.email} diff --git a/src/retail-rates.json b/src/retail-rates.json index fd7b0a0..690e99d 100644 --- a/src/retail-rates.json +++ b/src/retail-rates.json @@ -1,5 +1,5 @@ { - "version": "2026-09-24-dgemma-v1", + "version": "2026-09-24-beam-dgemma-v1", "models": [ { "provider": "typesafe", @@ -43,6 +43,13 @@ "outputUsdPerMillion": "0", "cachedInputUsdPerMillion": "0" }, + { + "provider": "beam", + "model": "jev/diffusiongemma", + "inputUsdPerMillion": "0", + "outputUsdPerMillion": "0", + "cachedInputUsdPerMillion": "0" + }, { "provider": "openrouter", "model": "ibm-granite/granite-4.0-h-micro", diff --git a/src/server/token-reservation.ts b/src/server/token-reservation.ts index c5b8077..49c5f07 100644 --- a/src/server/token-reservation.ts +++ b/src/server/token-reservation.ts @@ -9,6 +9,7 @@ export function providerCallBound(card: TokenRateCard, provider: ModelTokenUsage const inputLimit = provider === "typesafe" && model === JEV_ACCOUNT_MODEL ? 65_536 : provider === "beam" && model === "jev/laya" ? 512 : provider === "beam" && model === "jev/kev" ? 8_192 + : provider === "beam" && model === "jev/diffusiongemma" ? 32_768 : provider === "chunklaya" && model === "chunklaya/multilingual" ? 1_048_576 : provider === "dgemma" && model === "dgemma" ? 32_768 : provider === "openrouter" && model === "google/gemini-3.8-flash" ? 1_048_576 : null; diff --git a/src/spending/policy.ts b/src/spending/policy.ts index 1a72608..74fe4c1 100644 --- a/src/spending/policy.ts +++ b/src/spending/policy.ts @@ -41,6 +41,7 @@ export type Provider = ModelTokenUsage["provider"]; const models: Record = { "typesafe:jev-1.13.0": { context: 65536, input: 0.042, output: 0 }, "beam:jev/laya": { context: 512, input: 0.021, output: 0 }, + "beam:jev/diffusiongemma": { context: 32768, input: 0.021, output: 0 }, "beam:jev/kev": { context: 8192, input: 0.021, output: 0 }, // Our own pod, billed by the hour: no per-token provider price. The context is the service's document ceiling. "chunklaya:chunklaya/multilingual": { context: 1048576, input: 0, output: 0 }, diff --git a/wrangler.example.toml b/wrangler.example.toml index b611115..0da2832 100644 --- a/wrangler.example.toml +++ b/wrangler.example.toml @@ -83,8 +83,8 @@ LAYA_ENABLED = "true" # context now uses Jev screening and judgment with funded-workspace admission. CHUNKLAYA_ENABLED = "true" # model="dgemma" or an images array on POST /v1/systemone routes to the -# image-capable DiffusionGemma service when the Worker secrets DGEMMA_URL and -# DGEMMA_TOKEN are set; otherwise such requests answer dgemma_unavailable. +# Beam DiffusionGemma endpoint with BEAM_API_KEY. Without Beam, DGEMMA_URL +# and DGEMMA_TOKEN select a separately hosted service. No text-only fallback. DGEMMA_ENABLED = "true" # Enable only after account migration, pricing and Autumn reconciliation checks. APP_ACCOUNTS_ENABLED = "true" From 56f3147b00e82a31923960de9b2460bebcff59d4 Mon Sep 17 00:00:00 2001 From: Michael Ryaboy Date: Wed, 23 Sep 2026 20:22:22 -0700 Subject: [PATCH 2/4] Benchmark Beam against Jev and existing DiffusionGemma --- eval/diffusiongemma_latency.mjs | 96 +++++++++++++++++++++++++++++++++ 1 file changed, 96 insertions(+) create mode 100644 eval/diffusiongemma_latency.mjs diff --git a/eval/diffusiongemma_latency.mjs b/eval/diffusiongemma_latency.mjs new file mode 100644 index 0000000..c430425 --- /dev/null +++ b/eval/diffusiongemma_latency.mjs @@ -0,0 +1,96 @@ +// node --env-file=.dev.vars eval/diffusiongemma_latency.mjs +// Direct HTTP, one request in flight, alternating service order, no retries. +import assert from 'node:assert/strict'; +import { readFile, mkdir, writeFile } from 'node:fs/promises'; +import { performance } from 'node:perf_hooks'; + +const samples = Number(process.env.BENCH_SAMPLES || 20); +assert.ok(Number.isInteger(samples) && samples > 0); +const services = [ + { name: 'TypeSafe Jev', url: 'https://api.typesafe.ai/v1/systemone', model: 'jev-1.13.0', key: process.env.TYPESAFE_API_KEY }, + { name: 'Existing DiffusionGemma (classifier.dev)', url: 'https://classifier.dev/v1/systemone', model: 'dgemma', key: 'unused' }, + { name: 'Beam DiffusionGemma', url: 'https://app.beam.cloud/v1/systemone', model: 'jev/diffusiongemma', key: process.env.BEAM_API_KEY }, +]; +for (const service of services) assert.ok(service.key, `${service.name} credential is required`); +const criteria = { billing: null, technical: null, sales: null, feedback: null }; +const tickets = [ + ['Please refund the duplicate charge on my invoice.', 'billing'], + ['The application crashes whenever I open the settings page.', 'technical'], + ['Please send a quote for 500 enterprise seats.', 'sales'], + ['I love the new design. Thank you for making it easier to use!', 'feedback'], +]; +const category = (instructions) => ({ type: 'choice', instructions, criteria }); +const image = 'data:image/png;base64,' + (await readFile('captures/diffusiongemma-red.png')).toString('base64'); +const workloads = [ + { name: 'single_text', state: tickets[0][0], questions: { category: category('Which team should handle this ticket?') }, expected: { category: 'billing' } }, + { name: 'batch_16', state: Array.from({ length: 16 }, (_, i) => ({ id: `i${i}`, text: tickets[i % 4][0] })), + questions: Object.fromEntries(Array.from({ length: 16 }, (_, i) => [`i${i}`, category(`Which team should handle item i${i}?`)])), + expected: Object.fromEntries(Array.from({ length: 16 }, (_, i) => [`i${i}`, tickets[i % 4][1]])) }, + { name: 'long_text', state: { background: 'The company offers software subscriptions, product documentation, and customer assistance. '.repeat(100), ticket: tickets[0][0] }, + questions: { category: category('Which team should handle the ticket? The background is context, not the ticket.') }, expected: { category: 'billing' } }, + { name: 'image_3_decisions', state: 'Look at the attached image.', images: [image], questions: { + color: { type: 'choice', instructions: 'What color is the image?', criteria: { red: null, blue: null, green: null } }, + red: { type: 'noul', instructions: 'Is the image red?' }, + intensity: { type: 'score', instructions: 'How red is the image?', criteria: ['Not red', 'Some red', 'Entirely red'] }, + }, expected: { color: 'red' } }, +]; +const report = { startedAt: new Date().toISOString(), runtime: process.version, + protocol: { samplesPerWorkload: samples, warmupsPerServiceAndWorkload: 2, concurrency: 1, retries: 0, + order: 'alternate service order every round; same state/questions for both services', + latency: 'client end-to-end milliseconds including network, headers and complete JSON response; persistent Node fetch connections', + limitations: 'Same client, not co-located servers. Existing DiffusionGemma includes classifier.dev proxy overhead; TypeSafe and Beam use direct provider URLs. First request is not a verified cold start. Different models and tokenizers. Small synthetic correctness sanity check, not an accuracy or calibration benchmark. Image workload runs on both DiffusionGemma services; Jev is text-only.' }, + workloads: workloads.map(({ name, state, questions, images }) => ({ name, stateCharacters: JSON.stringify(state).length, decisions: Object.keys(questions).length, images: images?.length ?? 0 })), + rows: [], summaries: [] }; +const artifact = 'captures/diffusiongemma-latency.json'; +await mkdir('captures', { recursive: true }); +async function save() { await writeFile(artifact, JSON.stringify(report, null, 2)); } +async function call(service, work, round, warmup) { + const body = JSON.stringify({ model: service.model, state: work.state, questions: work.questions, ...(work.images ? { images: work.images } : {}) }); + const start = performance.now(); + const row = { service: service.name, workload: work.name, round, warmup, at: new Date().toISOString(), requestBytes: Buffer.byteLength(body) }; + try { + const response = await fetch(service.url, { method: 'POST', headers: { authorization: `Bearer ${service.key}`, 'content-type': 'application/json' }, body, signal: AbortSignal.timeout(65000) }); + row.headersMs = performance.now() - start; + const text = await response.text(); + row.totalMs = performance.now() - start; + row.status = response.status; + const payload = JSON.parse(text); + row.model = payload.model; + row.usage = payload.usage; + row.providerTiming = payload.diagnostics?.timing; + row.valid = response.ok && payload.model === service.model && Object.keys(work.questions).every(id => payload.answers?.[id]?.type === work.questions[id].type); + if (row.valid) { + row.correct = Object.entries(work.expected).filter(([id, choice]) => payload.answers[id].choice === choice).length; + row.decisionsChecked = Object.keys(work.expected).length; + row.answers = payload.answers; + } else row.error = payload.detail ?? payload.error ?? 'invalid response'; + } catch (error) { row.totalMs = performance.now() - start; row.valid = false; row.error = error.name; } + report.rows.push(row); + await save(); + console.log(`${warmup ? 'warmup' : 'sample'} ${work.name} ${service.name} ${round + 1}: ${row.status ?? row.error} ${Math.round(row.totalMs)}ms`); + return row; +} +const applicable = (work) => services.filter(service => !work.images || service.model !== 'jev-1.13.0'); +for (const work of workloads) for (const service of applicable(work)) for (let i = 0; i < 2; i++) { + const row = await call(service, work, i, true); + if (!row.valid) throw new Error(`${service.name} cannot run ${work.name}; see ${artifact}`); +} +for (let round = 0; round < samples; round++) for (const work of workloads) { + const order = applicable(work); + if (round % 2) order.reverse(); + for (const service of order) await call(service, work, round, false); +} +const percentile = (values, q) => values[Math.max(0, Math.ceil(values.length * q) - 1)]; +for (const work of workloads) for (const service of applicable(work)) { + const all = report.rows.filter(r => !r.warmup && r.service === service.name && r.workload === work.name); + const rows = all.filter(r => r.valid), times = rows.map(r => r.totalMs).sort((a, b) => a - b); + report.summaries.push({ service: service.name, workload: work.name, requests: all.length, successful: rows.length, + medianMs: percentile(times, .5), p95Ms: percentile(times, .95), minMs: times[0], maxMs: times.at(-1), + meanMs: times.reduce((a, b) => a + b, 0) / times.length, + medianInputTokens: percentile(rows.map(r => r.usage.input_tokens).sort((a, b) => a - b), .5), + correct: rows.reduce((n, r) => n + r.correct, 0), checked: rows.reduce((n, r) => n + r.decisionsChecked, 0), + }); +} +report.finishedAt = new Date().toISOString(); +await save(); +console.table(report.summaries); From 05b997ce87488b61146dc0d9a4bb2c91a6c883eb Mon Sep 17 00:00:00 2001 From: Michael Ryaboy Date: Wed, 23 Sep 2026 20:22:50 -0700 Subject: [PATCH 3/4] Use conventional medians in latency benchmark summaries --- eval/diffusiongemma_latency.mjs | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/eval/diffusiongemma_latency.mjs b/eval/diffusiongemma_latency.mjs index c430425..2c8f9b1 100644 --- a/eval/diffusiongemma_latency.mjs +++ b/eval/diffusiongemma_latency.mjs @@ -37,6 +37,7 @@ const workloads = [ const report = { startedAt: new Date().toISOString(), runtime: process.version, protocol: { samplesPerWorkload: samples, warmupsPerServiceAndWorkload: 2, concurrency: 1, retries: 0, order: 'alternate service order every round; same state/questions for both services', + percentiles: 'Median averages the two middle observations; p95 uses nearest rank', latency: 'client end-to-end milliseconds including network, headers and complete JSON response; persistent Node fetch connections', limitations: 'Same client, not co-located servers. Existing DiffusionGemma includes classifier.dev proxy overhead; TypeSafe and Beam use direct provider URLs. First request is not a verified cold start. Different models and tokenizers. Small synthetic correctness sanity check, not an accuracy or calibration benchmark. Image workload runs on both DiffusionGemma services; Jev is text-only.' }, workloads: workloads.map(({ name, state, questions, images }) => ({ name, stateCharacters: JSON.stringify(state).length, decisions: Object.keys(questions).length, images: images?.length ?? 0 })), @@ -80,7 +81,9 @@ for (let round = 0; round < samples; round++) for (const work of workloads) { if (round % 2) order.reverse(); for (const service of order) await call(service, work, round, false); } -const percentile = (values, q) => values[Math.max(0, Math.ceil(values.length * q) - 1)]; +const percentile = (values, q) => q === .5 && values.length % 2 === 0 + ? (values[values.length / 2 - 1] + values[values.length / 2]) / 2 + : values[Math.max(0, Math.ceil(values.length * q) - 1)]; for (const work of workloads) for (const service of applicable(work)) { const all = report.rows.filter(r => !r.warmup && r.service === service.name && r.workload === work.name); const rows = all.filter(r => r.valid), times = rows.map(r => r.totalMs).sort((a, b) => a - b); From 1ca2aa45da38f2312479292099bbf3db0dd4b30b Mon Sep 17 00:00:00 2001 From: Michael Ryaboy Date: Wed, 23 Sep 2026 20:25:00 -0700 Subject: [PATCH 4/4] Compare Beam and RunPod directly from one benchmark runner --- .github/workflows/benchmark-images.yml | 32 ++++++++++++++++++++++++++ eval/diffusiongemma_latency.mjs | 24 ++++++++++--------- 2 files changed, 45 insertions(+), 11 deletions(-) create mode 100644 .github/workflows/benchmark-images.yml diff --git a/.github/workflows/benchmark-images.yml b/.github/workflows/benchmark-images.yml new file mode 100644 index 0000000..decfcf7 --- /dev/null +++ b/.github/workflows/benchmark-images.yml @@ -0,0 +1,32 @@ +name: Beam and RunPod latency +on: + workflow_dispatch: + push: + branches: [codex/diffusiongemma-images] + paths: + - eval/diffusiongemma_latency.mjs + - .github/workflows/benchmark-images.yml +permissions: + contents: read +jobs: + benchmark: + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-node@v7 + with: + node-version: 22 + - name: Compare direct inference APIs + env: + BEAM_API_KEY: ${{ secrets.BEAM_API_KEY }} + DGEMMA_URL: ${{ secrets.DGEMMA_URL }} + DGEMMA_TOKEN: ${{ secrets.DGEMMA_TOKEN }} + BENCH_SAMPLES: "30" + run: node eval/diffusiongemma_latency.mjs + - name: Save latency measurements + if: always() + uses: actions/upload-artifact@v4 + with: + name: beam-runpod-latency + path: captures/beam-runpod-latency.json diff --git a/eval/diffusiongemma_latency.mjs b/eval/diffusiongemma_latency.mjs index 2c8f9b1..2bf150f 100644 --- a/eval/diffusiongemma_latency.mjs +++ b/eval/diffusiongemma_latency.mjs @@ -1,14 +1,16 @@ // node --env-file=.dev.vars eval/diffusiongemma_latency.mjs // Direct HTTP, one request in flight, alternating service order, no retries. import assert from 'node:assert/strict'; -import { readFile, mkdir, writeFile } from 'node:fs/promises'; +import { mkdir, writeFile } from 'node:fs/promises'; import { performance } from 'node:perf_hooks'; const samples = Number(process.env.BENCH_SAMPLES || 20); assert.ok(Number.isInteger(samples) && samples > 0); +assert.ok(process.env.DGEMMA_URL, 'DGEMMA_URL is required'); +const runpodUrl = new URL(process.env.DGEMMA_URL); +assert.equal(runpodUrl.protocol, 'https:', 'RunPod requires HTTPS'); const services = [ - { name: 'TypeSafe Jev', url: 'https://api.typesafe.ai/v1/systemone', model: 'jev-1.13.0', key: process.env.TYPESAFE_API_KEY }, - { name: 'Existing DiffusionGemma (classifier.dev)', url: 'https://classifier.dev/v1/systemone', model: 'dgemma', key: 'unused' }, + { name: 'RunPod DiffusionGemma', url: runpodUrl.href.replace(/\/+$/, '') + '/v1/systemone', model: 'dgemma', key: process.env.DGEMMA_TOKEN }, { name: 'Beam DiffusionGemma', url: 'https://app.beam.cloud/v1/systemone', model: 'jev/diffusiongemma', key: process.env.BEAM_API_KEY }, ]; for (const service of services) assert.ok(service.key, `${service.name} credential is required`); @@ -20,7 +22,8 @@ const tickets = [ ['I love the new design. Thank you for making it easier to use!', 'feedback'], ]; const category = (instructions) => ({ type: 'choice', instructions, criteria }); -const image = 'data:image/png;base64,' + (await readFile('captures/diffusiongemma-red.png')).toString('base64'); +// Deterministic 128 × 128 red PNG, shared by both services. +const image = 'data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAIAAAACACAIAAABMXPacAAABcUlEQVR4nO3UwQkAMAwDsey/dDuGHmfQAIdDe+9u4AJbHy+wA+wAeoK9AL/CvqAuXxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC+J8QZwviPMFcb4gzhfE+YI4XxDnC17bByBow7KSKBBvAAAAAElFTkSuQmCC'; const workloads = [ { name: 'single_text', state: tickets[0][0], questions: { category: category('Which team should handle this ticket?') }, expected: { category: 'billing' } }, { name: 'batch_16', state: Array.from({ length: 16 }, (_, i) => ({ id: `i${i}`, text: tickets[i % 4][0] })), @@ -34,15 +37,15 @@ const workloads = [ intensity: { type: 'score', instructions: 'How red is the image?', criteria: ['Not red', 'Some red', 'Entirely red'] }, }, expected: { color: 'red' } }, ]; -const report = { startedAt: new Date().toISOString(), runtime: process.version, +const report = { startedAt: new Date().toISOString(), runtime: process.version, runner: process.env.GITHUB_ACTIONS === 'true' ? 'GitHub Actions ubuntu-latest' : 'local', protocol: { samplesPerWorkload: samples, warmupsPerServiceAndWorkload: 2, concurrency: 1, retries: 0, order: 'alternate service order every round; same state/questions for both services', percentiles: 'Median averages the two middle observations; p95 uses nearest rank', latency: 'client end-to-end milliseconds including network, headers and complete JSON response; persistent Node fetch connections', - limitations: 'Same client, not co-located servers. Existing DiffusionGemma includes classifier.dev proxy overhead; TypeSafe and Beam use direct provider URLs. First request is not a verified cold start. Different models and tokenizers. Small synthetic correctness sanity check, not an accuracy or calibration benchmark. Image workload runs on both DiffusionGemma services; Jev is text-only.' }, + limitations: 'Direct provider APIs from the same client; no classifier.dev proxy. Server regions, hardware and inference settings may differ. First request is not a verified cold start. Repeated synthetic workloads are a correctness sanity check, not an accuracy or calibration benchmark.' }, workloads: workloads.map(({ name, state, questions, images }) => ({ name, stateCharacters: JSON.stringify(state).length, decisions: Object.keys(questions).length, images: images?.length ?? 0 })), rows: [], summaries: [] }; -const artifact = 'captures/diffusiongemma-latency.json'; +const artifact = 'captures/beam-runpod-latency.json'; await mkdir('captures', { recursive: true }); async function save() { await writeFile(artifact, JSON.stringify(report, null, 2)); } async function call(service, work, round, warmup) { @@ -71,20 +74,19 @@ async function call(service, work, round, warmup) { console.log(`${warmup ? 'warmup' : 'sample'} ${work.name} ${service.name} ${round + 1}: ${row.status ?? row.error} ${Math.round(row.totalMs)}ms`); return row; } -const applicable = (work) => services.filter(service => !work.images || service.model !== 'jev-1.13.0'); -for (const work of workloads) for (const service of applicable(work)) for (let i = 0; i < 2; i++) { +for (const work of workloads) for (const service of services) for (let i = 0; i < 2; i++) { const row = await call(service, work, i, true); if (!row.valid) throw new Error(`${service.name} cannot run ${work.name}; see ${artifact}`); } for (let round = 0; round < samples; round++) for (const work of workloads) { - const order = applicable(work); + const order = [...services]; if (round % 2) order.reverse(); for (const service of order) await call(service, work, round, false); } const percentile = (values, q) => q === .5 && values.length % 2 === 0 ? (values[values.length / 2 - 1] + values[values.length / 2]) / 2 : values[Math.max(0, Math.ceil(values.length * q) - 1)]; -for (const work of workloads) for (const service of applicable(work)) { +for (const work of workloads) for (const service of services) { const all = report.rows.filter(r => !r.warmup && r.service === service.name && r.workload === work.name); const rows = all.filter(r => r.valid), times = rows.map(r => r.totalMs).sort((a, b) => a - b); report.summaries.push({ service: service.name, workload: work.name, requests: all.length, successful: rows.length,