diff --git a/AGENTS.md b/AGENTS.md index e0cacd2..3a76a42 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -79,6 +79,12 @@ the plain text (`curl classifier.dev`), the HTML and the Markdown never drift. TypeSafe's. Nothing answers for the other: a refused body is `dgemma_input`, a busy service `dgemma_busy`, a down or unconfigured one `dgemma_unavailable`, and images under another model `images_unsupported`. Billed by the hour. + Images are dgemma's only media: the interposer silently drops keys it does + not read, so a dgemma body carrying `audio`, `audios`, `video` or `videos` + is refused with `dgemma_input` at the door rather than answered by a model + that never saw the media. Audio has no input path into DiffusionGemma; + video the model can read (vllm-project/vllm#57589) but the interposer does + not serve it yet — enabling it is a pod-side change, not a Worker one. - The updates roadmap is one constant, `ROADMAP` in `src/newsletter.ts`; the plain text, the signup form and the Markdown all render from it. Addresses go to the `subscriber` table in the shared application Neon database. Preserve consent, diff --git a/src/dgemma.ts b/src/dgemma.ts index 54ebaad..f060b26 100644 --- a/src/dgemma.ts +++ b/src/dgemma.ts @@ -10,6 +10,12 @@ import { addTokens, type Meter } from "./cost"; * nowhere else: there is no fallback to Jev, which cannot look at the image, * and none the other way. Everything else on the route stays TypeSafe's to * validate, so the official SDKs keep their native behaviour. + * + * Images are the only media. The interposer drops keys it does not read, so + * audio or video sent through this door would be answered confidently by a + * model that never saw it; those bodies are refused here instead. Audio has + * no path into DiffusionGemma at all, and video, which the model itself can + * read, is not served by the interposer yet. */ export const DGEMMA_MODEL = "dgemma"; @@ -45,6 +51,15 @@ export function dgemmaRoute(body: string): DgemmaRoute { if (hasImages && parsed.model !== undefined && !named) { return refuse("images_unsupported", `Jev does not accept images; set model to "${DGEMMA_MODEL}" or remove images`); } + // The interposer ignores keys it does not read, so audio or video traveling + // with this body would be silently dropped and the questions answered about + // the text alone. Refusing here is the honest answer. + for (const key of ["audio", "audios", "video", "videos"]) { + if (parsed[key] === undefined || parsed[key] === null) continue; + return refuse("dgemma_input", key.startsWith("audio") + ? `${DGEMMA_MODEL} reads text and images only; DiffusionGemma has no audio input, so remove ${key}` + : `${DGEMMA_MODEL} reads text and images only; video is not supported yet, so remove ${key}`); + } if (hasImages) { if (!Array.isArray(images) || images.length === 0 || images.length > DGEMMA_MAX_IMAGES) { return refuse("dgemma_input", `images must be an array of 1 to ${DGEMMA_MAX_IMAGES} data URLs`); diff --git a/src/docs.ts b/src/docs.ts index 6cf505d..0b34d62 100644 --- a/src/docs.ts +++ b/src/docs.ts @@ -157,6 +157,10 @@ TYPESAFE SDK COMPATIBILITY a text-only guess. A body it refuses is 400 dgemma_input with its reason, a saturated model is 429 dgemma_busy with Retry-After, and images sent under another model are 400 images_unsupported. + Images are the only media. Audio and video bodies (audio, audios, video or + videos keys) are 400 dgemma_input rather than silently ignored: audio has + no input path into DiffusionGemma at all, and video, which the model can + read, is not served yet. The output is structured answers, not generated image data. The TypeSafe JavaScript SDK 0.6.0 forwards images at runtime but its TypeScript request type omits that field; direct HTTP exposes the documented JSON contract. diff --git a/src/openapi.ts b/src/openapi.ts index aaa6225..32c310e 100644 --- a/src/openapi.ts +++ b/src/openapi.ts @@ -577,7 +577,8 @@ export const OPENAPI = { "Wire-compatible with TypeSafe's POST /v1/systemone. The official JavaScript and Python SDKs work unchanged when their base URL is https://classifier.dev. " + "Use any non-empty placeholder API key for anonymous per-IP limits, or a classifier_agent_ workspace key to use workspace quota, credits and usage history. Free workspaces have the public ceilings and Pro workspaces get 10x limits. " + "classifier.dev never forwards caller credentials to TypeSafe. Choice, Noul, Score, structured state, model aliases, usage, validation errors and request IDs retain TypeSafe's shapes. Quota is counted by named questions, not requests. TypeSafe reference: https://docs.typesafe.ai/. " + - "Images: set model to \"dgemma\" and add an images array of data URLs; the same questions are then answered about the images and the state by DiffusionGemma, never by Jev, and a body with images under another model is refused with images_unsupported.", + "Images: set model to \"dgemma\" and add an images array of data URLs; the same questions are then answered about the images and the state by DiffusionGemma, never by Jev, and a body with images under another model is refused with images_unsupported. " + + "Images are the only media: a dgemma body carrying audio, audios, video or videos is refused with dgemma_input rather than silently ignored — audio has no input path into DiffusionGemma, and video is not served yet.", tags: ["classify"], security: [{ accountKey: [] }, {}], requestBody: { required: true, content: { "application/json": { schema: { $ref: "#/components/schemas/TypeSafeSystemOneRequest" } } } }, @@ -596,7 +597,7 @@ export const OPENAPI = { "402": { description: "The workspace balance cannot cover the provider reservation; TypeSafe is not called.", content: { "application/json": { schema: { type: "object" } } } }, "403": { description: "The workspace key is inactive or the workspace cannot authorize usage.", content: { "application/json": { schema: { type: "object" } } } }, "413": { description: "The request body exceeds 1 MB.", content: { "application/json": { schema: { type: "object" } } } }, - "400": { description: "An image request was refused: images_unsupported (images under a model other than dgemma) or dgemma_input (a malformed image, too many images, or a schema the image-capable model refuses; its reason is the error).", content: { "application/json": { schema: { $ref: "#/components/schemas/Error" } } } }, + "400": { description: "An image request was refused: images_unsupported (images under a model other than dgemma) or dgemma_input (a malformed image, too many images, audio or video sent to the image-capable model, or a schema it refuses; its reason is the error).", content: { "application/json": { schema: { $ref: "#/components/schemas/Error" } } } }, "429": { description: "classifier.dev or TypeSafe rate limit; Retry-After or Retry-After-Ms is preserved. dgemma_busy when the image-capable model is saturated.", headers: { ...RATE_LIMIT_HEADERS, ...TYPESAFE_REQUEST_ID_HEADER, ...ACCOUNT_BILLING_HEADERS }, content: { "application/json": { schema: { type: "object" } } } }, "502": { description: "TypeSafe could not be reached.", headers: ACCOUNT_BILLING_HEADERS, content: { "application/json": { schema: { type: "object", properties: { error: { type: "string" } } } } } }, "503": { description: "The compatibility endpoint or workspace billing is temporarily unavailable; dgemma_unavailable when the image-capable model is down or not configured.", headers: ACCOUNT_BILLING_HEADERS, content: { "application/json": { schema: { type: "object", properties: { error: { type: "string" } } } } } }, @@ -1016,7 +1017,7 @@ export const OPENAPI = { images: { type: "array", minItems: 1, maxItems: 4, items: { type: "string", pattern: "^data:image/(png|jpeg|webp|gif);base64,", maxLength: 900000 }, - description: "Images the questions are asked about, ahead of the state, as data URLs; at most 4 and 900,000 data URL characters in total. Only model \"dgemma\" reads them.", + description: "Images the questions are asked about, ahead of the state, as data URLs; at most 4 and 900,000 data URL characters in total. Only model \"dgemma\" reads them, and images are its only media: audio and video are refused with dgemma_input.", }, }, example: { diff --git a/test/dgemma.test.ts b/test/dgemma.test.ts index 0ef4a0c..89a6108 100644 --- a/test/dgemma.test.ts +++ b/test/dgemma.test.ts @@ -98,6 +98,31 @@ test("images under another model, malformed images and too many images are refus expect(calls).toHaveLength(0); }); +test("audio and video are refused at the door, never silently dropped by the pod", async () => { + const calls = mockUpstreams(); + const WAV = "data:audio/wav;base64," + Buffer.from("riff").toString("base64"); + const MP4 = "data:video/mp4;base64,AAAA"; + // The pod's interposer ignores keys it does not read, so each of these would + // otherwise come back as a confident answer about media the model never saw. + for (const [body, word] of [ + [{ model: "dgemma", state: "x", audio: [WAV], questions: QUESTIONS }, "audio"], + [{ model: "dgemma", state: "x", audios: [WAV], questions: QUESTIONS }, "audio"], + [{ model: "dgemma", state: "x", video: [MP4], questions: QUESTIONS }, "video"], + [{ model: "dgemma", state: "x", videos: [MP4], questions: QUESTIONS }, "video"], + [{ state: "x", images: [PNG], video: [MP4], questions: QUESTIONS }, "video"], + ] as const) { + const response = await post(body); + expect(response.status).toBe(400); + const answer = await response.json() as { code: string; error: string }; + expect(answer.code).toBe("dgemma_input"); + expect(answer.error).toContain(word); + } + // Audio or video without images and without naming dgemma stays TypeSafe's to validate. + const plain = await post({ state: "x", audio: [WAV], questions: QUESTIONS }); + expect(plain.status).toBe(200); + expect(calls.map((c) => c.url)).toEqual(["https://api.typesafe.ai/v1/systemone"]); +}); + test("without the service configured or enabled, an image or dgemma request is unavailable, never answered by Jev", async () => { const calls = mockUpstreams(); for (const bindings of [{ ...env, DGEMMA_URL: undefined }, { ...env, DGEMMA_TOKEN: undefined }, { ...env, DGEMMA_ENABLED: "false" }]) {