Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
74 changes: 74 additions & 0 deletions extensions/ext-llm-openai/src/openai-tool-input.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,74 @@
import { assertEquals } from "#veryfront/testing/assert.ts";
import { describe, it } from "#veryfront/testing/bdd.ts";
import {
appendOpenAIStreamToolArgument,
joinOpenAIStreamToolArguments,
MAX_OPENAI_STREAM_TOOL_ARGUMENT_BYTES,
MAX_OPENAI_STREAM_TOOL_ARGUMENT_FRAGMENTS,
type OpenAIStreamToolArgumentBudget,
} from "./openai-tool-input.ts";

function emptyBudget(): OpenAIStreamToolArgumentBudget {
return { bytes: 0, fragments: 0 };
}

describe("ext-llm-openai/openai-tool-input", () => {
it("accepts far more non-empty fragments than the fragment cap while under the byte budget", () => {
// The provider's tokenizer decides how finely arguments are chunked, so a
// caller cannot control fragment count. Counting non-empty fragments against
// the flood guard rejected ordinary tool calls at roughly 2-8% of the byte
// budget they were nominally allowed -- every scheduled run of one project
// failed on this hourly. The byte budget is what bounds content.
const budget = emptyBudget();
const chunks: string[] = [];
const fragment = '{"a":1},';
const count = MAX_OPENAI_STREAM_TOOL_ARGUMENT_FRAGMENTS * 4;

let limit: string | undefined;
for (let i = 0; i < count; i++) {
limit = appendOpenAIStreamToolArgument(budget, chunks, fragment);
if (limit) break;
}

assertEquals(limit, undefined, `expected no limit for ${count} small fragments`);
assertEquals(chunks.length, count, "every non-empty fragment must be kept");
assertEquals(
budget.bytes < MAX_OPENAI_STREAM_TOOL_ARGUMENT_BYTES,
true,
"the payload must still be well inside the byte budget",
);
assertEquals(joinOpenAIStreamToolArguments(chunks).length, fragment.length * count);
});

it("still stops an unbounded flood of zero-byte fragments", () => {
// Zero-byte fragments never advance the byte budget, so they are the only
// ones that can arrive without bound. They are what this cap guards.
const budget = emptyBudget();
const chunks: string[] = [];

let limit: string | undefined;
for (let i = 0; i < MAX_OPENAI_STREAM_TOOL_ARGUMENT_FRAGMENTS + 10; i++) {
limit = appendOpenAIStreamToolArgument(budget, chunks, "");
if (limit) break;
}

assertEquals(limit, "fragments", "empty-fragment floods must still be rejected");
assertEquals(chunks.length, 0, "empty fragments are never appended");
});

it("still enforces the byte budget", () => {
const budget = emptyBudget();
const chunks: string[] = [];
const big = "x".repeat(MAX_OPENAI_STREAM_TOOL_ARGUMENT_BYTES);

assertEquals(appendOpenAIStreamToolArgument(budget, chunks, big), undefined);
assertEquals(appendOpenAIStreamToolArgument(budget, chunks, "y"), "bytes");
});

it("counts multi-byte characters by UTF-8 length, not code units", () => {
const budget = emptyBudget();
const chunks: string[] = [];
appendOpenAIStreamToolArgument(budget, chunks, "€");
assertEquals(budget.bytes, 3);
});
});
29 changes: 22 additions & 7 deletions extensions/ext-llm-openai/src/openai-tool-input.ts
Original file line number Diff line number Diff line change
Expand Up @@ -13,20 +13,35 @@ export function appendOpenAIStreamToolArgument(
chunks: string[],
fragment: string,
): "bytes" | "fragments" | undefined {
if (budget.fragments >= MAX_OPENAI_STREAM_TOOL_ARGUMENT_FRAGMENTS) {
return "fragments";
const fragmentBytes = TOOL_ARGUMENT_ENCODER.encode(fragment).byteLength;

// Zero-byte fragments never advance the byte budget, so they are the only
// ones that can arrive without bound -- and they are what the fragment cap
// exists to stop. Both provider streams exercise it exactly that way, by
// flooding empty deltas.
//
// Counting non-empty fragments here too made the cap bind long before the
// byte budget: at typical delta sizes, 4096 fragments is roughly 2-8% of the
// 1 MiB a tool call is nominally allowed, which left the byte limit
// unreachable. Fragment count is chosen by the provider's tokenizer rather
// than by the caller, so the same tool call passed or failed depending on how
// the stream happened to be chunked.
if (fragmentBytes === 0) {

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Badge Bound retained nonempty argument chunks

When an OpenAI-compatible provider emits very small nonempty deltas, this condition bypasses the fragment cap entirely, so both stream parsers can retain up to 1,048,576 separately allocated strings before the byte budget fires. The payload is limited to 1 MiB, but the array and per-string overhead can consume tens of megabytes per concurrent request, whereas the previous implementation retained at most 4,096 chunks. Allow provider-controlled chunking without retaining every fragment separately, for example by coalescing chunks while maintaining the byte limit.

Useful? React with 👍 / 👎.

if (budget.fragments >= MAX_OPENAI_STREAM_TOOL_ARGUMENT_FRAGMENTS) {
return "fragments";
}
budget.fragments++;
return undefined;
}

const fragmentBytes = TOOL_ARGUMENT_ENCODER.encode(fragment).byteLength;
if (fragmentBytes > MAX_OPENAI_STREAM_TOOL_ARGUMENT_BYTES - budget.bytes) {
return "bytes";
}

budget.fragments++;
// Content is bounded by bytes: every fragment kept here is at least one byte,
// so the array cannot outgrow the byte budget.
budget.bytes += fragmentBytes;
if (fragment.length > 0) {
chunks.push(fragment);
}
chunks.push(fragment);
return undefined;
}

Expand Down