Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion deno.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"name": "veryfront",
"version": "0.1.1010",
"version": "0.1.1011",
"license": "Apache-2.0",
"nodeModulesDir": "auto",
"minimumDependencyAge": {
Expand Down
85 changes: 85 additions & 0 deletions src/agent/conversation/run-event-normalization.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -106,6 +106,51 @@ describe("agent/conversation-run-event-normalization", () => {
);
});

it("keeps events with oversized message ids within the byte limit", () => {
const result = normalizeConversationRunEvent({
type: "TEXT_MESSAGE_CONTENT",
messageId: "m".repeat(300 * 1024),
delta: "x",
});

for (const event of result) {
assertEquals(
getConversationRunEventJsonByteLength(event) <= MAX_CONVERSATION_RUN_EVENT_PAYLOAD_BYTES,
true,
);
}
});

it("keeps events with oversized tool call ids within the byte limit", () => {
const result = normalizeConversationRunEvent({
type: "TOOL_CALL_RESULT",
toolCallId: "tc".repeat(160 * 1024),
content: "ok",
input: { blob: "x".repeat(300 * 1024) },
});

for (const event of result) {
assertEquals(
getConversationRunEventJsonByteLength(event) <= MAX_CONVERSATION_RUN_EVENT_PAYLOAD_BYTES,
true,
);
}
});

it("keeps events with oversized type fields within the byte limit", () => {
const result = normalizeConversationRunEvent({
type: "CUSTOM_EVENT_".repeat(32 * 1024),
payload: "x".repeat(300 * 1024),
});

for (const event of result) {
assertEquals(
getConversationRunEventJsonByteLength(event) <= MAX_CONVERSATION_RUN_EVENT_PAYLOAD_BYTES,
true,
);
}
});

it("keeps every split part of escape-heavy delta events within the byte limit", () => {
const escapeHeavyDelta = '"'.repeat(300 * 1024);
const parts = normalizeConversationRunEvent({
Expand All @@ -122,6 +167,46 @@ describe("agent/conversation-run-event-normalization", () => {
}
});

it("preserves all data when splitting escape-heavy delta events", () => {
// Splitting by raw UTF-8 bytes would leave each part oversized once JSON-escaped,
// forcing the size-limit backstop to truncate (drop) the tail. The split must be
// escape-aware so the concatenated parts reconstruct the original delta losslessly.
const escapeHeavyDelta = '"'.repeat(300 * 1024);
const parts = normalizeConversationRunEvent({
type: "TEXT_MESSAGE_CONTENT",
messageId: "m1",
delta: escapeHeavyDelta,
});

assertEquals(parts.length > 1, true);
assertEquals(parts.map((part) => part.delta).join(""), escapeHeavyDelta);
for (const part of parts) {
assertEquals(
getConversationRunEventJsonByteLength(part) <= MAX_CONVERSATION_RUN_EVENT_PAYLOAD_BYTES,
true,
);
}
});

it("preserves all data when splitting escape-heavy TOOL_CALL_ARGS deltas", () => {
// TOOL_CALL_ARGS reassembles into a tool's JSON arguments; a dropped tail would
// corrupt them, so escape-heavy args must split losslessly across parts.
const escapeHeavyArgs = '"'.repeat(300 * 1024);
const parts = normalizeConversationRunEvent({
type: "TOOL_CALL_ARGS",
toolCallId: "tc_args",
delta: escapeHeavyArgs,
});

assertEquals(parts.map((part) => part.delta).join(""), escapeHeavyArgs);
for (const part of parts) {
assertEquals(
getConversationRunEventJsonByteLength(part) <= MAX_CONVERSATION_RUN_EVENT_PAYLOAD_BYTES,
true,
);
}
});

it("normalizes whole event lists", () => {
const events = [
{ type: "TEXT_MESSAGE_CONTENT", delta: "ok" },
Expand Down
109 changes: 84 additions & 25 deletions src/agent/conversation/run-event-normalization.ts
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
export const MAX_CONVERSATION_RUN_EVENT_PAYLOAD_BYTES = 240 * 1024;
const OMITTED_CONVERSATION_RUN_EVENT_TYPE = "CUSTOM";
const MAX_SUMMARY_DEPTH = 4;
const MAX_SUMMARY_ARRAY_ITEMS = 8;
const MAX_SUMMARY_OBJECT_KEYS = 24;
Expand Down Expand Up @@ -74,13 +75,7 @@ function enforceEventSizeLimit(event: ConversationRunEventRecord): ConversationR
}
}

return {
type: event.type,
...(typeof event.messageId === "string" ? { messageId: event.messageId } : {}),
...(typeof event.toolCallId === "string" ? { toolCallId: event.toolCallId } : {}),
truncated: true,
note: "Conversation-run event payload exceeded the size limit and was omitted.",
};
return buildOmittedEvent(event);
}

/** Normalizes conversation run events. */
Expand Down Expand Up @@ -143,7 +138,7 @@ function summarizeToolResultEvent(event: ConversationRunEventRecord): Conversati
*/
function truncateEventStringFieldToLimit(
event: ConversationRunEventRecord,
field: "content" | "delta",
field: string,
suffix: string,
): ConversationRunEventRecord | null {
const value = event[field];
Expand Down Expand Up @@ -189,6 +184,50 @@ function truncateEventStringFieldToLimit(
return buildCandidate(best);
}

function addStringFieldWithinLimit(
event: ConversationRunEventRecord,
field: string,
value: string,
): ConversationRunEventRecord {
const candidate = { ...event, [field]: value };
if (
getConversationRunEventJsonByteLength(candidate) <= MAX_CONVERSATION_RUN_EVENT_PAYLOAD_BYTES
) {
return candidate;
}

return truncateEventStringFieldToLimit(candidate, field, " [truncated]") ?? event;
}

function buildOmittedEvent(event: ConversationRunEventRecord): ConversationRunEventRecord {
let omitted: ConversationRunEventRecord = {
type: OMITTED_CONVERSATION_RUN_EVENT_TYPE,
name: "conversation-run-event-omitted",
truncated: true,
note: "Conversation-run event payload exceeded the size limit and was omitted.",
};

omitted = addStringFieldWithinLimit(omitted, "originalType", event.type);

if (typeof event.messageId === "string") {
omitted = addStringFieldWithinLimit(omitted, "originalMessageId", event.messageId);
}

if (typeof event.toolCallId === "string") {
omitted = addStringFieldWithinLimit(omitted, "originalToolCallId", event.toolCallId);
}

if (getConversationRunEventJsonByteLength(omitted) <= MAX_CONVERSATION_RUN_EVENT_PAYLOAD_BYTES) {
return omitted;
}

return {
type: OMITTED_CONVERSATION_RUN_EVENT_TYPE,
name: "conversation-run-event-omitted",
truncated: true,
};
}

function summarizeGenericEvent(event: ConversationRunEventRecord): ConversationRunEventRecord {
const { type, ...rest } = event;
return {
Expand All @@ -203,25 +242,45 @@ function splitStringFieldEvent<TField extends "delta" | "content">(
event: ConversationRunEventRecord & Record<TField, string>,
field: TField,
): ConversationRunEventRecord[] {
const maxBytes = getStringFieldBudget(event, field);
const parts = splitUtf8String(event[field], maxBytes);
return parts.map((part) => ({ ...event, [field]: part }));
}
const value = event[field];
const buildPart = (slice: string): ConversationRunEventRecord => ({ ...event, [field]: slice });

function getStringFieldBudget(
event: ConversationRunEventRecord,
field: "delta" | "content",
): number {
const eventWithEmptyField = {
...event,
[field]: "",
};
const parts: ConversationRunEventRecord[] = [];
let startIndex = 0;

while (startIndex < value.length) {
// Largest prefix whose WHOLE serialized event fits the byte limit. Measuring the
// event (not the raw slice) keeps the split correct for escape-heavy content that
// expands under JSON.stringify — the same unit the API enforces — so every part
// fits without the size-limit backstop having to truncate (drop) any data.
let low = startIndex + 1;
let high = value.length;
let bestEndIndex = -1;

while (low <= high) {
const mid = Math.floor((low + high) / 2);
if (
getConversationRunEventJsonByteLength(buildPart(value.slice(startIndex, mid))) <=
MAX_CONVERSATION_RUN_EVENT_PAYLOAD_BYTES
) {
bestEndIndex = mid;
low = mid + 1;
} else {
high = mid - 1;
Comment on lines +260 to +269

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Badge Split deltas on Unicode-safe boundaries

When a near-limit envelope leaves only a few bytes for delta and the delta begins with an astral Unicode character, this binary search can test a lone high-surrogate slice as oversized ("\\ud83d" in JSON) even though the complete surrogate pair fits. Because JSON byte length is not monotonic over UTF-16 code-unit indexes, bestEndIndex can remain unset and the fallback omits the whole event; for example, a TEXT_MESSAGE_CONTENT with a messageId length of 245698 and delta: "😀😀" is reduced to the CUSTOM omitted marker instead of splitting into two valid emoji events. Search on code-point boundaries or otherwise avoid considering lone surrogates as split candidates.

Useful? React with 👍 / 👎.

}
}

if (bestEndIndex <= startIndex) {
// Even a single character overflows the envelope; hand off to the size-limit
// backstop rather than loop forever emitting zero-progress parts.
return [event];
}

parts.push(buildPart(value.slice(startIndex, bestEndIndex)));
startIndex = bestEndIndex;
}

return Math.max(
1,
MAX_CONVERSATION_RUN_EVENT_PAYLOAD_BYTES -
getConversationRunEventJsonByteLength(eventWithEmptyField),
);
return parts.length > 0 ? parts : [event];
}

function splitUtf8String(value: string, maxBytes: number): string[] {
Expand Down
2 changes: 1 addition & 1 deletion src/utils/version-constant.ts
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
// Keep in sync with deno.json version.
// scripts/release.ts updates this constant during releases.
/** Shared version value. */
export const VERSION = "0.1.1010";
export const VERSION = "0.1.1011";