Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
93 changes: 57 additions & 36 deletions apps/web/src/session-logic.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2353,43 +2353,64 @@ describe("session activity performance", () => {
expect(appendedEntries[1]).toBe(initialEntries[1]);
});

it("updates 20,000 ordered tool activities within 100 ms", () => {
const activities = Array.from({ length: 20_000 }, (_, index) =>
makeActivity({
id: `benchmark-tool-${index}`,
createdAt: new Date(1_700_000_000_000 + index).toISOString(),
kind: "tool.completed",
summary: "Ran command",
sequence: index,
payload: {
itemType: "command_execution",
title: "Ran command",
data: {
toolCallId: `benchmark-tool-${index}`,
item: { command: ["git", "status"] },
it("updates 20,000 ordered tool activities far faster than deriving them from scratch", () => {
const makeActivities = (count: number, prefix: string) =>
Array.from({ length: count }, (_, index) =>
makeActivity({
id: `${prefix}-${index}`,
createdAt: new Date(1_700_000_000_000 + index).toISOString(),
kind: "tool.completed",
summary: "Ran command",
sequence: index,
payload: {
itemType: "command_execution",
title: "Ran command",
data: { toolCallId: `${prefix}-${index}`, item: { command: ["git", "status"] } },
},
},
}),
);
deriveWorkLogEntries(activities);
const updatedActivities = [
...activities,
makeActivity({
id: "benchmark-tool-appended",
createdAt: new Date(1_700_000_000_000 + activities.length).toISOString(),
kind: "tool.completed",
summary: "Ran command",
sequence: activities.length,
payload: {
itemType: "command_execution",
title: "Ran command",
data: { toolCallId: "benchmark-tool-appended", item: { command: ["git", "diff"] } },
},
}),
];
}),
);

// Wall-clock on a shared CI runner swings several-fold between runs (the
// old 100 ms bound measured 100-108 ms on a 4-vCPU hosted runner for work
// that takes 4 ms on a quiet laptop), so compare the update against a
// from-scratch derivation measured in the same process instead. Best of
// three rounds shrugs off a GC pause or scheduling stall in any one
// sample. With the per-activity cache an update is roughly 6-8x cheaper;
// without it both calls walk the same code path and the ratio is ~1.
let fromScratchMs = Number.POSITIVE_INFINITY;
let updateMs = Number.POSITIVE_INFINITY;
for (let round = 0; round < 3; round++) {
const activities = makeActivities(20_000, `benchmark-tool-${round}`);
deriveWorkLogEntries(activities);
const updatedActivities = [
...activities,
makeActivity({
id: `benchmark-tool-appended-${round}`,
createdAt: new Date(1_700_000_000_000 + activities.length).toISOString(),
kind: "tool.completed",
summary: "Ran command",
sequence: activities.length,
payload: {
itemType: "command_execution",
title: "Ran command",
data: {
toolCallId: `benchmark-tool-appended-${round}`,
item: { command: ["git", "diff"] },
},
},
}),
];

const updateStartedAt = performance.now();
expect(deriveWorkLogEntries(updatedActivities)).toHaveLength(20_001);
updateMs = Math.min(updateMs, performance.now() - updateStartedAt);

const fresh = makeActivities(20_001, `benchmark-fresh-${round}`);
const fromScratchStartedAt = performance.now();
expect(deriveWorkLogEntries(fresh)).toHaveLength(20_001);
fromScratchMs = Math.min(fromScratchMs, performance.now() - fromScratchStartedAt);
}

const startedAt = performance.now();
expect(deriveWorkLogEntries(updatedActivities)).toHaveLength(20_001);
expect(performance.now() - startedAt).toBeLessThan(100);
expect(updateMs).toBeLessThan(fromScratchMs / 2);
});
});
Loading