From 70c8d94567db3e2b39c8884b9f6f1f91f21b5943 Mon Sep 17 00:00:00 2001
From: JSONbored <49853598+JSONbored@users.noreply.github.com>
Date: Tue, 28 Jul 2026 21:29:13 -0700
Subject: [PATCH] fix(stats): withhold reversal-grounded accuracy where a
reversal cannot be recorded
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
The fairness page showed 100% accuracy for every repository and all eight
weeks of the trend while the headline read 94.6%. Those 100%s are structural
zeros, not measurements.
`recordReversalSignals` has one call site, inside `handlePullRequestWebhookEvent`,
reached only via the `github-webhook` job — a member of
REVIEW_EXECUTION_JOB_TYPES, which a runtime that does not execute reviews
acks-and-drops. So `reversed` is pinned at 0 while merged/closed keep growing,
and `1 - 0/N` renders as a flawless score over a numerator that can never move.
The repo already recorded the outcome: reversal-superseded.ts notes "verified in
production: zero reversal events ever".
`accuracyPct` now returns null unless a reversal is OBSERVABLE on this
deployment — a recorded reversal proves it outright (and short-circuits the
probe), otherwise there must be at least one terminal auto-action in the
retained window for a reversal to attach to. This is the discipline the surface
already applies elsewhere, extended to the case where the numerator's WRITER,
not just its sample, is absent: a sparse rule's precision is null, and
gamingFlagsCaught is null below three instances.
Also fixes the weekly trend's denominator asymmetry: `merged`/`closed` folded in
registered-Orb rows while `reversed` stayed own-ledger-only, so the denominator
grew with every new install while the numerator did not — trending every week
toward 100% independent of real reversal behavior. #7449 fixed exactly this for
the lifetime figure and it was never carried across. Accuracy now divides the
own-ledger pairing; the displayed volume still includes the fleet.
UI: the withheld cells say why rather than leaving a bare em dash next to
healthy volume, and the headline now discloses `basis` — #9168 computes
"single_instance_self_report" precisely so one operator's self-reported number
is not read as fleet corroboration, and the page dropped the field entirely,
along with the sample size and confidence interval. The methodology note also
described the reversal formula while the headline rendered decisionAccuracy;
it now explains both and why they differ.
Refs #9676
---
.../site/fairness-report-page.test.tsx | 59 +++++++++++++++++++
.../components/site/fairness-report-page.tsx | 52 ++++++++++++++--
src/review/public-stats.ts | 46 ++++++++++++++-
src/services/public-accuracy-trend.ts | 48 +++++++++++----
test/unit/public-accuracy-trend.test.ts | 55 ++++++++++++++---
test/unit/public-stats.test.ts | 55 +++++++++++++++++
6 files changed, 288 insertions(+), 27 deletions(-)
diff --git a/apps/loopover-ui/src/components/site/fairness-report-page.test.tsx b/apps/loopover-ui/src/components/site/fairness-report-page.test.tsx
index 0725fe95f8..81631909be 100644
--- a/apps/loopover-ui/src/components/site/fairness-report-page.test.tsx
+++ b/apps/loopover-ui/src/components/site/fairness-report-page.test.tsx
@@ -220,6 +220,65 @@ describe("FairnessReportPage (#fairness-analytics)", () => {
expect(screen.getByText("2 human-reversed, lifetime")).toBeTruthy();
});
+ it("explains a withheld accuracy instead of letting it read as a dash-shaped mystery", async () => {
+ // Backend publishes null when no auto-action was ever recorded for a reversal to attach to; the page must
+ // say that rather than leaving a bare "—" next to a healthy-looking volume.
+ apiFetch.mockResolvedValue({
+ ok: true,
+ data: {
+ ...FIXTURE,
+ byProject: [
+ { project: "owner/repo", reviewed: 100, merged: 60, closed: 30, accuracyPct: null },
+ ],
+ accuracyTrend: [
+ { weekStart: "2026-07-13", merged: 30, closed: 15, reversed: 0, accuracyPct: null },
+ ],
+ },
+ status: 200,
+ durationMs: 10,
+ });
+ renderWithClient();
+
+ await waitFor(() => expect(screen.getByText("By repository")).toBeTruthy());
+ const notes = screen.getAllByText(/not measurable on this deployment, not 100%/);
+ expect(notes.length).toBe(2); // one under each affected table
+ });
+
+ it("does not show the unmeasurable-accuracy note when every accuracy is real", async () => {
+ apiFetch.mockResolvedValue({ ok: true, data: FIXTURE, status: 200, durationMs: 10 });
+ renderWithClient();
+
+ await waitFor(() => expect(screen.getByText("By repository")).toBeTruthy());
+ expect(screen.queryByText(/not measurable on this deployment/)).toBeNull();
+ });
+
+ it("#9168: discloses a single-instance self-report rather than presenting it as fleet corroboration", async () => {
+ apiFetch.mockResolvedValue({
+ ok: true,
+ data: {
+ ...FIXTURE,
+ fleetAccuracy: {
+ ...FIXTURE.fleetAccuracy,
+ instanceCount: 1,
+ basis: "single_instance_self_report",
+ decidedCount: 5225,
+ accuracyCiPct: { lo: 93.9, hi: 95.2 },
+ },
+ },
+ status: 200,
+ durationMs: 10,
+ });
+ renderWithClient();
+
+ await waitFor(() => expect(screen.getByText("Decision accuracy")).toBeTruthy());
+ expect(screen.getByText(/Self-reported by that single instance/).textContent).toContain(
+ "5,225 decided",
+ );
+ expect(screen.getByText(/Self-reported by that single instance/).textContent).toContain(
+ "93.9–95.2%",
+ );
+ });
+
it("#9068: renders the insufficient-instances state (not a fabricated zero) when gamingFlagsCaught is null", async () => {
apiFetch.mockResolvedValue({
ok: true,
diff --git a/apps/loopover-ui/src/components/site/fairness-report-page.tsx b/apps/loopover-ui/src/components/site/fairness-report-page.tsx
index cebc2164c2..7a5646a8e6 100644
--- a/apps/loopover-ui/src/components/site/fairness-report-page.tsx
+++ b/apps/loopover-ui/src/components/site/fairness-report-page.tsx
@@ -15,6 +15,19 @@ import type { PublicStats } from "@/components/site/proof-of-power-stats-model";
// count, with a short methodology note. Counts only; no PR content, contributor identities, or trust scores.
const pctFmt = new Intl.NumberFormat("en", { maximumFractionDigits: 1 });
+/** Why an accuracy cell reads "—". Reversal-grounded accuracy needs the deployment to have recorded the
+ * terminal auto-actions a reversal attaches to; where it hasn't, `1 - 0/N` would render as a flawless 100%
+ * over a numerator that can never move, so the backend publishes null and the page says so out loud. */
+function UnmeasurableAccuracyNote() {
+ return (
+
+ An accuracy of — means not measurable on this deployment,
+ not 100%: no auto-merge/auto-close was recorded here for a human reversal to be counted
+ against. The volume columns beside it are measured directly and are unaffected.
+
+ );
+}
+
const intFmt = new Intl.NumberFormat("en");
async function fetchPublicStats(): Promise {
@@ -119,6 +132,23 @@ export function FairnessReportPage() {
? `${intFmt.format(data.totals.reversed)} human-reversed, lifetime`
: "reversal-grounded, lifetime"}
+ {/* #9168 computes `basis` precisely so this number is not read as corroborated-across-operators
+ when it is one operator's own disclosed outcomes; the page used to drop the field entirely. */}
+ {fleetEligible && data.fleetAccuracy.basis === "single_instance_self_report" ? (
+
+ Self-reported by that single instance, not corroborated across operators
+ {data.fleetAccuracy.decidedCount != null
+ ? ` (${intFmt.format(data.fleetAccuracy.decidedCount)} decided`
+ : ""}
+ {data.fleetAccuracy.decidedCount != null &&
+ data.fleetAccuracy.accuracyCiPct != null
+ ? `, 95% CI ${pctFmt.format(data.fleetAccuracy.accuracyCiPct.lo)}–${pctFmt.format(data.fleetAccuracy.accuracyCiPct.hi)}%)`
+ : data.fleetAccuracy.decidedCount != null
+ ? ")"
+ : ""}
+ .
+
+ ) : null}
Anti-gaming flags caught
@@ -149,11 +179,15 @@ export function FairnessReportPage() {
- How accuracy is measured: 1
- minus the share of auto-merged/auto-closed PRs a human later overturned — a
- bot-closed PR a contributor reopened, or a bot-merged PR undone by a separate revert
- PR. Nothing here is a prediction or a self-assessment; it's counted after the fact
- from what actually happened on GitHub.
+ How accuracy is measured: the
+ headline scores the gate's own merge/close decisions — the share the
+ realized outcome confirmed, with holds excluded because a deferral to a human is not
+ a decision that can be right or wrong (#8820). The per-repository and weekly tables
+ below are a different, stricter measure: 1 minus the share of
+ auto-merged/auto-closed PRs a human later overturned — a bot-closed PR a contributor
+ reopened, or a bot-merged PR undone by a separate revert PR. Neither is a prediction
+ or a self-assessment; both are counted after the fact from what actually happened on
+ GitHub, which is also why the two can differ.
@@ -206,6 +240,9 @@ export function FairnessReportPage() {
+ {data.byProject.some((row) => row.accuracyPct == null) ? (
+
+ ) : null}
) : null}
@@ -256,6 +293,11 @@ export function FairnessReportPage() {
+ {data.accuracyTrend.some(
+ (week) => week.merged != null && week.accuracyPct == null,
+ ) ? (
+
+ ) : null}
{data.rulePrecision && data.rulePrecision.rules.length > 0 ? (
diff --git a/src/review/public-stats.ts b/src/review/public-stats.ts
index 117752cdd5..f5866ba45b 100644
--- a/src/review/public-stats.ts
+++ b/src/review/public-stats.ts
@@ -55,6 +55,7 @@ import { validateCalibrationPayload } from "./risk-control";
import { loadRepoFocusManifest } from "../signals/focus-manifest-loader";
import { resolveLoopOverSelfRepoFullName } from "../config/loopover-repo-focus-manifest";
import { errorMessage } from "../utils/json";
+import { retentionCutoffIsoForTable } from "../db/retention";
/** FALLBACK estimate of maintainer review/triage time saved per reviewed PR, used ONLY when the real per-PR
* average (`estimateReviewEffort`'s minutes, persisted at publish time — see `reviewEffortMinutes` in the
@@ -164,12 +165,47 @@ function filteredPct(reviewed: number, merged: number): number | null {
return Math.round(((reviewed - merged) / reviewed) * 1000) / 10;
}
-/** Reversal-grounded accuracy over the irreversible auto-actions (merged + closed); null until there is signal. */
+/** The auto-actions a reversal is recorded against. `recordReversalSignals` only ever fires for a PR this
+ * deployment itself auto-merged or auto-closed, so with zero of these rows in the retained window a
+ * `reversal_*` event cannot exist — and `1 - 0/N` is then a structural zero, not a measurement. */
+export const AUTO_ACTION_EVENT_TYPES = ["agent.action.close", "agent.action.merge"] as const;
+
+/**
+ * Whether a reversal is OBSERVABLE on this deployment: did it record any terminal auto-action in the window a
+ * reversal could still be attributed to? Reversal signals are written only by the review-execution pipeline
+ * (`recordReversalSignals`, reached via the `github-webhook` job), which a runtime that does not execute
+ * reviews acks-and-drops — so on such a runtime `reversed` is pinned at 0 forever while the merged/closed
+ * denominator keeps growing, and every reversal-grounded percentage converges on a fake 100%.
+ *
+ * Publishing that as accuracy is the exact failure #8820 called out for the fleet number ("the old formula
+ * overstated accuracy") and #7449 fixed for the global one. This is the same discipline the rest of this
+ * surface already applies — a sparse rule's precision is null, `gamingFlagsCaught` is null below three
+ * instances — extended to the case where the numerator's WRITER, not just its sample, is absent.
+ */
+export async function loadReversalObservability(env: Env, knownReversals = 0): Promise {
+ // A recorded reversal is itself proof the pipeline that records them runs here -- no probe needed, and no
+ // query on the hot path for any deployment that has ever overturned an auto-action.
+ if (knownReversals > 0) return true;
+ const rows = await safeAll<{ n: number }>(
+ env,
+ `SELECT COUNT(*) AS n FROM audit_events
+ WHERE event_type IN (${AUTO_ACTION_EVENT_TYPES.map((type) => `'${type}'`).join(", ")})
+ AND created_at >= ?`,
+ retentionCutoffIsoForTable("audit_events"),
+ );
+ return (rows[0]?.n ?? 0) > 0;
+}
+
+/** Reversal-grounded accuracy over the irreversible auto-actions (merged + closed); null until there is signal,
+ * and null when a reversal is not observable at all on this deployment (see {@link loadReversalObservability}) —
+ * an unmeasurable quantity stays unknown rather than rendering as a perfect score. */
function accuracyPct(
merged: number,
closed: number,
reversed: number,
+ reversalObservable: boolean,
): number | null {
+ if (!reversalObservable) return null;
const decided = merged + closed;
if (decided <= 0) return null;
// `reversed` counts engine auto-actions regardless of a PR's CURRENT disposition, so a reopened
@@ -426,6 +462,10 @@ export async function getPublicStats(
error: 0,
reversed: 0,
};
+ // Resolved before byProject so every published accuracy figure -- per repo and global -- answers the same
+ // question about the same deployment, rather than one of them silently disagreeing with the others.
+ const totalReversals = [...reversedByProject.values()].reduce((sum, n) => sum + n, 0);
+ const reversalObservable = await loadReversalObservability(env, totalReversals);
const byProject = dispositions
.map((d) => {
const merged = d.merged ?? 0;
@@ -445,7 +485,7 @@ export async function getPublicStats(
reviewed,
merged,
closed,
- accuracyPct: accuracyPct(merged, closed, reversed),
+ accuracyPct: accuracyPct(merged, closed, reversed, reversalObservable),
};
})
.filter((r) => r.reviewed > 0)
@@ -573,7 +613,7 @@ export async function getPublicStats(
// Option 1 of #7449: compute the global accuracy from the OWN-LEDGER merged/closed snapshot (not the
// fleet-folded totals.merged/closed), so its numerator (own-ledger reversed) and denominator are drawn
// from the same population. See the ownLedgerMerged/ownLedgerClosed snapshot above the Orb fold for why.
- accuracyPct: accuracyPct(ownLedgerMerged, ownLedgerClosed, totals.reversed),
+ accuracyPct: accuracyPct(ownLedgerMerged, ownLedgerClosed, totals.reversed, reversalObservable),
minutesSaved,
},
weekly: { reviewed: w.reviewed ?? 0, merged: w.merged ?? 0 },
diff --git a/src/services/public-accuracy-trend.ts b/src/services/public-accuracy-trend.ts
index 84217ddea9..9ba57b15d8 100644
--- a/src/services/public-accuracy-trend.ts
+++ b/src/services/public-accuracy-trend.ts
@@ -8,7 +8,7 @@
// #2568 pattern for the sibling per-repo quality trend) can recompute any historical week correctly on every
// request -- no cron-miss gap risk, no second copy of the number to keep in sync, and the SAME formula as the
// live figure by construction, so the two can never silently diverge or read as inconsistent to a public viewer.
-import { PUBLISHED_PR_KEYS, publicStatsProjects, safeAll } from "../review/public-stats";
+import { loadReversalObservability, PUBLISHED_PR_KEYS, publicStatsProjects, safeAll } from "../review/public-stats";
import { isoWeekStart } from "./public-quality-metrics";
export const PUBLIC_ACCURACY_TREND_WEEKS = 8;
@@ -24,7 +24,13 @@ export type PublicAccuracyTrendWeek = {
accuracyPct: number | null;
};
-type DayRow = { day: string; merged: number; closed: number; reversed: number };
+/** `merged`/`closed` are the DISPLAYED volume (own ledger + registered Orb fleet). `ownMerged`/`ownClosed` are
+ * the own-ledger-only pairing for `reversed`, which is own-ledger-only by construction (the Orb aggregate has
+ * no reversal concept). Accuracy divides the own-ledger numbers ONLY -- #7449 fixed exactly this asymmetry for
+ * the lifetime figure in public-stats.ts and it was never carried across to this trend, so the denominator grew
+ * with every newly registered install while the numerator stayed own-ledger-scoped, trending every week toward
+ * 100% independent of real reversal behavior. */
+type DayRow = { day: string; merged: number; closed: number; ownMerged: number; ownClosed: number; reversed: number };
const MS_PER_WEEK = 7 * 86_400_000;
@@ -33,20 +39,35 @@ function roundPct(value: number): number {
}
/** Same formula as public-stats.ts's accuracyPct, reused so the trend and the live number can never drift
- * apart into two competing definitions of "accuracy". */
-function publicBucketOf(bucket: { merged: number; closed: number; reversed: number }): Omit {
+ * apart into two competing definitions of "accuracy". `reversalObservable` false means this deployment records
+ * no terminal auto-actions for a reversal to be attributed to, so the week's accuracy is unknown, not perfect --
+ * the volume columns still publish, since those ARE measured. */
+function publicBucketOf(
+ bucket: { merged: number; closed: number; ownMerged: number; ownClosed: number; reversed: number },
+ reversalObservable: boolean,
+): Omit {
const decided = bucket.merged + bucket.closed;
if (decided < MIN_ACCURACY_TREND_SAMPLE) return { merged: null, closed: null, reversed: null, accuracyPct: null };
- const reversalRate = Math.min(1, bucket.reversed / decided);
- return { merged: bucket.merged, closed: bucket.closed, reversed: bucket.reversed, accuracyPct: roundPct(1 - reversalRate) };
+ const observed = { merged: bucket.merged, closed: bucket.closed, reversed: bucket.reversed };
+ if (!reversalObservable) return { ...observed, accuracyPct: null };
+ // Own-ledger-only denominator: `reversed` can only ever be attributed to own-ledger PRs (see DayRow).
+ const ownDecided = bucket.ownMerged + bucket.ownClosed;
+ if (ownDecided <= 0) return { ...observed, accuracyPct: null };
+ const reversalRate = Math.min(1, bucket.reversed / ownDecided);
+ return { ...observed, accuracyPct: roundPct(1 - reversalRate) };
}
/** Fold day-granularity rows into `weeks` trailing UTC-Monday buckets ending in the week containing `nowMs`.
* Pure -- mirrors buildPublicQualityTrend's own bucketing shape (public-quality-metrics.ts). */
-export function buildPublicAccuracyTrend(dayRows: DayRow[], nowMs: number, weeks: number = PUBLIC_ACCURACY_TREND_WEEKS): PublicAccuracyTrendWeek[] {
+export function buildPublicAccuracyTrend(
+ dayRows: DayRow[],
+ nowMs: number,
+ weeks: number = PUBLIC_ACCURACY_TREND_WEEKS,
+ reversalObservable = true,
+): PublicAccuracyTrendWeek[] {
const currentStartMs = Date.parse(isoWeekStart(nowMs));
const oldestStartMs = currentStartMs - (weeks - 1) * MS_PER_WEEK;
- const buckets = Array.from({ length: weeks }, () => ({ merged: 0, closed: 0, reversed: 0 }));
+ const buckets = Array.from({ length: weeks }, () => ({ merged: 0, closed: 0, ownMerged: 0, ownClosed: 0, reversed: 0 }));
for (const row of dayRows) {
const dayMs = Date.parse(`${row.day}T00:00:00.000Z`);
@@ -56,12 +77,14 @@ export function buildPublicAccuracyTrend(dayRows: DayRow[], nowMs: number, weeks
const bucket = buckets[weekOffset]!;
bucket.merged += row.merged;
bucket.closed += row.closed;
+ bucket.ownMerged += row.ownMerged;
+ bucket.ownClosed += row.ownClosed;
bucket.reversed += row.reversed;
}
return buckets.map((bucket, offset) => ({
weekStart: isoWeekStart(oldestStartMs + offset * MS_PER_WEEK),
- ...publicBucketOf(bucket),
+ ...publicBucketOf(bucket, reversalObservable),
}));
}
@@ -169,10 +192,11 @@ export async function loadPublicAccuracyTrend(env: Env, nowMs: number = Date.now
const projects = publicStatsProjects(env);
const sinceIso = new Date(Date.parse(isoWeekStart(nowMs)) - (PUBLIC_ACCURACY_TREND_WEEKS - 1) * MS_PER_WEEK).toISOString();
- const [ownLedger, reversals, orb] = await Promise.all([
+ const [ownLedger, reversals, orb, reversalObservable] = await Promise.all([
loadOwnLedgerDayRows(env, projects, sinceIso),
loadReversalDayRows(env, projects, sinceIso),
loadOrbDayRows(env, sinceIso),
+ loadReversalObservability(env),
]);
const days = new Set([...ownLedger.keys(), ...reversals.keys(), ...orb.keys()]);
@@ -180,8 +204,10 @@ export async function loadPublicAccuracyTrend(env: Env, nowMs: number = Date.now
day,
merged: (ownLedger.get(day)?.merged ?? 0) + (orb.get(day)?.merged ?? 0),
closed: (ownLedger.get(day)?.closed ?? 0) + (orb.get(day)?.closed ?? 0),
+ ownMerged: ownLedger.get(day)?.merged ?? 0,
+ ownClosed: ownLedger.get(day)?.closed ?? 0,
reversed: reversals.get(day) ?? 0,
}));
- return buildPublicAccuracyTrend(dayRows, nowMs);
+ return buildPublicAccuracyTrend(dayRows, nowMs, PUBLIC_ACCURACY_TREND_WEEKS, reversalObservable);
}
diff --git a/test/unit/public-accuracy-trend.test.ts b/test/unit/public-accuracy-trend.test.ts
index 29a3bc2cb2..9fed995593 100644
--- a/test/unit/public-accuracy-trend.test.ts
+++ b/test/unit/public-accuracy-trend.test.ts
@@ -11,15 +11,22 @@ import { createTestEnv } from "../helpers/d1";
const NOW = Date.parse("2026-06-22T12:00:00.000Z");
+/** A day row whose volume is entirely own-ledger (no registered-Orb fold), which is what every case below
+ * models -- so `ownMerged`/`ownClosed` mirror `merged`/`closed` and the accuracy denominator is unchanged.
+ * The Orb-fold asymmetry gets its own explicit cases further down. */
+function row(day: string, merged: number, closed: number, reversed: number) {
+ return { day, merged, closed, ownMerged: merged, ownClosed: closed, reversed };
+}
+
describe("buildPublicAccuracyTrend", () => {
it("buckets day rows into weekly totals and computes the SAME accuracy formula as the live number", () => {
const currentMonday = isoWeekStart(NOW);
const priorMonday = isoWeekStart(NOW - 7 * 86_400_000);
const trend = buildPublicAccuracyTrend(
[
- { day: priorMonday, merged: 4, closed: 2, reversed: 1 },
- { day: priorMonday, merged: 1, closed: 0, reversed: 0 }, // a second day in the SAME week -- must accumulate
- { day: currentMonday, merged: 3, closed: 3, reversed: 0 },
+ row(priorMonday, 4, 2, 1),
+ row(priorMonday, 1, 0, 0), // a second day in the SAME week -- must accumulate
+ row(currentMonday, 3, 3, 0),
],
NOW,
2,
@@ -45,33 +52,33 @@ describe("buildPublicAccuracyTrend", () => {
it("REGRESSION: ignores day rows outside the trailing window instead of letting them corrupt the oldest bucket", () => {
const currentMonday = isoWeekStart(NOW);
const tooOld = isoWeekStart(NOW - 30 * 86_400_000);
- const trend = buildPublicAccuracyTrend([{ day: tooOld, merged: 999, closed: 999, reversed: 999 }, { day: currentMonday, merged: MIN_ACCURACY_TREND_SAMPLE, closed: 0, reversed: 0 }], NOW, 2);
+ const trend = buildPublicAccuracyTrend([row(tooOld, 999, 999, 999), row(currentMonday, MIN_ACCURACY_TREND_SAMPLE, 0, 0)], NOW, 2);
expect(trend[0]).toMatchObject({ merged: null, closed: null, reversed: null });
expect(trend[1]).toMatchObject({ merged: MIN_ACCURACY_TREND_SAMPLE, closed: 0, reversed: 0 });
});
it("ignores an unparseable day string rather than throwing or corrupting a bucket", () => {
const currentMonday = isoWeekStart(NOW);
- const trend = buildPublicAccuracyTrend([{ day: "not-a-date", merged: 5, closed: 5, reversed: 5 }, { day: currentMonday, merged: MIN_ACCURACY_TREND_SAMPLE, closed: 0, reversed: 0 }], NOW, 1);
+ const trend = buildPublicAccuracyTrend([row("not-a-date", 5, 5, 5), row(currentMonday, MIN_ACCURACY_TREND_SAMPLE, 0, 0)], NOW, 1);
expect(trend).toHaveLength(1);
expect(trend[0]).toMatchObject({ merged: MIN_ACCURACY_TREND_SAMPLE, closed: 0, reversed: 0 });
});
it("REGRESSION: redacts counts and accuracyPct below MIN_ACCURACY_TREND_SAMPLE decided PRs", () => {
const week = isoWeekStart(NOW);
- const trend = buildPublicAccuracyTrend([{ day: week, merged: MIN_ACCURACY_TREND_SAMPLE - 1, closed: 0, reversed: 0 }], NOW, 1);
+ const trend = buildPublicAccuracyTrend([row(week, MIN_ACCURACY_TREND_SAMPLE - 1, 0, 0)], NOW, 1);
expect(trend[0]).toMatchObject({ merged: null, closed: null, reversed: null, accuracyPct: null });
});
it("returns a real percentage at exactly MIN_ACCURACY_TREND_SAMPLE decided PRs", () => {
const week = isoWeekStart(NOW);
- const trend = buildPublicAccuracyTrend([{ day: week, merged: MIN_ACCURACY_TREND_SAMPLE, closed: 0, reversed: 0 }], NOW, 1);
+ const trend = buildPublicAccuracyTrend([row(week, MIN_ACCURACY_TREND_SAMPLE, 0, 0)], NOW, 1);
expect(trend[0]?.accuracyPct).toBe(100);
});
it("clamps a reversed count that exceeds decided (a reopened auto-close dropped from merged+closed) to 0%, never negative", () => {
const week = isoWeekStart(NOW);
- const trend = buildPublicAccuracyTrend([{ day: week, merged: 0, closed: MIN_ACCURACY_TREND_SAMPLE, reversed: MIN_ACCURACY_TREND_SAMPLE + 5 }], NOW, 1);
+ const trend = buildPublicAccuracyTrend([row(week, 0, MIN_ACCURACY_TREND_SAMPLE, MIN_ACCURACY_TREND_SAMPLE + 5)], NOW, 1);
expect(trend[0]?.accuracyPct).toBe(0);
});
@@ -85,6 +92,38 @@ describe("buildPublicAccuracyTrend", () => {
expect(trend).toHaveLength(3);
for (const week of trend) expect(week).toMatchObject({ merged: null, closed: null, reversed: null, accuracyPct: null });
});
+
+ it("REGRESSION: divides reversals by the OWN-LEDGER denominator, not the Orb-folded volume", () => {
+ // #7449 fixed this asymmetry for the lifetime figure and it was never carried across to the trend: the Orb
+ // aggregate has no reversal concept, so folding its merged/closed into the denominator while the numerator
+ // stays own-ledger-only trends every week toward 100% as more installs register.
+ const week = isoWeekStart(NOW);
+ const trend = buildPublicAccuracyTrend(
+ [{ day: week, merged: 100, closed: 0, ownMerged: 4, ownClosed: 0, reversed: 1 }],
+ NOW,
+ 1,
+ );
+ // 1 - 1/4 = 75%, NOT 1 - 1/100 = 99%.
+ expect(trend[0]).toEqual({ weekStart: week, merged: 100, closed: 0, reversed: 1, accuracyPct: 75 });
+ });
+
+ it("reports null accuracy — never 100% — when a reversal is not observable on this deployment", () => {
+ // Regression: on a runtime that does not execute reviews, `reversed` is pinned at 0 forever, so every week
+ // rendered as a perfect 100% over rising volume. The volume itself IS measured and still publishes.
+ const week = isoWeekStart(NOW);
+ const trend = buildPublicAccuracyTrend([row(week, 40, 10, 0)], NOW, 1, false);
+ expect(trend[0]).toEqual({ weekStart: week, merged: 40, closed: 10, reversed: 0, accuracyPct: null });
+ });
+
+ it("reports null accuracy when a week's volume is entirely Orb-folded (no own-ledger PRs to attribute a reversal to)", () => {
+ const week = isoWeekStart(NOW);
+ const trend = buildPublicAccuracyTrend(
+ [{ day: week, merged: 50, closed: 10, ownMerged: 0, ownClosed: 0, reversed: 0 }],
+ NOW,
+ 1,
+ );
+ expect(trend[0]).toMatchObject({ merged: 50, closed: 10, reversed: 0, accuracyPct: null });
+ });
});
describe("loadPublicAccuracyTrend — end-to-end over the real live tables", () => {
diff --git a/test/unit/public-stats.test.ts b/test/unit/public-stats.test.ts
index ac36f6b3f5..d721bcec63 100644
--- a/test/unit/public-stats.test.ts
+++ b/test/unit/public-stats.test.ts
@@ -6,8 +6,10 @@ import {
getPublicStats,
isPublicStatsEnabled,
MINUTES_SAVED_PER_PR,
+ loadReversalObservability,
resolvePublicStatsManifestOverride,
} from "../../src/review/public-stats";
+import { recordAuditEvent, upsertPullRequestFromGitHub, upsertRepositoryFromGitHub } from "../../src/db/repositories";
const SELF_REPO = "JSONbored/loopover";
@@ -718,6 +720,36 @@ describe("getPublicStats — live aggregate over the review ledger", () => {
expect(out.totals.minutesSaved).not.toBe(2 * MINUTES_SAVED_PER_PR);
});
+ it("REGRESSION: publishes null accuracy, never 100%, when no reversal could have been recorded (real D1)", async () => {
+ // On a runtime that does not execute reviews, `recordReversalSignals` never runs, so `reversed` is pinned
+ // at 0 while merged/closed keep growing -- and `1 - 0/N` rendered as a perfect 100% on the fairness page
+ // for every repo and every week. That is a structural zero, not a measurement.
+ const env = createTestEnv({ LOOPOVER_PUBLIC_STATS_REPOS: "JSONbored/loopover" });
+ await upsertRepositoryFromGitHub(env, { name: "loopover", full_name: "JSONbored/loopover", private: false, owner: { login: "JSONbored" } }, 1);
+ await upsertPullRequestFromGitHub(env, "JSONbored/loopover", { number: 1, title: "PR 1", state: "closed", merged_at: "2026-06-20T09:00:00.000Z", user: { login: "a" }, head: { sha: "s1" }, labels: [] });
+ await recordAuditEvent(env, { eventType: "github_app.pr_public_surface_published", targetKey: "JSONbored/loopover#1", outcome: "completed", createdAt: "2026-06-20T09:00:00.000Z" });
+
+ const out = await getPublicStats(env, NOW);
+ expect(out.totals.merged).toBe(1);
+ expect(out.totals.reversed).toBe(0);
+ // The volume IS measured and still publishes; only the unmeasurable ratio is withheld.
+ expect(out.totals.accuracyPct).toBeNull();
+ expect(out.byProject.map((row) => row.accuracyPct)).toEqual([null]);
+ });
+
+ it("publishes a real accuracy once the deployment records the auto-actions a reversal attaches to (real D1)", async () => {
+ const env = createTestEnv({ LOOPOVER_PUBLIC_STATS_REPOS: "JSONbored/loopover" });
+ await upsertRepositoryFromGitHub(env, { name: "loopover", full_name: "JSONbored/loopover", private: false, owner: { login: "JSONbored" } }, 1);
+ await upsertPullRequestFromGitHub(env, "JSONbored/loopover", { number: 1, title: "PR 1", state: "closed", merged_at: "2026-06-20T09:00:00.000Z", user: { login: "a" }, head: { sha: "s1" }, labels: [] });
+ await recordAuditEvent(env, { eventType: "github_app.pr_public_surface_published", targetKey: "JSONbored/loopover#1", outcome: "completed", createdAt: "2026-06-20T09:00:00.000Z" });
+ // The engine actually merged it here -- so a reversal WOULD have been recorded had one happened.
+ await recordAuditEvent(env, { eventType: "agent.action.merge", targetKey: "JSONbored/loopover#1", outcome: "completed", createdAt: "2026-06-20T09:00:00.000Z" });
+
+ const out = await getPublicStats(env, NOW);
+ expect(out.totals.accuracyPct).toBe(100);
+ expect(out.byProject.map((row) => row.accuracyPct)).toEqual([100]);
+ });
+
it("deduplicates repeated public-surface publishes before averaging reviewEffortMinutes (real D1)", async () => {
const env = createTestEnv({ LOOPOVER_PUBLIC_STATS_REPOS: "JSONbored/loopover" });
const db = env.DB;
@@ -913,6 +945,29 @@ describe("getPublicStats — live aggregate over the review ledger", () => {
});
});
+describe("loadReversalObservability", () => {
+ it("short-circuits to true on a known reversal without querying at all", async () => {
+ const env = { DB: { prepare: () => { throw new Error("must not query"); } } } as unknown as Env;
+ expect(await loadReversalObservability(env, 1)).toBe(true);
+ });
+
+ it("is false on a deployment that has recorded no terminal auto-action", async () => {
+ expect(await loadReversalObservability(createTestEnv(), 0)).toBe(false);
+ });
+
+ it("is true once a terminal auto-action exists in the retained window", async () => {
+ const env = createTestEnv();
+ await recordAuditEvent(env, { eventType: "agent.action.close", targetKey: "JSONbored/loopover#3", outcome: "completed" });
+ expect(await loadReversalObservability(env, 0)).toBe(true);
+ });
+
+ it("degrades to false (never throws) when the probe query itself fails", async () => {
+ const broken = createTestEnv();
+ broken.DB = { prepare: () => { throw new Error("boom"); } } as never;
+ expect(await loadReversalObservability(broken, 0)).toBe(false);
+ });
+});
+
describe("fleetAccuracy.guaranteed (#8835/#9121/#9050/#9068)", () => {
// minimumCalibrationLabels(0.015, 0.05) = 199 -- every valid fixture below clears it with margin.
const calibrated = (overrides: Record = {}) => ({