diff --git a/apps/loopover-ui/public/openapi.json b/apps/loopover-ui/public/openapi.json
index 281e2017db..22e9a360e3 100644
--- a/apps/loopover-ui/public/openapi.json
+++ b/apps/loopover-ui/public/openapi.json
@@ -800,6 +800,30 @@
"filteredPct"
]
}
+ },
+ "fleetAccuracyTrend": {
+ "type": "array",
+ "items": {
+ "type": "object",
+ "properties": {
+ "weekStart": {
+ "type": "string"
+ },
+ "verdicts": {
+ "type": "number",
+ "nullable": true
+ },
+ "accuracyPct": {
+ "type": "number",
+ "nullable": true
+ }
+ },
+ "required": [
+ "weekStart",
+ "verdicts",
+ "accuracyPct"
+ ]
+ }
}
},
"required": [
@@ -811,6 +835,7 @@
"byProject",
"fleetAccuracy",
"accuracyTrend",
+ "fleetAccuracyTrend",
"reuseRateTrend",
"reviewVolumeTrend"
]
diff --git a/apps/loopover-ui/src/components/site/fairness-report-page.test.tsx b/apps/loopover-ui/src/components/site/fairness-report-page.test.tsx
index 81631909be..8c3624a5bf 100644
--- a/apps/loopover-ui/src/components/site/fairness-report-page.test.tsx
+++ b/apps/loopover-ui/src/components/site/fairness-report-page.test.tsx
@@ -81,6 +81,7 @@ const FIXTURE: PublicStats = {
windowDays: 90,
gamingFlagsCaught: 1,
},
+ fleetAccuracyTrend: [{ weekStart: "2026-07-13", verdicts: 40, accuracyPct: 92.5 }],
accuracyTrend: [
{ weekStart: "2026-07-13", merged: 30, closed: 15, reversed: 1, accuracyPct: 97.8 },
],
@@ -244,6 +245,33 @@ describe("FairnessReportPage (#fairness-analytics)", () => {
expect(notes.length).toBe(2); // one under each affected table
});
+ it("#9676: renders the fleet trend as its own section, never merged into the own-ledger table", async () => {
+ apiFetch.mockResolvedValue({ ok: true, data: FIXTURE, status: 200, durationMs: 10 });
+ renderWithClient();
+
+ await waitFor(() => expect(screen.getByText("Weekly trend — self-hosted fleet")).toBeTruthy());
+ // Two distinct sections, so a reader can never read one population's number off the other's row.
+ expect(screen.getByText("Weekly trend")).toBeTruthy();
+ expect(screen.getByText("Decisions scored")).toBeTruthy();
+ expect(screen.getByText("92.5%")).toBeTruthy();
+ });
+
+ it("#9676: hides the fleet trend entirely when no week has a scored verdict", async () => {
+ apiFetch.mockResolvedValue({
+ ok: true,
+ data: {
+ ...FIXTURE,
+ fleetAccuracyTrend: [{ weekStart: "2026-07-13", verdicts: null, accuracyPct: null }],
+ },
+ status: 200,
+ durationMs: 10,
+ });
+ renderWithClient();
+
+ await waitFor(() => expect(screen.getByText("Weekly trend")).toBeTruthy());
+ expect(screen.queryByText("Weekly trend — self-hosted fleet")).toBeNull();
+ });
+
it("does not show the unmeasurable-accuracy note when every accuracy is real", async () => {
apiFetch.mockResolvedValue({ ok: true, data: FIXTURE, status: 200, durationMs: 10 });
renderWithClient();
diff --git a/apps/loopover-ui/src/components/site/fairness-report-page.tsx b/apps/loopover-ui/src/components/site/fairness-report-page.tsx
index 7a5646a8e6..f11b583f0f 100644
--- a/apps/loopover-ui/src/components/site/fairness-report-page.tsx
+++ b/apps/loopover-ui/src/components/site/fairness-report-page.tsx
@@ -300,6 +300,57 @@ export function FairnessReportPage() {
) : null}
+ {/* #9676: the fleet-population sibling of the table above. Rendered as its own section, never
+ merged into it: the two measure different populations on different estimands, and blending
+ them into one column is the bug this whole surface is being corrected for. */}
+ {data.fleetAccuracyTrend &&
+ data.fleetAccuracyTrend.some((week) => week.verdicts != null) ? (
+
+
Weekly trend — self-hosted fleet
+
+ The same weeks, measured over the live self-hosted fleet instead of this
+ site's own frozen review history. This scores the gate's merge/close{" "}
+ decisions — the share the realized outcome confirmed — so it matches the
+ headline above rather than the reversal-grounded table beside it. Holds are
+ excluded: a deferral to a human is not a decision that can be right or wrong. Only
+ registered instances count.
+
+
+
+
+ Weekly scored fleet verdicts and the share the realized outcome confirmed.
+
+
+
+ |
+ Week
+ |
+
+ Decisions scored
+ |
+
+ Accuracy
+ |
+
+
+
+ {data.fleetAccuracyTrend.map((week) => (
+
+ | {week.weekStart} |
+
+ {week.verdicts != null ? intFmt.format(week.verdicts) : "—"}
+ |
+
+ {week.accuracyPct != null ? `${pctFmt.format(week.accuracyPct)}%` : "—"}
+ |
+
+ ))}
+
+
+
+
+ ) : null}
+
{data.rulePrecision && data.rulePrecision.rules.length > 0 ? (
Measured accuracy per rule
diff --git a/apps/loopover-ui/src/components/site/proof-of-power-stats.test.tsx b/apps/loopover-ui/src/components/site/proof-of-power-stats.test.tsx
index 6c31f744bb..775f3d01f5 100644
--- a/apps/loopover-ui/src/components/site/proof-of-power-stats.test.tsx
+++ b/apps/loopover-ui/src/components/site/proof-of-power-stats.test.tsx
@@ -98,6 +98,7 @@ const PAYLOAD: PublicStats = {
windowDays: 90,
gamingFlagsCaught: 0,
},
+ fleetAccuracyTrend: [{ weekStart: "2026-07-13", verdicts: 40, accuracyPct: 92.5 }],
accuracyTrend: [
{ weekStart: "2026-05-04", merged: 40, closed: 10, reversed: 2, accuracyPct: 96 },
{ weekStart: "2026-05-11", merged: 42, closed: 9, reversed: 1, accuracyPct: 98 },
diff --git a/packages/loopover-contract/src/public-api.ts b/packages/loopover-contract/src/public-api.ts
index 5bdf1a79af..f3716097e9 100644
--- a/packages/loopover-contract/src/public-api.ts
+++ b/packages/loopover-contract/src/public-api.ts
@@ -95,6 +95,20 @@ export const PublicStatsSchema = z.object({
accuracyPct: z.number().nullable(),
}),
),
+ /** Trailing weekly FLEET accuracy (#9676). A SEPARATE series from `accuracyTrend`, never blended with it:
+ * that one is reversal-grounded over the own-ledger population (raw `owner/repo#number` keys), this one is
+ * `decisionAccuracy` over registered self-host instances' `orb_signals` (per-instance HMAC'd keys). The two
+ * populations cannot be joined, and #8820 established `decisionAccuracy` -- not `1 - reversalRate` -- as the
+ * fleet estimand, so this matches the headline it sits under rather than the table beside it. `verdicts`
+ * counts scored merge/close decisions only; holds and policy actions are excluded. Null on a week means too
+ * few scored verdicts to publish. */
+ fleetAccuracyTrend: z.array(
+ z.object({
+ weekStart: z.string(),
+ verdicts: z.number().nullable(),
+ accuracyPct: z.number().nullable(),
+ }),
+ ),
/** Trailing weekly "how often we avoid redoing AI work" trend (#4448) -- a competence signal, not a cost
* claim. Counts cache hits/misses across every instrumented AI-touching capability (grounding,
* review-memory, impact-map, repo-culture-profile, ai_review, ai_slop, linked_issue_satisfaction,
diff --git a/src/api/routes.ts b/src/api/routes.ts
index 57aff8ebe8..718db374aa 100644
--- a/src/api/routes.ts
+++ b/src/api/routes.ts
@@ -313,6 +313,7 @@ import { ingestBittensorAnchorReport, parseBittensorAnchorReport } from "../revi
import { loadPublicLedgerAnchors } from "../review/ledger-anchor-persistence";
import { getPublicStats, isPublicStatsEnabled, resolvePublicStatsManifestOverride } from "../review/public-stats";
import { loadPublicAccuracyTrend } from "../services/public-accuracy-trend";
+import { loadPublicFleetAccuracyTrend } from "../services/public-fleet-accuracy-trend";
import { loadPublicRulePrecision } from "../review/public-rule-precision";
import { loadCalibrationTrend } from "../services/rule-calibration-trend";
import { isSatisfactionFloorAutotuneEnabled, loadSatisfactionFloorStatus, runSatisfactionFloorLoosening } from "../services/satisfaction-floor-loosening-run";
@@ -1272,9 +1273,13 @@ export function createApp() {
const publicStatsManifestOverride = await resolvePublicStatsManifestOverride(c.env);
if (!isPublicStatsEnabled(c.env, publicStatsManifestOverride)) return c.json({ error: "not_found" }, 404);
try {
- const [stats, accuracyTrend, reuseRateTrend, reviewVolumeTrend, rulePrecision] = await Promise.all([
+ const [stats, accuracyTrend, fleetAccuracyTrend, reuseRateTrend, reviewVolumeTrend, rulePrecision] = await Promise.all([
getPublicStats(c.env),
loadPublicAccuracyTrend(c.env),
+ // #9676: the fleet-population sibling of the series above. Deliberately a SECOND series rather than a
+ // merged one -- see public-fleet-accuracy-trend.ts's header for why the two populations cannot be
+ // joined, and why this one matches the headline's estimand instead of the table's.
+ loadPublicFleetAccuracyTrend(c.env),
loadPublicReuseRateTrend(c.env),
loadPublicReviewVolumeTrend(c.env),
// #8230: measured per-rule precision + the reproducibility freeze point. Same flag, same cache,
@@ -1282,7 +1287,7 @@ export function createApp() {
loadPublicRulePrecision(c.env),
]);
c.header("Cache-Control", "public, max-age=60, stale-while-revalidate=300");
- return c.json({ ...stats, accuracyTrend, reuseRateTrend, reviewVolumeTrend, rulePrecision });
+ return c.json({ ...stats, accuracyTrend, fleetAccuracyTrend, reuseRateTrend, reviewVolumeTrend, rulePrecision });
} catch {
return c.json({ error: "public_stats_unavailable" }, 503);
}
diff --git a/src/services/public-fleet-accuracy-trend.ts b/src/services/public-fleet-accuracy-trend.ts
new file mode 100644
index 0000000000..db9d272d89
--- /dev/null
+++ b/src/services/public-fleet-accuracy-trend.ts
@@ -0,0 +1,135 @@
+// Public fleet accuracy trend (#9676). The own-ledger weekly trend (public-accuracy-trend.ts) is
+// reversal-grounded over `audit_events`, and that ledger is frozen: the hosted Worker does not execute
+// reviews, so every recent week correctly reports `null` while Orb volume keeps growing. A table of nulls
+// beside a rising volume column is honest but tells a reader nothing about how the gate behaves TODAY.
+//
+// `orb_signals` already carries what is needed, with no new ingest, no new column and no new secret:
+// self-host runtimes export `gate_verdict`, `outcome` and `reversal_flag` per PR (src/selfhost/orb-collector.ts),
+// the hosted side validates and stores them (src/orb/ingest.ts), and computeFleetAnalytics already turns them
+// into the 90-day headline. This module buckets the SAME rows weekly.
+//
+// TWO THINGS THIS DELIBERATELY DOES NOT DO:
+//
+// 1. It never blends with the own-ledger series. `orb_signals` is keyed by per-instance HMACs of repo/PR;
+// `orb_pr_outcomes` and `audit_events` are keyed by raw `owner/repo#number`. The two count different
+// populations and cannot be joined, so a reversal from one can never be attributed to a decided row in
+// the other. Publishing one number over a mixed denominator is exactly the class of bug #9676 exists to
+// fix -- so this is a SEPARATE series with its own numerator and denominator.
+//
+// 2. It does not use `1 - reversalRate`. #8820 established that the fleet number is `decisionAccuracy` --
+// the share of the gate's own merge/close verdicts the realized outcome confirmed, holds excluded --
+// because the reversal formula both divides by deferrals that cannot be right or wrong and misses
+// outright mispredictions that carry no reversal marker. Publishing a weekly series on a different
+// estimand than the headline it sits under would make the page disagree with itself.
+//
+// The weekly fold calls `foldInstance` rather than reimplementing the confusion-matrix accounting. That
+// function's own doc comment already warns that a second copy silently makes comparisons
+// apples-to-oranges (it was exported for the federated bundle for the same reason), and the accounting is
+// subtle: `policy_action` rows are excluded from scoring but still count as activity, and a `superseded`
+// close is disconfirmed exactly like a literal reopen.
+import { foldInstance, type Cell } from "../orb/analytics";
+import { safeAll } from "../review/public-stats";
+import { isoWeekStart } from "./public-quality-metrics";
+
+export const PUBLIC_FLEET_TREND_WEEKS = 8;
+
+/** Below this many scored verdicts in a week, that week's accuracy is too noisy to publish. Mirrors
+ * MIN_ACCURACY_TREND_SAMPLE's role on the own-ledger series -- a public percentage over one or two
+ * decisions is a coin flip wearing a number. */
+export const MIN_FLEET_TREND_VERDICTS = 5;
+
+export type PublicFleetAccuracyTrendWeek = {
+ /** UTC Monday (YYYY-MM-DD) that starts the bucket. */
+ weekStart: string;
+ /** Scored merge/close verdicts in the week -- holds and policy actions excluded, matching the estimand. */
+ verdicts: number | null;
+ /** Share of those verdicts the realized outcome confirmed; null below the sample floor. */
+ accuracyPct: number | null;
+};
+
+/** One `orb_signals` confusion-matrix cell, bucketed to a week. */
+export type FleetTrendCell = Cell & { weekStart: string };
+
+/** The raw query shape: the same cell dimensions, bucketed to a UTC day by SQL. */
+type FleetTrendDayCell = Cell & { day: string };
+
+const MS_PER_WEEK = 7 * 86_400_000;
+
+function roundPct(value: number): number {
+ return Math.round(value * 1000) / 10;
+}
+
+/**
+ * PURE. Fold week-bucketed cells into `weeks` trailing UTC-Monday buckets ending in the week containing
+ * `nowMs`. Mirrors buildPublicAccuracyTrend's bucketing shape so the two series line up row-for-row on the
+ * page even though they measure different populations.
+ */
+export function buildPublicFleetAccuracyTrend(
+ cells: readonly FleetTrendCell[],
+ nowMs: number,
+ weeks: number = PUBLIC_FLEET_TREND_WEEKS,
+): PublicFleetAccuracyTrendWeek[] {
+ const currentStartMs = Date.parse(isoWeekStart(nowMs));
+ const oldestStartMs = currentStartMs - (weeks - 1) * MS_PER_WEEK;
+ const buckets: Cell[][] = Array.from({ length: weeks }, () => []);
+
+ for (const cell of cells) {
+ const weekMs = Date.parse(`${cell.weekStart}T00:00:00.000Z`);
+ if (!Number.isFinite(weekMs)) continue;
+ const offset = Math.round((weekMs - oldestStartMs) / MS_PER_WEEK);
+ if (offset < 0 || offset >= weeks) continue;
+ buckets[offset]!.push(cell);
+ }
+
+ return buckets.map((bucketCells, offset) => {
+ const weekStart = isoWeekStart(oldestStartMs + offset * MS_PER_WEEK);
+ if (bucketCells.length === 0) return { weekStart, verdicts: null, accuracyPct: null };
+ // Pooled across every registered instance in the week, matching how the headline pools (#9068): one
+ // synthetic id, because this series is a fleet aggregate and not a per-instance figure.
+ const folded = foldInstance("fleet", bucketCells);
+ const verdicts = folded.counts.mergeVerdicts + folded.counts.closeVerdicts;
+ if (verdicts < MIN_FLEET_TREND_VERDICTS || folded.decisionAccuracy === null) {
+ return { weekStart, verdicts: null, accuracyPct: null };
+ }
+ return { weekStart, verdicts, accuracyPct: roundPct(folded.decisionAccuracy) };
+ });
+}
+
+/**
+ * Load the weekly fleet accuracy series. Fail-safe: `safeAll` swallows a read error into an empty result, so
+ * a bad query yields all-null weeks rather than throwing the whole public stats payload.
+ *
+ * Only REGISTERED instances count, matching computeFleetAnalytics' own trust gate: the ingest is open, so a
+ * stranger's signals must not move a published number until a human opts them in.
+ */
+export async function loadPublicFleetAccuracyTrend(env: Env, nowMs: number = Date.now()): Promise
{
+ const sinceIso = new Date(Date.parse(isoWeekStart(nowMs)) - (PUBLIC_FLEET_TREND_WEEKS - 1) * MS_PER_WEEK).toISOString();
+ // Bucket by when the gate DECIDED, not when the row arrived: a batch exported late would otherwise pile
+ // weeks of decisions into the week it was received. `received_at` is the fallback only because
+ // `decision_timestamp` is nullable in the schema.
+ const cells = await safeAll(
+ env,
+ // instance_id is selected and grouped even though this series is a fleet aggregate: it keeps the rows a
+ // genuine `Cell`, which is what foldInstance consumes, and summing per-instance cells within a week
+ // yields exactly the pooled figure anyway. Narrowing the projection would mean hand-rolling a
+ // near-Cell type and re-deriving the accounting -- the duplication foldInstance is exported to prevent.
+ `SELECT substr(COALESCE(s.decision_timestamp, s.received_at), 1, 10) AS day, s.instance_id,
+ s.gate_verdict AS verdict, s.outcome, s.reversal_flag, s.gate_reasoncode_bucket, COUNT(*) AS n
+ FROM orb_signals s
+ JOIN orb_instances i ON i.instance_id = s.instance_id AND i.registered = 1
+ WHERE COALESCE(s.decision_timestamp, s.received_at) >= ?
+ GROUP BY day, s.instance_id, s.gate_verdict, s.outcome, s.reversal_flag, s.gate_reasoncode_bucket`,
+ sinceIso,
+ );
+ // isoWeekStart is applied here rather than in SQL: SQLite's strftime('%W') is not ISO-8601 week numbering,
+ // and the own-ledger trend already owns this exact conversion in JS.
+ const weekly: FleetTrendCell[] = [];
+ for (const cell of cells) {
+ const dayMs = Date.parse(`${cell.day}T00:00:00.000Z`);
+ // A row whose timestamp column is unparseable is dropped rather than bucketed to the epoch, which would
+ // silently land it outside the window anyway -- explicit is better than accidentally-correct.
+ if (!Number.isFinite(dayMs)) continue;
+ weekly.push({ ...cell, weekStart: isoWeekStart(dayMs) });
+ }
+ return buildPublicFleetAccuracyTrend(weekly, nowMs);
+}
diff --git a/test/unit/public-fleet-accuracy-trend.test.ts b/test/unit/public-fleet-accuracy-trend.test.ts
new file mode 100644
index 0000000000..08cca008aa
--- /dev/null
+++ b/test/unit/public-fleet-accuracy-trend.test.ts
@@ -0,0 +1,205 @@
+import { describe, expect, it } from "vitest";
+import {
+ buildPublicFleetAccuracyTrend,
+ loadPublicFleetAccuracyTrend,
+ MIN_FLEET_TREND_VERDICTS,
+ PUBLIC_FLEET_TREND_WEEKS,
+ type FleetTrendCell,
+} from "../../src/services/public-fleet-accuracy-trend";
+import { isoWeekStart } from "../../src/services/public-quality-metrics";
+import { createTestEnv } from "../helpers/d1";
+
+// #9676: the fleet-population weekly series. Load-bearing properties: it uses decisionAccuracy (the
+// headline's estimand, #8820) rather than 1 - reversalRate, it counts only scored merge/close verdicts, and
+// it only ever sees REGISTERED instances.
+
+const NOW = Date.parse("2026-06-22T12:00:00.000Z");
+
+const cell = (over: Partial = {}): FleetTrendCell => ({
+ weekStart: isoWeekStart(NOW),
+ instance_id: "inst",
+ verdict: "merge",
+ outcome: "merged",
+ reversal_flag: "none",
+ gate_reasoncode_bucket: "quality",
+ n: 1,
+ ...over,
+});
+
+describe("buildPublicFleetAccuracyTrend (#9676)", () => {
+ it("scores decisionAccuracy: confirmed verdicts over scored verdicts", () => {
+ const week = isoWeekStart(NOW);
+ const trend = buildPublicFleetAccuracyTrend(
+ [cell({ n: 8 }), cell({ outcome: "closed", n: 2 })], // 8 confirmed merges, 2 merges that closed instead
+ NOW,
+ 1,
+ );
+ expect(trend[0]).toEqual({ weekStart: week, verdicts: 10, accuracyPct: 80 });
+ });
+
+ it("INVARIANT: uses decisionAccuracy, NOT 1 - reversalRate -- holds must not dilute it", () => {
+ // 5 confirmed merges plus 95 holds. The reversal formula would divide by all 100 and report ~100%
+ // regardless of gate quality; decisionAccuracy scores only the 5 decisions the gate actually made.
+ const trend = buildPublicFleetAccuracyTrend([cell({ n: 5 }), cell({ verdict: "hold", n: 95 })], NOW, 1);
+ expect(trend[0]?.verdicts).toBe(5);
+ expect(trend[0]?.accuracyPct).toBe(100);
+ });
+
+ it("counts a reverted merge and a reopened/superseded close as the gate being wrong", () => {
+ const reverted = buildPublicFleetAccuracyTrend([cell({ n: 5 }), cell({ reversal_flag: "reverted", n: 5 })], NOW, 1);
+ expect(reverted[0]?.accuracyPct).toBe(50);
+ for (const flag of ["reopened", "superseded"] as const) {
+ const closes = buildPublicFleetAccuracyTrend(
+ [cell({ verdict: "close", outcome: "closed", n: 5 }), cell({ verdict: "close", outcome: "closed", reversal_flag: flag, n: 5 })],
+ NOW,
+ 1,
+ );
+ expect(closes[0]?.accuracyPct).toBe(50);
+ }
+ });
+
+ it("excludes policy_action rows from scoring entirely (#8825), rather than inflating either side", () => {
+ const trend = buildPublicFleetAccuracyTrend(
+ [cell({ n: 5 }), cell({ gate_reasoncode_bucket: "policy_action", verdict: "close", outcome: "closed", n: 50 })],
+ NOW,
+ 1,
+ );
+ expect(trend[0]).toEqual({ weekStart: isoWeekStart(NOW), verdicts: 5, accuracyPct: 100 });
+ });
+
+ it("redacts a week below the verdict floor rather than publishing a coin flip", () => {
+ const trend = buildPublicFleetAccuracyTrend([cell({ n: MIN_FLEET_TREND_VERDICTS - 1 })], NOW, 1);
+ expect(trend[0]).toMatchObject({ verdicts: null, accuracyPct: null });
+ });
+
+ it("publishes at exactly the floor", () => {
+ expect(buildPublicFleetAccuracyTrend([cell({ n: MIN_FLEET_TREND_VERDICTS })], NOW, 1)[0]?.accuracyPct).toBe(100);
+ });
+
+ it("redacts a week of holds only -- activity without a single scored decision", () => {
+ const trend = buildPublicFleetAccuracyTrend([cell({ verdict: "hold", n: 40 })], NOW, 1);
+ expect(trend[0]).toMatchObject({ verdicts: null, accuracyPct: null });
+ });
+
+ it("buckets into trailing UTC-Monday weeks and ignores rows outside the window", () => {
+ const current = isoWeekStart(NOW);
+ const prior = isoWeekStart(NOW - 7 * 86_400_000);
+ const ancient = isoWeekStart(NOW - 60 * 86_400_000);
+ const trend = buildPublicFleetAccuracyTrend(
+ [cell({ weekStart: prior, n: 10 }), cell({ weekStart: current, n: 5 }), cell({ weekStart: ancient, n: 999 })],
+ NOW,
+ 2,
+ );
+ expect(trend.map((w) => w.weekStart)).toEqual([prior, current]);
+ expect(trend.map((w) => w.verdicts)).toEqual([10, 5]);
+ });
+
+ it("ignores an unparseable weekStart rather than corrupting a bucket", () => {
+ const trend = buildPublicFleetAccuracyTrend([cell({ weekStart: "not-a-date", n: 99 }), cell({ n: 5 })], NOW, 1);
+ expect(trend[0]?.verdicts).toBe(5);
+ });
+
+ it("returns all-null weeks for empty input, and defaults to PUBLIC_FLEET_TREND_WEEKS", () => {
+ const trend = buildPublicFleetAccuracyTrend([], NOW);
+ expect(trend).toHaveLength(PUBLIC_FLEET_TREND_WEEKS);
+ for (const week of trend) expect(week).toMatchObject({ verdicts: null, accuracyPct: null });
+ });
+});
+
+describe("loadPublicFleetAccuracyTrend — end to end over orb_signals", () => {
+ async function seed(env: Env, input: { instance: string; registered: boolean; rows: Array<{ verdict: string; outcome: string; reversal?: string; decidedAt: string }> }): Promise {
+ await env.DB.prepare("INSERT OR IGNORE INTO orb_instances (instance_id, registered) VALUES (?, ?)").bind(input.instance, input.registered ? 1 : 0).run();
+ let i = 0;
+ for (const row of input.rows) {
+ i += 1;
+ await env.DB
+ .prepare(
+ `INSERT INTO orb_signals (instance_id, repo_hash, pr_hash, gate_verdict, outcome, reversal_flag, gate_reasoncode_bucket, decision_timestamp, received_at)
+ VALUES (?, ?, ?, ?, ?, ?, 'quality', ?, ?)`,
+ )
+ .bind(input.instance, `repo-${input.instance}`, `pr-${input.instance}-${i}`, row.verdict, row.outcome, row.reversal ?? "none", row.decidedAt, row.decidedAt)
+ .run();
+ }
+ }
+
+ const thisWeek = `${isoWeekStart(NOW)}T09:00:00.000Z`;
+
+ it("scores a registered instance's week from real rows", async () => {
+ const env = createTestEnv();
+ await seed(env, {
+ instance: "inst-a",
+ registered: true,
+ rows: [
+ ...Array.from({ length: 8 }, () => ({ verdict: "merge", outcome: "merged", decidedAt: thisWeek })),
+ ...Array.from({ length: 2 }, () => ({ verdict: "merge", outcome: "closed", decidedAt: thisWeek })),
+ ],
+ });
+ const trend = await loadPublicFleetAccuracyTrend(env, NOW);
+ expect(trend[trend.length - 1]).toEqual({ weekStart: isoWeekStart(NOW), verdicts: 10, accuracyPct: 80 });
+ });
+
+ it("INVARIANT: an UNREGISTERED instance never moves a published number", async () => {
+ // The ingest is open, so a stranger's signals must not reach the public surface until a human opts them
+ // in -- the same trust gate computeFleetAnalytics applies to the headline.
+ const env = createTestEnv();
+ await seed(env, {
+ instance: "stranger",
+ registered: false,
+ rows: Array.from({ length: 40 }, () => ({ verdict: "merge", outcome: "closed", decidedAt: thisWeek })),
+ });
+ const trend = await loadPublicFleetAccuracyTrend(env, NOW);
+ for (const week of trend) expect(week).toMatchObject({ verdicts: null, accuracyPct: null });
+ });
+
+ it("pools registered instances within a week", async () => {
+ const env = createTestEnv();
+ await seed(env, { instance: "a", registered: true, rows: Array.from({ length: 5 }, () => ({ verdict: "merge", outcome: "merged", decidedAt: thisWeek })) });
+ await seed(env, { instance: "b", registered: true, rows: Array.from({ length: 5 }, () => ({ verdict: "merge", outcome: "closed", decidedAt: thisWeek })) });
+ expect((await loadPublicFleetAccuracyTrend(env, NOW))[PUBLIC_FLEET_TREND_WEEKS - 1]).toMatchObject({ verdicts: 10, accuracyPct: 50 });
+ });
+
+ it("buckets by when the gate DECIDED, not when the row was received", async () => {
+ // A late export must not pile older decisions into the week it arrived.
+ const env = createTestEnv();
+ const priorWeek = `${isoWeekStart(NOW - 7 * 86_400_000)}T09:00:00.000Z`;
+ await env.DB.prepare("INSERT OR IGNORE INTO orb_instances (instance_id, registered) VALUES ('late', 1)").run();
+ for (let i = 0; i < 6; i += 1) {
+ await env.DB
+ .prepare(
+ `INSERT INTO orb_signals (instance_id, repo_hash, pr_hash, gate_verdict, outcome, reversal_flag, gate_reasoncode_bucket, decision_timestamp, received_at)
+ VALUES ('late', 'r', ?, 'merge', 'merged', 'none', 'quality', ?, ?)`,
+ )
+ .bind(`pr-${i}`, priorWeek, thisWeek)
+ .run();
+ }
+ const trend = await loadPublicFleetAccuracyTrend(env, NOW);
+ expect(trend[trend.length - 2]).toMatchObject({ verdicts: 6 });
+ expect(trend[trend.length - 1]).toMatchObject({ verdicts: null });
+ });
+
+ it("drops a row whose timestamp column is unparseable rather than bucketing it to the epoch", async () => {
+ // decision_timestamp is TEXT with no format constraint, so a malformed value is reachable in practice.
+ // It must be dropped explicitly, not silently land somewhere via Date.parse -> NaN arithmetic.
+ const env = createTestEnv();
+ await env.DB.prepare("INSERT OR IGNORE INTO orb_instances (instance_id, registered) VALUES ('bad', 1)").run();
+ for (let i = 0; i < 6; i += 1) {
+ await env.DB
+ .prepare(
+ `INSERT INTO orb_signals (instance_id, repo_hash, pr_hash, gate_verdict, outcome, reversal_flag, gate_reasoncode_bucket, decision_timestamp, received_at)
+ VALUES ('bad', 'r', ?, 'merge', 'merged', 'none', 'quality', 'not-a-timestamp', 'not-a-timestamp')`,
+ )
+ .bind(`pr-${i}`)
+ .run();
+ }
+ const trend = await loadPublicFleetAccuracyTrend(env, NOW);
+ for (const week of trend) expect(week).toMatchObject({ verdicts: null, accuracyPct: null });
+ });
+
+ it("degrades to all-null weeks (never throws) when the read fails", async () => {
+ const broken = createTestEnv();
+ broken.DB = { prepare: () => { throw new Error("boom"); } } as never;
+ const trend = await loadPublicFleetAccuracyTrend(broken, NOW);
+ expect(trend).toHaveLength(PUBLIC_FLEET_TREND_WEEKS);
+ for (const week of trend) expect(week).toMatchObject({ verdicts: null, accuracyPct: null });
+ });
+});