From 8d5b8c12c241b90d340b193f1263818f0698c9b8 Mon Sep 17 00:00:00 2001 From: JSONbored <49853598+JSONbored@users.noreply.github.com> Date: Sat, 25 Jul 2026 22:57:11 -0700 Subject: [PATCH] fix(selfhost): skip dimension-mismatched vectors in the SQLite vectorize adapter instead of silently truncating (#8766) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit cosineSimilarity's Math.min(a.length, b.length) meant a store containing mixed-width vectors (an operator swapping embedding models without a full reindex) silently produced wrong similarity scores over the shorter prefix. Qdrant hard-fails this at the collection level and pgvector raises — the SQLite adapter was the odd one out. A mismatched-width row is now skipped (degrade-to-fewer-candidates, the RAG pipeline's established fail-safe posture) with one WARN per query naming the query width and the deduped set of mismatched stored widths. --- src/selfhost/vectorize.ts | 28 ++++++++++++++-- test/unit/selfhost-vectorize.test.ts | 49 +++++++++++++++++++++++++++- 2 files changed, 73 insertions(+), 4 deletions(-) diff --git a/src/selfhost/vectorize.ts b/src/selfhost/vectorize.ts index dafb7b1ed7..30bc0ec501 100644 --- a/src/selfhost/vectorize.ts +++ b/src/selfhost/vectorize.ts @@ -61,12 +61,34 @@ export function createSqliteVectorize(driver: SqliteDriver): Vectorize { const { rows } = opts.namespace ? driver.query(`SELECT id, embedding, metadata FROM ${TABLE} WHERE namespace=?`, [opts.namespace]) : driver.query(`SELECT id, embedding, metadata FROM ${TABLE}`, []); - const scored: Match[] = rows.map((r) => { + // #8766: a stored vector whose width differs from the query's (an operator swapped embedding models + // without a full reindex) is SKIPPED, never scored — cosineSimilarity's Math.min would otherwise + // silently compute a wrong similarity over the shorter prefix. Skipping degrades to fewer candidates, + // the same fail-safe posture the RAG pipeline already takes everywhere else; Qdrant hard-fails this at + // the collection level and pgvector raises, so this brings the SQLite adapter to parity. One WARN per + // query (not per row) so a mixed-width store is visible without flooding the log. + const mismatchedWidths = new Set(); + const scored: Match[] = []; + for (const r of rows) { const values = JSON.parse(r.embedding as string) as number[]; + if (values.length !== vector.length) { + mismatchedWidths.add(values.length); + continue; + } const metadata = r.metadata ? (JSON.parse(r.metadata as string) as Record) : undefined; const score = cosineSimilarity(vector, values); - return metadata === undefined ? { id: r.id as string, score } : { id: r.id as string, score, metadata }; - }); + scored.push(metadata === undefined ? { id: r.id as string, score } : { id: r.id as string, score, metadata }); + } + if (mismatchedWidths.size > 0) { + console.warn( + JSON.stringify({ + event: "sqlite_vectorize_dimension_mismatch", + queryWidth: vector.length, + storedWidths: [...mismatchedWidths].sort((a, b) => a - b), + detail: "mismatched-width rows skipped — reindex after an embedding-model change", + }), + ); + } scored.sort((a, b) => b.score - a.score); return { matches: scored.slice(0, opts.topK ?? 12) }; }, diff --git a/test/unit/selfhost-vectorize.test.ts b/test/unit/selfhost-vectorize.test.ts index 742b6c4f08..ed2a9c82a4 100644 --- a/test/unit/selfhost-vectorize.test.ts +++ b/test/unit/selfhost-vectorize.test.ts @@ -1,5 +1,5 @@ import { DatabaseSync } from "node:sqlite"; -import { describe, expect, it } from "vitest"; +import { afterEach, describe, expect, it, vi } from "vitest"; import { nodeSqliteDriver } from "../../src/selfhost/d1-adapter"; import { cosineSimilarity, createSqliteVectorize } from "../../src/selfhost/vectorize"; @@ -83,3 +83,50 @@ describe("createSqliteVectorize (#979 local RAG)", () => { expect(res.matches.map((m) => m.id)).toContain("n2"); }); }); + +describe("dimension-mismatch guard (#8766)", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + it("SKIPS a stored vector whose width differs from the query's — never scores over a truncated prefix", async () => { + const warn = vi.spyOn(console, "warn").mockImplementation(() => undefined); + const v = makeVectorize(); + await v.upsert([ + { id: "ok", values: [1, 0], namespace: "n" }, + { id: "wide", values: [1, 0, 0, 0], namespace: "n" }, // stale row from a different-dimension model + ]); + const { matches } = await v.query([1, 0], { topK: 10, namespace: "n" }); + expect(matches.map((m) => m.id)).toEqual(["ok"]); + // One WARN per query, naming both widths. + expect(warn).toHaveBeenCalledTimes(1); + const logged = JSON.parse(warn.mock.calls[0]![0] as string); + expect(logged).toMatchObject({ event: "sqlite_vectorize_dimension_mismatch", queryWidth: 2, storedWidths: [4] }); + }); + + it("logs NOTHING when every stored width matches (the steady state stays silent)", async () => { + const warn = vi.spyOn(console, "warn").mockImplementation(() => undefined); + const v = makeVectorize(); + await v.upsert([{ id: "a", values: [1, 0], namespace: "n" }]); + const { matches } = await v.query([1, 0], { topK: 10, namespace: "n" }); + expect(matches).toHaveLength(1); + expect(warn).not.toHaveBeenCalled(); + }); + + it("dedupes the warned widths and still returns matching rows sorted by score", async () => { + const warn = vi.spyOn(console, "warn").mockImplementation(() => undefined); + const v = makeVectorize(); + await v.upsert([ + { id: "best", values: [1, 0], namespace: "n" }, + { id: "worse", values: [0.5, 0.5], namespace: "n" }, + { id: "w4a", values: [1, 0, 0, 0], namespace: "n" }, + { id: "w4b", values: [0, 1, 0, 0], namespace: "n" }, + { id: "w3", values: [1, 0, 0], namespace: "n" }, + ]); + const { matches } = await v.query([1, 0], { topK: 10, namespace: "n" }); + expect(matches.map((m) => m.id)).toEqual(["best", "worse"]); + const logged = JSON.parse((vi.mocked(console.warn).mock.calls[0]!)[0] as string); + expect(logged.storedWidths).toEqual([3, 4]); + expect(warn).toHaveBeenCalledTimes(1); + }); +});